From 29f3df9487006c730416a53551f39504a7935790 Mon Sep 17 00:00:00 2001 From: Gabriel Date: Tue, 29 Sep 2026 13:50:19 +0800 Subject: [PATCH 1/9] [improvement](lance) Adopt upstream segment prefilter optimization ### What problem does this PR solve? Related PR: lance-format/lance#9599, lance-format/lance-c#89 Problem Summary: Complete index-segment vector scans without predicates materialize an unnecessary row-ID allowlist. Pin upstream lance-c with the Lance development-branch optimization and metric-based regressions. Remove superseded local patches, retaining only the Foyer cache integration with object-store compatibility and bounded response-metadata reuse. ### Release note Avoid redundant prefilter row-ID construction for unfiltered vector queries covering complete visible index segments. ### Check List (For Author) - Test: Lance prefilter and segment-contract tests; lance-c and Foyer unit/C API suites; Rust format and Clippy; C/C++ executable tests; dependency downloader patch lifecycle, checksum, and fallback regressions. - Behavior changed: Yes, eligible vector scans skip redundant prefilter work; predicates, partial coverage, deletions, and flat fallback retain semantics. - Does this need documentation: No; no SQL or configuration interface change. --- thirdparty/download-thirdparty.sh | 26 +- thirdparty/patches/lance-c-0.1.9-pr-74.patch | 1159 -------- .../patches/lance-c-0.1.9-pr-75-pr-78.patch | 2441 ----------------- thirdparty/patches/lance-c-0.1.9-pr-77.patch | 1863 ------------- thirdparty/patches/lance-c-0.1.9-pr-79.patch | 1782 ------------ thirdparty/patches/lance-c-0.1.9-pr-80.patch | 225 -- thirdparty/patches/lance-c-0.1.9-pr-83.patch | 1914 ------------- .../patches/lance-c-0.1.9-prefilter-fts.patch | 148 - .../patches/lance-c-0.1.9-prefilter.patch | 147 - ...-0.1.9-pr-73.patch => lance-c-foyer.patch} | 655 ++--- thirdparty/test/lance-prefilter-patch-test.sh | 95 + thirdparty/vars.sh | 9 +- 12 files changed, 362 insertions(+), 10102 deletions(-) delete mode 100644 thirdparty/patches/lance-c-0.1.9-pr-74.patch delete mode 100644 thirdparty/patches/lance-c-0.1.9-pr-75-pr-78.patch delete mode 100644 thirdparty/patches/lance-c-0.1.9-pr-77.patch delete mode 100644 thirdparty/patches/lance-c-0.1.9-pr-79.patch delete mode 100644 thirdparty/patches/lance-c-0.1.9-pr-80.patch delete mode 100644 thirdparty/patches/lance-c-0.1.9-pr-83.patch delete mode 100644 thirdparty/patches/lance-c-0.1.9-prefilter-fts.patch delete mode 100644 thirdparty/patches/lance-c-0.1.9-prefilter.patch rename thirdparty/patches/{lance-c-0.1.9-pr-73.patch => lance-c-foyer.patch} (88%) create mode 100755 thirdparty/test/lance-prefilter-patch-test.sh diff --git a/thirdparty/download-thirdparty.sh b/thirdparty/download-thirdparty.sh index e40a6f3d8b872e..bb46d3640ea60d 100755 --- a/thirdparty/download-thirdparty.sh +++ b/thirdparty/download-thirdparty.sh @@ -718,30 +718,14 @@ if [[ " ${TP_ARCHIVES[*]} " =~ " AZURE " ]]; then echo "Finished patching ${AZURE_SOURCE}" fi -# Apply Doris lance-c patches as one chain to the pinned release archive. +# Foyer remains a local patch until its cache interface is accepted upstream. +# All search fixes are supplied by the immutable lance-c dependency revision. if [[ " ${TP_ARCHIVES[*]} " =~ " LANCE_C " ]]; then cd "${TP_SOURCE_DIR}/${LANCE_C_SOURCE}" - LANCE_C_PATCHED_MARK="${PATCHED_MARK}_community_pr83_prefilter" - # Older source caches carry a different PR #73 and cannot accept this chain incrementally. - if [[ -f "${PATCHED_MARK}" && ! -f "${LANCE_C_PATCHED_MARK}" ]]; then - echo "The lance-c patch chain changed; remove ${TP_SOURCE_DIR}/${LANCE_C_SOURCE} and rebuild." - exit 1 - fi - if [[ ! -f "${LANCE_C_PATCHED_MARK}" ]]; then - # PR #77 provides Lance v11 for the following community patches. PR #83 - # retains PR #79's scalar-segment path when adding multi-vector execution. - # The final patch pins the full-snapshot prefilter fix and its execution metrics. - for lance_patch in pr-74 pr-75-pr-78 pr-77 pr-73 pr-79 pr-80 pr-83 prefilter; do - patch --batch --forward --reject-file=- --fuzz=0 --no-backup-if-mismatch -s \ - -p1 <"${TP_PATCH_DIR}/${LANCE_C_SOURCE}-${lance_patch}.patch" - done - touch "${PATCHED_MARK}" "${LANCE_C_PATCHED_MARK}" - fi - # Cached sources may carry the earlier prefilter pin; upgrade FTS metrics independently. - if [[ ! -f "${PATCHED_MARK}_prefilter_fts" ]]; then + if [[ ! -f "${PATCHED_MARK}_foyer" ]]; then patch --batch --forward --reject-file=- --fuzz=0 --no-backup-if-mismatch -s \ - -p1 <"${TP_PATCH_DIR}/${LANCE_C_SOURCE}-prefilter-fts.patch" - touch "${PATCHED_MARK}_prefilter_fts" + -p1 <"${TP_PATCH_DIR}/lance-c-foyer.patch" + touch "${PATCHED_MARK}_foyer" fi cd - echo "Finished patching ${LANCE_C_SOURCE}" diff --git a/thirdparty/patches/lance-c-0.1.9-pr-74.patch b/thirdparty/patches/lance-c-0.1.9-pr-74.patch deleted file mode 100644 index 24c6d33457d84c..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-pr-74.patch +++ /dev/null @@ -1,1159 +0,0 @@ -From b07f970bf2cf3f983cc6043bc72a0fe444c0fd4c Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Thu, 3 Sep 2026 17:16:34 +0800 -Subject: [PATCH 1/2] fts - ---- - include/lance/lance.h | 74 ++++++++--- - include/lance/lance.hpp | 44 +++++-- - src/fts_query.rs | 235 +++++++++++++++++++++++++++++------ - src/scanner.rs | 72 ++++++++--- - tests/c_api_test.rs | 264 +++++++++++++++++++++++++++++++++++++--- - 5 files changed, 595 insertions(+), 94 deletions(-) - -diff --git a/include/lance/lance.h b/include/lance/lance.h -index 3bf291f..0c76edc 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -1695,39 +1695,85 @@ typedef enum { - LANCE_FTS_COVERAGE_INDEX_ONLY = 1, - } LanceFtsCoverageMode; - -+/** How the analyzed terms of one Match query are combined. */ -+typedef enum { -+ /** At least one analyzed term must match. */ -+ LANCE_FTS_MATCH_OPERATOR_OR = 0, -+ /** Every analyzed term must match. */ -+ LANCE_FTS_MATCH_OPERATOR_AND = 1, -+} LanceFtsMatchOperator; -+ -+/** -+ * Prepare an OR Match query context for one column. -+ * -+ * @deprecated Use lance_dataset_prepare_fts_match_query() to select the Match -+ * operator explicitly. This compatibility API is equivalent to -+ * LANCE_FTS_MATCH_OPERATOR_OR. -+ */ -+LanceFtsQueryContext* lance_dataset_prepare_fts_query( -+ const LanceDataset* dataset, -+ const char* column, -+ const char* query, -+ uint32_t max_fuzzy_distance, -+ int32_t coverage_mode -+); -+ - /** -- * Prepare an immutable, process-local FTS query context for one column. -+ * Prepare an immutable, process-local Match query context for one column. - * - * Preparation pins the dataset handle's current snapshot, enumerates all - * committed FTS segments for `column`, checks fragment coverage, opens those -- * segments, and computes one query-specific global BM25 scorer across their -- * indexed documents. The context can then be shared by any number of scanners -- * created from the exact same process-local dataset snapshot. It has no -- * serialization or cross-process transport format. Reopening the same URI and -- * manifest version creates a different identity and cannot reuse the context, -- * because storage options and object-store endpoints may differ. -+ * segments, and prepares one global BM25 scorer across their indexed -+ * documents. `match_operator` supports both AND and OR. -+ * -+ * The context can be shared by scanners created from the exact same -+ * process-local dataset snapshot. It has no serialization or cross-process -+ * transport format. Reopening the same URI and manifest version creates a -+ * different identity and cannot reuse the context because storage options and -+ * object-store endpoints may differ. - * - * In LANCE_FTS_COVERAGE_INDEX_ONLY mode, unindexed fragments are allowed and - * excluded from both the scorer corpus and query results. In STRICT mode any - * unindexed fragment makes this call fail. - * -- * Prepared contexts currently support exact Match queries only. -- * `max_fuzzy_distance` must be zero because fuzzy execution requires its -- * canonical expanded vocabulary to be prepared together with the scorer. -- * This restriction does not apply to lance_scanner_full_text_search(). -- * -- * @param max_fuzzy_distance Must be zero for prepared query contexts. -+ * @param match_operator Fixed-width LanceFtsMatchOperator discriminant. -+ * @param max_fuzzy_distance Reserved for prepared fuzzy matching and currently -+ * must be 0. The parameter is retained so enabling -+ * canonical cross-segment fuzzy vocabulary injection -+ * later does not require another C ABI change. - * @param coverage_mode Fixed-width LanceFtsCoverageMode discriminant. - * @return Context handle on success, or NULL on error. - */ --LanceFtsQueryContext* lance_dataset_prepare_fts_query( -+LanceFtsQueryContext* lance_dataset_prepare_fts_match_query( - const LanceDataset* dataset, - const char* column, - const char* query, -+ int32_t match_operator, - uint32_t max_fuzzy_distance, - int32_t coverage_mode - ); - -+/** -+ * Prepare an immutable, process-local Phrase query context for one column. -+ * -+ * The selected FTS index must store token positions. `slop == 0` requires an -+ * exact phrase; a positive value permits that many intervening positions. -+ * Dataset identity, coverage, sharing, and segment-scoped execution follow the -+ * same contract as lance_dataset_prepare_fts_match_query(). -+ * -+ * @param slop Maximum number of intervening token positions permitted between -+ * adjacent phrase terms. -+ * @param coverage_mode Fixed-width LanceFtsCoverageMode discriminant. -+ * @return Context handle on success, or NULL on error. -+ */ -+LanceFtsQueryContext* lance_dataset_prepare_fts_phrase_query( -+ const LanceDataset* dataset, -+ const char* column, -+ const char* query, -+ uint32_t slop, -+ int32_t coverage_mode -+); -+ - /** - * Close a context handle. NULL-safe. Scanners that already attached this - * context retain shared ownership and remain valid. -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index 6cf245f..973216a 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -128,6 +128,11 @@ enum class FtsCoverageMode : int32_t { - IndexOnly = LANCE_FTS_COVERAGE_INDEX_ONLY, - }; - -+enum class FtsMatchOperator : int32_t { -+ Or = LANCE_FTS_MATCH_OPERATOR_OR, -+ And = LANCE_FTS_MATCH_OPERATOR_AND, -+}; -+ - /// Tunable parameters for Dataset::write. Numeric fields default-out via 0; - /// `data_storage_version` defaults out via `std::nullopt`. - /// -@@ -764,18 +769,43 @@ class Dataset { - /// Create a Scanner builder for this dataset. - Scanner scan() const; - -- /// Prepare a query-specific global BM25 scorer over the committed FTS -- /// segments of this pinned snapshot. IndexOnly permits unindexed fragments; -- /// Strict rejects them. Prepared contexts currently require -- /// `max_fuzzy_distance == 0`. The context can only be attached to scanners -- /// created from this exact process-local dataset snapshot. -+ /// Compatibility wrapper for an OR Match query. -+ [[deprecated("Use prepare_fts_match_query() to select the Match operator")]] - FtsQueryContext prepare_fts_query( - const std::string& column, - const std::string& query, - uint32_t max_fuzzy_distance = 0, - FtsCoverageMode coverage_mode = FtsCoverageMode::Strict) const { -- auto* context = lance_dataset_prepare_fts_query( -- handle_.get(), column.c_str(), query.c_str(), max_fuzzy_distance, -+ return prepare_fts_match_query(column, query, FtsMatchOperator::Or, -+ max_fuzzy_distance, coverage_mode); -+ } -+ -+ /// Prepare a Match query with a global BM25 scorer. AND and OR are -+ /// supported. `max_fuzzy_distance` is reserved and currently must be zero; -+ /// keeping it here avoids another API change when canonical cross-segment -+ /// fuzzy vocabulary injection becomes available. -+ FtsQueryContext prepare_fts_match_query( -+ const std::string& column, -+ const std::string& query, -+ FtsMatchOperator match_operator = FtsMatchOperator::Or, -+ uint32_t max_fuzzy_distance = 0, -+ FtsCoverageMode coverage_mode = FtsCoverageMode::Strict) const { -+ auto* context = lance_dataset_prepare_fts_match_query( -+ handle_.get(), column.c_str(), query.c_str(), -+ static_cast(match_operator), max_fuzzy_distance, -+ static_cast(coverage_mode)); -+ if (!context) check_error(); -+ return FtsQueryContext(context); -+ } -+ -+ /// Prepare a Phrase query. Its FTS index must store token positions. -+ FtsQueryContext prepare_fts_phrase_query( -+ const std::string& column, -+ const std::string& query, -+ uint32_t slop = 0, -+ FtsCoverageMode coverage_mode = FtsCoverageMode::Strict) const { -+ auto* context = lance_dataset_prepare_fts_phrase_query( -+ handle_.get(), column.c_str(), query.c_str(), slop, - static_cast(coverage_mode)); - if (!context) check_error(); - return FtsQueryContext(context); -diff --git a/src/fts_query.rs b/src/fts_query.rs -index cd194c7..3bf7f3e 100644 ---- a/src/fts_query.rs -+++ b/src/fts_query.rs -@@ -14,7 +14,9 @@ use lance_core::{Error, Result}; - use lance_index::IndexCriteria; - use lance_index::metrics::NoOpMetricsCollector; - use lance_index::scalar::FullTextSearchQuery; --use lance_index::scalar::inverted::query::{FtsQuery, collect_query_tokens}; -+use lance_index::scalar::inverted::query::{ -+ FtsQuery, MatchQuery, Operator, PhraseQuery, collect_query_tokens, -+}; - use lance_index::scalar::inverted::{InvertedIndex, MemBM25Scorer, build_global_bm25_scorer}; - use lance_table::format::IndexMetadata; - use uuid::Uuid; -@@ -48,12 +50,53 @@ impl TryFrom for LanceFtsCoverageMode { - } - } - -+/// Operator used to combine the analyzed terms of a Match query. -+#[repr(i32)] -+#[derive(Clone, Copy, Debug, PartialEq, Eq)] -+pub enum LanceFtsMatchOperator { -+ /// At least one analyzed term must match. -+ Or = 0, -+ /// Every analyzed term must match. -+ And = 1, -+} -+ -+impl TryFrom for LanceFtsMatchOperator { -+ type Error = Error; -+ -+ fn try_from(value: i32) -> Result { -+ match value { -+ 0 => Ok(Self::Or), -+ 1 => Ok(Self::And), -+ _ => Err(Error::invalid_input(format!( -+ "invalid match_operator {value}; expected 0 (OR) or 1 (AND)" -+ ))), -+ } -+ } -+} -+ -+impl From for Operator { -+ fn from(value: LanceFtsMatchOperator) -> Self { -+ match value { -+ LanceFtsMatchOperator::Or => Self::Or, -+ LanceFtsMatchOperator::And => Self::And, -+ } -+ } -+} -+ -+/// Query-specific state that must be shared by every segment-scoped scan. -+pub(crate) enum PreparedFtsQuery { -+ /// Exact Match queries share one corpus-wide scorer. -+ Match(Arc), -+ /// Phrase does not expand terms, so a shared global scorer is sufficient. -+ Phrase(Arc), -+} -+ - /// Rust-owned immutable state behind [`LanceFtsQueryContext`]. - pub(crate) struct FtsQueryContextInner { - pub(crate) dataset: Arc, - pub(crate) query: FullTextSearchQuery, - pub(crate) segments: Vec, -- pub(crate) scorer: Arc, -+ pub(crate) prepared: PreparedFtsQuery, - } - - impl FtsQueryContextInner { -@@ -87,7 +130,7 @@ fn invalid_input(message: impl Into) -> Error { - async fn prepare_fts_query_context( - dataset: Arc, - column: String, -- query_text: String, -+ query: FullTextSearchQuery, - coverage_mode: LanceFtsCoverageMode, - ) -> Result { - let logical_index = dataset -@@ -193,34 +236,78 @@ async fn prepare_fts_query_context( - ))); - } - -- let query = FullTextSearchQuery::new(query_text).with_column(column.clone())?; -- let match_query = match &query.query { -- FtsQuery::Match(query) => query, -+ let prepared = match &query.query { -+ FtsQuery::Match(match_query) => { -+ let mut tokenizer = indices[0].tokenizer(); -+ let query_tokens = collect_query_tokens(&match_query.terms, &mut tokenizer); -+ let params = query -+ .params() -+ .with_fuzziness(match_query.fuzziness) -+ .with_max_expansions(match_query.max_expansions) -+ .with_prefix_length(match_query.prefix_length); -+ PreparedFtsQuery::Match(Arc::new( -+ build_global_bm25_scorer(&indices, &query_tokens, ¶ms).await?, -+ )) -+ } -+ FtsQuery::Phrase(phrase_query) => { -+ if !expected_params.has_positions() { -+ return Err(invalid_input(format!( -+ "FTS index '{}' for column '{column}' does not store token positions required by Phrase queries; recreate the index with positions enabled", -+ logical_index.name -+ ))); -+ } -+ let mut tokenizer = indices[0].tokenizer(); -+ let query_tokens = collect_query_tokens(&phrase_query.terms, &mut tokenizer); -+ let params = query.params().with_phrase_slop(Some(phrase_query.slop)); -+ PreparedFtsQuery::Phrase(Arc::new( -+ build_global_bm25_scorer(&indices, &query_tokens, ¶ms).await?, -+ )) -+ } - _ => { - return Err(Error::internal( -- "prepared FTS query unexpectedly produced a non-Match query".to_string(), -+ "prepared FTS query must be a single-column Match or Phrase query".to_string(), - )); - } - }; -- let mut tokenizer = indices[0].tokenizer(); -- let query_tokens = collect_query_tokens(&match_query.terms, &mut tokenizer); -- let params = query -- .params() -- .with_fuzziness(match_query.fuzziness) -- .with_max_expansions(match_query.max_expansions) -- .with_prefix_length(match_query.prefix_length); -- let scorer = Arc::new(build_global_bm25_scorer(&indices, &query_tokens, ¶ms).await?); - - Ok(FtsQueryContextInner { - dataset, - query, - segments, -- scorer, -+ prepared, - }) - } - --/// Prepare a process-local global BM25 scorer and the committed segment list --/// for one single-column Match query against the dataset's pinned snapshot. -+unsafe fn parse_query_inputs( -+ dataset: *const LanceDataset, -+ column: *const c_char, -+ query: *const c_char, -+ coverage_mode: i32, -+) -> Result<(Arc, String, String, LanceFtsCoverageMode)> { -+ if dataset.is_null() || column.is_null() || query.is_null() { -+ return Err(invalid_input("dataset, column, and query must not be NULL")); -+ } -+ let column = unsafe { helpers::parse_c_string(column)? } -+ .filter(|value| !value.is_empty()) -+ .ok_or_else(|| invalid_input("column must not be empty"))? -+ .to_string(); -+ let query = unsafe { helpers::parse_c_string(query)? } -+ .filter(|value| !value.is_empty()) -+ .ok_or_else(|| invalid_input("query must not be empty"))? -+ .to_string(); -+ let coverage_mode = LanceFtsCoverageMode::try_from(coverage_mode)?; -+ let snapshot = unsafe { &*dataset }.snapshot(); -+ Ok((snapshot, column, query, coverage_mode)) -+} -+ -+fn into_context(inner: FtsQueryContextInner) -> *mut LanceFtsQueryContext { -+ Box::into_raw(Box::new(LanceFtsQueryContext { -+ inner: Arc::new(inner), -+ })) -+} -+ -+/// Compatibility API for an OR Match query. -+#[deprecated(note = "use lance_dataset_prepare_fts_match_query to select the Match operator")] - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_dataset_prepare_fts_query( - dataset: *const LanceDataset, -@@ -231,46 +318,120 @@ pub unsafe extern "C" fn lance_dataset_prepare_fts_query( - ) -> *mut LanceFtsQueryContext { - ffi_try!( - unsafe { -- prepare_fts_query_inner(dataset, column, query, max_fuzzy_distance, coverage_mode) -+ prepare_fts_match_query_inner( -+ dataset, -+ column, -+ query, -+ LanceFtsMatchOperator::Or as i32, -+ max_fuzzy_distance, -+ coverage_mode, -+ ) -+ }, -+ null -+ ) -+} -+ -+/// Prepare a process-local Match query context. AND and OR are supported. -+/// `max_fuzzy_distance` is retained for the future prepared-fuzzy path but -+/// must be zero with the currently pinned Lance revision. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_dataset_prepare_fts_match_query( -+ dataset: *const LanceDataset, -+ column: *const c_char, -+ query: *const c_char, -+ match_operator: i32, -+ max_fuzzy_distance: u32, -+ coverage_mode: i32, -+) -> *mut LanceFtsQueryContext { -+ ffi_try!( -+ unsafe { -+ prepare_fts_match_query_inner( -+ dataset, -+ column, -+ query, -+ match_operator, -+ max_fuzzy_distance, -+ coverage_mode, -+ ) - }, - null - ) - } - --unsafe fn prepare_fts_query_inner( -+unsafe fn prepare_fts_match_query_inner( - dataset: *const LanceDataset, - column: *const c_char, - query: *const c_char, -+ match_operator: i32, - max_fuzzy_distance: u32, - coverage_mode: i32, - ) -> Result<*mut LanceFtsQueryContext> { -- if dataset.is_null() || column.is_null() || query.is_null() { -- return Err(invalid_input("dataset, column, and query must not be NULL")); -- } -- let column = unsafe { helpers::parse_c_string(column)? } -- .filter(|value| !value.is_empty()) -- .ok_or_else(|| invalid_input("column must not be empty"))? -- .to_string(); -- let query = unsafe { helpers::parse_c_string(query)? } -- .filter(|value| !value.is_empty()) -- .ok_or_else(|| invalid_input("query must not be empty"))? -- .to_string(); -- let coverage_mode = LanceFtsCoverageMode::try_from(coverage_mode)?; -+ let (snapshot, column, query_text, coverage_mode) = -+ unsafe { parse_query_inputs(dataset, column, query, coverage_mode)? }; -+ let operator: Operator = LanceFtsMatchOperator::try_from(match_operator)?.into(); -+ // The parameter remains in the public API so callers do not need another -+ // ABI change when Lance-C moves to a Lance revision that can inject the -+ // same canonical fuzzy vocabulary into every segment-scoped scan. The -+ // pinned Lance revision can share only the scorer, so accepting fuzzy here -+ // would allow different segments to choose different capped expansions. - if max_fuzzy_distance != 0 { - return Err(invalid_input(format!( -- "max_fuzzy_distance must be 0 for prepared FTS query contexts, got {max_fuzzy_distance}; fuzzy queries require a canonical prepared BM25 vocabulary" -+ "max_fuzzy_distance must be 0 for prepared FTS with the pinned Lance revision, got {max_fuzzy_distance}; the parameter is reserved until canonical fuzzy vocabulary injection is available" - ))); - } -- let snapshot = unsafe { &*dataset }.snapshot(); -+ let query = FullTextSearchQuery::new_query( -+ MatchQuery::new(query_text) -+ .with_column(Some(column.clone())) -+ .with_operator(operator) -+ .with_fuzziness(Some(0)) -+ .into(), -+ ); - let inner = block_on(prepare_fts_query_context( - snapshot, - column, - query, - coverage_mode, - ))?; -- Ok(Box::into_raw(Box::new(LanceFtsQueryContext { -- inner: Arc::new(inner), -- }))) -+ Ok(into_context(inner)) -+} -+ -+/// Prepare a process-local Phrase query context. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_dataset_prepare_fts_phrase_query( -+ dataset: *const LanceDataset, -+ column: *const c_char, -+ query: *const c_char, -+ slop: u32, -+ coverage_mode: i32, -+) -> *mut LanceFtsQueryContext { -+ ffi_try!( -+ unsafe { prepare_fts_phrase_query_inner(dataset, column, query, slop, coverage_mode) }, -+ null -+ ) -+} -+ -+unsafe fn prepare_fts_phrase_query_inner( -+ dataset: *const LanceDataset, -+ column: *const c_char, -+ query: *const c_char, -+ slop: u32, -+ coverage_mode: i32, -+) -> Result<*mut LanceFtsQueryContext> { -+ let (snapshot, column, query_text, coverage_mode) = -+ unsafe { parse_query_inputs(dataset, column, query, coverage_mode)? }; -+ let query = FullTextSearchQuery::new_query( -+ PhraseQuery::new(query_text) -+ .with_column(Some(column.clone())) -+ .with_slop(slop) -+ .into(), -+ ); -+ let inner = block_on(prepare_fts_query_context( -+ snapshot, -+ column, -+ query, -+ coverage_mode, -+ ))?; -+ Ok(into_context(inner)) - } - - /// Close a context handle. NULL-safe. Scanners that already attached the -diff --git a/src/scanner.rs b/src/scanner.rs -index 0c29b17..cbf1b13 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -18,7 +18,7 @@ use lance::Dataset; - use lance::dataset::scanner::{ - DatasetRecordBatchStream, ExecutionStatsCallback, ExecutionSummaryCounts, - }; --use lance::io::exec::fts::{FlatMatchQueryExec, MatchQueryExec}; -+use lance::io::exec::fts::{FlatMatchQueryExec, MatchQueryExec, PhraseQueryExec}; - use lance_core::Result; - use lance_index::scalar::FullTextSearchQuery; - use lance_io::stream::RecordBatchStream; -@@ -33,7 +33,8 @@ use crate::error::{ - panic_payload_message, set_lance_error, set_last_error, swallow_unwind, - }; - use crate::fts_query::{ -- FtsQueryContextInner, LanceFtsQueryContext, clone_context, parse_segment_uuids, -+ FtsQueryContextInner, LanceFtsQueryContext, PreparedFtsQuery, clone_context, -+ parse_segment_uuids, - }; - use crate::helpers; - use crate::runtime::{RT, block_on}; -@@ -328,17 +329,17 @@ impl PreparedScanner { - let (plan, rewritten) = rewrite_prepared_fts_plan( - plan, - &distributed_fts.segments, -- &distributed_fts.context.scorer, -+ &distributed_fts.context.prepared, - selected_segments_have_current_fragments, - )?; -- if rewritten.match_query_execs > 1 -+ if rewritten.indexed_query_execs > 1 - || rewritten.flat_match_query_execs > 1 -- || rewritten.match_query_execs + rewritten.flat_match_query_execs == 0 -- || (selected_segments_have_current_fragments && rewritten.match_query_execs != 1) -+ || rewritten.indexed_query_execs + rewritten.flat_match_query_execs == 0 -+ || (selected_segments_have_current_fragments && rewritten.indexed_query_execs != 1) - { - return Err(lance_core::Error::internal(format!( -- "unexpected prepared FTS plan for selected segments with current fragment coverage {selected_segments_have_current_fragments}: rewrote {} MatchQueryExec node(s) and removed {} FlatMatchQueryExec node(s)", -- rewritten.match_query_execs, rewritten.flat_match_query_execs -+ "unexpected prepared FTS plan for selected segments with current fragment coverage {selected_segments_have_current_fragments}: rewrote {} indexed FTS query node(s) and removed {} FlatMatchQueryExec node(s)", -+ rewritten.indexed_query_execs, rewritten.flat_match_query_execs - ))); - } - let stream = lance_datafusion::exec::execute_plan( -@@ -420,14 +421,14 @@ fn segments_have_current_fragments( - - #[derive(Default)] - struct PreparedFtsPlanRewriteCounts { -- match_query_execs: usize, -+ indexed_query_execs: usize, - flat_match_query_execs: usize, - } - - fn rewrite_prepared_fts_plan( - plan: Arc, - segments: &[IndexMetadata], -- scorer: &Arc, -+ prepared: &PreparedFtsQuery, - selected_segments_have_current_fragments: bool, - ) -> Result<(Arc, PreparedFtsPlanRewriteCounts)> { - // Lance's ordinary FTS planner adds a flat-search branch for fragments not -@@ -438,7 +439,7 @@ fn rewrite_prepared_fts_plan( - return Ok(( - Arc::new(EmptyExec::new(plan.schema())), - PreparedFtsPlanRewriteCounts { -- match_query_execs: 0, -+ indexed_query_execs: 0, - flat_match_query_execs: 1, - }, - )); -@@ -454,11 +455,11 @@ fn rewrite_prepared_fts_plan( - let (new_child, child_rewritten) = rewrite_prepared_fts_plan( - Arc::clone(child), - segments, -- scorer, -+ prepared, - selected_segments_have_current_fragments, - )?; - new_children.push(new_child); -- rewritten.match_query_execs += child_rewritten.match_query_execs; -+ rewritten.indexed_query_execs += child_rewritten.indexed_query_execs; - rewritten.flat_match_query_execs += child_rewritten.flat_match_query_execs; - } - plan.with_new_children(new_children).map_err(|error| { -@@ -469,10 +470,15 @@ fn rewrite_prepared_fts_plan( - }; - - if let Some(exec) = rebuilt.downcast_ref::() { -- rewritten.match_query_execs += 1; -+ rewritten.indexed_query_execs += 1; - if !selected_segments_have_current_fragments { - return Ok((Arc::new(EmptyExec::new(rebuilt.schema())), rewritten)); - } -+ let PreparedFtsQuery::Match(scorer) = prepared else { -+ return Err(lance_core::Error::internal( -+ "prepared Phrase state cannot be attached to MatchQueryExec".to_string(), -+ )); -+ }; - let replacement = MatchQueryExec::new_with_segments( - Arc::clone(exec.dataset()), - exec.query().clone(), -@@ -483,6 +489,26 @@ fn rewrite_prepared_fts_plan( - .with_base_scorer(Arc::clone(scorer)); - return Ok((Arc::new(replacement), rewritten)); - } -+ if let Some(exec) = rebuilt.downcast_ref::() { -+ rewritten.indexed_query_execs += 1; -+ if !selected_segments_have_current_fragments { -+ return Ok((Arc::new(EmptyExec::new(rebuilt.schema())), rewritten)); -+ } -+ let PreparedFtsQuery::Phrase(scorer) = prepared else { -+ return Err(lance_core::Error::internal( -+ "prepared Match state cannot be attached to PhraseQueryExec".to_string(), -+ )); -+ }; -+ let replacement = PhraseQueryExec::new_with_segments( -+ Arc::clone(exec.dataset()), -+ exec.query().clone(), -+ exec.params().clone(), -+ exec.prefilter_source().clone(), -+ segments.to_vec(), -+ ) -+ .with_base_scorer(Arc::clone(scorer)); -+ return Ok((Arc::new(replacement), rewritten)); -+ } - Ok((rebuilt, rewritten)) - } - -@@ -2152,7 +2178,8 @@ mod tests { - use crate::dataset::{lance_dataset_close, lance_dataset_open}; - use crate::error::{lance_last_error_code, lance_last_error_message}; - use crate::fts_query::{ -- LanceFtsCoverageMode, lance_dataset_prepare_fts_query, lance_fts_query_context_close, -+ LanceFtsCoverageMode, LanceFtsMatchOperator, lance_dataset_prepare_fts_match_query, -+ lance_fts_query_context_close, - }; - use std::ffi::{CStr, CString}; - use std::sync::atomic::{AtomicI32, AtomicUsize}; -@@ -2276,10 +2303,11 @@ mod tests { - let column = CString::new("name").unwrap(); - let query = CString::new("a").unwrap(); - let context = unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - column.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::IndexOnly as i32, - ) -@@ -2293,7 +2321,6 @@ mod tests { - let prepared = unsafe { &*scanner }.build_scanner().unwrap(); - let distributed = prepared.distributed_fts.as_ref().unwrap(); - let segments = distributed.segments.clone(); -- let scorer = Arc::clone(&distributed.context.scorer); - let plan = block_on(prepared.scanner.create_plan()).unwrap(); - assert_eq!( - prepared_fts_plan_shape(&plan), -@@ -2303,9 +2330,14 @@ mod tests { - - let has_current_fragments = - segments_have_current_fragments(&distributed.context.dataset, &segments).unwrap(); -- let (rewritten, counts) = -- rewrite_prepared_fts_plan(plan, &segments, &scorer, has_current_fragments).unwrap(); -- assert_eq!(counts.match_query_execs, 1); -+ let (rewritten, counts) = rewrite_prepared_fts_plan( -+ plan, -+ &segments, -+ &distributed.context.prepared, -+ has_current_fragments, -+ ) -+ .unwrap(); -+ assert_eq!(counts.indexed_query_execs, 1); - assert_eq!(counts.flat_match_query_execs, 1); - assert_eq!(prepared_fts_plan_shape(&rewritten), (1, 0, 0)); - -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 8805764..a3e4ef7 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -5762,6 +5762,203 @@ fn load_fts_segment_uuids(uri: &str, column: &str) -> Vec<[u8; 16]> { - }) - } - -+#[test] -+#[allow(deprecated)] -+fn test_prepared_fts_match_phrase_and_legacy_compatibility() { -+ let tmp = tempfile::tempdir().unwrap(); -+ let uri = tmp -+ .path() -+ .join("prepared_fts_queries") -+ .to_str() -+ .unwrap() -+ .to_string(); -+ let schema = Arc::new(Schema::new(vec![ -+ Field::new("id", DataType::Int32, false), -+ Field::new("text", DataType::Utf8, false), -+ ])); -+ let batch = RecordBatch::try_new( -+ schema.clone(), -+ vec![ -+ Arc::new(Int32Array::from(vec![1, 2, 3, 4, 5])), -+ Arc::new(StringArray::from(vec![ -+ "quick brown fox", -+ "quick blue fox", -+ "slow brown fox", -+ "quik brown fox", -+ "quick red brown fox", -+ ])), -+ ], -+ ) -+ .unwrap(); -+ lance_c::runtime::block_on(async { -+ Dataset::write( -+ arrow::record_batch::RecordBatchIterator::new(vec![Ok(batch)], schema), -+ &uri, -+ None, -+ ) -+ .await -+ .unwrap(); -+ }); -+ -+ let uri_c = c_str(&uri); -+ let column = c_str("text"); -+ let index_params = -+ c_str(r#"{"base_tokenizer":"simple","language":"English","with_position":true}"#); -+ let dataset = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; -+ assert_eq!( -+ unsafe { -+ lance_dataset_create_scalar_index( -+ dataset, -+ column.as_ptr(), -+ ptr::null(), -+ LanceScalarIndexType::Inverted as i32, -+ index_params.as_ptr(), -+ false, -+ ) -+ }, -+ 0 -+ ); -+ -+ let query = c_str("quick brown"); -+ let exact_or = unsafe { -+ lance_dataset_prepare_fts_match_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, -+ 0, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(!exact_or.is_null()); -+ assert_eq!(collect_context_fts_scores(dataset, exact_or, None).len(), 5); -+ unsafe { lance_fts_query_context_close(exact_or) }; -+ -+ let legacy_or = unsafe { -+ lance_dataset_prepare_fts_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ 0, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(!legacy_or.is_null()); -+ assert_eq!( -+ collect_context_fts_scores(dataset, legacy_or, None).len(), -+ 5 -+ ); -+ unsafe { lance_fts_query_context_close(legacy_or) }; -+ -+ let exact_and = unsafe { -+ lance_dataset_prepare_fts_match_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ LanceFtsMatchOperator::And as i32, -+ 0, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(!exact_and.is_null()); -+ let exact_and_scores = collect_context_fts_scores(dataset, exact_and, None); -+ let mut exact_and_ids = exact_and_scores.keys().copied().collect::>(); -+ exact_and_ids.sort_unstable(); -+ assert_eq!(exact_and_ids, vec![1, 5]); -+ unsafe { lance_fts_query_context_close(exact_and) }; -+ -+ let fuzzy_and = unsafe { -+ lance_dataset_prepare_fts_match_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ LanceFtsMatchOperator::And as i32, -+ 1, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(fuzzy_and.is_null()); -+ let message = take_last_error_message(); -+ assert!( -+ message.contains("max_fuzzy_distance must be 0"), -+ "{message}" -+ ); -+ -+ let phrase = unsafe { -+ lance_dataset_prepare_fts_phrase_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ 0, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(!phrase.is_null(), "{}", take_last_error_message()); -+ let phrase_scores = collect_context_fts_scores(dataset, phrase, None); -+ assert_eq!(phrase_scores.keys().copied().collect::>(), vec![1]); -+ unsafe { lance_fts_query_context_close(phrase) }; -+ -+ let phrase_with_slop = unsafe { -+ lance_dataset_prepare_fts_phrase_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ 1, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(!phrase_with_slop.is_null(), "{}", take_last_error_message()); -+ let phrase_with_slop_scores = collect_context_fts_scores(dataset, phrase_with_slop, None); -+ let mut phrase_with_slop_ids = phrase_with_slop_scores.keys().copied().collect::>(); -+ phrase_with_slop_ids.sort_unstable(); -+ assert_eq!(phrase_with_slop_ids, vec![1, 5]); -+ unsafe { lance_fts_query_context_close(phrase_with_slop) }; -+ -+ unsafe { lance_dataset_close(dataset) }; -+} -+ -+#[test] -+fn test_prepared_fts_phrase_requires_positions() { -+ let (_tmp, uri) = create_test_dataset(); -+ let uri_c = c_str(&uri); -+ let column = c_str("name"); -+ let query = c_str("alice smith"); -+ let index_params = c_str(r#"{"base_tokenizer":"simple","language":"English"}"#); -+ let dataset = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; -+ assert_eq!( -+ unsafe { -+ lance_dataset_create_scalar_index( -+ dataset, -+ column.as_ptr(), -+ ptr::null(), -+ LanceScalarIndexType::Inverted as i32, -+ index_params.as_ptr(), -+ false, -+ ) -+ }, -+ 0 -+ ); -+ -+ let context = unsafe { -+ lance_dataset_prepare_fts_phrase_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ 0, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(context.is_null()); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ let message = take_last_error_message(); -+ assert!( -+ message.contains("does not store token positions"), -+ "{message}" -+ ); -+ -+ unsafe { lance_dataset_close(dataset) }; -+} -+ - #[test] - fn test_prepared_fts_row_id_output_is_explicit() { - let (_tmp, uri) = create_test_dataset(); -@@ -5785,10 +5982,11 @@ fn test_prepared_fts_row_id_output_is_explicit() { - 0 - ); - let context = unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - column.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::Strict as i32, - ) -@@ -5880,10 +6078,11 @@ fn test_prepare_fts_query_index_only_allows_unindexed_fragment() { - - let dataset = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; - let strict = unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - column.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::Strict as i32, - ) -@@ -5898,10 +6097,11 @@ fn test_prepare_fts_query_index_only_allows_unindexed_fragment() { - assert!(message.contains("unindexed fragments"), "{message}"); - - let context = unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - column.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::IndexOnly as i32, - ) -@@ -5976,10 +6176,11 @@ fn test_prepared_fts_index_only_empty_segment_returns_empty_shard() { - let query = c_str("alice"); - let dataset = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; - let context = unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - column.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::IndexOnly as i32, - ) -@@ -6075,10 +6276,11 @@ fn test_prepared_fts_global_scorer_is_shared_across_segment_splits() { - - let dataset = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; - let context = unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - column.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::Strict as i32, - ) -@@ -6186,7 +6388,7 @@ fn test_prepared_fts_global_scorer_is_shared_across_segment_splits() { - } - - #[test] --fn test_prepare_fts_query_rejects_null_empty_invalid_mode_and_fuzzy() { -+fn test_prepare_fts_queries_reject_invalid_inputs() { - let (_tmp, uri) = create_test_dataset(); - let uri_c = c_str(&uri); - let dataset = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; -@@ -6196,10 +6398,11 @@ fn test_prepare_fts_query_rejects_null_empty_invalid_mode_and_fuzzy() { - - assert!( - unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - ptr::null(), - column.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::Strict as i32, - ) -@@ -6208,10 +6411,11 @@ fn test_prepare_fts_query_rejects_null_empty_invalid_mode_and_fuzzy() { - ); - assert!( - unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - empty.as_ptr(), - query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, - 0, - LanceFtsCoverageMode::Strict as i32, - ) -@@ -6219,20 +6423,39 @@ fn test_prepare_fts_query_rejects_null_empty_invalid_mode_and_fuzzy() { - .is_null() - ); - assert!( -- unsafe { lance_dataset_prepare_fts_query(dataset, column.as_ptr(), empty.as_ptr(), 0, 0) } -- .is_null() -+ unsafe { -+ lance_dataset_prepare_fts_match_query( -+ dataset, -+ column.as_ptr(), -+ empty.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, -+ 0, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ } -+ .is_null() - ); - assert!( -- unsafe { lance_dataset_prepare_fts_query(dataset, column.as_ptr(), query.as_ptr(), 0, 99) } -- .is_null() -+ unsafe { -+ lance_dataset_prepare_fts_match_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ LanceFtsMatchOperator::Or as i32, -+ 0, -+ 99, -+ ) -+ } -+ .is_null() - ); - assert!( - unsafe { -- lance_dataset_prepare_fts_query( -+ lance_dataset_prepare_fts_match_query( - dataset, - column.as_ptr(), - query.as_ptr(), -- 1, -+ 99, -+ 0, - LanceFtsCoverageMode::Strict as i32, - ) - } -@@ -6244,9 +6467,18 @@ fn test_prepare_fts_query_rejects_null_empty_invalid_mode_and_fuzzy() { - .to_string_lossy() - .into_owned() - }; -+ assert!(message.contains("invalid match_operator"), "{message}"); - assert!( -- message.contains("max_fuzzy_distance must be 0"), -- "{message}" -+ unsafe { -+ lance_dataset_prepare_fts_phrase_query( -+ dataset, -+ column.as_ptr(), -+ ptr::null(), -+ 0, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ } -+ .is_null() - ); - let scanner = unsafe { lance_scanner_new(dataset, ptr::null(), ptr::null()) }; - assert_eq!( - -From 9bb38749c9de185664511def3cbeb699ee60a38b Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Thu, 3 Sep 2026 17:49:02 +0800 -Subject: [PATCH 2/2] update slop i32 - ---- - include/lance/lance.h | 6 +++--- - include/lance/lance.hpp | 5 +++-- - src/fts_query.rs | 6 ++++-- - tests/c_api_test.rs | 14 ++++++++++++++ - 4 files changed, 24 insertions(+), 7 deletions(-) - -diff --git a/include/lance/lance.h b/include/lance/lance.h -index 0c76edc..25c0430 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -1761,8 +1761,8 @@ LanceFtsQueryContext* lance_dataset_prepare_fts_match_query( - * Dataset identity, coverage, sharing, and segment-scoped execution follow the - * same contract as lance_dataset_prepare_fts_match_query(). - * -- * @param slop Maximum number of intervening token positions permitted between -- * adjacent phrase terms. -+ * @param slop Maximum non-negative number of intervening token positions -+ * permitted between adjacent phrase terms. - * @param coverage_mode Fixed-width LanceFtsCoverageMode discriminant. - * @return Context handle on success, or NULL on error. - */ -@@ -1770,7 +1770,7 @@ LanceFtsQueryContext* lance_dataset_prepare_fts_phrase_query( - const LanceDataset* dataset, - const char* column, - const char* query, -- uint32_t slop, -+ int32_t slop, - int32_t coverage_mode - ); - -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index 973216a..1e69a06 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -798,11 +798,12 @@ class Dataset { - return FtsQueryContext(context); - } - -- /// Prepare a Phrase query. Its FTS index must store token positions. -+ /// Prepare a Phrase query. Its FTS index must store token positions and -+ /// slop must be non-negative. - FtsQueryContext prepare_fts_phrase_query( - const std::string& column, - const std::string& query, -- uint32_t slop = 0, -+ int32_t slop = 0, - FtsCoverageMode coverage_mode = FtsCoverageMode::Strict) const { - auto* context = lance_dataset_prepare_fts_phrase_query( - handle_.get(), column.c_str(), query.c_str(), slop, -diff --git a/src/fts_query.rs b/src/fts_query.rs -index 3bf7f3e..c9d3a43 100644 ---- a/src/fts_query.rs -+++ b/src/fts_query.rs -@@ -401,7 +401,7 @@ pub unsafe extern "C" fn lance_dataset_prepare_fts_phrase_query( - dataset: *const LanceDataset, - column: *const c_char, - query: *const c_char, -- slop: u32, -+ slop: i32, - coverage_mode: i32, - ) -> *mut LanceFtsQueryContext { - ffi_try!( -@@ -414,9 +414,11 @@ unsafe fn prepare_fts_phrase_query_inner( - dataset: *const LanceDataset, - column: *const c_char, - query: *const c_char, -- slop: u32, -+ slop: i32, - coverage_mode: i32, - ) -> Result<*mut LanceFtsQueryContext> { -+ let slop = u32::try_from(slop) -+ .map_err(|_| invalid_input(format!("slop must be non-negative, got {slop}")))?; - let (snapshot, column, query_text, coverage_mode) = - unsafe { parse_query_inputs(dataset, column, query, coverage_mode)? }; - let query = FullTextSearchQuery::new_query( -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index a3e4ef7..9411e15 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -5914,6 +5914,20 @@ fn test_prepared_fts_match_phrase_and_legacy_compatibility() { - assert_eq!(phrase_with_slop_ids, vec![1, 5]); - unsafe { lance_fts_query_context_close(phrase_with_slop) }; - -+ let negative_phrase_slop = unsafe { -+ lance_dataset_prepare_fts_phrase_query( -+ dataset, -+ column.as_ptr(), -+ query.as_ptr(), -+ -1, -+ LanceFtsCoverageMode::Strict as i32, -+ ) -+ }; -+ assert!(negative_phrase_slop.is_null()); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ let message = take_last_error_message(); -+ assert!(message.contains("slop must be non-negative"), "{message}"); -+ - unsafe { lance_dataset_close(dataset) }; - } - diff --git a/thirdparty/patches/lance-c-0.1.9-pr-75-pr-78.patch b/thirdparty/patches/lance-c-0.1.9-pr-75-pr-78.patch deleted file mode 100644 index 08687627e7ea21..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-pr-75-pr-78.patch +++ /dev/null @@ -1,2441 +0,0 @@ -Lance-C v0.1.9 scanner options: PR #75 followed by PR #78. -Apply after lance-c-0.1.9-pr-74.patch, before lance-c-0.1.9-pr-77.patch. -The upstream mail patches below are concatenated without modification. - -PR #75: https://github.com/lance-format/lance-c/pull/75 -Head: 043a1f7eac253d8ac6be3f970b60fcbf615295aa -PR #78: https://github.com/lance-format/lance-c/pull/78 -Head: e894f591aef358cd36fdbf915c4d5b95fd0e8348 - -From 0752592cc4bbc69f8f9333362a00f3bceee73100 Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Fri, 4 Sep 2026 23:35:29 +0800 -Subject: [PATCH 1/2] add some setter - ---- - include/lance/lance.h | 81 +++++++++++ - include/lance/lance.hpp | 47 +++++++ - src/scanner.rs | 279 +++++++++++++++++++++++++++++++++++++ - tests/c_api_test.rs | 230 ++++++++++++++++++++++++++++++ - tests/cpp/test_cpp_api.cpp | 7 + - 5 files changed, 644 insertions(+) - -diff --git a/include/lance/lance.h b/include/lance/lance.h -index 3bf291f..00f415b 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -934,6 +934,74 @@ LanceScanner* lance_scanner_new( - int32_t lance_scanner_set_limit(LanceScanner* scanner, int64_t limit); - int32_t lance_scanner_set_offset(LanceScanner* scanner, int64_t offset); - int32_t lance_scanner_set_batch_size(LanceScanner* scanner, int64_t batch_size); -+ -+/** -+ * Set the target output batch size in bytes. -+ * -+ * When set, this takes precedence over the row-based batch size. The value -+ * must be greater than zero and must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_batch_size_bytes( -+ LanceScanner* scanner, -+ uint64_t batch_size_bytes -+); -+ -+/** -+ * Set the scanner I/O buffer size in bytes. -+ * -+ * The value must be greater than zero and must be set before scanning starts. -+ * This bounds buffered I/O received from storage, but is not a hard limit on -+ * all memory used by the scanner. -+ * -+ * @param scanner Scanner handle. Must not be NULL. -+ * @param io_buffer_size_bytes I/O buffer size in bytes. Must be greater than zero. -+ * @return 0 on success, -1 on error. -+ */ -+int32_t lance_scanner_set_io_buffer_size( -+ LanceScanner* scanner, -+ uint64_t io_buffer_size_bytes -+); -+ -+/** -+ * Set the maximum number of batches decoded concurrently. -+ * -+ * @param batch_readahead Number of in-flight batch decode tasks. Must be greater than zero. -+ */ -+int32_t lance_scanner_set_batch_readahead( -+ LanceScanner* scanner, -+ size_t batch_readahead -+); -+ -+/** -+ * Set fragment readahead for unordered scans. -+ * -+ * This setting is only used when scan-in-order is disabled. The value must be -+ * greater than zero. -+ */ -+int32_t lance_scanner_set_fragment_readahead( -+ LanceScanner* scanner, -+ size_t fragment_readahead -+); -+ -+/** -+ * Set the target number of physical execution partitions. -+ * -+ * This controls the partition count used by the physical optimizer and can be -+ * used to bound scan CPU parallelism. The value must be greater than zero and -+ * must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_target_parallelism( -+ LanceScanner* scanner, -+ size_t target_parallelism -+); -+ -+/** -+ * Configure whether batches are returned in storage order (default: true). -+ * -+ * Disabling ordering can improve throughput by returning batches as soon as -+ * they are ready. -+ */ -+int32_t lance_scanner_set_scan_in_order(LanceScanner* scanner, bool scan_in_order); - int32_t lance_scanner_with_row_id(LanceScanner* scanner, bool enable); - - /** -@@ -1656,6 +1724,19 @@ int32_t lance_scanner_nearest( - ); - - int32_t lance_scanner_set_nprobes(LanceScanner* scanner, uint32_t n); -+ -+/** -+ * Set vector index partition-search concurrency for each query. -+ * -+ * A value of -1 uses the CPU pool size, 0 selects Lance's automatic policy, -+ * 1 uses the sequential path, and values greater than 1 request parallel -+ * partition search. The effective value is capped by available parallelism. -+ * Values below -1 are rejected. Must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_query_parallelism( -+ LanceScanner* scanner, -+ int32_t query_parallelism -+); - int32_t lance_scanner_set_refine_factor(LanceScanner* scanner, uint32_t f); - int32_t lance_scanner_set_ef(LanceScanner* scanner, uint32_t e); - int32_t lance_scanner_set_metric(LanceScanner* scanner, LanceMetricType metric); -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index 6cf245f..3c03f86 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1196,6 +1196,48 @@ class Scanner { - return *this; - } - -+ /// Set the target output batch size in bytes. -+ Scanner& batch_size_bytes(uint64_t bytes) { -+ if (lance_scanner_set_batch_size_bytes(handle_.get(), bytes) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Set the scanner I/O buffer size in bytes. -+ Scanner& io_buffer_size(uint64_t bytes) { -+ if (lance_scanner_set_io_buffer_size(handle_.get(), bytes) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Set the number of batches decoded concurrently. -+ Scanner& batch_readahead(size_t batches) { -+ if (lance_scanner_set_batch_readahead(handle_.get(), batches) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Set fragment readahead for unordered scans. -+ Scanner& fragment_readahead(size_t fragments) { -+ if (lance_scanner_set_fragment_readahead(handle_.get(), fragments) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Set the target number of physical execution partitions. -+ Scanner& target_parallelism(size_t partitions) { -+ if (lance_scanner_set_target_parallelism(handle_.get(), partitions) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Configure whether batches are returned in storage order. -+ Scanner& scan_in_order(bool ordered = true) { -+ if (lance_scanner_set_scan_in_order(handle_.get(), ordered) != 0) -+ check_error(); -+ return *this; -+ } -+ - /// Enable/disable row ID in output. - Scanner& with_row_id(bool enable = true) { - if (lance_scanner_with_row_id(handle_.get(), enable) != 0) -@@ -1313,6 +1355,11 @@ class Scanner { - if (lance_scanner_set_nprobes(handle_.get(), n) != 0) check_error(); - return *this; - } -+ Scanner& query_parallelism(int32_t parallelism) { -+ if (lance_scanner_set_query_parallelism(handle_.get(), parallelism) != 0) -+ check_error(); -+ return *this; -+ } - Scanner& refine_factor(uint32_t f) { - if (lance_scanner_set_refine_factor(handle_.get(), f) != 0) check_error(); - return *this; -diff --git a/src/scanner.rs b/src/scanner.rs -index 0c29b17..ebedafc 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -60,11 +60,18 @@ pub struct LanceScanner { - limit: Option, - offset: Option, - batch_size: Option, -+ batch_size_bytes: Option, -+ io_buffer_size: Option, -+ batch_readahead: Option, -+ fragment_readahead: Option, -+ target_parallelism: Option, -+ scan_in_order: Option, - with_row_id: bool, - fragment_ids: Option>, - index_segments: Option>, - nearest: Option, - nprobes: Option, -+ query_parallelism: Option, - refine_factor: Option, - ef: Option, - metric_override: Option, -@@ -129,11 +136,18 @@ impl LanceScanner { - limit: None, - offset: None, - batch_size: None, -+ batch_size_bytes: None, -+ io_buffer_size: None, -+ batch_readahead: None, -+ fragment_readahead: None, -+ target_parallelism: None, -+ scan_in_order: None, - with_row_id: false, - fragment_ids: None, - index_segments: None, - nearest: None, - nprobes: None, -+ query_parallelism: None, - refine_factor: None, - ef: None, - metric_override: None, -@@ -164,6 +178,15 @@ impl LanceScanner { - Arc::clone(&self.poisoned) - } - -+ fn ensure_scan_not_started(&self, setting_name: &str) -> Result<()> { -+ if self.scan_started.load(Ordering::Acquire) { -+ return Err(lance_core::Error::invalid_input_source( -+ format!("{setting_name} must be set before the scan starts").into(), -+ )); -+ } -+ Ok(()) -+ } -+ - /// Apply fragment selection to a scanner builder if fragment_ids is set. - fn apply_fragment_filter(&self, scanner: &mut lance::dataset::scanner::Scanner) -> Result<()> { - if let Some(ids) = &self.fragment_ids { -@@ -231,6 +254,24 @@ impl LanceScanner { - if let Some(bs) = self.batch_size { - scanner.batch_size(bs); - } -+ if let Some(batch_size_bytes) = self.batch_size_bytes { -+ scanner.batch_size_bytes(batch_size_bytes); -+ } -+ if let Some(io_buffer_size) = self.io_buffer_size { -+ scanner.io_buffer_size(io_buffer_size); -+ } -+ if let Some(batch_readahead) = self.batch_readahead { -+ scanner.batch_readahead(batch_readahead); -+ } -+ if let Some(fragment_readahead) = self.fragment_readahead { -+ scanner.fragment_readahead(fragment_readahead); -+ } -+ if let Some(target_parallelism) = self.target_parallelism { -+ scanner.target_parallelism(target_parallelism); -+ } -+ if let Some(scan_in_order) = self.scan_in_order { -+ scanner.scan_in_order(scan_in_order); -+ } - if self.with_row_id { - scanner.with_row_id(); - } -@@ -260,6 +301,9 @@ impl LanceScanner { - if let Some(np) = self.nprobes { - scanner.nprobes(np as usize); - } -+ if let Some(query_parallelism) = self.query_parallelism { -+ scanner.query_parallelism(query_parallelism); -+ } - if let Some(rf) = self.refine_factor { - scanner.refine(rf); - } -@@ -757,6 +801,205 @@ unsafe fn scanner_set_batch_size_inner(scanner: *mut LanceScanner, batch_size: i - Ok(0) - } - -+/// Set the target output batch size in bytes. Returns 0 on success. -+/// -+/// The size must be greater than zero and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_batch_size_bytes( -+ scanner: *mut LanceScanner, -+ batch_size_bytes: u64, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_batch_size_bytes_inner(scanner, batch_size_bytes) -+ }) -+} -+ -+unsafe fn scanner_set_batch_size_bytes_inner( -+ scanner: *mut LanceScanner, -+ batch_size_bytes: u64, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if batch_size_bytes == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "batch_size_bytes must be greater than 0, got 0".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("batch_size_bytes")?; -+ scanner.batch_size_bytes = Some(batch_size_bytes); -+ Ok(0) -+} -+ -+/// Set the scanner I/O buffer size in bytes. Returns 0 on success. -+/// -+/// The size must be greater than zero and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_io_buffer_size( -+ scanner: *mut LanceScanner, -+ io_buffer_size_bytes: u64, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_io_buffer_size_inner(scanner, io_buffer_size_bytes) -+ }) -+} -+ -+unsafe fn scanner_set_io_buffer_size_inner( -+ scanner: *mut LanceScanner, -+ io_buffer_size_bytes: u64, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if io_buffer_size_bytes == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "io_buffer_size_bytes must be greater than 0, got 0".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("io_buffer_size_bytes")?; -+ scanner.io_buffer_size = Some(io_buffer_size_bytes); -+ Ok(0) -+} -+ -+/// Set the number of batches to decode concurrently. Returns 0 on success. -+/// -+/// The value must be greater than zero and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_batch_readahead( -+ scanner: *mut LanceScanner, -+ batch_readahead: usize, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_batch_readahead_inner(scanner, batch_readahead) -+ }) -+} -+ -+unsafe fn scanner_set_batch_readahead_inner( -+ scanner: *mut LanceScanner, -+ batch_readahead: usize, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if batch_readahead == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "batch_readahead must be greater than 0, got 0".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("batch_readahead")?; -+ scanner.batch_readahead = Some(batch_readahead); -+ Ok(0) -+} -+ -+/// Set the number of fragments to read ahead for unordered scans. -+/// -+/// The value must be greater than zero and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_fragment_readahead( -+ scanner: *mut LanceScanner, -+ fragment_readahead: usize, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_fragment_readahead_inner(scanner, fragment_readahead) -+ }) -+} -+ -+unsafe fn scanner_set_fragment_readahead_inner( -+ scanner: *mut LanceScanner, -+ fragment_readahead: usize, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if fragment_readahead == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "fragment_readahead must be greater than 0, got 0".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("fragment_readahead")?; -+ scanner.fragment_readahead = Some(fragment_readahead); -+ Ok(0) -+} -+ -+/// Set the target number of physical execution partitions. -+/// -+/// The value must be greater than zero and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_target_parallelism( -+ scanner: *mut LanceScanner, -+ target_parallelism: usize, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_target_parallelism_inner(scanner, target_parallelism) -+ }) -+} -+ -+unsafe fn scanner_set_target_parallelism_inner( -+ scanner: *mut LanceScanner, -+ target_parallelism: usize, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if target_parallelism == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "target_parallelism must be greater than 0, got 0".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("target_parallelism")?; -+ scanner.target_parallelism = Some(target_parallelism); -+ Ok(0) -+} -+ -+/// Configure whether scan results are returned in storage order. -+/// -+/// Must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_scan_in_order( -+ scanner: *mut LanceScanner, -+ scan_in_order: bool, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_scan_in_order_inner(scanner, scan_in_order) -+ }) -+} -+ -+unsafe fn scanner_set_scan_in_order_inner( -+ scanner: *mut LanceScanner, -+ scan_in_order: bool, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("scan_in_order")?; -+ scanner.scan_in_order = Some(scan_in_order); -+ Ok(0) -+} -+ - /// Enable or disable row ID in scan output. Returns 0. - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_with_row_id( -@@ -1766,6 +2009,42 @@ scanner_set_u32!(lance_scanner_set_nprobes, nprobes); - scanner_set_u32!(lance_scanner_set_refine_factor, refine_factor); - scanner_set_u32!(lance_scanner_set_ef, ef); - -+/// Set vector index partition-search concurrency for each query. -+/// -+/// `-1` uses the CPU pool size, `0` selects Lance's automatic policy, and -+/// positive values request that many workers. Values below `-1` are invalid. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_query_parallelism( -+ scanner: *mut LanceScanner, -+ query_parallelism: i32, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_query_parallelism_inner(scanner, query_parallelism) -+ }) -+} -+ -+unsafe fn scanner_set_query_parallelism_inner( -+ scanner: *mut LanceScanner, -+ query_parallelism: i32, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if query_parallelism < -1 { -+ return Err(lance_core::Error::invalid_input_source( -+ format!("query_parallelism must be -1, 0, or greater than 0, got {query_parallelism}") -+ .into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("query_parallelism")?; -+ scanner.query_parallelism = Some(query_parallelism); -+ Ok(0) -+} -+ - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_set_metric(scanner: *mut LanceScanner, metric: i32) -> i32 { - scanner_poison_check!(scanner, -1); -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 8805764..42f842d 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -1207,6 +1207,154 @@ fn test_scanner_batch_size() { - unsafe { lance_dataset_close(ds) }; - } - -+#[test] -+fn test_scanner_execution_tuning_options() { -+ let (_tmp, uri) = create_multi_fragment_dataset(); -+ let c_uri = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert!(!scanner.is_null()); -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_size_bytes(scanner, 1024) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_io_buffer_size(scanner, 64 * 1024) }, -+ 0 -+ ); -+ assert_eq!(unsafe { lance_scanner_set_batch_readahead(scanner, 1) }, 0); -+ assert_eq!( -+ unsafe { lance_scanner_set_fragment_readahead(scanner, 1) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_target_parallelism(scanner, 1) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_scan_in_order(scanner, false) }, -+ 0 -+ ); -+ -+ let mut ffi_stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut ffi_stream) }, -+ 0 -+ ); -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut ffi_stream) }.unwrap(); -+ let total_rows: usize = reader.map(|batch| batch.unwrap().num_rows()).sum(); -+ assert_eq!(total_rows, 10); -+ -+ unsafe { lance_scanner_close(scanner) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ -+#[test] -+fn test_scanner_execution_tuning_options_reject_zero() { -+ let (_tmp, uri) = create_test_dataset(); -+ let c_uri = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert!(!scanner.is_null()); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_size_bytes(scanner, 0) }, -+ -1 -+ ); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!(take_last_error_message().contains("batch_size_bytes must be greater than 0, got 0")); -+ -+ assert_eq!(unsafe { lance_scanner_set_io_buffer_size(scanner, 0) }, -1); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!( -+ take_last_error_message().contains("io_buffer_size_bytes must be greater than 0, got 0") -+ ); -+ -+ assert_eq!(unsafe { lance_scanner_set_batch_readahead(scanner, 0) }, -1); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!(take_last_error_message().contains("batch_readahead must be greater than 0, got 0")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_fragment_readahead(scanner, 0) }, -+ -1 -+ ); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!(take_last_error_message().contains("fragment_readahead must be greater than 0, got 0")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_target_parallelism(scanner, 0) }, -+ -1 -+ ); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!(take_last_error_message().contains("target_parallelism must be greater than 0, got 0")); -+ -+ unsafe { lance_scanner_close(scanner) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ -+#[test] -+fn test_scanner_execution_tuning_options_reject_after_scan_start() { -+ let (_tmp, uri) = create_test_dataset(); -+ let c_uri = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert!(!scanner.is_null()); -+ -+ let mut ffi_stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut ffi_stream) }, -+ 0 -+ ); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_size_bytes(scanner, 1024) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("batch_size_bytes must be set before")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_io_buffer_size(scanner, 64 * 1024) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("io_buffer_size_bytes must be set before")); -+ -+ assert_eq!(unsafe { lance_scanner_set_batch_readahead(scanner, 1) }, -1); -+ assert!(take_last_error_message().contains("batch_readahead must be set before")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_fragment_readahead(scanner, 1) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("fragment_readahead must be set before")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_target_parallelism(scanner, 1) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("target_parallelism must be set before")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_scan_in_order(scanner, false) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("scan_in_order must be set before")); -+ -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut ffi_stream) }.unwrap(); -+ assert_eq!( -+ reader.map(|batch| batch.unwrap().num_rows()).sum::(), -+ 5 -+ ); -+ -+ unsafe { lance_scanner_close(scanner) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ - // --------------------------------------------------------------------------- - // Combined filter + projection + limit - // --------------------------------------------------------------------------- -@@ -1441,6 +1589,34 @@ fn test_null_safety_comprehensive() { - unsafe { lance_scanner_set_batch_size(ptr::null_mut(), 10) }, - -1 - ); -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_size_bytes(ptr::null_mut(), 1024) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_io_buffer_size(ptr::null_mut(), 64 * 1024) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_readahead(ptr::null_mut(), 1) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_fragment_readahead(ptr::null_mut(), 1) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_target_parallelism(ptr::null_mut(), 1) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_query_parallelism(ptr::null_mut(), 1) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_scan_in_order(ptr::null_mut(), true) }, -+ -1 -+ ); - assert_eq!( - unsafe { lance_scanner_with_row_id(ptr::null_mut(), true) }, - -1 -@@ -5216,6 +5392,7 @@ fn test_scanner_nearest_with_ivf_pq_index() { - 10, - ); - lance_scanner_set_nprobes(scanner, 4); -+ assert_eq!(lance_scanner_set_query_parallelism(scanner, 4), 0); - } - - let mut stream = FFI_ArrowArrayStream::empty(); -@@ -5234,6 +5411,59 @@ fn test_scanner_nearest_with_ivf_pq_index() { - unsafe { lance_dataset_close(ds) }; - } - -+#[test] -+fn test_scanner_query_parallelism_validation_and_lifecycle() { -+ let (_tmp, uri) = create_test_dataset(); -+ let uri_c = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert!(!scanner.is_null()); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_query_parallelism(scanner, -1) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_query_parallelism(scanner, 0) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_query_parallelism(scanner, 2) }, -+ 0 -+ ); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_query_parallelism(scanner, -2) }, -+ -1 -+ ); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!( -+ take_last_error_message() -+ .contains("query_parallelism must be -1, 0, or greater than 0, got -2") -+ ); -+ -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut stream) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_query_parallelism(scanner, 1) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("query_parallelism must be set before")); -+ -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut stream) }.unwrap(); -+ assert_eq!( -+ reader.map(|batch| batch.unwrap().num_rows()).sum::(), -+ 5 -+ ); -+ -+ unsafe { lance_scanner_close(scanner) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ - #[test] - fn test_scanner_nearest_dim_mismatch() { - let (_tmp, uri) = create_vector_dataset(64, 8); -diff --git a/tests/cpp/test_cpp_api.cpp b/tests/cpp/test_cpp_api.cpp -index 17b1ab6..28762b8 100644 ---- a/tests/cpp/test_cpp_api.cpp -+++ b/tests/cpp/test_cpp_api.cpp -@@ -134,6 +134,12 @@ static void test_scanner_fluent(const std::string& uri) { - scanner.limit(5) - .offset(0) - .batch_size(2) -+ .batch_size_bytes(1024) -+ .io_buffer_size(64 * 1024) -+ .batch_readahead(1) -+ .fragment_readahead(1) -+ .target_parallelism(1) -+ .scan_in_order(false) - .statistics_callback(capture_scan_statistics, &captured); - - ArrowArrayStream stream; -@@ -370,6 +376,7 @@ static void test_nearest_smoke(const std::string& uri) { - try { - scanner.nearest("embedding", q, 8, 5) - .nprobes(2) -+ .query_parallelism(2) - .refine_factor(1) - .ef(50) - .metric(LANCE_METRIC_L2) - -From 043a1f7eac253d8ac6be3f970b60fcbf615295aa Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Sat, 5 Sep 2026 00:07:06 +0800 -Subject: [PATCH 2/2] add check - ---- - include/lance/lance.h | 8 ++++---- - include/lance/lance.hpp | 2 +- - src/scanner.rs | 11 ++++++++++- - tests/c_api_test.rs | 17 ++++++++++++++++- - 4 files changed, 31 insertions(+), 7 deletions(-) - -diff --git a/include/lance/lance.h b/include/lance/lance.h -index 00f415b..541134f 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -949,12 +949,12 @@ int32_t lance_scanner_set_batch_size_bytes( - /** - * Set the scanner I/O buffer size in bytes. - * -- * The value must be greater than zero and must be set before scanning starts. -- * This bounds buffered I/O received from storage, but is not a hard limit on -- * all memory used by the scanner. -+ * The value must be between 1 and INT64_MAX, inclusive, and must be set before -+ * scanning starts. This bounds buffered I/O received from storage, but is not -+ * a hard limit on all memory used by the scanner. - * - * @param scanner Scanner handle. Must not be NULL. -- * @param io_buffer_size_bytes I/O buffer size in bytes. Must be greater than zero. -+ * @param io_buffer_size_bytes I/O buffer size in bytes, in the range [1, INT64_MAX]. - * @return 0 on success, -1 on error. - */ - int32_t lance_scanner_set_io_buffer_size( -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index 3c03f86..a60dcd4 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1203,7 +1203,7 @@ class Scanner { - return *this; - } - -- /// Set the scanner I/O buffer size in bytes. -+ /// Set the scanner I/O buffer size in bytes, in the range [1, INT64_MAX]. - Scanner& io_buffer_size(uint64_t bytes) { - if (lance_scanner_set_io_buffer_size(handle_.get(), bytes) != 0) - check_error(); -diff --git a/src/scanner.rs b/src/scanner.rs -index ebedafc..6414cc0 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -837,7 +837,7 @@ unsafe fn scanner_set_batch_size_bytes_inner( - - /// Set the scanner I/O buffer size in bytes. Returns 0 on success. - /// --/// The size must be greater than zero and must be set before the scan starts. -+/// The size must be between 1 and [`i64::MAX`] and must be set before the scan starts. - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_set_io_buffer_size( - scanner: *mut LanceScanner, -@@ -863,6 +863,15 @@ unsafe fn scanner_set_io_buffer_size_inner( - "io_buffer_size_bytes must be greater than 0, got 0".into(), - )); - } -+ if io_buffer_size_bytes > i64::MAX as u64 { -+ return Err(lance_core::Error::invalid_input_source( -+ format!( -+ "io_buffer_size_bytes must be at most {}, got {io_buffer_size_bytes}", -+ i64::MAX -+ ) -+ .into(), -+ )); -+ } - let scanner = unsafe { &mut *scanner }; - scanner.ensure_scan_not_started("io_buffer_size_bytes")?; - scanner.io_buffer_size = Some(io_buffer_size_bytes); -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 42f842d..ffdd915 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -1252,7 +1252,7 @@ fn test_scanner_execution_tuning_options() { - } - - #[test] --fn test_scanner_execution_tuning_options_reject_zero() { -+fn test_scanner_execution_tuning_options_reject_invalid_values() { - let (_tmp, uri) = create_test_dataset(); - let c_uri = c_str(&uri); - let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -@@ -1274,6 +1274,21 @@ fn test_scanner_execution_tuning_options_reject_zero() { - take_last_error_message().contains("io_buffer_size_bytes must be greater than 0, got 0") - ); - -+ assert_eq!( -+ unsafe { lance_scanner_set_io_buffer_size(scanner, i64::MAX as u64) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_io_buffer_size(scanner, u64::MAX) }, -+ -1 -+ ); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!(take_last_error_message().contains(&format!( -+ "io_buffer_size_bytes must be at most {}, got {}", -+ i64::MAX, -+ u64::MAX -+ ))); -+ - assert_eq!(unsafe { lance_scanner_set_batch_readahead(scanner, 0) }, -1); - assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); - assert!(take_last_error_message().contains("batch_readahead must be greater than 0, got 0")); -From 057135cdd4ac6ac7b5348a1832caf7081a8541fb Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Mon, 7 Sep 2026 12:35:45 +0800 -Subject: [PATCH 1/2] feat: expose scanner use_scalar_index options - ---- - include/lance/lance.h | 83 +++++++++ - include/lance/lance.hpp | 47 ++++++ - src/scanner.rs | 335 +++++++++++++++++++++++++++++++++++++ - tests/c_api_test.rs | 296 ++++++++++++++++++++++++++++++++ - tests/cpp/test_cpp_api.cpp | 8 + - 5 files changed, 769 insertions(+) - -diff --git a/include/lance/lance.h b/include/lance/lance.h -index 31213da..6ace2cb 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -147,6 +147,13 @@ typedef enum { - LANCE_METRIC_HAMMING = 3, - } LanceMetricType; - -+/** Speed / accuracy tradeoff for approximate vector search. */ -+typedef enum { -+ LANCE_APPROX_MODE_FAST = 0, -+ LANCE_APPROX_MODE_NORMAL = 1, -+ LANCE_APPROX_MODE_ACCURATE = 2, -+} LanceApproxMode; -+ - typedef enum { - LANCE_DTYPE_FLOAT32 = 0, - LANCE_DTYPE_FLOAT16 = 1, -@@ -1002,8 +1009,54 @@ int32_t lance_scanner_set_target_parallelism( - * they are ready. - */ - int32_t lance_scanner_set_scan_in_order(LanceScanner* scanner, bool scan_in_order); -+ -+/** -+ * Configure whether scalar indices may be used to optimize filters. -+ * -+ * Scalar indices are enabled by default. Disable this to force filter -+ * evaluation without scalar indices. This setting is independent of -+ * `lance_scanner_set_use_index`, which controls vector ANN index usage. -+ * Must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_use_scalar_index( -+ LanceScanner* scanner, -+ bool use_scalar_index -+); -+ -+/** -+ * Configure whether row-based output batches are strict. -+ * -+ * When enabled, every batch except the last has exactly the configured row -+ * batch size. This may require copying and cannot be combined with a byte-based -+ * batch-size limit. Must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_strict_batch_size( -+ LanceScanner* scanner, -+ bool strict_batch_size -+); -+ -+/** -+ * Configure whether file statistics may optimize the scan (default: true). -+ * Intended primarily for debugging and benchmarking. Must be set before -+ * scanning starts. -+ */ -+int32_t lance_scanner_set_use_stats(LanceScanner* scanner, bool use_stats); -+ - int32_t lance_scanner_with_row_id(LanceScanner* scanner, bool enable); - -+/** Include or omit the `_rowaddr` metadata column. Must be set before scanning. */ -+int32_t lance_scanner_with_row_address(LanceScanner* scanner, bool enable); -+ -+/** -+ * Configure whether deleted rows still present in storage are returned. -+ * Deleted rows have a NULL `_rowid`; callers should also enable row IDs. -+ * Must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_include_deleted_rows( -+ LanceScanner* scanner, -+ bool include_deleted_rows -+); -+ - /** - * Restrict scan to the given fragment IDs. Must be called before iteration. - * @param ids Array of fragment IDs -@@ -1725,6 +1778,36 @@ int32_t lance_scanner_nearest( - - int32_t lance_scanner_set_nprobes(LanceScanner* scanner, uint32_t n); - -+/** -+ * Set the minimum number of vector-index partitions to search. -+ * Must be greater than zero and no greater than `maximum_nprobes` when set. -+ * Must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_minimum_nprobes( -+ LanceScanner* scanner, -+ uint32_t minimum_nprobes -+); -+ -+/** -+ * Set the maximum number of vector-index partitions to search. -+ * Must be greater than zero and no less than `minimum_nprobes` when set. -+ * This only affects prefiltered searches that need more candidates. -+ * Must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_maximum_nprobes( -+ LanceScanner* scanner, -+ uint32_t maximum_nprobes -+); -+ -+/** -+ * Configure the speed / accuracy tradeoff for approximate vector search. -+ * Must be set before scanning starts. -+ */ -+int32_t lance_scanner_set_approx_mode( -+ LanceScanner* scanner, -+ LanceApproxMode approx_mode -+); -+ - /** - * Set vector index partition-search concurrency for each query. - * -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index 404d2df..e08f76a 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1269,6 +1269,27 @@ class Scanner { - return *this; - } - -+ /// Configure whether scalar indices may be used to optimize filters. -+ Scanner& use_scalar_index(bool enable = true) { -+ if (lance_scanner_set_use_scalar_index(handle_.get(), enable) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Configure whether row-based output batches are strict. -+ Scanner& strict_batch_size(bool strict_batch_size = true) { -+ if (lance_scanner_set_strict_batch_size(handle_.get(), strict_batch_size) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Configure whether file statistics may optimize the scan. -+ Scanner& use_stats(bool use_stats = true) { -+ if (lance_scanner_set_use_stats(handle_.get(), use_stats) != 0) -+ check_error(); -+ return *this; -+ } -+ - /// Enable/disable row ID in output. - Scanner& with_row_id(bool enable = true) { - if (lance_scanner_with_row_id(handle_.get(), enable) != 0) -@@ -1276,6 +1297,20 @@ class Scanner { - return *this; - } - -+ /// Include or omit the `_rowaddr` metadata column. -+ Scanner& with_row_address(bool enable = true) { -+ if (lance_scanner_with_row_address(handle_.get(), enable) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ /// Configure whether deleted rows still present in storage are returned. -+ Scanner& include_deleted_rows(bool include_deleted_rows = true) { -+ if (lance_scanner_set_include_deleted_rows(handle_.get(), include_deleted_rows) != 0) -+ check_error(); -+ return *this; -+ } -+ - /// Restrict scan to specific fragment IDs. - Scanner& fragment_ids(const uint64_t* ids, size_t len) { - if (lance_scanner_set_fragment_ids(handle_.get(), ids, len) != 0) -@@ -1386,6 +1421,18 @@ class Scanner { - if (lance_scanner_set_nprobes(handle_.get(), n) != 0) check_error(); - return *this; - } -+ Scanner& minimum_nprobes(uint32_t minimum_nprobes) { -+ if (lance_scanner_set_minimum_nprobes(handle_.get(), minimum_nprobes) != 0) check_error(); -+ return *this; -+ } -+ Scanner& maximum_nprobes(uint32_t maximum_nprobes) { -+ if (lance_scanner_set_maximum_nprobes(handle_.get(), maximum_nprobes) != 0) check_error(); -+ return *this; -+ } -+ Scanner& approx_mode(LanceApproxMode approx_mode) { -+ if (lance_scanner_set_approx_mode(handle_.get(), approx_mode) != 0) check_error(); -+ return *this; -+ } - Scanner& query_parallelism(int32_t parallelism) { - if (lance_scanner_set_query_parallelism(handle_.get(), parallelism) != 0) - check_error(); -diff --git a/src/scanner.rs b/src/scanner.rs -index d3ef3be..53cd5b6 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -21,6 +21,7 @@ use lance::dataset::scanner::{ - use lance::io::exec::fts::{FlatMatchQueryExec, MatchQueryExec, PhraseQueryExec}; - use lance_core::Result; - use lance_index::scalar::FullTextSearchQuery; -+use lance_index::vector::ApproxMode; - use lance_io::stream::RecordBatchStream; - use lance_table::format::IndexMetadata; - use uuid::Uuid; -@@ -51,6 +52,38 @@ pub enum LanceDataType { - Int8 = 4, - } - -+/// Speed / accuracy tradeoff for approximate vector search, mirroring the C -+/// enum `LanceApproxMode`. -+#[repr(i32)] -+#[derive(Clone, Copy, Debug, PartialEq, Eq)] -+pub enum LanceApproxMode { -+ Fast = 0, -+ Normal = 1, -+ Accurate = 2, -+} -+ -+impl LanceApproxMode { -+ fn from_i32(value: i32) -> Result { -+ match value { -+ 0 => Ok(Self::Fast), -+ 1 => Ok(Self::Normal), -+ 2 => Ok(Self::Accurate), -+ _ => Err(lance_core::Error::invalid_input_source( -+ format!("approx_mode must be 0 (FAST), 1 (NORMAL), or 2 (ACCURATE), got {value}") -+ .into(), -+ )), -+ } -+ } -+ -+ fn to_approx_mode(self) -> ApproxMode { -+ match self { -+ Self::Fast => ApproxMode::Fast, -+ Self::Normal => ApproxMode::Normal, -+ Self::Accurate => ApproxMode::Accurate, -+ } -+ } -+} -+ - /// Opaque scanner handle. Stores configuration until stream materialization. - pub struct LanceScanner { - dataset: Arc, -@@ -62,16 +95,24 @@ pub struct LanceScanner { - offset: Option, - batch_size: Option, - batch_size_bytes: Option, -+ strict_batch_size: Option, - io_buffer_size: Option, - batch_readahead: Option, - fragment_readahead: Option, - target_parallelism: Option, - scan_in_order: Option, -+ use_scalar_index: Option, -+ use_stats: Option, - with_row_id: bool, -+ with_row_address: bool, -+ include_deleted_rows: bool, - fragment_ids: Option>, - index_segments: Option>, - nearest: Option, - nprobes: Option, -+ minimum_nprobes: Option, -+ maximum_nprobes: Option, -+ approx_mode: Option, - query_parallelism: Option, - refine_factor: Option, - ef: Option, -@@ -138,16 +179,24 @@ impl LanceScanner { - offset: None, - batch_size: None, - batch_size_bytes: None, -+ strict_batch_size: None, - io_buffer_size: None, - batch_readahead: None, - fragment_readahead: None, - target_parallelism: None, - scan_in_order: None, -+ use_scalar_index: None, -+ use_stats: None, - with_row_id: false, -+ with_row_address: false, -+ include_deleted_rows: false, - fragment_ids: None, - index_segments: None, - nearest: None, - nprobes: None, -+ minimum_nprobes: None, -+ maximum_nprobes: None, -+ approx_mode: None, - query_parallelism: None, - refine_factor: None, - ef: None, -@@ -258,6 +307,9 @@ impl LanceScanner { - if let Some(batch_size_bytes) = self.batch_size_bytes { - scanner.batch_size_bytes(batch_size_bytes); - } -+ if let Some(strict_batch_size) = self.strict_batch_size { -+ scanner.strict_batch_size(strict_batch_size); -+ } - if let Some(io_buffer_size) = self.io_buffer_size { - scanner.io_buffer_size(io_buffer_size); - } -@@ -273,9 +325,21 @@ impl LanceScanner { - if let Some(scan_in_order) = self.scan_in_order { - scanner.scan_in_order(scan_in_order); - } -+ if let Some(use_scalar_index) = self.use_scalar_index { -+ scanner.use_scalar_index(use_scalar_index); -+ } -+ if let Some(use_stats) = self.use_stats { -+ scanner.use_stats(use_stats); -+ } - if self.with_row_id { - scanner.with_row_id(); - } -+ if self.with_row_address { -+ scanner.with_row_address(); -+ } -+ if self.include_deleted_rows { -+ scanner.include_deleted_rows(); -+ } - self.apply_fragment_filter(&mut scanner)?; - if self.index_segments.is_some() && self.nearest.is_none() { - return Err(lance_core::Error::invalid_input_source( -@@ -302,6 +366,15 @@ impl LanceScanner { - if let Some(np) = self.nprobes { - scanner.nprobes(np as usize); - } -+ if let Some(minimum_nprobes) = self.minimum_nprobes { -+ scanner.minimum_nprobes(minimum_nprobes as usize); -+ } -+ if let Some(maximum_nprobes) = self.maximum_nprobes { -+ scanner.maximum_nprobes(maximum_nprobes as usize); -+ } -+ if let Some(approx_mode) = self.approx_mode { -+ scanner.approx_mode(approx_mode.to_approx_mode()); -+ } - if let Some(query_parallelism) = self.query_parallelism { - scanner.query_parallelism(query_parallelism); - } -@@ -1035,6 +1108,92 @@ unsafe fn scanner_set_scan_in_order_inner( - Ok(0) - } - -+/// Configure whether scalar indices may be used to optimize filters. -+/// -+/// Scalar indices are enabled by default in Lance. Must be set before the scan -+/// starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_use_scalar_index( -+ scanner: *mut LanceScanner, -+ use_scalar_index: bool, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_use_scalar_index_inner(scanner, use_scalar_index) -+ }) -+} -+ -+unsafe fn scanner_set_use_scalar_index_inner( -+ scanner: *mut LanceScanner, -+ use_scalar_index: bool, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("use_scalar_index")?; -+ scanner.use_scalar_index = Some(use_scalar_index); -+ Ok(0) -+} -+ -+/// Configure whether output batches use the exact row-based batch size. -+/// -+/// Must be set before the scan starts. Lance rejects enabling this together -+/// with a byte-based batch-size limit when the scan is materialized. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_strict_batch_size( -+ scanner: *mut LanceScanner, -+ strict_batch_size: bool, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_strict_batch_size_inner(scanner, strict_batch_size) -+ }) -+} -+ -+unsafe fn scanner_set_strict_batch_size_inner( -+ scanner: *mut LanceScanner, -+ strict_batch_size: bool, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("strict_batch_size")?; -+ scanner.strict_batch_size = Some(strict_batch_size); -+ Ok(0) -+} -+ -+/// Configure whether file statistics may be used to optimize the scan. -+/// -+/// Statistics are enabled by default. Must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_use_stats( -+ scanner: *mut LanceScanner, -+ use_stats: bool, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_use_stats_inner(scanner, use_stats) -+ }) -+} -+ -+unsafe fn scanner_set_use_stats_inner(scanner: *mut LanceScanner, use_stats: bool) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("use_stats")?; -+ scanner.use_stats = Some(use_stats); -+ Ok(0) -+} -+ - /// Enable or disable row ID in scan output. Returns 0. - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_with_row_id( -@@ -1058,6 +1217,62 @@ unsafe fn scanner_with_row_id_inner(scanner: *mut LanceScanner, enable: bool) -> - Ok(0) - } - -+/// Enable or disable the `_rowaddr` metadata column in scan output. -+/// -+/// Must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_with_row_address( -+ scanner: *mut LanceScanner, -+ enable: bool, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_with_row_address_inner(scanner, enable) -+ }) -+} -+ -+unsafe fn scanner_with_row_address_inner(scanner: *mut LanceScanner, enable: bool) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("with_row_address")?; -+ scanner.with_row_address = enable; -+ Ok(0) -+} -+ -+/// Configure whether deleted rows still present in storage are returned. -+/// -+/// Deleted rows have a NULL `_rowid`, so callers should also enable row IDs. -+/// Must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_include_deleted_rows( -+ scanner: *mut LanceScanner, -+ include_deleted_rows: bool, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_include_deleted_rows_inner(scanner, include_deleted_rows) -+ }) -+} -+ -+unsafe fn scanner_set_include_deleted_rows_inner( -+ scanner: *mut LanceScanner, -+ include_deleted_rows: bool, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("include_deleted_rows")?; -+ scanner.include_deleted_rows = include_deleted_rows; -+ Ok(0) -+} -+ - /// Restrict the scan to the given fragment IDs. - /// Must be called before any iteration method. - /// -@@ -2044,6 +2259,126 @@ scanner_set_u32!(lance_scanner_set_nprobes, nprobes); - scanner_set_u32!(lance_scanner_set_refine_factor, refine_factor); - scanner_set_u32!(lance_scanner_set_ef, ef); - -+/// Set the minimum number of vector-index partitions to search. -+/// -+/// The value must be greater than zero, no greater than a configured -+/// `maximum_nprobes`, and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_minimum_nprobes( -+ scanner: *mut LanceScanner, -+ minimum_nprobes: u32, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_minimum_nprobes_inner(scanner, minimum_nprobes) -+ }) -+} -+ -+unsafe fn scanner_set_minimum_nprobes_inner( -+ scanner: *mut LanceScanner, -+ minimum_nprobes: u32, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if minimum_nprobes == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "minimum_nprobes must be greater than 0, got 0".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("minimum_nprobes")?; -+ if let Some(maximum_nprobes) = scanner.maximum_nprobes -+ && minimum_nprobes > maximum_nprobes -+ { -+ return Err(lance_core::Error::invalid_input_source( -+ format!( -+ "minimum_nprobes ({minimum_nprobes}) must not exceed maximum_nprobes ({maximum_nprobes})" -+ ) -+ .into(), -+ )); -+ } -+ scanner.minimum_nprobes = Some(minimum_nprobes); -+ Ok(0) -+} -+ -+/// Set the maximum number of vector-index partitions to search. -+/// -+/// The value must be greater than zero, no less than a configured -+/// `minimum_nprobes`, and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_maximum_nprobes( -+ scanner: *mut LanceScanner, -+ maximum_nprobes: u32, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_maximum_nprobes_inner(scanner, maximum_nprobes) -+ }) -+} -+ -+unsafe fn scanner_set_maximum_nprobes_inner( -+ scanner: *mut LanceScanner, -+ maximum_nprobes: u32, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ if maximum_nprobes == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "maximum_nprobes must be greater than 0, got 0".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("maximum_nprobes")?; -+ if let Some(minimum_nprobes) = scanner.minimum_nprobes -+ && maximum_nprobes < minimum_nprobes -+ { -+ return Err(lance_core::Error::invalid_input_source( -+ format!( -+ "maximum_nprobes ({maximum_nprobes}) must not be less than minimum_nprobes ({minimum_nprobes})" -+ ) -+ .into(), -+ )); -+ } -+ scanner.maximum_nprobes = Some(maximum_nprobes); -+ Ok(0) -+} -+ -+/// Configure the speed / accuracy tradeoff for approximate vector search. -+/// -+/// Must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_approx_mode( -+ scanner: *mut LanceScanner, -+ approx_mode: i32, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_approx_mode_inner(scanner, approx_mode) -+ }) -+} -+ -+unsafe fn scanner_set_approx_mode_inner( -+ scanner: *mut LanceScanner, -+ approx_mode: i32, -+) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let approx_mode = LanceApproxMode::from_i32(approx_mode)?; -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("approx_mode")?; -+ scanner.approx_mode = Some(approx_mode); -+ Ok(0) -+} -+ - /// Set vector index partition-search concurrency for each query. - /// - /// `-1` uses the CPU pool size, `0` selects Lance's automatic policy, and -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index bde742d..a550a93 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -1237,6 +1237,12 @@ fn test_scanner_execution_tuning_options() { - unsafe { lance_scanner_set_scan_in_order(scanner, false) }, - 0 - ); -+ assert_eq!( -+ unsafe { lance_scanner_set_use_scalar_index(scanner, false) }, -+ 0 -+ ); -+ assert_eq!(unsafe { lance_scanner_set_use_stats(scanner, false) }, 0); -+ assert_eq!(unsafe { lance_scanner_with_row_address(scanner, true) }, 0); - - let mut ffi_stream = FFI_ArrowArrayStream::empty(); - assert_eq!( -@@ -1244,6 +1250,7 @@ fn test_scanner_execution_tuning_options() { - 0 - ); - let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut ffi_stream) }.unwrap(); -+ assert!(reader.schema().field_with_name("_rowaddr").is_ok()); - let total_rows: usize = reader.map(|batch| batch.unwrap().num_rows()).sum(); - assert_eq!(total_rows, 10); - -@@ -1360,6 +1367,30 @@ fn test_scanner_execution_tuning_options_reject_after_scan_start() { - ); - assert!(take_last_error_message().contains("scan_in_order must be set before")); - -+ assert_eq!( -+ unsafe { lance_scanner_set_use_scalar_index(scanner, false) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("use_scalar_index must be set before")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_strict_batch_size(scanner, true) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("strict_batch_size must be set before")); -+ -+ assert_eq!(unsafe { lance_scanner_set_use_stats(scanner, false) }, -1); -+ assert!(take_last_error_message().contains("use_stats must be set before")); -+ -+ assert_eq!(unsafe { lance_scanner_with_row_address(scanner, true) }, -1); -+ assert!(take_last_error_message().contains("with_row_address must be set before")); -+ -+ assert_eq!( -+ unsafe { lance_scanner_set_include_deleted_rows(scanner, true) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("include_deleted_rows must be set before")); -+ - let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut ffi_stream) }.unwrap(); - assert_eq!( - reader.map(|batch| batch.unwrap().num_rows()).sum::(), -@@ -1370,6 +1401,79 @@ fn test_scanner_execution_tuning_options_reject_after_scan_start() { - unsafe { lance_dataset_close(ds) }; - } - -+#[test] -+fn test_scanner_strict_batch_size_across_fragments() { -+ let (_tmp, uri) = create_multi_fragment_dataset(); -+ let c_uri = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert!(!scanner.is_null()); -+ assert_eq!(unsafe { lance_scanner_set_batch_size(scanner, 3) }, 0); -+ assert_eq!( -+ unsafe { lance_scanner_set_strict_batch_size(scanner, true) }, -+ 0 -+ ); -+ -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut stream) }, -+ 0 -+ ); -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut stream) }.unwrap(); -+ let batch_sizes = reader -+ .map(|batch| batch.unwrap().num_rows()) -+ .collect::>(); -+ assert_eq!(batch_sizes, vec![3, 3, 3, 1]); -+ -+ unsafe { lance_scanner_close(scanner) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ -+#[test] -+fn test_scanner_include_deleted_rows() { -+ let (_tmp, uri) = create_multi_fragment_dataset(); -+ let c_uri = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ -+ let predicate = c_str("id >= 8"); -+ let mut num_deleted = 0; -+ assert_eq!( -+ unsafe { lance_dataset_delete(ds, predicate.as_ptr(), &mut num_deleted) }, -+ 0 -+ ); -+ assert_eq!(num_deleted, 2); -+ -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert!(!scanner.is_null()); -+ assert_eq!(unsafe { lance_scanner_with_row_id(scanner, true) }, 0); -+ assert_eq!( -+ unsafe { lance_scanner_set_include_deleted_rows(scanner, true) }, -+ 0 -+ ); -+ -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut stream) }, -+ 0 -+ ); -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut stream) }.unwrap(); -+ let batches = reader.map(|batch| batch.unwrap()).collect::>(); -+ assert_eq!(batches.iter().map(RecordBatch::num_rows).sum::(), 10); -+ assert_eq!( -+ batches -+ .iter() -+ .map(|batch| batch.column_by_name("_rowid").unwrap().null_count()) -+ .sum::(), -+ 2 -+ ); -+ -+ unsafe { lance_scanner_close(scanner) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ - // --------------------------------------------------------------------------- - // Combined filter + projection + limit - // --------------------------------------------------------------------------- -@@ -1632,10 +1736,42 @@ fn test_null_safety_comprehensive() { - unsafe { lance_scanner_set_scan_in_order(ptr::null_mut(), true) }, - -1 - ); -+ assert_eq!( -+ unsafe { lance_scanner_set_use_scalar_index(ptr::null_mut(), false) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_strict_batch_size(ptr::null_mut(), true) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_use_stats(ptr::null_mut(), false) }, -+ -1 -+ ); - assert_eq!( - unsafe { lance_scanner_with_row_id(ptr::null_mut(), true) }, - -1 - ); -+ assert_eq!( -+ unsafe { lance_scanner_with_row_address(ptr::null_mut(), true) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_include_deleted_rows(ptr::null_mut(), true) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_minimum_nprobes(ptr::null_mut(), 1) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_maximum_nprobes(ptr::null_mut(), 1) }, -+ -1 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_approx_mode(ptr::null_mut(), LanceApproxMode::Normal as i32,) }, -+ -1 -+ ); - - // Scanner iteration with NULL. - let mut ffi_stream2 = FFI_ArrowArrayStream::empty(); -@@ -3077,6 +3213,90 @@ fn test_create_scalar_index_btree() { - unsafe { lance_dataset_close(ds) }; - } - -+#[test] -+fn test_scanner_set_use_scalar_index_controls_filter_planning() { -+ let (_tmp, uri) = create_test_dataset(); -+ let uri_c = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ -+ let column = c_str("id"); -+ assert_eq!( -+ unsafe { -+ lance_dataset_create_scalar_index( -+ ds, -+ column.as_ptr(), -+ ptr::null(), -+ LanceScalarIndexType::BTree as i32, -+ ptr::null(), -+ false, -+ ) -+ }, -+ 0 -+ ); -+ -+ let run_scan = |use_scalar_index: bool| { -+ let filter = c_str("id = 3"); -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), filter.as_ptr()) }; -+ assert!(!scanner.is_null()); -+ assert_eq!( -+ unsafe { lance_scanner_set_use_scalar_index(scanner, use_scalar_index) }, -+ 0 -+ ); -+ -+ let mut captured = CapturedScanStatistics::default(); -+ assert_eq!( -+ unsafe { -+ lance_scanner_set_statistics_callback( -+ scanner, -+ Some(capture_scan_statistics), -+ (&mut captured as *mut CapturedScanStatistics).cast(), -+ ) -+ }, -+ 0 -+ ); -+ -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut stream) }, -+ 0 -+ ); -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut stream) }.unwrap(); -+ let ids = reader -+ .flat_map(|batch| { -+ let batch = batch.unwrap(); -+ batch -+ .column_by_name("id") -+ .unwrap() -+ .as_any() -+ .downcast_ref::() -+ .unwrap() -+ .values() -+ .to_vec() -+ }) -+ .collect::>(); -+ -+ assert_eq!(captured.calls, 1); -+ unsafe { lance_scanner_close(scanner) }; -+ (ids, captured) -+ }; -+ -+ let (indexed_ids, indexed_statistics) = run_scan(true); -+ let (unindexed_ids, unindexed_statistics) = run_scan(false); -+ assert_eq!(indexed_ids, vec![3]); -+ assert_eq!(unindexed_ids, indexed_ids); -+ assert!( -+ indexed_statistics.indices_loaded > 0, -+ "enabled scan should load the scalar index" -+ ); -+ assert_eq!( -+ unindexed_statistics.indices_loaded, 0, -+ "disabled scan should bypass the scalar index" -+ ); -+ -+ unsafe { lance_dataset_close(ds) }; -+} -+ - #[test] - fn test_scalar_index_segment_build_is_fragment_scoped_and_uncommitted() { - let (_tmp, uri) = create_many_small_fragments(2); -@@ -5409,6 +5629,12 @@ fn test_scanner_nearest_with_ivf_pq_index() { - 10, - ); - lance_scanner_set_nprobes(scanner, 4); -+ assert_eq!(lance_scanner_set_minimum_nprobes(scanner, 2), 0); -+ assert_eq!(lance_scanner_set_maximum_nprobes(scanner, 6), 0); -+ assert_eq!( -+ lance_scanner_set_approx_mode(scanner, LanceApproxMode::Accurate as i32), -+ 0 -+ ); - assert_eq!(lance_scanner_set_query_parallelism(scanner, 4), 0); - } - -@@ -5428,6 +5654,76 @@ fn test_scanner_nearest_with_ivf_pq_index() { - unsafe { lance_dataset_close(ds) }; - } - -+#[test] -+fn test_scanner_adaptive_nprobes_and_approx_mode_validation_and_lifecycle() { -+ let (_tmp, uri) = create_test_dataset(); -+ let uri_c = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(uri_c.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert!(!scanner.is_null()); -+ -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 0) }, -1); -+ assert!(take_last_error_message().contains("minimum_nprobes must be greater than 0, got 0")); -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 0) }, -1); -+ assert!(take_last_error_message().contains("maximum_nprobes must be greater than 0, got 0")); -+ assert_eq!(unsafe { lance_scanner_set_approx_mode(scanner, 3) }, -1); -+ assert!( -+ take_last_error_message() -+ .contains("approx_mode must be 0 (FAST), 1 (NORMAL), or 2 (ACCURATE), got 3") -+ ); -+ -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 2) }, 0); -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 3) }, -1); -+ assert!( -+ take_last_error_message() -+ .contains("minimum_nprobes (3) must not exceed maximum_nprobes (2)") -+ ); -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 1) }, 0); -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 1) }, 0); -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 2) }, -1); -+ assert!( -+ take_last_error_message() -+ .contains("minimum_nprobes (2) must not exceed maximum_nprobes (1)") -+ ); -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 2) }, 0); -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 2) }, 0); -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 1) }, -1); -+ assert!( -+ take_last_error_message() -+ .contains("maximum_nprobes (1) must not be less than minimum_nprobes (2)") -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_approx_mode(scanner, LanceApproxMode::Fast as i32) }, -+ 0 -+ ); -+ -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut stream) }, -+ 0 -+ ); -+ -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 1) }, -1); -+ assert!(take_last_error_message().contains("minimum_nprobes must be set before")); -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 1) }, -1); -+ assert!(take_last_error_message().contains("maximum_nprobes must be set before")); -+ assert_eq!( -+ unsafe { lance_scanner_set_approx_mode(scanner, LanceApproxMode::Normal as i32) }, -+ -1 -+ ); -+ assert!(take_last_error_message().contains("approx_mode must be set before")); -+ -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut stream) }.unwrap(); -+ assert_eq!( -+ reader.map(|batch| batch.unwrap().num_rows()).sum::(), -+ 5 -+ ); -+ -+ unsafe { lance_scanner_close(scanner) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ - #[test] - fn test_scanner_query_parallelism_validation_and_lifecycle() { - let (_tmp, uri) = create_test_dataset(); -diff --git a/tests/cpp/test_cpp_api.cpp b/tests/cpp/test_cpp_api.cpp -index 28762b8..60b8fc4 100644 ---- a/tests/cpp/test_cpp_api.cpp -+++ b/tests/cpp/test_cpp_api.cpp -@@ -140,6 +140,11 @@ static void test_scanner_fluent(const std::string& uri) { - .fragment_readahead(1) - .target_parallelism(1) - .scan_in_order(false) -+ .use_scalar_index(false) -+ .strict_batch_size(false) -+ .use_stats(false) -+ .with_row_address(true) -+ .include_deleted_rows(false) - .statistics_callback(capture_scan_statistics, &captured); - - ArrowArrayStream stream; -@@ -376,6 +381,9 @@ static void test_nearest_smoke(const std::string& uri) { - try { - scanner.nearest("embedding", q, 8, 5) - .nprobes(2) -+ .minimum_nprobes(1) -+ .maximum_nprobes(2) -+ .approx_mode(LANCE_APPROX_MODE_NORMAL) - .query_parallelism(2) - .refine_factor(1) - .ef(50) - -From e894f591aef358cd36fdbf915c4d5b95fd0e8348 Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Mon, 7 Sep 2026 14:27:10 +0800 -Subject: [PATCH 2/2] update - ---- - include/lance/lance.h | 18 +++- - include/lance/lance.hpp | 7 +- - src/scanner.rs | 210 +++++++++++++++++++++++++++++++--------- - tests/c_api_test.rs | 77 +++++++++++++++ - 4 files changed, 261 insertions(+), 51 deletions(-) - -diff --git a/include/lance/lance.h b/include/lance/lance.h -index 6ace2cb..8173ae5 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -946,7 +946,8 @@ int32_t lance_scanner_set_batch_size(LanceScanner* scanner, int64_t batch_size); - * Set the target output batch size in bytes. - * - * When set, this takes precedence over the row-based batch size. The value -- * must be greater than zero and must be set before scanning starts. -+ * must be greater than zero and must be set before scanning starts. The call -+ * is rejected without changing scanner state if strict batch sizing is enabled. - */ - int32_t lance_scanner_set_batch_size_bytes( - LanceScanner* scanner, -@@ -1028,7 +1029,8 @@ int32_t lance_scanner_set_use_scalar_index( - * - * When enabled, every batch except the last has exactly the configured row - * batch size. This may require copying and cannot be combined with a byte-based -- * batch-size limit. Must be set before scanning starts. -+ * batch-size limit. The call is rejected without changing scanner state if a -+ * byte limit is already set. Must be set before scanning starts. - */ - int32_t lance_scanner_set_strict_batch_size( - LanceScanner* scanner, -@@ -1776,11 +1778,19 @@ int32_t lance_scanner_nearest( - uint32_t k - ); - --int32_t lance_scanner_set_nprobes(LanceScanner* scanner, uint32_t n); -+/** -+ * Set both the minimum and maximum vector-index partition-search bounds. -+ * -+ * This replaces both bounds configured by earlier calls to any nprobes -+ * setter. The value must be greater than zero. Must be set before scanning. -+ */ -+int32_t lance_scanner_set_nprobes(LanceScanner* scanner, uint32_t nprobes); - - /** - * Set the minimum number of vector-index partitions to search. -+ * This replaces only the minimum bound; the current maximum is preserved. - * Must be greater than zero and no greater than `maximum_nprobes` when set. -+ * An invalid resulting range is rejected without changing either bound. - * Must be set before scanning starts. - */ - int32_t lance_scanner_set_minimum_nprobes( -@@ -1790,8 +1800,10 @@ int32_t lance_scanner_set_minimum_nprobes( - - /** - * Set the maximum number of vector-index partitions to search. -+ * This replaces only the maximum bound; the current minimum is preserved. - * Must be greater than zero and no less than `minimum_nprobes` when set. - * This only affects prefiltered searches that need more candidates. -+ * An invalid resulting range is rejected without changing either bound. - * Must be set before scanning starts. - */ - int32_t lance_scanner_set_maximum_nprobes( -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index e08f76a..c12c0c6 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1417,14 +1417,17 @@ class Scanner { - return *this; - } - -- Scanner& nprobes(uint32_t n) { -- if (lance_scanner_set_nprobes(handle_.get(), n) != 0) check_error(); -+ /// Replace both minimum and maximum partition-search bounds. -+ Scanner& nprobes(uint32_t nprobes) { -+ if (lance_scanner_set_nprobes(handle_.get(), nprobes) != 0) check_error(); - return *this; - } -+ /// Replace only the minimum partition-search bound. - Scanner& minimum_nprobes(uint32_t minimum_nprobes) { - if (lance_scanner_set_minimum_nprobes(handle_.get(), minimum_nprobes) != 0) check_error(); - return *this; - } -+ /// Replace only the maximum partition-search bound. - Scanner& maximum_nprobes(uint32_t maximum_nprobes) { - if (lance_scanner_set_maximum_nprobes(handle_.get(), maximum_nprobes) != 0) check_error(); - return *this; -diff --git a/src/scanner.rs b/src/scanner.rs -index 53cd5b6..4ceeb0e 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -109,9 +109,7 @@ pub struct LanceScanner { - fragment_ids: Option>, - index_segments: Option>, - nearest: Option, -- nprobes: Option, -- minimum_nprobes: Option, -- maximum_nprobes: Option, -+ nprobes: NprobesRange, - approx_mode: Option, - query_parallelism: Option, - refine_factor: Option, -@@ -148,6 +146,72 @@ struct NearestQuery { - k: u32, - } - -+/// The effective adaptive partition-search range shared by all three nprobes -+/// setters. Updates are computed and validated before replacing this state. -+#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -+struct NprobesRange { -+ minimum: Option, -+ maximum: Option, -+} -+ -+impl NprobesRange { -+ fn exact(nprobes: u32) -> Result { -+ if nprobes == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "nprobes must be greater than 0, got 0".into(), -+ )); -+ } -+ Ok(Self { -+ minimum: Some(nprobes), -+ maximum: Some(nprobes), -+ }) -+ } -+ -+ fn with_minimum(self, minimum_nprobes: u32) -> Result { -+ if minimum_nprobes == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "minimum_nprobes must be greater than 0, got 0".into(), -+ )); -+ } -+ if let Some(maximum_nprobes) = self.maximum -+ && minimum_nprobes > maximum_nprobes -+ { -+ return Err(lance_core::Error::invalid_input_source( -+ format!( -+ "minimum_nprobes ({minimum_nprobes}) must not exceed maximum_nprobes ({maximum_nprobes})" -+ ) -+ .into(), -+ )); -+ } -+ Ok(Self { -+ minimum: Some(minimum_nprobes), -+ ..self -+ }) -+ } -+ -+ fn with_maximum(self, maximum_nprobes: u32) -> Result { -+ if maximum_nprobes == 0 { -+ return Err(lance_core::Error::invalid_input_source( -+ "maximum_nprobes must be greater than 0, got 0".into(), -+ )); -+ } -+ if let Some(minimum_nprobes) = self.minimum -+ && maximum_nprobes < minimum_nprobes -+ { -+ return Err(lance_core::Error::invalid_input_source( -+ format!( -+ "maximum_nprobes ({maximum_nprobes}) must not be less than minimum_nprobes ({minimum_nprobes})" -+ ) -+ .into(), -+ )); -+ } -+ Ok(Self { -+ maximum: Some(maximum_nprobes), -+ ..self -+ }) -+ } -+} -+ - /// Poll status for `lance_scanner_poll_next`. - #[repr(C)] - #[derive(Debug, PartialEq, Eq)] -@@ -193,9 +257,7 @@ impl LanceScanner { - fragment_ids: None, - index_segments: None, - nearest: None, -- nprobes: None, -- minimum_nprobes: None, -- maximum_nprobes: None, -+ nprobes: NprobesRange::default(), - approx_mode: None, - query_parallelism: None, - refine_factor: None, -@@ -363,13 +425,10 @@ impl LanceScanner { - } - if let Some(n) = &self.nearest { - scanner.nearest(&n.column, n.query.as_ref(), n.k as usize)?; -- if let Some(np) = self.nprobes { -- scanner.nprobes(np as usize); -- } -- if let Some(minimum_nprobes) = self.minimum_nprobes { -+ if let Some(minimum_nprobes) = self.nprobes.minimum { - scanner.minimum_nprobes(minimum_nprobes as usize); - } -- if let Some(maximum_nprobes) = self.maximum_nprobes { -+ if let Some(maximum_nprobes) = self.nprobes.maximum { - scanner.maximum_nprobes(maximum_nprobes as usize); - } - if let Some(approx_mode) = self.approx_mode { -@@ -930,6 +989,14 @@ unsafe fn scanner_set_batch_size_bytes_inner( - } - let scanner = unsafe { &mut *scanner }; - scanner.ensure_scan_not_started("batch_size_bytes")?; -+ if scanner.strict_batch_size == Some(true) { -+ return Err(lance_core::Error::invalid_input_source( -+ format!( -+ "strict_batch_size=true cannot be combined with batch_size_bytes={batch_size_bytes}" -+ ) -+ .into(), -+ )); -+ } - scanner.batch_size_bytes = Some(batch_size_bytes); - Ok(0) - } -@@ -1140,8 +1207,8 @@ unsafe fn scanner_set_use_scalar_index_inner( - - /// Configure whether output batches use the exact row-based batch size. - /// --/// Must be set before the scan starts. Lance rejects enabling this together --/// with a byte-based batch-size limit when the scan is materialized. -+/// Must be set before the scan starts. Enabling this together with a -+/// byte-based batch-size limit is rejected without changing scanner state. - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_set_strict_batch_size( - scanner: *mut LanceScanner, -@@ -1164,6 +1231,14 @@ unsafe fn scanner_set_strict_batch_size_inner( - } - let scanner = unsafe { &mut *scanner }; - scanner.ensure_scan_not_started("strict_batch_size")?; -+ if strict_batch_size && let Some(batch_size_bytes) = scanner.batch_size_bytes { -+ return Err(lance_core::Error::invalid_input_source( -+ format!( -+ "strict_batch_size=true cannot be combined with batch_size_bytes={batch_size_bytes}" -+ ) -+ .into(), -+ )); -+ } - scanner.strict_batch_size = Some(strict_batch_size); - Ok(0) - } -@@ -2255,10 +2330,38 @@ macro_rules! scanner_set_u32 { - }; - } - --scanner_set_u32!(lance_scanner_set_nprobes, nprobes); - scanner_set_u32!(lance_scanner_set_refine_factor, refine_factor); - scanner_set_u32!(lance_scanner_set_ef, ef); - -+/// Set both vector-index partition-search bounds to the same value. -+/// -+/// This replaces any values previously configured through -+/// `minimum_nprobes` or `maximum_nprobes`. The value must be greater than zero -+/// and must be set before the scan starts. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_nprobes( -+ scanner: *mut LanceScanner, -+ nprobes: u32, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ scanner_set_nprobes_inner(scanner, nprobes) -+ }) -+} -+ -+unsafe fn scanner_set_nprobes_inner(scanner: *mut LanceScanner, nprobes: u32) -> Result { -+ if scanner.is_null() { -+ return Err(lance_core::Error::invalid_input_source( -+ "scanner is NULL".into(), -+ )); -+ } -+ let scanner = unsafe { &mut *scanner }; -+ scanner.ensure_scan_not_started("nprobes")?; -+ let next = NprobesRange::exact(nprobes)?; -+ scanner.nprobes = next; -+ Ok(0) -+} -+ - /// Set the minimum number of vector-index partitions to search. - /// - /// The value must be greater than zero, no greater than a configured -@@ -2283,24 +2386,10 @@ unsafe fn scanner_set_minimum_nprobes_inner( - "scanner is NULL".into(), - )); - } -- if minimum_nprobes == 0 { -- return Err(lance_core::Error::invalid_input_source( -- "minimum_nprobes must be greater than 0, got 0".into(), -- )); -- } - let scanner = unsafe { &mut *scanner }; - scanner.ensure_scan_not_started("minimum_nprobes")?; -- if let Some(maximum_nprobes) = scanner.maximum_nprobes -- && minimum_nprobes > maximum_nprobes -- { -- return Err(lance_core::Error::invalid_input_source( -- format!( -- "minimum_nprobes ({minimum_nprobes}) must not exceed maximum_nprobes ({maximum_nprobes})" -- ) -- .into(), -- )); -- } -- scanner.minimum_nprobes = Some(minimum_nprobes); -+ let next = scanner.nprobes.with_minimum(minimum_nprobes)?; -+ scanner.nprobes = next; - Ok(0) - } - -@@ -2328,24 +2417,10 @@ unsafe fn scanner_set_maximum_nprobes_inner( - "scanner is NULL".into(), - )); - } -- if maximum_nprobes == 0 { -- return Err(lance_core::Error::invalid_input_source( -- "maximum_nprobes must be greater than 0, got 0".into(), -- )); -- } - let scanner = unsafe { &mut *scanner }; - scanner.ensure_scan_not_started("maximum_nprobes")?; -- if let Some(minimum_nprobes) = scanner.minimum_nprobes -- && maximum_nprobes < minimum_nprobes -- { -- return Err(lance_core::Error::invalid_input_source( -- format!( -- "maximum_nprobes ({maximum_nprobes}) must not be less than minimum_nprobes ({minimum_nprobes})" -- ) -- .into(), -- )); -- } -- scanner.maximum_nprobes = Some(maximum_nprobes); -+ let next = scanner.nprobes.with_maximum(maximum_nprobes)?; -+ scanner.nprobes = next; - Ok(0) - } - -@@ -2885,6 +2960,49 @@ mod tests { - ) - } - -+ #[test] -+ fn nprobes_setters_share_one_validated_range() { -+ let (_tmp, uri) = create_test_dataset(); -+ let (dataset, scanner) = open_dataset_and_scanner(&uri); -+ let assert_range = |minimum, maximum| { -+ assert_eq!( -+ unsafe { &*scanner }.nprobes, -+ NprobesRange { minimum, maximum } -+ ); -+ }; -+ -+ assert_range(None, None); -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 2) }, 0); -+ assert_range(Some(2), None); -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 5) }, 0); -+ assert_range(Some(2), Some(5)); -+ -+ // The combined setter replaces both bounds. -+ assert_eq!(unsafe { lance_scanner_set_nprobes(scanner, 4) }, 0); -+ assert_range(Some(4), Some(4)); -+ -+ // A failed partial update leaves both bounds unchanged. -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 5) }, -1); -+ assert_range(Some(4), Some(4)); -+ -+ // Widening the maximum first makes the new minimum valid. -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 6) }, 0); -+ assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 5) }, 0); -+ assert_range(Some(5), Some(6)); -+ -+ // A later combined call deterministically replaces the widened range. -+ assert_eq!(unsafe { lance_scanner_set_nprobes(scanner, 3) }, 0); -+ assert_range(Some(3), Some(3)); -+ assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 2) }, -1); -+ assert_eq!(unsafe { lance_scanner_set_nprobes(scanner, 0) }, -1); -+ assert_range(Some(3), Some(3)); -+ -+ unsafe { -+ lance_scanner_close(scanner); -+ lance_dataset_close(dataset); -+ } -+ } -+ - #[test] - fn prepared_fts_index_only_plan_does_not_scan_indexed_fragment_row_ids() { - let (_tmp, uri) = create_test_dataset(); -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index a550a93..3b3424b 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -1318,6 +1318,78 @@ fn test_scanner_execution_tuning_options_reject_invalid_values() { - unsafe { lance_dataset_close(ds) }; - } - -+#[test] -+fn test_scanner_strict_batch_size_and_bytes_conflict_is_recoverable() { -+ let (_tmp, uri) = create_test_dataset(); -+ let c_uri = c_str(&uri); -+ let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -+ assert!(!ds.is_null()); -+ -+ let consume = |scanner| { -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ unsafe { lance_scanner_to_arrow_stream(scanner, &mut stream) }, -+ 0 -+ ); -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut stream) }.unwrap(); -+ assert_eq!( -+ reader.map(|batch| batch.unwrap().num_rows()).sum::(), -+ 5 -+ ); -+ }; -+ -+ // A byte limit already exists: strict=true is rejected without starting -+ // the scan or replacing the prior strict setting. -+ let bytes_first = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_size_bytes(bytes_first, 1024) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_strict_batch_size(bytes_first, true) }, -+ -1 -+ ); -+ assert!( -+ take_last_error_message() -+ .contains("strict_batch_size=true cannot be combined with batch_size_bytes=1024") -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_use_stats(bytes_first, false) }, -+ 0, -+ "the rejected setter must not mark the scan as started" -+ ); -+ consume(bytes_first); -+ -+ // Strict sizing already exists: the byte limit is rejected without -+ // mutation. The caller can disable strict sizing and retry on this handle. -+ let strict_first = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; -+ assert_eq!( -+ unsafe { lance_scanner_set_strict_batch_size(strict_first, true) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_size_bytes(strict_first, 1024) }, -+ -1 -+ ); -+ assert!( -+ take_last_error_message() -+ .contains("strict_batch_size=true cannot be combined with batch_size_bytes=1024") -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_strict_batch_size(strict_first, false) }, -+ 0 -+ ); -+ assert_eq!( -+ unsafe { lance_scanner_set_batch_size_bytes(strict_first, 1024) }, -+ 0 -+ ); -+ consume(strict_first); -+ -+ unsafe { lance_scanner_close(bytes_first) }; -+ unsafe { lance_scanner_close(strict_first) }; -+ unsafe { lance_dataset_close(ds) }; -+} -+ - #[test] - fn test_scanner_execution_tuning_options_reject_after_scan_start() { - let (_tmp, uri) = create_test_dataset(); -@@ -1760,6 +1832,7 @@ fn test_null_safety_comprehensive() { - unsafe { lance_scanner_set_include_deleted_rows(ptr::null_mut(), true) }, - -1 - ); -+ assert_eq!(unsafe { lance_scanner_set_nprobes(ptr::null_mut(), 1) }, -1); - assert_eq!( - unsafe { lance_scanner_set_minimum_nprobes(ptr::null_mut(), 1) }, - -1 -@@ -5663,6 +5736,8 @@ fn test_scanner_adaptive_nprobes_and_approx_mode_validation_and_lifecycle() { - let scanner = unsafe { lance_scanner_new(ds, ptr::null(), ptr::null()) }; - assert!(!scanner.is_null()); - -+ assert_eq!(unsafe { lance_scanner_set_nprobes(scanner, 0) }, -1); -+ assert!(take_last_error_message().contains("nprobes must be greater than 0, got 0")); - assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 0) }, -1); - assert!(take_last_error_message().contains("minimum_nprobes must be greater than 0, got 0")); - assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 0) }, -1); -@@ -5704,6 +5779,8 @@ fn test_scanner_adaptive_nprobes_and_approx_mode_validation_and_lifecycle() { - 0 - ); - -+ assert_eq!(unsafe { lance_scanner_set_nprobes(scanner, 1) }, -1); -+ assert!(take_last_error_message().contains("nprobes must be set before")); - assert_eq!(unsafe { lance_scanner_set_minimum_nprobes(scanner, 1) }, -1); - assert!(take_last_error_message().contains("minimum_nprobes must be set before")); - assert_eq!(unsafe { lance_scanner_set_maximum_nprobes(scanner, 1) }, -1); diff --git a/thirdparty/patches/lance-c-0.1.9-pr-77.patch b/thirdparty/patches/lance-c-0.1.9-pr-77.patch deleted file mode 100644 index 341a1c692dcebd..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-pr-77.patch +++ /dev/null @@ -1,1863 +0,0 @@ -From eaf06c0374e62de6b519f55efe17de0c58e88c0a Mon Sep 17 00:00:00 2001 -From: "jianjian.xie" -Date: Fri, 4 Sep 2026 22:04:55 -0700 -Subject: [PATCH] build: bump lance to v11.0.0 - -Summary: -Intent: -- Move the lance git pins from e934cc2c to ab6b5bbe (lance v11.0.0 release tag) - so lance-c tracks a released upstream version instead of an arbitrary commit. -- Pick up the v11 blob APIs (read_blob_ranges, Option-based take_blobs results) - needed to answer #76 without a second pin bump. - -Changes: -- Point all lance, lance-core, lance-file, lance-index, lance-io, lance-linalg, - lance-table, lance-datafusion, and lance-datagen dependencies at ab6b5bbe. -- Re-resolve Cargo.lock; blake3, jiff, and reqwest 0.13 were unlocked explicitly - because v11 raised their minimum versions, the rest follows from lance v11 - (opendal 0.58, lance-namespace-reqwest-client 0.11, etc.). -- Adapt to upstream signature changes: build_global_bm25_scorer takes an optional - metrics collector, MatchQueryExec/PhraseQueryExec::new_with_segments are now - fallible, DataFile::new takes a ConcreteFileVersion, and pb::IndexMetadata - gained covering_fields. -- Keep the DOT PQ strict-subset guard; upstream make_global_pq is unchanged at - v11, so only the referenced revision in the comment and error text moved. - -Test Plan: -- cargo fmt, cargo check --all-targets, cargo clippy --all-targets -D warnings. -- cargo test: 367 passed, 0 failed, 2 ignored. -- cargo test --test compile_and_run_test -- --ignored: 2 passed (C and C++ - compile-and-run against the rebuilt library). - -Co-Authored-By: Claude Fable 5.1 ---- - Cargo.lock | 653 ++++++++++++++++++++++++------------------- - Cargo.toml | 22 +- - src/fts_query.rs | 4 +- - src/index_segment.rs | 4 +- - src/scanner.rs | 4 +- - tests/c_api_test.rs | 8 +- - 6 files changed, 391 insertions(+), 304 deletions(-) - -diff --git a/Cargo.lock b/Cargo.lock -index 60c1caf..bc37cb9 100644 ---- a/Cargo.lock -+++ b/Cargo.lock -@@ -222,7 +222,7 @@ dependencies = [ - "arrow-schema", - "arrow-select", - "atoi", -- "base64", -+ "base64 0.22.1", - "chrono", - "comfy-table", - "half", -@@ -332,7 +332,7 @@ version = "58.3.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "f633dbfdf39c039ada1bf9e34c694816eb71fbb7dc78f613993b7245e078a1ed" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - "serde_core", - "serde_json", - ] -@@ -434,6 +434,16 @@ dependencies = [ - "loom", - ] - -+[[package]] -+name = "asyncband" -+version = "0.6.7" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "94a214ba60d6231afd0e805e3c27c45a1626d9debaa5a5061c45a1ea1b2f1ed0" -+dependencies = [ -+ "hashbrown 0.17.1", -+ "slab", -+] -+ - [[package]] - name = "atoi" - version = "2.0.0" -@@ -504,7 +514,6 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "a054912289d18629dc78375ba2c3726a3afe3ff71b4edba9dedfca0e3446d1fc" - dependencies = [ - "aws-lc-sys", -- "untrusted 0.7.1", - "zeroize", - ] - -@@ -631,7 +640,7 @@ dependencies = [ - "bytes", - "form_urlencoded", - "hex", -- "hmac", -+ "hmac 0.12.1", - "http 0.2.12", - "http 1.4.0", - "percent-encoding", -@@ -829,6 +838,12 @@ version = "0.22.1" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" - -+[[package]] -+name = "base64" -+version = "0.23.1" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "ac07cdecf99051d9a5238b80f35af32cdeba5b336e55d957b318b50137e18da5" -+ - [[package]] - name = "base64-simd" - version = "0.8.0" -@@ -858,6 +873,12 @@ dependencies = [ - "num-traits", - ] - -+[[package]] -+name = "bitflags" -+version = "1.3.2" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a" -+ - [[package]] - name = "bitflags" - version = "2.11.0" -@@ -887,16 +908,15 @@ dependencies = [ - - [[package]] - name = "blake3" --version = "1.8.3" -+version = "1.8.7" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "2468ef7d57b3fb7e16b576e8377cdbde2320c60e1491e961d11da40fc4f02a2d" -+checksum = "6d9e454fc11f76977dc803893aff6304ed33d6a26efae8696573bea74baa27ae" - dependencies = [ -- "arrayref", - "arrayvec", - "cc", - "cfg-if 1.0.4", - "constant_time_eq", -- "cpufeatures 0.2.17", -+ "cpufeatures 0.3.0", - ] - - [[package]] -@@ -1127,6 +1147,12 @@ dependencies = [ - "cc", - ] - -+[[package]] -+name = "cmov" -+version = "0.5.4" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "0c9ea0ac24bc397ab3c98583a3c9ba74fa56b09a4449bbe172b9b1ddb016027a" -+ - [[package]] - name = "colorchoice" - version = "1.0.5" -@@ -1295,12 +1321,13 @@ dependencies = [ - ] - - [[package]] --name = "crc32c" --version = "0.6.8" -+name = "crc-fast" -+version = "1.10.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "3a47af21622d091a8f0fb295b88bc886ac74efcc613efc19f5d0b21de5c89e47" -+checksum = "e75b2483e97a5a7da73ac68a05b629f9c53cff58d8ed1c77866079e18b00dba5" - dependencies = [ -- "rustc_version", -+ "digest 0.10.7", -+ "spin 0.10.1", - ] - - [[package]] -@@ -1437,6 +1464,15 @@ version = "0.0.7" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "52560adf09603e58c9a7ee1fe1dcb95a16927b17c127f0ac02d6e768a0e25bc1" - -+[[package]] -+name = "ctutils" -+version = "0.4.2" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "7d5515a3834141de9eafb9717ad39eea8247b5674e6066c404e8c4b365d2a29e" -+dependencies = [ -+ "cmov", -+] -+ - [[package]] - name = "darling" - version = "0.23.0" -@@ -1785,7 +1821,7 @@ checksum = "5f64c983bbbdcb729d921a2b2ac3375598719b5cc0c30345ad664936f3176fc7" - dependencies = [ - "arrow", - "arrow-buffer", -- "base64", -+ "base64 0.22.1", - "blake2", - "blake3", - "chrono", -@@ -2112,6 +2148,37 @@ dependencies = [ - "url", - ] - -+[[package]] -+name = "defmt" -+version = "1.1.1" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "e2953bfe4f93bbd20cc71198842756f77d161884c99ebbabc41d80231ded88d1" -+dependencies = [ -+ "bitflags 1.3.2", -+ "defmt-macros", -+] -+ -+[[package]] -+name = "defmt-macros" -+version = "1.1.1" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "bad9c72e7ca2137e0dc3813245a0d282fd6daad32fd800af018306a9169b5fe8" -+dependencies = [ -+ "defmt-parser", -+ "proc-macro2", -+ "quote", -+ "syn 2.0.117", -+] -+ -+[[package]] -+name = "defmt-parser" -+version = "1.0.0" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "10d60334b3b2e7c9d91ef8150abfb6fa4c1c39ebbcf4a81c2e346aad939fee3e" -+dependencies = [ -+ "thiserror 2.0.18", -+] -+ - [[package]] - name = "der" - version = "0.7.10" -@@ -2154,6 +2221,7 @@ dependencies = [ - "block-buffer 0.12.1", - "const-oid 0.10.2", - "crypto-common 0.2.2", -+ "ctutils", - ] - - [[package]] -@@ -2322,7 +2390,7 @@ version = "25.12.19" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "35f6839d7b3b98adde531effaf34f0c2badc6f4735d26fe74709d8e513a96ef3" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - "rustc_version", - ] - -@@ -2369,6 +2437,12 @@ dependencies = [ - "percent-encoding", - ] - -+[[package]] -+name = "frostem" -+version = "1.20260821.5" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "36a80a7406da302e04bfd2ca987907590d3a1f3c69958947c43890abd7426b2f" -+ - [[package]] - name = "fs_extra" - version = "1.3.0" -@@ -2377,8 +2451,8 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c" - - [[package]] - name = "fsst" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", - "rand 0.9.2", -@@ -2727,14 +2801,23 @@ dependencies = [ - - [[package]] - name = "goosefs-sdk" --version = "0.1.5" -+version = "0.1.9" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "9ae079b88ffe7772d12cfc5c40a5a324babb357893d95b5e3a22ae857f236c5f" -+checksum = "e1ea4eee6dcbc31b25ab4fd577adc55b677d2bed3aa3016c44c58fbe1b2298a5" - dependencies = [ -+ "arc-swap", - "async-trait", - "bytes", - "dashmap", -+ "fastrand", -+ "futures", - "hostname", -+ "io-uring", -+ "itoa", -+ "libc", -+ "lru", -+ "memmap2", -+ "moka", - "prost", - "prost-types", - "rand 0.9.2", -@@ -2747,6 +2830,7 @@ dependencies = [ - "tonic-prost", - "tracing", - "uuid", -+ "xxhash-rust", - ] - - [[package]] -@@ -2900,6 +2984,15 @@ dependencies = [ - "digest 0.10.7", - ] - -+[[package]] -+name = "hmac" -+version = "0.13.0" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "6303bc9732ae41b04cb554b844a762b4115a61bfaa81e3e83050991eeb56863f" -+dependencies = [ -+ "digest 0.11.3", -+] -+ - [[package]] - name = "hostname" - version = "0.4.2" -@@ -3053,7 +3146,7 @@ version = "0.1.20" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "96547c2556ec9d12fb1578c4eaf448b04993e7fb79cbaad930a656880a6bdfa0" - dependencies = [ -- "base64", -+ "base64 0.22.1", - "bytes", - "futures-channel", - "futures-util", -@@ -3347,7 +3440,7 @@ version = "0.7.12" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "4d09b98f7eace8982db770e4408e7470b028ce513ac28fecdc6bf4c30fe92b62" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - "cfg-if 1.0.4", - "libc", - ] -@@ -3400,10 +3493,12 @@ checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" - - [[package]] - name = "jiff" --version = "0.2.23" -+version = "0.2.35" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "1a3546dc96b6d42c5f24902af9e2538e82e39ad350b0c766eb3fbf2d8f3d8359" -+checksum = "668b7183bd07af9a4885f5c35b0cc5c83c4607a913c16b7e17291832910d2dcc" - dependencies = [ -+ "defmt", -+ "jiff-core", - "jiff-static", - "jiff-tzdb-platform", - "js-sys", -@@ -3412,15 +3507,25 @@ dependencies = [ - "portable-atomic-util", - "serde_core", - "wasm-bindgen", -- "windows-sys 0.61.2", -+ "windows-link", -+] -+ -+[[package]] -+name = "jiff-core" -+version = "0.1.0" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "7feca88439efe53da3754500c1851dedf3cb36c524dd5cf8225cc0794de95d09" -+dependencies = [ -+ "defmt", - ] - - [[package]] - name = "jiff-static" --version = "0.2.23" -+version = "0.2.35" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "2a8c8b344124222efd714b73bb41f8b5120b27a7cc1c75593a6ff768d9d05aa4" -+checksum = "3a69dcb3a21cfb32ce1cd056169337ca284af0766dd766e7878819b251a49204" - dependencies = [ -+ "jiff-core", - "proc-macro2", - "quote", - "syn 2.0.117", -@@ -3530,24 +3635,6 @@ dependencies = [ - "serde_json", - ] - --[[package]] --name = "jsonwebtoken" --version = "10.4.0" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "eba32bfb4ffdeaca3e34431072faf01745c9b26d25504aa7a6cf5684334fc4fc" --dependencies = [ -- "aws-lc-rs", -- "base64", -- "getrandom 0.2.17", -- "js-sys", -- "pem", -- "serde", -- "serde_json", -- "signature", -- "simple_asn1", -- "zeroize", --] -- - [[package]] - name = "konst" - version = "0.4.3" -@@ -3567,8 +3654,8 @@ checksum = "e037a2e1d8d5fdbd49b16a4ea09d5d6401c1f29eca5ff29d03d3824dba16256a" - - [[package]] - name = "lance" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arc-swap", - "arrow", -@@ -3584,7 +3671,6 @@ dependencies = [ - "async-recursion", - "async-trait", - "async_cell", -- "aws-credential-types", - "byteorder", - "bytes", - "chrono", -@@ -3599,7 +3685,6 @@ dependencies = [ - "either", - "fst", - "futures", -- "half", - "humantime", - "itertools 0.14.0", - "lance-arrow", -@@ -3641,8 +3726,8 @@ dependencies = [ - - [[package]] - name = "lance-arrow" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", - "arrow-buffer", -@@ -3664,7 +3749,7 @@ dependencies = [ - [[package]] - name = "lance-arrow-scalar" - version = "58.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", - "arrow-buffer", -@@ -3678,7 +3763,7 @@ dependencies = [ - [[package]] - name = "lance-arrow-stats" - version = "58.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", - "arrow-schema", -@@ -3687,8 +3772,8 @@ dependencies = [ - - [[package]] - name = "lance-bitpacking" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrayref", - "crunchy", -@@ -3728,20 +3813,19 @@ dependencies = [ - - [[package]] - name = "lance-core" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", - "arrow-buffer", - "arrow-data", - "arrow-schema", - "async-trait", -- "byteorder", -+ "blake3", - "bytes", - "datafusion-common", - "datafusion-sql", - "futures", -- "itertools 0.14.0", - "lance-arrow", - "lance-derive", - "libc", -@@ -3752,13 +3836,13 @@ dependencies = [ - "object_store", - "pin-project", - "prost", -+ "quick_cache", - "rand 0.9.2", - "roaring", - "serde_json", - "snafu", - "tempfile", - "tokio", -- "tokio-stream", - "tokio-util", - "tracing", - "twox-hash", -@@ -3767,8 +3851,8 @@ dependencies = [ - - [[package]] - name = "lance-datafusion" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow", - "arrow-array", -@@ -3788,7 +3872,6 @@ dependencies = [ - "jsonb", - "lance-arrow", - "lance-core", -- "lance-datagen", - "lance-geo", - "log", - "pin-project", -@@ -3800,8 +3883,8 @@ dependencies = [ - - [[package]] - name = "lance-datagen" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow", - "arrow-array", -@@ -3818,8 +3901,8 @@ dependencies = [ - - [[package]] - name = "lance-derive" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "proc-macro2", - "quote", -@@ -3828,8 +3911,8 @@ dependencies = [ - - [[package]] - name = "lance-encoding" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-arith", - "arrow-array", -@@ -3854,8 +3937,6 @@ dependencies = [ - "num-traits", - "prost", - "prost-build", -- "rand 0.9.2", -- "strum", - "tokio", - "tracing", - "xxhash-rust", -@@ -3864,12 +3945,13 @@ dependencies = [ - - [[package]] - name = "lance-file" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-arith", - "arrow-array", - "arrow-buffer", -+ "arrow-cast", - "arrow-data", - "arrow-schema", - "arrow-select", -@@ -3895,8 +3977,8 @@ dependencies = [ - - [[package]] - name = "lance-geo" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "datafusion", - "geo-traits", -@@ -3910,13 +3992,14 @@ dependencies = [ - - [[package]] - name = "lance-index" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arc-swap", - "arrow", - "arrow-arith", - "arrow-array", -+ "arrow-ipc", - "arrow-ord", - "arrow-schema", - "arrow-select", -@@ -3925,7 +4008,6 @@ dependencies = [ - "async-trait", - "bitvec", - "bytes", -- "chrono", - "crossbeam-queue", - "datafusion", - "datafusion-common", -@@ -3945,7 +4027,6 @@ dependencies = [ - "lance-bitpacking", - "lance-core", - "lance-datafusion", -- "lance-datagen", - "lance-encoding", - "lance-file", - "lance-geo", -@@ -3975,13 +4056,12 @@ dependencies = [ - "tempfile", - "tokio", - "tracing", -- "uuid", - ] - - [[package]] - name = "lance-index-core" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", - "arrow-schema", -@@ -4003,18 +4083,12 @@ dependencies = [ - - [[package]] - name = "lance-io" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow", -- "arrow-arith", - "arrow-array", -- "arrow-buffer", -- "arrow-cast", -- "arrow-data", - "arrow-schema", -- "arrow-select", -- "async-recursion", - "async-trait", - "aws-config", - "aws-credential-types", -@@ -4022,10 +4096,8 @@ dependencies = [ - "bytes", - "chrono", - "futures", -- "goosefs-sdk", - "http 1.4.0", - "io-uring", -- "lance-arrow", - "lance-core", - "lance-namespace", - "log", -@@ -4037,34 +4109,37 @@ dependencies = [ - "pin-project", - "prost", - "rand 0.9.2", -+ "reqsign-core", -+ "reqsign-file-read-tokio", -+ "reqsign-google", - "serde", -+ "serde_json", - "tempfile", - "tokio", - "tracing", - "url", -+ "uuid", - ] - - [[package]] - name = "lance-linalg" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", -- "arrow-buffer", - "arrow-schema", - "cc", - "half", - "lance-arrow", - "lance-core", - "num-traits", -- "rand 0.9.2", - "rayon", - ] - - [[package]] - name = "lance-namespace" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow", - "async-trait", -@@ -4076,9 +4151,9 @@ dependencies = [ - - [[package]] - name = "lance-namespace-reqwest-client" --version = "0.8.6" -+version = "0.11.1" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "ba3f0a235e3ed5f8805205649ccc7d7d0f3df23ce1294242c9265ad488d7f19d" -+checksum = "1d06b1fbb5d41f93bc652b61e2872af92e8a6c5f6b4ce8839a8ecfa05365d359" - dependencies = [ - "reqwest 0.12.28", - "serde", -@@ -4090,14 +4165,13 @@ dependencies = [ - - [[package]] - name = "lance-select" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow-array", - "arrow-buffer", - "arrow-schema", - "byteorder", -- "bytes", - "itertools 0.14.0", - "lance-core", - "roaring", -@@ -4106,8 +4180,8 @@ dependencies = [ - - [[package]] - name = "lance-table" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ - "arrow", - "arrow-array", -@@ -4115,6 +4189,7 @@ dependencies = [ - "arrow-ipc", - "arrow-schema", - "async-trait", -+ "blake3", - "byteorder", - "bytes", - "chrono", -@@ -4144,11 +4219,11 @@ dependencies = [ - - [[package]] - name = "lance-tokenizer" --version = "9.1.0-beta.3" --source = "git+https://github.com/lance-format/lance.git?rev=e934cc2c#e934cc2ceda2bd5f5aa37a953cc29f71d24bd5c0" -+version = "11.0.0" -+source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" - dependencies = [ -+ "frostem", - "icu_segmenter", -- "rust-stemmers", - "serde", - "stop-words", - "unicode-normalization", -@@ -4160,7 +4235,7 @@ version = "1.5.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe" - dependencies = [ -- "spin", -+ "spin 0.9.8", - ] - - [[package]] -@@ -4291,9 +4366,9 @@ dependencies = [ - - [[package]] - name = "log" --version = "0.4.29" -+version = "0.4.34" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897" -+checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" - - [[package]] - name = "loom" -@@ -4308,6 +4383,15 @@ dependencies = [ - "tracing-subscriber", - ] - -+[[package]] -+name = "lru" -+version = "0.18.4" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "ff9840bcc50b71349309900da0ce7279aa336ae71d73250b07998932c7d97c25" -+dependencies = [ -+ "hashbrown 0.17.1", -+] -+ - [[package]] - name = "lru-slab" - version = "0.1.2" -@@ -4399,6 +4483,15 @@ version = "2.8.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "f8ca58f447f06ed17d5fc4043ce1b10dd205e060fb3ce5b979b8ed8e59ff3f79" - -+[[package]] -+name = "memmap2" -+version = "0.9.11" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "d1219ed1b7f229ee7104d281dd01d6802fe28bb6e95d292942c4daacdeb798c0" -+dependencies = [ -+ "libc", -+] -+ - [[package]] - name = "mime" - version = "0.3.17" -@@ -4619,7 +4712,7 @@ version = "0.3.2" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "2a180dd8642fa45cdb7dd721cd4c11b1cadd4929ce112ebd8b9f5803cc79d536" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - ] - - [[package]] -@@ -4648,7 +4741,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "622acbc9100d3c10e2ee15804b0caa40e55c933d5aa53814cd520805b7958a49" - dependencies = [ - "async-trait", -- "base64", -+ "base64 0.22.1", - "bytes", - "chrono", - "form_urlencoded", -@@ -4664,7 +4757,7 @@ dependencies = [ - "md-5 0.10.6", - "parking_lot", - "percent-encoding", -- "quick-xml", -+ "quick-xml 0.39.4", - "rand 0.10.1", - "reqwest 0.12.28", - "ring", -@@ -4683,9 +4776,9 @@ dependencies = [ - - [[package]] - name = "object_store_opendal" --version = "0.57.0" -+version = "0.58.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "0eb12a624a41fce745838d0ef3701ff6c47797c13cd18ad3612fd2a3134fdbd8" -+checksum = "88f165780495c17aa3ce86846600504198c3fffd99073521552751c2430fa6ac" - dependencies = [ - "async-trait", - "bytes", -@@ -4718,12 +4811,13 @@ checksum = "269bca4c2591a28585d6bf10d9ed0332b7d76900a1b02bec41bdc3a2cdcda107" - - [[package]] - name = "opendal" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "96c9c85ce253ff87225e7669979d877a20c98a06604ec9d6dd5f4473e08f1ae1" -+checksum = "33dbff14cc9bb085224256d6a81289d2f3202e85b06f408d42534b42162a4231" - dependencies = [ - "ctor 1.0.13", - "opendal-core", -+ "opendal-http-transport-reqwest", - "opendal-layer-concurrent-limit", - "opendal-layer-logging", - "opendal-layer-retry", -@@ -4741,24 +4835,22 @@ dependencies = [ - - [[package]] - name = "opendal-core" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "c4f8607c90e2c963a91467f50fb49fbc7fb3d573f88cea219ca59ccd3740b309" -+checksum = "48dbcef97d3eb7591db2c18d5cae95c836bcce07359b98d98dd6f4e861eb77b7" - dependencies = [ - "anyhow", -- "base64", -+ "asyncband", -+ "base64 0.23.1", - "bytes", - "futures", - "http 1.4.0", -- "http-body 1.0.1", - "jiff", - "log", - "md-5 0.11.0", -- "mea", - "percent-encoding", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-core", -- "reqwest 0.13.3", - "serde", - "serde_json", - "tokio", -@@ -4767,23 +4859,37 @@ dependencies = [ - "web-time", - ] - -+[[package]] -+name = "opendal-http-transport-reqwest" -+version = "0.58.2" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "85663452ea32bbc17e8f79ab29788c846d116ec7de31451be9c787e462dcb36c" -+dependencies = [ -+ "bytes", -+ "futures", -+ "http 1.4.0", -+ "http-body 1.0.1", -+ "opendal-core", -+ "reqwest 0.13.4", -+] -+ - [[package]] - name = "opendal-layer-concurrent-limit" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "0d6f81ba6960e3fae1882f253b114b21d7e444e1534f209c7737a79f6243eb6f" -+checksum = "03f9e144b5228d741c3763ade8711d9b72e5fb6d998e779f2d7a09da0b5a3eba" - dependencies = [ -+ "asyncband", - "futures", - "http 1.4.0", -- "mea", - "opendal-core", - ] - - [[package]] - name = "opendal-layer-logging" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "58ada45c6d81d1aa4c9305d0c7d4bc317c59c85866a0908a2d75a7a978aa5ee2" -+checksum = "c2de17c61cd32e9d8d7d8efb91795e714dbccbafbc3c4e219e1542f4d6324161" - dependencies = [ - "log", - "opendal-core", -@@ -4791,9 +4897,9 @@ dependencies = [ - - [[package]] - name = "opendal-layer-retry" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "7b2a25a718afb81fad81cb9a0580a1cb989221fa2317f888c6a37f8dad408eb7" -+checksum = "e94db301964a25366090484d61e6da16d5979cc8faf02c3e12210dc74fafed38" - dependencies = [ - "backon", - "log", -@@ -4802,9 +4908,9 @@ dependencies = [ - - [[package]] - name = "opendal-layer-timeout" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "1e91f731724c213af81e9d03517859c8fc47b4578e64ad61ae4f099f10fe36e3" -+checksum = "08956ddda07465449bfd48825f4f0f25e0351278ac974eaa659895d9d74f2c80" - dependencies = [ - "opendal-core", - "tokio", -@@ -4812,17 +4918,17 @@ dependencies = [ - - [[package]] - name = "opendal-service-azblob" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "0030644366ef5d8cbe3a4a5822bf99a4aafddc1666e9d24b44d158d9062fc76a" -+checksum = "6d278d2fb57661947d1c9fb44047e432b2782c48dc085b7abc99fe6e18c26cf8" - dependencies = [ -- "base64", -+ "base64 0.23.1", - "bytes", - "http 1.4.0", - "log", - "opendal-core", - "opendal-service-azure-common", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-azure-storage", - "reqsign-core", - "reqsign-file-read-tokio", -@@ -4833,17 +4939,18 @@ dependencies = [ - - [[package]] - name = "opendal-service-azdls" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "6dea4908d490143a9b0b7f7a790e139ff829b06a023f670455ed3d44f664b361" -+checksum = "2d564484a8f7d091827e825cfc91ed45bd48e64d262451ee041fe843db81bd8a" - dependencies = [ -- "base64", -+ "asyncband", -+ "base64 0.23.1", - "bytes", - "http 1.4.0", - "log", - "opendal-core", - "opendal-service-azure-common", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-azure-storage", - "reqsign-core", - "reqsign-file-read-tokio", -@@ -4853,9 +4960,9 @@ dependencies = [ - - [[package]] - name = "opendal-service-azure-common" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "9b489f13c42e69d69bdd72952b634356ec43a7881a20259b38b540fcecdf4051" -+checksum = "cfcc1bfdac4f54d9018c462dd32ad9e2f68fdf584f825c811550c8ffc87894cd" - dependencies = [ - "http 1.4.0", - "opendal-core", -@@ -4863,15 +4970,15 @@ dependencies = [ - - [[package]] - name = "opendal-service-cos" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "aa8cafe9729213375c7331019b0cb756ad3e1aff7f45cd32c45eae91ebde8901" -+checksum = "bb021c128ebde42e6f3e719d4cfed27e994017aa503a7a1e5818bdb61fd67dc9" - dependencies = [ - "bytes", - "http 1.4.0", - "log", - "opendal-core", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-core", - "reqsign-file-read-tokio", - "reqsign-tencent-cos", -@@ -4880,9 +4987,9 @@ dependencies = [ - - [[package]] - name = "opendal-service-gcs" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "48de101aac565ed06af4b47903c24eafd249075553ec1fb18256751c45148d47" -+checksum = "da1f2a8c975fd22fea01f0409bcb8774f1ac02ad79863a3490098203121555d3" - dependencies = [ - "async-trait", - "bytes", -@@ -4890,7 +4997,7 @@ dependencies = [ - "log", - "opendal-core", - "percent-encoding", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-core", - "reqsign-file-read-tokio", - "reqsign-google", -@@ -4901,9 +5008,9 @@ dependencies = [ - - [[package]] - name = "opendal-service-goosefs" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "69e43048bde419947ba826fbdc2f134d6c03f44ebf48bd33a03b72f9fc45fcb4" -+checksum = "89fd71b80078f2983bd363e322fbebfa6a46d5fc56f17e4f23d76d5edb31226b" - dependencies = [ - "bytes", - "goosefs-sdk", -@@ -4915,9 +5022,9 @@ dependencies = [ - - [[package]] - name = "opendal-service-hf" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "c4922661976a1d40794a2adfbdb888cc3c23097690f825a92f773af38908a848" -+checksum = "c8a3c8ec0c2918fa23f258fa28822f222ec8c1e3a501ded665b78bc65a9d7734" - dependencies = [ - "bytes", - "hf-xet", -@@ -4925,22 +5032,21 @@ dependencies = [ - "log", - "opendal-core", - "percent-encoding", -- "reqwest 0.13.3", - "serde", - "serde_json", - ] - - [[package]] - name = "opendal-service-oss" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "328fa55e8888cbdfe00826bfea2a79042422b720e8369e9e021e46121dea5ace" -+checksum = "8e3ce7a2ceb925e0f28f545b169eb57d04cec6116aae94b8584d2bdbe7d456c6" - dependencies = [ - "bytes", - "http 1.4.0", - "log", - "opendal-core", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-aliyun-oss", - "reqsign-core", - "reqsign-file-read-tokio", -@@ -4949,18 +5055,18 @@ dependencies = [ - - [[package]] - name = "opendal-service-s3" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "313d46c9f5ae70bca26b7c3e3fbb9b639292625f28af73aa016f47e788af9deb" -+checksum = "c64335f9f24ccb62ac36f1d976342b48611a75ba61979813a4f78a4ebd94de42" - dependencies = [ -- "base64", -+ "base64 0.23.1", - "bytes", -- "crc32c", -+ "crc-fast", - "http 1.4.0", - "log", - "md-5 0.11.0", - "opendal-core", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-aws-v4", - "reqsign-core", - "reqsign-file-read-tokio", -@@ -4970,14 +5076,14 @@ dependencies = [ - - [[package]] - name = "opendal-service-tos" --version = "0.57.0" -+version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "6f2f7a4c32e5202eb4ac72e76c4b5e30c86ab60762811172f4111103b9d673a1" -+checksum = "70c3c507c3a565b2feb4c7b5f653436acd34ca59ffc842a7671b5498b13b76bf" - dependencies = [ - "bytes", - "http 1.4.0", - "opendal-core", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-core", - "reqsign-file-read-tokio", - "reqsign-volcengine-tos", -@@ -5084,7 +5190,7 @@ version = "0.8.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "898bac3fa00d0ba57a4e8289837e965baa2dee8c3749f3b11d45a64b4223d9c3" - dependencies = [ -- "base64", -+ "base64 0.22.1", - "serde", - ] - -@@ -5122,16 +5228,16 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "f8ed6a7761f76e3b9f92dfb0a60a6a6477c61024b775147ff0973a02653abaf2" - dependencies = [ - "digest 0.10.7", -- "hmac", -+ "hmac 0.12.1", - ] - - [[package]] - name = "pem" --version = "3.0.6" -+version = "4.0.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "1d30c53c26bc5b31a98cd02d20f25a7c8567146caf63ed593a9d87b2775291be" -+checksum = "d354a98a3d1251555de99e8fdd8afda05573c31b82f59063a7b0a29b5527f120" - dependencies = [ -- "base64", -+ "base64 0.23.1", - "serde_core", - ] - -@@ -5392,6 +5498,28 @@ dependencies = [ - "serde", - ] - -+[[package]] -+name = "quick-xml" -+version = "0.41.0" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "e660451e55124f798a69a5af3f49ccfbefbd41910eefd25caf2393e1f3473ec1" -+dependencies = [ -+ "memchr", -+ "serde", -+] -+ -+[[package]] -+name = "quick_cache" -+version = "0.6.24" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "b9c6658afe513a3b484e3abfdaa0d03ef3c0bbf017542c178dd55f94eb3051f9" -+dependencies = [ -+ "ahash", -+ "equivalent", -+ "hashbrown 0.16.1", -+ "parking_lot", -+] -+ - [[package]] - name = "quinn" - version = "0.11.9" -@@ -5616,7 +5744,7 @@ version = "0.5.18" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - ] - - [[package]] -@@ -5697,9 +5825,9 @@ dependencies = [ - - [[package]] - name = "reqsign-aliyun-oss" --version = "3.0.0" -+version = "3.1.4" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "57ac2757f3140aa2e213b554148ae0b52733e624fc6723f0cc6bb3d440176c95" -+checksum = "68d24d281f734a463093b7b93aae8b16f5f8496a54fbeec4d2d10b0488295296" - dependencies = [ - "anyhow", - "form_urlencoded", -@@ -5713,18 +5841,18 @@ dependencies = [ - ] - - [[package]] --name = "reqsign-aws-v4" --version = "3.0.0" -+name = "reqsign-aws-core" -+version = "3.1.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "44eaca382e94505a49f1a4849658d153aebf79d9c1a58e5dd3b10361511e9f43" -+checksum = "4d63b56638bb3cc7bd376a7cdce1ba3089777a08f47e4097888f2d784cc3f46c" - dependencies = [ -- "anyhow", - "bytes", - "form_urlencoded", -+ "hex", - "http 1.4.0", - "log", - "percent-encoding", -- "quick-xml", -+ "quick-xml 0.41.0", - "reqsign-core", - "rust-ini", - "serde", -@@ -5733,18 +5861,32 @@ dependencies = [ - "sha1", - ] - -+[[package]] -+name = "reqsign-aws-v4" -+version = "3.2.0" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "4a0c499f4ed12d04c3d4c78fe4cb01aee22c9dae22848c14db2c6313d9df9f43" -+dependencies = [ -+ "bytes", -+ "http 1.4.0", -+ "log", -+ "quick-xml 0.41.0", -+ "reqsign-aws-core", -+ "reqsign-core", -+ "serde", -+] -+ - [[package]] - name = "reqsign-azure-storage" --version = "3.0.0" -+version = "3.2.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "7a321980405d596bd34aaf95c4722a3de4128a67fd19e74a81a83aa3fdf082e6" -+checksum = "e8177b4f08620ab7f2e9cab7d7ccb9da1b61c66b889fed46cc0880b0fe75eb6b" - dependencies = [ - "anyhow", -- "base64", -+ "base64 0.23.1", - "bytes", - "form_urlencoded", - "http 1.4.0", -- "jsonwebtoken", - "log", - "pem", - "percent-encoding", -@@ -5757,31 +5899,33 @@ dependencies = [ - - [[package]] - name = "reqsign-core" --version = "3.0.0" -+version = "3.3.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "b10302cf0a7d7e7352ba211fc92c3c5bebf1286153e49cc5aa87348078a8e102" -+checksum = "f4ac1510872d9481205975d264deb39c109797e5068cc882ed9064270eaae5fa" - dependencies = [ - "anyhow", -- "base64", -+ "base64 0.23.1", - "bytes", -- "form_urlencoded", - "futures", - "hex", -- "hmac", -+ "hmac 0.13.0", - "http 1.4.0", - "jiff", - "log", - "percent-encoding", -+ "rsa", -+ "serde", -+ "serde_json", - "sha1", -- "sha2 0.10.9", -+ "sha2 0.11.0", - "windows-sys 0.61.2", - ] - - [[package]] - name = "reqsign-file-read-tokio" --version = "3.0.0" -+version = "3.0.5" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "e2d89295b3d17abea31851cc8de55d843d89c52132c864963c38d41920613dc5" -+checksum = "95c3371bfc7e5c7f9627a04133af3583fd6c28715e7c83f79db38f3b384f535f" - dependencies = [ - "anyhow", - "reqsign-core", -@@ -5790,13 +5934,13 @@ dependencies = [ - - [[package]] - name = "reqsign-google" --version = "3.0.0" -+version = "3.1.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "35cc609b49c69e76ecaceb775a03f792d1ed3e7755ab3548d4534fd801e3242e" -+checksum = "f81a9d38870892443489c0abb5332edfa81d5a14c437af9caef7c194897c92c1" - dependencies = [ -+ "bytes", - "form_urlencoded", - "http 1.4.0", -- "jsonwebtoken", - "log", - "percent-encoding", - "reqsign-aws-v4", -@@ -5804,15 +5948,14 @@ dependencies = [ - "rsa", - "serde", - "serde_json", -- "sha2 0.10.9", - "tokio", - ] - - [[package]] - name = "reqsign-tencent-cos" --version = "3.0.0" -+version = "3.0.5" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "e128f19525861dbded59e1e7c17653a8ed63d573ca04aed708d552dbef5bb32a" -+checksum = "b15c5a4df7c3f16823ae242675c5ebfb52d640cc9a50d1fcf943263247fa1730" - dependencies = [ - "anyhow", - "http 1.4.0", -@@ -5825,9 +5968,9 @@ dependencies = [ - - [[package]] - name = "reqsign-volcengine-tos" --version = "3.0.0" -+version = "3.1.1" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "f9d757602a7ef2b6025c0da77e6d2e23fbdef35930fa466b15ffbf0a3f13acf7" -+checksum = "173387eb5ae4cf6a0a7098665ebcc6729862819dc3d95a81aee840767803e3d9" - dependencies = [ - "anyhow", - "http 1.4.0", -@@ -5842,7 +5985,7 @@ version = "0.12.28" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "eddd3ca559203180a307f12d114c268abf583f59b03cb906fd0b3ff8646c1147" - dependencies = [ -- "base64", -+ "base64 0.22.1", - "bytes", - "encoding_rs", - "futures-core", -@@ -5884,11 +6027,11 @@ dependencies = [ - - [[package]] - name = "reqwest" --version = "0.13.3" -+version = "0.13.4" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "62e0021ea2c22aed41653bc7e1419abb2c97e038ff2c33d0e1309e49a97deec0" -+checksum = "219c5811de6525e5416c7d5d53bb656d3afdbc6c5af816e0802bcfa42dbdc1c3" - dependencies = [ -- "base64", -+ "base64 0.22.1", - "bytes", - "futures-core", - "futures-util", -@@ -5931,7 +6074,7 @@ dependencies = [ - "anyhow", - "async-trait", - "http 1.4.0", -- "reqwest 0.13.3", -+ "reqwest 0.13.4", - "thiserror 2.0.18", - "tower-service", - ] -@@ -5946,7 +6089,7 @@ dependencies = [ - "cfg-if 1.0.4", - "getrandom 0.2.17", - "libc", -- "untrusted 0.9.0", -+ "untrusted", - "windows-sys 0.52.0", - ] - -@@ -6008,16 +6151,6 @@ dependencies = [ - "ordered-multimap", - ] - --[[package]] --name = "rust-stemmers" --version = "1.2.0" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "e46a2036019fdb888131db7a4c847a1063a7493f971ed94ea82c67eada63ca54" --dependencies = [ -- "serde", -- "serde_derive", --] -- - [[package]] - name = "rustc-hash" - version = "2.1.1" -@@ -6039,7 +6172,7 @@ version = "1.1.4" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - "errno", - "libc", - "linux-raw-sys", -@@ -6119,7 +6252,7 @@ dependencies = [ - "aws-lc-rs", - "ring", - "rustls-pki-types", -- "untrusted 0.9.0", -+ "untrusted", - ] - - [[package]] -@@ -6244,7 +6377,7 @@ version = "3.7.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "b7f4bc775c73d9a02cde8bf7b2ec4c9d12743edf609006c7facc23998404cd1d" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - "core-foundation 0.10.1", - "core-foundation-sys", - "libc", -@@ -6373,7 +6506,7 @@ version = "3.20.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "e72c1c2cb7b223fafb600a619537a871c2818583d619401b785e7c0b746ccde2" - dependencies = [ -- "base64", -+ "base64 0.22.1", - "bs58", - "chrono", - "hex", -@@ -6414,13 +6547,13 @@ dependencies = [ - - [[package]] - name = "sha1" --version = "0.10.6" -+version = "0.11.0" - source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "e3bf829a2d51ab4a5ddf1352d8470c140cadc8301b2ae1789db023f01cedd6ba" -+checksum = "aacc4cc499359472b4abe1bf11d0b12e688af9a805fa5e3016f9a386dc2d0214" - dependencies = [ - "cfg-if 1.0.4", -- "cpufeatures 0.2.17", -- "digest 0.10.7", -+ "cpufeatures 0.3.0", -+ "digest 0.11.3", - ] - - [[package]] -@@ -6523,18 +6656,6 @@ version = "0.1.5" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e" - --[[package]] --name = "simple_asn1" --version = "0.6.4" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "0d585997b0ac10be3c5ee635f1bab02d512760d14b7c468801ac8a01d9ae5f1d" --dependencies = [ -- "num-bigint", -- "num-traits", -- "thiserror 2.0.18", -- "time", --] -- - [[package]] - name = "siphasher" - version = "1.0.2" -@@ -6602,6 +6723,12 @@ version = "0.9.8" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "6980e8d7511241f8acf4aebddbb1ff938df5eebe98691418c4468d0b72a96a67" - -+[[package]] -+name = "spin" -+version = "0.10.1" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "023a211cb3138dbc438680b32560ad89f699977624c9f8dbb95a47d5b4c07dd3" -+ - [[package]] - name = "spki" - version = "0.7.3" -@@ -6682,28 +6809,6 @@ version = "0.11.1" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" - --[[package]] --name = "strum" --version = "0.26.3" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "8fec0f0aef304996cf250b31b5a10dee7980c85da9d759361292b8bca5a18f06" --dependencies = [ -- "strum_macros", --] -- --[[package]] --name = "strum_macros" --version = "0.26.4" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "4c6bee85a5a24955dc440386795aa378cd9cf82acd5f764469152d2270e581be" --dependencies = [ -- "heck", -- "proc-macro2", -- "quote", -- "rustversion", -- "syn 2.0.117", --] -- - [[package]] - name = "substrait" - version = "0.63.0" -@@ -6804,7 +6909,7 @@ version = "0.7.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "a13f3d0daba03132c0aa9767f98351b3488edc2c100cda2d2ec2b04f3d8d3c8b" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - "core-foundation 0.9.4", - "system-configuration-sys", - ] -@@ -7078,7 +7183,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "ac2a5518c70fa84342385732db33fb3f44bc4cc748936eb5833d2df34d6445ef" - dependencies = [ - "async-trait", -- "base64", -+ "base64 0.22.1", - "bytes", - "h2", - "http 1.4.0", -@@ -7136,7 +7241,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "d4e6559d53cc268e5031cd8429d05415bc4cb4aefc4aa5d6cc35fbf5b924a1f8" - dependencies = [ - "async-compression", -- "bitflags", -+ "bitflags 2.11.0", - "bytes", - "futures-core", - "futures-util", -@@ -7370,12 +7475,6 @@ version = "0.2.11" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "673aac59facbab8a9007c7f6108d11f63b603f7cabff99fabf650fea5c32b861" - --[[package]] --name = "untrusted" --version = "0.7.1" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "a156c684c91ea7d62626509bce3cb4e1d9ed5c4d978f7b4352658f96a4c26b4a" -- - [[package]] - name = "untrusted" - version = "0.9.0" -@@ -7622,7 +7721,7 @@ version = "0.244.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe" - dependencies = [ -- "bitflags", -+ "bitflags 2.11.0", - "hashbrown 0.15.5", - "indexmap 2.14.0", - "semver", -@@ -8054,7 +8153,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2" - dependencies = [ - "anyhow", -- "bitflags", -+ "bitflags 2.11.0", - "indexmap 2.14.0", - "log", - "serde", -@@ -8132,7 +8231,7 @@ checksum = "3e1e496dcbe6a09017acdfaf48e1a646735e7ff5b2a49e2c7e081cca77a59bc8" - dependencies = [ - "anyhow", - "async-trait", -- "base64", -+ "base64 0.22.1", - "bytes", - "clap", - "crc32fast", -@@ -8143,7 +8242,7 @@ dependencies = [ - "more-asserts", - "rand 0.10.1", - "redb", -- "reqwest 0.13.3", -+ "reqwest 0.13.4", - "reqwest-middleware", - "serde", - "serde_json", -@@ -8169,7 +8268,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "cb838aa8eb67d730af301584cf003caad407487606058292a6750711b603fbee" - dependencies = [ - "async-trait", -- "base64", -+ "base64 0.22.1", - "blake3", - "bytemuck", - "bytes", -@@ -8256,7 +8355,7 @@ dependencies = [ - "oneshot", - "pin-project", - "rand 0.10.1", -- "reqwest 0.13.3", -+ "reqwest 0.13.4", - "serde", - "serde_json", - "shellexpand", -@@ -8352,20 +8451,6 @@ name = "zeroize" - version = "1.8.2" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "b97154e67e32c85465826e8bcc1c59429aaaf107c1e4a9e53c8d8ccd5eff88d0" --dependencies = [ -- "zeroize_derive", --] -- --[[package]] --name = "zeroize_derive" --version = "1.4.3" --source = "registry+https://github.com/rust-lang/crates.io-index" --checksum = "85a5b4158499876c763cb03bc4e49185d3cccbabb15b33c627f7884f43db852e" --dependencies = [ -- "proc-macro2", -- "quote", -- "syn 2.0.117", --] - - [[package]] - name = "zerotrie" -diff --git a/Cargo.toml b/Cargo.toml -index d072a5d..3920a65 100644 ---- a/Cargo.toml -+++ b/Cargo.toml -@@ -18,14 +18,14 @@ rust-version = "1.91.0" - crate-type = ["cdylib", "staticlib", "rlib"] - - [dependencies] --lance = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c", features = ["substrait"] } --lance-core = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } --lance-file = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } --lance-index = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } --lance-io = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } --lance-linalg = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } --lance-table = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } --lance-datafusion = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c", features = ["substrait"] } -+lance = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe", features = ["substrait"] } -+lance-core = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance-file = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance-index = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance-io = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance-linalg = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance-table = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance-datafusion = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe", features = ["substrait"] } - datafusion = { version = "54.0.0", default-features = false } - arrow = { version = "58.0.0", features = ["prettyprint", "ffi"] } - arrow-array = "58.0.0" -@@ -45,9 +45,9 @@ snafu = "0.9" - uuid = { version = "1", features = ["v4"] } - - [dev-dependencies] --lance = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c", features = ["substrait"] } --lance-datagen = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } --lance-file = { git = "https://github.com/lance-format/lance.git", rev = "e934cc2c" } -+lance = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe", features = ["substrait"] } -+lance-datagen = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance-file = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } - tokio = { version = "1", features = ["rt-multi-thread", "macros"] } - arrow-array = "58.0.0" - arrow-schema = "58.0.0" -diff --git a/src/fts_query.rs b/src/fts_query.rs -index c9d3a43..71874e0 100644 ---- a/src/fts_query.rs -+++ b/src/fts_query.rs -@@ -246,7 +246,7 @@ async fn prepare_fts_query_context( - .with_max_expansions(match_query.max_expansions) - .with_prefix_length(match_query.prefix_length); - PreparedFtsQuery::Match(Arc::new( -- build_global_bm25_scorer(&indices, &query_tokens, ¶ms).await?, -+ build_global_bm25_scorer(&indices, &query_tokens, ¶ms, None).await?, - )) - } - FtsQuery::Phrase(phrase_query) => { -@@ -260,7 +260,7 @@ async fn prepare_fts_query_context( - let query_tokens = collect_query_tokens(&phrase_query.terms, &mut tokenizer); - let params = query.params().with_phrase_slop(Some(phrase_query.slop)); - PreparedFtsQuery::Phrase(Arc::new( -- build_global_bm25_scorer(&indices, &query_tokens, ¶ms).await?, -+ build_global_bm25_scorer(&indices, &query_tokens, ¶ms, None).await?, - )) - } - _ => { -diff --git a/src/index_segment.rs b/src/index_segment.rs -index a46c4f3..d4e143c 100644 ---- a/src/index_segment.rs -+++ b/src/index_segment.rs -@@ -889,7 +889,7 @@ unsafe fn new_vector_builder_inner( - // TODO(upstream-lance): Remove this fail-fast once Lance's distributed - // vector-index path reconstructs a supplied PQ codebook with an L2 - // ProductQuantizer, matching the ordinary full-dataset path. Pinned Lance -- // revision e934cc2c rewraps supplied codebooks with DistanceType::Dot in -+ // revision ab6b5bbe rewraps supplied codebooks with DistanceType::Dot in - // `make_global_pq`, which silently switches PQ code assignment away from - // the L2 contract shared by full-dataset builds and index readers. - if matches!( -@@ -910,7 +910,7 @@ unsafe fn new_vector_builder_inner( - let selected_fragment_ids: HashSet = fragment_ids.iter().copied().collect(); - if selected_fragment_ids != all_fragment_ids { - return Err(invalid_input(format!( -- "pq_codebook is supplied for metric=DOT, index_type={:?}, mode={:?}, and an effective strict fragment subset ({} of {} fragments): pinned Lance revision e934cc2c reconstructs the supplied codebook with a DOT ProductQuantizer in the distributed build path (make_global_pq), silently breaking the L2 PQ-assignment contract; cover the full dataset in one segment (pass NULL fragment_ids or list every fragment) or wait for upstream Lance DOT support", -+ "pq_codebook is supplied for metric=DOT, index_type={:?}, mode={:?}, and an effective strict fragment subset ({} of {} fragments): pinned Lance revision ab6b5bbe reconstructs the supplied codebook with a DOT ProductQuantizer in the distributed build path (make_global_pq), silently breaking the L2 PQ-assignment contract; cover the full dataset in one segment (pass NULL fragment_ids or list every fragment) or wait for upstream Lance DOT support", - params.index_type, - parsed.mode, - selected_fragment_ids.len(), -diff --git a/src/scanner.rs b/src/scanner.rs -index 7e898ce..d3ef3be 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -529,7 +529,7 @@ fn rewrite_prepared_fts_plan( - exec.params().clone(), - exec.prefilter_source().clone(), - segments.to_vec(), -- ) -+ )? - .with_base_scorer(Arc::clone(scorer)); - return Ok((Arc::new(replacement), rewritten)); - } -@@ -549,7 +549,7 @@ fn rewrite_prepared_fts_plan( - exec.params().clone(), - exec.prefilter_source().clone(), - segments.to_vec(), -- ) -+ )? - .with_base_scorer(Arc::clone(scorer)); - return Ok((Arc::new(replacement), rewritten)); - } -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index b763cef..bde742d 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -2449,8 +2449,7 @@ fn test_robotics_e2e_write_then_finalize() { - format!("data/{}", filename), - field_ids, - column_indices, -- meta.major_version as u32, -- meta.minor_version as u32, -+ meta.version, - None, // file_size_bytes - None, // base_id - ); -@@ -3405,6 +3404,7 @@ fn test_index_segment_metadata_parse_rejects_malformed_and_dangerous_input() { - created_at: Some(u64::MAX), - base_id: None, - files: Vec::new(), -+ covering_fields: Vec::new(), - } - .encode_to_vec(); - assert_eq!( -@@ -3429,6 +3429,7 @@ fn test_index_segment_metadata_parse_rejects_malformed_and_dangerous_input() { - created_at: None, - base_id: None, - files: Vec::new(), -+ covering_fields: Vec::new(), - } - .encode_to_vec(); - assert_eq!( -@@ -3456,6 +3457,7 @@ fn test_index_segment_metadata_parse_rejects_malformed_and_dangerous_input() { - created_at: None, - base_id: None, - files: Vec::new(), -+ covering_fields: Vec::new(), - } - .encode_to_vec(); - assert_eq!( -@@ -4033,7 +4035,7 @@ fn test_vector_index_segment_rejects_strict_subset_dot_pq() { - let message = take_last_error_message(); - assert!(message.contains("metric=DOT"), "{message}"); - assert!(message.contains("strict fragment subset"), "{message}"); -- assert!(message.contains("e934cc2c"), "{message}"); -+ assert!(message.contains("ab6b5bbe"), "{message}"); - assert!(message.contains("1 of 2 fragments"), "{message}"); - assert!(!centroids.is_released()); - assert!(!codebook.is_released()); diff --git a/thirdparty/patches/lance-c-0.1.9-pr-79.patch b/thirdparty/patches/lance-c-0.1.9-pr-79.patch deleted file mode 100644 index ad32a48065f09d..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-pr-79.patch +++ /dev/null @@ -1,1782 +0,0 @@ -From d819fbdfa52031d84d1fa01f2d06c51a6712c40b Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Tue, 8 Sep 2026 22:30:48 +0800 -Subject: [PATCH 1/5] scalar index segment - ---- - docs/scalar-segment-scans.md | 78 +++++++++++ - include/lance/lance.h | 24 ++++ - include/lance/lance.hpp | 14 ++ - src/lib.rs | 1 + - src/scalar_segment.rs | 224 +++++++++++++++++++++++++++++++ - src/scanner.rs | 61 +++++++++ - tests/c_api_test.rs | 249 +++++++++++++++++++++++++++++++++++ - 7 files changed, 651 insertions(+) - create mode 100644 docs/scalar-segment-scans.md - create mode 100644 src/scalar_segment.rs - -diff --git a/docs/scalar-segment-scans.md b/docs/scalar-segment-scans.md -new file mode 100644 -index 0000000..dc7c141 ---- /dev/null -+++ b/docs/scalar-segment-scans.md -@@ -0,0 +1,78 @@ -+# Scalar index segment scans -+ -+An ordinary scanner can use one physical BTree/Bitmap segment to generate -+candidates, then read those candidates with the complete scanner filter. This -+does not run a global search of the other segments of the logical index. It does -+not subdivide a physical segment or make its own index search incremental. -+ -+## Configuring a task -+ -+Open a fixed dataset version. Select the physical index UUID from that version's -+metadata and pass the task's complete fragment domain explicitly: -+ -+```c -+LanceScanner *scanner = lance_scanner_new(dataset, columns, full_filter_sql); -+/* Check every return value in production. */ -+lance_scanner_set_fragment_ids(scanner, fragment_ids, fragment_count); -+lance_scanner_set_scalar_index_segment(scanner, segment_uuid_16_bytes); -+lance_scanner_set_limit(scanner, 20000); -+/* The scanner-owning thread calls lance_scanner_next as usual. */ -+``` -+ -+SQL, Substrait and additional SQL filters keep their existing precedence and AND -+composition. The caller does not supply a separate driver predicate: Lance-C -+uses the typed filter planner and selects a necessary indexed leaf belonging to -+the requested logical index. It only descends through AND, never through OR or -+NOT. It then searches the selected UUID and applies the complete filter while -+reading candidates with automatic scalar-index planning disabled. -+ -+Each task's fragment IDs define its result domain, including on fallback. A -+distributed planner must assign disjoint domains whose union covers the intended -+scan. Unindexed fragments need their own tasks, or an explicit domain including -+them (which causes that task to use fallback). Merely listing indexed segments -+does not include appended, unindexed data automatically. -+ -+An unknown UUID, absent fragment or invalid option combination is an error. A -+known segment with incomplete/unknown coverage, no suitable driver, unsupported -+index type, nested key, overlays, fragment reuse, non-exact results or unsupported -+row-ID domain falls back to a non-indexed scan of the entire explicit domain. -+I/O and corruption errors are propagated, not converted to empty results or -+successful fallback. -+ -+The first implementation supports live-row ordinary scans and cannot be combined -+with vector/FTS queries. Physical row-address -+results on stable-row-ID datasets currently fall back; results already expressed -+in the correct row-ID domain use the candidate path. Deletes and all remaining -+predicates are handled by the ordinary reader. No candidate-count limit is -+applied: LIMIT/OFFSET remain after the scanner's complete filter. -+ -+If the host has additional predicates outside Lance, do not set a local limit -+before those predicates. Never divide the global limit by the number of tasks. -+Global OFFSET belongs to the coordinator, not independently to each task. -+ -+## Stopping after the host limit -+ -+This mode uses the existing scanner/stream lifecycle. Once the host has enough -+rows, it stops requesting further batches and closes the scanner after any active -+call has returned. Do not call `lance_scanner_close` concurrently with `next`. -+Exported Arrow streams remain owned by the caller and must also be released after -+their active consumers have finished. -+ -+A host stop flag does not interrupt an in-progress `lance_scanner_next`: current -+index evaluation or I/O may finish before the host observes stop and closes the -+stream. No separate cancellation signal or thread is introduced. The host remains -+responsible for enforcing the global LIMIT across concurrent tasks. -+ -+## Memory and statistics -+ -+Candidate masks stay in Rust, and record batches are streamed. Each active task -+can still hold a complete segment's candidate set; scanner I/O buffer size does -+not cap that allocation. Control task concurrency and physical segment size. -+ -+Successful exhaustion merges segment-search metrics into the existing statistics -+callback exactly once. New metrics include `scalar_segments_requested`, -+`scalar_segments_searched`, `scalar_segment_candidate_rows`, -+`scalar_segment_prepare_time`, `scalar_segment_search_time`, and -+`scalar_segment_fallback_*` reasons. `prepare_time` includes search time. Early -+release, cancellation and errors retain the existing callback contract: final -+statistics are not guaranteed. Metrics do not establish global task concurrency. -diff --git a/include/lance/lance.h b/include/lance/lance.h -index 8173ae5..cbd8330 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -1858,6 +1858,30 @@ int32_t lance_scanner_set_index_segments( - size_t len - ); - -+/** -+ * Accelerate an ordinary scalar-filtered scan with one physical index segment. -+ * segment_uuid points to 16 UUID bytes in RFC 4122 order; NULL clears the setting. -+ * Must be configured before scanning. Requires explicit nonempty fragment_ids, -+ * which define BOTH the read and fallback domain, independently of the segment. -+ * Missing snapshot UUIDs / fragment IDs are errors. Extra segment coverage is -+ * excluded by fragment_ids; incomplete coverage falls back to a full filtered -+ * scan of those fragment_ids. Callers distributing work must assign disjoint -+ * fragment domains and separately include any unindexed data they wish to read. -+ * -+ * BTree/Bitmap searches use a necessary AND-conjunct of the full scanner filter -+ * on the selected logical index. All predicates are reapplied during candidate -+ * reads; other scalar indices are disabled. OR/NOT-only filters, overlays, -+ * fragment reuse, unsupported index types / result domains -+ * and missing coverage use the same domain without an index. No filter also -+ * falls back. LIMIT/OFFSET apply after the complete scanner filter, never to the -+ * unfiltered candidate set. Vector/FTS queries are rejected. -+ * -+ * UUID bytes are copied. Metadata and final option compatibility are validated -+ * when creating the stream. Index corruption or I/O failures remain errors. -+ */ -+int32_t lance_scanner_set_scalar_index_segment( -+ LanceScanner* scanner, const uint8_t* segment_uuid); -+ - /* ─── Full-text search (Phase 2) ─── */ - - /** -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index c12c0c6..8e1c6cf 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1311,6 +1311,20 @@ class Scanner { - return *this; - } - -+ /// Restrict scalar candidate generation to one segment; fragment_ids is -+ /// required and defines the complete read/fallback domain. See lance.h. -+ Scanner& scalar_index_segment(const std::array& segment_uuid) { -+ if (lance_scanner_set_scalar_index_segment(handle_.get(), segment_uuid.data()) != 0) -+ check_error(); -+ return *this; -+ } -+ -+ Scanner& clear_scalar_index_segment() { -+ if (lance_scanner_set_scalar_index_segment(handle_.get(), nullptr) != 0) -+ check_error(); -+ return *this; -+ } -+ - /// Restrict scan to specific fragment IDs. - Scanner& fragment_ids(const uint64_t* ids, size_t len) { - if (lance_scanner_set_fragment_ids(handle_.get(), ids, len) != 0) -diff --git a/src/lib.rs b/src/lib.rs -index 8b212f5..c7ca4cf 100644 ---- a/src/lib.rs -+++ b/src/lib.rs -@@ -39,6 +39,7 @@ mod index_segment; - mod merge_insert; - mod restore; - pub mod runtime; -+mod scalar_segment; - mod scanner; - mod session; - pub mod stream_guard; -diff --git a/src/scalar_segment.rs b/src/scalar_segment.rs -new file mode 100644 -index 0000000..2ab1297 ---- /dev/null -+++ b/src/scalar_segment.rs -@@ -0,0 +1,224 @@ -+// SPDX-License-Identifier: Apache-2.0 -+// SPDX-FileCopyrightText: Copyright The Lance Authors -+ -+//! Segment-scoped candidate generation for ordinary scans. The explicit fragment -+//! list is the read domain, including on fallback; a segment is only an accelerator. -+ -+use std::collections::HashSet; -+use std::sync::Arc; -+use std::time::Instant; -+ -+use datafusion::physical_plan::metrics::ExecutionPlanMetricsSet; -+use lance::Dataset; -+use lance::dataset::scanner::{ -+ ExecutionStatsCallback, ExecutionSummaryCounts, RowAddrMask, Scanner, -+}; -+use lance::index::{DatasetIndexExt, DatasetIndexInternalExt}; -+use lance::io::exec::utils::IndexMetrics; -+use lance_core::{Error, Result}; -+use lance_datafusion::planner::Planner; -+use lance_datafusion::utils::MetricsExt; -+use lance_index::IndexType; -+use lance_index::scalar::SearchResult; -+use lance_index::scalar::expression::{PlannerIndexExt, ScalarIndexExpr, ScalarIndexSearch}; -+use uuid::Uuid; -+ -+pub(crate) struct PreparedScalarSegment { -+ pub dataset: Arc, -+ pub segment_uuid: Uuid, -+ pub fragment_ids: Vec, -+ pub callback: Option, -+} -+ -+fn invalid(message: impl Into) -> Error { -+ Error::invalid_input_source(message.into().into()) -+} -+ -+// Only descend through AND: a leaf below OR or NOT need not contain all matches -+// of the full expression. The original expression is always reapplied by reader. -+fn driver<'a>(expr: &'a ScalarIndexExpr, index_name: &str) -> Option<&'a ScalarIndexSearch> { -+ match expr { -+ ScalarIndexExpr::Query(search) if search.index_name == index_name => Some(search), -+ ScalarIndexExpr::And(lhs, rhs) => { -+ driver(lhs, index_name).or_else(|| driver(rhs, index_name)) -+ } -+ _ => None, -+ } -+} -+ -+impl PreparedScalarSegment { -+ pub async fn configure(self, mut reader: Scanner) -> Result { -+ // Never let either candidate reads or fallback re-enter a global index search. -+ reader.use_scalar_index(false); -+ let mut stats = ExecutionSummaryCounts::default(); -+ stats -+ .all_counts -+ .insert("scalar_segments_requested".into(), 1); -+ let plan_metrics = ExecutionPlanMetricsSet::new(); -+ let metrics = IndexMetrics::new(&plan_metrics, 0); -+ let started = Instant::now(); -+ let reason = self -+ .configure_candidates(&mut reader, &metrics, &mut stats) -+ .await?; -+ metrics.flush_io(); -+ stats.all_times.insert( -+ "scalar_segment_prepare_time".into(), -+ started.elapsed().as_nanos().min(usize::MAX as u128) as usize, -+ ); -+ if let Some(reason) = reason { -+ stats -+ .all_counts -+ .insert("scalar_segment_fallbacks".into(), 1); -+ stats -+ .all_counts -+ .insert(format!("scalar_segment_fallback_{reason}"), 1); -+ } -+ for (name, count) in plan_metrics.clone_inner().iter_counts() { -+ let name = name.as_ref(); -+ match name { -+ "iops" => stats.iops += count.value(), -+ "requests" => stats.requests += count.value(), -+ "bytes_read" => stats.bytes_read += count.value(), -+ "indices_loaded" => stats.indices_loaded += count.value(), -+ "parts_loaded" => stats.parts_loaded += count.value(), -+ "index_comparisons" => stats.index_comparisons += count.value(), -+ _ => *stats.all_counts.entry(name.to_string()).or_default() += count.value(), -+ } -+ } -+ if let Some(callback) = self.callback { -+ // Preserve the callback's once-per-successfully-exhausted-stream contract. -+ // Candidate work is not part of the underlying reader's plan metrics. -+ reader.scan_stats_callback(Arc::new(move |read| { -+ let mut combined = read.clone(); -+ combined.iops += stats.iops; -+ combined.requests += stats.requests; -+ combined.bytes_read += stats.bytes_read; -+ combined.indices_loaded += stats.indices_loaded; -+ combined.parts_loaded += stats.parts_loaded; -+ combined.index_comparisons += stats.index_comparisons; -+ for (name, value) in &stats.all_counts { -+ *combined.all_counts.entry(name.clone()).or_default() += value; -+ } -+ for (name, value) in &stats.all_times { -+ *combined.all_times.entry(name.clone()).or_default() += value; -+ } -+ callback(&combined); -+ })); -+ } -+ Ok(reader) -+ } -+ -+ async fn configure_candidates( -+ &self, -+ reader: &mut Scanner, -+ metrics: &IndexMetrics, -+ stats: &mut ExecutionSummaryCounts, -+ ) -> Result> { -+ let fragments = self.dataset.get_fragments(); -+ let visible: HashSet = fragments.iter().map(|f| f.id() as u64).collect(); -+ if self.fragment_ids.iter().any(|id| !visible.contains(id)) { -+ return Err(invalid( -+ "scalar segment fragment_ids contains a fragment absent from the dataset snapshot", -+ )); -+ } -+ let indices = self.dataset.load_indices().await?; -+ let index_meta = indices -+ .iter() -+ .find(|i| i.uuid == self.segment_uuid) -+ .ok_or_else(|| { -+ invalid(format!( -+ "scalar index segment {} is absent from the dataset snapshot", -+ self.segment_uuid -+ )) -+ })?; -+ let field_id = index_meta -+ .keyed_field() -+ .ok_or_else(|| invalid("scalar segment must index a single key field"))?; -+ let field = -+ self.dataset.schema().field_by_id(field_id).ok_or_else(|| { -+ invalid("scalar segment key field is absent from the dataset schema") -+ })?; -+ // Keep V1 to flat scalar fields. A dotted name is not sufficient to prove -+ // the field path of an evolved or nested schema. -+ if !self -+ .dataset -+ .schema() -+ .fields -+ .iter() -+ .any(|f| f.id == field.id) -+ { -+ return Ok(Some("nested_field")); -+ } -+ let scope: HashSet = self.fragment_ids.iter().copied().collect(); -+ let Some(coverage) = index_meta.fragment_bitmap.as_ref() else { -+ return Ok(Some("unknown_coverage")); -+ }; -+ if self -+ .fragment_ids -+ .iter() -+ .any(|id| u32::try_from(*id).map_or(true, |id| !coverage.contains(id))) -+ { -+ // Scan the ENTIRE explicit read domain, not just the covered part. -+ return Ok(Some("partial_coverage")); -+ } -+ if fragments -+ .iter() -+ .filter(|f| scope.contains(&(f.id() as u64))) -+ .any(|f| !f.metadata().overlays.is_empty() || f.metadata().physical_rows.is_none()) -+ { -+ return Ok(Some("fragment_state")); -+ } -+ // Fragment reuse can change the domain of an old segment. Until its -+ // coverage mapping is handled here, preserve correctness with a scoped scan. -+ if self.dataset.frag_reuse_index_uuid().await.is_some() { -+ return Ok(Some("fragment_reuse")); -+ } -+ let Some(filter) = reader.get_expr_filter()? else { -+ return Ok(Some("no_filter")); -+ }; -+ let planner = Planner::new(Arc::new(self.dataset.schema().into())); -+ let index_info = self.dataset.scalar_index_info().await?; -+ let filter_plan = planner.create_filter_plan(filter, &index_info, true)?; -+ let Some(search) = filter_plan -+ .index_query -+ .as_ref() -+ .and_then(|expr| driver(expr, &index_meta.name)) -+ else { -+ return Ok(Some("no_driver")); -+ }; -+ if search.column != field.name { -+ return Ok(Some("field_path")); -+ } -+ let index = self -+ .dataset -+ .open_scalar_index(&search.column, &self.segment_uuid, metrics) -+ .await?; -+ if !matches!(index.index_type(), IndexType::BTree | IndexType::Bitmap) { -+ return Ok(Some("index_type")); -+ } -+ // External masks use _rowid, not necessarily physical row addresses. -+ if index.results_are_row_addresses() && self.dataset.manifest.uses_stable_row_ids() { -+ return Ok(Some("row_id_domain")); -+ } -+ let started = Instant::now(); -+ let result = index.search(search.query.as_ref(), metrics).await?; -+ stats.all_times.insert( -+ "scalar_segment_search_time".into(), -+ started.elapsed().as_nanos().min(usize::MAX as u128) as usize, -+ ); -+ stats -+ .all_counts -+ .insert("scalar_segments_searched".into(), 1); -+ let SearchResult::Exact(rows) = result else { -+ return Ok(Some("inexact_result")); -+ }; -+ stats.all_counts.insert( -+ "scalar_segment_candidate_rows".into(), -+ rows.len().unwrap_or(0) as usize, -+ ); -+ // Do not truncate candidates at LIMIT. The reader evaluates the complete -+ // filter before applying its existing limit/offset operators. -+ reader.with_row_addr_prefilter(RowAddrMask::from_allowed(rows.selected_rows().clone())); -+ Ok(None) -+ } -+} -diff --git a/src/scanner.rs b/src/scanner.rs -index 4ceeb0e..ac3cff6 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -39,6 +39,7 @@ use crate::fts_query::{ - }; - use crate::helpers; - use crate::runtime::{RT, block_on}; -+use crate::scalar_segment::PreparedScalarSegment; - use crate::stream_guard::GuardedReader; - - /// Data type tag for query vectors, mirroring the C enum `LanceDataType`. -@@ -108,6 +109,7 @@ pub struct LanceScanner { - include_deleted_rows: bool, - fragment_ids: Option>, - index_segments: Option>, -+ scalar_index_segment: Option, - nearest: Option, - nprobes: NprobesRange, - approx_mode: Option, -@@ -256,6 +258,7 @@ impl LanceScanner { - include_deleted_rows: false, - fragment_ids: None, - index_segments: None, -+ scalar_index_segment: None, - nearest: None, - nprobes: NprobesRange::default(), - approx_mode: None, -@@ -470,12 +473,37 @@ impl LanceScanner { - None - }; - self.apply_filter(&mut scanner)?; -+ let scalar_segment = if let Some(segment_uuid) = self.scalar_index_segment { -+ if self.nearest.is_some() -+ || self.fts_query.is_some() -+ || self.fts_context.is_some() -+ || self.index_segments.is_some() -+ || self.fts_index_segments.is_some() -+ { -+ return Err(lance_core::Error::invalid_input_source( -+ "scalar_index_segment requires an ordinary scan of live rows".into(), -+ )); -+ } -+ let fragment_ids = self.fragment_ids.as_ref().filter(|ids| !ids.is_empty()) -+ .ok_or_else(|| lance_core::Error::invalid_input_source( -+ "scalar_index_segment requires explicit nonempty fragment_ids for its read and fallback domain".into(), -+ ))?; -+ Some(PreparedScalarSegment { -+ dataset: Arc::clone(&self.dataset), -+ segment_uuid, -+ fragment_ids: fragment_ids.clone(), -+ callback: self.scan_statistics_callback.clone(), -+ }) -+ } else { -+ None -+ }; - if let Some(callback) = &self.scan_statistics_callback { - scanner.scan_stats_callback(callback.clone()); - } - Ok(PreparedScanner { - scanner, - distributed_fts, -+ scalar_segment, - }) - } - } -@@ -490,10 +518,18 @@ struct PreparedFtsExecution { - struct PreparedScanner { - scanner: lance::dataset::scanner::Scanner, - distributed_fts: Option, -+ scalar_segment: Option, - } - - impl PreparedScanner { - async fn try_into_stream(self) -> Result { -+ if let Some(scalar_segment) = self.scalar_segment { -+ return scalar_segment -+ .configure(self.scanner) -+ .await? -+ .try_into_stream() -+ .await; -+ } - let Some(distributed_fts) = self.distributed_fts else { - return self.scanner.try_into_stream().await; - }; -@@ -858,6 +894,31 @@ macro_rules! scanner_ffi_try { - }}; - } - -+/// Select one physical scalar index segment. NULL clears the selection. -+/// Requires explicit fragment_ids and an ordinary live-row scan. See the C header. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_set_scalar_index_segment( -+ scanner: *mut LanceScanner, -+ segment_uuid: *const u8, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, { -+ let scanner = unsafe { scanner.as_mut() } -+ .ok_or_else(|| lance_core::Error::invalid_input_source("scanner is NULL".into()))?; -+ scanner.ensure_scan_not_started("scalar_index_segment")?; -+ let segment = if segment_uuid.is_null() { -+ None -+ } else { -+ Some( -+ Uuid::from_slice(unsafe { std::slice::from_raw_parts(segment_uuid, 16) }) -+ .map_err(|e| lance_core::Error::invalid_input_source(e.into()))?, -+ ) -+ }; -+ scanner.scalar_index_segment = segment; -+ Ok(0) -+ }) -+} -+ - // --------------------------------------------------------------------------- - // Scanner lifecycle + builder - // --------------------------------------------------------------------------- -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 3b3424b..a8a7eab 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -12513,3 +12513,252 @@ fn test_add_columns_stream_null_dataset_consumes_stream() { - assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); - assert_stream_consumed(&stream, &drop_count); - } -+ -+// Segment scans deliberately use an unprojected nullable key and a residual -+// predicate so a candidate LIMIT or loss of filter columns changes the answer. -+fn create_scalar_segment_fixture( -+ kind: lance_index::IndexType, -+ stable: bool, -+) -> (tempfile::TempDir, String, Vec<[u8; 16]>) { -+ use lance::dataset::WriteParams; -+ use lance::index::DatasetIndexExt; -+ use lance_index::scalar::{BuiltinIndexType, ScalarIndexParams}; -+ let tmp = tempfile::tempdir().unwrap(); -+ let uri = tmp.path().join("segments").to_str().unwrap().to_owned(); -+ let uuids = lance_c::runtime::block_on(async { -+ let schema = Arc::new(Schema::new(vec![ -+ Field::new("id", DataType::Int32, false), -+ Field::new("key", DataType::Int32, true), -+ ])); -+ let batch = RecordBatch::try_new( -+ schema.clone(), -+ vec![ -+ Arc::new(Int32Array::from_iter_values(0..12)), -+ Arc::new(Int32Array::from( -+ (0..12) -+ .map(|id| if id % 4 == 0 { None } else { Some(id % 3) }) -+ .collect::>(), -+ )), -+ ], -+ ) -+ .unwrap(); -+ let mut ds = Dataset::write( -+ arrow::record_batch::RecordBatchIterator::new(vec![Ok(batch)], schema), -+ &uri, -+ Some(WriteParams { -+ max_rows_per_file: 4, -+ enable_stable_row_ids: stable, -+ ..Default::default() -+ }), -+ ) -+ .await -+ .unwrap(); -+ let params = ScalarIndexParams::for_builtin(if kind == lance_index::IndexType::Bitmap { -+ BuiltinIndexType::Bitmap -+ } else { -+ BuiltinIndexType::BTree -+ }); -+ let fragments = ds.get_fragments(); -+ assert_eq!(fragments.len(), 3); -+ let mut segments = Vec::new(); -+ for fragment in fragments.iter().take(2) { -+ segments.push( -+ ds.create_index_builder(&["key"], kind, ¶ms) -+ .name("key_idx".into()) -+ .fragments(vec![fragment.id() as u32]) -+ .execute_uncommitted() -+ .await -+ .unwrap(), -+ ); -+ } -+ let uuids = segments.iter().map(|s| *s.uuid.as_bytes()).collect(); -+ ds.commit_existing_index_segments("key_idx", "key", segments) -+ .await -+ .unwrap(); -+ uuids -+ }); -+ (tmp, uri, uuids) -+} -+ -+fn scalar_segment_ids( -+ uri: &str, -+ uuid: &[u8; 16], -+ fragments: &[u64], -+ filter: &str, -+ limit: Option, -+ offset: i64, -+) -> (Vec, CapturedScanStatistics) { -+ let uri = c_str(uri); -+ let filter = c_str(filter); -+ let id = c_str("id"); -+ let columns = [id.as_ptr(), ptr::null()]; -+ let mut captured = CapturedScanStatistics::default(); -+ let mut ids = Vec::new(); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ assert!(!ds.is_null()); -+ let scanner = lance_scanner_new(ds, columns.as_ptr(), filter.as_ptr()); -+ assert!(!scanner.is_null()); -+ assert_eq!( -+ lance_scanner_set_fragment_ids(scanner, fragments.as_ptr(), fragments.len()), -+ 0 -+ ); -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(scanner, uuid.as_ptr()), -+ 0 -+ ); -+ if let Some(limit) = limit { -+ assert_eq!(lance_scanner_set_limit(scanner, limit), 0); -+ } -+ assert_eq!(lance_scanner_set_offset(scanner, offset), 0); -+ assert_eq!( -+ lance_scanner_set_statistics_callback( -+ scanner, -+ Some(capture_scan_statistics), -+ (&mut captured as *mut CapturedScanStatistics).cast() -+ ), -+ 0 -+ ); -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ let rc = lance_scanner_to_arrow_stream(scanner, &mut stream); -+ assert_eq!( -+ rc, -+ 0, -+ "{}", -+ if rc != 0 { -+ take_last_error_message() -+ } else { -+ String::new() -+ } -+ ); -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(scanner, ptr::null()), -+ -1 -+ ); -+ { -+ let reader = ArrowArrayStreamReader::from_raw(&mut stream).unwrap(); -+ for batch in reader { -+ let batch = batch.unwrap(); -+ assert_eq!(batch.num_columns(), 1); -+ ids.extend( -+ batch -+ .column(0) -+ .as_any() -+ .downcast_ref::() -+ .unwrap() -+ .values() -+ .iter() -+ .copied(), -+ ); -+ } -+ } -+ lance_scanner_close(scanner); -+ lance_dataset_close(ds); -+ } -+ (ids, captured) -+} -+ -+#[test] -+fn test_scalar_segment_scope_residual_limit_and_unindexed_fallback() { -+ for kind in [ -+ lance_index::IndexType::BTree, -+ lance_index::IndexType::Bitmap, -+ ] { -+ let (_tmp, uri, uuids) = create_scalar_segment_fixture(kind, false); -+ let (ids, stats) = -+ scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", None, 0); -+ assert_eq!(ids, vec![2, 3]); -+ assert_eq!(stats.calls, 1); -+ assert!( -+ stats -+ .metrics -+ .iter() -+ .any(|(name, _, value)| name == "scalar_segments_searched" && *value == 1) -+ ); -+ let (ids, _) = -+ scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", Some(1), 1); -+ assert_eq!( -+ ids, -+ vec![3], -+ "offset and limit must apply after residual filtering" -+ ); -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[1], &[1], "key >= 0 AND id >= 2", None, 0); -+ assert_eq!(ids, vec![5, 6, 7]); -+ let (ids, stats) = -+ scalar_segment_ids(&uri, &uuids[0], &[0, 2], "key >= 0 AND id >= 2", None, 0); -+ assert_eq!( -+ ids, -+ vec![2, 3, 9, 10, 11], -+ "partial coverage must not omit unindexed rows" -+ ); -+ assert!( -+ stats -+ .metrics -+ .iter() -+ .any(|(name, _, _)| name == "scalar_segment_fallback_partial_coverage") -+ ); -+ let (ids, stats) = scalar_segment_ids(&uri, &uuids[0], &[0], "key = 99 OR id = 0", None, 0); -+ assert_eq!( -+ ids, -+ vec![0], -+ "OR must not use just one branch as candidates" -+ ); -+ assert!( -+ stats -+ .metrics -+ .iter() -+ .any(|(name, _, _)| name == "scalar_segment_fallback_no_driver") -+ ); -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[0], &[0], "key = 99", None, 0); -+ assert!(ids.is_empty()); -+ } -+} -+ -+#[test] -+fn test_scalar_segment_stable_row_ids_and_deletes() { -+ use lance::index::DatasetIndexExt; -+ let (_tmp, uri, uuids) = create_scalar_segment_fixture(lance_index::IndexType::BTree, true); -+ lance_c::runtime::block_on(async { -+ let mut ds = Dataset::open(&uri).await.unwrap(); -+ ds.delete("id = 2").await.unwrap(); -+ assert_eq!(ds.load_indices().await.unwrap().len(), 2); -+ }); -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", None, 0); -+ assert_eq!(ids, vec![3]); -+} -+ -+#[test] -+fn test_scalar_segment_requires_explicit_domain_and_checks_uuid() { -+ let (_tmp, uri, uuids) = create_scalar_segment_fixture(lance_index::IndexType::BTree, false); -+ let uri = c_str(&uri); -+ let filter = c_str("key >= 0"); -+ unsafe { -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(ptr::null_mut(), ptr::null()), -+ -1 -+ ); -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ let scanner = lance_scanner_new(ds, ptr::null(), filter.as_ptr()); -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(scanner, uuids[0].as_ptr()), -+ 0 -+ ); -+ let mut batch = ptr::null_mut(); -+ assert_eq!(lance_scanner_next(scanner, &mut batch), -1); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ lance_scanner_close(scanner); -+ let scanner = lance_scanner_new(ds, ptr::null(), filter.as_ptr()); -+ assert_eq!( -+ lance_scanner_set_fragment_ids(scanner, [0u64].as_ptr(), 1), -+ 0 -+ ); -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(scanner, [0u8; 16].as_ptr()), -+ 0 -+ ); -+ assert_eq!(lance_scanner_next(scanner, &mut batch), -1); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ lance_scanner_close(scanner); -+ lance_dataset_close(ds); -+ } -+} - -From 17240674d739293a494c07dfe5aec92a97185363 Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Tue, 8 Sep 2026 23:17:23 +0800 -Subject: [PATCH 2/5] update - ---- - docs/scalar-segment-scans.md | 3 ++ - include/lance/lance.h | 4 +-- - src/scalar_segment.rs | 7 +++++ - tests/c_api_test.rs | 60 ++++++++++++++++++++++++++++++++++-- - 4 files changed, 70 insertions(+), 4 deletions(-) - -diff --git a/docs/scalar-segment-scans.md b/docs/scalar-segment-scans.md -index dc7c141..f72dbfc 100644 ---- a/docs/scalar-segment-scans.md -+++ b/docs/scalar-segment-scans.md -@@ -36,6 +36,9 @@ An unknown UUID, absent fragment or invalid option combination is an error. A - known segment with incomplete/unknown coverage, no suitable driver, unsupported - index type, nested key, overlays, fragment reuse, non-exact results or unsupported - row-ID domain falls back to a non-indexed scan of the entire explicit domain. -+Legacy (v1) storage also takes this fallback because ordinary scans cannot consume -+external row masks; it reports `scalar_segment_fallback_legacy_storage` without -+searching the index. - I/O and corruption errors are propagated, not converted to empty results or - successful fallback. - -diff --git a/include/lance/lance.h b/include/lance/lance.h -index cbd8330..ab15247 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -1870,8 +1870,8 @@ int32_t lance_scanner_set_index_segments( - * - * BTree/Bitmap searches use a necessary AND-conjunct of the full scanner filter - * on the selected logical index. All predicates are reapplied during candidate -- * reads; other scalar indices are disabled. OR/NOT-only filters, overlays, -- * fragment reuse, unsupported index types / result domains -+ * reads; other scalar indices are disabled. Legacy storage, OR/NOT-only filters, -+ * overlays, fragment reuse, unsupported index types / result domains - * and missing coverage use the same domain without an index. No filter also - * falls back. LIMIT/OFFSET apply after the complete scanner filter, never to the - * unfiltered candidate set. Vector/FTS queries are rejected. -diff --git a/src/scalar_segment.rs b/src/scalar_segment.rs -index 2ab1297..5e121c9 100644 ---- a/src/scalar_segment.rs -+++ b/src/scalar_segment.rs -@@ -138,6 +138,13 @@ impl PreparedScalarSegment { - self.dataset.schema().field_by_id(field_id).ok_or_else(|| { - invalid("scalar segment key field is absent from the dataset schema") - })?; -+ // Match Lance's plain-scan external-mask restriction. Keep the scoped, -+ // full-filtered reader intact and avoid index work on legacy storage. -+ if self.dataset.manifest().data_storage_format.lance_file_format() -+ == lance_file::version::ConcreteFileVersion::V1 -+ { -+ return Ok(Some("legacy_storage")); -+ } - // Keep V1 to flat scalar fields. A dotted name is not sufficient to prove - // the field path of an evolved or nested schema. - if !self -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index a8a7eab..0cb998b 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -12519,6 +12519,15 @@ fn test_add_columns_stream_null_dataset_consumes_stream() { - fn create_scalar_segment_fixture( - kind: lance_index::IndexType, - stable: bool, -+) -> (tempfile::TempDir, String, Vec<[u8; 16]>) { -+ create_scalar_segment_fixture_with_options(kind, stable, None, &[&[0], &[1]]) -+} -+ -+fn create_scalar_segment_fixture_with_options( -+ kind: lance_index::IndexType, -+ stable: bool, -+ storage_version: Option, -+ segment_fragments: &[&[u32]], - ) -> (tempfile::TempDir, String, Vec<[u8; 16]>) { - use lance::dataset::WriteParams; - use lance::index::DatasetIndexExt; -@@ -12548,6 +12557,7 @@ fn create_scalar_segment_fixture( - Some(WriteParams { - max_rows_per_file: 4, - enable_stable_row_ids: stable, -+ data_storage_version: storage_version, - ..Default::default() - }), - ) -@@ -12561,11 +12571,11 @@ fn create_scalar_segment_fixture( - let fragments = ds.get_fragments(); - assert_eq!(fragments.len(), 3); - let mut segments = Vec::new(); -- for fragment in fragments.iter().take(2) { -+ for fragment_ids in segment_fragments { - segments.push( - ds.create_index_builder(&["key"], kind, ¶ms) - .name("key_idx".into()) -- .fragments(vec![fragment.id() as u32]) -+ .fragments(fragment_ids.to_vec()) - .execute_uncommitted() - .await - .unwrap(), -@@ -12714,6 +12724,52 @@ fn test_scalar_segment_scope_residual_limit_and_unindexed_fallback() { - } - } - -+#[test] -+fn test_scalar_segment_legacy_storage_falls_back() { -+ use lance_file::version::{ConcreteFileVersion, LanceFileVersion}; -+ -+ // Three fragments, with one segment covering 0 and 1. Reading only fragment -+ // 0 must retain the full predicate and must not leak rows from fragment 1. -+ let (_tmp, uri, uuids) = create_scalar_segment_fixture_with_options( -+ lance_index::IndexType::BTree, -+ false, -+ Some(LanceFileVersion::Legacy), -+ &[&[0, 1]], -+ ); -+ assert_eq!(uuids.len(), 1); -+ lance_c::runtime::block_on(async { -+ let ds = Dataset::open(&uri).await.unwrap(); -+ assert_eq!( -+ ds.manifest().data_storage_format.lance_file_format(), -+ ConcreteFileVersion::V1 -+ ); -+ }); -+ -+ let (ids, stats) = -+ scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", None, 0); -+ assert_eq!(ids, vec![2, 3]); -+ assert_eq!(stats.calls, 1); -+ assert_eq!(stats.indices_loaded, 0); -+ assert_eq!(stats.index_comparisons, 0); -+ assert!(stats.metrics.iter().any(|(name, _, value)| { -+ name == "scalar_segment_fallback_legacy_storage" && *value == 1 -+ })); -+ assert!( -+ !stats -+ .metrics -+ .iter() -+ .any(|(name, _, value)| name == "scalar_segments_searched" && *value != 0) -+ ); -+ -+ let (ids, _) = -+ scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", Some(1), 1); -+ assert_eq!( -+ ids, -+ vec![3], -+ "fallback must retain LIMIT/OFFSET after filtering" -+ ); -+} -+ - #[test] - fn test_scalar_segment_stable_row_ids_and_deletes() { - use lance::index::DatasetIndexExt; - -From f9ce263e4e32540c26afa6ccd18021ca81c090f1 Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Tue, 8 Sep 2026 23:21:02 +0800 -Subject: [PATCH 3/5] update - ---- - src/scalar_segment.rs | 6 +++++- - tests/c_api_test.rs | 6 ++---- - 2 files changed, 7 insertions(+), 5 deletions(-) - -diff --git a/src/scalar_segment.rs b/src/scalar_segment.rs -index 5e121c9..01df58b 100644 ---- a/src/scalar_segment.rs -+++ b/src/scalar_segment.rs -@@ -140,7 +140,11 @@ impl PreparedScalarSegment { - })?; - // Match Lance's plain-scan external-mask restriction. Keep the scoped, - // full-filtered reader intact and avoid index work on legacy storage. -- if self.dataset.manifest().data_storage_format.lance_file_format() -+ if self -+ .dataset -+ .manifest() -+ .data_storage_format -+ .lance_file_format() - == lance_file::version::ConcreteFileVersion::V1 - { - return Ok(Some("legacy_storage")); -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 0cb998b..efd4284 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -12745,8 +12745,7 @@ fn test_scalar_segment_legacy_storage_falls_back() { - ); - }); - -- let (ids, stats) = -- scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", None, 0); -+ let (ids, stats) = scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", None, 0); - assert_eq!(ids, vec![2, 3]); - assert_eq!(stats.calls, 1); - assert_eq!(stats.indices_loaded, 0); -@@ -12761,8 +12760,7 @@ fn test_scalar_segment_legacy_storage_falls_back() { - .any(|(name, _, value)| name == "scalar_segments_searched" && *value != 0) - ); - -- let (ids, _) = -- scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", Some(1), 1); -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0 AND id >= 2", Some(1), 1); - assert_eq!( - ids, - vec![3], - -From 221d8800f56790ab8b48b12f3aaaf9da06359380 Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Wed, 9 Sep 2026 12:59:08 +0800 -Subject: [PATCH 4/5] add LabelList - ---- - docs/scalar-segment-scans.md | 12 ++- - include/lance/lance.h | 8 +- - include/lance/lance.hpp | 4 +- - src/scalar_segment.rs | 7 +- - tests/c_api_test.rs | 190 ++++++++++++++++++++++++++++++++--- - 5 files changed, 200 insertions(+), 21 deletions(-) - -diff --git a/docs/scalar-segment-scans.md b/docs/scalar-segment-scans.md -index f72dbfc..72f6188 100644 ---- a/docs/scalar-segment-scans.md -+++ b/docs/scalar-segment-scans.md -@@ -1,10 +1,18 @@ - # Scalar index segment scans - --An ordinary scanner can use one physical BTree/Bitmap segment to generate --candidates, then read those candidates with the complete scanner filter. This -+An ordinary scanner can use one physical BTree, Bitmap, or LabelList segment to -+generate candidates, then read them with the complete scanner filter. This - does not run a global search of the other segments of the logical index. It does - not subdivide a physical segment or make its own index search incremental. - -+LabelList supports indexed array membership predicates. Every candidate search -+must return `SearchResult::Exact`. -+LabelList query values should match the array element type, for example -+`array_contains(int32_labels, CAST(42 AS INT))`; a cast on the indexed column -+can prevent the planner from finding an index driver and cause fallback. -+`AtMost` and `AtLeast` results still fall back; this mode does not enable FMIndex, -+NGram, BloomFilter, ZoneMap, or Inverted indices. -+ - ## Configuring a task - - Open a fixed dataset version. Select the physical index UUID from that version's -diff --git a/include/lance/lance.h b/include/lance/lance.h -index ab15247..b5c6902 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -1868,9 +1868,11 @@ int32_t lance_scanner_set_index_segments( - * scan of those fragment_ids. Callers distributing work must assign disjoint - * fragment domains and separately include any unindexed data they wish to read. - * -- * BTree/Bitmap searches use a necessary AND-conjunct of the full scanner filter -- * on the selected logical index. All predicates are reapplied during candidate -- * reads; other scalar indices are disabled. Legacy storage, OR/NOT-only filters, -+ * BTree/Bitmap/LabelList searches use a necessary AND-conjunct of the -+ * full scanner filter on the selected logical index and require an Exact result. -+ * AtMost/AtLeast results fall back to a full filtered scan of fragment_ids. -+ * All predicates are reapplied during candidate reads; other scalar indices -+ * are disabled. Legacy storage, OR/NOT-only filters, - * overlays, fragment reuse, unsupported index types / result domains - * and missing coverage use the same domain without an index. No filter also - * falls back. LIMIT/OFFSET apply after the complete scanner filter, never to the -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index 8e1c6cf..ebb8141 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1311,8 +1311,8 @@ class Scanner { - return *this; - } - -- /// Restrict scalar candidate generation to one segment; fragment_ids is -- /// required and defines the complete read/fallback domain. See lance.h. -+ /// Generate exact candidates from one BTree/Bitmap/LabelList segment. -+ /// fragment_ids is required and defines the complete read/fallback domain. See lance.h. - Scanner& scalar_index_segment(const std::array& segment_uuid) { - if (lance_scanner_set_scalar_index_segment(handle_.get(), segment_uuid.data()) != 0) - check_error(); -diff --git a/src/scalar_segment.rs b/src/scalar_segment.rs -index 01df58b..1faf500 100644 ---- a/src/scalar_segment.rs -+++ b/src/scalar_segment.rs -@@ -204,7 +204,12 @@ impl PreparedScalarSegment { - .dataset - .open_scalar_index(&search.column, &self.segment_uuid, metrics) - .await?; -- if !matches!(index.index_type(), IndexType::BTree | IndexType::Bitmap) { -+ // These implementations can return exact candidates. Keep the runtime -+ // Exact check below: a type alone is not a guarantee for every query. -+ if !matches!( -+ index.index_type(), -+ IndexType::BTree | IndexType::Bitmap | IndexType::LabelList -+ ) { - return Ok(Some("index_type")); - } - // External masks use _rowid, not necessarily physical row addresses. -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index efd4284..3322d42 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -12528,6 +12528,21 @@ fn create_scalar_segment_fixture_with_options( - stable: bool, - storage_version: Option, - segment_fragments: &[&[u32]], -+) -> (tempfile::TempDir, String, Vec<[u8; 16]>) { -+ let key = Arc::new(Int32Array::from( -+ (0..12) -+ .map(|id| if id % 4 == 0 { None } else { Some(id % 3) }) -+ .collect::>(), -+ )); -+ create_scalar_segment_fixture_from_key(kind, stable, storage_version, segment_fragments, key) -+} -+ -+fn create_scalar_segment_fixture_from_key( -+ kind: lance_index::IndexType, -+ stable: bool, -+ storage_version: Option, -+ segment_fragments: &[&[u32]], -+ key: arrow_array::ArrayRef, - ) -> (tempfile::TempDir, String, Vec<[u8; 16]>) { - use lance::dataset::WriteParams; - use lance::index::DatasetIndexExt; -@@ -12537,17 +12552,14 @@ fn create_scalar_segment_fixture_with_options( - let uuids = lance_c::runtime::block_on(async { - let schema = Arc::new(Schema::new(vec![ - Field::new("id", DataType::Int32, false), -- Field::new("key", DataType::Int32, true), -+ Field::new("key", key.data_type().clone(), true), - ])); -+ let row_count = key.len(); - let batch = RecordBatch::try_new( - schema.clone(), - vec![ -- Arc::new(Int32Array::from_iter_values(0..12)), -- Arc::new(Int32Array::from( -- (0..12) -- .map(|id| if id % 4 == 0 { None } else { Some(id % 3) }) -- .collect::>(), -- )), -+ Arc::new(Int32Array::from_iter_values(0..row_count as i32)), -+ key, - ], - ) - .unwrap(); -@@ -12563,13 +12575,9 @@ fn create_scalar_segment_fixture_with_options( - ) - .await - .unwrap(); -- let params = ScalarIndexParams::for_builtin(if kind == lance_index::IndexType::Bitmap { -- BuiltinIndexType::Bitmap -- } else { -- BuiltinIndexType::BTree -- }); -+ let params = ScalarIndexParams::for_builtin(BuiltinIndexType::try_from(kind).unwrap()); - let fragments = ds.get_fragments(); -- assert_eq!(fragments.len(), 3); -+ assert_eq!(fragments.len(), row_count.div_ceil(4)); - let mut segments = Vec::new(); - for fragment_ids in segment_fragments { - segments.push( -@@ -12724,6 +12732,162 @@ fn test_scalar_segment_scope_residual_limit_and_unindexed_fallback() { - } - } - -+#[test] -+fn test_scalar_segment_label_list_exact_candidates() { -+ use arrow_array::builder::{Int32Builder, ListBuilder}; -+ use lance::index::DatasetIndexExt; -+ use lance_index::IndexType; -+ -+ for stable in [false, true] { -+ let mut lists = ListBuilder::new(Int32Builder::new()); -+ for row in 0..16 { -+ match row { -+ 0 | 9 | 13 => lists.append(false), -+ 1 | 10 | 14 => lists.append(true), -+ 4 => { -+ lists.values().append_value(7); -+ lists.append(true); -+ } -+ _ => { -+ lists.values().append_value(42); -+ if row == 3 || row == 11 || row == 15 { -+ lists.values().append_value(7); -+ } -+ if row == 6 { -+ lists.values().append_null(); -+ } -+ lists.append(true); -+ } -+ } -+ } -+ // S0 covers fragments 0 and 1, S1 covers 2, and 3 is unindexed. -+ let (_tmp, uri, uuids) = create_scalar_segment_fixture_from_key( -+ IndexType::LabelList, -+ stable, -+ None, -+ &[&[0, 1], &[2]], -+ Arc::new(lists.finish()), -+ ); -+ let predicate = "array_contains(key, CAST(42 AS INT))"; -+ let filter = format!("{predicate} AND id >= 3"); -+ for (fragments, expected) in [(vec![0, 1], vec![3, 5, 6, 7]), (vec![0], vec![3])] { -+ let (ids, stats) = scalar_segment_ids(&uri, &uuids[0], &fragments, &filter, None, 0); -+ assert_eq!(ids, expected, "stable={stable}"); -+ assert_eq!(stats.calls, 1); -+ assert!( -+ stats -+ .metrics -+ .iter() -+ .any(|(name, _, value)| { name == "scalar_segments_searched" && *value == 1 }), -+ "stable={stable}, metrics={:?}", -+ stats.metrics -+ ); -+ assert!( -+ !stats -+ .metrics -+ .iter() -+ .any(|(name, _, value)| { name == "scalar_segment_fallbacks" && *value != 0 }) -+ ); -+ } -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[0], &[0, 1], &filter, Some(1), 1); -+ assert_eq!(ids, vec![5], "limit/offset must follow the residual filter"); -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[1], &[2], &filter, None, 0); -+ assert_eq!(ids, vec![8, 11]); -+ let (ids, stats) = scalar_segment_ids(&uri, &uuids[0], &[0, 3], &filter, None, 0); -+ assert_eq!(ids, vec![3, 12, 15]); -+ assert!(stats.metrics.iter().any(|(name, _, value)| { -+ name == "scalar_segment_fallback_partial_coverage" && *value == 1 -+ })); -+ let (ids, stats) = scalar_segment_ids( -+ &uri, -+ &uuids[0], -+ &[0], -+ &format!("{predicate} OR id = 0"), -+ None, -+ 0, -+ ); -+ assert_eq!(ids, vec![0, 2, 3]); -+ assert!(stats.metrics.iter().any(|(name, _, value)| { -+ name == "scalar_segment_fallback_no_driver" && *value == 1 -+ })); -+ -+ for (predicate, expected) in [ -+ ( -+ "array_has_all(key, [CAST(42 AS INT), CAST(7 AS INT)])", -+ vec![3], -+ ), -+ ( -+ "array_has_any(key, [CAST(42 AS INT), CAST(99 AS INT)])", -+ vec![2, 3, 5, 6, 7], -+ ), -+ ("array_contains(key, CAST(99 AS INT))", vec![]), -+ ("array_contains(key, CAST(NULL AS INT))", vec![]), -+ ("array_has_any(key, [])", vec![]), -+ ] { -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[0], &[0, 1], predicate, None, 0); -+ assert_eq!(ids, expected, "{predicate}, stable={stable}"); -+ } -+ // An untyped integer literal casts this Int32 list to Int64. Such -+ // a column expression must retain the scan fallback. -+ let (ids, stats) = -+ scalar_segment_ids(&uri, &uuids[0], &[0, 1], "array_contains(key, 42)", None, 0); -+ assert_eq!(ids, vec![2, 3, 5, 6, 7]); -+ assert!(stats.metrics.iter().any(|(name, _, value)| { -+ name == "scalar_segment_fallback_no_driver" && *value == 1 -+ })); -+ -+ lance_c::runtime::block_on(async { -+ let mut ds = Dataset::open(&uri).await.unwrap(); -+ ds.delete("id = 3").await.unwrap(); -+ assert_eq!(ds.load_indices().await.unwrap().len(), 2); -+ }); -+ let (ids, _) = scalar_segment_ids(&uri, &uuids[0], &[0, 1], &filter, None, 0); -+ assert_eq!(ids, vec![5, 6, 7]); -+ } -+} -+ -+#[test] -+fn test_scalar_segment_text_indices_still_fall_back() { -+ for kind in [lance_index::IndexType::Fm, lance_index::IndexType::NGram] { -+ for stable in [false, true] { -+ let key = Arc::new(StringArray::from(vec![ -+ Some("needle"), -+ None, -+ Some(""), -+ Some("other"), -+ Some("needle"), -+ Some("other"), -+ Some(""), -+ None, -+ ])); -+ let (_tmp, uri, uuids) = -+ create_scalar_segment_fixture_from_key(kind, stable, None, &[&[0, 1]], key); -+ for (predicate, expected) in [ -+ ("contains(key, 'needle')", vec![0]), -+ ("contains(key, '')", vec![0, 2, 3]), -+ ] { -+ let (ids, stats) = scalar_segment_ids(&uri, &uuids[0], &[0], predicate, None, 0); -+ assert_eq!(ids, expected, "{kind:?}, stable={stable}, {predicate}"); -+ assert!( -+ stats.metrics.iter().any(|(name, _, value)| { -+ name == "scalar_segment_fallbacks" && *value == 1 -+ }) -+ ); -+ if predicate == "contains(key, 'needle')" { -+ assert!(stats.metrics.iter().any(|(name, _, value)| { -+ name == "scalar_segment_fallback_index_type" && *value == 1 -+ })); -+ } -+ assert!( -+ !stats.metrics.iter().any(|(name, _, value)| { -+ name == "scalar_segments_searched" && *value != 0 -+ }) -+ ); -+ } -+ } -+ } -+} -+ - #[test] - fn test_scalar_segment_legacy_storage_falls_back() { - use lance_file::version::{ConcreteFileVersion, LanceFileVersion}; - -From 30dda06a1cc6bad37d04ed207115409ccef91715 Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Wed, 9 Sep 2026 13:27:54 +0800 -Subject: [PATCH 5/5] update - ---- - docs/scalar-segment-scans.md | 55 +++++++++-- - include/lance/lance.h | 16 ++- - include/lance/lance.hpp | 5 + - src/scalar_segment.rs | 12 ++- - src/scanner.rs | 13 ++- - tests/c_api_test.rs | 187 +++++++++++++++++++++++++++++++++++ - 6 files changed, 272 insertions(+), 16 deletions(-) - -diff --git a/docs/scalar-segment-scans.md b/docs/scalar-segment-scans.md -index 72f6188..efd0498 100644 ---- a/docs/scalar-segment-scans.md -+++ b/docs/scalar-segment-scans.md -@@ -27,11 +27,19 @@ lance_scanner_set_limit(scanner, 20000); - /* The scanner-owning thread calls lance_scanner_next as usual. */ - ``` - -+The segment setter copies the UUID; passing NULL clears it. Final option -+compatibility and snapshot metadata are checked when preparing the stream, not -+by the setter. Configure all options before the first `next`, Arrow stream -+export, or asynchronous scan. A failed preparation also freezes the options; -+create a new scanner to retry with different settings. -+ - SQL, Substrait and additional SQL filters keep their existing precedence and AND - composition. The caller does not supply a separate driver predicate: Lance-C - uses the typed filter planner and selects a necessary indexed leaf belonging to - the requested logical index. It only descends through AND, never through OR or --NOT. It then searches the selected UUID and applies the complete filter while -+NOT, and chooses the first matching leaf in the planner's expression tree; -+this is not a selectivity-based choice or a guarantee of SQL text order. -+It then searches the selected UUID and applies the complete filter while - reading candidates with automatic scalar-index planning disabled. - - Each task's fragment IDs define its result domain, including on fallback. A -@@ -40,9 +48,16 @@ scan. Unindexed fragments need their own tasks, or an explicit domain including - them (which causes that task to use fallback). Merely listing indexed segments - does not include appended, unindexed data automatically. - --An unknown UUID, absent fragment or invalid option combination is an error. A --known segment with incomplete/unknown coverage, no suitable driver, unsupported --index type, nested key, overlays, fragment reuse, non-exact results or unsupported -+`use_scalar_index=false` disables segment search regardless of setter order. -+The scanner validates the selected snapshot UUID and fragment domain, then scans -+that domain with the full filter and LIMIT/OFFSET without opening the index or -+generating candidates. It reports `scalar_segment_fallback_disabled`. -+ -+An unknown UUID, absent fragment, invalid option combination or segment metadata -+without one valid schema key field is an error. A known segment with -+incomplete/unknown coverage, no suitable driver, unsupported -+index type, nested key, overlays, unknown physical row counts, fragment reuse, -+non-exact results or unsupported - row-ID domain falls back to a non-indexed scan of the entire explicit domain. - Legacy (v1) storage also takes this fallback because ordinary scans cannot consume - external row masks; it reports `scalar_segment_fallback_legacy_storage` without -@@ -50,8 +65,15 @@ searching the index. - I/O and corruption errors are propagated, not converted to empty results or - successful fallback. - --The first implementation supports live-row ordinary scans and cannot be combined --with vector/FTS queries. Physical row-address -+The first implementation supports live-row ordinary scans and rejects vector/FTS -+queries and `include_deleted_rows=true` at stream creation, even when -+`use_scalar_index=false`. An index built after a delete does not contain the -+tombstoned rows, so even exact segment candidates -+cannot satisfy a scan that includes deleted rows. To read those rows, clear the -+segment setting and use an ordinary scan with `with_row_id=true`, -+`include_deleted_rows=true`, and `use_scalar_index=false`. -+Fragments removed from the current snapshot are not scanned by this option. -+Physical row-address - results on stable-row-ID datasets currently fall back; results already expressed - in the correct row-ID domain use the candidate path. Deletes and all remaining - predicates are handled by the ordinary reader. No candidate-count limit is -@@ -84,6 +106,23 @@ Successful exhaustion merges segment-search metrics into the existing statistics - callback exactly once. New metrics include `scalar_segments_requested`, - `scalar_segments_searched`, `scalar_segment_candidate_rows`, - `scalar_segment_prepare_time`, `scalar_segment_search_time`, and --`scalar_segment_fallback_*` reasons. `prepare_time` includes search time. Early --release, cancellation and errors retain the existing callback contract: final -+`scalar_segment_fallbacks` plus `scalar_segment_fallback_*` reasons. -+`scalar_segment_prepare_time` includes search time. Metrics describe each -+successfully exhausted stream, including separately exported streams: -+ -+- `scalar_segments_requested` is 1 even on fallback. -+- `scalar_segments_searched` is 1 after a completed index search, including one -+ whose inexact result causes fallback. A fallback can therefore include index work. -+- `scalar_segment_candidate_rows` counts TRUE rows in the exact segment result before fragment -+ restriction, residual filtering, deletion handling and LIMIT/OFFSET. It is not -+ the output or physical-read row count; a result without a known cardinality is -+ reported as 0. The reader's mask may also include NULL candidates that the full -+ filter subsequently discards. -+- A fallback records one reason, the first eligibility check that fails. Disabled -+ scalar indices and legacy storage bypass index opening and search after snapshot -+ validation. Other fallback reasons may be found after opening or searching an index. -+- Search and candidate metrics may be absent when their stage did not execute; -+ consumers should treat absent counts as 0. -+ -+Early release, cancellation and errors retain the existing callback contract: final - statistics are not guaranteed. Metrics do not establish global task concurrency. -diff --git a/include/lance/lance.h b/include/lance/lance.h -index b5c6902..ed5bb6c 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -1015,7 +1015,9 @@ int32_t lance_scanner_set_scan_in_order(LanceScanner* scanner, bool scan_in_orde - * Configure whether scalar indices may be used to optimize filters. - * - * Scalar indices are enabled by default. Disable this to force filter -- * evaluation without scalar indices. This setting is independent of -+ * evaluation without scalar indices, including an explicitly selected scalar -+ * segment (which falls back to a scan of its explicit fragment_ids). -+ * This setting is independent of - * `lance_scanner_set_use_index`, which controls vector ANN index usage. - * Must be set before scanning starts. - */ -@@ -1051,7 +1053,11 @@ int32_t lance_scanner_with_row_address(LanceScanner* scanner, bool enable); - - /** - * Configure whether deleted rows still present in storage are returned. -- * Deleted rows have a NULL `_rowid`; callers should also enable row IDs. -+ * Requires with_row_id=true; deleted rows have a NULL `_rowid`. -+ * For filtered scans, also set use_scalar_index=false: indices built after a -+ * deletion may omit tombstoned rows. Incompatible with scalar_index_segment, -+ * even when scalar indices are disabled. -+ * Fragments removed from the current snapshot are not scanned. - * Must be set before scanning starts. - */ - int32_t lance_scanner_set_include_deleted_rows( -@@ -1867,16 +1873,20 @@ int32_t lance_scanner_set_index_segments( - * excluded by fragment_ids; incomplete coverage falls back to a full filtered - * scan of those fragment_ids. Callers distributing work must assign disjoint - * fragment domains and separately include any unindexed data they wish to read. -+ * The segment metadata must identify one key field present in the schema. - * - * BTree/Bitmap/LabelList searches use a necessary AND-conjunct of the - * full scanner filter on the selected logical index and require an Exact result. -+ * use_scalar_index=false skips segment search and uses the scoped fallback; -+ * snapshot UUID and fragment validation still applies. - * AtMost/AtLeast results fall back to a full filtered scan of fragment_ids. - * All predicates are reapplied during candidate reads; other scalar indices - * are disabled. Legacy storage, OR/NOT-only filters, - * overlays, fragment reuse, unsupported index types / result domains - * and missing coverage use the same domain without an index. No filter also - * falls back. LIMIT/OFFSET apply after the complete scanner filter, never to the -- * unfiltered candidate set. Vector/FTS queries are rejected. -+ * unfiltered candidate set. Vector/FTS queries and include_deleted_rows=true -+ * are rejected even when use_scalar_index=false; segment mode is live-row-only. - * - * UUID bytes are copied. Metadata and final option compatibility are validated - * when creating the stream. Index corruption or I/O failures remain errors. -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index ebb8141..d96cf9f 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1270,6 +1270,7 @@ class Scanner { - } - - /// Configure whether scalar indices may be used to optimize filters. -+ /// False also disables explicit scalar segment search, retaining its fragment domain. - Scanner& use_scalar_index(bool enable = true) { - if (lance_scanner_set_use_scalar_index(handle_.get(), enable) != 0) - check_error(); -@@ -1305,6 +1306,8 @@ class Scanner { - } - - /// Configure whether deleted rows still present in storage are returned. -+ /// Requires with_row_id(true); use_scalar_index(false) is needed for filtered scans. -+ /// Incompatible with scalar_index_segment. See lance.h. - Scanner& include_deleted_rows(bool include_deleted_rows = true) { - if (lance_scanner_set_include_deleted_rows(handle_.get(), include_deleted_rows) != 0) - check_error(); -@@ -1313,6 +1316,8 @@ class Scanner { - - /// Generate exact candidates from one BTree/Bitmap/LabelList segment. - /// fragment_ids is required and defines the complete read/fallback domain. See lance.h. -+ /// Requires live rows only: include_deleted_rows(true) is rejected at stream creation. -+ /// use_scalar_index(false) selects the scoped fallback without searching the segment. - Scanner& scalar_index_segment(const std::array& segment_uuid) { - if (lance_scanner_set_scalar_index_segment(handle_.get(), segment_uuid.data()) != 0) - check_error(); -diff --git a/src/scalar_segment.rs b/src/scalar_segment.rs -index 1faf500..747dac6 100644 ---- a/src/scalar_segment.rs -+++ b/src/scalar_segment.rs -@@ -27,6 +27,7 @@ pub(crate) struct PreparedScalarSegment { - pub dataset: Arc, - pub segment_uuid: Uuid, - pub fragment_ids: Vec, -+ pub use_scalar_index: bool, - pub callback: Option, - } - -@@ -138,6 +139,11 @@ impl PreparedScalarSegment { - self.dataset.schema().field_by_id(field_id).ok_or_else(|| { - invalid("scalar segment key field is absent from the dataset schema") - })?; -+ // Explicitly disabling scalar indices also disables this accelerator. -+ // Keep snapshot validation above, but do not plan, open or search an index. -+ if !self.use_scalar_index { -+ return Ok(Some("disabled")); -+ } - // Match Lance's plain-scan external-mask restriction. Keep the scoped, - // full-filtered reader intact and avoid index work on legacy storage. - if self -@@ -149,8 +155,8 @@ impl PreparedScalarSegment { - { - return Ok(Some("legacy_storage")); - } -- // Keep V1 to flat scalar fields. A dotted name is not sufficient to prove -- // the field path of an evolved or nested schema. -+ // Keep this implementation to flat scalar fields. A dotted name cannot -+ // prove the field path of an evolved or nested schema. - if !self - .dataset - .schema() -@@ -234,6 +240,8 @@ impl PreparedScalarSegment { - ); - // Do not truncate candidates at LIMIT. The reader evaluates the complete - // filter before applying its existing limit/offset operators. -+ // The raw selected bitmap can overlap NULL rows; the full filter removes -+ // those as well. The metric above counts semantic TRUE rows, not mask size. - reader.with_row_addr_prefilter(RowAddrMask::from_allowed(rows.selected_rows().clone())); - Ok(None) - } -diff --git a/src/scanner.rs b/src/scanner.rs -index ac3cff6..bda554d 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -479,9 +479,10 @@ impl LanceScanner { - || self.fts_context.is_some() - || self.index_segments.is_some() - || self.fts_index_segments.is_some() -+ || self.include_deleted_rows - { - return Err(lance_core::Error::invalid_input_source( -- "scalar_index_segment requires an ordinary scan of live rows".into(), -+ "scalar_index_segment requires an ordinary scan of live rows; vector/FTS queries and include_deleted_rows=true are unsupported".into(), - )); - } - let fragment_ids = self.fragment_ids.as_ref().filter(|ids| !ids.is_empty()) -@@ -492,6 +493,7 @@ impl LanceScanner { - dataset: Arc::clone(&self.dataset), - segment_uuid, - fragment_ids: fragment_ids.clone(), -+ use_scalar_index: self.use_scalar_index.unwrap_or(true), - callback: self.scan_statistics_callback.clone(), - }) - } else { -@@ -896,6 +898,8 @@ macro_rules! scanner_ffi_try { - - /// Select one physical scalar index segment. NULL clears the selection. - /// Requires explicit fragment_ids and an ordinary live-row scan. See the C header. -+/// include_deleted_rows=true is rejected when preparing the scan, even if -+/// use_scalar_index=false selects the scoped non-indexed fallback. - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_set_scalar_index_segment( - scanner: *mut LanceScanner, -@@ -1239,7 +1243,8 @@ unsafe fn scanner_set_scan_in_order_inner( - /// Configure whether scalar indices may be used to optimize filters. - /// - /// Scalar indices are enabled by default in Lance. Must be set before the scan --/// starts. -+/// starts. False also disables explicit scalar segment search while preserving -+/// the configured fragment domain and snapshot validation. - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_set_use_scalar_index( - scanner: *mut LanceScanner, -@@ -1381,7 +1386,9 @@ unsafe fn scanner_with_row_address_inner(scanner: *mut LanceScanner, enable: boo - - /// Configure whether deleted rows still present in storage are returned. - /// --/// Deleted rows have a NULL `_rowid`, so callers should also enable row IDs. -+/// Requires with_row_id=true; deleted rows have a NULL `_rowid`. -+/// Filtered scans also need use_scalar_index=false because indices may omit -+/// tombstoned rows. Incompatible with scalar_index_segment. - /// Must be set before the scan starts. - #[unsafe(no_mangle)] - pub unsafe extern "C" fn lance_scanner_set_include_deleted_rows( -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 3322d42..513f8d0 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -12945,6 +12945,193 @@ fn test_scalar_segment_stable_row_ids_and_deletes() { - assert_eq!(ids, vec![3]); - } - -+#[test] -+fn test_scalar_segment_honors_use_scalar_index_false() { -+ let (_tmp, uri, uuids) = create_scalar_segment_fixture(lance_index::IndexType::BTree, false); -+ let (ids, stats) = scalar_segment_ids(&uri, &uuids[0], &[0], "key >= 0", None, 0); -+ assert_eq!(ids, vec![1, 2, 3]); -+ assert!( -+ stats -+ .metrics -+ .iter() -+ .any(|(name, _, value)| name == "scalar_segments_searched" && *value == 1) -+ ); -+ -+ let uri = c_str(&uri); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ assert!(!ds.is_null()); -+ for disable_first in [false, true] { -+ for (fragments, filter, limit, offset, expected) in [ -+ (vec![0u64], "key >= 0", None, 0, vec![1, 2, 3]), -+ (vec![0, 2], "key >= 0 AND id >= 2", Some(2), 1, vec![3, 9]), -+ ] { -+ let filter = c_str(filter); -+ let scanner = lance_scanner_new(ds, ptr::null(), filter.as_ptr()); -+ assert!(!scanner.is_null()); -+ assert_eq!( -+ lance_scanner_set_fragment_ids(scanner, fragments.as_ptr(), fragments.len()), -+ 0 -+ ); -+ if disable_first { -+ assert_eq!(lance_scanner_set_use_scalar_index(scanner, false), 0); -+ } -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(scanner, uuids[0].as_ptr()), -+ 0 -+ ); -+ if !disable_first { -+ assert_eq!(lance_scanner_set_use_scalar_index(scanner, false), 0); -+ } -+ if let Some(limit) = limit { -+ assert_eq!(lance_scanner_set_limit(scanner, limit), 0); -+ } -+ assert_eq!(lance_scanner_set_offset(scanner, offset), 0); -+ let mut captured = CapturedScanStatistics::default(); -+ assert_eq!( -+ lance_scanner_set_statistics_callback( -+ scanner, -+ Some(capture_scan_statistics), -+ (&mut captured as *mut CapturedScanStatistics).cast(), -+ ), -+ 0 -+ ); -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!(lance_scanner_to_arrow_stream(scanner, &mut stream), 0); -+ let mut ids = Vec::new(); -+ for batch in ArrowArrayStreamReader::from_raw(&mut stream).unwrap() { -+ let batch = batch.unwrap(); -+ ids.extend_from_slice( -+ batch -+ .column_by_name("id") -+ .unwrap() -+ .as_any() -+ .downcast_ref::() -+ .unwrap() -+ .values(), -+ ); -+ } -+ assert_eq!(ids, expected); -+ assert_eq!(captured.calls, 1); -+ for metric in ["scalar_segments_searched", "scalar_segment_candidate_rows"] { -+ assert_eq!( -+ captured -+ .metrics -+ .iter() -+ .filter(|(name, _, _)| name == metric) -+ .map(|(_, _, value)| *value) -+ .sum::(), -+ 0, -+ "{metric}" -+ ); -+ } -+ assert_eq!(captured.indices_loaded, 0); -+ assert_eq!(captured.index_comparisons, 0); -+ assert!(captured.metrics.iter().any(|(name, _, value)| name -+ == "scalar_segment_fallback_disabled" -+ && *value == 1)); -+ lance_scanner_close(scanner); -+ } -+ } -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn test_scalar_segment_rejects_include_deleted_rows_after_index_rebuild() { -+ use lance::index::DatasetIndexExt; -+ use lance_index::scalar::{BuiltinIndexType, ScalarIndexParams}; -+ -+ let (_tmp, uri, _) = create_scalar_segment_fixture(lance_index::IndexType::BTree, false); -+ let uuid = lance_c::runtime::block_on(async { -+ let mut ds = Dataset::open(&uri).await.unwrap(); -+ ds.delete("id = 2").await.unwrap(); -+ ds.drop_index("key_idx").await.unwrap(); -+ // A segment built after the delete cannot return the tombstoned row, -+ // even though its search result is Exact for the indexed live rows. -+ let params = ScalarIndexParams::for_builtin(BuiltinIndexType::BTree); -+ let segment = ds -+ .create_index_builder(&["key"], lance_index::IndexType::BTree, ¶ms) -+ .name("key_idx".into()) -+ .fragments(vec![0]) -+ .execute_uncommitted() -+ .await -+ .unwrap(); -+ let uuid = *segment.uuid.as_bytes(); -+ ds.commit_existing_index_segments("key_idx", "key", vec![segment]) -+ .await -+ .unwrap(); -+ uuid -+ }); -+ -+ let (ids, _) = scalar_segment_ids(&uri, &uuid, &[0], "key >= 0 AND id >= 2", None, 0); -+ assert_eq!(ids, vec![3], "live-row segment scans remain supported"); -+ -+ let uri = c_str(&uri); -+ let filter = c_str("key >= 0 AND id >= 2"); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ assert!(!ds.is_null()); -+ // Check both setter orders: compatibility is validated at stream creation. -+ for segment_first in [None, Some(false), Some(true)] { -+ let scanner = lance_scanner_new(ds, ptr::null(), filter.as_ptr()); -+ assert!(!scanner.is_null()); -+ assert_eq!( -+ lance_scanner_set_fragment_ids(scanner, [0u64].as_ptr(), 1), -+ 0 -+ ); -+ assert_eq!(lance_scanner_with_row_id(scanner, true), 0); -+ if segment_first.is_none() { -+ assert_eq!(lance_scanner_set_use_scalar_index(scanner, false), 0); -+ } -+ if segment_first == Some(true) { -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(scanner, uuid.as_ptr()), -+ 0 -+ ); -+ } -+ assert_eq!(lance_scanner_set_include_deleted_rows(scanner, true), 0); -+ if segment_first == Some(false) { -+ assert_eq!( -+ lance_scanner_set_scalar_index_segment(scanner, uuid.as_ptr()), -+ 0 -+ ); -+ } -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ let rc = lance_scanner_to_arrow_stream(scanner, &mut stream); -+ if segment_first.is_some() { -+ assert_eq!(rc, -1, "segment scans must not silently omit deleted rows"); -+ assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); -+ assert!(take_last_error_message().contains("include_deleted_rows=true")); -+ } else { -+ assert_eq!(rc, 0); -+ let reader = ArrowArrayStreamReader::from_raw(&mut stream).unwrap(); -+ let mut ids = Vec::new(); -+ for batch in reader { -+ let batch = batch.unwrap(); -+ ids.extend_from_slice( -+ batch -+ .column_by_name("id") -+ .unwrap() -+ .as_any() -+ .downcast_ref::() -+ .unwrap() -+ .values(), -+ ); -+ } -+ ids.sort_unstable(); -+ assert_eq!( -+ ids, -+ vec![2, 3], -+ "ordinary scans can still read tombstoned rows" -+ ); -+ } -+ lance_scanner_close(scanner); -+ } -+ lance_dataset_close(ds); -+ } -+} -+ - #[test] - fn test_scalar_segment_requires_explicit_domain_and_checks_uuid() { - let (_tmp, uri, uuids) = create_scalar_segment_fixture(lance_index::IndexType::BTree, false); diff --git a/thirdparty/patches/lance-c-0.1.9-pr-80.patch b/thirdparty/patches/lance-c-0.1.9-pr-80.patch deleted file mode 100644 index edc42cad0a1271..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-pr-80.patch +++ /dev/null @@ -1,225 +0,0 @@ -From 7fcd9c4ff7c03c10bdc9d8a600b3f7b0cf28a9ad Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Thu, 10 Sep 2026 15:50:53 +0800 -Subject: [PATCH] oss provider error - -Doris integration: rebase only Cargo.toml/Cargo.lock hunk context over PR #73. -All added and removed lines are identical to upstream PR #80 at 7fcd9c4. - ---- - Cargo.lock | 1 + - Cargo.toml | 2 + - src/runtime.rs | 5 ++ - tests/compile_and_run_test.rs | 16 ++++++ - tests/cpp/test_oss_transport.c | 40 +++++++++++++ - tests/static_oss_transport_test.py | 91 ++++++++++++++++++++++++++++++ - 6 files changed, 155 insertions(+) - create mode 100644 tests/cpp/test_oss_transport.c - create mode 100644 tests/static_oss_transport_test.py - -diff --git a/Cargo.lock b/Cargo.lock ---- a/Cargo.lock -+++ b/Cargo.lock -@@ -3970,6 +3970,7 @@ - "libc", - "log", - "object_store", -+ "opendal", - "pin-project", - "prost", - "snafu", -diff --git a/Cargo.toml b/Cargo.toml ---- a/Cargo.toml -+++ b/Cargo.toml -@@ -43,6 +43,8 @@ - log = "0.4" - libc = "0.2" - object_store = "0.13.2" -+# Explicitly install the HTTP transport when embedded in a static C/C++ executable. -+opendal = { version = "=0.58.2", default-features = false, features = ["http-transport-reqwest"] } - pin-project = "1.0" - prost = "0.14" - snafu = "0.9" -diff --git a/src/runtime.rs b/src/runtime.rs -index 0153d3f..3bd8964 100644 ---- a/src/runtime.rs -+++ b/src/runtime.rs -@@ -8,6 +8,11 @@ use std::sync::LazyLock; - /// Global multi-threaded Tokio runtime, shared across all FFI calls. - /// Initialized lazily on first access. - pub static RT: LazyLock = LazyLock::new(|| { -+ // A native linker can omit OpenDAL's automatic constructor from liblance_c.a. -+ // Keep initialization reachable from the FFI entry points, before any HTTP I/O. -+ // Installation is idempotent and preserves an already installed transport. -+ opendal::install_default(); -+ - tokio::runtime::Builder::new_multi_thread() - .enable_all() - .build() -diff --git a/tests/compile_and_run_test.rs b/tests/compile_and_run_test.rs -index b419ac9..8566d10 100644 ---- a/tests/compile_and_run_test.rs -+++ b/tests/compile_and_run_test.rs -@@ -249,3 +249,19 @@ fn test_cpp_compilation_and_execution() { - - run_test_binary(&binary, &dataset_uri, &write_uri); - } -+ -+/// A fresh C executable must initialize OpenDAL even when archive constructors are omitted. -+#[cfg(target_os = "linux")] -+#[test] -+#[ignore = "requires a C compiler, Python 3, and building the static library"] -+fn test_static_oss_transport() { -+ let (shared_library, _) = build_lance_c(); -+ let static_library = shared_library.with_file_name("liblance_c.a"); -+ assert!(static_library.exists(), "static library was not built"); -+ let status = Command::new("python3") -+ .arg(Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/static_oss_transport_test.py")) -+ .arg(static_library) -+ .status() -+ .expect("failed to run the static OSS transport test"); -+ assert!(status.success(), "static OSS HTTP transport test failed"); -+} -diff --git a/tests/cpp/test_oss_transport.c b/tests/cpp/test_oss_transport.c -new file mode 100644 -index 0000000..c96ee7e ---- /dev/null -+++ b/tests/cpp/test_oss_transport.c -@@ -0,0 +1,40 @@ -+/* SPDX-License-Identifier: Apache-2.0 */ -+/* SPDX-FileCopyrightText: Copyright The Lance Authors */ -+ -+#include "lance/lance.h" -+#include -+#include -+ -+/* The Python harness serves a missing manifest on a local HTTP endpoint. */ -+int main(int argc, char **argv) { -+ if (argc != 3) return 2; -+ const char *options[] = { -+ "oss_endpoint", argv[1], -+ "oss_region", "cn-test", -+ "oss_access_key_id", "test-key", -+ "oss_secret_access_key", "test-secret", -+ "addressing_style", "path", -+ NULL -+ }; -+ LanceSession *session = NULL; -+ LanceDataset *dataset = NULL; -+ const char *uri = "oss://test-bucket/missing.lance"; -+ if (strcmp(argv[2], "shared") == 0) { -+ session = lance_session_new(0, 0); -+ if (session == NULL) return 3; -+ dataset = lance_dataset_open_with_session(uri, options, 1, session); -+ } else { -+ dataset = lance_dataset_open(uri, options, 1); -+ } -+ /* The object does not exist, but the request must reach the HTTP server. */ -+ const char *error = lance_last_error_message(); -+ int failed = dataset != NULL || error == NULL; -+ if (error != NULL) { -+ fprintf(stderr, "%s\n", error); -+ failed |= strstr(error, "default HTTP transport is not installed") != NULL; -+ lance_free_string(error); -+ } -+ lance_dataset_close(dataset); -+ lance_session_close(session); -+ return failed ? 1 : 0; -+} -diff --git a/tests/static_oss_transport_test.py b/tests/static_oss_transport_test.py -new file mode 100644 -index 0000000..7d0e87c ---- /dev/null -+++ b/tests/static_oss_transport_test.py -@@ -0,0 +1,91 @@ -+# SPDX-License-Identifier: Apache-2.0 -+# SPDX-FileCopyrightText: Copyright The Lance Authors -+ -+"""Exercise native OSS HTTP initialization from fresh, statically linked C processes. -+ -+Build lance-c first, then run on Linux: -+ python3 tests/static_oss_transport_test.py target/release/liblance_c.a -+ -+No OSS account is needed. A local HTTP server returns 404 for a missing manifest. -+The assertion is that an HTTP request reaches it, not merely that opening fails. -+Unlike a Rust test binary, the C executable must pull initialization from the archive. -+""" -+ -+import argparse -+from http.server import BaseHTTPRequestHandler, HTTPServer -+import os -+from pathlib import Path -+import shlex -+import subprocess -+import sys -+import tempfile -+import threading -+ -+ -+def main(): -+ parser = argparse.ArgumentParser(description=__doc__) -+ parser.add_argument("library", type=Path) -+ args = parser.parse_args() -+ if not sys.platform.startswith("linux"): -+ parser.error("this static-link regression test currently supports Linux") -+ library = args.library.resolve(strict=True) -+ root = Path(__file__).resolve().parents[1] -+ requests = [] -+ -+ class Handler(BaseHTTPRequestHandler): -+ def missing(self): -+ requests.append((self.command, self.path)) -+ self.send_response(404) -+ self.send_header("Content-Length", "0") -+ self.send_header("Connection", "close") -+ self.end_headers() -+ -+ do_HEAD = missing -+ do_GET = missing -+ -+ def log_message(self, *_args): -+ pass -+ -+ with tempfile.TemporaryDirectory(prefix="lance-static-oss-") as directory: -+ executable = Path(directory) / "test_oss_transport" -+ # Pass the archive explicitly; -llance_c could silently select the shared library. -+ # Do not use --whole-archive: ordinary native linking must retain initialization. -+ subprocess.run( -+ shlex.split(os.environ.get("CC", "cc")) -+ + ["-std=c11", "-Wall", "-Wextra", "-Werror", "-Wl,--gc-sections", -+ "-I", str(root / "include"), str(root / "tests/cpp/test_oss_transport.c"), -+ str(library), "-lgcc_s", "-lutil", "-lrt", "-lpthread", "-lm", "-ldl", -+ "-o", str(executable)], -+ check=True, -+ ) -+ environment = { -+ key: value for key, value in os.environ.items() -+ if not key.startswith(("AWS_", "OSS_", "ALIBABA_CLOUD_")) -+ and key.lower() not in ("http_proxy", "https_proxy", "all_proxy", "no_proxy") -+ } -+ environment["NO_PROXY"] = "127.0.0.1,localhost" -+ # Each mode starts a new process so an earlier call cannot hide missing initialization. -+ for mode in ("ordinary", "shared"): -+ requests.clear() -+ with HTTPServer(("127.0.0.1", 0), Handler) as server: -+ thread = threading.Thread( -+ target=server.serve_forever, kwargs={"poll_interval": 0.05}, daemon=True -+ ) -+ thread.start() -+ try: -+ result = subprocess.run( -+ [str(executable), f"http://127.0.0.1:{server.server_port}", mode], -+ env=environment, capture_output=True, text=True, timeout=30, -+ ) -+ finally: -+ server.shutdown() -+ thread.join() -+ assert result.returncode == 0, f"{mode}: {result.stderr}" -+ assert any("/_versions/" in path for _, path in requests), ( -+ f"{mode}: no manifest HTTP request reached the server: {result.stderr}" -+ ) -+ print(f"PASS: {mode} OSS open reached the local HTTP server") -+ -+ -+if __name__ == "__main__": -+ main() diff --git a/thirdparty/patches/lance-c-0.1.9-pr-83.patch b/thirdparty/patches/lance-c-0.1.9-pr-83.patch deleted file mode 100644 index 86c0c23e227d31..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-pr-83.patch +++ /dev/null @@ -1,1914 +0,0 @@ -From 0a30ee6c5a9d1455ceb36f4745acb79e53a00461 Mon Sep 17 00:00:00 2001 -Subject: [PATCH] feat: support multi-vector queries through C and C++ APIs (#83) - -Upstream: https://github.com/lance-format/lance-c/pull/83 -Commit: 0a30ee6c5a9d1455ceb36f4745acb79e53a00461 - -Adapted to the v0.1.9 community patch chain used by Doris. Resolve context -conflicts with PR #73 and retain PR #79 scalar_segment fields, execution -branch, and tests alongside the upstream multi-vector additions. The -multi-vector implementation and its integration tests are unchanged. - -diff --git a/README.md b/README.md -index d9a5f2b..29c89c4 100644 ---- a/README.md -+++ b/README.md -@@ -70,6 +70,40 @@ Based on the [liblance RFC](https://github.com/lance-format/lance/discussions/60 - | [x] | Filter pushdown | `lance_scanner_set_substrait_filter()` accepts a serialized Substrait `ExtendedExpression`; `lance_scanner_additional_sql_filter()` adds SQL predicates with AND before scanning starts | - | [x] | Data-file cache | Optional Foyer memory/disk cache for immutable `data/*.lance` reads | - -+## Multi-vector search -+ -+Use `lance_scanner_nearest_multivector` or the C++ `Scanner::nearest_multivector` -+method for a `List>` column: -+ -+```cpp -+const float query[] = {1.0f, 0.0f, 0.0f, 1.0f}; -+auto scanner = dataset.scan(); -+scanner.nearest_multivector("embeddings", query, 2, 2, LANCE_DTYPE_FLOAT32, 10) -+ .metric(LANCE_METRIC_COSINE) -+ .prefilter(true); -+``` -+ -+The copied, row-major matrix is **one query** containing two subvectors. Results -+rank logical rows by the sum of each query subvector's minimum distance to a -+stored subvector. Empty or null outer rows do not rank. Inner vectors must be -+non-nullable; actual stored null or non-finite elements encountered during -+scoring fail the stream. Float types and dimensions must match the column. -+Cosine pairs with zero norm have undefined distance and are ignored. A row is -+excluded if any query subvector has no defined match; a zero-norm query subvector -+therefore produces no results. Column names use Lance field-path syntax, -+including nested paths such as `payload.embeddings` and backtick-quoted names. -+ -+L2 is the default on every fragment. Cosine multi-vector indexes are supported -+by the pinned Lance version; incompatible metrics use exact search. Indexed -+candidates are refined against stored values (`refine_factor` defaults to 1). -+ANN candidate selection remains approximate. Limit and offset apply after -+restoring distance order, including fragment-scoped searches. Strict row batching -+is applied after that final result window, preserving full batches except the last. -+ -+Queries accept at most 128 subvectors. Both `num_vectors * k` and -+`refine_factor * k` must be at most 100,000 to bound plan expansion and candidate -+allocation. The existing single-vector API and its defaults are unchanged. -+ - ## Building - - There are four supported entry points; pick whichever matches your toolchain. -diff --git a/include/lance/lance.h b/include/lance/lance.h -index a201bfb..e404612 100644 ---- a/include/lance/lance.h -+++ b/include/lance/lance.h -@@ -1847,6 +1847,21 @@ int32_t lance_scanner_nearest( - uint32_t k - ); - -+/** -+ * Set one multi-vector query on a List> column. -+ * Inner vectors must be non-nullable and contain no null elements; the outer list may be nullable. -+ * query_data contains dimension * num_vectors aligned elements in row-major order. -+ * Both sizes and k must be positive. At most 128 query subvectors are accepted; -+ * num_vectors * k and refine_factor * k must each be at most 100000. -+ * Values are copied before returning. The default metric is L2 on every fragment. -+ * Scores sum each query vector's minimum distance; refinement defaults to 1. -+ * Returns 0 on success, -1 on error. Stored invalid elements fail during execution. -+ */ -+int32_t lance_scanner_nearest_multivector( -+ LanceScanner* scanner, const char* column, const void* query_data, -+ size_t dimension, size_t num_vectors, LanceDataType element_type, uint32_t k -+); -+ - /** - * Set both the minimum and maximum vector-index partition-search bounds. - * -diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index a231a00..146c6b1 100644 ---- a/include/lance/lance.hpp -+++ b/include/lance/lance.hpp -@@ -1465,6 +1465,16 @@ public: - return *this; - } - -+ /// One multi-vector query, copied from dimension * num_vectors row-major elements. -+ Scanner& nearest_multivector(const std::string& column, const void* query_data, -+ size_t dimension, size_t num_vectors, -+ LanceDataType element_type, uint32_t k) { -+ if (lance_scanner_nearest_multivector(handle_.get(), column.c_str(), query_data, -+ dimension, num_vectors, element_type, k) != 0) -+ check_error(); -+ return *this; -+ } -+ - /// Replace both minimum and maximum partition-search bounds. - Scanner& nprobes(uint32_t nprobes) { - if (lance_scanner_set_nprobes(handle_.get(), nprobes) != 0) check_error(); -diff --git a/src/lib.rs b/src/lib.rs -index 197da5f..55f6ecf 100644 ---- a/src/lib.rs -+++ b/src/lib.rs -@@ -39,6 +39,7 @@ mod index; - mod index_model; - mod index_segment; - mod merge_insert; -+mod multivector; - mod restore; - pub mod runtime; - mod scalar_segment; -diff --git a/src/multivector.rs b/src/multivector.rs -new file mode 100644 -index 0000000..c08df28 ---- /dev/null -+++ b/src/multivector.rs -@@ -0,0 +1,514 @@ -+// SPDX-License-Identifier: Apache-2.0 -+// SPDX-FileCopyrightText: Copyright The Lance Authors -+ -+//! Correct multi-vector scoring before the pinned Lance plan's candidate limits. -+ -+use std::collections::HashMap; -+use std::sync::Arc; -+ -+use arrow_array::types::{Float16Type, Float32Type, Float64Type}; -+use arrow_array::{ -+ Array, ArrayRef, ArrowPrimitiveType, BooleanArray, FixedSizeListArray, Float32Array, ListArray, -+ RecordBatch, UInt64Array, -+}; -+use arrow_schema::{DataType, SchemaRef}; -+use datafusion::error::{DataFusionError, Result}; -+use datafusion::execution::context::TaskContext; -+use datafusion::physical_plan::{ -+ DisplayAs, DisplayFormatType, ExecutionPlan, PlanProperties, SendableRecordBatchStream, -+ stream::RecordBatchStreamAdapter, -+}; -+use futures::{StreamExt, TryStreamExt, stream}; -+use lance::io::exec::KNNVectorDistanceExec; -+use lance_linalg::distance::{Cosine, DistanceType, Dot, L2}; -+ -+// Lance creates one ANN branch per query vector, -+// each overfetching 10 * k candidates before scoring; wire bytes alone cannot bound this work. -+pub(crate) const MAX_QUERY_VECTORS: usize = 128; -+pub(crate) const MAX_QUERY_VECTOR_CANDIDATES: usize = 100_000; -+ -+fn invalid(message: impl Into) -> DataFusionError { -+ DataFusionError::Execution(message.into()) -+} -+ -+/// Rewrite inside TopK/refinement, before any score can discard a candidate. -+pub(crate) fn rewrite(plan: Arc) -> Result> { -+ let children = plan -+ .children() -+ .into_iter() -+ .map(|child| rewrite(child.clone())) -+ .collect::>>()?; -+ let plan = if children.is_empty() { -+ plan -+ } else { -+ plan.with_new_children(children)? -+ }; -+ let mode = if let Some(exact) = plan.downcast_ref::() { -+ if exact.is_batch { -+ return Err(invalid( -+ "expected one logical multi-vector query, not batch queries", -+ )); -+ } -+ Some(Scoring::Exact { -+ query: exact.query.clone(), -+ column: exact.column.clone(), -+ metric: exact.distance_type, -+ }) -+ // This pinned Lance node is not publicly re-exported, so match its stable plan name. -+ } else if plan.name() == "MultivectorScoringExec" { -+ Some(Scoring::Indexed) -+ } else { -+ None -+ }; -+ Ok(match mode { -+ Some(mode) => Arc::new(MultiVectorScoreExec { -+ original: plan, -+ mode, -+ }), -+ None => plan, -+ }) -+} -+ -+/// Apply the final distance-ordered window without invalidating output batching. -+pub(crate) fn apply_result_window( -+ plan: Arc, -+ offset: usize, -+ limit: Option, -+) -> Result> { -+ use datafusion::physical_expr::{PhysicalSortExpr, expressions}; -+ use datafusion::physical_plan::{ -+ coalesce_partitions::CoalescePartitionsExec, limit::GlobalLimitExec, sorts::sort::SortExec, -+ }; -+ if plan -+ .downcast_ref::() -+ .is_some() -+ { -+ // Offset can split a previously strict batch. Keep Lance's final rechunker -+ // outside the window, preserving its resolved batch size, including defaults. -+ let input = apply_result_window(plan.children()[0].clone(), offset, limit)?; -+ return plan.with_new_children(vec![input]); -+ } -+ let sort = PhysicalSortExpr { -+ expr: expressions::col("_distance", plan.schema().as_ref())?, -+ options: arrow::compute::SortOptions { -+ descending: false, -+ nulls_first: false, -+ }, -+ }; -+ // Fragment-scoped payload takes can reorder batches. Restore distance order -+ // before the window; the nearest plan already bounds candidate rows by k. -+ let sorted = Arc::new(SortExec::new( -+ [sort].into(), -+ Arc::new(CoalescePartitionsExec::new(plan)), -+ )); -+ Ok(Arc::new(GlobalLimitExec::new(sorted, offset, limit))) -+} -+ -+#[derive(Clone, Debug)] -+enum Scoring { -+ Exact { -+ query: ArrayRef, -+ column: String, -+ metric: DistanceType, -+ }, -+ Indexed, -+} -+ -+#[derive(Debug)] -+struct MultiVectorScoreExec { -+ original: Arc, -+ mode: Scoring, -+} -+ -+impl DisplayAs for MultiVectorScoreExec { -+ fn fmt_as(&self, _: DisplayFormatType, f: &mut std::fmt::Formatter) -> std::fmt::Result { -+ write!(f, "MultiVectorScore: {}", self.original.name()) -+ } -+} -+ -+impl ExecutionPlan for MultiVectorScoreExec { -+ fn name(&self) -> &str { -+ "MultiVectorScoreExec" -+ } -+ fn properties(&self) -> &Arc { -+ self.original.properties() -+ } -+ fn children(&self) -> Vec<&Arc> { -+ self.original.children() -+ } -+ fn required_input_distribution(&self) -> Vec { -+ self.original.required_input_distribution() -+ } -+ fn with_new_children( -+ self: Arc, -+ children: Vec>, -+ ) -> Result> { -+ Ok(Arc::new(Self { -+ original: self.original.clone().with_new_children(children)?, -+ mode: self.mode.clone(), -+ })) -+ } -+ fn execute( -+ &self, -+ partition: usize, -+ context: Arc, -+ ) -> Result { -+ let schema = self.schema(); -+ match &self.mode { -+ Scoring::Exact { -+ query, -+ column, -+ metric, -+ } => { -+ let input = self.children()[0].execute(partition, context)?; -+ let query = query.clone(); -+ let column = column.clone(); -+ let metric = *metric; -+ let output_schema = schema.clone(); -+ let output = input -+ .map(move |batch| { -+ let query = query.clone(); -+ let column = column.clone(); -+ let schema = output_schema.clone(); -+ async move { -+ let batch = batch?; -+ tokio::task::spawn_blocking(move || { -+ exact_batch(batch, query, &column, metric, schema) -+ }) -+ .await -+ .map_err(|e| DataFusionError::External(Box::new(e)))? -+ } -+ }) -+ .buffered(lance_core::utils::tokio::get_num_compute_intensive_cpus()); -+ Ok(Box::pin(RecordBatchStreamAdapter::new(schema, output))) -+ } -+ Scoring::Indexed => { -+ let inputs = self -+ .children() -+ .into_iter() -+ .map(|child| child.execute(partition, context.clone())) -+ .collect::>>()?; -+ let output_schema = schema.clone(); -+ let output = -+ stream::once(async move { indexed_batch(inputs, output_schema).await }); -+ Ok(Box::pin(RecordBatchStreamAdapter::new(schema, output))) -+ } -+ } -+ } -+} -+ -+fn row_distance( -+ query: &dyn Array, -+ vectors: &FixedSizeListArray, -+ metric: DistanceType, -+) -> Result> -+where -+ T::Native: L2 + Cosine + Dot + Into, -+{ -+ let q = query -+ .as_any() -+ .downcast_ref::>() -+ .ok_or_else(|| invalid("multi-vector query element type mismatch"))?; -+ let values = vectors -+ .values() -+ .as_any() -+ .downcast_ref::>() -+ .ok_or_else(|| invalid("multi-vector stored element type mismatch"))?; -+ if vectors.null_count() != 0 -+ || values.null_count() != 0 -+ || values -+ .values() -+ .iter() -+ .any(|v| !Into::::into(*v).is_finite()) -+ { -+ return Err(invalid( -+ "multi-vector stored subvectors must contain only finite, non-null elements", -+ )); -+ } -+ let dimension = vectors.value_length() as usize; -+ let distance = metric.func(); -+ // Subtracting each small distance from 1 rounds it away before TopK. Sum minima -+ // directly, using f64 only for the accumulator; the base kernels and output remain f32. -+ let mut score = 0.0f64; -+ for query_vector in q.values().chunks_exact(dimension) { -+ let best = values -+ .values() -+ .chunks_exact(dimension) -+ .map(|vector| distance(query_vector, vector)) -+ // Finite zero-norm vectors have undefined cosine distance. Ignore those -+ // pairs; a query with no defined match masks this row, not the whole scan. -+ .filter(|distance| !distance.is_nan()) -+ .min_by(f32::total_cmp); -+ let Some(best) = best else { -+ return Ok(None); -+ }; -+ score += best as f64; -+ } -+ let score = score as f32; -+ if !score.is_finite() { -+ return Err(invalid("multi-vector distance is not finite")); -+ } -+ Ok(Some(score)) -+} -+ -+fn vector_column(batch: &RecordBatch, column: &str) -> Result { -+ if let Some(array) = batch.column_by_name(column) { -+ return Ok(array.clone()); -+ } -+ // The planner resolves field paths, including quoted dotted names. Its private -+ // KNN resolver is not exported, so use the same parser and struct traversal here. -+ let parts = lance_core::datatypes::parse_field_path(column) -+ .map_err(|e| invalid(format!("invalid vector column path '{column}': {e}")))?; -+ let root = parts -+ .first() -+ .ok_or_else(|| invalid("empty vector column path"))?; -+ let mut array = batch -+ .column_by_name(root) -+ .cloned() -+ .ok_or_else(|| invalid(format!("missing vector column '{column}'")))?; -+ for part in &parts[1..] { -+ array = array -+ .as_any() -+ .downcast_ref::() -+ .and_then(|parent| parent.column_by_name(part)) -+ .cloned() -+ .ok_or_else(|| { -+ invalid(format!( -+ "missing struct field '{part}' in vector column '{column}'" -+ )) -+ })?; -+ } -+ Ok(array) -+} -+ -+fn exact_batch( -+ batch: RecordBatch, -+ query: ArrayRef, -+ column: &str, -+ metric: DistanceType, -+ schema: SchemaRef, -+) -> Result { -+ if batch.num_rows() == 0 { -+ return Ok(RecordBatch::new_empty(schema)); -+ } -+ let array = vector_column(&batch, column)?; -+ let vectors = array -+ .as_any() -+ .downcast_ref::() -+ .ok_or_else(|| invalid("multi-vector scoring requires a List column"))?; -+ let row_ids = batch.column_by_name("_rowid"); -+ let mut scores = Vec::with_capacity(batch.num_rows()); -+ for (i, vector) in vectors.iter().enumerate() { -+ if row_ids.is_some_and(|ids| ids.is_null(i)) { -+ scores.push(None); -+ continue; -+ } -+ let Some(vector) = vector else { -+ scores.push(None); -+ continue; -+ }; -+ let vector = vector -+ .as_any() -+ .downcast_ref::() -+ .ok_or_else(|| invalid("multi-vector row must be a FixedSizeList"))?; -+ if vector.is_empty() { -+ scores.push(None); -+ continue; -+ } -+ let score = match query.data_type() { -+ DataType::Float16 => row_distance::(query.as_ref(), vector, metric), -+ DataType::Float32 => row_distance::(query.as_ref(), vector, metric), -+ DataType::Float64 => row_distance::(query.as_ref(), vector, metric), -+ _ => Err(invalid("unsupported multi-vector element type")), -+ }?; -+ scores.push(score); -+ } -+ let mask = BooleanArray::from_iter(scores.iter().map(|score| Some(score.is_some()))); -+ let distances: ArrayRef = Arc::new(Float32Array::from(scores)); -+ let columns = schema -+ .fields() -+ .iter() -+ .map(|field| { -+ if field.name() == "_distance" { -+ Ok(distances.clone()) -+ } else { -+ batch -+ .column_by_name(field.name()) -+ .cloned() -+ .ok_or_else(|| invalid(format!("missing score output column {}", field.name()))) -+ } -+ }) -+ .collect::>>()?; -+ Ok(arrow::compute::filter_record_batch( -+ &RecordBatch::try_new(schema, columns)?, -+ &mask, -+ )?) -+} -+ -+async fn indexed_batch( -+ inputs: Vec, -+ schema: SchemaRef, -+) -> Result { -+ // A query child may emit several batches. Reduce its entire stream exactly once; -+ // treating each batch as a query adds spurious missing-query contributions. -+ let queries = futures::future::try_join_all(inputs.into_iter().map(|mut input| async move { -+ let mut rows = HashMap::::new(); -+ let mut maximum: Option = None; -+ while let Some(batch) = input.try_next().await? { -+ let ids = batch -+ .column_by_name("_rowid") -+ .and_then(|a| a.as_any().downcast_ref::()) -+ .ok_or_else(|| invalid("indexed multi-vector scorer requires row IDs"))?; -+ let distances = batch -+ .column_by_name("_distance") -+ .and_then(|a| a.as_any().downcast_ref::()) -+ .ok_or_else(|| invalid("indexed multi-vector scorer requires distances"))?; -+ for i in 0..batch.num_rows() { -+ let distance = distances.value(i); -+ if ids.is_null(i) || distances.is_null(i) || !distance.is_finite() { -+ return Err(invalid( -+ "indexed multi-vector candidate has invalid row ID or distance", -+ )); -+ } -+ maximum = Some(maximum.map_or(distance, |old| old.max(distance))); -+ rows.entry(ids.value(i)) -+ .and_modify(|old| *old = old.min(distance)) -+ .or_insert(distance); -+ } -+ } -+ Ok::<_, DataFusionError>((rows, maximum.unwrap_or(1.0))) -+ })) -+ .await?; -+ let mut results = HashMap::::new(); -+ let mut missed = 0.0f64; -+ for (rows, maximum) in queries { -+ for (id, score) in &mut results { -+ *score += *rows.get(id).unwrap_or(&maximum) as f64; -+ } -+ for (id, score) in rows { -+ results.entry(id).or_insert(score as f64 + missed); -+ } -+ missed += maximum as f64; -+ } -+ let (ids, scores): (Vec<_>, Vec<_>) = results.into_iter().unzip(); -+ Ok(RecordBatch::try_new( -+ schema, -+ vec![ -+ Arc::new(Float32Array::from_iter_values( -+ scores.into_iter().map(|score| score as f32), -+ )), -+ Arc::new(UInt64Array::from(ids)), -+ ], -+ )?) -+} -+ -+pub(crate) fn validate_query(values: &dyn Array) -> Result<()> { -+ fn finite(values: &dyn Array) -> bool -+ where -+ T::Native: Into, -+ { -+ values -+ .as_any() -+ .downcast_ref::>() -+ .is_some_and(|array| { -+ array.null_count() == 0 -+ && array -+ .values() -+ .iter() -+ .all(|v| Into::::into(*v).is_finite()) -+ }) -+ } -+ let valid = match values.data_type() { -+ DataType::Float16 => finite::(values), -+ DataType::Float32 => finite::(values), -+ DataType::Float64 => finite::(values), -+ _ => false, -+ }; -+ if valid { -+ Ok(()) -+ } else { -+ Err(invalid( -+ "multi-vector query must contain only finite, non-null elements", -+ )) -+ } -+} -+ -+#[cfg(test)] -+mod tests { -+ use super::*; -+ use arrow_schema::{Field, Schema}; -+ use lance_datafusion::exec::{ -+ LanceExecutionOptions, OneShotExec, StrictBatchSizeExec, execute_plan, -+ }; -+ -+ #[test] -+ fn result_window_preserves_strict_batching_and_distance_order() { -+ crate::runtime::block_on(async { -+ for size in [2, 3] { -+ for (offset, limit) in [ -+ (1, Some(4)), -+ (1, Some(3)), -+ (5, Some(4)), -+ (6, Some(4)), -+ (1, None), -+ ] { -+ let schema = Arc::new(Schema::new(vec![Field::new( -+ "_distance", -+ DataType::Float32, -+ false, -+ )])); -+ let batches = [[5., 0.], [4., 1.], [3., 2.]] -+ .into_iter() -+ .map(|values| { -+ RecordBatch::try_new( -+ schema.clone(), -+ vec![Arc::new(Float32Array::from(values.to_vec()))], -+ ) -+ .map_err(DataFusionError::from) -+ }) -+ .collect::>(); -+ let input = Arc::new(OneShotExec::new(Box::pin( -+ RecordBatchStreamAdapter::new(schema, stream::iter(batches)), -+ ))); -+ let plan = Arc::new(StrictBatchSizeExec::new(input, size)); -+ let plan = apply_result_window(plan, offset, limit).unwrap(); -+ let batches: Vec<_> = execute_plan( -+ plan, -+ LanceExecutionOptions { -+ batch_size: Some(2), -+ ..Default::default() -+ }, -+ ) -+ .unwrap() -+ .try_collect() -+ .await -+ .unwrap(); -+ let actual: Vec<_> = batches -+ .iter() -+ .flat_map(|b| { -+ b.column(0) -+ .as_any() -+ .downcast_ref::() -+ .unwrap() -+ .values() -+ .to_vec() -+ }) -+ .collect(); -+ let expected: Vec<_> = (0..6) -+ .skip(offset) -+ .take(limit.unwrap_or(6)) -+ .map(|i| i as f32) -+ .collect(); -+ assert_eq!(actual, expected); -+ assert_eq!( -+ batches -+ .iter() -+ .map(RecordBatch::num_rows) -+ .collect::>(), -+ expected.chunks(size).map(<[f32]>::len).collect::>() -+ ); -+ } -+ } -+ }); -+ } -+} -diff --git a/src/scanner.rs b/src/scanner.rs -index bda554d..f4b3f36 100644 ---- a/src/scanner.rs -+++ b/src/scanner.rs -@@ -363,8 +363,18 @@ impl LanceScanner { - if let Some(cols) = &self.columns { - scanner.project(cols)?; - } -+ let multi_vector = self.nearest.as_ref().is_some_and(|query| { -+ matches!( -+ query.query.data_type(), -+ arrow_schema::DataType::FixedSizeList(_, _) -+ ) -+ }); - if self.limit.is_some() || self.offset.is_some() { - scanner.limit(self.limit, self.offset)?; -+ if multi_vector { -+ // Retain Lance's window validation, but defer truncation until the final sort. -+ scanner.limit(None, None)?; -+ } - } - if let Some(bs) = self.batch_size { - scanner.batch_size(bs); -@@ -440,7 +450,27 @@ impl LanceScanner { - if let Some(query_parallelism) = self.query_parallelism { - scanner.query_parallelism(query_parallelism); - } -- if let Some(rf) = self.refine_factor { -+ if multi_vector { -+ if matches!( -+ self.metric_override, -+ Some(crate::index::LanceMetricType::Hamming) -+ ) { -+ return Err(lance_core::Error::invalid_input_source( -+ "multi-vector queries support only l2, cosine, and dot metrics".into(), -+ )); -+ } -+ let refine = self.refine_factor.unwrap_or(1); -+ if refine == 0 -+ || n.k as usize -+ > crate::multivector::MAX_QUERY_VECTOR_CANDIDATES / refine as usize -+ { -+ return Err(lance_core::Error::invalid_input_source( -+ "multi-vector refined candidate count must be in 1..=100000".into(), -+ )); -+ } -+ // Validate actual stored values and refine candidate scores before TopK. -+ scanner.refine(refine); -+ } else if let Some(rf) = self.refine_factor { - scanner.refine(rf); - } - if let Some(ef) = self.ef { -@@ -448,6 +478,9 @@ impl LanceScanner { - } - if let Some(m) = self.metric_override { - scanner.distance_metric(m.to_distance()); -+ } else if multi_vector { -+ // Resolve the same default on indexed and uncovered fragments. -+ scanner.distance_metric(lance_linalg::distance::DistanceType::L2); - } - if let Some(ui) = self.use_index { - scanner.use_index(ui); -@@ -506,6 +539,12 @@ impl LanceScanner { - scanner, - distributed_fts, - scalar_segment, -+ multi_vector_window: multi_vector.then_some(( -+ self.offset.unwrap_or(0) as usize, -+ self.limit.map(|n| n as usize), -+ )), -+ batch_size: self.batch_size, -+ scan_statistics_callback: self.scan_statistics_callback.clone(), - }) - } - } -@@ -521,6 +560,9 @@ struct PreparedScanner { - scanner: lance::dataset::scanner::Scanner, - distributed_fts: Option, - scalar_segment: Option, -+ multi_vector_window: Option<(usize, Option)>, -+ batch_size: Option, -+ scan_statistics_callback: Option, - } - - impl PreparedScanner { -@@ -532,6 +574,19 @@ impl PreparedScanner { - .try_into_stream() - .await; - } -+ if let Some((offset, limit)) = self.multi_vector_window { -+ let plan = crate::multivector::rewrite(self.scanner.create_plan().await?)?; -+ let plan = crate::multivector::apply_result_window(plan, offset, limit)?; -+ let stream = lance_datafusion::exec::execute_plan( -+ plan, -+ lance_datafusion::exec::LanceExecutionOptions { -+ batch_size: self.batch_size, -+ execution_stats_callback: self.scan_statistics_callback, -+ ..Default::default() -+ }, -+ )?; -+ return Ok(DatasetRecordBatchStream::new(stream)); -+ } - let Some(distributed_fts) = self.distributed_fts else { - return self.scanner.try_into_stream().await; - }; -@@ -2744,6 +2799,21 @@ unsafe fn scanner_nearest_inner( - } - let column_str = unsafe { helpers::parse_c_string(column)? }.unwrap(); - -+ let query = unsafe { decode_query_values(query_data, query_len, element_type)? }; -+ -+ s.nearest = Some(NearestQuery { -+ column: column_str.to_string(), -+ query, -+ k, -+ }); -+ Ok(0) -+} -+ -+unsafe fn decode_query_values( -+ query_data: *const c_void, -+ query_len: usize, -+ element_type: i32, -+) -> Result { - let dtype = match element_type { - 0 => LanceDataType::Float32, - 1 => LanceDataType::Float16, -@@ -2782,9 +2852,112 @@ unsafe fn scanner_nearest_inner( - } - }; - -+ Ok(query) -+} -+ -+/// Set one multi-vector query, supplied as a row-major matrix of floating-point values. -+/// The caller must supply dimension * num_vectors aligned elements matching the column type. -+#[unsafe(no_mangle)] -+pub unsafe extern "C" fn lance_scanner_nearest_multivector( -+ scanner: *mut LanceScanner, -+ column: *const c_char, -+ query_data: *const c_void, -+ dimension: usize, -+ num_vectors: usize, -+ element_type: i32, -+ k: u32, -+) -> i32 { -+ scanner_poison_check!(scanner, -1); -+ scanner_ffi_try!(scanner, unsafe { -+ nearest_multivector_inner( -+ scanner, -+ column, -+ query_data, -+ dimension, -+ num_vectors, -+ element_type, -+ k, -+ ) -+ },) -+} -+ -+unsafe fn nearest_multivector_inner( -+ scanner: *mut LanceScanner, -+ column: *const c_char, -+ query_data: *const c_void, -+ dimension: usize, -+ num_vectors: usize, -+ element_type: i32, -+ k: u32, -+) -> Result { -+ use arrow_schema::{DataType, Field}; -+ let invalid = |message: &str| lance_core::Error::invalid_input_source(message.into()); -+ if scanner.is_null() || column.is_null() || query_data.is_null() { -+ return Err(invalid("scanner, column, and query_data must not be NULL")); -+ } -+ if dimension == 0 || dimension > i32::MAX as usize || num_vectors == 0 || k == 0 { -+ return Err(invalid( -+ "dimension, num_vectors, and k must be positive; dimension must fit int32", -+ )); -+ } -+ if num_vectors > crate::multivector::MAX_QUERY_VECTORS -+ || num_vectors > crate::multivector::MAX_QUERY_VECTOR_CANDIDATES / k as usize -+ { -+ return Err(invalid( -+ "multi-vector query exceeds 128 subvectors or 100000 subvector-candidates", -+ )); -+ } -+ let (data_type, width) = match element_type { -+ 0 => (DataType::Float32, 4), -+ 1 => (DataType::Float16, 2), -+ 2 => (DataType::Float64, 8), -+ _ => { -+ return Err(invalid( -+ "multi-vector queries require float16, float32, or float64", -+ )); -+ } -+ }; -+ let count = dimension -+ .checked_mul(num_vectors) -+ .filter(|count| *count <= isize::MAX as usize / width) -+ .ok_or_else(|| invalid("query matrix byte size overflows"))?; -+ let s = unsafe { &mut *scanner }; -+ if s.fts_query.is_some() || s.fts_context.is_some() { -+ return Err(invalid( -+ "nearest and full-text search are mutually exclusive", -+ )); -+ } -+ let column = unsafe { helpers::parse_c_string(column)? }.unwrap(); -+ let field = s -+ .dataset -+ .schema() -+ .field(column) -+ .ok_or_else(|| invalid("multi-vector column does not exist"))?; -+ match field.data_type() { -+ DataType::List(child) if !child.is_nullable() => match child.data_type() { -+ DataType::FixedSizeList(element, dim) -+ if *dim == dimension as i32 && *element.data_type() == data_type => {} -+ _ => return Err(invalid("multi-vector dimension/type mismatch")), -+ }, -+ _ => { -+ return Err(invalid( -+ "multi-vector column must be List of non-nullable FixedSizeList", -+ )); -+ } -+ } -+ // A primitive array is interpreted as one vector by Lance. Preserve matrix shape even -+ // for a single subvector. Lance does not preserve element nullability in its schema. -+ let values = unsafe { decode_query_values(query_data, count, element_type)? }; -+ crate::multivector::validate_query(values.as_ref())?; -+ let query = arrow_array::FixedSizeListArray::try_new( -+ Arc::new(Field::new("item", data_type, false)), -+ dimension as i32, -+ values, -+ None, -+ )?; - s.nearest = Some(NearestQuery { -- column: column_str.to_string(), -- query, -+ column: column.to_string(), -+ query: Arc::new(query), - k, - }); - Ok(0) -diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 04699da..24bef9f 100644 ---- a/tests/c_api_test.rs -+++ b/tests/c_api_test.rs -@@ -13388,3 +13388,28 @@ fn test_scalar_segment_requires_explicit_domain_and_checks_uuid() { - lance_dataset_close(ds); - } - } -+ -+#[test] -+fn test_multivector_nearest_rejects_null_handle() { -+ let column = c_str("vectors"); -+ let query = [1.0f32, 0.0]; -+ let status = unsafe { -+ lance_scanner_nearest_multivector( -+ ptr::null_mut(), -+ column.as_ptr(), -+ query.as_ptr().cast(), -+ 2, -+ 1, -+ 0, -+ 1, -+ ) -+ }; -+ assert_eq!(status, -1); -+ let error = lance_last_error_message(); -+ assert!(!error.is_null()); -+ let message = unsafe { std::ffi::CStr::from_ptr(error) } -+ .to_string_lossy() -+ .into_owned(); -+ unsafe { lance_free_string(error) }; -+ assert!(message.contains("NULL")); -+} -diff --git a/tests/cpp/test_cpp_api.cpp b/tests/cpp/test_cpp_api.cpp -index fc6fc82..11badc6 100644 ---- a/tests/cpp/test_cpp_api.cpp -+++ b/tests/cpp/test_cpp_api.cpp -@@ -427,6 +427,20 @@ static void test_nearest_smoke(const std::string& uri) { - PASS(); - } - -+static void test_multivector_rejects_flat_column(const std::string& uri) { -+ TEST(test_multivector_rejects_flat_column); -+ auto scanner = lance::Dataset::open(uri).scan(); -+ const float query[8] = {1.0f, 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, 0.0f, 0.0f}; -+ bool caught = false; -+ try { -+ scanner.nearest_multivector("embedding", query, 8, 1, LANCE_DTYPE_FLOAT32, 1); -+ } catch (const lance::Error&) { -+ caught = true; -+ } -+ assert(caught); -+ PASS(); -+} -+ - static void test_index_segments_smoke(const std::string& /*uri*/) { - TEST(test_index_segments_smoke); - -@@ -971,6 +985,7 @@ int main(int argc, char** argv) { - test_error_exception(uri); - test_index_lifecycle(uri); - test_nearest_smoke(uri); -+ test_multivector_rejects_flat_column(uri); - test_index_segments_smoke(uri); - test_index_segment_builder(uri); - test_vector_models_and_reusable_segments(uri); -diff --git a/tests/multivector_test.rs b/tests/multivector_test.rs -new file mode 100644 -index 0000000..f02e850 ---- /dev/null -+++ b/tests/multivector_test.rs -@@ -0,0 +1,965 @@ -+// SPDX-License-Identifier: Apache-2.0 -+// SPDX-FileCopyrightText: Copyright The Lance Authors -+ -+use std::ffi::{CString, c_void}; -+use std::ptr; -+use std::sync::Arc; -+ -+use arrow::buffer::{NullBuffer, OffsetBuffer}; -+use arrow::ffi_stream::{ArrowArrayStreamReader, FFI_ArrowArrayStream}; -+use arrow::record_batch::RecordBatchIterator; -+use arrow_array::{Array, FixedSizeListArray, Float32Array, Int32Array, ListArray, RecordBatch}; -+use arrow_schema::{DataType, Field, Schema}; -+use lance::Dataset; -+use lance_c::*; -+ -+fn fixture() -> (tempfile::TempDir, CString) { -+ let dir = tempfile::tempdir().unwrap(); -+ let path = dir.path().join("vectors.lance"); -+ let element = Arc::new(Field::new("item", DataType::Float32, false)); -+ let vectors = FixedSizeListArray::try_new( -+ element, -+ 2, -+ Arc::new(Float32Array::from(vec![ -+ 1., 0., 0., 1., 2., 0., 3., 0., 0., 3., -+ ])), -+ None, -+ ) -+ .unwrap(); -+ let child = Arc::new(Field::new("item", vectors.data_type().clone(), false)); -+ let rows = ListArray::try_new( -+ child, -+ OffsetBuffer::new(vec![0i32, 2, 3, 5, 5, 5].into()), -+ Arc::new(vectors), -+ Some(NullBuffer::from(vec![true, true, true, true, false])), -+ ) -+ .unwrap(); -+ let schema = Arc::new(Schema::new(vec![ -+ Field::new("id", DataType::Int32, false), -+ Field::new("vectors", rows.data_type().clone(), true), -+ ])); -+ let batch = RecordBatch::try_new( -+ schema.clone(), -+ vec![ -+ Arc::new(Int32Array::from(vec![1, 2, 3, 4, 5])), -+ Arc::new(rows), -+ ], -+ ) -+ .unwrap(); -+ lance_c::runtime::block_on(async { -+ Dataset::write( -+ RecordBatchIterator::new(vec![Ok(batch)], schema), -+ path.to_str().unwrap(), -+ None, -+ ) -+ .await -+ .unwrap(); -+ }); -+ (dir, CString::new(path.to_str().unwrap()).unwrap()) -+} -+ -+unsafe fn collect(scanner: *mut LanceScanner) -> Vec<(i32, f32)> { -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ let status = unsafe { lance_scanner_to_arrow_stream(scanner, &mut stream) }; -+ if status != 0 { -+ panic!( -+ "{}", -+ unsafe { std::ffi::CStr::from_ptr(lance_last_error_message()) }.to_string_lossy() -+ ); -+ } -+ let reader = unsafe { ArrowArrayStreamReader::from_raw(&mut stream) }.unwrap(); -+ reader -+ .flat_map(|batch| { -+ let batch = batch.unwrap(); -+ let ids = batch -+ .column_by_name("id") -+ .unwrap() -+ .as_any() -+ .downcast_ref::() -+ .unwrap(); -+ let distances = batch -+ .column_by_name("_distance") -+ .unwrap() -+ .as_any() -+ .downcast_ref::() -+ .unwrap(); -+ (0..batch.num_rows()) -+ .map(|i| (ids.value(i), distances.value(i))) -+ .collect::>() -+ }) -+ .collect() -+} -+ -+#[test] -+fn multivector_search_scores_logical_rows_and_excludes_empty_and_null_rows() { -+ let (_dir, uri) = fixture(); -+ let column = CString::new("vectors").unwrap(); -+ let q = [1.0f32, 0., 0., 1.]; -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ assert!(!ds.is_null()); -+ for count in [1, 2] { -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ q.as_ptr() as *const c_void, -+ 2, -+ count, -+ 0, -+ 10 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, false), 0); -+ assert_eq!(lance_scanner_set_prefilter(scan, true), 0); -+ let rows = collect(scan); -+ let expected = if count == 2 { -+ vec![(1, 0.), (2, 6.), (3, 8.)] -+ } else { -+ vec![(1, 0.), (2, 1.), (3, 4.)] -+ }; -+ assert_eq!(rows, expected); -+ lance_scanner_close(scan); -+ } -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn multivector_rejects_invalid_shape_and_preserves_previous_query() { -+ let (_dir, uri) = fixture(); -+ let column = CString::new("vectors").unwrap(); -+ let q = [1.0f32, 0., 0., 1.]; -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ q.as_ptr() as *const c_void, -+ 2, -+ 2, -+ 0, -+ 10 -+ ), -+ 0 -+ ); -+ for (dim, count, dtype, k) in [ -+ (0, 2, 0, 10), -+ (2, 0, 0, 10), -+ (3, 1, 0, 10), -+ (2, 2, 3, 10), -+ (2, 2, 4, 10), -+ (2, 2, 1, 10), -+ (2, 2, 0, 0), -+ (usize::MAX, 2, 0, 10), -+ (2, usize::MAX, 0, 10), -+ ] { -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ q.as_ptr() as *const c_void, -+ dim, -+ count, -+ dtype, -+ k -+ ), -+ -1 -+ ); -+ } -+ assert_eq!( -+ lance_scanner_nearest_multivector(scan, column.as_ptr(), ptr::null(), 2, 2, 0, 10), -+ -1 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, false), 0); -+ for (limit, offset) in [(-1, 0), (10, -1)] { -+ assert_eq!(lance_scanner_set_limit(scan, limit), 0); -+ assert_eq!(lance_scanner_set_offset(scan, offset), 0); -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!(lance_scanner_to_arrow_stream(scan, &mut stream), -1); -+ } -+ assert_eq!(lance_scanner_set_limit(scan, 10), 0); -+ assert_eq!(lance_scanner_set_offset(scan, 0), 0); -+ assert_eq!(collect(scan), vec![(1, 0.), (2, 6.), (3, 8.)]); -+ lance_scanner_close(scan); -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn fragment_scoped_indexed_search_orders_candidates_before_offset() { -+ use lance::index::DatasetIndexExt; -+ use lance::index::vector::VectorIndexParams; -+ use lance_index::IndexType; -+ use lance_linalg::distance::MetricType; -+ -+ let (_dir, uri) = fixture(); -+ lance_c::runtime::block_on(async { -+ let mut ds = Dataset::open(uri.to_str().unwrap()).await.unwrap(); -+ let batch = ds.scan().try_into_batch().await.unwrap(); -+ ds.create_index( -+ &["vectors"], -+ IndexType::Vector, -+ None, -+ &VectorIndexParams::ivf_flat(1, MetricType::Cosine), -+ false, -+ ) -+ .await -+ .unwrap(); -+ ds.append( -+ RecordBatchIterator::new(vec![Ok(batch.clone())], batch.schema()), -+ None, -+ ) -+ .await -+ .unwrap(); -+ }); -+ let column = CString::new("vectors").unwrap(); -+ let filter = CString::new("id >= 2").unwrap(); -+ let q = [1.0f32, 0., 0., 1.]; -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ for _ in 0..10 { -+ let scan = lance_scanner_new(ds, ptr::null(), filter.as_ptr()); -+ assert_eq!( -+ lance_scanner_set_fragment_ids(scan, [0u64, 1].as_ptr(), 2), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_prefilter(scan, true), 0); -+ assert_eq!(lance_scanner_set_batch_size(scan, 1), 0); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ q.as_ptr() as *const c_void, -+ 2, -+ 2, -+ 0, -+ 4 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_metric(scan, 1), 0); -+ assert_eq!(lance_scanner_set_refine_factor(scan, 1), 0); -+ assert_eq!(lance_scanner_set_offset(scan, 1), 0); -+ assert_eq!(lance_scanner_set_limit(scan, 2), 0); -+ assert_eq!(collect(scan), vec![(3, 0.), (2, 1.)]); -+ lance_scanner_close(scan); -+ } -+ lance_dataset_close(ds); -+ } -+} -+ -+fn custom_fixture( -+ rows: Vec>>, -+ dim: i32, -+ indexed: bool, -+) -> (tempfile::TempDir, CString) { -+ custom_fixture_at(rows, dim, indexed, "vectors") -+} -+ -+fn custom_fixture_at( -+ rows: Vec>>, -+ dim: i32, -+ indexed: bool, -+ column: &str, -+) -> (tempfile::TempDir, CString) { -+ use lance::index::DatasetIndexExt; -+ use lance::index::vector::VectorIndexParams; -+ use lance_index::IndexType; -+ use lance_linalg::distance::MetricType; -+ let dir = tempfile::tempdir().unwrap(); -+ let path = dir.path().join("vectors.lance"); -+ let mut offsets = vec![0i32]; -+ let mut values = Vec::new(); -+ for row in &rows { -+ values.extend_from_slice(row); -+ offsets.push(values.len() as i32 / dim); -+ } -+ let vectors = FixedSizeListArray::try_new( -+ Arc::new(Field::new("item", DataType::Float32, true)), -+ dim, -+ Arc::new(Float32Array::from(values)), -+ None, -+ ) -+ .unwrap(); -+ let rows_array = ListArray::try_new( -+ Arc::new(Field::new("item", vectors.data_type().clone(), false)), -+ OffsetBuffer::new(offsets.into()), -+ Arc::new(vectors), -+ None, -+ ) -+ .unwrap(); -+ let parts = lance_core::datatypes::parse_field_path(column).unwrap(); -+ let mut array: arrow_array::ArrayRef = Arc::new(rows_array); -+ for name in parts[1..].iter().rev() { -+ let field = Arc::new(Field::new(name, array.data_type().clone(), true)); -+ let labels: arrow_array::ArrayRef = Arc::new(Int32Array::from(vec![7; rows.len()])); -+ array = Arc::new(arrow_array::StructArray::from(vec![ -+ (field, array), -+ ( -+ Arc::new(Field::new("label", DataType::Int32, false)), -+ labels, -+ ), -+ ])); -+ } -+ let schema = Arc::new(Schema::new(vec![ -+ Field::new("id", DataType::Int32, false), -+ Field::new(&parts[0], array.data_type().clone(), true), -+ ])); -+ let batch = RecordBatch::try_new( -+ schema.clone(), -+ vec![ -+ Arc::new(Int32Array::from_iter_values(1..=rows.len() as i32)), -+ array, -+ ], -+ ) -+ .unwrap(); -+ lance_c::runtime::block_on(async { -+ let mut ds = Dataset::write( -+ RecordBatchIterator::new(vec![Ok(batch)], schema), -+ path.to_str().unwrap(), -+ None, -+ ) -+ .await -+ .unwrap(); -+ if indexed { -+ ds.create_index( -+ &[column], -+ IndexType::Vector, -+ None, -+ &VectorIndexParams::ivf_flat(1, MetricType::Cosine), -+ false, -+ ) -+ .await -+ .unwrap(); -+ } -+ }); -+ (dir, CString::new(path.to_str().unwrap()).unwrap()) -+} -+ -+#[test] -+fn exact_top_one_preserves_small_distances_before_truncation() { -+ let (_dir, uri) = custom_fixture(vec![vec![Some(0.00015)], vec![Some(0.0001)]], 1, false); -+ let column = CString::new("vectors").unwrap(); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ for count in [1, 2] { -+ for batch_size in [1, 1024] { -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ [0.0f32, 0.0].as_ptr().cast(), -+ 1, -+ count, -+ 0, -+ 1 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, false), 0); -+ assert_eq!(lance_scanner_set_batch_size(scan, batch_size), 0); -+ let rows = collect(scan); -+ assert_eq!(rows.len(), 1); -+ assert_eq!(rows[0].0, 2); -+ assert!((rows[0].1 - count as f32 * 1e-8).abs() < 1e-14, "{rows:?}"); -+ lance_scanner_close(scan); -+ } -+ } -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn indexed_top_one_is_independent_of_child_batch_boundaries() { -+ let rows = vec![ -+ vec![Some(1.), Some(0.)], -+ vec![Some(0.), Some(1.)], -+ vec![Some(1.), Some(1.)], -+ vec![Some(2.), Some(1.)], -+ vec![Some(1.), Some(2.)], -+ ]; -+ let (_dir, uri) = custom_fixture(rows, 2, true); -+ let column = CString::new("vectors").unwrap(); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ for batch_size in [1, 2, 1024] { -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ [1.0f32, 0., 0., 1.].as_ptr().cast(), -+ 2, -+ 2, -+ 0, -+ 1 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_metric(scan, 1), 0); -+ assert_eq!(lance_scanner_set_refine_factor(scan, 1), 0); -+ assert_eq!(lance_scanner_set_batch_size(scan, batch_size), 0); -+ let result = collect(scan); -+ assert_eq!(result[0].0, 3, "batch_size={batch_size}"); -+ assert!((result[0].1 - (2.0 - 2.0f32.sqrt())).abs() < 1e-6); -+ lance_scanner_close(scan); -+ } -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn actual_null_and_nonfinite_stored_elements_fail_search() { -+ let column = CString::new("vectors").unwrap(); -+ for value in [ -+ None, -+ Some(f32::NAN), -+ Some(f32::INFINITY), -+ Some(f32::NEG_INFINITY), -+ ] { -+ let (_dir, uri) = custom_fixture(vec![vec![value, Some(0.)]], 2, false); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ [0.0f32, 0.].as_ptr().cast(), -+ 2, -+ 1, -+ 0, -+ 1 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, false), 0); -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!(lance_scanner_to_arrow_stream(scan, &mut stream), 0); -+ let mut reader = ArrowArrayStreamReader::from_raw(&mut stream).unwrap(); -+ let result = reader.next(); -+ assert!( -+ matches!(result, Some(Err(_))), -+ "value={value:?}: {result:?}" -+ ); -+ drop(reader); -+ lance_scanner_close(scan); -+ lance_dataset_close(ds); -+ } -+ } -+} -+ -+#[test] -+fn rejects_excessive_multivector_plan_width_and_candidate_work() { -+ let (_dir, uri) = fixture(); -+ let column = CString::new("vectors").unwrap(); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ let q = [0.0f32; 258]; -+ for (count, k) in [(129, 1), (128, 782), (1, 100001)] { -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ q.as_ptr().cast(), -+ 2, -+ count, -+ 0, -+ k -+ ), -+ -1 -+ ); -+ } -+ lance_scanner_close(scan); -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn omitted_metric_is_l2_on_indexed_and_unindexed_fragments() { -+ let (_dir, uri) = custom_fixture(vec![vec![Some(2.), Some(0.)]], 2, true); -+ lance_c::runtime::block_on(async { -+ let mut ds = Dataset::open(uri.to_str().unwrap()).await.unwrap(); -+ let old = ds.scan().try_into_batch().await.unwrap(); -+ let vectors = FixedSizeListArray::try_new( -+ Arc::new(Field::new("item", DataType::Float32, true)), -+ 2, -+ Arc::new(Float32Array::from(vec![1.0, 0.1])), -+ None, -+ ) -+ .unwrap(); -+ let lists = ListArray::try_new( -+ Arc::new(Field::new("item", vectors.data_type().clone(), false)), -+ OffsetBuffer::new(vec![0i32, 1].into()), -+ Arc::new(vectors), -+ None, -+ ) -+ .unwrap(); -+ let batch = RecordBatch::try_new( -+ old.schema(), -+ vec![Arc::new(Int32Array::from(vec![2])), Arc::new(lists)], -+ ) -+ .unwrap(); -+ ds.append( -+ RecordBatchIterator::new(vec![Ok(batch)], old.schema()), -+ None, -+ ) -+ .await -+ .unwrap(); -+ }); -+ let column = CString::new("vectors").unwrap(); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ for (fragment, id, score) in [(0u64, 1, 1.0f32), (1u64, 2, 0.01f32)] { -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ [1.0f32, 0.].as_ptr().cast(), -+ 2, -+ 1, -+ 0, -+ 1 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_prefilter(scan, true), 0); -+ assert_eq!(lance_scanner_set_fragment_ids(scan, &fragment, 1), 0); -+ let rows = collect(scan); -+ assert_eq!(rows[0].0, id); -+ assert!( -+ (rows[0].1 - score).abs() < 1e-6, -+ "fragment={fragment}: {rows:?}" -+ ); -+ lance_scanner_close(scan); -+ } -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn query_values_and_refinement_are_validated_before_execution() { -+ let (_dir, uri) = fixture(); -+ let column = CString::new("vectors").unwrap(); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ for value in [f32::NAN, f32::INFINITY, f32::NEG_INFINITY] { -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ [value, 0.0f32].as_ptr().cast(), -+ 2, -+ 1, -+ 0, -+ 1 -+ ), -+ -1 -+ ); -+ } -+ let query = [1.0f32; 256]; -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ query.as_ptr().cast(), -+ 2, -+ 128, -+ 0, -+ 781 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_metric(scan, 3), 0); -+ let mut invalid_metric_stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!( -+ lance_scanner_to_arrow_stream(scan, &mut invalid_metric_stream), -+ -1 -+ ); -+ assert_eq!(lance_scanner_set_metric(scan, 0), 0); -+ assert_eq!(lance_scanner_set_refine_factor(scan, u32::MAX), 0); -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!(lance_scanner_to_arrow_stream(scan, &mut stream), -1); -+ lance_scanner_close(scan); -+ lance_dataset_close(ds); -+ } -+} -+ -+#[test] -+fn float16_and_float64_queries_support_all_distance_metrics() { -+ let (dir, source) = fixture(); -+ for (code, dtype) in [(1, DataType::Float16), (2, DataType::Float64)] { -+ let path = dir.path().join(format!("typed_{code}.lance")); -+ lance_c::runtime::block_on(async { -+ let source = Dataset::open(source.to_str().unwrap()).await.unwrap(); -+ let batch = source.scan().try_into_batch().await.unwrap(); -+ let vector_type = DataType::List(Arc::new(Field::new( -+ "item", -+ DataType::FixedSizeList(Arc::new(Field::new("item", dtype, true)), 2), -+ false, -+ ))); -+ let vectors = -+ arrow::compute::cast(batch.column_by_name("vectors").unwrap(), &vector_type) -+ .unwrap(); -+ let schema = Arc::new(Schema::new(vec![ -+ Field::new("id", DataType::Int32, false), -+ Field::new("vectors", vector_type, true), -+ ])); -+ let batch = -+ RecordBatch::try_new(schema.clone(), vec![batch.column(0).clone(), vectors]) -+ .unwrap(); -+ Dataset::write( -+ RecordBatchIterator::new(vec![Ok(batch)], schema), -+ path.to_str().unwrap(), -+ None, -+ ) -+ .await -+ .unwrap(); -+ }); -+ let uri = CString::new(path.to_str().unwrap()).unwrap(); -+ let column = CString::new("vectors").unwrap(); -+ let f16_query = [1.0f32, 0., 0., 1.].map(half::f16::from_f32); -+ let f64_query = [1.0f64, 0., 0., 1.]; -+ let query = if code == 1 { -+ f16_query.as_ptr().cast() -+ } else { -+ f64_query.as_ptr().cast() -+ }; -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ for (metric, expected) in [(0, [0., 6., 8.]), (1, [0., 1., 0.]), (2, [0., 0., -4.])] { -+ let scanner = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scanner, -+ column.as_ptr(), -+ query, -+ 2, -+ 2, -+ code, -+ 10 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_metric(scanner, metric), 0); -+ assert_eq!(lance_scanner_set_use_index(scanner, false), 0); -+ let mut rows = collect(scanner); -+ rows.sort_by_key(|row| row.0); -+ assert_eq!( -+ rows, -+ vec![(1, expected[0]), (2, expected[1]), (3, expected[2])] -+ ); -+ lance_scanner_close(scanner); -+ } -+ lance_dataset_close(ds); -+ } -+ } -+} -+ -+#[test] -+fn cosine_zero_norm_rows_do_not_abort_exact_or_refined_search() { -+ let rows = vec![ -+ vec![Some(0.), Some(0.)], -+ vec![Some(1.), Some(0.)], -+ vec![Some(0.), Some(0.), Some(0.), Some(1.)], -+ ]; -+ for indexed in [false, true] { -+ let (_dir, uri) = custom_fixture(rows.clone(), 2, indexed); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ for query in [[1.0f32, 0., 0., 1.], [0., 0., 0., 1.]] { -+ for count in [1, 2] { -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ c"vectors".as_ptr(), -+ query.as_ptr().cast(), -+ 2, -+ count, -+ 0, -+ 3 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, indexed), 0); -+ assert_eq!(lance_scanner_set_metric(scan, 1), 0); -+ assert_eq!(lance_scanner_set_refine_factor(scan, 2), 0); -+ let mut actual = collect(scan); -+ actual.sort_by_key(|row| row.0); -+ let expected = if query[0] == 0. { -+ vec![] -+ } else if count == 1 { -+ vec![(2, 0.), (3, 1.)] -+ } else { -+ vec![(2, 1.), (3, 1.)] -+ }; -+ assert_eq!( -+ actual, expected, -+ "indexed={indexed}, query={query:?}, count={count}" -+ ); -+ lance_scanner_close(scan); -+ } -+ } -+ lance_dataset_close(ds); -+ } -+ } -+} -+ -+#[test] -+fn nested_and_quoted_columns_support_exact_and_refined_search() { -+ for column in [ -+ "payload.vectors", -+ "payload.`vectors.with.dot`", -+ "payload.inner.`vectors.with.dot`", -+ ] { -+ let rows = vec![ -+ vec![Some(1.), Some(0.), Some(0.), Some(1.)], -+ vec![Some(2.), Some(0.)], -+ vec![Some(1.), Some(1.)], -+ ]; -+ for indexed in [false, true] { -+ let (_dir, uri) = custom_fixture_at(rows.clone(), 2, indexed, column); -+ let column = CString::new(column).unwrap(); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ [1.0f32, 0., 0., 1.].as_ptr().cast(), -+ 2, -+ 2, -+ 0, -+ 3 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, indexed), 0); -+ assert_eq!(lance_scanner_set_metric(scan, 1), 0); -+ assert_eq!(lance_scanner_set_refine_factor(scan, 2), 0); -+ let actual = collect(scan); -+ assert_eq!(actual.len(), 3); -+ assert_eq!( -+ actual.iter().map(|row| row.0).collect::>(), -+ vec![1, 3, 2] -+ ); -+ assert_eq!(actual[0].1, 0.); -+ assert!((actual[1].1 - (2. - 2.0f32.sqrt())).abs() < 1e-6); -+ assert_eq!(actual[2].1, 1.); -+ lance_scanner_close(scan); -+ lance_dataset_close(ds); -+ } -+ } -+ } -+} -+ -+#[test] -+fn nested_projections_preserve_schema_and_values_in_exact_indexed_and_hybrid_search() { -+ for column in ["payload.vectors", "payload.`vectors.with.dot`"] { -+ for indexed in [false, true] { -+ for appended in [false, true] { -+ let (_dir, uri) = custom_fixture_at( -+ vec![vec![Some(1.), Some(0.)], vec![Some(0.), Some(1.)]], -+ 2, -+ indexed, -+ column, -+ ); -+ if appended { -+ lance_c::runtime::block_on(async { -+ let mut ds = Dataset::open(uri.to_str().unwrap()).await.unwrap(); -+ let batch = ds.scan().try_into_batch().await.unwrap(); -+ ds.append( -+ RecordBatchIterator::new(vec![Ok(batch.clone())], batch.schema()), -+ None, -+ ) -+ .await -+ .unwrap(); -+ }); -+ } -+ for projection in [ -+ vec![], -+ vec!["id"], -+ vec!["payload.label"], -+ vec!["id", column], -+ ] { -+ let expected = lance_c::runtime::block_on(async { -+ let ds = Dataset::open(uri.to_str().unwrap()).await.unwrap(); -+ let mut scanner = ds.scan(); -+ scanner.scan_in_order(true); -+ if !projection.is_empty() { -+ scanner.project(&projection).unwrap(); -+ } -+ scanner.try_into_batch().await.unwrap() -+ }); -+ // The appended fragment repeats the same two values; distance order groups -+ // both exact matches before the orthogonal rows, regardless of tie order. -+ let indices = arrow_array::UInt32Array::from(if appended { -+ vec![0, 2, 1, 3] -+ } else { -+ vec![0, 1] -+ }); -+ let expected = RecordBatch::try_new( -+ expected.schema(), -+ expected -+ .columns() -+ .iter() -+ .map(|array| { -+ arrow::compute::take(array.as_ref(), &indices, None).unwrap() -+ }) -+ .collect(), -+ ) -+ .unwrap(); -+ unsafe { -+ let names: Vec<_> = projection -+ .iter() -+ .map(|name| CString::new(*name).unwrap()) -+ .collect(); -+ let mut columns: Vec<_> = names.iter().map(|name| name.as_ptr()).collect(); -+ columns.push(ptr::null()); -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ let scan = lance_scanner_new( -+ ds, -+ if projection.is_empty() { -+ ptr::null() -+ } else { -+ columns.as_ptr() -+ }, -+ ptr::null(), -+ ); -+ let column = CString::new(column).unwrap(); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ column.as_ptr(), -+ [1.0f32, 0.].as_ptr().cast(), -+ 2, -+ 1, -+ 0, -+ 10 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, indexed), 0); -+ assert_eq!(lance_scanner_set_metric(scan, 1), 0); -+ assert_eq!(lance_scanner_set_batch_size(scan, 1), 0); -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!(lance_scanner_to_arrow_stream(scan, &mut stream), 0); -+ let batches = ArrowArrayStreamReader::from_raw(&mut stream) -+ .unwrap() -+ .collect::, _>>() -+ .unwrap(); -+ lance_scanner_close(scan); -+ lance_dataset_close(ds); -+ let actual = -+ arrow::compute::concat_batches(&batches[0].schema(), &batches).unwrap(); -+ let distances = actual -+ .column_by_name("_distance") -+ .unwrap() -+ .as_any() -+ .downcast_ref::() -+ .unwrap(); -+ assert_eq!( -+ distances.values().as_ref(), -+ if appended { -+ &[0., 0., 1., 1.][..] -+ } else { -+ &[0., 1.][..] -+ } -+ ); -+ let positions: Vec<_> = actual -+ .schema() -+ .fields() -+ .iter() -+ .enumerate() -+ .filter_map(|(i, field)| (field.name() != "_distance").then_some(i)) -+ .collect(); -+ assert_eq!( -+ actual.project(&positions).unwrap(), -+ expected, -+ "column={column:?} indexed={indexed} appended={appended} projection={projection:?}" -+ ); -+ } -+ } -+ } -+ } -+ } -+} -+ -+#[test] -+fn strict_batches_apply_after_multivector_offset_and_limit() { -+ let (_dir, uri) = custom_fixture( -+ (1..=6).map(|i| vec![Some(i as f32), Some(0.)]).collect(), -+ 2, -+ false, -+ ); -+ unsafe { -+ let ds = lance_dataset_open(uri.as_ptr(), ptr::null(), 0); -+ for batch_size in [Some(2), None] { -+ for (offset, limit) in [(1, 4), (1, 3), (5, 4), (6, 4), (0, 6)] { -+ let scan = lance_scanner_new(ds, ptr::null(), ptr::null()); -+ assert_eq!( -+ lance_scanner_nearest_multivector( -+ scan, -+ c"vectors".as_ptr(), -+ [1.0f32, 0.].as_ptr().cast(), -+ 2, -+ 1, -+ 0, -+ 6 -+ ), -+ 0 -+ ); -+ assert_eq!(lance_scanner_set_use_index(scan, false), 0); -+ if let Some(size) = batch_size { -+ assert_eq!(lance_scanner_set_batch_size(scan, size), 0); -+ } -+ assert_eq!(lance_scanner_set_strict_batch_size(scan, true), 0); -+ assert_eq!(lance_scanner_set_offset(scan, offset), 0); -+ assert_eq!(lance_scanner_set_limit(scan, limit), 0); -+ let mut stream = FFI_ArrowArrayStream::empty(); -+ assert_eq!(lance_scanner_to_arrow_stream(scan, &mut stream), 0); -+ let batches = ArrowArrayStreamReader::from_raw(&mut stream) -+ .unwrap() -+ .collect::, _>>() -+ .unwrap(); -+ lance_scanner_close(scan); -+ let sizes: Vec<_> = batches.iter().map(RecordBatch::num_rows).collect(); -+ let ids: Vec<_> = batches -+ .iter() -+ .flat_map(|b| { -+ b.column_by_name("id") -+ .unwrap() -+ .as_any() -+ .downcast_ref::() -+ .unwrap() -+ .values() -+ .to_vec() -+ }) -+ .collect(); -+ let expected_ids: Vec<_> = -+ (1..=6).skip(offset as usize).take(limit as usize).collect(); -+ let expected_sizes: Vec<_> = expected_ids -+ .chunks(batch_size.unwrap_or(8192) as usize) -+ .map(<[i32]>::len) -+ .collect(); -+ assert_eq!(ids, expected_ids); -+ assert_eq!( -+ sizes, expected_sizes, -+ "batch_size={batch_size:?} offset={offset} limit={limit}" -+ ); -+ } -+ } -+ lance_dataset_close(ds); -+ } -+} diff --git a/thirdparty/patches/lance-c-0.1.9-prefilter-fts.patch b/thirdparty/patches/lance-c-0.1.9-prefilter-fts.patch deleted file mode 100644 index 890093886fbcbb..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-prefilter-fts.patch +++ /dev/null @@ -1,148 +0,0 @@ -Subject: [PATCH] Include FTS prefilter metrics and full-snapshot test coverage - -Upstream: https://github.com/lance-format/lance/pull/9460 -Commit: f202fe41ac18323ca0cd7bc6efd8fd30b8722116 - -Backport the applicable review updates from Lance PR #9471 to v11. -Keep the C API and other dependencies unchanged, and upgrade previously -patched source caches with the same revision as fresh builds. - -diff --git a/Cargo.toml b/Cargo.toml ---- a/Cargo.toml -+++ b/Cargo.toml -@@ -20,10 +20,10 @@ - [dependencies] --lance = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6", features = ["substrait"] } --lance-core = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } --lance-file = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } --lance-index = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } --lance-io = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } --lance-linalg = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } --lance-table = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } --lance-datafusion = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6", features = ["substrait"] } -+lance = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116", features = ["substrait"] } -+lance-core = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } -+lance-file = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } -+lance-index = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } -+lance-io = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } -+lance-linalg = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } -+lance-table = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } -+lance-datafusion = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116", features = ["substrait"] } - datafusion = { version = "54.0.0", default-features = false } -@@ -53,5 +53,5 @@ - [dev-dependencies] --lance = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6", features = ["substrait"] } --lance-datagen = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } --lance-file = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116", features = ["substrait"] } -+lance-datagen = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } -+lance-file = { git = "https://github.com/lance-format/lance.git", rev = "f202fe41ac18323ca0cd7bc6efd8fd30b8722116" } - tokio = { version = "1", features = ["rt-multi-thread", "macros"] } -diff --git a/Cargo.lock b/Cargo.lock ---- a/Cargo.lock -+++ b/Cargo.lock -@@ -2608,3 +2608,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -3820,3 +3820,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -3892,3 +3892,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -3914,3 +3914,3 @@ - version = "58.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -3928,3 +3928,3 @@ - version = "58.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -3938,3 +3938,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -3984,3 +3984,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4022,3 +4022,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4054,3 +4054,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4072,3 +4072,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4082,3 +4082,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4116,3 +4116,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4148,3 +4148,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4163,3 +4163,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4231,3 +4231,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4254,3 +4254,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4294,3 +4294,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4309,3 +4309,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4336,3 +4336,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4351,3 +4351,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ -@@ -4390,3 +4390,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" -+source = "git+https://github.com/lance-format/lance.git?rev=f202fe41ac18323ca0cd7bc6efd8fd30b8722116#f202fe41ac18323ca0cd7bc6efd8fd30b8722116" - dependencies = [ diff --git a/thirdparty/patches/lance-c-0.1.9-prefilter.patch b/thirdparty/patches/lance-c-0.1.9-prefilter.patch deleted file mode 100644 index c3df2f5d26244d..00000000000000 --- a/thirdparty/patches/lance-c-0.1.9-prefilter.patch +++ /dev/null @@ -1,147 +0,0 @@ -Subject: [PATCH] Use Lance full-snapshot prefilter optimization and loader metrics - -Upstream: https://github.com/lance-format/lance/pull/9460 -Commit: f75f3343b5e125c42da1bd7acc8d8217cd5660a6 - -Pin the tested Lance v11 change without upgrading the release or changing -the C API. Keep all Lance crates on the same revision. - -diff --git a/Cargo.toml b/Cargo.toml ---- a/Cargo.toml -+++ b/Cargo.toml -@@ -20,10 +20,10 @@ - [dependencies] --lance = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe", features = ["substrait"] } --lance-core = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } --lance-file = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } --lance-index = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } --lance-io = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } --lance-linalg = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } --lance-table = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } --lance-datafusion = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe", features = ["substrait"] } -+lance = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6", features = ["substrait"] } -+lance-core = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance-file = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance-index = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance-io = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance-linalg = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance-table = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance-datafusion = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6", features = ["substrait"] } - datafusion = { version = "54.0.0", default-features = false } -@@ -53,5 +53,5 @@ - [dev-dependencies] --lance = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe", features = ["substrait"] } --lance-datagen = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } --lance-file = { git = "https://github.com/lance-format/lance.git", rev = "ab6b5bbe" } -+lance = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6", features = ["substrait"] } -+lance-datagen = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } -+lance-file = { git = "https://github.com/lance-format/lance.git", rev = "f75f3343b5e125c42da1bd7acc8d8217cd5660a6" } - tokio = { version = "1", features = ["rt-multi-thread", "macros"] } -diff --git a/Cargo.lock b/Cargo.lock ---- a/Cargo.lock -+++ b/Cargo.lock -@@ -2608,3 +2608,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -3820,3 +3820,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -3892,3 +3892,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -3914,3 +3914,3 @@ - version = "58.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -3928,3 +3928,3 @@ - version = "58.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -3938,3 +3938,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -3984,3 +3984,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4022,3 +4022,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4054,3 +4054,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4072,3 +4072,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4082,3 +4082,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4116,3 +4116,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4148,3 +4148,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4163,3 +4163,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4231,3 +4231,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4254,3 +4254,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4294,3 +4294,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4309,3 +4309,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4336,3 +4336,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4351,3 +4351,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ -@@ -4390,3 +4390,3 @@ - version = "11.0.0" --source = "git+https://github.com/lance-format/lance.git?rev=ab6b5bbe#ab6b5bbe46009ed78746b444df8db59a8bc5d842" -+source = "git+https://github.com/lance-format/lance.git?rev=f75f3343b5e125c42da1bd7acc8d8217cd5660a6#f75f3343b5e125c42da1bd7acc8d8217cd5660a6" - dependencies = [ diff --git a/thirdparty/patches/lance-c-0.1.9-pr-73.patch b/thirdparty/patches/lance-c-foyer.patch similarity index 88% rename from thirdparty/patches/lance-c-0.1.9-pr-73.patch rename to thirdparty/patches/lance-c-foyer.patch index 33a8985b25aecc..426b9257f13d78 100644 --- a/thirdparty/patches/lance-c-0.1.9-pr-73.patch +++ b/thirdparty/patches/lance-c-foyer.patch @@ -1,49 +1,13 @@ -From cc373477a6485cb24ed7ea88ef506e93caacedd0 Mon Sep 17 00:00:00 2001 -From: zhangstar333 -Date: Thu, 3 Sep 2026 13:00:43 +0800 -Subject: [PATCH] foyer - ---- - Cargo.lock | 207 +++++- - Cargo.toml | 4 + - README.md | 22 + - include/lance/lance.h | 63 ++ - include/lance/lance.hpp | 29 + - src/data_cache.rs | 68 ++ - src/dataset.rs | 11 + - src/foyer_data_cache.rs | 1215 ++++++++++++++++++++++++++++++++++++ - src/lib.rs | 4 + - src/restore.rs | 8 + - src/session.rs | 13 +- - src/writer.rs | 1 + - tests/c_api_test.rs | 221 +++++++ - tests/cpp/test_c_api.c | 32 + - tests/cpp/test_cpp_api.cpp | 23 + - 15 files changed, 1917 insertions(+), 4 deletions(-) - create mode 100644 src/data_cache.rs - create mode 100644 src/foyer_data_cache.rs - +# Retained Foyer data-cache integration from lance-format/lance-c#73. +# Rebased onto the pinned upstream lance-c revision; search fixes live upstream. diff --git a/Cargo.lock b/Cargo.lock -index bc37cb9..199f997 100644 --- a/Cargo.lock +++ b/Cargo.lock -@@ -444,6 +444,12 @@ dependencies = [ - "slab", - ] - -+[[package]] -+name = "asyncband" -+version = "0.7.1" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "2e52766975a4f080528a898235c51e82e65df9db713419067b5368040eeb5659" -+ - [[package]] - name = "atoi" - version = "2.0.0" -@@ -1293,6 +1299,17 @@ version = "0.8.7" +@@ -1192,6 +1192,17 @@ + version = "0.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" - ++ +[[package]] +name = "core_affinity" +version = "0.8.3" @@ -54,28 +18,13 @@ index bc37cb9..199f997 100644 + "num_cpus", + "winapi", +] -+ + [[package]] name = "countio" - version = "0.3.0" -@@ -2148,6 +2165,12 @@ dependencies = [ - "url", - ] - -+[[package]] -+name = "datasketches" -+version = "0.3.0" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "46c4cf71a36b46dcfc00e5014c0c20ccad2b1b6a008304d7d57d2749b2d41b3d" -+ - [[package]] - name = "defmt" - version = "1.1.1" -@@ -2366,6 +2389,16 @@ version = "0.2.3" - source = "registry+https://github.com/rust-lang/crates.io-index" +@@ -2242,6 +2253,16 @@ checksum = "f8eb564c5c7423d25c886fb561d1e4ee69f72354d16918afa32c08811f6b6a55" - -+[[package]] + + [[package]] +name = "fastant" +version = "0.1.11" +source = "registry+https://github.com/rust-lang/crates.io-index" @@ -85,13 +34,16 @@ index bc37cb9..199f997 100644 + "web-time", +] + - [[package]] ++[[package]] name = "fastrand" version = "2.3.0" -@@ -2437,12 +2470,133 @@ dependencies = [ + source = "registry+https://github.com/rust-lang/crates.io-index" +@@ -2310,6 +2331,26 @@ + checksum = "cb4cb245038516f5f85277875cdaa4f7d2c9a0fa0468de06ed190163b1581fcf" + dependencies = [ "percent-encoding", - ] - ++] ++ +[[package]] +name = "foyer" +version = "0.22.5" @@ -99,7 +51,7 @@ index bc37cb9..199f997 100644 +checksum = "f911e6f0b4909f23d65a95c5d27bcf2f92855b8a0f629b47b743d53d14f2828b" +dependencies = [ + "anyhow", -+ "asyncband 0.7.1", ++ "asyncband", + "equivalent", + "foyer-common", + "foyer-memory", @@ -110,67 +62,21 @@ index bc37cb9..199f997 100644 + "pin-project", + "serde", + "tracing", -+] -+ -+[[package]] -+name = "foyer-common" -+version = "0.22.5" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "05cdcae6cedec72c28e97ada0b453e33b1df57fbc209708091107be2a283d577" -+dependencies = [ -+ "anyhow", -+ "bytes", -+ "cfg-if 1.0.4", -+ "foyer-tokio", -+ "mixtrics", -+ "parking_lot", -+ "pin-project", -+ "twox-hash", -+] -+ -+[[package]] -+name = "foyer-intrusive-collections" -+version = "0.10.0-dev" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "6e4fee46bea69e0596130e3210e65d3424e0ac1e6df3bde6636304bdf1ca4a3b" -+dependencies = [ -+ "memoffset", -+] -+ -+[[package]] -+name = "foyer-memory" -+version = "0.22.5" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "8a51c8ce8e1e323a1e087ac45bf629d953676fc5e0037d14a799fadd272b42db" -+dependencies = [ -+ "anyhow", -+ "asyncband 0.7.1", -+ "bitflags 2.11.0", -+ "datasketches", -+ "equivalent", -+ "foyer-common", -+ "foyer-intrusive-collections", -+ "foyer-tokio", -+ "futures-util", -+ "hashbrown 0.17.1", -+ "itertools 0.15.0", -+ "mixtrics", -+ "parking_lot", -+ "paste", -+ "pin-project", -+ "serde", -+ "tracing", -+] -+ -+[[package]] + ] + + [[package]] +@@ -2363,6 +2404,38 @@ + ] + + [[package]] +name = "foyer-storage" -+version = "0.22.5" ++version = "0.22.6" +source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "127a64057f63e361123cf62b3fd50a36147783687d4cd36e7087c27f9909265b" ++checksum = "118192f532ba013f0cdc607efc22324b9ed855baed185608af0daa986e835c6b" +dependencies = [ + "allocator-api2", + "anyhow", -+ "asyncband 0.7.1", ++ "asyncband", + "bytes", + "core_affinity", + "equivalent", @@ -195,20 +101,14 @@ index bc37cb9..199f997 100644 +] + +[[package]] -+name = "foyer-tokio" -+version = "0.22.5" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "cbdbb9f39443cb348a069baa1a0ec73bcea848a4a383eb2da1a5ea7a0ca05941" -+dependencies = [ -+ "tokio", -+] -+ - [[package]] - name = "frostem" + name = "foyer-tokio" + version = "0.22.6" + source = "registry+https://github.com/rust-lang/crates.io-index" +@@ -2376,6 +2449,16 @@ version = "1.20260821.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "36a80a7406da302e04bfd2ca987907590d3a1f3c69958947c43890abd7426b2f" - ++ +[[package]] +name = "fs4" +version = "0.13.1" @@ -218,31 +118,13 @@ index bc37cb9..199f997 100644 + "rustix", + "windows-sys 0.59.0", +] -+ + [[package]] name = "fs_extra" - version = "1.3.0" -@@ -3485,6 +3639,15 @@ dependencies = [ - "either", - ] - -+[[package]] -+name = "itertools" -+version = "0.15.0" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "8b4baf93f58d4425749ca49a51c50ebab072c5df6994d08fed93541c331481dc" -+dependencies = [ -+ "either", -+] -+ - [[package]] - name = "itoa" - version = "1.0.18" -@@ -3788,8 +3951,11 @@ dependencies = [ - "arrow", +@@ -3720,8 +3803,10 @@ "arrow-array", "arrow-schema", -+ "async-trait", + "async-trait", + "bytes", "chrono", "datafusion", @@ -250,149 +132,92 @@ index bc37cb9..199f997 100644 "futures", "half", "lance", -@@ -3803,6 +3969,7 @@ dependencies = [ +@@ -3735,6 +3820,7 @@ "lance-table", "libc", "log", -+ "object_store", ++ "object_store 0.14.2", + "opendal", "pin-project", "prost", - "snafu", -@@ -4492,6 +4659,15 @@ dependencies = [ - "libc", - ] - -+[[package]] -+name = "memoffset" -+version = "0.9.1" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "488016bfae457b036d996092f6cb448677611ce4449e970ceaf42695203f218a" -+dependencies = [ -+ "autocfg", -+] -+ - [[package]] - name = "mime" - version = "0.3.17" -@@ -4529,6 +4705,16 @@ dependencies = [ - "windows-sys 0.61.2", - ] - -+[[package]] -+name = "mixtrics" -+version = "0.2.5" -+source = "registry+https://github.com/rust-lang/crates.io-index" -+checksum = "2c46b5adfb7a3ae4996d327a5bdc90e78fec025806dd312bdbe6f07a755e0ec9" -+dependencies = [ -+ "itertools 0.15.0", -+ "parking_lot", -+] -+ - [[package]] - name = "moka" - version = "0.12.15" -@@ -4840,7 +5026,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "48dbcef97d3eb7591db2c18d5cae95c836bcce07359b98d98dd6f4e861eb77b7" - dependencies = [ - "anyhow", -- "asyncband", -+ "asyncband 0.6.7", - "base64 0.23.1", - "bytes", - "futures", -@@ -4879,7 +5065,7 @@ version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "03f9e144b5228d741c3763ade8711d9b72e5fb6d998e779f2d7a09da0b5a3eba" - dependencies = [ -- "asyncband", -+ "asyncband 0.6.7", - "futures", - "http 1.4.0", - "opendal-core", -@@ -4943,7 +5129,7 @@ version = "0.58.2" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "2d564484a8f7d091827e825cfc91ed45bd48e64d262451ee041fe843db81bd8a" - dependencies = [ -- "asyncband", -+ "asyncband 0.6.7", - "base64 0.23.1", - "bytes", - "http 1.4.0", -@@ -6668,6 +6854,12 @@ version = "0.4.12" - source = "registry+https://github.com/rust-lang/crates.io-index" +@@ -6629,6 +6715,12 @@ checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" - -+[[package]] + + [[package]] +name = "small_ctor" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "88414a5ca1f85d82cc34471e975f0f74f6aa54c40f062efa42c0080e7f763f81" + - [[package]] ++[[package]] name = "smallvec" version = "1.15.1" -@@ -7930,6 +8122,15 @@ dependencies = [ - "windows-targets 0.52.6", - ] - + source = "registry+https://github.com/rust-lang/crates.io-index" +@@ -7893,6 +7985,15 @@ + version = "0.52.0" + source = "registry+https://github.com/rust-lang/crates.io-index" + checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" ++dependencies = [ ++ "windows-targets 0.52.6", ++] ++ +[[package]] +name = "windows-sys" +version = "0.59.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1e38bc4d79ed67fd075bcc251a1c39b32a1776bbe92e5bef1f0bf1f8c531853b" -+dependencies = [ -+ "windows-targets 0.52.6", -+] -+ - [[package]] - name = "windows-sys" - version = "0.60.2" + dependencies = [ + "windows-targets 0.52.6", + ] diff --git a/Cargo.toml b/Cargo.toml -index 3920a65..356c9dd 100644 --- a/Cargo.toml +++ b/Cargo.toml -@@ -30,6 +30,8 @@ datafusion = { version = "54.0.0", default-features = false } +@@ -35,6 +35,7 @@ arrow = { version = "58.0.0", features = ["prettyprint", "ffi"] } arrow-array = "58.0.0" arrow-schema = "58.0.0" -+async-trait = "0.1" +bytes = "1" # Direct to name `chrono::TimeDelta` (the field type of lance's public # `AutoCleanupParams`) and `chrono::DateTime`/`Utc` (index metadata # timestamps); already in the graph transitively via lance. -@@ -37,8 +39,10 @@ chrono = { version = "0.4", default-features = false } +@@ -42,10 +43,12 @@ half = "2" tokio = { version = "1", features = ["rt-multi-thread", "sync"] } futures = "0.3" +foyer = "=0.22.5" log = "0.4" libc = "0.2" -+object_store = "0.13.2" + # Explicitly install the HTTP transport when embedded in a static C/C++ executable. + opendal = { version = "=0.59.2", default-features = false, features = ["http-transport-reqwest"] } ++object_store = "0.14.1" pin-project = "1.0" prost = "0.14" snafu = "0.9" diff --git a/README.md b/README.md -index 2056671..d9a5f2b 100644 --- a/README.md +++ b/README.md -@@ -68,6 +68,7 @@ Based on the [liblance RFC](https://github.com/lance-format/lance/discussions/60 +@@ -68,6 +68,7 @@ | [x] | Async scan | Callback-based `lance_scanner_scan_async()` for non-blocking scans | | [x] | Dataset metadata | `lance_dataset_version()`, `lance_dataset_count_rows()`, `lance_dataset_latest_version()` | | [x] | Filter pushdown | `lance_scanner_set_substrait_filter()` accepts a serialized Substrait `ExtendedExpression`; `lance_scanner_additional_sql_filter()` adds SQL predicates with AND before scanning starts | +| [x] | Data-file cache | Optional Foyer memory/disk cache for immutable `data/*.lance` reads | - - ## Building - -@@ -197,6 +198,27 @@ auto ds = lance::Dataset::open_with_session(session, "data.lance"); + + ## Multi-vector search + +@@ -229,6 +230,29 @@ + 1ULL * 1024 * 1024 * 1024); + auto ds = lance::Dataset::open_with_session(session, "data.lance"); auto stats = session.cache_stats(); - ``` - ++``` ++ +To add a process-local memory/disk cache for remote Lance data-file reads, +create the session with Foyer configuration. The cache is deliberately narrow: +whole-object, single-range, and batched range reads of direct `data/*.lance` +children are cached. Conditional and versioned reads, plus manifests, deletion +files, and index files, keep using Lance's normal paths. Use one shared session -+for datasets that share the cache directory. ++for datasets that share the cache directory. Immutable data-file response metadata ++uses a separate in-memory cache budget of one eighth of the configured data ++memory capacity, avoiding a remote HEAD request on repeated range reads. + +```cpp +lance::DataCacheOptions data_cache{ @@ -406,20 +231,16 @@ index 2056671..d9a5f2b 100644 + 1ULL * 1024 * 1024 * 1024, + data_cache); +auto ds = lance::Dataset::open_with_session(session, "s3://bucket/data.lance"); -+``` -+ + ``` + ### Open at a specific version - - `lance_dataset_open` takes a `version` argument — `0` means the latest, any diff --git a/include/lance/lance.h b/include/lance/lance.h -index 31213da..152351b 100644 --- a/include/lance/lance.h +++ b/include/lance/lance.h -@@ -206,6 +206,36 @@ typedef struct LanceSessionCacheStats { - uint64_t metadata_cache_size_bytes; +@@ -215,6 +215,36 @@ } LanceSessionCacheStats; - -+/** + + /** + * Configuration for the optional Foyer cache of immutable Lance data files. + * + * Whole-object, single-range, and batched range reads of direct @@ -449,13 +270,16 @@ index 31213da..152351b 100644 + uint64_t bytes_read_from_remote; +} LanceDataCacheStatistics; + - /** ++/** * Create a session that can share metadata and index caches across datasets. * -@@ -217,6 +247,25 @@ LanceSession* lance_session_new( + * Cache limits are specified in bytes. Pass 0 to request zero capacity. +@@ -223,6 +253,25 @@ + LanceSession* lance_session_new( + uint64_t index_cache_size_bytes, uint64_t metadata_cache_size_bytes - ); - ++); ++ +/** + * Create a shared Lance session with a Foyer data-file cache. + * @@ -473,15 +297,15 @@ index 31213da..152351b 100644 + uint64_t index_cache_size_bytes, + uint64_t metadata_cache_size_bytes, + const LanceDataCacheOptions* data_cache_options -+); -+ + ); + /** - * Close a session handle. Safe to call with NULL. Datasets previously opened - * with the session remain valid and retain the shared cache state. -@@ -273,6 +322,20 @@ LanceDataset* lance_dataset_open_with_session( +@@ -279,6 +328,20 @@ + const char* const* storage_opts, + uint64_t version, const LanceSession* session - ); - ++); ++ +/** + * Copy this dataset handle's cumulative data-cache statistics. + * @@ -494,19 +318,16 @@ index 31213da..152351b 100644 +int32_t lance_dataset_get_data_cache_statistics( + const LanceDataset* dataset, + LanceDataCacheStatistics* out_statistics -+); -+ + ); + /** Close and free a dataset handle. Safe to call with NULL. */ - void lance_dataset_close(LanceDataset* dataset); - diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp -index 404d2df..3fe9dbc 100644 --- a/include/lance/lance.hpp +++ b/include/lance/lance.hpp -@@ -176,6 +176,13 @@ struct SqlColumn { - +@@ -176,12 +176,34 @@ + // ─── Shared Session ────────────────────────────────────────────────────────── - + +struct DataCacheOptions { + std::string directory; + uint64_t memory_capacity_bytes; @@ -516,11 +337,13 @@ index 404d2df..3fe9dbc 100644 + class Session { Handle handle_; - -@@ -185,6 +192,21 @@ class Session { - if (!handle_) check_error(); - } - + + public: + Session(uint64_t index_cache_size_bytes, uint64_t metadata_cache_size_bytes) + : handle_(lance_session_new(index_cache_size_bytes, metadata_cache_size_bytes)) { ++ if (!handle_) check_error(); ++ } ++ + Session(uint64_t index_cache_size_bytes, + uint64_t metadata_cache_size_bytes, + const DataCacheOptions& data_cache_options) { @@ -533,29 +356,24 @@ index 404d2df..3fe9dbc 100644 + handle_ = Handle( + lance_session_new_with_data_cache( + index_cache_size_bytes, metadata_cache_size_bytes, &options)); -+ if (!handle_) check_error(); + if (!handle_) check_error(); + } + +@@ -348,6 +370,13 @@ + uri.c_str(), opts_ptr, version, session.c_handle()); + if (!ds) check_error(); + return Dataset(ds); + } + - LanceSessionCacheStats cache_stats() const { - LanceSessionCacheStats stats{}; - if (lance_session_get_cache_stats(handle_.get(), &stats) != 0) -@@ -265,6 +287,13 @@ class Dataset { - return Dataset(ds); - } - + LanceDataCacheStatistics data_cache_statistics() const { + LanceDataCacheStatistics statistics{}; + if (lance_dataset_get_data_cache_statistics(handle_.get(), &statistics) != 0) + check_error(); + return statistics; -+ } -+ + } + /// Write an Arrow record batch stream to a Lance dataset and return the - /// open dataset at the committed version. - /// diff --git a/src/data_cache.rs b/src/data_cache.rs -new file mode 100644 -index 0000000..430b4c8 --- /dev/null +++ b/src/data_cache.rs @@ -0,0 +1,68 @@ @@ -628,28 +446,27 @@ index 0000000..430b4c8 + Ok(0) +} diff --git a/src/dataset.rs b/src/dataset.rs -index cc1f87c..76fd39e 100644 --- a/src/dataset.rs +++ b/src/dataset.rs -@@ -14,6 +14,7 @@ use lance::Dataset; +@@ -14,6 +14,7 @@ use lance::dataset::builder::DatasetBuilder; use lance_core::Result; - + +use crate::data_cache::DatasetDataCache; use crate::error::{ffi_try, swallow_unwind}; use crate::helpers; use crate::runtime::block_on; -@@ -23,6 +24,7 @@ use crate::stream_guard::guarded_ffi_stream_from_reader; +@@ -23,6 +24,7 @@ /// Opaque handle representing an opened Lance dataset. pub struct LanceDataset { pub(crate) inner: RwLock>, + pub(crate) data_cache: Option>, } - + impl LanceDataset { -@@ -182,8 +184,16 @@ unsafe fn open_dataset_inner( +@@ -182,8 +184,16 @@ } - + let dataset = block_on(builder.load())?; + let (dataset, data_cache) = + if let Some(factory) = session.and_then(|session| session.data_cache_factory.clone()) { @@ -664,7 +481,7 @@ index cc1f87c..76fd39e 100644 }; Ok(Box::into_raw(Box::new(handle))) } -@@ -519,6 +529,7 @@ mod tests { +@@ -519,6 +529,7 @@ .unwrap(); let handle = LanceDataset { inner: RwLock::new(Arc::new(dataset)), @@ -673,11 +490,9 @@ index cc1f87c..76fd39e 100644 (tmp, handle) } diff --git a/src/foyer_data_cache.rs b/src/foyer_data_cache.rs -new file mode 100644 -index 0000000..9a10e43 --- /dev/null +++ b/src/foyer_data_cache.rs -@@ -0,0 +1,1215 @@ +@@ -0,0 +1,1268 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + @@ -694,16 +509,17 @@ index 0000000..9a10e43 +use async_trait::async_trait; +use bytes::{Bytes, BytesMut}; +use foyer::{ -+ BlockEngineConfig, DeviceBuilder, FsDeviceBuilder, HybridCache, HybridCacheBuilder, -+ HybridCachePolicy, PsyncIoEngineConfig, ++ BlockEngineConfig, Cache, CacheBuilder, DeviceBuilder, FsDeviceBuilder, HybridCache, ++ HybridCacheBuilder, HybridCachePolicy, PsyncIoEngineConfig, +}; +use futures::stream::BoxStream; +use lance_io::object_store::WrappingObjectStore; ++use object_store::list::PaginatedListStore; +use object_store::path::Path; +use object_store::{ -+ CopyOptions, GetOptions, GetResult, GetResultPayload, ListResult, MultipartUpload, ObjectMeta, -+ ObjectStore, ObjectStoreExt, PutMultipartOptions, PutOptions, PutPayload, PutResult, -+ RenameOptions, Result, ++ Attribute, Attributes, CopyOptions, GetOptions, GetResult, GetResultPayload, ListResult, ++ MultipartUpload, ObjectMeta, ObjectStore, ObjectStoreExt, PutMultipartOptions, PutOptions, ++ PutPayload, PutResult, RenameOptions, Result, +}; + +use crate::data_cache::{DataCacheFactory, DatasetDataCache, LanceDataCacheStatistics}; @@ -846,6 +662,7 @@ index 0000000..9a10e43 +#[derive(Clone)] +pub(crate) struct FoyerDataCache { + cache: HybridCache, ++ metadata: Cache, + read_block_size: usize, + wrapped_stores: Arc>>, +} @@ -879,6 +696,33 @@ index 0000000..9a10e43 + .with_capacity(disk_capacity) + .build()?; + let engine = BlockEngineConfig::new(device).with_block_size(engine_block_size); ++ // Single-range reads use get_opts(), which must return object metadata. ++ // Share a bounded metadata cache across dataset scopes so a data-cache hit ++ // does not require another HEAD request for an immutable data file. ++ let metadata_capacity = memory_capacity / 8; ++ let metadata = CacheBuilder::new(metadata_capacity) ++ .with_shards(1) ++ .with_weighter( ++ |key: &String, (meta, attributes): &(ObjectMeta, Attributes)| { ++ key.len() ++ + std::mem::size_of::<(ObjectMeta, Attributes)>() ++ + meta.location.as_ref().len() ++ + meta.e_tag.as_ref().map_or(0, String::len) ++ + meta.version.as_ref().map_or(0, String::len) ++ + attributes ++ .iter() ++ .map(|(attribute, value)| { ++ std::mem::size_of::<(Attribute, object_store::AttributeValue)>() ++ + value.as_ref().len() ++ + match attribute { ++ Attribute::Metadata(name) => name.len(), ++ _ => 0, ++ } ++ }) ++ .sum::() ++ }, ++ ) ++ .build(); + let cache = HybridCacheBuilder::new() + .with_name("lance_data") + .with_policy(HybridCachePolicy::WriteOnInsertion) @@ -895,6 +739,7 @@ index 0000000..9a10e43 + .await?; + Ok(Self { + cache, ++ metadata, + read_block_size, + wrapped_stores: Arc::new(Mutex::new(HashMap::new())), + }) @@ -1047,6 +892,15 @@ index 0000000..9a10e43 + self.cache.remember_wrapper(&wrapped, &original); + wrapped + } ++ ++ fn wrap_paginated( ++ &self, ++ _store_prefix: &str, ++ original: Arc, ++ ) -> Option> { ++ // Data caching does not hide or rewrite paths, so keep listing pushdown. ++ Some(original) ++ } +} + +#[derive(Debug)] @@ -1087,24 +941,35 @@ index 0000000..9a10e43 + } + + async fn cached_get(&self, location: &Path, options: GetOptions) -> Result { -+ // Fetch metadata separately so the returned GetResult retains the origin's identity while -+ // its payload uses the same block cache as get_ranges(). This also provides the object size -+ // needed to resolve bounded, offset, and suffix ranges. -+ let GetResult { -+ meta: metadata, -+ attributes, -+ .. -+ } = self ++ let metadata_key = self + .reader -+ .original -+ .get_opts( -+ location, -+ GetOptions { -+ head: true, -+ ..Default::default() -+ }, -+ ) -+ .await?; ++ .cache ++ .size_key(&self.reader.store_prefix, location); ++ let cached_metadata = self.reader.cache.metadata.get(&metadata_key); ++ let (metadata, attributes, extensions) = if let Some(entry) = cached_metadata { ++ let (metadata, attributes) = entry.value(); ++ (metadata.clone(), attributes.clone(), Default::default()) ++ } else { ++ let result = self ++ .reader ++ .original ++ .get_opts( ++ location, ++ GetOptions { ++ head: true, ++ ..Default::default() ++ }, ++ ) ++ .await?; ++ // Extensions may carry request-specific state and cannot be replayed. ++ if result.extensions.is_empty() { ++ self.reader.cache.metadata.insert( ++ metadata_key, ++ (result.meta.clone(), result.attributes.clone()), ++ ); ++ } ++ (result.meta, result.attributes, result.extensions) ++ }; + let object_size = metadata.size; + self.reader.cache.cache.insert( + self.reader @@ -1157,6 +1022,7 @@ index 0000000..9a10e43 + meta: metadata, + range, + attributes, ++ extensions, + }) + } +} @@ -1674,7 +1540,7 @@ index 0000000..9a10e43 + original.put(path, data.clone().into()).await.unwrap(); + } + -+ let (wrapped, statistics) = wrap_for_test(&cache, original); ++ let (wrapped, statistics) = wrap_for_test(&cache, original.clone()); + for (path, requested, expected_range) in cases { + let expected = data.slice(expected_range.start as usize..expected_range.end as usize); + let before = statistics.snapshot(); @@ -1691,14 +1557,16 @@ index 0000000..9a10e43 + before.bytes_read_from_remote + expected.len() as u64 + ); + ++ // Cached data reads must not depend on an extra origin HEAD request. ++ let expected_meta = original.head(&path).await.unwrap(); ++ original.delete(&path).await.unwrap(); + let second = wrapped + .get_opts(&path, GetOptions::new().with_range(Some(requested))) + .await -+ .unwrap() -+ .bytes() -+ .await + .unwrap(); -+ assert_eq!(second, expected); ++ assert_eq!(second.meta, expected_meta); ++ assert_eq!(second.range, expected_range); ++ assert_eq!(second.bytes().await.unwrap(), expected); + assert_eq!( + statistics.snapshot().bytes_read_from_cache, + before.bytes_read_from_cache + expected.len() as u64 @@ -1894,12 +1762,11 @@ index 0000000..9a10e43 + } +} diff --git a/src/lib.rs b/src/lib.rs -index 8b212f5..923528a 100644 --- a/src/lib.rs +++ b/src/lib.rs -@@ -25,11 +25,13 @@ mod alter_columns; - mod async_dispatcher; +@@ -26,11 +26,13 @@ mod batch; + mod blob; mod compact; +mod data_cache; mod data_statistics; @@ -1911,15 +1778,15 @@ index 8b212f5..923528a 100644 mod fragment_writer; mod fts_query; mod helpers; -@@ -51,6 +53,7 @@ pub use add_columns::*; - pub use alter_columns::*; +@@ -55,6 +57,7 @@ pub use batch::*; + pub use blob::*; pub use compact::*; +pub use data_cache::{LanceDataCacheStatistics, lance_dataset_get_data_cache_statistics}; pub use data_statistics::*; pub use dataset::*; pub use delete::*; -@@ -58,6 +61,7 @@ pub use drop_columns::*; +@@ -62,6 +65,7 @@ pub use error::{ LanceErrorCode, lance_free_string, lance_last_error_code, lance_last_error_message, }; @@ -1928,13 +1795,12 @@ index 8b212f5..923528a 100644 pub use fts_query::*; pub use index::*; diff --git a/src/restore.rs b/src/restore.rs -index 7804b55..fa2d26d 100644 --- a/src/restore.rs +++ b/src/restore.rs -@@ -65,8 +65,16 @@ unsafe fn restore_inner(dataset: *const LanceDataset, version: u64) -> Result<*m +@@ -65,8 +65,16 @@ Ok::<_, lance_core::Error>(checked_out) })?; - + + let (restored, data_cache) = if let Some(data_cache) = &ds.data_cache { + let (restored, data_cache) = data_cache.attach_fresh(restored); + (restored, Some(data_cache)) @@ -1949,30 +1815,28 @@ index 7804b55..fa2d26d 100644 Ok(Box::into_raw(Box::new(handle))) } diff --git a/src/session.rs b/src/session.rs -index 60a1623..9ed8cfe 100644 --- a/src/session.rs +++ b/src/session.rs -@@ -8,12 +8,14 @@ use std::sync::Arc; +@@ -8,12 +8,14 @@ use lance::session::Session; use lance_core::Result; - + +use crate::data_cache::DataCacheFactory; use crate::error::{ffi_try, swallow_unwind}; use crate::runtime::block_on; - + -/// Opaque handle for sharing Lance metadata and index caches across datasets. +/// Opaque handle for shared Lance caches across datasets. pub struct LanceSession { pub(crate) inner: Arc, + pub(crate) data_cache_factory: Option>, } - + /// Snapshot of a session's metadata and index cache statistics. -@@ -47,6 +49,14 @@ pub extern "C" fn lance_session_new( - fn session_new_inner( +@@ -48,6 +50,14 @@ index_cache_size_bytes: u64, metadata_cache_size_bytes: u64, -+) -> Result<*mut LanceSession> { + ) -> Result<*mut LanceSession> { + session_new_with_data_cache_factory(index_cache_size_bytes, metadata_cache_size_bytes, None) +} + @@ -1980,22 +1844,22 @@ index 60a1623..9ed8cfe 100644 + index_cache_size_bytes: u64, + metadata_cache_size_bytes: u64, + data_cache_factory: Option>, - ) -> Result<*mut LanceSession> { ++) -> Result<*mut LanceSession> { let index_cache_size_bytes = u64_to_usize(index_cache_size_bytes, "index_cache_size_bytes")?; let metadata_cache_size_bytes = -@@ -58,6 +68,7 @@ fn session_new_inner( + u64_to_usize(metadata_cache_size_bytes, "metadata_cache_size_bytes")?; +@@ -58,6 +68,7 @@ ); Ok(Box::into_raw(Box::new(LanceSession { inner: Arc::new(session), + data_cache_factory, }))) } - + diff --git a/src/writer.rs b/src/writer.rs -index 1971510..ba51c87 100644 --- a/src/writer.rs +++ b/src/writer.rs -@@ -282,6 +282,7 @@ unsafe fn write_dataset_inner( +@@ -282,6 +282,7 @@ if !out_dataset.is_null() { let handle = LanceDataset { inner: RwLock::new(Arc::new(dataset)), @@ -2004,13 +1868,12 @@ index 1971510..ba51c87 100644 // SAFETY: `out_dataset` is non-NULL (checked above) and the caller // guarantees it points to caller-owned, writable storage of size diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index bde742d..a4ea8f8 100644 --- a/tests/c_api_test.rs +++ b/tests/c_api_test.rs -@@ -96,10 +96,83 @@ fn create_large_dataset(num_rows: i32) -> (tempfile::TempDir, String) { +@@ -100,8 +100,81 @@ (tmp, uri) } - + +/// Helper: create two fragments large enough for Lance's batched range-read +/// path, which is the path wrapped by the Foyer data cache. +fn create_large_multi_fragment_dataset(num_rows_per_fragment: i32) -> (tempfile::TempDir, String) { @@ -2050,8 +1913,8 @@ index bde742d..a4ea8f8 100644 + fn c_str(s: &str) -> CString { CString::new(s).unwrap() - } - ++} ++ +fn file_object_store_uri(path: &str) -> CString { + let path = path.replace('\\', "/"); + let leading_slash = if path.starts_with('/') { "" } else { "/" }; @@ -2086,16 +1949,13 @@ index bde742d..a4ea8f8 100644 + .iter() + .map(RecordBatch::num_rows) + .sum() -+} -+ + } + #[derive(Default)] - struct CapturedScanStatistics { - calls: usize, -@@ -313,6 +386,120 @@ fn test_shared_session_rejects_null_inputs() { - } +@@ -446,6 +519,120 @@ } - -+#[test] + + #[test] +fn test_session_with_data_cache_serves_repeated_scan() { + let (tmp, uri) = create_large_multi_fragment_dataset(10_000); + let c_uri = file_object_store_uri(&uri); @@ -2209,13 +2069,16 @@ index bde742d..a4ea8f8 100644 + assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); +} + - #[test] ++#[test] fn test_open_nonexistent() { let c_uri = c_str("memory://nonexistent_dataset_xyz"); -@@ -2977,6 +3164,40 @@ fn test_dataset_restore_to_prior_version() { + let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; +@@ -3316,6 +3503,40 @@ + + unsafe { lance_dataset_close(restored) }; unsafe { lance_dataset_close(ds) }; - } - ++} ++ +#[test] +fn test_restored_handle_has_independent_data_cache_statistics() { + let (_tmp, uri) = create_large_multi_fragment_dataset(10_000); @@ -2248,19 +2111,18 @@ index bde742d..a4ea8f8 100644 + lance_dataset_close(restored); + lance_dataset_close(source); + } -+} -+ + } + #[test] - fn test_dataset_restore_to_current_latest_writes_new_manifest() { - // Restoring to the current latest still writes a new manifest. The diff --git a/tests/cpp/test_c_api.c b/tests/cpp/test_c_api.c -index c49ecfa..dd674eb 100644 --- a/tests/cpp/test_c_api.c +++ b/tests/cpp/test_c_api.c -@@ -126,6 +126,37 @@ static void test_shared_session(const char *uri) { +@@ -169,6 +169,37 @@ + lance_dataset_close(ds); + printf("metadata_entries=%llu... OK\n", (unsigned long long)stats.metadata_cache_entries); - } - ++} ++ +static void test_data_cache_session(const char *uri, const char *write_uri) { + printf(" test_data_cache_session... "); + @@ -2290,27 +2152,27 @@ index c49ecfa..dd674eb 100644 + "dataset should remain valid after data-cache session close"); + lance_dataset_close(ds); + printf("OK\n"); -+} -+ + } + static void test_scan(const char *uri) { - printf(" test_scan... "); - -@@ -973,6 +1004,7 @@ int main(int argc, char **argv) { - +@@ -1323,6 +1354,7 @@ + test_open_and_metadata(uri); test_shared_session(uri); + test_data_cache_session(uri, write_uri); test_scan(uri); test_scan_with_limit(uri); - test_versions(uri); + test_scanner_blob_handling(blob_uri); diff --git a/tests/cpp/test_cpp_api.cpp b/tests/cpp/test_cpp_api.cpp -index 28762b8..5f34bde 100644 --- a/tests/cpp/test_cpp_api.cpp +++ b/tests/cpp/test_cpp_api.cpp -@@ -99,6 +99,28 @@ static void test_shared_session(const std::string& uri) { - PASS(); - } - +@@ -128,6 +128,28 @@ + + printf("metadata_entries=%llu... ", + (unsigned long long)stats.metadata_cache_entries); ++ PASS(); ++} ++ +static void test_data_cache_session(const std::string& uri, + const std::string& write_uri) { + TEST(test_data_cache_session); @@ -2330,14 +2192,11 @@ index 28762b8..5f34bde 100644 + session.reset(); + assert(ds.count_rows() > 0); + -+ PASS(); -+} -+ - static void test_dataset_schema(const std::string& uri) { - TEST(test_dataset_schema); - -@@ -929,6 +951,7 @@ int main(int argc, char** argv) { - + PASS(); + } + +@@ -1211,6 +1233,7 @@ + test_dataset_open(uri); test_shared_session(uri); + test_data_cache_session(uri, write_uri); diff --git a/thirdparty/test/lance-prefilter-patch-test.sh b/thirdparty/test/lance-prefilter-patch-test.sh new file mode 100755 index 00000000000000..c22c3a2c7ded03 --- /dev/null +++ b/thirdparty/test/lance-prefilter-patch-test.sh @@ -0,0 +1,95 @@ +#!/usr/bin/env bash +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." &>/dev/null && pwd)" +ARCHIVE_DIR="${1:?Usage: $0 directory-containing-the-pinned-lance-c-archive}" +ARCHIVE_DIR="$(cd "${ARCHIVE_DIR}" && pwd)" +TP_DIR="${ROOT}" +# Load only repository-owned definitions, never extracted dependency code. +source "${ROOT}/vars.sh" +mkdir -p "${ROOT}/src" +tmpdir="$(mktemp -d "${ROOT}/src/lance-prefilter-test.XXXXXX")" +trap 'rm -rf "${tmpdir}"' EXIT + +fail() { + echo "FAIL: $*" >&2 + exit 1 +} + +prepare() { + local dest="$1" + mkdir -p "${dest}/src" + cp "${ROOT}/vars.sh" "${dest}/vars.sh" + ln -s "${ROOT}/patches" "${dest}/patches" + cp "${ARCHIVE_DIR}/${LANCE_C_NAME}" "${dest}/src/" +} + +run_download() { + TP_DIR="$1" DORIS_HOME="${tmpdir}" bash "${ROOT}/download-thirdparty.sh" lance_c +} + +check_sources() { + local source="$1/src/${LANCE_C_SOURCE}" + grep -q 'fn test_scanner_nearest_segment_prefilter_statistics' "${source}/tests/c_api_test.rs" \ + || fail "missing upstream segment-prefilter regression" + grep -q 'lance_session_new_with_data_cache' "${source}/src/foyer_data_cache.rs" \ + || fail "missing retained Foyer API" + grep -q 'lance_dataset_get_data_cache_statistics' "${source}/src/data_cache.rs" \ + || fail "missing retained cache statistics API" + grep -q 'source = "git+https://github.com/lance-format/lance.git' "${source}/Cargo.lock" \ + || fail "Lance must come from the upstream git dependency" + [[ -f "${source}/patched_mark_foyer" ]] || fail "missing Foyer patch marker" +} + +prepare "${tmpdir}/fresh" +run_download "${tmpdir}/fresh" +check_sources "${tmpdir}/fresh" +cp "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/Cargo.toml" "${tmpdir}/manifest" +cp "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/Cargo.lock" "${tmpdir}/lock" +run_download "${tmpdir}/fresh" +cmp "${tmpdir}/manifest" "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/Cargo.toml" +cmp "${tmpdir}/lock" "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/Cargo.lock" +echo "PASS: fresh archive and idempotent Foyer patch" + +# Re-extraction must apply Foyer again; no separate Lance source is required. +rm -rf "${tmpdir}/fresh/src/${LANCE_C_SOURCE}" +run_download "${tmpdir}/fresh" +check_sources "${tmpdir}/fresh" +cmp "${tmpdir}/lock" "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/Cargo.lock" +echo "PASS: re-extracted archive" + +prepare "${tmpdir}/cached" +tar xzf "${ARCHIVE_DIR}/${LANCE_C_NAME}" -C "${tmpdir}/cached/src" +# A generic marker from an earlier build must not suppress the Foyer patch. +touch "${tmpdir}/cached/src/${LANCE_C_SOURCE}/patched_mark" +run_download "${tmpdir}/cached" +check_sources "${tmpdir}/cached" +cmp "${tmpdir}/lock" "${tmpdir}/cached/src/${LANCE_C_SOURCE}/Cargo.lock" +echo "PASS: cached sources with an existing generic marker" + +prepare "${tmpdir}/invalid" +tar xzf "${ARCHIVE_DIR}/${LANCE_C_NAME}" -C "${tmpdir}/invalid/src" +printf '%s\n' 'incompatible manifest' > "${tmpdir}/invalid/src/${LANCE_C_SOURCE}/Cargo.toml" +if run_download "${tmpdir}/invalid" > "${tmpdir}/invalid.log" 2>&1; then + fail "expected an incompatible source to reject the patch" +fi +[[ ! -f "${tmpdir}/invalid/src/${LANCE_C_SOURCE}/patched_mark_foyer" ]] \ + || fail "failed patch was marked complete" +echo "PASS: patch failure does not mark sources ready" diff --git a/thirdparty/vars.sh b/thirdparty/vars.sh index f86c204d579350..98ca1a6f234aeb 100644 --- a/thirdparty/vars.sh +++ b/thirdparty/vars.sh @@ -580,10 +580,11 @@ PUGIXML_SOURCE=pugixml-1.15 PUGIXML_MD5SUM="3b894c29455eb33a40b165c6e2de5895" # lance-c -LANCE_C_DOWNLOAD="https://github.com/lance-format/lance-c/archive/refs/tags/v0.1.9.tar.gz" -LANCE_C_NAME="lance-c-v0.1.9.tar.gz" -LANCE_C_SOURCE="lance-c-0.1.9" -LANCE_C_MD5SUM="7138ed44e92d4bc91d5b522a6b92ed64" +# Complete-segment prefilter fixes are supplied by upstream lance-c, not local patches. +LANCE_C_DOWNLOAD="https://codeload.github.com/lance-format/lance-c/tar.gz/1bebabace62e3a9ed366f541003e534c92fc4248" +LANCE_C_NAME="lance-c-1bebabace62e3a9ed366f541003e534c92fc4248.tar.gz" +LANCE_C_SOURCE="lance-c-1bebabace62e3a9ed366f541003e534c92fc4248" +LANCE_C_MD5SUM="461b6430f8ab87607aea9f6697df17a6" # all thirdparties which need to be downloaded is set in array TP_ARCHIVES export TP_ARCHIVES=( From 4a86e23553670f3a630037bb85ecc020d42c0fb3 Mon Sep 17 00:00:00 2001 From: Gabriel Date: Tue, 29 Sep 2026 14:34:40 +0800 Subject: [PATCH 2/9] [test](lance) Sync upstream prefilter review regressions ### What problem does this PR solve? Related PR: lance-format/lance#9599, lance-format/lance#9537, lance-format/lance-c#89 Problem Summary: Pin the reviewed Lance prefilter documentation and expanded partial-coverage regression through upstream lance-c. Include the C API regression verifying identical 4-bit PQ candidates and distances with and without an all-row prefilter. The dependency already contains the final merged exact FastScan scoring fix. Refresh the immutable archive checksum; the retained Foyer patch is unchanged. ### Release note None ### Check List (For Author) - Test: 138 Lance prefilter tests, 62 PQ tests, and 443 lance-c Rust tests passed. Fresh/idempotent/re-extracted/invalid-patch downloader regressions passed on both Doris branches; shell syntax and diff checks passed. - Behavior changed: No; this update carries upstream documentation and regression coverage. - Does this need documentation: No --- thirdparty/vars.sh | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/thirdparty/vars.sh b/thirdparty/vars.sh index 98ca1a6f234aeb..342b4c4d8137ab 100644 --- a/thirdparty/vars.sh +++ b/thirdparty/vars.sh @@ -581,10 +581,10 @@ PUGIXML_MD5SUM="3b894c29455eb33a40b165c6e2de5895" # lance-c # Complete-segment prefilter fixes are supplied by upstream lance-c, not local patches. -LANCE_C_DOWNLOAD="https://codeload.github.com/lance-format/lance-c/tar.gz/1bebabace62e3a9ed366f541003e534c92fc4248" -LANCE_C_NAME="lance-c-1bebabace62e3a9ed366f541003e534c92fc4248.tar.gz" -LANCE_C_SOURCE="lance-c-1bebabace62e3a9ed366f541003e534c92fc4248" -LANCE_C_MD5SUM="461b6430f8ab87607aea9f6697df17a6" +LANCE_C_DOWNLOAD="https://codeload.github.com/lance-format/lance-c/tar.gz/83636c486e22ac62e4b2fd2e225628746cccafc3" +LANCE_C_NAME="lance-c-83636c486e22ac62e4b2fd2e225628746cccafc3.tar.gz" +LANCE_C_SOURCE="lance-c-83636c486e22ac62e4b2fd2e225628746cccafc3" +LANCE_C_MD5SUM="34035a1c31bd67487fa6bbaa7d1eaf03" # all thirdparties which need to be downloaded is set in array TP_ARCHIVES export TP_ARCHIVES=( From a7ff3f86f1aa270bd5a4fffc58cd12a3627f2cd4 Mon Sep 17 00:00:00 2001 From: Gabriel Date: Tue, 29 Sep 2026 16:48:47 +0800 Subject: [PATCH 3/9] feat: expose Lance ANN search stage profiles --- be/src/format_v2/table/lance_reader.cpp | 49 ++++++++++++++ be/test/format_v2/table/lance_reader_test.cpp | 9 +++ docs/lance-ann-profile.md | 67 +++++++++++++++++++ thirdparty/vars.sh | 8 +-- 4 files changed, 129 insertions(+), 4 deletions(-) create mode 100644 docs/lance-ann-profile.md diff --git a/be/src/format_v2/table/lance_reader.cpp b/be/src/format_v2/table/lance_reader.cpp index 724495bf990329..52cf45533e5548 100644 --- a/be/src/format_v2/table/lance_reader.cpp +++ b/be/src/format_v2/table/lance_reader.cpp @@ -697,6 +697,55 @@ void LanceTableReader::_init_scanner_profile() { TUnit::UNIT, LANCE_READER_PROFILE, 1)}, }; _lance_time_metrics = { + // Partition stages accumulate across concurrent work and overlap their parent timers. + // DistanceTopK includes fused candidate filtering, scoring, and heap updates. + {"index_open_time", ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexOpenTime", + LANCE_READER_PROFILE, 1)}, + {"index_partition_load_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexPartitionLoadTime", + LANCE_READER_PROFILE, 1)}, + {"index_partition_prepare_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexPartitionPrepareTime", + LANCE_READER_PROFILE, 1)}, + {"index_prefilter_wait_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexPrefilterWaitTime", + LANCE_READER_PROFILE, 1)}, + {"index_cpu_queue_wait_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexCpuQueueWaitTime", + LANCE_READER_PROFILE, 1)}, + {"index_search_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexSearchTime", + LANCE_READER_PROFILE, 1)}, + {"index_query_prepare_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexQueryPrepareTime", + LANCE_READER_PROFILE, 1)}, + {"index_distance_topk_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexDistanceTopKTime", + LANCE_READER_PROFILE, 1)}, + {"index_result_materialize_time", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceIndexResultMaterializeTime", + LANCE_READER_PROFILE, 1)}, + {"ANNIVFPartitionExec_elapsed_compute", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceANNPartitionExecTime", + LANCE_READER_PROFILE, 1)}, + {"ANNSubIndexExec_elapsed_compute", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceANNSubIndexExecTime", + LANCE_READER_PROFILE, 1)}, + {"ANNIvfBatchExec_elapsed_compute", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceANNBatchExecTime", + LANCE_READER_PROFILE, 1)}, + {"SortExec_elapsed_compute", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceSortComputeTime", + LANCE_READER_PROFILE, 1)}, + {"SortPreservingMergeExec_elapsed_compute", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceSortMergeComputeTime", + LANCE_READER_PROFILE, 1)}, + {"TakeExec_elapsed_compute", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceTakeExecTime", LANCE_READER_PROFILE, + 1)}, + {"KNNVectorDistanceExec_elapsed_compute", + ADD_CHILD_TIMER_WITH_LEVEL(_scanner_profile, "LanceVectorDistanceComputeTime", + LANCE_READER_PROFILE, 1)}, // These are wall times in the ANN row-id loader. LoadTime includes input polling // and set construction; it must not be added to its component timers. {"prefilter_load_time", diff --git a/be/test/format_v2/table/lance_reader_test.cpp b/be/test/format_v2/table/lance_reader_test.cpp index a751e3aa80e86b..93531688634ca2 100644 --- a/be/test/format_v2/table/lance_reader_test.cpp +++ b/be/test/format_v2/table/lance_reader_test.cpp @@ -912,6 +912,15 @@ TEST(LanceTableReaderVectorSearchTest, MultiVectorScoresFiltersOffsetsAndIndexed } EXPECT_TRUE(reader.close().ok()); if (indexed) { + // Warm searches still perform scoring even when every index partition is cached. + for (const char* name : + {"LanceIndexPartitionLoadTime", "LanceIndexCpuQueueWaitTime", + "LanceIndexSearchTime", "LanceIndexQueryPrepareTime", + "LanceIndexDistanceTopKTime", "LanceIndexResultMaterializeTime"}) { + auto* counter = profile.get_counter(name); + ASSERT_NE(nullptr, counter) << name; + EXPECT_GT(counter->value(), 0) << name; + } // Read metrics after close: lance-c publishes its final execution summary // when the stream is released, including for an early top-k stop. for (const char* name : {"LancePrefilterLoads", "LancePrefilterInputRows", diff --git a/docs/lance-ann-profile.md b/docs/lance-ann-profile.md new file mode 100644 index 00000000000000..51fdc15d095ef6 --- /dev/null +++ b/docs/lance-ann-profile.md @@ -0,0 +1,67 @@ + + +# Lance ANN profile timings + +`FileScannerV2` accumulates scanner initialization, open, block-read, and close +wall time. Range acquisition and split preparation are nested within those +calls, so their counters are not additional time. Scanner worker scheduling +wait is reported separately. `LanceScannerReadTime` measures time spent calling +the Lance scanner, including Rust execution and waits. Doris scanner CPU time +does not include work done on Lance's CPU pool. + +The following counters expose the work within an indexed vector search: + +| Counter | Scope | +| --- | --- | +| `LanceIndexOpenTime` | Index-handle lookup/open, including metadata reads on a miss. | +| `LanceIVFPartitionRankingTime` | Partition ranking, including its CPU dispatch wait. | +| `LanceIndexPartitionLoadTime` | Partition cache lookup, coalesced-load wait, and read/decode on a miss. Also measured on cache hits. | +| `LanceIndexPartitionPrepareTime` | Partition load and per-partition filter preparation. On the streaming path, shared-filter waiting overlaps loading. | +| `LanceIndexPrefilterWaitTime` | Waiting for the shared prefilter to become ready. This is distinct from building the filter. | +| `LanceIndexCpuQueueWaitTime` | Delay before a dispatched search or result-materialization CPU task starts. | +| `LanceIndexSearchTime` | Search of prepared partitions on the CPU pool, including query preparation and any per-partition result construction. | +| `LanceIndexQueryPrepareTime` | Distance-calculator / lookup-table construction in the IVF flat sub-index (including quantized storage). | +| `LanceIndexDistanceTopKTime` | Candidate filtering, distance evaluation, and heap updates in that sub-index. These operations are fused in fast-scan paths. | +| `LanceIndexResultMaterializeTime` | Converting result heaps into Arrow arrays and batches, excluding final global sorting. | +| `LanceANNPartitionExecTime`, `LanceANNSubIndexExecTime`, `LanceANNBatchExecTime` | Baseline elapsed times reported by the corresponding Lance ANN operators. These include asynchronous waits. | +| `LanceSortComputeTime`, `LanceSortMergeComputeTime` | DataFusion sort / sort-preserving merge operator compute times. | +| `LanceTakeExecTime` | Baseline time reported by Lance's take operator within the scan plan. Doris second-phase row-ID fetch has separate counters. | +| `LanceVectorDistanceComputeTime` | Baseline reported by the vector-distance operator, e.g. refinement or an unindexed tail. | + +Timings accumulate across partitions, tasks, and index segments. They are +**nested and may overlap**, so summing them does not reconstruct query wall time. +In particular, partition preparation contains loading; search contains query +preparation and distance/TopK work; ANN operator baselines contain downstream +search stages and waits. A zero counter can mean the corresponding operator or +path was not used. Detailed sub-index timers currently cover IVF flat sub-indices; +other sub-indices are visible through the encompassing search timer. + +`LancePrefilterLoadTime` includes `LancePrefilterInputTime` and +`LancePrefilterBuildTime`; do not add these three together. A segment-scoped +search without a predicate can avoid constructing the row-ID allowlist, while +still respecting deletions and fragment visibility. Filter-readiness waiting can +therefore remain nonzero even when the row-ID materialization counters are zero. + +For a warm query with no execution I/O and no prefilter materialization, inspect +CPU queue wait, query preparation, distance/TopK, and result/sort timers. For cold +queries, inspect partition load together with execution bytes, requests, and +partition cache misses. Use repeated queries and the operator-level elapsed +times to assess latency; cumulative parallel stage times alone are not a critical +path trace. diff --git a/thirdparty/vars.sh b/thirdparty/vars.sh index 342b4c4d8137ab..bafd8c94b4c4d1 100644 --- a/thirdparty/vars.sh +++ b/thirdparty/vars.sh @@ -581,10 +581,10 @@ PUGIXML_MD5SUM="3b894c29455eb33a40b165c6e2de5895" # lance-c # Complete-segment prefilter fixes are supplied by upstream lance-c, not local patches. -LANCE_C_DOWNLOAD="https://codeload.github.com/lance-format/lance-c/tar.gz/83636c486e22ac62e4b2fd2e225628746cccafc3" -LANCE_C_NAME="lance-c-83636c486e22ac62e4b2fd2e225628746cccafc3.tar.gz" -LANCE_C_SOURCE="lance-c-83636c486e22ac62e4b2fd2e225628746cccafc3" -LANCE_C_MD5SUM="34035a1c31bd67487fa6bbaa7d1eaf03" +LANCE_C_DOWNLOAD="https://codeload.github.com/lance-format/lance-c/tar.gz/4299a5ee5a7849f7075ba5f1bb327dc165cb30a2" +LANCE_C_NAME="lance-c-4299a5ee5a7849f7075ba5f1bb327dc165cb30a2.tar.gz" +LANCE_C_SOURCE="lance-c-4299a5ee5a7849f7075ba5f1bb327dc165cb30a2" +LANCE_C_MD5SUM="fb1bdfacb63821ba85b3c8aeb879c73d" # all thirdparties which need to be downloaded is set in array TP_ARCHIVES export TP_ARCHIVES=( From 8c7f614fb7a5eaa2e3b2cbacbc8f1e568c0f168d Mon Sep 17 00:00:00 2001 From: Gabriel Date: Tue, 29 Sep 2026 18:48:18 +0800 Subject: [PATCH 4/9] [fix](lance) Update parallel partition preparation timing --- thirdparty/vars.sh | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/thirdparty/vars.sh b/thirdparty/vars.sh index bafd8c94b4c4d1..60936f378ca622 100644 --- a/thirdparty/vars.sh +++ b/thirdparty/vars.sh @@ -581,10 +581,10 @@ PUGIXML_MD5SUM="3b894c29455eb33a40b165c6e2de5895" # lance-c # Complete-segment prefilter fixes are supplied by upstream lance-c, not local patches. -LANCE_C_DOWNLOAD="https://codeload.github.com/lance-format/lance-c/tar.gz/4299a5ee5a7849f7075ba5f1bb327dc165cb30a2" -LANCE_C_NAME="lance-c-4299a5ee5a7849f7075ba5f1bb327dc165cb30a2.tar.gz" -LANCE_C_SOURCE="lance-c-4299a5ee5a7849f7075ba5f1bb327dc165cb30a2" -LANCE_C_MD5SUM="fb1bdfacb63821ba85b3c8aeb879c73d" +LANCE_C_DOWNLOAD="https://codeload.github.com/lance-format/lance-c/tar.gz/4e6b2edbee36103758feb4b3f14ace2475e3be4d" +LANCE_C_NAME="lance-c-4e6b2edbee36103758feb4b3f14ace2475e3be4d.tar.gz" +LANCE_C_SOURCE="lance-c-4e6b2edbee36103758feb4b3f14ace2475e3be4d" +LANCE_C_MD5SUM="bb47fd5cee55428a625ef5278b7005f8" # all thirdparties which need to be downloaded is set in array TP_ARCHIVES export TP_ARCHIVES=( From 1bdd1ea96845010c131496c4acdcc21beb8c966c Mon Sep 17 00:00:00 2001 From: Gabriel Date: Tue, 29 Sep 2026 22:02:53 +0800 Subject: [PATCH 5/9] [fix](lance) Pin merged and native-tested upstream dependency --- thirdparty/vars.sh | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/thirdparty/vars.sh b/thirdparty/vars.sh index 60936f378ca622..4bda48054baa23 100644 --- a/thirdparty/vars.sh +++ b/thirdparty/vars.sh @@ -581,10 +581,10 @@ PUGIXML_MD5SUM="3b894c29455eb33a40b165c6e2de5895" # lance-c # Complete-segment prefilter fixes are supplied by upstream lance-c, not local patches. -LANCE_C_DOWNLOAD="https://codeload.github.com/lance-format/lance-c/tar.gz/4e6b2edbee36103758feb4b3f14ace2475e3be4d" -LANCE_C_NAME="lance-c-4e6b2edbee36103758feb4b3f14ace2475e3be4d.tar.gz" -LANCE_C_SOURCE="lance-c-4e6b2edbee36103758feb4b3f14ace2475e3be4d" -LANCE_C_MD5SUM="bb47fd5cee55428a625ef5278b7005f8" +LANCE_C_DOWNLOAD="https://codeload.github.com/lance-format/lance-c/tar.gz/abba808a6e105f033b9575c28263201b739f89d5" +LANCE_C_NAME="lance-c-abba808a6e105f033b9575c28263201b739f89d5.tar.gz" +LANCE_C_SOURCE="lance-c-abba808a6e105f033b9575c28263201b739f89d5" +LANCE_C_MD5SUM="35c202659051cd059977bbc377f3beb0" # all thirdparties which need to be downloaded is set in array TP_ARCHIVES export TP_ARCHIVES=( From eb51a98db63ba73911acee7d11e5f9b09c4a6e85 Mon Sep 17 00:00:00 2001 From: Gabriel Date: Tue, 29 Sep 2026 22:43:07 +0800 Subject: [PATCH 6/9] [test](lance) Sync upstream prefilter timing coverage --- thirdparty/vars.sh | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/thirdparty/vars.sh b/thirdparty/vars.sh index 4bda48054baa23..506017449e7541 100644 --- a/thirdparty/vars.sh +++ b/thirdparty/vars.sh @@ -581,10 +581,10 @@ PUGIXML_MD5SUM="3b894c29455eb33a40b165c6e2de5895" # lance-c # Complete-segment prefilter fixes are supplied by upstream lance-c, not local patches. -LANCE_C_DOWNLOAD="https://codeload.github.com/lance-format/lance-c/tar.gz/abba808a6e105f033b9575c28263201b739f89d5" -LANCE_C_NAME="lance-c-abba808a6e105f033b9575c28263201b739f89d5.tar.gz" -LANCE_C_SOURCE="lance-c-abba808a6e105f033b9575c28263201b739f89d5" -LANCE_C_MD5SUM="35c202659051cd059977bbc377f3beb0" +LANCE_C_DOWNLOAD="https://codeload.github.com/lance-format/lance-c/tar.gz/9bd730add2ac70316c1d642b8459011e2dd92022" +LANCE_C_NAME="lance-c-9bd730add2ac70316c1d642b8459011e2dd92022.tar.gz" +LANCE_C_SOURCE="lance-c-9bd730add2ac70316c1d642b8459011e2dd92022" +LANCE_C_MD5SUM="63851b09bf1689032579f1a094ff2f37" # all thirdparties which need to be downloaded is set in array TP_ARCHIVES export TP_ARCHIVES=( From 065e490a0116b3d3c78978cb3c805d0a3a56f6ce Mon Sep 17 00:00:00 2001 From: Gabriel Date: Wed, 30 Sep 2026 12:42:10 +0800 Subject: [PATCH 7/9] [fix](lance) Avoid repeated HTTP HEAD requests on Foyer cache hits ### What problem does this PR solve? Problem Summary: HTTP response extensions prevented immutable object metadata from entering the Foyer metadata cache. Every cached single-range read could therefore issue another origin HEAD. Cache ObjectMeta and Attributes regardless of transport extensions, without caching or replaying the extensions themselves. Fingerprint the retained patch so previously patched third-party source trees are refreshed when the patch changes. Preserve reuse for identical patches. ### Release note Avoid redundant metadata requests for cached Lance data-file reads. ### Check List (For Author) - Test: Real HTTP regression and Foyer unit tests; third-party downloader lifecycle checks; Rust formatting, shell syntax, and diff checks. - Behavior changed: Yes, cached immutable metadata avoids repeated origin HEAD requests. - Does this need documentation: No --- thirdparty/download-thirdparty.sh | 11 +- thirdparty/patches/lance-c-foyer.patch | 159 +++++++++++++++++- thirdparty/test/lance-prefilter-patch-test.sh | 16 ++ 3 files changed, 177 insertions(+), 9 deletions(-) diff --git a/thirdparty/download-thirdparty.sh b/thirdparty/download-thirdparty.sh index bb46d3640ea60d..431d5a5d013443 100755 --- a/thirdparty/download-thirdparty.sh +++ b/thirdparty/download-thirdparty.sh @@ -721,11 +721,20 @@ fi # Foyer remains a local patch until its cache interface is accepted upstream. # All search fixes are supplied by the immutable lance-c dependency revision. if [[ " ${TP_ARCHIVES[*]} " =~ " LANCE_C " ]]; then + foyer_patch_checksum="$(cksum < "${TP_PATCH_DIR}/lance-c-foyer.patch")" + foyer_patch_marker="${TP_SOURCE_DIR}/${LANCE_C_SOURCE}/${PATCHED_MARK}_foyer" + # A new local patch must also replace previously patched cached sources. + # Empty markers from older builds cannot identify the applied patch version. + if [[ -f "${foyer_patch_marker}" ]] && + [[ "$(cat "${foyer_patch_marker}")" != "${foyer_patch_checksum}" ]]; then + rm -rf "${TP_SOURCE_DIR}/${LANCE_C_SOURCE}" + "${TAR_CMD}" xzf "${TP_SOURCE_DIR}/${LANCE_C_NAME}" -C "${TP_SOURCE_DIR}/" + fi cd "${TP_SOURCE_DIR}/${LANCE_C_SOURCE}" if [[ ! -f "${PATCHED_MARK}_foyer" ]]; then patch --batch --forward --reject-file=- --fuzz=0 --no-backup-if-mismatch -s \ -p1 <"${TP_PATCH_DIR}/lance-c-foyer.patch" - touch "${PATCHED_MARK}_foyer" + printf '%s\n' "${foyer_patch_checksum}" > "${PATCHED_MARK}_foyer" fi cd - echo "Finished patching ${LANCE_C_SOURCE}" diff --git a/thirdparty/patches/lance-c-foyer.patch b/thirdparty/patches/lance-c-foyer.patch index 426b9257f13d78..3b470bd307213f 100644 --- a/thirdparty/patches/lance-c-foyer.patch +++ b/thirdparty/patches/lance-c-foyer.patch @@ -492,7 +492,7 @@ diff --git a/src/dataset.rs b/src/dataset.rs diff --git a/src/foyer_data_cache.rs b/src/foyer_data_cache.rs --- /dev/null +++ b/src/foyer_data_cache.rs -@@ -0,0 +1,1268 @@ +@@ -0,0 +1,1411 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + @@ -961,13 +961,12 @@ diff --git a/src/foyer_data_cache.rs b/src/foyer_data_cache.rs + }, + ) + .await?; -+ // Extensions may carry request-specific state and cannot be replayed. -+ if result.extensions.is_empty() { -+ self.reader.cache.metadata.insert( -+ metadata_key, -+ (result.meta.clone(), result.attributes.clone()), -+ ); -+ } ++ // HTTP responses carry transport extensions even for immutable files. ++ // Cache metadata independently; request-specific extensions are never replayed. ++ self.reader.cache.metadata.insert( ++ metadata_key, ++ (result.meta.clone(), result.attributes.clone()), ++ ); + (result.meta, result.attributes, result.extensions) + }; + let object_size = metadata.size; @@ -1360,6 +1359,150 @@ diff --git a/src/foyer_data_cache.rs b/src/foyer_data_cache.rs + } + + #[tokio::test] ++ async fn cached_http_ranges_do_not_repeat_head_requests() { ++ use object_store::aws::AmazonS3Builder; ++ use tokio::io::{AsyncReadExt, AsyncWriteExt}; ++ ++ let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); ++ let address = listener.local_addr().unwrap(); ++ let heads = Arc::new(AtomicU64::new(0)); ++ let gets = Arc::new(AtomicU64::new(0)); ++ let server_heads = heads.clone(); ++ let server_gets = gets.clone(); ++ let data = Bytes::from((0..8192).map(|value| value as u8).collect::>()); ++ let server_data = data.clone(); ++ let server = tokio::spawn(async move { ++ loop { ++ let (mut socket, _) = listener.accept().await.unwrap(); ++ let mut request = Vec::new(); ++ while !request.ends_with(b"\r\n\r\n") { ++ request.push(socket.read_u8().await.unwrap()); ++ assert!(request.len() < 16 * 1024); ++ } ++ let request = String::from_utf8(request).unwrap(); ++ let is_head = request.starts_with("HEAD "); ++ let range = request.lines().find_map(|line| { ++ line.to_ascii_lowercase() ++ .strip_prefix("range: bytes=") ++ .map(|range| { ++ let (start, end) = range.split_once('-').unwrap(); ++ start.parse::().unwrap()..end.parse::().unwrap() + 1 ++ }) ++ }); ++ let (status, content_range, range) = match range { ++ Some(range) => ( ++ "206 Partial Content", ++ format!( ++ "Content-Range: bytes {}-{}/{}\r\n", ++ range.start, ++ range.end - 1, ++ server_data.len() ++ ), ++ range, ++ ), ++ None => ("200 OK", String::new(), 0..server_data.len()), ++ }; ++ if is_head { ++ server_heads.fetch_add(1, Ordering::SeqCst); ++ } else { ++ server_gets.fetch_add(1, Ordering::SeqCst); ++ } ++ let headers = format!( ++ "HTTP/1.1 {status}\r\nContent-Length: {}\r\n{content_range}Last-Modified: Tue, 01 Sep 2026 00:00:00 GMT\r\nETag: \"sample\"\r\nContent-Type: application/octet-stream\r\nConnection: close\r\n\r\n", ++ range.len() ++ ); ++ socket.write_all(headers.as_bytes()).await.unwrap(); ++ if !is_head { ++ socket.write_all(&server_data[range]).await.unwrap(); ++ } ++ } ++ }); ++ ++ // Use a real HTTP client: in-memory stores do not exercise transport extensions. ++ let original = Arc::new( ++ AmazonS3Builder::new() ++ .with_bucket_name("example-bucket") ++ .with_region("us-east-1") ++ .with_endpoint(format!("http://{address}")) ++ .with_allow_http(true) ++ // An explicit excluded proxy also disables environment proxy discovery. ++ .with_proxy_url("http://127.0.0.1:1") ++ .with_proxy_excludes("127.0.0.1") ++ .with_skip_signature(true) ++ .build() ++ .unwrap(), ++ ); ++ let directory = tempfile::tempdir().unwrap(); ++ let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) ++ .await ++ .unwrap(); ++ let path = Path::from("table.lance/data/sample.lance"); ++ let (wrapped, _) = wrap_for_test(&cache, original.clone()); ++ let first = wrapped ++ .get_opts( ++ &path, ++ GetOptions { ++ range: Some(GetRange::Bounded(100..200)), ++ ..Default::default() ++ }, ++ ) ++ .await ++ .unwrap(); ++ assert!( ++ !first.extensions.is_empty(), ++ "HTTP response must exercise transport extensions" ++ ); ++ let meta = first.meta.clone(); ++ let attributes = first.attributes.clone(); ++ assert_eq!(first.range, 100..200); ++ assert_eq!(first.bytes().await.unwrap(), data.slice(100..200)); ++ assert_eq!(heads.load(Ordering::SeqCst), 1); ++ assert_eq!(gets.load(Ordering::SeqCst), 1); ++ ++ // Metadata must be shared with fresh dataset scopes, just like cached blocks. ++ let (fresh, _) = wrap_for_test(&cache, original); ++ let mut replayed_extensions = false; ++ for store in [&wrapped, &fresh] { ++ let result = store ++ .get_opts( ++ &path, ++ GetOptions { ++ range: Some(GetRange::Bounded(120..180)), ++ ..Default::default() ++ }, ++ ) ++ .await ++ .unwrap(); ++ replayed_extensions |= !result.extensions.is_empty(); ++ assert_eq!(result.meta, meta); ++ assert_eq!(result.attributes, attributes); ++ assert_eq!(result.range, 120..180); ++ assert_eq!(result.bytes().await.unwrap(), data.slice(120..180)); ++ } ++ assert_eq!( ++ heads.load(Ordering::SeqCst), ++ 1, ++ "cache hits must not issue HEAD" ++ ); ++ assert_eq!( ++ gets.load(Ordering::SeqCst), ++ 1, ++ "cache hits must not issue GET" ++ ); ++ assert!( ++ !replayed_extensions, ++ "cached metadata must not replay response extensions" ++ ); ++ wrapped.head(&path).await.unwrap(); ++ assert_eq!( ++ heads.load(Ordering::SeqCst), ++ 2, ++ "explicit HEAD must bypass cache" ++ ); ++ server.abort(); ++ } ++ ++ #[tokio::test] + async fn caches_only_immutable_data_file_ranges() { + let directory = tempfile::tempdir().unwrap(); + let cache = FoyerDataCache::try_new(directory.path(), 256 * 1024, 1024 * 1024, 64 * 1024) diff --git a/thirdparty/test/lance-prefilter-patch-test.sh b/thirdparty/test/lance-prefilter-patch-test.sh index c22c3a2c7ded03..ecf9c0c718968f 100755 --- a/thirdparty/test/lance-prefilter-patch-test.sh +++ b/thirdparty/test/lance-prefilter-patch-test.sh @@ -63,7 +63,10 @@ run_download "${tmpdir}/fresh" check_sources "${tmpdir}/fresh" cp "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/Cargo.toml" "${tmpdir}/manifest" cp "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/Cargo.lock" "${tmpdir}/lock" +touch "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/cache_reuse_sentinel" run_download "${tmpdir}/fresh" +[[ -f "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/cache_reuse_sentinel" ]] \ + || fail "unchanged patch unnecessarily replaced cached sources" cmp "${tmpdir}/manifest" "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/Cargo.toml" cmp "${tmpdir}/lock" "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/Cargo.lock" echo "PASS: fresh archive and idempotent Foyer patch" @@ -84,6 +87,19 @@ check_sources "${tmpdir}/cached" cmp "${tmpdir}/lock" "${tmpdir}/cached/src/${LANCE_C_SOURCE}/Cargo.lock" echo "PASS: cached sources with an existing generic marker" +# A cached source tree may already contain the previous Foyer patch. The patch +# fingerprint must invalidate it even when the upstream archive is unchanged. +cached_source="${tmpdir}/cached/src/${LANCE_C_SOURCE}" +for stale_marker in '' '0 0'; do + printf '%s\n' 'stale patch contents' > "${cached_source}/src/foyer_data_cache.rs" + printf '%s\n' "${stale_marker}" > "${cached_source}/patched_mark_foyer" + run_download "${tmpdir}/cached" + check_sources "${tmpdir}/cached" + cmp "${tmpdir}/fresh/src/${LANCE_C_SOURCE}/src/foyer_data_cache.rs" \ + "${cached_source}/src/foyer_data_cache.rs" +done +echo "PASS: old and mismatched Foyer markers refresh cached sources" + prepare "${tmpdir}/invalid" tar xzf "${ARCHIVE_DIR}/${LANCE_C_NAME}" -C "${tmpdir}/invalid/src" printf '%s\n' 'incompatible manifest' > "${tmpdir}/invalid/src/${LANCE_C_SOURCE}/Cargo.toml" From 7586f8c570793cdaf694ccae1773d6938893acc4 Mon Sep 17 00:00:00 2001 From: Gabriel Date: Wed, 30 Sep 2026 14:41:40 +0800 Subject: [PATCH 8/9] [chore](lance) generate Foyer patch from unified source --- thirdparty/patches/lance-c-foyer.patch | 291 +++++++++++++------------ 1 file changed, 157 insertions(+), 134 deletions(-) diff --git a/thirdparty/patches/lance-c-foyer.patch b/thirdparty/patches/lance-c-foyer.patch index 3b470bd307213f..3e5c1c8a99f15a 100644 --- a/thirdparty/patches/lance-c-foyer.patch +++ b/thirdparty/patches/lance-c-foyer.patch @@ -1,13 +1,16 @@ -# Retained Foyer data-cache integration from lance-format/lance-c#73. -# Rebased onto the pinned upstream lance-c revision; search fixes live upstream. +# Foyer data-cache integration for lance-format/lance-c#73. +# Base: 9bd730add2ac70316c1d642b8459011e2dd92022 +# Source: https://github.com/Gabriel39/lance-c/commit/98271597a0535902b7abce912cd19819e84bc681 +# Regenerate the payload in lance-c; do not maintain separate downstream edits: +# git diff --full-index --binary 9bd730add2ac70316c1d642b8459011e2dd92022 98271597a0535902b7abce912cd19819e84bc681 -- . | sed 's/^ $//' diff --git a/Cargo.lock b/Cargo.lock +index 78812f1330e146db295a14276f90f9e28654f399..daf9f85daf3e0ea59bb906e8e3c32470519b084b 100644 --- a/Cargo.lock +++ b/Cargo.lock -@@ -1192,6 +1192,17 @@ - version = "0.8.7" +@@ -1193,6 +1193,17 @@ version = "0.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" -+ + +[[package]] +name = "core_affinity" +version = "0.8.3" @@ -18,13 +21,15 @@ diff --git a/Cargo.lock b/Cargo.lock + "num_cpus", + "winapi", +] - ++ [[package]] name = "countio" -@@ -2242,6 +2253,16 @@ + version = "0.3.0" +@@ -2241,6 +2252,16 @@ version = "0.2.3" + source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f8eb564c5c7423d25c886fb561d1e4ee69f72354d16918afa32c08811f6b6a55" - [[package]] ++[[package]] +name = "fastant" +version = "0.1.11" +source = "registry+https://github.com/rust-lang/crates.io-index" @@ -34,16 +39,13 @@ diff --git a/Cargo.lock b/Cargo.lock + "web-time", +] + -+[[package]] + [[package]] name = "fastrand" version = "2.3.0" - source = "registry+https://github.com/rust-lang/crates.io-index" -@@ -2310,6 +2331,26 @@ - checksum = "cb4cb245038516f5f85277875cdaa4f7d2c9a0fa0468de06ed190163b1581fcf" - dependencies = [ +@@ -2312,6 +2333,26 @@ dependencies = [ "percent-encoding", -+] -+ + ] + +[[package]] +name = "foyer" +version = "0.22.5" @@ -62,13 +64,16 @@ diff --git a/Cargo.lock b/Cargo.lock + "pin-project", + "serde", + "tracing", - ] - ++] ++ [[package]] -@@ -2363,6 +2404,38 @@ + name = "foyer-common" + version = "0.22.6" +@@ -2362,6 +2403,38 @@ dependencies = [ + "tracing", ] - [[package]] ++[[package]] +name = "foyer-storage" +version = "0.22.6" +source = "registry+https://github.com/rust-lang/crates.io-index" @@ -100,15 +105,13 @@ diff --git a/Cargo.lock b/Cargo.lock + "zstd", +] + -+[[package]] + [[package]] name = "foyer-tokio" version = "0.22.6" - source = "registry+https://github.com/rust-lang/crates.io-index" -@@ -2376,6 +2449,16 @@ - version = "1.20260821.5" +@@ -2377,6 +2450,16 @@ version = "1.20260821.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "36a80a7406da302e04bfd2ca987907590d3a1f3c69958947c43890abd7426b2f" -+ + +[[package]] +name = "fs4" +version = "0.13.1" @@ -118,10 +121,11 @@ diff --git a/Cargo.lock b/Cargo.lock + "rustix", + "windows-sys 0.59.0", +] - ++ [[package]] name = "fs_extra" -@@ -3720,8 +3803,10 @@ + version = "1.3.0" +@@ -3720,8 +3803,10 @@ dependencies = [ "arrow-array", "arrow-schema", "async-trait", @@ -132,7 +136,7 @@ diff --git a/Cargo.lock b/Cargo.lock "futures", "half", "lance", -@@ -3735,6 +3820,7 @@ +@@ -3735,6 +3820,7 @@ dependencies = [ "lance-table", "libc", "log", @@ -140,39 +144,40 @@ diff --git a/Cargo.lock b/Cargo.lock "opendal", "pin-project", "prost", -@@ -6629,6 +6715,12 @@ +@@ -6628,6 +6714,12 @@ version = "0.4.12" + source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" - [[package]] ++[[package]] +name = "small_ctor" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "88414a5ca1f85d82cc34471e975f0f74f6aa54c40f062efa42c0080e7f763f81" + -+[[package]] + [[package]] name = "smallvec" version = "1.15.1" - source = "registry+https://github.com/rust-lang/crates.io-index" -@@ -7893,6 +7985,15 @@ - version = "0.52.0" - source = "registry+https://github.com/rust-lang/crates.io-index" - checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" -+dependencies = [ -+ "windows-targets 0.52.6", -+] -+ +@@ -7897,6 +7989,15 @@ dependencies = [ + "windows-targets 0.52.6", + ] + +[[package]] +name = "windows-sys" +version = "0.59.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1e38bc4d79ed67fd075bcc251a1c39b32a1776bbe92e5bef1f0bf1f8c531853b" - dependencies = [ - "windows-targets 0.52.6", - ] ++dependencies = [ ++ "windows-targets 0.52.6", ++] ++ + [[package]] + name = "windows-sys" + version = "0.60.2" diff --git a/Cargo.toml b/Cargo.toml +index 18654b862440a90aa6bf0dd846e8f1ad6036fd7c..fd134a7151e1df61da201d867a642e9bcbf645d5 100644 --- a/Cargo.toml +++ b/Cargo.toml -@@ -35,6 +35,7 @@ +@@ -35,6 +35,7 @@ datafusion = { version = "54.0.0", default-features = false } arrow = { version = "58.0.0", features = ["prettyprint", "ffi"] } arrow-array = "58.0.0" arrow-schema = "58.0.0" @@ -180,7 +185,7 @@ diff --git a/Cargo.toml b/Cargo.toml # Direct to name `chrono::TimeDelta` (the field type of lance's public # `AutoCleanupParams`) and `chrono::DateTime`/`Utc` (index metadata # timestamps); already in the graph transitively via lance. -@@ -42,10 +43,12 @@ +@@ -42,10 +43,12 @@ chrono = { version = "0.4", default-features = false } half = "2" tokio = { version = "1", features = ["rt-multi-thread", "sync"] } futures = "0.3" @@ -194,9 +199,10 @@ diff --git a/Cargo.toml b/Cargo.toml prost = "0.14" snafu = "0.9" diff --git a/README.md b/README.md +index 9ceccdc4f3f6b5d13dc271be3d2659c7be400306..9d5a318d911ea8f8efba82a447275ebf6e88901f 100644 --- a/README.md +++ b/README.md -@@ -68,6 +68,7 @@ +@@ -68,6 +68,7 @@ Based on the [liblance RFC](https://github.com/lance-format/lance/discussions/60 | [x] | Async scan | Callback-based `lance_scanner_scan_async()` for non-blocking scans | | [x] | Dataset metadata | `lance_dataset_version()`, `lance_dataset_count_rows()`, `lance_dataset_latest_version()` | | [x] | Filter pushdown | `lance_scanner_set_substrait_filter()` accepts a serialized Substrait `ExtendedExpression`; `lance_scanner_additional_sql_filter()` adds SQL predicates with AND before scanning starts | @@ -204,12 +210,10 @@ diff --git a/README.md b/README.md ## Multi-vector search -@@ -229,6 +230,29 @@ - 1ULL * 1024 * 1024 * 1024); - auto ds = lance::Dataset::open_with_session(session, "data.lance"); +@@ -231,6 +232,29 @@ auto ds = lance::Dataset::open_with_session(session, "data.lance"); auto stats = session.cache_stats(); -+``` -+ + ``` + +To add a process-local memory/disk cache for remote Lance data-file reads, +create the session with Foyer configuration. The cache is deliberately narrow: +whole-object, single-range, and batched range reads of direct `data/*.lance` @@ -231,16 +235,20 @@ diff --git a/README.md b/README.md + 1ULL * 1024 * 1024 * 1024, + data_cache); +auto ds = lance::Dataset::open_with_session(session, "s3://bucket/data.lance"); - ``` - ++``` ++ ### Open at a specific version + + `lance_dataset_open` takes a `version` argument — `0` means the latest, any diff --git a/include/lance/lance.h b/include/lance/lance.h +index 7630913fd7849d0bff2cb0f31e1ab1199cb9f0fb..1c1d0752efdda5023b29eafe5aa9a87e03dc5303 100644 --- a/include/lance/lance.h +++ b/include/lance/lance.h -@@ -215,6 +215,36 @@ +@@ -214,6 +214,36 @@ typedef struct LanceSessionCacheStats { + uint64_t metadata_cache_size_bytes; } LanceSessionCacheStats; - /** ++/** + * Configuration for the optional Foyer cache of immutable Lance data files. + * + * Whole-object, single-range, and batched range reads of direct @@ -270,16 +278,13 @@ diff --git a/include/lance/lance.h b/include/lance/lance.h + uint64_t bytes_read_from_remote; +} LanceDataCacheStatistics; + -+/** + /** * Create a session that can share metadata and index caches across datasets. * - * Cache limits are specified in bytes. Pass 0 to request zero capacity. -@@ -223,6 +253,25 @@ - LanceSession* lance_session_new( - uint64_t index_cache_size_bytes, +@@ -225,6 +255,25 @@ LanceSession* lance_session_new( uint64_t metadata_cache_size_bytes -+); -+ + ); + +/** + * Create a shared Lance session with a Foyer data-file cache. + * @@ -297,15 +302,15 @@ diff --git a/include/lance/lance.h b/include/lance/lance.h + uint64_t index_cache_size_bytes, + uint64_t metadata_cache_size_bytes, + const LanceDataCacheOptions* data_cache_options - ); - - /** -@@ -279,6 +328,20 @@ - const char* const* storage_opts, - uint64_t version, - const LanceSession* session +); + + /** + * Close a session handle. Safe to call with NULL. Datasets previously opened + * with the session remain valid and retain the shared cache state. +@@ -281,6 +330,20 @@ LanceDataset* lance_dataset_open_with_session( + const LanceSession* session + ); + +/** + * Copy this dataset handle's cumulative data-cache statistics. + * @@ -318,13 +323,16 @@ diff --git a/include/lance/lance.h b/include/lance/lance.h +int32_t lance_dataset_get_data_cache_statistics( + const LanceDataset* dataset, + LanceDataCacheStatistics* out_statistics - ); - ++); ++ /** Close and free a dataset handle. Safe to call with NULL. */ + void lance_dataset_close(LanceDataset* dataset); + diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp +index 286724e96736e354785e046f2fcfe5ae7c65147b..5070d423453a5d60d31d03ce758ad7dbe83a6ea2 100644 --- a/include/lance/lance.hpp +++ b/include/lance/lance.hpp -@@ -176,12 +176,34 @@ +@@ -176,6 +176,13 @@ struct SqlColumn { // ─── Shared Session ────────────────────────────────────────────────────────── @@ -338,12 +346,10 @@ diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp class Session { Handle handle_; - public: - Session(uint64_t index_cache_size_bytes, uint64_t metadata_cache_size_bytes) - : handle_(lance_session_new(index_cache_size_bytes, metadata_cache_size_bytes)) { -+ if (!handle_) check_error(); -+ } -+ +@@ -185,6 +192,21 @@ public: + if (!handle_) check_error(); + } + + Session(uint64_t index_cache_size_bytes, + uint64_t metadata_cache_size_bytes, + const DataCacheOptions& data_cache_options) { @@ -356,24 +362,29 @@ diff --git a/include/lance/lance.hpp b/include/lance/lance.hpp + handle_ = Handle( + lance_session_new_with_data_cache( + index_cache_size_bytes, metadata_cache_size_bytes, &options)); - if (!handle_) check_error(); - } - -@@ -348,6 +370,13 @@ - uri.c_str(), opts_ptr, version, session.c_handle()); - if (!ds) check_error(); - return Dataset(ds); ++ if (!handle_) check_error(); + } + + LanceSessionCacheStats cache_stats() const { + LanceSessionCacheStats stats{}; + if (lance_session_get_cache_stats(handle_.get(), &stats) != 0) +@@ -350,6 +372,13 @@ public: + return Dataset(ds); + } + + LanceDataCacheStatistics data_cache_statistics() const { + LanceDataCacheStatistics statistics{}; + if (lance_dataset_get_data_cache_statistics(handle_.get(), &statistics) != 0) + check_error(); + return statistics; - } - ++ } ++ /// Write an Arrow record batch stream to a Lance dataset and return the + /// open dataset at the committed version. + /// diff --git a/src/data_cache.rs b/src/data_cache.rs +new file mode 100644 +index 0000000000000000000000000000000000000000..430b4c891f91d31f202e2318e18db64c7a12ae9d --- /dev/null +++ b/src/data_cache.rs @@ -0,0 +1,68 @@ @@ -446,9 +457,10 @@ diff --git a/src/data_cache.rs b/src/data_cache.rs + Ok(0) +} diff --git a/src/dataset.rs b/src/dataset.rs +index cc1f87ce3f7dfe8dd33aeb6ab7f54d02e31a220a..76fd39ef9e456d1fab6978b6b2cb3c27fab1c2f3 100644 --- a/src/dataset.rs +++ b/src/dataset.rs -@@ -14,6 +14,7 @@ +@@ -14,6 +14,7 @@ use lance::Dataset; use lance::dataset::builder::DatasetBuilder; use lance_core::Result; @@ -456,7 +468,7 @@ diff --git a/src/dataset.rs b/src/dataset.rs use crate::error::{ffi_try, swallow_unwind}; use crate::helpers; use crate::runtime::block_on; -@@ -23,6 +24,7 @@ +@@ -23,6 +24,7 @@ use crate::stream_guard::guarded_ffi_stream_from_reader; /// Opaque handle representing an opened Lance dataset. pub struct LanceDataset { pub(crate) inner: RwLock>, @@ -464,7 +476,7 @@ diff --git a/src/dataset.rs b/src/dataset.rs } impl LanceDataset { -@@ -182,8 +184,16 @@ +@@ -182,8 +184,16 @@ unsafe fn open_dataset_inner( } let dataset = block_on(builder.load())?; @@ -481,7 +493,7 @@ diff --git a/src/dataset.rs b/src/dataset.rs }; Ok(Box::into_raw(Box::new(handle))) } -@@ -519,6 +529,7 @@ +@@ -519,6 +529,7 @@ mod tests { .unwrap(); let handle = LanceDataset { inner: RwLock::new(Arc::new(dataset)), @@ -490,6 +502,8 @@ diff --git a/src/dataset.rs b/src/dataset.rs (tmp, handle) } diff --git a/src/foyer_data_cache.rs b/src/foyer_data_cache.rs +new file mode 100644 +index 0000000000000000000000000000000000000000..c892311be04c4a2b21026954cb386f930a4aab98 --- /dev/null +++ b/src/foyer_data_cache.rs @@ -0,0 +1,1411 @@ @@ -1905,9 +1919,10 @@ diff --git a/src/foyer_data_cache.rs b/src/foyer_data_cache.rs + } +} diff --git a/src/lib.rs b/src/lib.rs +index 34817600548d8d68c0081597d2f0ea9a41efc8d8..4a96bb58652b467d3ff278596b4f4fd95eea97fb 100644 --- a/src/lib.rs +++ b/src/lib.rs -@@ -26,11 +26,13 @@ +@@ -26,11 +26,13 @@ mod async_dispatcher; mod batch; mod blob; mod compact; @@ -1921,7 +1936,7 @@ diff --git a/src/lib.rs b/src/lib.rs mod fragment_writer; mod fts_query; mod helpers; -@@ -55,6 +57,7 @@ +@@ -55,6 +57,7 @@ pub use alter_columns::*; pub use batch::*; pub use blob::*; pub use compact::*; @@ -1929,7 +1944,7 @@ diff --git a/src/lib.rs b/src/lib.rs pub use data_statistics::*; pub use dataset::*; pub use delete::*; -@@ -62,6 +65,7 @@ +@@ -62,6 +65,7 @@ pub use drop_columns::*; pub use error::{ LanceErrorCode, lance_free_string, lance_last_error_code, lance_last_error_message, }; @@ -1938,9 +1953,10 @@ diff --git a/src/lib.rs b/src/lib.rs pub use fts_query::*; pub use index::*; diff --git a/src/restore.rs b/src/restore.rs +index 7804b55818c3b0ed2f14de5cc9f632098e0eb25e..fa2d26da5670c78224a3e33f98f747b24190b6d6 100644 --- a/src/restore.rs +++ b/src/restore.rs -@@ -65,8 +65,16 @@ +@@ -65,8 +65,16 @@ unsafe fn restore_inner(dataset: *const LanceDataset, version: u64) -> Result<*m Ok::<_, lance_core::Error>(checked_out) })?; @@ -1958,9 +1974,10 @@ diff --git a/src/restore.rs b/src/restore.rs Ok(Box::into_raw(Box::new(handle))) } diff --git a/src/session.rs b/src/session.rs +index 60a16234557eaeb93e8eaad971a9eba73a562a83..9ed8cfe4b5207afc78c65163e8cc4c7dc1faab09 100644 --- a/src/session.rs +++ b/src/session.rs -@@ -8,12 +8,14 @@ +@@ -8,12 +8,14 @@ use std::sync::Arc; use lance::session::Session; use lance_core::Result; @@ -1976,10 +1993,11 @@ diff --git a/src/session.rs b/src/session.rs } /// Snapshot of a session's metadata and index cache statistics. -@@ -48,6 +50,14 @@ +@@ -47,6 +49,14 @@ pub extern "C" fn lance_session_new( + fn session_new_inner( index_cache_size_bytes: u64, metadata_cache_size_bytes: u64, - ) -> Result<*mut LanceSession> { ++) -> Result<*mut LanceSession> { + session_new_with_data_cache_factory(index_cache_size_bytes, metadata_cache_size_bytes, None) +} + @@ -1987,11 +2005,10 @@ diff --git a/src/session.rs b/src/session.rs + index_cache_size_bytes: u64, + metadata_cache_size_bytes: u64, + data_cache_factory: Option>, -+) -> Result<*mut LanceSession> { + ) -> Result<*mut LanceSession> { let index_cache_size_bytes = u64_to_usize(index_cache_size_bytes, "index_cache_size_bytes")?; let metadata_cache_size_bytes = - u64_to_usize(metadata_cache_size_bytes, "metadata_cache_size_bytes")?; -@@ -58,6 +68,7 @@ +@@ -58,6 +68,7 @@ fn session_new_inner( ); Ok(Box::into_raw(Box::new(LanceSession { inner: Arc::new(session), @@ -2000,9 +2017,10 @@ diff --git a/src/session.rs b/src/session.rs } diff --git a/src/writer.rs b/src/writer.rs +index 1971510de4a13ee6f04fc39c159b088e78e21d55..ba51c87a8cebb6897c48fb55509f71dabb58f3b4 100644 --- a/src/writer.rs +++ b/src/writer.rs -@@ -282,6 +282,7 @@ +@@ -282,6 +282,7 @@ unsafe fn write_dataset_inner( if !out_dataset.is_null() { let handle = LanceDataset { inner: RwLock::new(Arc::new(dataset)), @@ -2011,9 +2029,10 @@ diff --git a/src/writer.rs b/src/writer.rs // SAFETY: `out_dataset` is non-NULL (checked above) and the caller // guarantees it points to caller-owned, writable storage of size diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs +index 1a25e34a8ca50623812ffedaa7bb5a8c54f7850e..b63a26a7c2135026e7ccf761565cc7746fde79f5 100644 --- a/tests/c_api_test.rs +++ b/tests/c_api_test.rs -@@ -100,8 +100,81 @@ +@@ -100,10 +100,83 @@ fn create_large_dataset(num_rows: i32) -> (tempfile::TempDir, String) { (tmp, uri) } @@ -2056,8 +2075,8 @@ diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs + fn c_str(s: &str) -> CString { CString::new(s).unwrap() -+} -+ + } + +fn file_object_store_uri(path: &str) -> CString { + let path = path.replace('\\', "/"); + let leading_slash = if path.starts_with('/') { "" } else { "/" }; @@ -2092,13 +2111,16 @@ diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs + .iter() + .map(RecordBatch::num_rows) + .sum() - } - ++} ++ #[derive(Default)] -@@ -446,6 +519,120 @@ + struct CapturedScanStatistics { + calls: usize, +@@ -445,6 +518,120 @@ fn test_shared_session_rejects_null_inputs() { + } } - #[test] ++#[test] +fn test_session_with_data_cache_serves_repeated_scan() { + let (tmp, uri) = create_large_multi_fragment_dataset(10_000); + let c_uri = file_object_store_uri(&uri); @@ -2212,16 +2234,13 @@ diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs + assert_eq!(lance_last_error_code(), LanceErrorCode::InvalidArgument); +} + -+#[test] + #[test] fn test_open_nonexistent() { let c_uri = c_str("memory://nonexistent_dataset_xyz"); - let ds = unsafe { lance_dataset_open(c_uri.as_ptr(), ptr::null(), 0) }; -@@ -3316,6 +3503,40 @@ - - unsafe { lance_dataset_close(restored) }; +@@ -3318,6 +3505,40 @@ fn test_dataset_restore_to_prior_version() { unsafe { lance_dataset_close(ds) }; -+} -+ + } + +#[test] +fn test_restored_handle_has_independent_data_cache_statistics() { + let (_tmp, uri) = create_large_multi_fragment_dataset(10_000); @@ -2254,18 +2273,19 @@ diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs + lance_dataset_close(restored); + lance_dataset_close(source); + } - } - ++} ++ #[test] + fn test_dataset_restore_to_current_latest_writes_new_manifest() { + // Restoring to the current latest still writes a new manifest. The diff --git a/tests/cpp/test_c_api.c b/tests/cpp/test_c_api.c +index 5df9cde4d8fa971823b9e6c974a8a0db6c9ec9ca..1444e55161631ac8ad976f427d867ad91d358195 100644 --- a/tests/cpp/test_c_api.c +++ b/tests/cpp/test_c_api.c -@@ -169,6 +169,37 @@ - lance_dataset_close(ds); - printf("metadata_entries=%llu... OK\n", +@@ -171,6 +171,37 @@ static void test_shared_session(const char *uri) { (unsigned long long)stats.metadata_cache_entries); -+} -+ + } + +static void test_data_cache_session(const char *uri, const char *write_uri) { + printf(" test_data_cache_session... "); + @@ -2295,10 +2315,12 @@ diff --git a/tests/cpp/test_c_api.c b/tests/cpp/test_c_api.c + "dataset should remain valid after data-cache session close"); + lance_dataset_close(ds); + printf("OK\n"); - } - ++} ++ static void test_scan(const char *uri) { -@@ -1323,6 +1354,7 @@ + printf(" test_scan... "); + +@@ -1323,6 +1354,7 @@ int main(int argc, char **argv) { test_open_and_metadata(uri); test_shared_session(uri); @@ -2307,15 +2329,13 @@ diff --git a/tests/cpp/test_c_api.c b/tests/cpp/test_c_api.c test_scan_with_limit(uri); test_scanner_blob_handling(blob_uri); diff --git a/tests/cpp/test_cpp_api.cpp b/tests/cpp/test_cpp_api.cpp +index 0332d1a2687354984d0551e10716df08344a4013..ef64ab0d02bed2f1750c3d388792e4aa44265e24 100644 --- a/tests/cpp/test_cpp_api.cpp +++ b/tests/cpp/test_cpp_api.cpp -@@ -128,6 +128,28 @@ +@@ -131,6 +131,28 @@ static void test_shared_session(const std::string& uri) { + PASS(); + } - printf("metadata_entries=%llu... ", - (unsigned long long)stats.metadata_cache_entries); -+ PASS(); -+} -+ +static void test_data_cache_session(const std::string& uri, + const std::string& write_uri) { + TEST(test_data_cache_session); @@ -2335,10 +2355,13 @@ diff --git a/tests/cpp/test_cpp_api.cpp b/tests/cpp/test_cpp_api.cpp + session.reset(); + assert(ds.count_rows() > 0); + - PASS(); - } ++ PASS(); ++} ++ + static void test_dataset_schema(const std::string& uri) { + TEST(test_dataset_schema); -@@ -1211,6 +1233,7 @@ +@@ -1211,6 +1233,7 @@ int main(int argc, char** argv) { test_dataset_open(uri); test_shared_session(uri); From 4c1f03e50960bb9297b0cfb251cf01d9968f6e32 Mon Sep 17 00:00:00 2001 From: Gabriel Date: Wed, 30 Sep 2026 15:21:37 +0800 Subject: [PATCH 9/9] [fix](lance) Isolate Foyer cache origins and avoid warm metadata writes ### What problem does this PR solve? Related PR: #68613 Problem Summary: Identical bucket/path names on distinct storage endpoints could reuse cached metadata or bytes. Warm range reads also repeatedly enqueued size records to disk. Regenerate the Foyer patch from its reviewed source fix and add macOS nounset coverage to the patch harness. ### Release note Cache entries are isolated by live object-store instance. New instances and process restarts start cold; same-instance hot reads avoid redundant size writes. ### Check List (For Author) - Test: 461 Rust tests; generated source equality; GNU patch and git apply; downloader lifecycle, shell syntax, and simulated macOS platform checks. - Behavior changed: Yes, cache isolation and warm-read disk write behavior. - Does this need documentation: Yes, included in the upstream README patch. --- thirdparty/patches/lance-c-foyer.patch | 268 ++++++++++++++++-- thirdparty/test/lance-prefilter-patch-test.sh | 13 + 2 files changed, 259 insertions(+), 22 deletions(-) diff --git a/thirdparty/patches/lance-c-foyer.patch b/thirdparty/patches/lance-c-foyer.patch index 3e5c1c8a99f15a..73f244c49fd57d 100644 --- a/thirdparty/patches/lance-c-foyer.patch +++ b/thirdparty/patches/lance-c-foyer.patch @@ -1,8 +1,8 @@ # Foyer data-cache integration for lance-format/lance-c#73. # Base: 9bd730add2ac70316c1d642b8459011e2dd92022 -# Source: https://github.com/Gabriel39/lance-c/commit/98271597a0535902b7abce912cd19819e84bc681 +# Source: https://github.com/Gabriel39/lance-c/commit/24c7ca4bcb9422c113b0d3e07e4efe1173b0bc9f # Regenerate the payload in lance-c; do not maintain separate downstream edits: -# git diff --full-index --binary 9bd730add2ac70316c1d642b8459011e2dd92022 98271597a0535902b7abce912cd19819e84bc681 -- . | sed 's/^ $//' +# git diff --full-index --binary 9bd730add2ac70316c1d642b8459011e2dd92022 24c7ca4bcb9422c113b0d3e07e4efe1173b0bc9f -- . | sed 's/^ $//' diff --git a/Cargo.lock b/Cargo.lock index 78812f1330e146db295a14276f90f9e28654f399..daf9f85daf3e0ea59bb906e8e3c32470519b084b 100644 --- a/Cargo.lock @@ -199,7 +199,7 @@ index 18654b862440a90aa6bf0dd846e8f1ad6036fd7c..fd134a7151e1df61da201d867a642e9b prost = "0.14" snafu = "0.9" diff --git a/README.md b/README.md -index 9ceccdc4f3f6b5d13dc271be3d2659c7be400306..9d5a318d911ea8f8efba82a447275ebf6e88901f 100644 +index 9ceccdc4f3f6b5d13dc271be3d2659c7be400306..94de4d1c26d8ab869d1726e2b416d21f21661d6d 100644 --- a/README.md +++ b/README.md @@ -68,6 +68,7 @@ Based on the [liblance RFC](https://github.com/lance-format/lance/discussions/60 @@ -210,7 +210,7 @@ index 9ceccdc4f3f6b5d13dc271be3d2659c7be400306..9d5a318d911ea8f8efba82a447275ebf ## Multi-vector search -@@ -231,6 +232,29 @@ auto ds = lance::Dataset::open_with_session(session, "data.lance"); +@@ -231,6 +232,35 @@ auto ds = lance::Dataset::open_with_session(session, "data.lance"); auto stats = session.cache_stats(); ``` @@ -222,6 +222,12 @@ index 9ceccdc4f3f6b5d13dc271be3d2659c7be400306..9d5a318d911ea8f8efba82a447275ebf +for datasets that share the cache directory. Immutable data-file response metadata +uses a separate in-memory cache budget of one eighth of the configured data +memory capacity, avoiding a remote HEAD request on repeated range reads. ++Cache entries are isolated by the underlying object-store instance because the ++wrapper interface does not expose a complete backend identity. Datasets sharing ++the same live store can reuse entries; a new store instance or process restart ++starts a new cache namespace. Reopening the disk tier while that same store is ++still alive can recover its entries. Identical bucket/path names on different ++endpoints never share metadata, sizes, or data blocks. + +```cpp +lance::DataCacheOptions data_cache{ @@ -503,10 +509,10 @@ index cc1f87ce3f7dfe8dd33aeb6ab7f54d02e31a220a..76fd39ef9e456d1fab6978b6b2cb3c27 } diff --git a/src/foyer_data_cache.rs b/src/foyer_data_cache.rs new file mode 100644 -index 0000000000000000000000000000000000000000..c892311be04c4a2b21026954cb386f930a4aab98 +index 0000000000000000000000000000000000000000..cfc47f1528d11b6be068c09dd4703ff4308a9d16 --- /dev/null +++ b/src/foyer_data_cache.rs -@@ -0,0 +1,1411 @@ +@@ -0,0 +1,1620 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright The Lance Authors + @@ -518,7 +524,7 @@ index 0000000000000000000000000000000000000000..c892311be04c4a2b21026954cb386f93 +use std::ops::Range; +use std::path::Path as FsPath; +use std::sync::atomic::{AtomicU64, Ordering}; -+use std::sync::{Arc, Mutex, Weak}; ++use std::sync::{Arc, LazyLock, Mutex, Weak}; + +use async_trait::async_trait; +use bytes::{Bytes, BytesMut}; @@ -542,9 +548,32 @@ index 0000000000000000000000000000000000000000..c892311be04c4a2b21026954cb386f93 +use crate::runtime::block_on; +use crate::session::{LanceSession, session_new_with_data_cache_factory}; + -+const CACHE_KEY_VERSION: &str = "lance-data-v1"; ++const CACHE_KEY_VERSION: &str = "lance-data-v2"; +const FOYER_PAGE_SIZE: usize = 4096; + ++struct OriginNamespace { ++ origin: Weak, ++ namespace: uuid::Uuid, ++} ++ ++static ORIGIN_NAMESPACES: LazyLock>> = ++ LazyLock::new(|| Mutex::new(HashMap::new())); ++ ++fn origin_namespace(origin: &Arc) -> uuid::Uuid { ++ // store_prefix omits endpoint/credentials, and this interface exposes no stable ++ // backend identity. Only the same live origin may reuse metadata or data. ++ // Persist a random namespace, never an address that another process can reuse. ++ let mut namespaces = ORIGIN_NAMESPACES.lock().unwrap(); ++ namespaces.retain(|_, entry| entry.origin.strong_count() != 0); ++ namespaces ++ .entry(Arc::as_ptr(origin) as *const () as usize) ++ .or_insert_with(|| OriginNamespace { ++ origin: Arc::downgrade(origin), ++ namespace: uuid::Uuid::new_v4(), ++ }) ++ .namespace ++} ++ +/// Configuration for the optional Foyer data-file cache. +#[repr(C)] +#[derive(Clone, Copy, Debug)] @@ -893,7 +922,7 @@ index 0000000000000000000000000000000000000000..c892311be04c4a2b21026954cb386f93 + let original = self.cache.unwrap_store(original); + let reader = DataCacheReader { + cache: self.cache.clone(), -+ store_prefix: store_prefix.to_owned(), ++ store_prefix: format!("{}\0{store_prefix}", origin_namespace(&original)), + original: original.clone(), + statistics: self.statistics.clone(), + }; @@ -978,18 +1007,29 @@ index 0000000000000000000000000000000000000000..c892311be04c4a2b21026954cb386f93 + // HTTP responses carry transport extensions even for immutable files. + // Cache metadata independently; request-specific extensions are never replayed. + self.reader.cache.metadata.insert( -+ metadata_key, ++ metadata_key.clone(), + (result.meta.clone(), result.attributes.clone()), + ); + (result.meta, result.attributes, result.extensions) + }; + let object_size = metadata.size; -+ self.reader.cache.cache.insert( ++ let size_bytes = object_size.to_le_bytes(); ++ // WriteOnInsertion enqueues disk I/O even for an identical value. Look in ++ // both tiers so warm reads, including recovered entries, remain read-only. ++ let size_is_cached = match self.reader.cache.cache.get(&metadata_key).await { ++ Ok(Some(entry)) => entry.value().as_ref() == size_bytes, ++ Ok(None) => false, ++ Err(error) => { ++ log::warn!("Foyer data-cache size lookup failed for {location}: {error}"); ++ false ++ } ++ }; ++ if !size_is_cached { + self.reader + .cache -+ .size_key(&self.reader.store_prefix, location), -+ Bytes::copy_from_slice(&object_size.to_le_bytes()), -+ ); ++ .cache ++ .insert(metadata_key, Bytes::copy_from_slice(&size_bytes)); ++ } + + let range = match options.range.clone() { + Some(requested) => match requested.as_range(object_size) { @@ -1372,8 +1412,14 @@ index 0000000000000000000000000000000000000000..c892311be04c4a2b21026954cb386f93 + (scope.wrap("memory://test", original), scope) + } + -+ #[tokio::test] -+ async fn cached_http_ranges_do_not_repeat_head_requests() { ++ async fn http_store( ++ data: Bytes, ++ ) -> ( ++ Arc, ++ Arc, ++ Arc, ++ tokio::task::JoinHandle<()>, ++ ) { + use object_store::aws::AmazonS3Builder; + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + @@ -1383,7 +1429,6 @@ index 0000000000000000000000000000000000000000..c892311be04c4a2b21026954cb386f93 + let gets = Arc::new(AtomicU64::new(0)); + let server_heads = heads.clone(); + let server_gets = gets.clone(); -+ let data = Bytes::from((0..8192).map(|value| value as u8).collect::>()); + let server_data = data.clone(); + let server = tokio::spawn(async move { + loop { @@ -1446,6 +1491,13 @@ index 0000000000000000000000000000000000000000..c892311be04c4a2b21026954cb386f93 + .build() + .unwrap(), + ); ++ (original, heads, gets, server) ++ } ++ ++ #[tokio::test] ++ async fn cached_http_ranges_do_not_repeat_head_requests() { ++ let data = Bytes::from((0..8192).map(|value| value as u8).collect::>()); ++ let (original, heads, gets, server) = http_store(data.clone()).await; + let directory = tempfile::tempdir().unwrap(); + let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) + .await @@ -1517,6 +1569,126 @@ index 0000000000000000000000000000000000000000..c892311be04c4a2b21026954cb386f93 + } + + #[tokio::test] ++ async fn same_bucket_and_path_on_distinct_endpoints_remain_isolated() { ++ let (first, _, _, first_server) = http_store(Bytes::from(vec![11; 8192])).await; ++ let (second, second_heads, second_gets, second_server) = ++ http_store(Bytes::from(vec![29; 4096])).await; ++ let directory = tempfile::tempdir().unwrap(); ++ let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) ++ .await ++ .unwrap(); ++ let path = Path::from("table.lance/data/shared.lance"); ++ let first = cache.create_scope().wrap("s3$example-bucket", first); ++ let second = cache.create_scope().wrap("s3$example-bucket", second); ++ assert_eq!( ++ first.get(&path).await.unwrap().bytes().await.unwrap(), ++ Bytes::from(vec![11; 8192]) ++ ); ++ let result = second.get(&path).await.unwrap(); ++ assert_eq!( ++ result.meta.size, 4096, ++ "metadata must belong to the second endpoint" ++ ); ++ assert_eq!(result.bytes().await.unwrap(), Bytes::from(vec![29; 4096])); ++ assert_eq!(second_heads.load(Ordering::SeqCst), 1); ++ assert_eq!(second_gets.load(Ordering::SeqCst), 1); ++ first_server.abort(); ++ second_server.abort(); ++ } ++ ++ #[tokio::test] ++ async fn batched_ranges_do_not_reuse_another_store_or_mask_not_found() { ++ let directory = tempfile::tempdir().unwrap(); ++ let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) ++ .await ++ .unwrap(); ++ let first = Arc::new(InMemory::new()); ++ let second = Arc::new(InMemory::new()); ++ let missing = Arc::new(InMemory::new()); ++ let path = Path::from("table.lance/data/shared.lance"); ++ first ++ .put(&path, Bytes::from(vec![11; 8192]).into()) ++ .await ++ .unwrap(); ++ second ++ .put(&path, Bytes::from(vec![29; 8192]).into()) ++ .await ++ .unwrap(); ++ let (first, _) = wrap_for_test(&cache, first); ++ let (second, _) = wrap_for_test(&cache, second); ++ let (missing, _) = wrap_for_test(&cache, missing); ++ assert_eq!( ++ first ++ .get_ranges(&path, &[0..4096, 4096..8192]) ++ .await ++ .unwrap()[0], ++ Bytes::from(vec![11; 4096]) ++ ); ++ assert_eq!( ++ second ++ .get_ranges(&path, &[0..4096, 4096..8192]) ++ .await ++ .unwrap()[0], ++ Bytes::from(vec![29; 4096]) ++ ); ++ assert!(matches!( ++ missing.get(&path).await, ++ Err(object_store::Error::NotFound { .. }) ++ )); ++ assert!(matches!( ++ missing.get_ranges(&path, &[0..16, 16..32]).await, ++ Err(object_store::Error::NotFound { .. }) ++ )); ++ } ++ ++ #[tokio::test] ++ async fn warm_range_reads_do_not_rewrite_size_entries_to_disk() { ++ let directory = tempfile::tempdir().unwrap(); ++ let original = Arc::new(InMemory::new()); ++ let path = Path::from("table.lance/data/sample.lance"); ++ original ++ .put(&path, Bytes::from(vec![7; 8192]).into()) ++ .await ++ .unwrap(); ++ let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) ++ .await ++ .unwrap(); ++ let (wrapped, _) = wrap_for_test(&cache, original.clone()); ++ assert_eq!( ++ wrapped ++ .get(&path) ++ .await ++ .unwrap() ++ .bytes() ++ .await ++ .unwrap() ++ .len(), ++ 8192 ++ ); ++ drop(wrapped); ++ cache.cache.close().await.unwrap(); ++ drop(cache); ++ ++ let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) ++ .await ++ .unwrap(); ++ let (wrapped, _) = wrap_for_test(&cache, original); ++ for _ in 0..5 { ++ assert_eq!( ++ wrapped.get_range(&path, 100..200).await.unwrap(), ++ Bytes::from(vec![7; 100]) ++ ); ++ } ++ // Drain the real disk writer so asynchronous enqueues cannot hide behind the assertion. ++ cache.cache.close().await.unwrap(); ++ assert_eq!( ++ cache.cache.statistics().disk_write_bytes(), ++ 0, ++ "warm reads must not enqueue size records for disk storage" ++ ); ++ } ++ ++ #[tokio::test] + async fn caches_only_immutable_data_file_ranges() { + let directory = tempfile::tempdir().unwrap(); + let cache = FoyerDataCache::try_new(directory.path(), 256 * 1024, 1024 * 1024, 64 * 1024) @@ -1794,6 +1966,49 @@ index 0000000000000000000000000000000000000000..c892311be04c4a2b21026954cb386f93 + } + + #[tokio::test] ++ async fn recovered_disk_entries_do_not_outlive_their_origin_identity() { ++ let directory = tempfile::tempdir().unwrap(); ++ let path = Path::from("table.lance/data/shared.lance"); ++ let original = Arc::new(InMemory::new()); ++ let weak = Arc::downgrade(&original); ++ original ++ .put(&path, Bytes::from(vec![11; 8192]).into()) ++ .await ++ .unwrap(); ++ let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) ++ .await ++ .unwrap(); ++ let (wrapped, _) = wrap_for_test(&cache, original); ++ assert_eq!(wrapped.get_range(&path, 0..8192).await.unwrap().len(), 8192); ++ drop(wrapped); ++ assert!( ++ weak.upgrade().is_none(), ++ "the namespace registry must not retain stores" ++ ); ++ cache.cache.close().await.unwrap(); ++ drop(cache); ++ ++ let recovered = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 4096) ++ .await ++ .unwrap(); ++ let replacement = Arc::new(InMemory::new()); ++ replacement ++ .put(&path, Bytes::from(vec![29; 8192]).into()) ++ .await ++ .unwrap(); ++ let (wrapped, statistics) = wrap_for_test(&recovered, replacement); ++ assert_eq!( ++ wrapped ++ .get_ranges(&path, &[0..4096, 4096..8192]) ++ .await ++ .unwrap(), ++ vec![Bytes::from(vec![29; 4096]); 2] ++ ); ++ assert_eq!(statistics.snapshot().bytes_read_from_cache, 0); ++ assert_eq!(statistics.snapshot().bytes_read_from_remote, 8192); ++ } ++ ++ #[tokio::test] + async fn dataset_scopes_share_cache_without_sharing_statistics() { + let directory = tempfile::tempdir().unwrap(); + let cache = FoyerDataCache::try_new(directory.path(), 128 * 1024, 1024 * 1024, 64 * 1024) @@ -2029,7 +2244,7 @@ index 1971510de4a13ee6f04fc39c159b088e78e21d55..ba51c87a8cebb6897c48fb55509f71da // SAFETY: `out_dataset` is non-NULL (checked above) and the caller // guarantees it points to caller-owned, writable storage of size diff --git a/tests/c_api_test.rs b/tests/c_api_test.rs -index 1a25e34a8ca50623812ffedaa7bb5a8c54f7850e..b63a26a7c2135026e7ccf761565cc7746fde79f5 100644 +index 1a25e34a8ca50623812ffedaa7bb5a8c54f7850e..c341859e9434d4578b2833eca6ec313a28f31f32 100644 --- a/tests/c_api_test.rs +++ b/tests/c_api_test.rs @@ -100,10 +100,83 @@ fn create_large_dataset(num_rows: i32) -> (tempfile::TempDir, String) { @@ -2116,7 +2331,7 @@ index 1a25e34a8ca50623812ffedaa7bb5a8c54f7850e..b63a26a7c2135026e7ccf761565cc774 #[derive(Default)] struct CapturedScanStatistics { calls: usize, -@@ -445,6 +518,120 @@ fn test_shared_session_rejects_null_inputs() { +@@ -445,6 +518,129 @@ fn test_shared_session_rejects_null_inputs() { } } @@ -2141,6 +2356,12 @@ index 1a25e34a8ca50623812ffedaa7bb5a8c54f7850e..b63a26a7c2135026e7ccf761565cc774 + !cached_dataset.is_null(), + "second dataset open should succeed" + ); ++ // A fresh open may create a different underlying store. Warm its isolated ++ // namespace before removing origin files; only the same live store can reuse ++ // entries because bucket/path alone does not identify a storage backend. ++ assert_eq!(scanned_row_count(cached_dataset), 20_000); ++ let reopened_statistics = data_cache_statistics(cached_dataset); ++ assert!(reopened_statistics.bytes_read_from_remote > 0); + unsafe { lance_session_close(session) }; + + for entry in std::fs::read_dir(tmp.path().join("large_ds/data")).unwrap() { @@ -2148,8 +2369,11 @@ index 1a25e34a8ca50623812ffedaa7bb5a8c54f7850e..b63a26a7c2135026e7ccf761565cc774 + } + assert_eq!(scanned_row_count(cached_dataset), 20_000); + let cached_statistics = data_cache_statistics(cached_dataset); -+ assert!(cached_statistics.bytes_read_from_cache > 0); -+ assert_eq!(cached_statistics.bytes_read_from_remote, 0); ++ assert!(cached_statistics.bytes_read_from_cache > reopened_statistics.bytes_read_from_cache); ++ assert_eq!( ++ cached_statistics.bytes_read_from_remote, ++ reopened_statistics.bytes_read_from_remote ++ ); + + assert_eq!(data_cache_statistics(dataset), first_statistics); + @@ -2237,7 +2461,7 @@ index 1a25e34a8ca50623812ffedaa7bb5a8c54f7850e..b63a26a7c2135026e7ccf761565cc774 #[test] fn test_open_nonexistent() { let c_uri = c_str("memory://nonexistent_dataset_xyz"); -@@ -3318,6 +3505,40 @@ fn test_dataset_restore_to_prior_version() { +@@ -3318,6 +3514,40 @@ fn test_dataset_restore_to_prior_version() { unsafe { lance_dataset_close(ds) }; } diff --git a/thirdparty/test/lance-prefilter-patch-test.sh b/thirdparty/test/lance-prefilter-patch-test.sh index ecf9c0c718968f..e3dabff6c176d6 100755 --- a/thirdparty/test/lance-prefilter-patch-test.sh +++ b/thirdparty/test/lance-prefilter-patch-test.sh @@ -21,6 +21,19 @@ set -euo pipefail ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." &>/dev/null && pwd)" ARCHIVE_DIR="${1:?Usage: $0 directory-containing-the-pinned-lance-c-archive}" ARCHIVE_DIR="$(cd "${ARCHIVE_DIR}" && pwd)" +# Platform definitions must remain safe when this harness enables nounset. +for test_arch in x86_64 arm64; do + bash -eu -c ' + uname() { if [[ "$1" == -s ]]; then echo Darwin; else echo "$TEST_ARCH"; fi; } + unset ARROW_ADBC_FLIGHTSQL_SOURCE + TP_DIR="$1" + TEST_ARCH="$2" + source "$TP_DIR/vars.sh" + [[ " ${TP_ARCHIVES[*]} " != *" ARROW_ADBC_FLIGHTSQL "* ]] + ' _ "${ROOT}" "${test_arch}" +done +echo "PASS: macOS platform definitions under nounset" + TP_DIR="${ROOT}" # Load only repository-owned definitions, never extracted dependency code. source "${ROOT}/vars.sh"