From 18ef6c2e15168cd1dba0c8a9860dddd4a491fba2 Mon Sep 17 00:00:00 2001 From: Likith B Date: Tue, 8 Sep 2026 19:01:27 +0530 Subject: [PATCH 1/3] MB-73803: Introducing Numeric Index Section --- document/field_numeric_v2.go | 159 +++++++ index/scorch/snapshot_index.go | 207 +++++++++ index_test.go | 12 +- mapping.go | 4 + mapping/document.go | 10 +- mapping/field.go | 35 +- numericv2/encode.go | 88 ++++ numericv2/encode_test.go | 223 ++++++++++ numericv2/query.go | 69 +++ search/query/numeric_range_v2.go | 110 +++++ search/query/query.go | 11 + search/searcher/search_numeric_v2.go | 128 ++++++ search_numeric_v2_test.go | 610 +++++++++++++++++++++++++++ util/bitset.go | 295 ++++++++++++- util/bitset_iterator_test.go | 376 +++++++++++++++++ util/bitset_pool_test.go | 116 +++++ 16 files changed, 2422 insertions(+), 31 deletions(-) create mode 100644 document/field_numeric_v2.go create mode 100644 numericv2/encode.go create mode 100644 numericv2/encode_test.go create mode 100644 numericv2/query.go create mode 100644 search/query/numeric_range_v2.go create mode 100644 search/searcher/search_numeric_v2.go create mode 100644 search_numeric_v2_test.go create mode 100644 util/bitset_iterator_test.go create mode 100644 util/bitset_pool_test.go diff --git a/document/field_numeric_v2.go b/document/field_numeric_v2.go new file mode 100644 index 000000000..56e2eccfe --- /dev/null +++ b/document/field_numeric_v2.go @@ -0,0 +1,159 @@ +// Copyright (c) 2026 Couchbase, Inc. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package document + +import ( + "fmt" + "reflect" + + "github.com/blevesearch/bleve/v2/numeric" + "github.com/blevesearch/bleve/v2/numericv2" + "github.com/blevesearch/bleve/v2/size" + index "github.com/blevesearch/bleve_index_api" +) + +var reflectStaticSizeNumericV2Field int + +func init() { + var f NumericV2Field + reflectStaticSizeNumericV2Field = int(reflect.TypeOf(f).Size()) +} + +// DefaultNumericV2IndexingOptions mirrors the v1 numeric defaults, minus +// IncludeInAll: a number_v2 field produces no tokens, so it can never +// participate in the _all composite field. +const DefaultNumericV2IndexingOptions = index.StoreField | index.IndexField | index.DocValues + +// NumericV2Field is a numeric field indexed into the number_v2 section rather +// than as prefix-coded terms in the inverted index. It carries the value in two +// encodings because its two consumers need different things: the section's +// sorted search array wants the sortable uint64, while the sort and facet paths +// visit doc values as prefix-coded terms. +type NumericV2Field struct { + name string + arrayPositions []uint64 + options index.FieldIndexingOptions + value numeric.PrefixCoded + numPlainTextBytes uint64 +} + +func (n *NumericV2Field) Size() int { + return reflectStaticSizeNumericV2Field + size.SizeOfPtr + + len(n.name) + + len(n.arrayPositions)*size.SizeOfUint64 + + len(n.value) +} + +func (n *NumericV2Field) Name() string { + return n.name +} + +func (n *NumericV2Field) ArrayPositions() []uint64 { + return n.arrayPositions +} + +func (n *NumericV2Field) Options() index.FieldIndexingOptions { + return n.options +} + +func (n *NumericV2Field) EncodedFieldType() byte { + return 'm' +} + +// Analyze is a no-op: this field type is not tokenized. The number_v2 section +// reads the value directly, and nothing about it needs analysis. +func (n *NumericV2Field) Analyze() { +} + +func (n *NumericV2Field) AnalyzedLength() int { + return 0 +} + +func (n *NumericV2Field) AnalyzedTokenFrequencies() index.TokenFrequencies { + return nil +} + +// Value returns the stored-field representation, which is the same +// prefix-coded encoding a v1 NumericField stores. Keeping the two identical is +// what lets the stored-field decode path reuse NewNumericFieldFromBytes. +func (n *NumericV2Field) Value() []byte { + return n.value +} + +// DocValueTerm returns the prefix-coded, zero-shift term written to this +// field's doc values. The sort and facet paths validate doc value bytes as +// prefix-coded and keep only shift-zero terms, so this encoding is required +// rather than incidental: handing them the raw sortable uint64 would make +// every document sort and facet as missing. +func (n *NumericV2Field) DocValueTerm() []byte { + return n.value +} + +// SortableValue returns the value encoded as a uint64 whose unsigned ordering +// matches the float64 ordering of the original number. +func (n *NumericV2Field) SortableValue() uint64 { + i64, err := n.value.Int64() + if err != nil { + return 0 + } + return numericv2.EncodeInt64(i64) +} + +func (n *NumericV2Field) Number() (float64, error) { + i64, err := n.value.Int64() + if err != nil { + return 0.0, err + } + return numeric.Int64ToFloat64(i64), nil +} + +func (n *NumericV2Field) GoString() string { + return fmt.Sprintf("&document.NumericV2Field{Name:%s, Options: %s, Value: %s}", + n.name, n.options, n.value) +} + +func (n *NumericV2Field) NumPlainTextBytes() uint64 { + return n.numPlainTextBytes +} + +func NewNumericV2FieldFromBytes(name string, arrayPositions []uint64, value []byte) *NumericV2Field { + return &NumericV2Field{ + name: name, + arrayPositions: arrayPositions, + value: value, + options: DefaultNumericV2IndexingOptions, + numPlainTextBytes: uint64(len(value)), + } +} + +func NewNumericV2Field(name string, arrayPositions []uint64, number float64) *NumericV2Field { + return NewNumericV2FieldWithIndexingOptions(name, arrayPositions, number, + DefaultNumericV2IndexingOptions) +} + +func NewNumericV2FieldWithIndexingOptions(name string, arrayPositions []uint64, + number float64, options index.FieldIndexingOptions) *NumericV2Field { + numberInt64 := numeric.Float64ToInt64(number) + prefixCoded := numeric.MustNewPrefixCodedInt64(numberInt64, 0) + return &NumericV2Field{ + name: name, + arrayPositions: arrayPositions, + value: prefixCoded, + options: options, + // not correct, just a place holder until we revisit how fields are + // represented and can fix this better + numPlainTextBytes: uint64(8), + } +} diff --git a/index/scorch/snapshot_index.go b/index/scorch/snapshot_index.go index a363ecb71..39275cdaf 100644 --- a/index/scorch/snapshot_index.go +++ b/index/scorch/snapshot_index.go @@ -28,6 +28,7 @@ import ( "github.com/RoaringBitmap/roaring/v2" "github.com/blevesearch/bleve/v2/document" geov2 "github.com/blevesearch/bleve/v2/geov2" + "github.com/blevesearch/bleve/v2/numericv2" "github.com/blevesearch/bleve/v2/size" "github.com/blevesearch/bleve/v2/util" index "github.com/blevesearch/bleve_index_api" @@ -53,6 +54,8 @@ type asynchSegmentResult struct { var reflectStaticSizeIndexSnapshot int var reflectStaticSizeIndexSnapshotGeoShapeV2Reader int +var reflectStaticSizeIndexSnapshotNumericV2Reader int +var reflectStaticSizeBitsetIterator int var reflectStaticSizeRoaringIntIterator int func init() { @@ -69,6 +72,10 @@ func init() { } var gcr IndexSnapshotGeoShapeV2Reader reflectStaticSizeIndexSnapshotGeoShapeV2Reader = int(reflect.TypeOf(gcr).Size()) + var ncr IndexSnapshotNumericV2Reader + reflectStaticSizeIndexSnapshotNumericV2Reader = int(reflect.TypeOf(ncr).Size()) + var bsi util.BitsetIterator + reflectStaticSizeBitsetIterator = int(reflect.TypeOf(bsi).Size()) var rip roaring.IntIterator reflectStaticSizeRoaringIntIterator = int(reflect.TypeOf(rip).Size()) } @@ -556,6 +563,8 @@ func (is *IndexSnapshot) Document(id string) (rv index.Document, err error) { rvd.AddField(document.NewGeoPointFieldFromBytes(name, arrayPos, value)) case 's': rvd.AddField(document.NewGeoShapeFieldFromBytes(name, arrayPos, value)) + case 'm': + rvd.AddField(document.NewNumericV2FieldFromBytes(name, arrayPos, value)) } return true @@ -1502,3 +1511,201 @@ func (g *IndexSnapshotGeoShapeV2Reader) Size() int { return rv } + +func (i *IndexSnapshot) NumericV2FieldReader(ctx context.Context, field string) ( + index.NumericV2FieldReader, error) { + + rv := &IndexSnapshotNumericV2Reader{ + field: field, + hits: make([]*util.Bitset, len(i.segment)), + // the zero value of BitsetIterator is a valid empty iterator, so a + // segment with no data for the field needs no special casing below + iterators: make([]util.BitsetIterator, len(i.segment)), + counts: make([]uint64, len(i.segment)), + snapshot: i, + } + + return rv, nil +} + +type IndexSnapshotNumericV2Reader struct { + field string + + // hits holds the per-segment result bitsets; iterators alias their words, + // so neither survives Close. + hits []*util.Bitset + iterators []util.BitsetIterator + // counts caches each segment's cardinality, computed once during Search so + // that Count is not a repeated scan. + counts []uint64 + + segmentOffset int + + snapshot *IndexSnapshot +} + +// Search evaluates the numeric range across all segments in the index snapshot. +// +// Every hit is materialised up front rather than streamed, because the segment +// stores values in value order while a searcher must emit document order; there +// is no streaming formulation. +func (n *IndexSnapshotNumericV2Reader) Search(min, max *float64, + inclusiveMin, inclusiveMax *bool) error { + + numSegments := len(n.snapshot.segment) + // the query holds only the encoded bounds, so it is safe to share + // across the per-segment goroutines + query := numericv2.NewRangeQuery(min, max, inclusiveMin, inclusiveMax) + + var wg sync.WaitGroup + wg.Add(numSegments) + + var errm sync.Mutex + var err error + // search each segment concurrently + for i := 0; i < numSegments; i++ { + go func(segID int) { + defer wg.Done() + err2 := n.searchSeg(segID, query) + if err2 != nil { + errm.Lock() + if err == nil { + err = err2 + } + errm.Unlock() + } + }(i) + } + wg.Wait() + + return err +} + +// searchSeg evaluates the range against a single segment in the snapshot. +func (n *IndexSnapshotNumericV2Reader) searchSeg(segID int, + query *numericv2.Query) error { + + snapshot := n.snapshot.segment[segID] + numSeg, ok := snapshot.segment.(segment.NumericV2Segment) + if !ok { + return nil + } + + data, err := numSeg.NumericV2Data(n.field) + if err != nil { + return err + } + // return if the segment has no numeric data for the field + if data == nil { + return nil + } + // release the reference on the segment's cached arrays once the evaluation + // is done, so that the cache is free to evict them + defer data.Close() + + // the stored doc numbers are already in segment space, so the snapshot's + // deleted bitmap applies directly with no translation + hits := query.Evaluate(data, snapshot.deleted, snapshot.segment.Count()) + + // the bitset is already the ideal representation for a dense result: it is + // iterated in place rather than converted into a roaring bitmap first, + // which would cost a rebuild and then an interface-dispatched container + // walk per hit to recover what the words already hold. + n.hits[segID] = hits + n.iterators[segID] = hits.Iterator() + n.counts[segID] = uint64(hits.Count()) + + return nil +} + +// Next returns the next hit across all segments in the index snapshot. +func (n *IndexSnapshotNumericV2Reader) Next(preAlloced *index.NumericV2FieldDoc) ( + *index.NumericV2FieldDoc, error) { + rv := preAlloced + if rv == nil { + rv = &index.NumericV2FieldDoc{} + } + + for n.segmentOffset < len(n.iterators) { + docNum, ok := n.iterators[n.segmentOffset].Next() + if !ok { + n.segmentOffset++ + continue + } + + rv.ID = index.NewIndexInternalID(rv.ID, + uint64(docNum)+n.snapshot.offsets[n.segmentOffset]) + return rv, nil + } + + return nil, nil +} + +// Advance moves the reader to the first hit at or beyond the specified document. +func (n *IndexSnapshotNumericV2Reader) Advance(ID index.IndexInternalID, + preAlloced *index.NumericV2FieldDoc) (*index.NumericV2FieldDoc, error) { + rv := preAlloced + if rv == nil { + rv = &index.NumericV2FieldDoc{} + } + + num := ID.Value() + + segIdx, localDocNum := n.snapshot.segmentIndexAndLocalDocNumFromGlobal(num) + if segIdx >= len(n.iterators) { + return nil, fmt.Errorf("error advancing to doc number %d, segment "+ + "index %d out of bounds", num, segIdx) + } + + if n.segmentOffset > segIdx { + return nil, fmt.Errorf("error advancing to doc number %d, segment "+ + "index %d is less than current segment offset %d", num, segIdx, n.segmentOffset) + } + + n.segmentOffset = segIdx + n.iterators[n.segmentOffset].AdvanceTo(int(localDocNum)) + + return n.Next(rv) +} + +// Close drops the reader's per-segment results. The segment-level arrays the +// search read from are not touched: those are owned and evicted by the +// segment's own cache, and were already released at the end of searchSeg. +func (n *IndexSnapshotNumericV2Reader) Close() error { + // the iterators alias the bitsets' words, so the pooled memory can only go + // back once iteration is finished -- which is exactly what Close means here + for _, hits := range n.hits { + if hits != nil { + hits.Release() + } + } + // drop the references so a stray call after Close cannot read them + n.hits = nil + n.iterators = nil + return nil +} + +// Count returns the total number of matching documents across all segments. +func (n *IndexSnapshotNumericV2Reader) Count() uint64 { + var rv uint64 + for _, count := range n.counts { + rv += count + } + return rv +} + +// Size returns the estimated size in bytes of the +// IndexSnapshotNumericV2Reader, including its postings and iterators. +func (n *IndexSnapshotNumericV2Reader) Size() int { + rv := reflectStaticSizeIndexSnapshotNumericV2Reader + size.SizeOfPtr + + len(n.field) + size.SizeOfInt + + for _, hits := range n.hits { + rv += hits.SizeInBytes() + } + + rv += reflectStaticSizeBitsetIterator * len(n.iterators) + rv += size.SizeOfUint64 * len(n.counts) + + return rv +} diff --git a/index_test.go b/index_test.go index 203f3f459..4628045e6 100644 --- a/index_test.go +++ b/index_test.go @@ -612,9 +612,9 @@ func TestBytesRead(t *testing.T) { stats, _ := idx.StatsMap()["index"].(map[string]interface{}) prevBytesRead, _ := stats["num_bytes_read_at_query_time"].(uint64) - expectedBytesRead := uint64(21574) + expectedBytesRead := uint64(21984) if supportForVectorSearch { - expectedBytesRead = 21984 + expectedBytesRead = 22394 } if prevBytesRead != expectedBytesRead && res.Cost == prevBytesRead { @@ -770,9 +770,9 @@ func TestBytesReadStored(t *testing.T) { stats, _ := idx.StatsMap()["index"].(map[string]interface{}) bytesRead, _ := stats["num_bytes_read_at_query_time"].(uint64) - expectedBytesRead := uint64(11435) + expectedBytesRead := uint64(11845) if supportForVectorSearch { - expectedBytesRead = 11845 + expectedBytesRead = 12255 } if bytesRead != expectedBytesRead && bytesRead == res.Cost { @@ -847,9 +847,9 @@ func TestBytesReadStored(t *testing.T) { stats, _ = idx1.StatsMap()["index"].(map[string]interface{}) bytesRead, _ = stats["num_bytes_read_at_query_time"].(uint64) - expectedBytesRead = uint64(3622) + expectedBytesRead = uint64(4032) if supportForVectorSearch { - expectedBytesRead = 4032 + expectedBytesRead = 4442 } if bytesRead != expectedBytesRead && bytesRead == res.Cost { diff --git a/mapping.go b/mapping.go index ba77f15ea..a88c76b09 100644 --- a/mapping.go +++ b/mapping.go @@ -92,6 +92,10 @@ func NewGeoShapeV2FieldMapping() *mapping.FieldMapping { return mapping.NewGeoShapeV2FieldMapping() } +func NewNumberV2FieldMapping() *mapping.FieldMapping { + return mapping.NewNumberV2FieldMapping() +} + func NewIPFieldMapping() *mapping.FieldMapping { return mapping.NewIPFieldMapping() } diff --git a/mapping/document.go b/mapping/document.go index 943056b78..0966eea51 100644 --- a/mapping/document.go +++ b/mapping/document.go @@ -104,7 +104,15 @@ func (dm *DocumentMapping) Validate(cache *registry.Cache, func validateFieldType(field *FieldMapping) error { switch field.Type { - case "text", "datetime", "number", "boolean", "geopoint", "geoshape", "geoshape_v2", "IP": + case "text", "datetime", "number", "number_v2", "boolean", "geopoint", + "geoshape", "geoshape_v2", "IP": + if field.Type == "number_v2" && field.IncludeInAll { + // a number_v2 field produces no tokens, so it can never take part + // in the _all composite field; reject at index-definition time + // rather than silently matching nothing at query time + return fmt.Errorf("field: '%s', type 'number_v2' cannot be included in _all", + field.Name) + } return nil default: return fmt.Errorf("field: '%s', unknown field type: '%s'", diff --git a/mapping/field.go b/mapping/field.go index bf8d4d452..8be2be603 100644 --- a/mapping/field.go +++ b/mapping/field.go @@ -213,6 +213,20 @@ func NewGeoShapeV2FieldMapping() *FieldMapping { } } +// NewNumberV2FieldMapping returns a default field mapping for numbers indexed +// into the number_v2 section. The defaults match NewNumericFieldMapping, except +// that IncludeInAll is false and cannot be enabled: the field produces no +// tokens, so it could never contribute to the _all composite field. +func NewNumberV2FieldMapping() *FieldMapping { + return &FieldMapping{ + Type: "number_v2", + Store: true, + Index: true, + IncludeInAll: false, + DocValues: true, + } +} + // NewIPFieldMapping returns a default field mapping for IP points func NewIPFieldMapping() *FieldMapping { return &FieldMapping{ @@ -282,14 +296,21 @@ func (fm *FieldMapping) processString(propertyValueString string, pathString str func (fm *FieldMapping) processFloat64(propertyValFloat float64, pathString string, path []string, indexes []uint64, context *walkContext) { fieldName := getFieldName(pathString, path, fm) - if fm.Type == "number" { - options := fm.Options() - field := document.NewNumericFieldWithIndexingOptions(fieldName, indexes, propertyValFloat, options) - context.doc.AddField(field) + var field document.Field + switch fm.Type { + case "number": + field = document.NewNumericFieldWithIndexingOptions(fieldName, indexes, + propertyValFloat, fm.Options()) + case "number_v2": + field = document.NewNumericV2FieldWithIndexingOptions(fieldName, indexes, + propertyValFloat, fm.Options()) + default: + return + } + context.doc.AddField(field) - if !fm.IncludeInAll { - context.excludedFromAll = append(context.excludedFromAll, fieldName) - } + if !fm.IncludeInAll { + context.excludedFromAll = append(context.excludedFromAll, fieldName) } } diff --git a/numericv2/encode.go b/numericv2/encode.go new file mode 100644 index 000000000..69298a73a --- /dev/null +++ b/numericv2/encode.go @@ -0,0 +1,88 @@ +// Copyright (c) 2026 Couchbase, Inc. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// Package numericv2 holds the encoding and range evaluation for number_v2 +// fields, which are indexed as a single sorted array of values per segment +// rather than as prefix-coded terms in the inverted index. +package numericv2 + +import ( + "math" + + "github.com/blevesearch/bleve/v2/numeric" +) + +// signBit lifts an order-preserving int64 into an order-preserving uint64. +// Because it is addition of 2^63 modulo 2^64, it commutes with the +1 and -1 +// steps Bounds applies for exclusive endpoints. +const signBit = uint64(1) << 63 + +// EncodeInt64 maps a sortable int64, as produced by numeric.Float64ToInt64, to +// a uint64 whose unsigned ordering matches. This is the same transform +// numeric.PrefixCoded applies internally before splitting into 7-bit bytes. +func EncodeInt64(i int64) uint64 { + return uint64(i) ^ signBit +} + +// Encode maps a float64 to a uint64 whose unsigned ordering matches the +// float64 ordering of the input. +func Encode(f float64) uint64 { + return EncodeInt64(numeric.Float64ToInt64(f)) +} + +// Decode is the inverse of Encode. +func Decode(v uint64) float64 { + return numeric.Int64ToFloat64(int64(v ^ signBit)) +} + +// Bounds converts a query range into the inclusive uint64 interval [lo, hi] to +// scan. It deliberately mirrors searcher.NewNumericRangeSearcher step for step: +// an absent endpoint becomes the corresponding infinity, min is inclusive by +// default and max is not, and the adjustments for exclusive endpoints are +// guarded at the int64 extremes so they cannot wrap. +// +// A ±1 step in this space moves to the adjacent representable float64, so +// exclusive endpoints here are exact rather than approximate. When the range is +// empty, lo comes back greater than hi. +func Bounds(min, max *float64, inclusiveMin, inclusiveMax *bool) (lo, hi uint64) { + // account for unbounded edges + if min == nil { + negInf := math.Inf(-1) + min = &negInf + } + if max == nil { + inf := math.Inf(1) + max = &inf + } + if inclusiveMin == nil { + defaultInclusiveMin := true + inclusiveMin = &defaultInclusiveMin + } + if inclusiveMax == nil { + defaultInclusiveMax := false + inclusiveMax = &defaultInclusiveMax + } + + minInt64 := numeric.Float64ToInt64(*min) + if !*inclusiveMin && minInt64 != math.MaxInt64 { + minInt64++ + } + + maxInt64 := numeric.Float64ToInt64(*max) + if !*inclusiveMax && maxInt64 != math.MinInt64 { + maxInt64-- + } + + return EncodeInt64(minInt64), EncodeInt64(maxInt64) +} diff --git a/numericv2/encode_test.go b/numericv2/encode_test.go new file mode 100644 index 000000000..dcc3dbd78 --- /dev/null +++ b/numericv2/encode_test.go @@ -0,0 +1,223 @@ +// Copyright (c) 2026 Couchbase, Inc. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package numericv2 + +import ( + "math" + "sort" + "testing" + + "github.com/blevesearch/bleve/v2/numeric" +) + +// TestEncodeIsMonotone checks that Encode preserves float64 ordering across the +// awkward regions: negatives, zero, denormals and the infinities. +func TestEncodeIsMonotone(t *testing.T) { + vals := []float64{ + math.Inf(-1), + -math.MaxFloat64, + -1e300, -1e10, -1.5, -1, -0.5, + -math.SmallestNonzeroFloat64, + 0, + math.SmallestNonzeroFloat64, + 0.5, 1, 1.5, 1e10, 1e300, + math.MaxFloat64, + math.Inf(1), + } + + if !sort.SliceIsSorted(vals, func(i, j int) bool { return vals[i] < vals[j] }) { + t.Fatal("test input is not sorted") + } + + for i := 1; i < len(vals); i++ { + prev, cur := Encode(vals[i-1]), Encode(vals[i]) + if prev >= cur { + t.Fatalf("Encode not monotone at %v -> %v: %d >= %d", + vals[i-1], vals[i], prev, cur) + } + } + + // -0.0 and +0.0 compare equal as floats but are distinct bit patterns; + // what matters is that neither breaks ordering against its neighbours + if Encode(math.Copysign(0, -1)) > Encode(0) { + t.Fatal("negative zero encodes above positive zero") + } +} + +func TestEncodeDecodeRoundTrip(t *testing.T) { + for _, f := range []float64{-1e300, -1.5, -1, 0, 0.5, 1, 42, 1e300} { + if got := Decode(Encode(f)); got != f { + t.Fatalf("round trip of %v gave %v", f, got) + } + } +} + +func f64(v float64) *float64 { return &v } +func b(v bool) *bool { return &v } + +// TestBoundsMatchesInvertedPath is the load-bearing test for the encoding: it +// recomputes the int64 bounds exactly the way NewNumericRangeSearcher does and +// requires Bounds to agree after the sign-bit lift. If these ever diverge, the +// two numeric paths silently return different hits for the same query. +func TestBoundsMatchesInvertedPath(t *testing.T) { + // mirror of the arithmetic at the top of NewNumericRangeSearcher + reference := func(min, max *float64, incMin, incMax *bool) (int64, int64) { + if min == nil { + negInf := math.Inf(-1) + min = &negInf + } + if max == nil { + inf := math.Inf(1) + max = &inf + } + if incMin == nil { + d := true + incMin = &d + } + if incMax == nil { + d := false + incMax = &d + } + minInt64 := numeric.Float64ToInt64(*min) + if !*incMin && minInt64 != math.MaxInt64 { + minInt64++ + } + maxInt64 := numeric.Float64ToInt64(*max) + if !*incMax && maxInt64 != math.MinInt64 { + maxInt64-- + } + return minInt64, maxInt64 + } + + mins := []*float64{nil, f64(math.Inf(-1)), f64(-1e300), f64(-1), f64(0), f64(1), f64(42.5), f64(math.MaxFloat64)} + maxs := []*float64{nil, f64(math.Inf(1)), f64(-1e300), f64(-1), f64(0), f64(1), f64(42.5), f64(math.MaxFloat64)} + incs := []*bool{nil, b(true), b(false)} + + for _, min := range mins { + for _, max := range maxs { + for _, incMin := range incs { + for _, incMax := range incs { + wantLo, wantHi := reference(min, max, incMin, incMax) + gotLo, gotHi := Bounds(min, max, incMin, incMax) + + if gotLo != EncodeInt64(wantLo) { + t.Fatalf("lo mismatch for [%v,%v] inc(%v,%v): got %d, want %d", + deref(min), deref(max), derefB(incMin), derefB(incMax), + gotLo, EncodeInt64(wantLo)) + } + if gotHi != EncodeInt64(wantHi) { + t.Fatalf("hi mismatch for [%v,%v] inc(%v,%v): got %d, want %d", + deref(min), deref(max), derefB(incMin), derefB(incMax), + gotHi, EncodeInt64(wantHi)) + } + } + } + } + } +} + +// TestBoundsExtremeGuards pins down what the MaxInt64/MinInt64 guards in +// Bounds actually protect. They are not infinity guards: Float64ToInt64(+Inf) +// is 0x7FF0000000000000, comfortably short of MaxInt64, so an exclusive bound +// at an infinity increments normally -- into a NaN bit pattern, exactly as the +// inverted-index path does. The int64 extremes correspond to NaN payloads, so +// the guards are only reachable through NaN, which Validate rejects. They still +// have to stay, for bit-exact parity with NewNumericRangeSearcher. +func TestBoundsExtremeGuards(t *testing.T) { + // the floats that actually sit at the int64 extremes are NaNs + maxKey := math.Float64frombits(0x7FFFFFFFFFFFFFFF) + minKey := math.Float64frombits(0xFFFFFFFFFFFFFFFF) + + if got := numeric.Float64ToInt64(maxKey); got != math.MaxInt64 { + t.Fatalf("expected maxKey to map to MaxInt64, got %d", got) + } + if got := numeric.Float64ToInt64(minKey); got != math.MinInt64 { + t.Fatalf("expected minKey to map to MinInt64, got %d", got) + } + + // an exclusive minimum at the top must not wrap to zero + lo, _ := Bounds(&maxKey, nil, b(false), nil) + if lo != EncodeInt64(math.MaxInt64) { + t.Fatalf("exclusive min at the int64 max wrapped: got %d, want %d", + lo, uint64(EncodeInt64(math.MaxInt64))) + } + + // an exclusive maximum at the bottom must not wrap to the top + _, hi := Bounds(nil, &minKey, nil, b(false)) + if hi != EncodeInt64(math.MinInt64) { + t.Fatalf("exclusive max at the int64 min wrapped: got %d, want %d", + hi, uint64(EncodeInt64(math.MinInt64))) + } + + // and the infinities, which do increment, must still move upward + loInf, _ := Bounds(f64(math.Inf(1)), nil, b(false), nil) + if loInf <= Encode(math.Inf(1)) { + t.Fatalf("exclusive min at +Inf did not move upward: got %d", loInf) + } + _, hiInf := Bounds(nil, f64(math.Inf(-1)), nil, b(false)) + if hiInf >= Encode(math.Inf(-1)) { + t.Fatalf("exclusive max at -Inf did not move downward: got %d", hiInf) + } +} + +// TestBoundsExclusiveIsAdjacentFloat documents that a step in this space moves +// to the neighbouring representable float64, so exclusive bounds are exact. +func TestBoundsExclusiveIsAdjacentFloat(t *testing.T) { + lo, _ := Bounds(f64(1.0), nil, b(false), nil) + if got, want := Decode(lo), math.Nextafter(1.0, math.Inf(1)); got != want { + t.Fatalf("exclusive min above 1.0: got %v, want %v", got, want) + } + + _, hi := Bounds(nil, f64(1.0), nil, b(false)) + if got, want := Decode(hi), math.Nextafter(1.0, math.Inf(-1)); got != want { + t.Fatalf("exclusive max below 1.0: got %v, want %v", got, want) + } +} + +// TestBoundsEmptyRange checks that an inverted or empty range yields lo > hi, +// which Evaluate relies on to short-circuit. +func TestBoundsEmptyRange(t *testing.T) { + // min == max with both endpoints exclusive is empty + lo, hi := Bounds(f64(5), f64(5), b(false), b(false)) + if lo <= hi { + t.Fatalf("expected an empty range, got lo=%d hi=%d", lo, hi) + } + + // an inverted range is empty + lo, hi = Bounds(f64(10), f64(1), nil, nil) + if lo <= hi { + t.Fatalf("expected an empty range, got lo=%d hi=%d", lo, hi) + } + + // min == max inclusive on both sides matches exactly one value + lo, hi = Bounds(f64(5), f64(5), b(true), b(true)) + if lo != hi { + t.Fatalf("expected a single-value range, got lo=%d hi=%d", lo, hi) + } +} + +func deref(f *float64) interface{} { + if f == nil { + return "nil" + } + return *f +} + +func derefB(v *bool) interface{} { + if v == nil { + return "nil" + } + return *v +} diff --git a/numericv2/query.go b/numericv2/query.go new file mode 100644 index 000000000..27d4e6edb --- /dev/null +++ b/numericv2/query.go @@ -0,0 +1,69 @@ +// Copyright (c) 2026 Couchbase, Inc. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package numericv2 + +import ( + "sort" + + "github.com/RoaringBitmap/roaring/v2" + "github.com/blevesearch/bleve/v2/util" + segment "github.com/blevesearch/scorch_segment_api/v2" +) + +// Query evaluates a numeric range against one segment's number_v2 data. +type Query struct { + lo uint64 + hi uint64 +} + +// NewRangeQuery builds a query for the given range. A nil endpoint is +// unbounded; inclusiveMin defaults to true and inclusiveMax to false. +func NewRangeQuery(min, max *float64, inclusiveMin, inclusiveMax *bool) *Query { + lo, hi := Bounds(min, max, inclusiveMin, inclusiveMax) + return &Query{lo: lo, hi: hi} +} + +// Evaluate returns the segment document numbers whose value falls in the range. +// +// The values are sorted, so two binary searches bound the matching run and +// every entry inside it is a confirmed hit -- there is no refinement pass. The +// bitset both de-duplicates (a multi-valued field can put one document in the +// run more than once) and yields document numbers in ascending order. Since the +// stored document numbers are already in segment space, the snapshot's deleted +// bitmap can be applied directly as the bitset's exclusion set. +// +// The returned bitset comes from a shared pool: the caller owns it and must +// Release it once done with it. +func (q *Query) Evaluate(data segment.NumericV2Data, deleted *roaring.Bitmap, + numDocs uint64) *util.Bitset { + hits := util.AcquireBitset(int(numDocs), deleted) + if q.lo > q.hi { + return hits + } + + values := data.Values() + docNums := data.DocNums() + + start := sort.Search(len(values), func(i int) bool { return values[i] >= q.lo }) + // search only the unscanned suffix + end := start + sort.Search(len(values)-start, + func(i int) bool { return values[start+i] > q.hi }) + + for i := start; i < end; i++ { + hits.Add(int(docNums[i])) + } + + return hits +} diff --git a/search/query/numeric_range_v2.go b/search/query/numeric_range_v2.go new file mode 100644 index 000000000..539d70343 --- /dev/null +++ b/search/query/numeric_range_v2.go @@ -0,0 +1,110 @@ +// Copyright (c) 2026 Couchbase, Inc. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package query + +import ( + "context" + "fmt" + "math" + + "github.com/blevesearch/bleve/v2/mapping" + "github.com/blevesearch/bleve/v2/search" + "github.com/blevesearch/bleve/v2/search/searcher" + index "github.com/blevesearch/bleve_index_api" +) + +// NumericRangeV2 is the range itself. It is nested under a versioned key rather +// than living at the top level, because ParseQuery already claims top-level +// min/max for NumericRangeQuery and TermRangeQuery. +type NumericRangeV2 struct { + Min *float64 `json:"min,omitempty"` + Max *float64 `json:"max,omitempty"` + InclusiveMin *bool `json:"inclusive_min,omitempty"` + InclusiveMax *bool `json:"inclusive_max,omitempty"` +} + +// NumericRangeV2Query searches a number_v2 field. The endpoint semantics match +// NumericRangeQuery exactly: either endpoint may be omitted for an unbounded +// range, the minimum is inclusive by default and the maximum is not. +type NumericRangeV2Query struct { + RangeV2 NumericRangeV2 `json:"range_v2"` + FieldVal string `json:"field,omitempty"` + BoostVal *Boost `json:"boost,omitempty"` +} + +// NewNumericRangeV2Query creates a new query for ranges of numeric values over +// a number_v2 field. Either, but not both, endpoints can be nil. The minimum +// value is inclusive; the maximum value is exclusive. +func NewNumericRangeV2Query(min, max *float64) *NumericRangeV2Query { + return NewNumericRangeV2InclusiveQuery(min, max, nil, nil) +} + +// NewNumericRangeV2InclusiveQuery creates a new query for ranges of numeric +// values over a number_v2 field, with explicit control over endpoint inclusion. +func NewNumericRangeV2InclusiveQuery(min, max *float64, + minInclusive, maxInclusive *bool) *NumericRangeV2Query { + return &NumericRangeV2Query{ + RangeV2: NumericRangeV2{ + Min: min, + Max: max, + InclusiveMin: minInclusive, + InclusiveMax: maxInclusive, + }, + } +} + +func (q *NumericRangeV2Query) SetBoost(b float64) { + boost := Boost(b) + q.BoostVal = &boost +} + +func (q *NumericRangeV2Query) Boost() float64 { + return q.BoostVal.Value() +} + +func (q *NumericRangeV2Query) SetField(f string) { + q.FieldVal = f +} + +func (q *NumericRangeV2Query) Field() string { + return q.FieldVal +} + +func (q *NumericRangeV2Query) Validate() error { + if q.RangeV2.Min == nil && q.RangeV2.Max == nil { + return fmt.Errorf("number_v2 range query must specify min or max") + } + if q.RangeV2.Min != nil && math.IsNaN(*q.RangeV2.Min) { + return fmt.Errorf("number_v2 range query min must not be NaN") + } + if q.RangeV2.Max != nil && math.IsNaN(*q.RangeV2.Max) { + return fmt.Errorf("number_v2 range query max must not be NaN") + } + return nil +} + +func (q *NumericRangeV2Query) Searcher(ctx context.Context, i index.IndexReader, + m mapping.IndexMapping, options search.SearcherOptions) (search.Searcher, error) { + field := q.FieldVal + if q.FieldVal == "" { + field = m.DefaultSearchField() + } + + ctx = context.WithValue(ctx, search.QueryTypeKey, search.Numeric) + + return searcher.NewNumericV2Searcher(ctx, i, q.RangeV2.Min, q.RangeV2.Max, + q.RangeV2.InclusiveMin, q.RangeV2.InclusiveMax, field, + q.BoostVal.Value(), options) +} diff --git a/search/query/query.go b/search/query/query.go index 711b9deb5..249865b30 100644 --- a/search/query/query.go +++ b/search/query/query.go @@ -233,6 +233,17 @@ func ParseQuery(input []byte) (Query, error) { } return &rv, nil } + // checked ahead of the top-level min/max branches below: a number_v2 range + // is nested under its own key precisely because those are already taken + _, hasRangeV2 := tmp["range_v2"] + if hasRangeV2 { + var rv NumericRangeV2Query + err := util.UnmarshalJSON(input, &rv) + if err != nil { + return nil, err + } + return &rv, nil + } _, hasMin := tmp["min"].(float64) _, hasMax := tmp["max"].(float64) if hasMin || hasMax { diff --git a/search/searcher/search_numeric_v2.go b/search/searcher/search_numeric_v2.go new file mode 100644 index 000000000..bb4ba0b5a --- /dev/null +++ b/search/searcher/search_numeric_v2.go @@ -0,0 +1,128 @@ +// Copyright (c) 2026 Couchbase, Inc. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package searcher + +import ( + "context" + "fmt" + "reflect" + + "github.com/blevesearch/bleve/v2/search" + "github.com/blevesearch/bleve/v2/search/scorer" + index "github.com/blevesearch/bleve_index_api" +) + +var reflectStaticSizeNumericV2Searcher int + +func init() { + var nv2s NumericV2Searcher + reflectStaticSizeNumericV2Searcher = int(reflect.TypeOf(nv2s).Size()) +} + +// NumericV2Searcher is a filtering searcher over a number_v2 field. Every +// document the underlying reader returns is a confirmed match, so the searcher +// contributes a constant score rather than a computed one. +type NumericV2Searcher struct { + numericIndexReader index.NumericV2FieldReader + scorer *scorer.ConstantScorer + + nd index.NumericV2FieldDoc +} + +func NewNumericV2Searcher(ctx context.Context, indexReader index.IndexReader, + min, max *float64, inclusiveMin, inclusiveMax *bool, field string, + boost float64, options search.SearcherOptions, +) (search.Searcher, error) { + + nr, ok := indexReader.(index.NumericV2IndexReader) + if !ok { + return nil, fmt.Errorf("indexReader does not support number_v2 queries") + } + + // get the NumericV2FieldReader for the specified field + numericIndexReader, err := nr.NumericV2FieldReader(ctx, field) + if err != nil { + return nil, err + } + + // perform the search on the NumericV2FieldReader for the specified range + err = numericIndexReader.Search(min, max, inclusiveMin, inclusiveMax) + if err != nil { + return nil, err + } + + return &NumericV2Searcher{ + numericIndexReader: numericIndexReader, + scorer: scorer.NewConstantScorer(1, boost, options), + nd: index.NumericV2FieldDoc{}, + }, nil +} + +// Next returns the next document match, scored by the ConstantScorer. +func (n *NumericV2Searcher) Next(ctx *search.SearchContext) (*search.DocumentMatch, error) { + match, err := n.numericIndexReader.Next(n.nd.Reset()) + if err != nil { + return nil, err + } + if match == nil { + return nil, nil + } + + return n.scorer.Score(ctx, match.ID), nil +} + +// Advance moves the searcher to the first document with an ID greater than or +// equal to the specified ID. +func (n *NumericV2Searcher) Advance(ctx *search.SearchContext, + ID index.IndexInternalID) (*search.DocumentMatch, error) { + match, err := n.numericIndexReader.Advance(ID, n.nd.Reset()) + if err != nil { + return nil, err + } + if match == nil { + return nil, nil + } + + return n.scorer.Score(ctx, match.ID), nil +} + +func (n *NumericV2Searcher) Close() error { + return n.numericIndexReader.Close() +} + +func (n *NumericV2Searcher) Count() uint64 { + return n.numericIndexReader.Count() +} + +func (n *NumericV2Searcher) DocumentMatchPoolSize() int { + return 1 +} + +func (n *NumericV2Searcher) Min() int { + return 0 +} + +func (n *NumericV2Searcher) SetQueryNorm(norm float64) { + n.scorer.SetQueryNorm(norm) +} + +func (n *NumericV2Searcher) Size() int { + return reflectStaticSizeNumericV2Searcher + n.numericIndexReader.Size() + + n.scorer.Size() + n.nd.Size() +} + +func (n *NumericV2Searcher) Weight() float64 { + return n.scorer.Weight() +} diff --git a/search_numeric_v2_test.go b/search_numeric_v2_test.go new file mode 100644 index 000000000..9ad8fad02 --- /dev/null +++ b/search_numeric_v2_test.go @@ -0,0 +1,610 @@ +// Copyright (c) 2026 Couchbase, Inc. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package bleve + +import ( + "context" + "fmt" + "math" + "os" + "reflect" + "sort" + "testing" + + "github.com/blevesearch/bleve/v2/index/scorch" + "github.com/blevesearch/bleve/v2/index/scorch/mergeplan" + "github.com/blevesearch/bleve/v2/mapping" + "github.com/blevesearch/bleve/v2/search" + "github.com/blevesearch/bleve/v2/search/query" +) + +// The corpus is indexed twice, into a "v1" field of type number and a "v2" +// field of type number_v2, so every assertion below can be a comparison rather +// than a hand-written expectation. +const ( + nv2FieldV1 = "priceV1" + nv2FieldV2 = "priceV2" +) + +func nv2TestMapping(t *testing.T) mapping.IndexMapping { + t.Helper() + im := NewIndexMapping() + + v1 := NewNumericFieldMapping() + v1.Name = nv2FieldV1 + v2 := NewNumberV2FieldMapping() + v2.Name = nv2FieldV2 + + dm := NewDocumentMapping() + dm.AddFieldMappingsAt(nv2FieldV1, v1) + dm.AddFieldMappingsAt(nv2FieldV2, v2) + im.DefaultMapping = dm + + if err := im.Validate(); err != nil { + t.Fatalf("mapping did not validate: %v", err) + } + return im +} + +// nv2Corpus spans negatives, zero, fractions and duplicates. +var nv2Corpus = []float64{-1000, -42.5, -1, 0, 0.5, 1, 1, 7, 42.5, 99, 1000, 123456.75} + +func nv2OpenIndex(t *testing.T, name string) (Index, func()) { + t.Helper() + path := fmt.Sprintf("%s/%s", t.TempDir(), name) + idx, err := New(path, nv2TestMapping(t)) + if err != nil { + t.Fatal(err) + } + return idx, func() { + if cerr := idx.Close(); cerr != nil { + t.Fatalf("error closing index: %v", cerr) + } + _ = os.RemoveAll(path) + } +} + +// nv2Index builds an index over the corpus. When batches > 1 the documents are +// spread over several batches so that the search spans multiple segments. +func nv2Index(t *testing.T, name string, batches int) (Index, func()) { + t.Helper() + idx, cleanup := nv2OpenIndex(t, name) + + perBatch := (len(nv2Corpus) + batches - 1) / batches + for start := 0; start < len(nv2Corpus); start += perBatch { + batch := idx.NewBatch() + end := start + perBatch + if end > len(nv2Corpus) { + end = len(nv2Corpus) + } + for i := start; i < end; i++ { + doc := map[string]interface{}{ + nv2FieldV1: nv2Corpus[i], + nv2FieldV2: nv2Corpus[i], + } + if err := batch.Index(fmt.Sprintf("d%02d", i), doc); err != nil { + cleanup() + t.Fatal(err) + } + } + if err := idx.Batch(batch); err != nil { + cleanup() + t.Fatal(err) + } + } + return idx, cleanup +} + +func nv2HitIDs(t *testing.T, idx Index, q query.Query, sorts []string) []string { + t.Helper() + req := NewSearchRequestOptions(q, len(nv2Corpus)+10, 0, false) + if len(sorts) > 0 { + req.SortBy(sorts) + } + res, err := idx.Search(req) + if err != nil { + t.Fatal(err) + } + ids := make([]string, 0, len(res.Hits)) + for _, hit := range res.Hits { + ids = append(ids, hit.ID) + } + return ids +} + +func f64p(v float64) *float64 { return &v } +func boolp(v bool) *bool { return &v } + +// nv2Ranges is the shared range matrix: open ends, closed ends, every +// inclusivity combination, empty ranges and exact hits on corpus values. +type nv2Range struct { + min, max *float64 + incMin *bool + incMax *bool +} + +func nv2Ranges() []nv2Range { + var rv []nv2Range + ends := []*float64{nil, f64p(-1000), f64p(-42.5), f64p(-1), f64p(0), f64p(0.5), + f64p(1), f64p(42.5), f64p(99), f64p(1000), f64p(123456.75), f64p(1e9)} + incs := []*bool{nil, boolp(true), boolp(false)} + + for _, min := range ends { + for _, max := range ends { + if min == nil && max == nil { + continue // rejected by Validate on both query types + } + for _, incMin := range incs { + for _, incMax := range incs { + rv = append(rv, nv2Range{min, max, incMin, incMax}) + } + } + } + } + return rv +} + +func (r nv2Range) label() string { + fmtEnd := func(f *float64) string { + if f == nil { + return "nil" + } + return fmt.Sprintf("%g", *f) + } + fmtInc := func(b *bool) string { + if b == nil { + return "nil" + } + return fmt.Sprintf("%t", *b) + } + return fmt.Sprintf("[%s,%s] inc(%s,%s)", fmtEnd(r.min), fmtEnd(r.max), + fmtInc(r.incMin), fmtInc(r.incMax)) +} + +func (r nv2Range) v1(field string) query.Query { + q := query.NewNumericRangeInclusiveQuery(r.min, r.max, r.incMin, r.incMax) + q.SetField(field) + return q +} + +func (r nv2Range) v2(field string) query.Query { + q := query.NewNumericRangeV2InclusiveQuery(r.min, r.max, r.incMin, r.incMax) + q.SetField(field) + return q +} + +// TestNumericV2MatchesNumericV1 is the load-bearing test for the whole feature: +// the same corpus, the same ranges, through both the inverted-index numeric path +// and the number_v2 section, must produce identical hit sets. An encoding or +// inclusivity mismatch shows up here and essentially nowhere else. +func TestNumericV2MatchesNumericV1(t *testing.T) { + for _, batches := range []int{1, 4} { + t.Run(fmt.Sprintf("batches=%d", batches), func(t *testing.T) { + idx, cleanup := nv2Index(t, fmt.Sprintf("nv2-diff-%d", batches), batches) + defer cleanup() + + for _, r := range nv2Ranges() { + got := nv2HitIDs(t, idx, r.v2(nv2FieldV2), []string{"_id"}) + want := nv2HitIDs(t, idx, r.v1(nv2FieldV1), []string{"_id"}) + if !reflect.DeepEqual(got, want) { + t.Fatalf("range %s: number_v2 gave %v, number gave %v", + r.label(), got, want) + } + } + }) + } +} + +// TestNumericV2WithDeletionsAndUpdates exercises the deleted-document path and +// then forces a merge, re-checking parity after each step. +func TestNumericV2WithDeletionsAndUpdates(t *testing.T) { + idx, cleanup := nv2Index(t, "nv2-deletes", 3) + defer cleanup() + + // delete a few documents and update another to a new value + batch := idx.NewBatch() + batch.Delete("d00") // -1000 + batch.Delete("d06") // one of the two 1s + if err := batch.Index("d07", map[string]interface{}{ + nv2FieldV1: 5000.0, + nv2FieldV2: 5000.0, + }); err != nil { + t.Fatal(err) + } + if err := idx.Batch(batch); err != nil { + t.Fatal(err) + } + + check := func(stage string) { + for _, r := range nv2Ranges() { + got := nv2HitIDs(t, idx, r.v2(nv2FieldV2), []string{"_id"}) + want := nv2HitIDs(t, idx, r.v1(nv2FieldV1), []string{"_id"}) + if !reflect.DeepEqual(got, want) { + t.Fatalf("%s, range %s: number_v2 gave %v, number gave %v", + stage, r.label(), got, want) + } + } + } + + check("after deletes") + + // the deleted documents must be gone from the v2 results too + all := nv2HitIDs(t, idx, query.NewNumericRangeV2Query(f64p(math.Inf(-1)), f64p(math.Inf(1))), []string{"_id"}) + for _, id := range all { + if id == "d00" || id == "d06" { + t.Fatalf("deleted document %s still matched", id) + } + } + + // force everything into one segment and re-check: the merge path has its + // own doc number remapping and doc value stream + sc, ok := idx.(*indexImpl).i.(*scorch.Scorch) + if !ok { + t.Fatalf("expected a scorch index, got %T", idx.(*indexImpl).i) + } + if err := sc.ForceMerge(context.Background(), &mergeplan.SingleSegmentMergePlanOptions); err != nil { + t.Fatalf("force merge: %v", err) + } + + check("after merge") +} + +// TestNumericV2SortParity checks that sorting on a number_v2 field agrees with +// sorting on a number field across sort types, modes and missing-value +// placements. This is what the doc value block exists for. +func TestNumericV2SortParity(t *testing.T) { + idx, cleanup := nv2Index(t, "nv2-sort", 3) + defer cleanup() + + all := query.NewMatchAllQuery() + + cases := []struct { + name string + mkSort func(field string) search.SearchSort + }{ + {"auto-asc", func(f string) search.SearchSort { + return &search.SortField{Field: f} + }}, + {"auto-desc", func(f string) search.SearchSort { + return &search.SortField{Field: f, Desc: true} + }}, + {"as-number-asc", func(f string) search.SearchSort { + return &search.SortField{Field: f, Type: search.SortFieldAsNumber} + }}, + {"as-number-desc", func(f string) search.SearchSort { + return &search.SortField{Field: f, Type: search.SortFieldAsNumber, Desc: true} + }}, + {"as-number-min", func(f string) search.SearchSort { + return &search.SortField{Field: f, Type: search.SortFieldAsNumber, Mode: search.SortFieldMin} + }}, + {"as-number-max", func(f string) search.SearchSort { + return &search.SortField{Field: f, Type: search.SortFieldAsNumber, Mode: search.SortFieldMax} + }}, + {"missing-first", func(f string) search.SearchSort { + return &search.SortField{Field: f, Type: search.SortFieldAsNumber, Missing: search.SortFieldMissingFirst} + }}, + {"missing-last", func(f string) search.SearchSort { + return &search.SortField{Field: f, Type: search.SortFieldAsNumber, Missing: search.SortFieldMissingLast} + }}, + } + + run := func(field string, s search.SearchSort) []string { + req := NewSearchRequestOptions(all, len(nv2Corpus)+10, 0, false) + // tie-break on _id so the comparison is deterministic + req.Sort = search.SortOrder{s, &search.SortField{Field: "_id"}} + res, err := idx.Search(req) + if err != nil { + t.Fatal(err) + } + ids := make([]string, 0, len(res.Hits)) + for _, hit := range res.Hits { + ids = append(ids, hit.ID) + } + return ids + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + got := run(nv2FieldV2, tc.mkSort(nv2FieldV2)) + want := run(nv2FieldV1, tc.mkSort(nv2FieldV1)) + if !reflect.DeepEqual(got, want) { + t.Fatalf("sort %s: number_v2 gave %v, number gave %v", tc.name, got, want) + } + if len(got) != len(nv2Corpus) { + t.Fatalf("sort %s: expected %d hits, got %d", tc.name, len(nv2Corpus), len(got)) + } + }) + } +} + +// TestNumericV2FacetParity checks that numeric range facets over a number_v2 +// field agree with the same facets over a number field, including the missing +// and other counts. +func TestNumericV2FacetParity(t *testing.T) { + idx, cleanup := nv2Index(t, "nv2-facet", 3) + defer cleanup() + + buckets := []struct { + name string + min, max *float64 + }{ + {"negative", nil, f64p(0)}, + {"small", f64p(0), f64p(10)}, + {"medium", f64p(10), f64p(1000)}, + {"large", f64p(1000), nil}, + } + + facetFor := func(field string) *search.FacetResult { + req := NewSearchRequestOptions(query.NewMatchAllQuery(), 0, 0, false) + fr := NewFacetRequest(field, 10) + for _, b := range buckets { + fr.AddNumericRange(b.name, b.min, b.max) + } + req.AddFacet("prices", fr) + res, err := idx.Search(req) + if err != nil { + t.Fatal(err) + } + return res.Facets["prices"] + } + + gotF, wantF := facetFor(nv2FieldV2), facetFor(nv2FieldV1) + + if gotF == nil || wantF == nil { + t.Fatalf("missing facet result: v2=%v v1=%v", gotF, wantF) + } + if gotF.Total != wantF.Total || gotF.Missing != wantF.Missing || gotF.Other != wantF.Other { + t.Fatalf("facet totals differ: v2 total=%d missing=%d other=%d, "+ + "v1 total=%d missing=%d other=%d", + gotF.Total, gotF.Missing, gotF.Other, + wantF.Total, wantF.Missing, wantF.Other) + } + + counts := func(fr *search.FacetResult) map[string]int { + rv := map[string]int{} + for _, nr := range fr.NumericRanges { + rv[nr.Name] = nr.Count + } + return rv + } + if got, want := counts(gotF), counts(wantF); !reflect.DeepEqual(got, want) { + t.Fatalf("facet bucket counts differ: v2=%v v1=%v", got, want) + } + + // sanity: the buckets must actually have counted the corpus + var total int + for _, c := range counts(gotF) { + total += c + } + if total != len(nv2Corpus) { + t.Fatalf("expected the buckets to cover all %d docs, counted %d", + len(nv2Corpus), total) + } +} + +// TestNumericV2Conjunction exercises Advance, which a bare range query never +// reaches, by combining the range with another clause. +func TestNumericV2Conjunction(t *testing.T) { + idx, cleanup := nv2Index(t, "nv2-conj", 3) + defer cleanup() + + r := nv2Range{min: f64p(0), max: f64p(1000), incMin: boolp(true), incMax: boolp(true)} + + for _, other := range []query.Query{ + query.NewMatchAllQuery(), + query.NewDocIDQuery([]string{"d03", "d05", "d06", "d07", "d10"}), + } { + v2 := query.NewConjunctionQuery([]query.Query{r.v2(nv2FieldV2), other}) + v1 := query.NewConjunctionQuery([]query.Query{r.v1(nv2FieldV1), other}) + + got := nv2HitIDs(t, idx, v2, []string{"_id"}) + want := nv2HitIDs(t, idx, v1, []string{"_id"}) + if !reflect.DeepEqual(got, want) { + t.Fatalf("conjunction: number_v2 gave %v, number gave %v", got, want) + } + if len(got) == 0 { + t.Fatal("expected the conjunction to match something") + } + + dis2 := query.NewDisjunctionQuery([]query.Query{r.v2(nv2FieldV2), other}) + dis1 := query.NewDisjunctionQuery([]query.Query{r.v1(nv2FieldV1), other}) + got = nv2HitIDs(t, idx, dis2, []string{"_id"}) + want = nv2HitIDs(t, idx, dis1, []string{"_id"}) + if !reflect.DeepEqual(got, want) { + t.Fatalf("disjunction: number_v2 gave %v, number gave %v", got, want) + } + } +} + +// TestNumericV2ConstantScore verifies that every hit scores identically, and +// that boost feeds through the scorer the way a constant-score clause should. +// +// Note that boost cannot change the score of a lone clause: the collector +// normalises by the query norm, which for a single searcher is 1/boost, so the +// two cancel and the score is exactly the constant. Boost only has a visible +// effect relative to sibling clauses, which is what the disjunction below +// checks. +func TestNumericV2ConstantScore(t *testing.T) { + idx, cleanup := nv2Index(t, "nv2-score", 2) + defer cleanup() + + wideRange := func(boost float64) *query.NumericRangeV2Query { + q := query.NewNumericRangeV2Query(f64p(-1e9), f64p(1e9)) + q.SetField(nv2FieldV2) + q.SetBoost(boost) + return q + } + + scores := func(q query.Query) map[string]float64 { + req := NewSearchRequestOptions(q, len(nv2Corpus)+10, 0, false) + res, err := idx.Search(req) + if err != nil { + t.Fatal(err) + } + rv := map[string]float64{} + for _, hit := range res.Hits { + rv[hit.ID] = hit.Score + } + return rv + } + + // every hit of a lone constant-score clause scores the same + base := scores(wideRange(1)) + if len(base) != len(nv2Corpus) { + t.Fatalf("expected %d hits, got %d", len(nv2Corpus), len(base)) + } + var first float64 + for _, s := range base { + if first == 0 { + first = s + } + if s != first { + t.Fatalf("expected a constant score, got %v", base) + } + } + + // and boost cancels against the query norm for that lone clause + boostedAlone := scores(wideRange(7)) + for id, s := range boostedAlone { + if s != base[id] { + t.Fatalf("boost changed a lone clause's score for %s: %v vs %v", + id, s, base[id]) + } + } + + // the scorer's weight, which drives that normalisation, does scale + if got, want := wideRange(3).Boost(), 3.0; got != want { + t.Fatalf("boost not retained on the query: got %v, want %v", got, want) + } + + // relative to a sibling clause, boost does move the score: two disjoint + // ranges in a disjunction, the negative half boosted much harder + negatives := query.NewNumericRangeV2InclusiveQuery(f64p(-1e9), f64p(0), boolp(true), boolp(false)) + negatives.SetField(nv2FieldV2) + negatives.SetBoost(100) + positives := query.NewNumericRangeV2InclusiveQuery(f64p(0), f64p(1e9), boolp(true), boolp(true)) + positives.SetField(nv2FieldV2) + positives.SetBoost(1) + + got := scores(query.NewDisjunctionQuery([]query.Query{negatives, positives})) + + // d00 = -1000 and d01 = -42.5 are in the boosted half; d07 = 7 is not + for _, negID := range []string{"d00", "d01"} { + if got[negID] <= got["d07"] { + t.Fatalf("expected the boosted clause to score higher: %s=%v vs d07=%v", + negID, got[negID], got["d07"]) + } + } +} + +// TestNumericV2QueryParsing checks the range_v2 discriminator, and in +// particular that a plain min/max query still parses as a v1 NumericRangeQuery. +func TestNumericV2QueryParsing(t *testing.T) { + v2JSON := []byte(`{"field":"price","range_v2":{"min":10,"max":100,` + + `"inclusive_min":true,"inclusive_max":false}}`) + q, err := query.ParseQuery(v2JSON) + if err != nil { + t.Fatal(err) + } + nq, ok := q.(*query.NumericRangeV2Query) + if !ok { + t.Fatalf("expected a NumericRangeV2Query, got %T", q) + } + if nq.RangeV2.Min == nil || *nq.RangeV2.Min != 10 || + nq.RangeV2.Max == nil || *nq.RangeV2.Max != 100 { + t.Fatalf("range not parsed: %+v", nq.RangeV2) + } + if nq.RangeV2.InclusiveMin == nil || !*nq.RangeV2.InclusiveMin || + nq.RangeV2.InclusiveMax == nil || *nq.RangeV2.InclusiveMax { + t.Fatalf("inclusivity not parsed: %+v", nq.RangeV2) + } + if nq.Field() != "price" { + t.Fatalf("field not parsed: %q", nq.Field()) + } + + // the v1 form must be unaffected + v1JSON := []byte(`{"field":"price","min":10,"max":100}`) + q, err = query.ParseQuery(v1JSON) + if err != nil { + t.Fatal(err) + } + if _, ok := q.(*query.NumericRangeQuery); !ok { + t.Fatalf("expected a NumericRangeQuery for the v1 form, got %T", q) + } +} + +// TestNumericV2Validation covers the query and mapping rejections. +func TestNumericV2Validation(t *testing.T) { + if err := query.NewNumericRangeV2Query(nil, nil).Validate(); err == nil { + t.Fatal("expected an unbounded range to be rejected") + } + nan := math.NaN() + if err := query.NewNumericRangeV2Query(&nan, nil).Validate(); err == nil { + t.Fatal("expected a NaN minimum to be rejected") + } + if err := query.NewNumericRangeV2Query(nil, &nan).Validate(); err == nil { + t.Fatal("expected a NaN maximum to be rejected") + } + + // IncludeInAll must be rejected at index-definition time + im := NewIndexMapping() + fm := NewNumberV2FieldMapping() + fm.Name = nv2FieldV2 + fm.IncludeInAll = true + dm := NewDocumentMapping() + dm.AddFieldMappingsAt(nv2FieldV2, fm) + im.DefaultMapping = dm + if err := im.Validate(); err == nil { + t.Fatal("expected IncludeInAll on a number_v2 field to be rejected") + } +} + +// TestNumericV2MultiValued indexes an array of numbers and checks that a +// document is returned once, not once per matching value. +func TestNumericV2MultiValued(t *testing.T) { + idx, cleanup := nv2OpenIndex(t, "nv2-multi") + defer cleanup() + + docs := map[string][]float64{ + "a": {1, 5, 9}, + "b": {2}, + "c": {100, 200}, + } + batch := idx.NewBatch() + for id, vals := range docs { + if err := batch.Index(id, map[string]interface{}{ + nv2FieldV1: vals, + nv2FieldV2: vals, + }); err != nil { + t.Fatal(err) + } + } + if err := idx.Batch(batch); err != nil { + t.Fatal(err) + } + + // a range covering several of a's values must return a exactly once + r := nv2Range{min: f64p(0), max: f64p(10), incMin: boolp(true), incMax: boolp(true)} + got := nv2HitIDs(t, idx, r.v2(nv2FieldV2), []string{"_id"}) + want := nv2HitIDs(t, idx, r.v1(nv2FieldV1), []string{"_id"}) + sort.Strings(got) + sort.Strings(want) + if !reflect.DeepEqual(got, want) { + t.Fatalf("multi-valued: number_v2 gave %v, number gave %v", got, want) + } + if len(got) != 2 { + t.Fatalf("expected documents a and b, got %v", got) + } +} diff --git a/util/bitset.go b/util/bitset.go index 418ada2e5..7dd970f56 100644 --- a/util/bitset.go +++ b/util/bitset.go @@ -16,25 +16,73 @@ package util import ( "math/bits" + "sync" "github.com/RoaringBitmap/roaring/v2" ) +// Bitset is a two-level bitmap. The lower level (data) holds one bit per value. +// The upper level (summary) holds one bit per lower-level word, set whenever +// that word is non-empty, so iteration can skip 64 empty words at a time +// instead of loading each one. +// +// That matters because a bitset is sized by the document count of the segment +// it covers, not by the number of bits actually set: without the summary, a +// query matching 0.1% of a five-million-document segment still walks all 78,000 +// words to find its 5,000 hits. +// +// The invariant is exact -- a summary bit is set if and only if the +// corresponding data word is non-zero -- so every mutating method below has to +// maintain it. A stale set bit would still be correct, since iteration +// re-checks the word it names and skips it when empty; exactness is what +// preserves the performance the summary exists for. +// +// Maintaining it inside Add costs one extra store per value, against a region +// 64x smaller than the data which stays cache-resident. Measured against the +// same benchmark at four densities on a five-million-document segment, that +// buys 3x at 0.1% and roughly 1.1x at 1%, breaks even at 10%, and costs about +// 9% at 50% where nearly every word is occupied and there is nothing to skip. +// Two alternatives were tried and rejected: making the store conditional on the +// word having been empty regressed 10% density by 19% on an unpredictable +// branch, and rebuilding the summary in one pass per iteration regressed 1% +// density by 27% by paying an O(words) pass on top of the walk. type Bitset struct { + // backing owns the memory; data and summary are views over it, so that a + // pooled bitset is a single allocation. + backing []uint64 data []uint64 + summary []uint64 numBits int exclude *roaring.Bitmap } +// bitsetSizes returns the number of lower-level and upper-level words needed to +// hold values up to maxVal. +func bitsetSizes(maxVal int) (words, summaryWords int) { + // We need (maxVal / 64) + 1 buckets to hold up to maxVal + words = (maxVal / 64) + 1 + // and one summary bit per bucket, 64 buckets to a summary word + summaryWords = ((words - 1) / 64) + 1 + return words, summaryWords +} + +// split carves the two levels out of one backing slice. data is capped at its +// own length so that an accidental append cannot scribble over the summary. +func (b *Bitset) split(backing []uint64, words int) { + b.backing = backing + b.data = backing[:words:words] + b.summary = backing[words:] +} + // NewBitset initializes a bitset capable of holding numbers up to maxVal func NewBitset(maxVal int, exclude *roaring.Bitmap) *Bitset { - // We need (maxVal / 64) + 1 buckets to hold up to maxVal - size := (maxVal / 64) + 1 - return &Bitset{ - data: make([]uint64, size), + words, summaryWords := bitsetSizes(maxVal) + rv := &Bitset{ numBits: maxVal, exclude: exclude, } + rv.split(make([]uint64, words+summaryWords), words) + return rv } // Add inserts a value into the bitset (safely handles duplicates) @@ -47,6 +95,10 @@ func (b *Bitset) Add(val int) { // Set the bit to 1 using bitwise OR b.data[bucket] |= (1 << bit) + + // and mark the bucket as occupied, so iteration can skip 64 empty buckets + // at a time instead of loading each one + b.summary[bucket>>6] |= 1 << uint(bucket&63) } // Remove deletes a value from the bitset @@ -56,6 +108,11 @@ func (b *Bitset) Remove(val int) { // Set the bit to 0 using bitwise AND with the complement b.data[bucket] &^= (1 << bit) + + // keep the summary exact: clear its bit once the bucket empties + if b.data[bucket] == 0 { + b.summary[bucket>>6] &^= 1 << uint(bucket&63) + } } // Contains checks if a value exists in the bitset @@ -92,22 +149,35 @@ func (b *Bitset) Invert() { } } } + + // every word just changed; Invert is already O(words) so rebuilding here + // rather than deferring costs nothing extra + b.rebuildSummary() +} + +// rebuildSummary recomputes the upper level from the lower one. Only needed +// after a bulk rewrite of data; Add and Remove maintain it incrementally. +func (b *Bitset) rebuildSummary() { + clear(b.summary) + for i, word := range b.data { + if word != 0 { + b.summary[i>>6] |= 1 << uint(i&63) + } + } } // Iterate calls the provided function for every integer recorded in the bitset, in ascending order func (b *Bitset) Iterate(f func(int)) { - for bucketIdx, bucket := range b.data { - // If the entire 64-bit block is 0, skip it entirely for speed - if bucket == 0 { - continue - } + for summaryIdx, summaryWord := range b.summary { + // each set summary bit names a non-empty data word; empty words are + // skipped 64 at a time + for summaryWord != 0 { + bucketIdx := summaryIdx<<6 + bits.TrailingZeros64(summaryWord) + summaryWord &= summaryWord - 1 - // Check all 64 bits in this bucket - for bitIdx := 0; bitIdx < 64; bitIdx++ { - if (bucket & (1 << uint(bitIdx))) != 0 { - // Reconstruct the original integer - originalVal := (bucketIdx << 6) + bitIdx - f(originalVal) + // walk only the set bits of the bucket, lowest first + for bucket := b.data[bucketIdx]; bucket != 0; bucket &= bucket - 1 { + f(bucketIdx<<6 + bits.TrailingZeros64(bucket)) } } } @@ -124,7 +194,198 @@ func (b *Bitset) Count() int { } func (b *Bitset) Clear() { - for i := range b.data { - b.data[i] = 0 + clear(b.data) + clear(b.summary) +} + +// SizeInBytes returns the memory footprint of the bitset's backing words. +func (b *Bitset) SizeInBytes() int { + if b == nil { + return 0 + } + return len(b.backing) * 8 +} + +// ----------------------------------------------------------------------------- +// iteration + +// BitsetIterator walks the set bits of a Bitset in ascending order, using the +// summary level to skip runs of empty words. +// +// It is a value type on purpose: the caller keeps it in a slice and calls +// through it once per hit, so an interface or a pointer chase per call would +// cost more than the bit extraction itself. The zero value is a valid, empty +// iterator, which lets callers leave a slot unset rather than nil-checking on +// the hot path. +type BitsetIterator struct { + words []uint64 + summary []uint64 + + // summaryIdx is the summary word being consumed, and summaryWord its bits + // that have not been visited yet; each names a non-empty data word. + summaryIdx int + summaryWord uint64 + + // wordIdx is the data word being consumed, and word its bits that have not + // been returned yet, so the lowest set bit is always the next value. + wordIdx int + word uint64 +} + +// Iterator returns an iterator over the bitset's set bits. It aliases the +// bitset's memory, so it must not be used after the bitset is released. +func (b *Bitset) Iterator() BitsetIterator { + rv := BitsetIterator{words: b.data, summary: b.summary} + if len(b.summary) > 0 { + rv.summaryWord = b.summary[0] + } + return rv +} + +// Next returns the next set bit in ascending order, or false once exhausted. +func (it *BitsetIterator) Next() (int, bool) { + for it.word == 0 { + if !it.nextWord() { + return 0, false + } + } + + bit := bits.TrailingZeros64(it.word) + // clear the lowest set bit + it.word &= it.word - 1 + + return it.wordIdx<<6 + bit, true +} + +// nextWord loads the next non-empty data word, consuming one summary bit and +// advancing through summary words as needed. Returns false once exhausted. +func (it *BitsetIterator) nextWord() bool { + for it.summaryWord == 0 { + it.summaryIdx++ + if it.summaryIdx >= len(it.summary) { + // clamp, so repeated calls past the end stay put + it.summaryIdx = len(it.summary) + return false + } + it.summaryWord = it.summary[it.summaryIdx] + } + + it.wordIdx = it.summaryIdx<<6 + bits.TrailingZeros64(it.summaryWord) + it.summaryWord &= it.summaryWord - 1 + it.word = it.words[it.wordIdx] + return true +} + +// AdvanceTo positions the iterator so that the next call to Next returns the +// first set bit greater than or equal to val. It only ever moves forward: a val +// at or behind the current position leaves the iterator untouched. +func (it *BitsetIterator) AdvanceTo(val int) { + if val < 0 { + return + } + + targetWord := val >> 6 + if targetWord < it.wordIdx { + // already past it + return + } + + if targetWord == it.wordIdx && it.word != 0 { + // still inside the word we are partway through: dropping the bits + // below val is enough, unless that empties it + it.word &= ^uint64(0) << uint(val&63) + if it.word != 0 { + return + } + } + + // the current word cannot serve val, so move the summary to the word + // containing val and let Next pick up from there + it.word = 0 + + targetSummary := targetWord >> 6 + if targetSummary >= len(it.summary) { + it.summaryIdx = len(it.summary) + it.summaryWord = 0 + return + } + if targetSummary > it.summaryIdx { + it.summaryIdx = targetSummary + it.summaryWord = it.summary[targetSummary] + } + if it.summaryIdx == targetSummary { + // drop the summary bits for words below the target + it.summaryWord &= ^uint64(0) << uint(targetWord&63) + } + + if !it.nextWord() { + return + } + if it.wordIdx == targetWord { + it.word &= ^uint64(0) << uint(val&63) + } +} + +// ----------------------------------------------------------------------------- +// pooling + +// bitsetPool recycles the backing word slices. A bitset is sized by the +// document count of the segment it covers, not by the number of bits actually +// set, so a selective query allocates just as much as a broad one. Reusing the +// words keeps that cost off the allocator entirely. +// +// Pointers to slices are pooled rather than slices themselves, so that handing +// one to sync.Pool does not itself allocate an interface box. +var bitsetPool sync.Pool + +// AcquireBitset returns a zeroed bitset capable of holding values up to maxVal, +// reusing pooled memory when a large enough slice is available. +// +// The caller owns the bitset and should Release it once done. Anything derived +// from it -- in particular an Iterator, which aliases the same words -- must +// not be used after Release. +func AcquireBitset(maxVal int, exclude *roaring.Bitmap) *Bitset { + // both levels come out of one allocation, so the pooled slice has to be + // large enough for the summary as well as the data + words, summaryWords := bitsetSizes(maxVal) + total := words + summaryWords + + var backing []uint64 + if v := bitsetPool.Get(); v != nil { + pooled := v.(*[]uint64) + if cap(*pooled) >= total { + backing = (*pooled)[:total] + clear(backing) + } else { + // too small to serve this request, but still worth keeping + bitsetPool.Put(pooled) + } + } + if backing == nil { + backing = make([]uint64, total) + } + + rv := &Bitset{ + numBits: maxVal, + exclude: exclude, + } + rv.split(backing, words) + return rv +} + +// Release returns the bitset's memory to the pool. The bitset is left empty, so +// a later Add or Contains will panic rather than corrupt recycled memory, and a +// second Release is a no-op. +func (b *Bitset) Release() { + if b == nil || b.backing == nil { + return } + // return the whole backing allocation, not b.data: that view is cap-limited + // to the data level and excludes the summary, so pooling it would make every + // later acquire fail the capacity check and allocate instead + backing := b.backing + b.backing = nil + b.data = nil + b.summary = nil + bitsetPool.Put(&backing) } diff --git a/util/bitset_iterator_test.go b/util/bitset_iterator_test.go new file mode 100644 index 000000000..1ba5917c4 --- /dev/null +++ b/util/bitset_iterator_test.go @@ -0,0 +1,376 @@ +// Copyright (c) 2026 Couchbase, Inc. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package util + +import ( + "math/rand" + "reflect" + "testing" + + "github.com/RoaringBitmap/roaring/v2" +) + +func roaringOf(vals ...uint32) *roaring.Bitmap { + return roaring.BitmapOf(vals...) +} + +func drainIterator(it BitsetIterator) []int { + var rv []int + for { + v, ok := it.Next() + if !ok { + return rv + } + rv = append(rv, v) + } +} + +func collectIterate(b *Bitset) []int { + var rv []int + b.Iterate(func(v int) { rv = append(rv, v) }) + return rv +} + +// naiveBits is the independent reference: it reads the lower level directly, +// bit by bit, ignoring the summary entirely. Both Iterate and Iterator are now +// summary-driven, so cross-checking them against each other would pass even if +// the summary were wrong -- they have to be checked against this instead. +func naiveBits(b *Bitset) []int { + var rv []int + for wordIdx, word := range b.data { + for bit := 0; bit < 64; bit++ { + if word&(1<>6]&(1<>6]&(1<= target { + want = append(want, v) + } + } + + it := b.Iterator() + it.AdvanceTo(target) + got := drainIterator(it) + + if len(want) == 0 && len(got) == 0 { + continue + } + if !reflect.DeepEqual(got, want) { + t.Fatalf("trial %d: AdvanceTo(%d) gave %v, want %v", trial, target, got, want) + } + } +} + +func TestBitsetSizeInBytes(t *testing.T) { + // 639/64+1 = 10 data words, plus 1 summary word covering them + b := NewBitset(639, nil) + if got, want := b.SizeInBytes(), 11*8; got != want { + t.Fatalf("got %d, want %d", got, want) + } + + // the summary is 1/64th of the data, rounded up + b = NewBitset(64*64*3-1, nil) // 192 data words + if got, want := len(b.data), 192; got != want { + t.Fatalf("data words: got %d, want %d", got, want) + } + if got, want := len(b.summary), 3; got != want { + t.Fatalf("summary words: got %d, want %d", got, want) + } +} + +// TestBitsetSummaryInvariant exercises every mutating path and checks the +// summary stays exact after each. +func TestBitsetSummaryInvariant(t *testing.T) { + b := NewBitset(500, nil) + assertSummaryExact(t, b) + + for _, v := range []int{0, 63, 64, 65, 200, 499, 500} { + b.Add(v) + assertSummaryExact(t, b) + } + + // removing one of two bits in a word must keep the summary bit set + b.Add(66) + assertSummaryExact(t, b) + b.Remove(65) + assertSummaryExact(t, b) + if b.summary[1>>6]&(1<<1) == 0 { + t.Fatal("summary bit cleared while word 1 still holds bits 64 and 66") + } + + // and emptying a word entirely must clear it. Word 3 (bits 192-255) holds + // only 200, so removing that is the case that actually exercises the + // clearing path -- word 1 keeps bit 64 no matter what else is removed. + b.Remove(200) + assertSummaryExact(t, b) + if b.summary[3>>6]&(1<<3) != 0 { + t.Fatal("summary bit still set after word 3 was emptied") + } + b.Remove(66) + assertSummaryExact(t, b) + + b.Clear() + assertSummaryExact(t, b) + if b.Count() != 0 { + t.Fatalf("Clear left %d bits", b.Count()) + } + + // Invert rewrites every word, so the summary is rebuilt wholesale + b.Add(5) + b.Invert() + assertSummaryExact(t, b) + if b.Contains(5) { + t.Fatal("Invert did not clear bit 5") + } + + // and again with an exclude set, which Invert also has to honour + excl := roaringOf(7, 300) + b2 := NewBitset(500, excl) + b2.Add(9) + b2.Invert() + assertSummaryExact(t, b2) + if b2.Contains(7) || b2.Contains(300) { + t.Fatal("Invert did not respect the exclude set") + } +} diff --git a/util/bitset_pool_test.go b/util/bitset_pool_test.go new file mode 100644 index 000000000..6b3f3ac91 --- /dev/null +++ b/util/bitset_pool_test.go @@ -0,0 +1,116 @@ +package util + +import "testing" + +// TestAcquireBitsetIsZeroed is the test that matters for pooling: a recycled +// bitset must not carry bits from whoever used it last. +func TestAcquireBitsetIsZeroed(t *testing.T) { + const maxVal = 4096 + + for i := 0; i < 32; i++ { + b := AcquireBitset(maxVal, nil) + for v := 0; v <= maxVal; v++ { + b.Add(v) + } + b.Release() + } + + for i := 0; i < 32; i++ { + b := AcquireBitset(maxVal, nil) + if got := b.Count(); got != 0 { + t.Fatalf("acquire %d returned %d stale bits", i, got) + } + b.Release() + } +} + +// TestAcquireBitsetSmallerReuse covers reusing a large pooled slice for a +// smaller bitset: the tail beyond the new size must be invisible. +func TestAcquireBitsetSmallerReuse(t *testing.T) { + big := AcquireBitset(8192, nil) + for v := 0; v <= 8192; v++ { + big.Add(v) + } + big.Release() + + for i := 0; i < 16; i++ { + small := AcquireBitset(100, nil) + if got := small.Count(); got != 0 { + t.Fatalf("small bitset saw %d stale bits", got) + } + small.Release() + } +} + +func TestBitsetReleaseIsIdempotent(t *testing.T) { + b := AcquireBitset(100, nil) + b.Add(5) + b.Release() + b.Release() // must not double-insert the same slice into the pool + + var nilBitset *Bitset + nilBitset.Release() // must not panic +} + +// TestAcquireBitsetSummaryIsPooled guards the interaction that broke when +// pooling and the two-level summary were combined: a pooled bitset has to carry +// both levels out of one allocation, and Release has to return the whole +// backing rather than the cap-limited data view. +func TestAcquireBitsetSummaryIsPooled(t *testing.T) { + const maxVal = 100000 + words, summaryWords := bitsetSizes(maxVal) + + for i := 0; i < 16; i++ { + b := AcquireBitset(maxVal, nil) + + if len(b.backing) != words+summaryWords { + t.Fatalf("backing is %d words, want %d", len(b.backing), words+summaryWords) + } + if len(b.summary) != summaryWords { + t.Fatalf("summary is %d words, want %d", len(b.summary), summaryWords) + } + // data must not be able to grow into the summary + if cap(b.data) != words { + t.Fatalf("data cap is %d, want %d", cap(b.data), words) + } + + // exercise both levels, which is what panicked when summary was nil + for v := 0; v <= maxVal; v += 7 { + b.Add(v) + } + for wordIdx, word := range b.data { + want := word != 0 + got := b.summary[wordIdx>>6]&(1<= words+summaryWords { + return // found it: Release returned the whole backing + } + } + t.Fatalf("no pooled slice held %d words; Release did not return the full backing", + words+summaryWords) +} From 129845cd9bbf2e982796234356bcaeea44b23977 Mon Sep 17 00:00:00 2001 From: Likith B Date: Wed, 30 Sep 2026 16:16:16 +0530 Subject: [PATCH 2/3] code cleanup --- document/field_numeric_v2.go | 22 ++-------- index/scorch/snapshot_index.go | 61 ++++++++++------------------ mapping/document.go | 7 ---- mapping/field.go | 4 +- numericv2/encode.go | 37 ++++++----------- numericv2/encode_test.go | 38 ++++++++--------- numericv2/query.go | 2 +- search/query/numeric_range_v2.go | 9 +--- search/query/query.go | 3 +- search/searcher/search_numeric_v2.go | 3 -- search_numeric_v2_test.go | 12 ------ util/bitset.go | 32 ++------------- 12 files changed, 64 insertions(+), 166 deletions(-) diff --git a/document/field_numeric_v2.go b/document/field_numeric_v2.go index 56e2eccfe..6c981edf5 100644 --- a/document/field_numeric_v2.go +++ b/document/field_numeric_v2.go @@ -32,15 +32,11 @@ func init() { } // DefaultNumericV2IndexingOptions mirrors the v1 numeric defaults, minus -// IncludeInAll: a number_v2 field produces no tokens, so it can never -// participate in the _all composite field. +// IncludeInAll const DefaultNumericV2IndexingOptions = index.StoreField | index.IndexField | index.DocValues // NumericV2Field is a numeric field indexed into the number_v2 section rather -// than as prefix-coded terms in the inverted index. It carries the value in two -// encodings because its two consumers need different things: the section's -// sorted search array wants the sortable uint64, while the sort and facet paths -// visit doc values as prefix-coded terms. +// than as prefix-coded terms in the inverted index. type NumericV2Field struct { name string arrayPositions []uint64 @@ -85,22 +81,12 @@ func (n *NumericV2Field) AnalyzedTokenFrequencies() index.TokenFrequencies { return nil } -// Value returns the stored-field representation, which is the same -// prefix-coded encoding a v1 NumericField stores. Keeping the two identical is -// what lets the stored-field decode path reuse NewNumericFieldFromBytes. +// Value returns the prefix-coded, zero-shift representation, which is both the +// stored-field form and the term written to this field's doc values. func (n *NumericV2Field) Value() []byte { return n.value } -// DocValueTerm returns the prefix-coded, zero-shift term written to this -// field's doc values. The sort and facet paths validate doc value bytes as -// prefix-coded and keep only shift-zero terms, so this encoding is required -// rather than incidental: handing them the raw sortable uint64 would make -// every document sort and facet as missing. -func (n *NumericV2Field) DocValueTerm() []byte { - return n.value -} - // SortableValue returns the value encoded as a uint64 whose unsigned ordering // matches the float64 ordering of the original number. func (n *NumericV2Field) SortableValue() uint64 { diff --git a/index/scorch/snapshot_index.go b/index/scorch/snapshot_index.go index 39275cdaf..af7e81eb6 100644 --- a/index/scorch/snapshot_index.go +++ b/index/scorch/snapshot_index.go @@ -73,7 +73,9 @@ func init() { var gcr IndexSnapshotGeoShapeV2Reader reflectStaticSizeIndexSnapshotGeoShapeV2Reader = int(reflect.TypeOf(gcr).Size()) var ncr IndexSnapshotNumericV2Reader - reflectStaticSizeIndexSnapshotNumericV2Reader = int(reflect.TypeOf(ncr).Size()) + // via a pointer: the struct holds a sync.Once, and taking its type by value + // would copy a lock + reflectStaticSizeIndexSnapshotNumericV2Reader = int(reflect.TypeOf(&ncr).Elem().Size()) var bsi util.BitsetIterator reflectStaticSizeBitsetIterator = int(reflect.TypeOf(bsi).Size()) var rip roaring.IntIterator @@ -1514,14 +1516,10 @@ func (g *IndexSnapshotGeoShapeV2Reader) Size() int { func (i *IndexSnapshot) NumericV2FieldReader(ctx context.Context, field string) ( index.NumericV2FieldReader, error) { - rv := &IndexSnapshotNumericV2Reader{ - field: field, - hits: make([]*util.Bitset, len(i.segment)), - // the zero value of BitsetIterator is a valid empty iterator, so a - // segment with no data for the field needs no special casing below + field: field, + hits: make([]*util.Bitset, len(i.segment)), iterators: make([]util.BitsetIterator, len(i.segment)), - counts: make([]uint64, len(i.segment)), snapshot: i, } @@ -1531,13 +1529,13 @@ func (i *IndexSnapshot) NumericV2FieldReader(ctx context.Context, field string) type IndexSnapshotNumericV2Reader struct { field string - // hits holds the per-segment result bitsets; iterators alias their words, - // so neither survives Close. + // hits holds the per-segment result bitsets hits []*util.Bitset iterators []util.BitsetIterator - // counts caches each segment's cardinality, computed once during Search so - // that Count is not a repeated scan. - counts []uint64 + + // count is the total cardinality, computed on the first call and cached. + countOnce sync.Once + count uint64 segmentOffset int @@ -1545,16 +1543,10 @@ type IndexSnapshotNumericV2Reader struct { } // Search evaluates the numeric range across all segments in the index snapshot. -// -// Every hit is materialised up front rather than streamed, because the segment -// stores values in value order while a searcher must emit document order; there -// is no streaming formulation. func (n *IndexSnapshotNumericV2Reader) Search(min, max *float64, inclusiveMin, inclusiveMax *bool) error { numSegments := len(n.snapshot.segment) - // the query holds only the encoded bounds, so it is safe to share - // across the per-segment goroutines query := numericv2.NewRangeQuery(min, max, inclusiveMin, inclusiveMax) var wg sync.WaitGroup @@ -1595,25 +1587,15 @@ func (n *IndexSnapshotNumericV2Reader) searchSeg(segID int, if err != nil { return err } - // return if the segment has no numeric data for the field if data == nil { return nil } - // release the reference on the segment's cached arrays once the evaluation - // is done, so that the cache is free to evict them defer data.Close() - // the stored doc numbers are already in segment space, so the snapshot's - // deleted bitmap applies directly with no translation hits := query.Evaluate(data, snapshot.deleted, snapshot.segment.Count()) - // the bitset is already the ideal representation for a dense result: it is - // iterated in place rather than converted into a roaring bitmap first, - // which would cost a rebuild and then an interface-dispatched container - // walk per hit to recover what the words already hold. n.hits[segID] = hits n.iterators[segID] = hits.Iterator() - n.counts[segID] = uint64(hits.Count()) return nil } @@ -1668,30 +1650,30 @@ func (n *IndexSnapshotNumericV2Reader) Advance(ID index.IndexInternalID, return n.Next(rv) } -// Close drops the reader's per-segment results. The segment-level arrays the -// search read from are not touched: those are owned and evicted by the -// segment's own cache, and were already released at the end of searchSeg. +// Close drops the reader's per-segment results func (n *IndexSnapshotNumericV2Reader) Close() error { - // the iterators alias the bitsets' words, so the pooled memory can only go - // back once iteration is finished -- which is exactly what Close means here for _, hits := range n.hits { if hits != nil { hits.Release() } } - // drop the references so a stray call after Close cannot read them + n.hits = nil n.iterators = nil return nil } // Count returns the total number of matching documents across all segments. +// The value is computed on the first call and cached for subsequent calls. func (n *IndexSnapshotNumericV2Reader) Count() uint64 { - var rv uint64 - for _, count := range n.counts { - rv += count - } - return rv + n.countOnce.Do(func() { + for _, hits := range n.hits { + if hits != nil { + n.count += uint64(hits.Count()) + } + } + }) + return n.count } // Size returns the estimated size in bytes of the @@ -1705,7 +1687,6 @@ func (n *IndexSnapshotNumericV2Reader) Size() int { } rv += reflectStaticSizeBitsetIterator * len(n.iterators) - rv += size.SizeOfUint64 * len(n.counts) return rv } diff --git a/mapping/document.go b/mapping/document.go index 0966eea51..48ee0c7b2 100644 --- a/mapping/document.go +++ b/mapping/document.go @@ -106,13 +106,6 @@ func validateFieldType(field *FieldMapping) error { switch field.Type { case "text", "datetime", "number", "number_v2", "boolean", "geopoint", "geoshape", "geoshape_v2", "IP": - if field.Type == "number_v2" && field.IncludeInAll { - // a number_v2 field produces no tokens, so it can never take part - // in the _all composite field; reject at index-definition time - // rather than silently matching nothing at query time - return fmt.Errorf("field: '%s', type 'number_v2' cannot be included in _all", - field.Name) - } return nil default: return fmt.Errorf("field: '%s', unknown field type: '%s'", diff --git a/mapping/field.go b/mapping/field.go index 8be2be603..1ffa241b0 100644 --- a/mapping/field.go +++ b/mapping/field.go @@ -214,9 +214,7 @@ func NewGeoShapeV2FieldMapping() *FieldMapping { } // NewNumberV2FieldMapping returns a default field mapping for numbers indexed -// into the number_v2 section. The defaults match NewNumericFieldMapping, except -// that IncludeInAll is false and cannot be enabled: the field produces no -// tokens, so it could never contribute to the _all composite field. +// into the number_v2 section. func NewNumberV2FieldMapping() *FieldMapping { return &FieldMapping{ Type: "number_v2", diff --git a/numericv2/encode.go b/numericv2/encode.go index 69298a73a..b45ea2d0e 100644 --- a/numericv2/encode.go +++ b/numericv2/encode.go @@ -25,7 +25,7 @@ import ( // signBit lifts an order-preserving int64 into an order-preserving uint64. // Because it is addition of 2^63 modulo 2^64, it commutes with the +1 and -1 -// steps Bounds applies for exclusive endpoints. +// steps bounds applies for exclusive endpoints. const signBit = uint64(1) << 63 // EncodeInt64 maps a sortable int64, as produced by numeric.Float64ToInt64, to @@ -35,27 +35,20 @@ func EncodeInt64(i int64) uint64 { return uint64(i) ^ signBit } -// Encode maps a float64 to a uint64 whose unsigned ordering matches the +// encode maps a float64 to a uint64 whose unsigned ordering matches the // float64 ordering of the input. -func Encode(f float64) uint64 { +func encode(f float64) uint64 { return EncodeInt64(numeric.Float64ToInt64(f)) } -// Decode is the inverse of Encode. -func Decode(v uint64) float64 { +// decode is the inverse of Encode. +func decode(v uint64) float64 { return numeric.Int64ToFloat64(int64(v ^ signBit)) } -// Bounds converts a query range into the inclusive uint64 interval [lo, hi] to -// scan. It deliberately mirrors searcher.NewNumericRangeSearcher step for step: -// an absent endpoint becomes the corresponding infinity, min is inclusive by -// default and max is not, and the adjustments for exclusive endpoints are -// guarded at the int64 extremes so they cannot wrap. -// -// A ±1 step in this space moves to the adjacent representable float64, so -// exclusive endpoints here are exact rather than approximate. When the range is -// empty, lo comes back greater than hi. -func Bounds(min, max *float64, inclusiveMin, inclusiveMax *bool) (lo, hi uint64) { +// bounds converts a query range into the inclusive uint64 interval [lo, hi] to +// scan. +func bounds(min, max *float64, inclusiveMin, inclusiveMax *bool) (lo, hi uint64) { // account for unbounded edges if min == nil { negInf := math.Inf(-1) @@ -65,22 +58,16 @@ func Bounds(min, max *float64, inclusiveMin, inclusiveMax *bool) (lo, hi uint64) inf := math.Inf(1) max = &inf } - if inclusiveMin == nil { - defaultInclusiveMin := true - inclusiveMin = &defaultInclusiveMin - } - if inclusiveMax == nil { - defaultInclusiveMax := false - inclusiveMax = &defaultInclusiveMax - } minInt64 := numeric.Float64ToInt64(*min) - if !*inclusiveMin && minInt64 != math.MaxInt64 { + // the minimum is inclusive unless the caller says otherwise + if inclusiveMin != nil && !*inclusiveMin && minInt64 != math.MaxInt64 { minInt64++ } maxInt64 := numeric.Float64ToInt64(*max) - if !*inclusiveMax && maxInt64 != math.MinInt64 { + // the maximum is exclusive unless the caller says otherwise + if (inclusiveMax == nil || !*inclusiveMax) && maxInt64 != math.MinInt64 { maxInt64-- } diff --git a/numericv2/encode_test.go b/numericv2/encode_test.go index dcc3dbd78..955815ebd 100644 --- a/numericv2/encode_test.go +++ b/numericv2/encode_test.go @@ -42,7 +42,7 @@ func TestEncodeIsMonotone(t *testing.T) { } for i := 1; i < len(vals); i++ { - prev, cur := Encode(vals[i-1]), Encode(vals[i]) + prev, cur := encode(vals[i-1]), encode(vals[i]) if prev >= cur { t.Fatalf("Encode not monotone at %v -> %v: %d >= %d", vals[i-1], vals[i], prev, cur) @@ -51,14 +51,14 @@ func TestEncodeIsMonotone(t *testing.T) { // -0.0 and +0.0 compare equal as floats but are distinct bit patterns; // what matters is that neither breaks ordering against its neighbours - if Encode(math.Copysign(0, -1)) > Encode(0) { + if encode(math.Copysign(0, -1)) > encode(0) { t.Fatal("negative zero encodes above positive zero") } } func TestEncodeDecodeRoundTrip(t *testing.T) { for _, f := range []float64{-1e300, -1.5, -1, 0, 0.5, 1, 42, 1e300} { - if got := Decode(Encode(f)); got != f { + if got := decode(encode(f)); got != f { t.Fatalf("round trip of %v gave %v", f, got) } } @@ -69,7 +69,7 @@ func b(v bool) *bool { return &v } // TestBoundsMatchesInvertedPath is the load-bearing test for the encoding: it // recomputes the int64 bounds exactly the way NewNumericRangeSearcher does and -// requires Bounds to agree after the sign-bit lift. If these ever diverge, the +// requires bounds to agree after the sign-bit lift. If these ever diverge, the // two numeric paths silently return different hits for the same query. func TestBoundsMatchesInvertedPath(t *testing.T) { // mirror of the arithmetic at the top of NewNumericRangeSearcher @@ -110,7 +110,7 @@ func TestBoundsMatchesInvertedPath(t *testing.T) { for _, incMin := range incs { for _, incMax := range incs { wantLo, wantHi := reference(min, max, incMin, incMax) - gotLo, gotHi := Bounds(min, max, incMin, incMax) + gotLo, gotHi := bounds(min, max, incMin, incMax) if gotLo != EncodeInt64(wantLo) { t.Fatalf("lo mismatch for [%v,%v] inc(%v,%v): got %d, want %d", @@ -129,7 +129,7 @@ func TestBoundsMatchesInvertedPath(t *testing.T) { } // TestBoundsExtremeGuards pins down what the MaxInt64/MinInt64 guards in -// Bounds actually protect. They are not infinity guards: Float64ToInt64(+Inf) +// bounds actually protect. They are not infinity guards: Float64ToInt64(+Inf) // is 0x7FF0000000000000, comfortably short of MaxInt64, so an exclusive bound // at an infinity increments normally -- into a NaN bit pattern, exactly as the // inverted-index path does. The int64 extremes correspond to NaN payloads, so @@ -148,26 +148,26 @@ func TestBoundsExtremeGuards(t *testing.T) { } // an exclusive minimum at the top must not wrap to zero - lo, _ := Bounds(&maxKey, nil, b(false), nil) + lo, _ := bounds(&maxKey, nil, b(false), nil) if lo != EncodeInt64(math.MaxInt64) { t.Fatalf("exclusive min at the int64 max wrapped: got %d, want %d", lo, uint64(EncodeInt64(math.MaxInt64))) } // an exclusive maximum at the bottom must not wrap to the top - _, hi := Bounds(nil, &minKey, nil, b(false)) + _, hi := bounds(nil, &minKey, nil, b(false)) if hi != EncodeInt64(math.MinInt64) { t.Fatalf("exclusive max at the int64 min wrapped: got %d, want %d", hi, uint64(EncodeInt64(math.MinInt64))) } // and the infinities, which do increment, must still move upward - loInf, _ := Bounds(f64(math.Inf(1)), nil, b(false), nil) - if loInf <= Encode(math.Inf(1)) { + loInf, _ := bounds(f64(math.Inf(1)), nil, b(false), nil) + if loInf <= encode(math.Inf(1)) { t.Fatalf("exclusive min at +Inf did not move upward: got %d", loInf) } - _, hiInf := Bounds(nil, f64(math.Inf(-1)), nil, b(false)) - if hiInf >= Encode(math.Inf(-1)) { + _, hiInf := bounds(nil, f64(math.Inf(-1)), nil, b(false)) + if hiInf >= encode(math.Inf(-1)) { t.Fatalf("exclusive max at -Inf did not move downward: got %d", hiInf) } } @@ -175,13 +175,13 @@ func TestBoundsExtremeGuards(t *testing.T) { // TestBoundsExclusiveIsAdjacentFloat documents that a step in this space moves // to the neighbouring representable float64, so exclusive bounds are exact. func TestBoundsExclusiveIsAdjacentFloat(t *testing.T) { - lo, _ := Bounds(f64(1.0), nil, b(false), nil) - if got, want := Decode(lo), math.Nextafter(1.0, math.Inf(1)); got != want { + lo, _ := bounds(f64(1.0), nil, b(false), nil) + if got, want := decode(lo), math.Nextafter(1.0, math.Inf(1)); got != want { t.Fatalf("exclusive min above 1.0: got %v, want %v", got, want) } - _, hi := Bounds(nil, f64(1.0), nil, b(false)) - if got, want := Decode(hi), math.Nextafter(1.0, math.Inf(-1)); got != want { + _, hi := bounds(nil, f64(1.0), nil, b(false)) + if got, want := decode(hi), math.Nextafter(1.0, math.Inf(-1)); got != want { t.Fatalf("exclusive max below 1.0: got %v, want %v", got, want) } } @@ -190,19 +190,19 @@ func TestBoundsExclusiveIsAdjacentFloat(t *testing.T) { // which Evaluate relies on to short-circuit. func TestBoundsEmptyRange(t *testing.T) { // min == max with both endpoints exclusive is empty - lo, hi := Bounds(f64(5), f64(5), b(false), b(false)) + lo, hi := bounds(f64(5), f64(5), b(false), b(false)) if lo <= hi { t.Fatalf("expected an empty range, got lo=%d hi=%d", lo, hi) } // an inverted range is empty - lo, hi = Bounds(f64(10), f64(1), nil, nil) + lo, hi = bounds(f64(10), f64(1), nil, nil) if lo <= hi { t.Fatalf("expected an empty range, got lo=%d hi=%d", lo, hi) } // min == max inclusive on both sides matches exactly one value - lo, hi = Bounds(f64(5), f64(5), b(true), b(true)) + lo, hi = bounds(f64(5), f64(5), b(true), b(true)) if lo != hi { t.Fatalf("expected a single-value range, got lo=%d hi=%d", lo, hi) } diff --git a/numericv2/query.go b/numericv2/query.go index 27d4e6edb..657b91d44 100644 --- a/numericv2/query.go +++ b/numericv2/query.go @@ -31,7 +31,7 @@ type Query struct { // NewRangeQuery builds a query for the given range. A nil endpoint is // unbounded; inclusiveMin defaults to true and inclusiveMax to false. func NewRangeQuery(min, max *float64, inclusiveMin, inclusiveMax *bool) *Query { - lo, hi := Bounds(min, max, inclusiveMin, inclusiveMax) + lo, hi := bounds(min, max, inclusiveMin, inclusiveMax) return &Query{lo: lo, hi: hi} } diff --git a/search/query/numeric_range_v2.go b/search/query/numeric_range_v2.go index 539d70343..caabdb3bc 100644 --- a/search/query/numeric_range_v2.go +++ b/search/query/numeric_range_v2.go @@ -25,9 +25,6 @@ import ( index "github.com/blevesearch/bleve_index_api" ) -// NumericRangeV2 is the range itself. It is nested under a versioned key rather -// than living at the top level, because ParseQuery already claims top-level -// min/max for NumericRangeQuery and TermRangeQuery. type NumericRangeV2 struct { Min *float64 `json:"min,omitempty"` Max *float64 `json:"max,omitempty"` @@ -44,15 +41,13 @@ type NumericRangeV2Query struct { BoostVal *Boost `json:"boost,omitempty"` } -// NewNumericRangeV2Query creates a new query for ranges of numeric values over -// a number_v2 field. Either, but not both, endpoints can be nil. The minimum -// value is inclusive; the maximum value is exclusive. +// NewNumericRangeV2Query creates a new query for ranges of numeric values. func NewNumericRangeV2Query(min, max *float64) *NumericRangeV2Query { return NewNumericRangeV2InclusiveQuery(min, max, nil, nil) } // NewNumericRangeV2InclusiveQuery creates a new query for ranges of numeric -// values over a number_v2 field, with explicit control over endpoint inclusion. +// values. func NewNumericRangeV2InclusiveQuery(min, max *float64, minInclusive, maxInclusive *bool) *NumericRangeV2Query { return &NumericRangeV2Query{ diff --git a/search/query/query.go b/search/query/query.go index 249865b30..376b5c807 100644 --- a/search/query/query.go +++ b/search/query/query.go @@ -233,8 +233,7 @@ func ParseQuery(input []byte) (Query, error) { } return &rv, nil } - // checked ahead of the top-level min/max branches below: a number_v2 range - // is nested under its own key precisely because those are already taken + _, hasRangeV2 := tmp["range_v2"] if hasRangeV2 { var rv NumericRangeV2Query diff --git a/search/searcher/search_numeric_v2.go b/search/searcher/search_numeric_v2.go index bb4ba0b5a..3e2cce56b 100644 --- a/search/searcher/search_numeric_v2.go +++ b/search/searcher/search_numeric_v2.go @@ -31,9 +31,6 @@ func init() { reflectStaticSizeNumericV2Searcher = int(reflect.TypeOf(nv2s).Size()) } -// NumericV2Searcher is a filtering searcher over a number_v2 field. Every -// document the underlying reader returns is a confirmed match, so the searcher -// contributes a constant score rather than a computed one. type NumericV2Searcher struct { numericIndexReader index.NumericV2FieldReader scorer *scorer.ConstantScorer diff --git a/search_numeric_v2_test.go b/search_numeric_v2_test.go index 9ad8fad02..c9df9909a 100644 --- a/search_numeric_v2_test.go +++ b/search_numeric_v2_test.go @@ -557,18 +557,6 @@ func TestNumericV2Validation(t *testing.T) { if err := query.NewNumericRangeV2Query(nil, &nan).Validate(); err == nil { t.Fatal("expected a NaN maximum to be rejected") } - - // IncludeInAll must be rejected at index-definition time - im := NewIndexMapping() - fm := NewNumberV2FieldMapping() - fm.Name = nv2FieldV2 - fm.IncludeInAll = true - dm := NewDocumentMapping() - dm.AddFieldMappingsAt(nv2FieldV2, fm) - im.DefaultMapping = dm - if err := im.Validate(); err == nil { - t.Fatal("expected IncludeInAll on a number_v2 field to be rejected") - } } // TestNumericV2MultiValued indexes an array of numbers and checks that a diff --git a/util/bitset.go b/util/bitset.go index 7dd970f56..4b198e02a 100644 --- a/util/bitset.go +++ b/util/bitset.go @@ -26,29 +26,13 @@ import ( // that word is non-empty, so iteration can skip 64 empty words at a time // instead of loading each one. // -// That matters because a bitset is sized by the document count of the segment -// it covers, not by the number of bits actually set: without the summary, a -// query matching 0.1% of a five-million-document segment still walks all 78,000 -// words to find its 5,000 hits. -// // The invariant is exact -- a summary bit is set if and only if the // corresponding data word is non-zero -- so every mutating method below has to // maintain it. A stale set bit would still be correct, since iteration // re-checks the word it names and skips it when empty; exactness is what // preserves the performance the summary exists for. -// -// Maintaining it inside Add costs one extra store per value, against a region -// 64x smaller than the data which stays cache-resident. Measured against the -// same benchmark at four densities on a five-million-document segment, that -// buys 3x at 0.1% and roughly 1.1x at 1%, breaks even at 10%, and costs about -// 9% at 50% where nearly every word is occupied and there is nothing to skip. -// Two alternatives were tried and rejected: making the store conditional on the -// word having been empty regressed 10% density by 19% on an unpredictable -// branch, and rebuilding the summary in one pass per iteration regressed 1% -// density by 27% by paying an O(words) pass on top of the walk. type Bitset struct { - // backing owns the memory; data and summary are views over it, so that a - // pooled bitset is a single allocation. + // backing owns the memory; data and summary are views over it backing []uint64 data []uint64 summary []uint64 @@ -150,8 +134,6 @@ func (b *Bitset) Invert() { } } - // every word just changed; Invert is already O(words) so rebuilding here - // rather than deferring costs nothing extra b.rebuildSummary() } @@ -211,23 +193,15 @@ func (b *Bitset) SizeInBytes() int { // BitsetIterator walks the set bits of a Bitset in ascending order, using the // summary level to skip runs of empty words. -// -// It is a value type on purpose: the caller keeps it in a slice and calls -// through it once per hit, so an interface or a pointer chase per call would -// cost more than the bit extraction itself. The zero value is a valid, empty -// iterator, which lets callers leave a slot unset rather than nil-checking on -// the hot path. type BitsetIterator struct { words []uint64 summary []uint64 - // summaryIdx is the summary word being consumed, and summaryWord its bits - // that have not been visited yet; each names a non-empty data word. + // summaryIdx is the summary word being consumed summaryIdx int summaryWord uint64 - // wordIdx is the data word being consumed, and word its bits that have not - // been returned yet, so the lowest set bit is always the next value. + // wordIdx is the data word being consumed wordIdx int word uint64 } From 55147047fa8be27be388da66c1e857c4a2bd7bc9 Mon Sep 17 00:00:00 2001 From: Likith B Date: Thu, 1 Oct 2026 16:22:21 +0530 Subject: [PATCH 3/3] interface change --- document/field_numeric_v2.go | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/document/field_numeric_v2.go b/document/field_numeric_v2.go index 6c981edf5..af253b573 100644 --- a/document/field_numeric_v2.go +++ b/document/field_numeric_v2.go @@ -87,6 +87,12 @@ func (n *NumericV2Field) Value() []byte { return n.value } +// DocValue returns the term written to this field's doc values. +// Duplicate needed to serve readers at the segment level +func (n *NumericV2Field) DocValue() []byte { + return n.value +} + // SortableValue returns the value encoded as a uint64 whose unsigned ordering // matches the float64 ordering of the original number. func (n *NumericV2Field) SortableValue() uint64 {