Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
21 commits
Select commit Hold shift + click to select a range
d28e060
perf: block-at-a-time scan and windowed disjunction accumulator
capemox Aug 12, 2026
6680848
search: drain the accumulator via a matched bitset instead of a lo..h…
capemox Aug 26, 2026
ac52bfe
perf: decouple accumulator window from BlockSize
capemox Aug 26, 2026
18368db
search: pool DocScoreBlocks instead of allocating per clause
capemox Aug 13, 2026
69867fc
search: return pooled blocks on the abandoned-construction path
capemox Aug 26, 2026
d25d39f
perf: default to zapx v18's bitpacked postings format
capemox Sep 6, 2026
2f8acf8
perf: cost-gated intersection strategy for conjunction optimization
capemox Sep 6, 2026
7241a50
perf: block-max WAND for standalone term queries
capemox Sep 6, 2026
4f3bd8a
perf: gate block-max WAND on BM25 scoring
capemox Sep 7, 2026
0435338
perf: block-max WAND for conjunction queries
capemox Sep 8, 2026
0b5a871
perf: vectorize BM25/tf-idf scoring in ScoreBulk
capemox Sep 8, 2026
1a2f2ea
Point zapx/v18 replace at the wand/bulk-scan worktree
capemox Sep 17, 2026
5fc5c78
Add independent-reference conjunction test; fix flaky TestBytesRead stat
capemox Sep 17, 2026
67ee082
Add adaptive bailout for conjunction block-WAND when the prefilter do…
capemox Sep 18, 2026
1412fa7
Revert "Add adaptive bailout for conjunction block-WAND when the pref…
capemox Sep 18, 2026
f73a075
Depend on freeway's simd package instead of a duplicated local copy
capemox Sep 19, 2026
c7d14d7
Repoint zapx/scorch_segment_api at bitpack-simd-based wand/block-max
capemox Sep 19, 2026
48efc5f
Skip TestTermRangeSearchTooManyTerms with a clear reason, not a red f…
capemox Sep 19, 2026
5c891d4
Reapply "Add adaptive bailout for conjunction block-WAND when the pre…
capemox Sep 19, 2026
3105bbb
Skip ScoreBulk's SIMD batch dispatch for single-document calls
capemox Sep 19, 2026
c1e2309
Add AdvanceDocNum: skip the ID encode/decode round trip in scoreCandi…
capemox Sep 19, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
175 changes: 175 additions & 0 deletions bulk_pagination_tiedscore_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,175 @@
// Copyright (c) 2026 Couchbase, Inc.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.

package bleve

import (
"fmt"
"os"
"strings"
"testing"

"github.com/blevesearch/bleve/v2/mapping"
"github.com/blevesearch/bleve/v2/search/query"
index "github.com/blevesearch/bleve_index_api"
)

// buildBulkTiedScoreIndex indexes n BM25-scored documents with deliberately
// coarse term-membership (~33%/~25%/~50%, via mod-3/mod-4/mod-2 -- pairwise
// coprime so memberships are independent) and only 6 distinct filler
// lengths, so groups of hundreds of documents share the exact same
// (term-membership, field-length) combination and therefore the exact same
// BM25 score -- the scenario a HitNumber-based tie-break actually has to
// arbitrate, repeatedly, across a real-sized corpus.
func buildBulkTiedScoreIndex(t *testing.T, n int) Index {
t.Helper()

dir, err := os.MkdirTemp("", "bulktied")
if err != nil {
t.Fatal(err)
}
t.Cleanup(func() { _ = os.RemoveAll(dir) })

im := mapping.NewIndexMapping()
im.DefaultAnalyzer = "standard"
im.ScoringModel = index.BM25Scoring
idx, err := New(dir+"/i.bleve", im)
if err != nil {
t.Fatal(err)
}
t.Cleanup(func() { _ = idx.Close() })

batch := idx.NewBatch()
for i := 0; i < n; i++ {
var terms []string
if i%3 == 0 { // ~33%
terms = append(terms, "alpha")
}
if i%4 == 0 { // ~25%
terms = append(terms, "beta")
}
if i%2 == 0 { // ~50%
terms = append(terms, "common")
}
fillerCount := i%6 + 1 // only 6 distinct lengths -> heavy score collisions
filler := make([]string, fillerCount)
for j := range filler {
filler[j] = "filler"
}
body := strings.Join(append(terms, filler...), " ")
if err := batch.Index(fmt.Sprintf("d%05d", i), map[string]interface{}{
"body": body,
}); err != nil {
t.Fatal(err)
}
if batch.Size() >= 500 {
if err := idx.Batch(batch); err != nil {
t.Fatal(err)
}
batch = idx.NewBatch()
}
}
if batch.Size() > 0 {
if err := idx.Batch(batch); err != nil {
t.Fatal(err)
}
}
return idx
}

// TestBulkCollectPaginationWithTiedScores checks that collectBulk's
// HitNumber-based tie-break stays self-consistent under pagination: since
// hc.total increments once per document ScoreBlock hands it (see
// TopNCollector.collectBulk), a document's HitNumber -- and so its relative
// order among score-ties -- must be the same whether it's produced by a
// small paginated request or reconstructed by slicing one exhaustive scan.
// This is the same shape of bug found on wand/block-max (a threshold-driven
// path selectively skipping documents shifted which of two tied-score docs
// "arrived" first), checked fresh here because collectBulk's own block-max
// WAND layer (search_conjunction_block.go's EnableConjunctionBlockMaxWAND)
// is a different mechanism that could reintroduce the same class of issue.
func TestBulkCollectPaginationWithTiedScores(t *testing.T) {
const n = 6000
idx := buildBulkTiedScoreIndex(t, n)

bodyTerm := func(term string) query.Query {
q := query.NewTermQuery(term)
q.SetField("body")
return q
}

cases := []struct {
name string
q query.Query
}{
{"term", bodyTerm("common")},
{"conjunction", query.NewConjunctionQuery([]query.Query{bodyTerm("alpha"), bodyTerm("common")})},
{"disjunction", query.NewDisjunctionQuery([]query.Query{bodyTerm("alpha"), bodyTerm("beta")})},
}

pages := []struct {
name string
from, size int
}{
{"top5", 0, 5},
{"top50", 0, 50},
{"top500", 0, 500},
{"mid-30-40", 30, 40},
{"mid-300-100", 300, 100},
{"deep-999-137", 999, 137},
{"deep-1999-250", 1999, 250},
}

for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
exhaustive := NewSearchRequest(c.q)
exhaustive.Size = n
exRes, err := idx.Search(exhaustive)
if err != nil {
t.Fatal(err)
}
if len(exRes.Hits) < 500 {
t.Fatalf("corpus/query not dense enough: only %d hits", len(exRes.Hits))
}

for _, p := range pages {
t.Run(p.name, func(t *testing.T) {
req := NewSearchRequestOptions(c.q, p.size, p.from, false)
res, err := idx.Search(req)
if err != nil {
t.Fatal(err)
}
lo, hi := p.from, p.from+p.size
if hi > len(exRes.Hits) {
hi = len(exRes.Hits)
}
if lo > hi {
lo = hi
}
want := exRes.Hits[lo:hi]
if len(res.Hits) != len(want) {
t.Fatalf("From=%d,Size=%d returned %d hits, want %d",
p.from, p.size, len(res.Hits), len(want))
}
for i := range want {
if res.Hits[i].ID != want[i].ID || res.Hits[i].Score != want[i].Score {
t.Fatalf("hit %d: From=%d,Size=%d got (id=%s,score=%v), want (id=%s,score=%v) from exhaustive baseline",
i, p.from, p.size, res.Hits[i].ID, res.Hits[i].Score, want[i].ID, want[i].Score)
}
}
})
}
})
}
}
216 changes: 216 additions & 0 deletions conjunction_block_bound_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,216 @@
// Copyright (c) 2026 Couchbase, Inc.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.

package bleve

import (
"fmt"
"os"
"strings"
"testing"

"github.com/blevesearch/bleve/v2/mapping"
"github.com/blevesearch/bleve/v2/search/searcher"
index "github.com/blevesearch/bleve_index_api"
)

// buildSkewedSegmentIndex indexes n BM25-scored documents whose field length
// alternates in stark, batch-sized steps (short, long, short, long, ...)
// rather than being uniformly distributed within every segment -- so each
// segment's own local average field length is, by construction, far from
// the corpus-wide average across all segments (a segment made of one
// all-short batch or one all-long batch is never a representative sample
// of the whole).
//
// This specifically targets a real bug class found in the bulk-scan port:
// zapx's write-time block-max bound picked, per block, whichever real
// document scored highest under a BM25 estimate computed from that block's
// own SEGMENT's local average field length -- but the query-time scorer
// ranks documents using the corpus-wide average (bleve_index_api's
// IndexSnapshot.FieldCardinality/DocCount, spanning every segment). BM25's
// ranking across documents with different (norm, tf) pairs is not invariant
// to which average is used, so a bound picked as "the best real document"
// under the wrong (local) average can score lower under the right (global)
// one than a document the write-time estimate passed over -- silently
// dropping a genuine top-K match, either via the whole-window skip or the
// phase-1 score prefilter in search_conjunction_block.go. A uniformly-
// distributed corpus (see buildConjunctionWANDIndex) doesn't reliably
// trigger this: with enough i.i.d. samples per segment, the local and
// global averages converge to nearly the same value by the law of large
// numbers, and the bound stays sound in practice even with the underlying
// bug present. This corpus deliberately breaks that convergence.
func buildSkewedSegmentIndex(t *testing.T, n int) Index {
t.Helper()

dir, err := os.MkdirTemp("", "conjskew")
if err != nil {
t.Fatal(err)
}
t.Cleanup(func() { _ = os.RemoveAll(dir) })

im := mapping.NewIndexMapping()
im.DefaultAnalyzer = "standard"
im.ScoringModel = index.BM25Scoring
idx, err := New(dir+"/i.bleve", im)
if err != nil {
t.Fatal(err)
}
t.Cleanup(func() { _ = idx.Close() })

batch := idx.NewBatch()
for i := 0; i < n; i++ {
var terms []string
if i%97 == 0 {
terms = append(terms, "alpha")
}
if i%13 == 0 {
terms = append(terms, "beta")
}
if i%5 != 0 {
terms = append(terms, "common")
}
// A stark step, not a gentle trend or a fine-grained alternation:
// the corpus's first half is uniformly short, its second half
// uniformly long. A per-batch alternation was tried first and
// didn't survive scorch's background merging -- adjacent short/long
// batches merge into a blended segment whose local average drifts
// back toward the corpus-wide one, exactly the convergence this
// corpus needs to avoid. One contiguous block per extreme is robust
// to that: segment merges within a block stay homogeneous (short
// merges with short, long with long) for far longer, since a
// same-tier size-based merge policy prefers merging segments of
// similar size drawn from nearby, similarly-aged batches.
var fillerCount int
if i < n/2 {
fillerCount = 1
} else {
fillerCount = 120
}
filler := make([]string, fillerCount)
for j := range filler {
filler[j] = "filler"
}
body := strings.Join(append(terms, filler...), " ")
if err := batch.Index(fmt.Sprintf("d%06d", i), map[string]interface{}{
"body": body,
}); err != nil {
t.Fatal(err)
}
if batch.Size() >= 500 {
if err := idx.Batch(batch); err != nil {
t.Fatal(err)
}
batch = idx.NewBatch()
}
}
if batch.Size() > 0 {
if err := idx.Batch(batch); err != nil {
t.Fatal(err)
}
}
return idx
}

// TestConjunctionBlockMaxWANDAgainstIndependentBaseline is the companion
// TestConjunctionBlockMaxWANDMatchesBaseline was missing: that test only
// ever compares the block-max WAND path against WAND-off *within the same
// build*, both reading through the very same TermQueryScorer.avgDocLength
// (a corpus-wide statistic) -- so it can't distinguish "correct" from "a
// write-time bound computed against a different, wrong statistic," since
// WAND-off never touches the write-time bound machinery at all and neither
// comparison side is an independent computation of the *true* per-document
// score. This test's WAND-off side plays that independent-reference role
// properly: it never reads zapx's block-max metadata (the exact thing that
// was wrong), only real per-document (tf, norm) pairs decoded one at a
// time -- so it stays correct regardless of any bug in how the write-time
// bound was chosen. See buildSkewedSegmentIndex's doc comment for why this
// needs a segment-skewed corpus rather than the smaller uniform one
// TestConjunctionBlockMaxWANDMatchesBaseline already uses: that one's
// per-segment/global averages are close enough, even with a bug present,
// that this exact class of bound violation had been shipping unnoticed.
func TestConjunctionBlockMaxWANDAgainstIndependentBaseline(t *testing.T) {
const n = 60000
idx := buildSkewedSegmentIndex(t, n)

t.Cleanup(func() { searcher.EnableConjunctionBlockMaxWAND = true })

shapes := []struct {
name string
terms []string
}{
{"skewed-2term", []string{"alpha", "common"}},
{"mid-2term", []string{"beta", "common"}},
{"3term", []string{"alpha", "beta", "common"}},
}

for _, shape := range shapes {
for _, size := range []int{5, 10, 50} {
t.Run(fmt.Sprintf("%s/size=%d", shape.name, size), func(t *testing.T) {
q := andQuery(shape.terms...)

searcher.EnableConjunctionBlockMaxWAND = true
reqOn := NewSearchRequest(q)
reqOn.Size = size
resOn, err := idx.Search(reqOn)
if err != nil {
t.Fatalf("WAND-on search failed: %v", err)
}

// The independent reference: real per-document (tf, norm)
// pairs decoded one at a time via the plain scalar leapfrog
// path, which never reads zapx's write-time block-max bound
// at all -- unlike an exhaustive Size=n WAND-on scan, which
// would still run through the very same buggy bound
// machinery and just happen not to need it (nothing gets
// excluded when every match is wanted), so it would
// silently pass even with the bug this test targets present.
searcher.EnableConjunctionBlockMaxWAND = false
reqOff := NewSearchRequest(q)
reqOff.Size = size
resOff, err := idx.Search(reqOff)
if err != nil {
t.Fatalf("WAND-off search failed: %v", err)
}
searcher.EnableConjunctionBlockMaxWAND = true

if len(resOn.Hits) != len(resOff.Hits) {
offIDs := map[string]float64{}
for _, h := range resOff.Hits {
offIDs[h.ID] = h.Score
}
for _, h := range resOn.Hits {
delete(offIDs, h.ID)
}
for id, score := range offIDs {
t.Logf("missing from WAND-on top-%d: id=%s score=%v (present in the WAND-off baseline, should have made the cut)",
size, id, score)
}
t.Fatalf("hit count mismatch: WAND-on=%d WAND-off=%d",
len(resOn.Hits), len(resOff.Hits))
}
for i := range resOff.Hits {
on, off := resOn.Hits[i], resOff.Hits[i]
if on.ID != off.ID {
t.Errorf("hit %d: ID mismatch: WAND-on=%s WAND-off=%s (score on=%v off=%v)",
i, on.ID, off.ID, on.Score, off.Score)
}
if on.Score != off.Score {
t.Errorf("hit %d (id=%s): score mismatch: WAND-on=%v WAND-off=%v",
i, on.ID, on.Score, off.Score)
}
}
})
}
}
}
Loading
Loading