From 5edf0d1629a2702a9732c3014b5168b0ba431eba Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Sat, 29 Aug 2026 09:38:34 +0000 Subject: [PATCH 01/41] =?UTF-8?q?events:=208=20KiB=20DataCF=20blocks=20?= =?UTF-8?q?=E2=80=94=20the=20block=20is=20getEvents'=20decompression=20uni?= =?UTF-8?q?t?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A point read decompresses one block to serve one ~250B event, so the 32 KiB block paid ~128 events of zstd work per cache miss. Measured under the K-stratified events corpus at limit=1000 (8 vs 32 KiB): service p50 -53% / p99 -59% and peak RSS -41% at stress density (sac-6000, 2,500-ledger fixture); neutral at pubnet density; +0.8% disk; ingest wall unchanged; 4 KiB adds nothing further. Hot tier only - RocksDB reads any block size, so existing chunks are unaffected and pick the new size up on natural rotation. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/stores/event/hot_store.go | 13 ++++++++++--- 1 file changed, 10 insertions(+), 3 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go index ec2e3730c..006ae91a1 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go @@ -29,8 +29,15 @@ const ( // // - DataCF holds XDR-encoded event payloads: compressible (zstd // typically 2-3× on XDR) and read in batches via -// BatchedMultiGetCF. Larger blocks give zstd more context per -// compression unit and align with batch-fetch shapes. +// BatchedMultiGetCF. The block is the decompression unit of a +// point read, so its size trades compression context against +// per-miss work: getEvents fetches scattered ~250B events, and a +// 32 KiB block made every cache miss decompress ~128 events to +// serve one. Measured under the K-stratified events corpus at +// limit=1000, 8 KiB vs 32 KiB: service p50 −53% / p99 −59% and +// −41% peak RSS at stress density (sac-6000), neutral on pubnet; +// +0.8% on disk; ingest wall unchanged; 4 KiB adds nothing more +// (the curve is flat below 8 KiB). // - IndexCF stores 20-byte (term_hash || event_id) keys with // empty values — nothing in the values to compress, and small // blocks reduce wasted I/O per random Lookup miss (each Lookup @@ -38,7 +45,7 @@ const ( // - OffsetsCF stores 8-byte (ledger_seq -> event_count) rows in // the tens-of-thousands per chunk — same shape as IndexCF. const ( - dataCFBlockSize = 32 * 1024 + dataCFBlockSize = 8 * 1024 indexCFBlockSize = 4 * 1024 offsetsCFBlockSize = 4 * 1024 ) From a1de1b6d491255a662e139d4e3b3e19723bcbfe5 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Sat, 29 Aug 2026 09:41:35 +0000 Subject: [PATCH 02/41] events: batch arenas for fetched payload copies; page-sized first match batch MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Both tiers' FetchEvents copied every fetched payload into its own allocation — hot out of RocksDB's pinned pages, cold out of the packfile record buffer — at ~500-1500 payloads per page. A batch's payloads live and die together, so the copies now land in per-batch arenas: one backing allocation per hot batch (values share it and are read-only), a chunked arena on the cold side where sizes stream in unknown. Measured on the serving alloc profile the per-value clones were ~10% of hot serving allocations. batchSizes now honors a page-sized hint upward too (capped at eight default batches), so a limit=1000 page is one storage round trip instead of three. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/rocksdb/rocksdb.go | 16 ++++++-- .../internal/rpcv2/stores/event/arena.go | 24 +++++++++++ .../internal/rpcv2/stores/event/arena_test.go | 41 +++++++++++++++++++ .../rpcv2/stores/event/cold_reader.go | 9 ++-- .../internal/rpcv2/stores/event/match.go | 13 +++--- 5 files changed, 92 insertions(+), 11 deletions(-) create mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go create mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go diff --git a/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go b/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go index edb3dfde4..ea1e1e703 100644 --- a/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go +++ b/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go @@ -275,14 +275,24 @@ func (s *Store) BatchMultiGet(cf string, keys [][]byte) ([][]byte, error) { } defer pinned.Destroy() + // Copy out of the pinned cache pages (Destroy invalidates them) — + // through one arena, not a clone per value: a batch is fetched, + // decoded, and dropped as a unit, so per-value allocations only add + // GC work. The returned slices therefore share one backing array: + // they are read-only, and retaining any of them retains the batch. + total := 0 + for _, p := range pinned { + total += len(p.Data()) + } + arena := make([]byte, 0, total) results := make([][]byte, len(keys)) for i, p := range pinned { if !p.Exists() { continue } - // p.Data() points into the pinned cache page; copy before - // Destroy invalidates it. - results[i] = bytes.Clone(p.Data()) + n := len(arena) + arena = append(arena, p.Data()...) + results[i] = arena[n:len(arena):len(arena)] } return results, nil } diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go new file mode 100644 index 000000000..4e6e1c14e --- /dev/null +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go @@ -0,0 +1,24 @@ +package event + +// byteArena hands out stable copies of transient byte slices from large +// chunked allocations, so a fetch that copies hundreds of small payloads +// costs a handful of allocations instead of one per payload. Chunks are +// only ever appended within capacity, so previously returned copies never +// move. Zero value is ready; not safe for concurrent use. +type byteArena struct { + buf []byte +} + +// arenaChunkSize is the arena's allocation unit. Big enough that a +// 512-candidate fetch of ~250B payloads fits in one or two chunks, small +// enough that a mostly-idle arena wastes little. +const arenaChunkSize = 64 << 10 + +func (a *byteArena) copy(b []byte) []byte { + if len(b) > cap(a.buf)-len(a.buf) { + a.buf = make([]byte, 0, max(arenaChunkSize, len(b))) + } + n := len(a.buf) + a.buf = append(a.buf, b...) + return a.buf[n : n+len(b) : n+len(b)] +} diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go new file mode 100644 index 000000000..b37523579 --- /dev/null +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go @@ -0,0 +1,41 @@ +package event + +import ( + "bytes" + "fmt" + "testing" + + "github.com/stretchr/testify/require" +) + +// TestByteArenaCopiesAreStable pins the arena's one load-bearing property: +// a returned copy never moves or changes, however much is copied after it — +// including copies that force new chunks and copies larger than a chunk. +func TestByteArenaCopiesAreStable(t *testing.T) { + var a byteArena + src := make([]byte, 300) + var got [][]byte + var want [][]byte + for i := 0; i < 3000; i++ { // ~900KB total: crosses many 64KB chunks + for j := range src { + src[j] = byte(i + j) + } + c := a.copy(src) + got = append(got, c) + want = append(want, bytes.Clone(src)) + } + huge := bytes.Repeat([]byte{0xAB}, 3*arenaChunkSize) + hugeCopy := a.copy(huge) + huge[0] = 0xCD // mutate the source; the copy must not see it + require.Equal(t, byte(0xAB), hugeCopy[0]) + require.Len(t, hugeCopy, 3*arenaChunkSize) + for i := range got { + require.True(t, bytes.Equal(got[i], want[i]), fmt.Sprintf("copy %d changed", i)) + } + // Appending to a returned copy must not scribble into the arena: the + // three-index slice pins capacity to length. + c := a.copy([]byte{1, 2, 3}) + next := a.copy([]byte{9, 9, 9}) + _ = append(c, 7) //nolint:staticcheck // the append must copy, not extend in place + require.Equal(t, []byte{9, 9, 9}, next) +} diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go index 9c0a3efa9..6d5e21ca0 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go @@ -474,12 +474,15 @@ func (c *ColdReader) FetchEvents(ctx context.Context, eventIDs []uint32) ([]Payl positions[i] = int(id) } results := make([]Payload, len(eventIDs)) + var arena byteArena if err := c.events.ReadItems(ctx, positions, func(idx int, data []byte) error { // packfile.ReadItems passes a borrowed data slice valid only for // the duration of fn (see Reader.ReadItems docstring). FetchEvents - // returns the Payloads in a slice that outlives fn, so clone before - // Unmarshal aliases the bytes into ContractEventBytes. - return results[idx].Unmarshal(bytes.Clone(data)) + // returns the Payloads in a slice that outlives fn, so copy before + // Unmarshal aliases the bytes into ContractEventBytes — through an + // arena, since a batch's payloads live and die together and + // per-payload clones only add GC work. + return results[idx].Unmarshal(arena.copy(data)) }); err != nil { // packfile.ReadItems also validates sorted positions as defense in // depth; translate its sentinel to ours so callers can errors.Is diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go index d263667dd..3d490666f 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go @@ -219,14 +219,17 @@ type Match struct { type termPlan [][]int // batchSizes resolves the first and following internal batch sizes -// from the caller's hint. The hint applies only when it is positive -// and below the default. Both results are clamped positive, so a zero -// test seam cannot stall a stream (a zero step never advances). +// from the caller's hint. A positive hint sizes the first batch in both +// directions — smaller for a small page, larger so a page-sized request +// is one storage round trip instead of several — capped at eight default +// batches so a wild hint cannot demand an unbounded fetch. Both results +// are clamped positive, so a zero test seam cannot stall a stream (a +// zero step never advances). func batchSizes(hint int) (int, int) { rest := max(1, matchBatchSize) first := rest - if hint > 0 && hint < rest { - first = hint + if hint > 0 { + first = min(hint, 8*rest) } return first, rest } From 1e6aea8d23254cccbaccfc3ff58783edd076bdde Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Sat, 29 Aug 2026 10:27:59 +0000 Subject: [PATCH 03/41] =?UTF-8?q?events:=20ascending=20match=20pulls=20can?= =?UTF-8?q?didates=20through=20a=20cursor=20tree=20=E2=80=94=20no=20materi?= =?UTF-8?q?alized=20union?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit unionForFilters built the candidate set by materializing: OR the bitmaps of every group, AND every filter's groups, OR across filters, AND a range bitmap for the window — several fresh roaring bitmaps over full multi-million-posting terms per request, plus a roaring.New + AddMany per sparse term inside ConcurrentBitmaps.Get, all to serve one 1000-event page. On the serving allocation profile that construction was ~80% of hot getEvents allocations. Ascending queries now assemble the same union/intersect algebra as a tree of peekable galloping cursors (match_iter.go) over borrowed sources: a dense term walks roaring's IntPeekable in place (its iteration path reads containers through getContainerAtIndex, on the COW-safe list), a sparse term walks the mirror's []uint32 in place via a new no-materialize postings lookup, and the pinned window becomes advance(Start) plus an End clamp at the leaves — which also preserves the phantom-ID clipping the window AND provided. HotStore exposes the postings seam through an unexported optional interface; ColdReader and out-of-package Readers fall back to LookupKeys with their bitmaps wrapped, unchanged. Descending keeps the materialized path: roaring's reverse iterator has no gallop to build combinators on. In-tree A/B (4M-event chunk, 1000-event page): ascending 391µs/1.43MB/ 596 allocs → 160µs/103KB/34 allocs per page (−59% ns, −94% allocs); descending bit-identical. Kit cells (50rps 60s, 8K-block fixtures, K-corpus limit=1000, service layer): sac p50 3.54→3.37ms, p99 7.40→5.37ms, leg allocs −11%; pubnet p50 5.20→2.26ms, p99 11.59→8.77ms, leg allocs −79% (2.65MB→0.56MB/request). Two differential gates are deliberately complementary: the full- Matches one (asc vs desc over randomized corpora/filters/windows/both lookup seams) proves selection parity but postFilter masks index-side over-selection there; the candidateIter-level one against a naive bitmap-algebra reference catches exactly that. A race-detector test streams Matches against live AddTo sparse→dense promotions to pin the borrowed-snapshot contract. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../rpcv2/stores/event/concurrent_bitmaps.go | 74 +++ .../internal/rpcv2/stores/event/hot_store.go | 42 +- .../internal/rpcv2/stores/event/match.go | 260 ++++++-- .../internal/rpcv2/stores/event/match_iter.go | 359 ++++++++++ .../rpcv2/stores/event/match_iter_test.go | 628 ++++++++++++++++++ .../stores/event/matches_differential_test.go | 360 ++++++++++ 6 files changed, 1651 insertions(+), 72 deletions(-) create mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go create mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go create mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go index ad2d2cefd..aab5a930b 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go @@ -135,6 +135,59 @@ func (d *denseState) snapshot() *roaring.Bitmap { return bm } +// postings is an iterable view of ONE term's event IDs, in whichever +// representation the index already holds: the sparse mode's sorted +// []uint32, a dense term's live denseState, or a bitmap that came +// from outside the mirror (the cold tier's, and the ascending path's +// own bulk answers). Exactly one of the three is set; the zero value +// means the term is absent. +// +// It exists so the ascending match path can iterate a sparse term in +// place. Get has to promise a *roaring.Bitmap, so it pays a +// roaring.New + AddMany over the id list on every sparse lookup — +// measurably the bulk of the hot read path's container churn, and +// pure waste when the consumer only walks the ids in order. +// +// Ownership: ids borrows the atomically published termState's slice, +// which no writer ever mutates (a sparse AddTo publishes a NEW +// termState), so it may be held indefinitely; it must be read, never +// written or appended to. dense is the live term, and every bitmap it +// yields comes from denseState.snapshot — the writer's own wbm and +// the raw pub pointer are never handed out — so what bitmap() +// returns obeys Get's read-only contract verbatim, forbidden and safe +// method lists included. +// +// A postings over a dense term is a view, not a frozen copy: two +// materializations of it can straddle an ingest and the later one +// hold more ids. Callers pin a window before the lookup and clip +// every cursor to it at the leaf (see bitmapIter.end), so ids a write +// added after the pin sit above the window and are never yielded. +type postings struct { + ids []uint32 + bm *roaring.Bitmap + dense *denseState +} + +// present reports whether the term is in the index at all. A term +// that is present but holds no ids (possible only for an empty bitmap +// handed to NewConcurrentBitmapsFromBitmaps) is present: it yields an +// exhausted cursor, which intersects and unions to the same result an +// absent term's caller-side skip would produce. +func (p postings) present() bool { return p.bm != nil || p.ids != nil || p.dense != nil } + +// bitmap is the term's ids as a roaring bitmap, or nil when the term +// is sparse or absent — sparse callers walk ids instead. A dense term +// is snapshotted here, the only place outside Get that materializes +// one, which is what keeps the writer's wbm off every read path. +// Repeat calls cost nothing while no write lands: denseState caches +// the snapshot it published. +func (p postings) bitmap() *roaring.Bitmap { + if p.dense != nil { + return p.dense.snapshot() + } + return p.bm +} + // AddTo records each eventID under key. Callers feed events in // event-ID order relative to the chunk, so a duplicate is a retry of an // already-added prefix and is skipped. @@ -172,6 +225,27 @@ func (s *ConcurrentBitmaps) AddTo(key TermKey, eventIDs ...uint32) { p.Store(termStateFromIDs(appendSorted(ids, eventIDs))) } +// lookupPostings is Get without the sparse-mode materialization: it +// hands back the term's live representation rather than converting it +// to a bitmap. Same concurrency story as Get — the RLock covers the +// map lookup only, and the entry load is lock-free. A dense term is +// handed back as its denseState, so the bitmap a caller eventually +// reads is denseState.snapshot's, immutable and current as of the +// call that asks for it. A miss returns the zero postings. +func (s *ConcurrentBitmaps) lookupPostings(key TermKey) postings { + s.rwmu.RLock() + p := s.terms[key] + s.rwmu.RUnlock() + if p == nil { + return postings{} + } + st := p.Load() + if st.dense != nil { + return postings{dense: st.dense} + } + return postings{ids: st.ids} +} + // appendSorted appends the ids in src that are greater than dst's // last element. func appendSorted(dst, src []uint32) []uint32 { diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go index 006ae91a1..23696f806 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go @@ -112,8 +112,12 @@ type HotStore struct { offsets *ConcurrentLedgerOffsets } -// Compile-time guard: *HotStore satisfies Reader. -var _ Reader = (*HotStore)(nil) +// Compile-time guards: *HotStore satisfies Reader, and the optional +// postingReader seam match.go's ascending path probes for. +var ( + _ Reader = (*HotStore)(nil) + _ postingReader = (*HotStore)(nil) +) // NewWithStore wraps an ALREADY-OPEN rocksdb.Store as an events HotStore on the // three events CFs (CFNames()), running the mandatory warmup to rebuild the @@ -451,6 +455,40 @@ func (h *HotStore) IngestLedgerToBatch( return func() { h.applyLedger(startID, termKeys) }, nil } +// lookupPostings is the no-materialize half of LookupKeys, and the +// hot store's implementation of the optional postingReader seam the +// ascending match path probes for (see match.go). It returns each +// term's live mirror representation — sorted ids for a sparse term, +// the roaring bitmap for a dense one — so a query that only walks ids +// in ascending order never pays Get's roaring.New + AddMany per +// sparse term. +// +// Results are positionally aligned with keys; a miss is the zero +// postings, which postings.present() reports as absent. Same +// borrowed-snapshot contract as LookupKeys: read-only, valid +// indefinitely. +// +// The cold reader deliberately does NOT implement this. Its postings +// arrive as freshly-unmarshaled bitmaps out of index.pack, so there +// is no un-materialized representation to expose; match.go's fallback +// wraps its LookupKeys bitmaps in the same cursor type. +func (h *HotStore) lookupPostings(ctx context.Context, keys []TermKey) ([]postings, error) { + if h.chunkStore.IsClosed() { + return nil, stores.ErrStoreClosed + } + if err := ctx.Err(); err != nil { + return nil, err + } + if len(keys) == 0 { + return nil, nil + } + results := make([]postings, len(keys)) + for i, key := range keys { + results[i] = h.mirror.lookupPostings(key) + } + return results, nil +} + // index returns the in-memory term mirror. Test-only write hook: no production // path reads it. Kept unexported until #772 decides whether the v2 read path // hooks into it. diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go index 3d490666f..9d8f28d58 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go @@ -7,9 +7,16 @@ package event // Matches. // // Optimization shape: terms are deduped across filters and issued as -// a single Reader.LookupKeys call at iteration start; payload fetches +// a single batched index lookup at iteration start; payload fetches // then stream in internal batches. On the cold path this is one // MPHF+index.pack round trip per Matches call, not per batch. +// +// The candidate set is then built one of two ways. ASCENDING pulls +// ids through the un-materialized iterator tree in match_iter.go: no +// intermediate bitmap, and no index work past what the consumer +// pulls. DESCENDING materializes a union bitmap through roaring's +// aggregation, because roaring's reverse iterator offers no gallop to +// build ascending-style combinators on. import ( "bytes" @@ -271,11 +278,7 @@ func Matches( if window.isEmpty() { return } - union, matchAll, err := unionForFilters(ctx, r, filters, window) - if err != nil { - yield(Match{}, err) - return - } + plans, uniqueKeys, matchAll := planIndexTerms(filters) // Match-all path: empty filter slice or any filter that asks the // index for no terms. Serves without touching the index: the // window is dense, so it streams Reader.FetchRange directly. @@ -283,10 +286,38 @@ func Matches( streamRange(ctx, r, window, descending, firstBatch, yield) return } - if union.IsEmpty() { + // Direction splits the plan from here. Ascending pulls + // candidates through the un-materialized iterator tree + // (match_iter.go): no intermediate bitmap, and the walk stops + // the moment the consumer does. Descending materializes the + // union the old way — roaring's reverse iterator has no + // AdvanceIfNeeded, so there is no reverse gallop to build the + // combinators on. + if descending { + union, err := unionForFilters(ctx, r, plans, uniqueKeys, window) + if err != nil { + yield(Match{}, err) + return + } + if union.IsEmpty() { + return + } + streamUnion(ctx, r, filters, union, firstBatch, yield) + return + } + sources, err := lookupPostings(ctx, r, uniqueKeys) + if err != nil { + yield(Match{}, err) return } - streamUnion(ctx, r, filters, union, descending, firstBatch, yield) + candidates := candidateIter(plans, sources, window) + // The structural twin of the descending path's + // union.IsEmpty(): one peek settles whether anything matches, + // without reading a single id further than that. + if _, ok := candidates.peek(); !ok { + return + } + streamCandidates(ctx, r, filters, candidates, firstBatch, yield) } } @@ -315,36 +346,28 @@ func validateMatchCall(ctx context.Context, r Reader, filters []Filter, window I return nil } -// unionForFilters runs the index side once per Matches call (the -// numbered steps below). matchAll reports that some filter (or the -// empty slice) constrains nothing, detected before any index I/O; the -// caller then streams the window directly. Otherwise the result is -// empty when no candidate falls in the window, and is never a -// borrowed mirror snapshot (the window AND allocates on the borrowing -// path), so downstream iteration is safe. -func unionForFilters( - ctx context.Context, r Reader, filters []Filter, window IDRange, -) (*roaring.Bitmap, bool, error) { - // ───── 1. Dedupe terms across filters ───── - // - // filterPlans[i] holds the slots filter i needs out of the batched - // lookup: the bitmaps within a group are OR-ed, the groups AND-ed. - // - // A filter that asks the index for no terms constrains nothing, and - // so does an empty filter slice: both take the match-all path. - // Reading the condition off the term groups themselves is what keeps - // an unconstrained filter from intersecting nothing and coming back - // empty instead. +// planIndexTerms is step 1 of the index side, shared by both +// directions and run before any index I/O: dedupe the terms the +// filters name across the whole query, and resolve each filter's +// groups to slots in the single batched lookup that follows. plans[i] +// holds the slots filter i needs; the terms within a group are OR-ed +// and the groups AND-ed. +// +// matchAll reports that some filter (or the empty slice) constrains +// nothing, so the caller streams the window directly. Reading that +// condition off the term groups themselves is what keeps an +// unconstrained filter from intersecting nothing and coming back +// empty instead. +func planIndexTerms(filters []Filter) ([]termPlan, []TermKey, bool) { if len(filters) == 0 { - return nil, true, nil + return nil, nil, true } - filterPlans := make([]termPlan, len(filters)) var uniqueKeys []TermKey - + plans := make([]termPlan, len(filters)) for i := range filters { groups := filters[i].termGroups() if len(groups) == 0 { - return nil, true, nil + return nil, nil, true } plan := make(termPlan, len(groups)) for g, keys := range groups { @@ -354,13 +377,70 @@ func unionForFilters( } plan[g] = slots } - filterPlans[i] = plan + plans[i] = plan } + return plans, uniqueKeys, false +} +// postingReader is the optional, hot-tier half of the index read +// surface: a Reader that can expose a term's postings WITHOUT +// materializing a bitmap for it. +// +// Deliberately not folded into Reader. Cold postings genuinely are +// bitmaps — ColdReader unmarshals a fresh one per term out of +// index.pack — so it has nothing un-materialized to hand back and +// could only implement the method by re-wrapping what LookupKeys +// already returns. Hoisting it into Reader would also impose it on +// every out-of-package implementation (the query package's test +// fakes) for no gain. The assertion in lookupPostings picks the fast +// path where it exists and wraps bitmaps everywhere else. +type postingReader interface { + lookupPostings(ctx context.Context, keys []TermKey) ([]postings, error) +} + +// lookupPostings resolves keys to per-term postings, positionally +// aligned with keys, in one batched call. It takes the +// no-materialize path when r offers one and otherwise falls back to +// LookupKeys; a nil bitmap stays the zero postings, i.e. absent. +func lookupPostings(ctx context.Context, r Reader, keys []TermKey) ([]postings, error) { + if pr, ok := r.(postingReader); ok { + sources, err := pr.lookupPostings(ctx, keys) + if err != nil { + return nil, fmt.Errorf("events: query lookup: %w", err) + } + return sources, nil + } + bitmaps, err := r.LookupKeys(ctx, keys) + if err != nil { + return nil, fmt.Errorf("events: query lookup: %w", err) + } + sources := make([]postings, len(bitmaps)) + for i, bm := range bitmaps { + if bm != nil { + sources[i] = postings{bm: bm} + } + } + return sources, nil +} + +// unionForFilters materializes the DESCENDING path's candidate set: +// steps 2-5 below, over the plan planIndexTerms already resolved. The +// result is empty when no candidate falls in the window, and is never +// a borrowed mirror snapshot (the window AND allocates on the +// borrowing path), so downstream iteration is safe. +// +// The ascending path does not come through here at all — see +// candidateIter, which answers the same question with no intermediate +// bitmap. This shape survives because roaring offers no reverse +// gallop to build ascending-style combinators on. +func unionForFilters( + ctx context.Context, r Reader, filterPlans []termPlan, uniqueKeys []TermKey, + window IDRange, +) (*roaring.Bitmap, error) { // ───── 2. Single batched lookup for all unique terms ───── bitmaps, err := r.LookupKeys(ctx, uniqueKeys) if err != nil { - return nil, false, fmt.Errorf("events: query lookup: %w", err) + return nil, fmt.Errorf("events: query lookup: %w", err) } // ───── 3. Per-filter intersect ───── @@ -407,7 +487,7 @@ func unionForFilters( } if len(perFilter) == 0 { - return roaring.New(), false, nil + return roaring.New(), nil } // ───── 4. Union across filters ───── @@ -442,32 +522,20 @@ func unionForFilters( } else { union.And(rangeBM) // FastOr output is owned; in-place is fine } - return union, false, nil + return union, nil } -// streamUnion walks the union bitmap in internal batches: collect -// candidate ordinals up to the batch size, fetch, post-filter, yield -// the survivors. Drops advance the walk with no yield. The first -// batch is sized to firstBatch (see Matches); later batches use the -// default. -// -// FetchEvents requires ascending ids, so a descending batch is -// collected highest-first and flipped before the fetch, then the -// fetched matches are flipped back. Stepping one id at a time is fine -// here: the fetch I/O dominates a 512-step loop. +// streamUnion walks the DESCENDING path's materialized union bitmap +// in internal batches: collect candidate ordinals up to the batch +// size, fetch, post-filter, yield the survivors. Drops advance the +// walk with no yield. The first batch is sized to firstBatch (see +// Matches); later batches use the default. Stepping one id at a time +// is fine here: the fetch I/O dominates a 512-step loop. func streamUnion( ctx context.Context, r Reader, filters []Filter, union *roaring.Bitmap, - descending bool, firstBatch int, yield func(Match, error) bool, + firstBatch int, yield func(Match, error) bool, ) { - var it interface { - HasNext() bool - Next() uint32 - } - if descending { - it = union.ReverseIterator() - } else { - it = union.Iterator() - } + it := union.ReverseIterator() batch, rest := batchSizes(firstBatch) ids := make([]uint32, 0, batch) for { @@ -483,29 +551,81 @@ func streamUnion( if len(ids) == 0 { return } - if descending { - slices.Reverse(ids) + if !emitBatch(ctx, r, filters, ids, true, yield) { + return } - payloads, err := r.FetchEvents(ctx, ids) - if err != nil { + } +} + +// streamCandidates is streamUnion's ASCENDING twin over the +// un-materialized iterator tree. Same batch loop, same post-filter, +// same yields; the difference is that candidates are pulled out of +// the tree one at a time as the batch fills, so a consumer that stops +// after one page never touched the postings past it. +func streamCandidates( + ctx context.Context, r Reader, filters []Filter, candidates idIter, + firstBatch int, yield func(Match, error) bool, +) { + batch, rest := batchSizes(firstBatch) + ids := make([]uint32, 0, batch) + for { + if err := ctx.Err(); err != nil { yield(Match{}, err) return } - // Drop bitmap-side false positives (see postFilter for the rationale). - matched, err := postFilter(payloads, ids, filters) - if err != nil { - yield(Match{}, err) + ids = ids[:0] + for len(ids) < batch { + id, ok := candidates.peek() + if !ok { + break + } + ids = append(ids, id) + candidates.next() + } + batch = rest + if len(ids) == 0 { return } - if descending { - slices.Reverse(matched) + if !emitBatch(ctx, r, filters, ids, false, yield) { + return } - for i := range matched { - if !yield(matched[i], nil) { - return - } + } +} + +// emitBatch fetches one batch of candidate ordinals, drops the +// bitmap-side false positives and yields the survivors, reporting +// whether the stream should continue. +// +// FetchEvents requires ascending ids, so a descending batch — which +// arrives highest-first — is flipped in place before the fetch and +// the surviving matches are flipped back before they are yielded. +func emitBatch( + ctx context.Context, r Reader, filters []Filter, ids []uint32, + descending bool, yield func(Match, error) bool, +) bool { + if descending { + slices.Reverse(ids) + } + payloads, err := r.FetchEvents(ctx, ids) + if err != nil { + yield(Match{}, err) + return false + } + // Drop bitmap-side false positives (see postFilter for the rationale). + matched, err := postFilter(payloads, ids, filters) + if err != nil { + yield(Match{}, err) + return false + } + if descending { + slices.Reverse(matched) + } + for i := range matched { + if !yield(matched[i], nil) { + return false } } + return true } // ValidateFilters rejects filters that would silently never match diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go new file mode 100644 index 000000000..e9d94fda9 --- /dev/null +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go @@ -0,0 +1,359 @@ +package event + +// match_iter.go is the ascending match path's un-materialized query +// plan: a tree of peekable ascending cursors that pulls candidate +// event IDs straight out of the index's postings, in order, and stops +// as soon as the consumer stops. +// +// Why it exists: unionForFilters' materializing shape — OR every +// group, AND every filter's groups, OR across filters, AND a window +// bitmap — builds several fresh roaring bitmaps over the FULL +// multi-million-posting terms just to serve one 1000-event page, and +// materializes a bitmap per sparse term on top (see +// ConcurrentBitmaps.Get). On the serving allocation profile that +// construction was ~80% of hot getEvents allocations. The tree below +// answers the same question with no intermediate bitmap at all: the +// only per-request allocations are the handful of cursor structs. +// +// Direction: ascending only. roaring's reverse iterator has no +// AdvanceIfNeeded, so there is no reverse gallop to build the +// combinators on; descending keeps the materialized path in match.go +// unchanged. + +import ( + "slices" + + "github.com/RoaringBitmap/roaring/v2" +) + +// idIter is a peekable cursor over a strictly ascending run of +// chunk-relative event IDs. +// +// Structural early exit: nothing in the tree reads further than the +// consumer pulls. peek resolves exactly one id — the combinators +// cache it and do no lookahead — so a batch loop that stops after N +// ids has touched only the postings those N ids needed. +type idIter interface { + // peek returns the id under the cursor without consuming it. ok + // is false once the cursor is exhausted, which is permanent. + peek() (id uint32, ok bool) + // next steps past the id peek returns. A no-op on an exhausted + // cursor. + next() + // advance moves the cursor to the first id >= floor, exhausting + // it when there is none. Never moves backwards, so a floor at or + // below the current id is a no-op. + advance(floor uint32) +} + +// emptyIter is the permanently exhausted cursor: an absent term, or a +// query where no filter survived group resolution. Zero-size, so an +// idIter holding it never allocates. +type emptyIter struct{} + +func (emptyIter) peek() (uint32, bool) { return 0, false } +func (emptyIter) next() {} +func (emptyIter) advance(uint32) {} + +// sliceIter walks a sorted []uint32 postings list in place — the hot +// mirror's sparse-term representation, read directly instead of being +// inflated into a roaring bitmap per lookup. +// +// The window is applied by reslicing at construction (see +// postings.iter), so the cursor itself carries no bounds check and +// the underlying array is never copied. The mirror publishes a fresh +// slice on every AddTo and never mutates a published one, so the +// borrowed backing array is immutable for the cursor's lifetime. +type sliceIter struct { + ids []uint32 + i int +} + +func (s *sliceIter) peek() (uint32, bool) { + if s.i >= len(s.ids) { + return 0, false + } + return s.ids[s.i], true +} + +func (s *sliceIter) next() { + if s.i < len(s.ids) { + s.i++ + } +} + +func (s *sliceIter) advance(floor uint32) { + if s.i >= len(s.ids) || s.ids[s.i] >= floor { + return + } + // Binary search over the unread tail: one gallop can skip most of + // a long postings list. + j, _ := slices.BinarySearch(s.ids[s.i:], floor) + s.i += j +} + +// bitmapIter walks a roaring bitmap's set bits through the library's +// IntPeekable cursor, whose PeekNext/AdvanceIfNeeded are the +// roaring-side gallop. Both read through getContainerAtIndex, not the +// *Writable* accessor, so they are on the COW-safe read list in +// ConcurrentBitmaps.Get's contract and are safe on a borrowed mirror +// snapshot. +// +// end clamps the caller's pinned window at the LEAF rather than at +// the root. Every source stops at the window's upper bound, so no +// combinator above it ever aligns on ids outside the window — which +// is what makes killing the materialized range-AND free rather than a +// cost shifted upward. It is also what clips phantom IDs from a +// concurrent hot-store ingest: the mirror publishes index entries +// before offsets, so a lookup can briefly surface IDs past +// EventCount, and the stream must stay inside the snapshot the caller +// pinned at request entry. +type bitmapIter struct { + it roaring.IntPeekable + end uint32 +} + +func (b *bitmapIter) peek() (uint32, bool) { + // PeekNext is only defined while HasNext holds. + if !b.it.HasNext() { + return 0, false + } + if v := b.it.PeekNext(); v < b.end { + return v, true + } + return 0, false +} + +func (b *bitmapIter) next() { + if _, ok := b.peek(); ok { + b.it.Next() + } +} + +func (b *bitmapIter) advance(floor uint32) { + b.it.AdvanceIfNeeded(floor) +} + +// iter returns an ascending cursor over the postings clipped to +// window, without materializing anything: a sparse term is resliced +// in place, a dense one is walked through roaring's own iterator. +// Callers must check present() first — absent postings are the +// group-missed signal, not an empty cursor. +func (p postings) iter(window IDRange) idIter { + if bm := p.bitmap(); bm != nil { + it := bm.Iterator() + it.AdvanceIfNeeded(window.Start) + return &bitmapIter{it: it, end: window.End} + } + ids := p.ids + lo, _ := slices.BinarySearch(ids, window.Start) + ids = ids[lo:] + // BinarySearch returns the first index at or above the target, so + // this drops exactly the ids at or past the window's exclusive End. + hi, _ := slices.BinarySearch(ids, window.End) + return &sliceIter{ids: ids[:hi]} +} + +// unionIter is the OR of its children: the ascending merge of their +// ids with equal ids across children collapsed to one. +// +// Children are scanned linearly for the minimum rather than kept in a +// heap. K is the number of terms in one group (at most the +// topic-count bucket family) or the number of filters in a query — +// single digits in practice — and advance has to move all K children +// regardless, which would force a full re-heapify on every gallop. +type unionIter struct { + children []idIter + cur uint32 + ok bool + primed bool +} + +func (u *unionIter) peek() (uint32, bool) { + if !u.primed { + u.cur, u.ok = 0, false + for _, c := range u.children { + if v, ok := c.peek(); ok && (!u.ok || v < u.cur) { + u.cur, u.ok = v, true + } + } + u.primed = true + } + return u.cur, u.ok +} + +func (u *unionIter) next() { + v, ok := u.peek() + if !ok { + return + } + // Dedup: EVERY child sitting on the winning id steps past it, so + // an id several children hold is yielded once. Stepping only the + // winner would re-emit it from each of the others in turn — and + // FetchEvents rejects a duplicate id outright. + for _, c := range u.children { + if cv, cok := c.peek(); cok && cv == v { + c.next() + } + } + u.primed = false +} + +func (u *unionIter) advance(floor uint32) { + if v, ok := u.peek(); !ok || v >= floor { + return + } + for _, c := range u.children { + c.advance(floor) + } + u.primed = false +} + +// intersectIter is the AND of its children, by galloping alignment: +// every child is advanced to the running maximum of the peeks until +// they all agree on one id. A child that exhausts ends the +// intersection — permanently, since cursors only move forward. +type intersectIter struct { + children []idIter + cur uint32 + ok bool + primed bool +} + +func (n *intersectIter) peek() (uint32, bool) { + if !n.primed { + n.cur, n.ok = n.align() + n.primed = true + } + return n.cur, n.ok +} + +// align raises every child to the smallest id all of them hold. +func (n *intersectIter) align() (uint32, bool) { + cand, ok := n.children[0].peek() + if !ok { + return 0, false + } + for { + raised := false + for _, c := range n.children { + c.advance(cand) + v, cok := c.peek() + if !cok { + return 0, false + } + if v > cand { + // This child overshot the floor. Raise it and repeat + // the pass so the children already visited are pulled + // up to the new floor too. cand strictly increases + // per pass, so the loop terminates. + cand, raised = v, true + } + } + if !raised { + return cand, true + } + } +} + +func (n *intersectIter) next() { + v, ok := n.peek() + if !ok { + return + } + // Post-align every child sits on v; step them all past it. + for _, c := range n.children { + if cv, cok := c.peek(); cok && cv == v { + c.next() + } + } + n.primed = false +} + +func (n *intersectIter) advance(floor uint32) { + if v, ok := n.peek(); !ok || v >= floor { + return + } + for _, c := range n.children { + c.advance(floor) + } + n.primed = false +} + +// unionOf and intersectOf collapse a single input to the input +// itself, rather than wrapping it in a combinator that would re-scan +// a one-element slice on every step. This is the iterator-side twin +// of the singleton guards the materialized path needs for a harder +// reason: roaring's FastAnd/FastOr have historically Cloned a +// single-input slice, so unionForFilters must never hand them one. +func unionOf(children []idIter) idIter { + switch len(children) { + case 0: + return emptyIter{} + case 1: + return children[0] + default: + return &unionIter{children: children} + } +} + +func intersectOf(children []idIter) idIter { + switch len(children) { + case 0: + // Unreachable: a filter that names no term group takes the + // match-all path before any index I/O. + return emptyIter{} + case 1: + return children[0] + default: + return &intersectIter{children: children} + } +} + +// candidateIter assembles the ascending candidate cursor for one +// Matches call out of the batched lookup's postings: terms within a +// group OR, a filter's groups AND, filters OR — the same three steps +// unionForFilters performs on bitmaps, and clamped to the same +// window, but with nothing materialized in between. +// +// A filter with an entirely absent group contributes nothing, exactly +// as in the materialized path; if that leaves no filter at all the +// result is the exhausted cursor, the un-materialized form of +// "union.IsEmpty()". +func candidateIter(plans []termPlan, sources []postings, window IDRange) idIter { + perFilter := make([]idIter, 0, len(plans)) + for _, plan := range plans { + groups := make([]idIter, 0, len(plan)) + missed := false + for _, slots := range plan { + g := groupIter(sources, slots, window) + if g == nil { + missed = true + break + } + groups = append(groups, g) + } + if missed { + continue + } + perFilter = append(perFilter, intersectOf(groups)) + } + return unionOf(perFilter) +} + +// groupIter ORs the postings at slots into one cursor, returning nil +// when every one of them is absent from the index — the mirror of +// unionSlots' nil return, and the signal that the owning filter can +// match nothing. +func groupIter(sources []postings, slots []int, window IDRange) idIter { + present := make([]idIter, 0, len(slots)) + for _, slot := range slots { + if p := sources[slot]; p.present() { + present = append(present, p.iter(window)) + } + } + if len(present) == 0 { + return nil + } + return unionOf(present) +} diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go new file mode 100644 index 000000000..93eb06dcf --- /dev/null +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go @@ -0,0 +1,628 @@ +package event + +// match_iter_test.go covers the ascending path's un-materialized +// query plan: the cursor sources, the union/intersect combinators, +// and the tree candidateIter assembles from them. The end-to-end +// semantics stay pinned black-box by match_test.go; what is pinned +// here is the machinery underneath it, plus a randomized differential +// check that the tree and the descending path's materialized +// bitmap algebra answer identically. + +import ( + "context" + "errors" + "iter" + "math/rand" + "slices" + "sync" + "testing" + + "github.com/RoaringBitmap/roaring/v2" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + protocol "github.com/stellar/go-stellar-sdk/protocols/rpc" + "github.com/stellar/go-stellar-sdk/xdr" + + "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/chunk" +) + +// wholeWindow is the no-op window: every cursor test that is not about +// clamping uses it. +var wholeWindow = IDRange{Start: 0, End: ^uint32(0)} + +// drain pulls a cursor dry. Always returns a non-nil slice so an empty +// result compares equal to a materialized bitmap's ToArray(). +func drain(it idIter) []uint32 { + out := []uint32{} + for { + v, ok := it.peek() + if !ok { + return out + } + out = append(out, v) + it.next() + } +} + +// sparseSource / denseSource build the two representations the index +// actually holds, so a test can pin that both cursor sources behave +// identically. +func sparseSource(ids ...uint32) postings { return postings{ids: ids} } + +func denseSource(ids ...uint32) postings { + bm := roaring.New() + bm.AddMany(ids) + return postings{bm: bm} +} + +// sourceKinds runs fn against both representations of the same id set, +// so every cursor-level assertion is made twice. +func sourceKinds(ids ...uint32) map[string]func() postings { + return map[string]func() postings{ + "sparse": func() postings { return sparseSource(ids...) }, + "dense": func() postings { return denseSource(ids...) }, + } +} + +func TestPostingsPresent(t *testing.T) { + assert.False(t, postings{}.present(), "the zero postings is the absent term") + assert.True(t, sparseSource(1).present()) + assert.True(t, denseSource(1).present()) + assert.True(t, postings{bm: roaring.New()}.present(), + "a present-but-empty bitmap is present; it just yields nothing") + assert.Empty(t, drain(postings{bm: roaring.New()}.iter(wholeWindow))) +} + +// TestIDIterSources pins peek/next/advance on both leaf sources: peek +// does not consume, advance lands on the first id at or above min, +// advance never moves backwards, and both are idempotent at +// exhaustion. +func TestIDIterSources(t *testing.T) { + for name, mk := range sourceKinds(3, 7, 8, 20, 100) { + t.Run(name, func(t *testing.T) { + it := mk().iter(wholeWindow) + assert.Equal(t, []uint32{3, 7, 8, 20, 100}, drain(mk().iter(wholeWindow))) + + v, ok := it.peek() + require.True(t, ok) + assert.Equal(t, uint32(3), v) + v2, _ := it.peek() + assert.Equal(t, v, v2, "peek must not consume") + + it.advance(7) + v, ok = it.peek() + require.True(t, ok) + assert.Equal(t, uint32(7), v, "advance lands on an id equal to min") + + it.advance(4) + v, _ = it.peek() + assert.Equal(t, uint32(7), v, "advance must never move backwards") + + it.advance(9) + v, ok = it.peek() + require.True(t, ok) + assert.Equal(t, uint32(20), v, "advance skips the gap to the next id above min") + + it.advance(1000) + _, ok = it.peek() + assert.False(t, ok, "advance past the last id exhausts the cursor") + it.next() // must not panic + it.advance(0) + _, ok = it.peek() + assert.False(t, ok, "exhaustion is permanent") + }) + } +} + +// TestIDIterSourcesWindowClampsBothEnds pins the window applied at the +// leaf: ids below Start are skipped at construction and ids at or +// above the exclusive End exhaust the cursor. +func TestIDIterSourcesWindowClampsBothEnds(t *testing.T) { + for name, mk := range sourceKinds(1, 4, 5, 6, 9, 10, 11) { + t.Run(name, func(t *testing.T) { + assert.Equal(t, []uint32{4, 5, 6, 9}, + drain(mk().iter(IDRange{Start: 4, End: 10})), + "Start is inclusive, End exclusive") + assert.Equal(t, []uint32{1, 4, 5, 6, 9, 10, 11}, + drain(mk().iter(IDRange{Start: 0, End: 12}))) + assert.Empty(t, drain(mk().iter(IDRange{Start: 12, End: 20})), + "a window entirely above the postings yields nothing") + assert.Empty(t, drain(mk().iter(IDRange{Start: 0, End: 1})), + "a window entirely below the postings yields nothing") + assert.Equal(t, []uint32{11}, + drain(mk().iter(IDRange{Start: 11, End: 12})), + "a one-id window ending just past the last id keeps it") + + // advance must not resurrect ids the End clamp removed. + it := mk().iter(IDRange{Start: 0, End: 7}) + it.advance(9) + _, ok := it.peek() + assert.False(t, ok, "advancing past End must not escape the window") + }) + } +} + +func TestEmptyIter(t *testing.T) { + it := idIter(emptyIter{}) + _, ok := it.peek() + assert.False(t, ok) + it.next() + it.advance(5) + _, ok = it.peek() + assert.False(t, ok) + assert.Empty(t, drain(it)) +} + +// TestUnionIterDedups is the combinator's load-bearing property: an id +// several children hold is yielded once. Emitting it per child would +// hand FetchEvents a duplicate, which it rejects outright. +func TestUnionIterDedups(t *testing.T) { + u := &unionIter{children: []idIter{ + sparseSource(1, 3, 5, 7).iter(wholeWindow), + denseSource(3, 4, 5).iter(wholeWindow), + sparseSource(5).iter(wholeWindow), + }} + assert.Equal(t, []uint32{1, 3, 4, 5, 7}, drain(u)) +} + +func TestUnionIterAdvanceAndEdges(t *testing.T) { + mk := func() idIter { + return &unionIter{children: []idIter{ + sparseSource(2, 6, 10).iter(wholeWindow), + denseSource(4, 6, 12).iter(wholeWindow), + emptyIter{}, + }} + } + assert.Equal(t, []uint32{2, 4, 6, 10, 12}, drain(mk()), + "an exhausted child contributes nothing and does not stop the union") + + u := mk() + u.advance(5) + v, ok := u.peek() + require.True(t, ok) + assert.Equal(t, uint32(6), v) + u.advance(3) + v, _ = u.peek() + assert.Equal(t, uint32(6), v, "advance backwards is a no-op") + assert.Equal(t, []uint32{6, 10, 12}, drain(u)) + + u = mk() + u.advance(100) + _, ok = u.peek() + assert.False(t, ok) + + allEmpty := &unionIter{children: []idIter{emptyIter{}, emptyIter{}}} + assert.Empty(t, drain(allEmpty)) +} + +// TestIntersectIterGallops covers the AND: a plain overlap, a +// three-way overlap that forces several alignment passes, disjoint +// children, and an empty child short-circuiting the whole thing. +func TestIntersectIterGallops(t *testing.T) { + t.Run("overlap", func(t *testing.T) { + n := &intersectIter{children: []idIter{ + sparseSource(1, 2, 3, 4, 5, 6).iter(wholeWindow), + denseSource(2, 4, 6, 8).iter(wholeWindow), + }} + assert.Equal(t, []uint32{2, 4, 6}, drain(n)) + }) + + t.Run("three way with long gallops", func(t *testing.T) { + // Each child holds a long run the others skip, so alignment + // has to gallop repeatedly and in both orders. + a := make([]uint32, 0, 400) + b := make([]uint32, 0, 400) + c := make([]uint32, 0, 400) + for i := range uint32(400) { + a = append(a, i*2) // even + b = append(b, i*3) // multiples of 3 + c = append(c, i*10) // multiples of 10 + } + n := &intersectIter{children: []idIter{ + denseSource(a...).iter(wholeWindow), + denseSource(b...).iter(wholeWindow), + sparseSource(c...).iter(wholeWindow), + }} + want := []uint32{} + for i := uint32(0); i <= 780; i += 30 { // lcm(2,3,10) = 30, capped by c's max + want = append(want, i) + } + assert.Equal(t, want, drain(n)) + }) + + t.Run("disjoint", func(t *testing.T) { + n := &intersectIter{children: []idIter{ + sparseSource(1, 3, 5).iter(wholeWindow), + sparseSource(2, 4, 6).iter(wholeWindow), + }} + assert.Empty(t, drain(n)) + }) + + t.Run("empty child", func(t *testing.T) { + n := &intersectIter{children: []idIter{ + sparseSource(1, 2, 3).iter(wholeWindow), + emptyIter{}, + sparseSource(2).iter(wholeWindow), + }} + assert.Empty(t, drain(n)) + _, ok := n.peek() + assert.False(t, ok, "an exhausted child ends the intersection permanently") + }) +} + +func TestIntersectIterAdvance(t *testing.T) { + n := &intersectIter{children: []idIter{ + denseSource(1, 2, 3, 4, 5, 6, 7, 8).iter(wholeWindow), + sparseSource(2, 4, 6, 8).iter(wholeWindow), + }} + n.advance(5) + v, ok := n.peek() + require.True(t, ok) + assert.Equal(t, uint32(6), v) + n.advance(1) + v, _ = n.peek() + assert.Equal(t, uint32(6), v, "advance backwards is a no-op") + assert.Equal(t, []uint32{6, 8}, drain(n)) +} + +// TestSingleChildCollapse pins that a one-input union or intersect is +// the input itself, not a wrapper. The materialized path needs the +// same guard for a harder reason (roaring's FastAnd/FastOr Clone a +// singleton input); here it just keeps a one-constraint filter from +// re-scanning a one-element slice per step. +func TestSingleChildCollapse(t *testing.T) { + leaf := sparseSource(1, 2).iter(wholeWindow) + assert.Same(t, leaf, unionOf([]idIter{leaf})) + assert.Same(t, leaf, intersectOf([]idIter{leaf})) + assert.Equal(t, emptyIter{}, unionOf(nil)) + assert.Equal(t, emptyIter{}, intersectOf(nil)) + assert.IsType(t, &unionIter{}, unionOf([]idIter{leaf, leaf})) + assert.IsType(t, &intersectIter{}, intersectOf([]idIter{leaf, leaf})) +} + +// TestGroupIterAbsentGroup pins the absent-group signal: a group whose +// every term is missing from the index returns nil, which drops the +// owning filter from the union entirely. +func TestGroupIterAbsentGroup(t *testing.T) { + sources := []postings{ + sparseSource(1, 2), + {}, // absent + {}, // absent + denseSource(2, 3), + } + assert.Nil(t, groupIter(sources, []int{1}, wholeWindow)) + assert.Nil(t, groupIter(sources, []int{1, 2}, wholeWindow), + "a group is absent only when every one of its terms is") + assert.Equal(t, []uint32{1, 2}, drain(groupIter(sources, []int{0, 1}, wholeWindow)), + "a partly-present group ORs only the present terms") + assert.Equal(t, []uint32{1, 2, 3}, drain(groupIter(sources, []int{0, 3}, wholeWindow))) +} + +func TestCandidateIterDropsFilterWithAbsentGroup(t *testing.T) { + sources := []postings{ + sparseSource(1, 2, 3), + {}, // absent + sparseSource(9), + } + // Filter 0 needs slot 1, which is absent → contributes nothing. + // Filter 1 is slot 2 alone → survives. + plans := []termPlan{{{0}, {1}}, {{2}}} + assert.Equal(t, []uint32{9}, drain(candidateIter(plans, sources, wholeWindow))) + + // Every filter dropped → the exhausted cursor, the + // un-materialized form of union.IsEmpty(). + allMissed := []termPlan{{{0}, {1}}} + it := candidateIter(allMissed, sources, wholeWindow) + _, ok := it.peek() + assert.False(t, ok) +} + +// referenceCandidates is an independent, deliberately naive +// materialized implementation of the same query algebra +// candidateIter answers: OR within a group, AND across a filter's +// groups, OR across filters, AND the window. +func referenceCandidates(plans []termPlan, sources []postings, window IDRange) []uint32 { + materialize := func(p postings) *roaring.Bitmap { + if bm := p.bitmap(); bm != nil { + return bm + } + bm := roaring.New() + bm.AddMany(p.ids) + return bm + } + union := roaring.New() + for _, plan := range plans { + var acc *roaring.Bitmap + missed := false + for _, slots := range plan { + group := roaring.New() + present := false + for _, s := range slots { + if sources[s].present() { + present = true + group.Or(materialize(sources[s])) + } + } + if !present { + missed = true + break + } + if acc == nil { + acc = group + } else { + acc.And(group) + } + } + if missed { + continue + } + union.Or(acc) + } + windowBM := roaring.New() + windowBM.AddRange(uint64(window.Start), uint64(window.End)) + union.And(windowBM) + return union.ToArray() +} + +// TestCandidateIterMatchesMaterializedAlgebra is the differential +// gate: over randomized plans, source shapes (absent / sparse / dense +// / present-but-empty) and windows, the un-materialized tree must +// yield exactly what the bitmap algebra does. +func TestCandidateIterMatchesMaterializedAlgebra(t *testing.T) { + rng := rand.New(rand.NewSource(20260829)) + const idSpace = 400 + for trial := range 500 { + nSources := 1 + rng.Intn(6) + sources := make([]postings, nSources) + for i := range sources { + switch rng.Intn(5) { + case 0: + // absent + case 1: + sources[i] = postings{bm: roaring.New()} // present, empty + default: + n := rng.Intn(40) + seen := make(map[uint32]struct{}, n) + for range n { + seen[uint32(rng.Intn(idSpace))] = struct{}{} + } + ids := make([]uint32, 0, len(seen)) + for id := range seen { + ids = append(ids, id) + } + slices.Sort(ids) + if rng.Intn(2) == 0 { + sources[i] = postings{ids: ids} + } else { + sources[i] = denseSource(ids...) + } + } + } + plans := make([]termPlan, 1+rng.Intn(3)) + for f := range plans { + plan := make(termPlan, 1+rng.Intn(3)) + for g := range plan { + slots := make([]int, 1+rng.Intn(3)) + for s := range slots { + slots[s] = rng.Intn(nSources) + } + plan[g] = slots + } + plans[f] = plan + } + start := uint32(rng.Intn(idSpace)) + end := start + uint32(rng.Intn(idSpace)) + window := IDRange{Start: start, End: end} + + want := referenceCandidates(plans, sources, window) + got := drain(candidateIter(plans, sources, window)) + require.Equal(t, want, got, + "trial %d: window %v plans %v", trial, window, plans) + } +} + +// ─── the A/B benchmark ────────────────────────────────────────────── + +// stubIndex is a Reader over an in-memory mirror and one shared event +// payload: enough for Matches to run end to end without RocksDB, so a +// benchmark measures the match layer rather than the storage tier. +// FetchEvents reuses its result buffer, keeping the per-batch fetch +// cost identical in both directions. +type stubIndex struct { + mirror *ConcurrentBitmaps + count uint32 + raw []byte + buf []Payload +} + +func (s *stubIndex) ChunkID() chunk.ID { return chunk.ID(0) } +func (s *stubIndex) EventCount() (uint32, error) { return s.count, nil } + +func (s *stubIndex) Offsets() (*LedgerOffsets, error) { + return nil, errors.New("stubIndex: Offsets is not part of the match path") +} + +func (s *stubIndex) LookupKeys(_ context.Context, keys []TermKey) ([]*roaring.Bitmap, error) { + out := make([]*roaring.Bitmap, len(keys)) + for i, k := range keys { + bm, err := s.mirror.Get(k) + if err != nil { + return nil, err + } + out[i] = bm + } + return out, nil +} + +func (s *stubIndex) FetchEvents(_ context.Context, ids []uint32) ([]Payload, error) { + if err := validateSortedEventIDs(ids); err != nil { + return nil, err + } + s.buf = s.buf[:0] + for range ids { + s.buf = append(s.buf, Payload{ContractEventBytes: s.raw}) + } + return s.buf, nil +} + +func (s *stubIndex) FetchRange(_ context.Context, start, count uint32) iter.Seq2[Payload, error] { + return func(yield func(Payload, error) bool) { + if err := validateFetchRange(start, count, s.count, s.ChunkID()); err != nil { + yield(Payload{}, err) + return + } + for range count { + if !yield(Payload{ContractEventBytes: s.raw}, nil) { + return + } + } + } +} + +func (s *stubIndex) All(ctx context.Context) iter.Seq2[Payload, error] { + return s.FetchRange(ctx, 0, s.count) +} + +// hotLikeIndex carries the same optional no-materialize seam HotStore +// does, so the ascending benchmark exercises the production fast path +// (sparse terms read in place) rather than the bitmap fallback. +type hotLikeIndex struct{ *stubIndex } + +func (h *hotLikeIndex) lookupPostings(_ context.Context, keys []TermKey) ([]postings, error) { + out := make([]postings, len(keys)) + for i, k := range keys { + out[i] = h.mirror.lookupPostings(k) + } + return out, nil +} + +var ( + _ Reader = (*stubIndex)(nil) + _ Reader = (*hotLikeIndex)(nil) + _ postingReader = (*hotLikeIndex)(nil) +) + +const ( + // ~4M events: half a production chunk (~9M), enough that the + // materialized path's intermediates are the multi-container + // bitmaps the real one builds. + benchEvents = 1 << 22 + benchPage = 1000 // getEvents' max page size +) + +type benchIndex struct { + reader *hotLikeIndex + filters []Filter + window IDRange +} + +// newBenchIndex builds the synthetic chunk once for both directions: +// three dense terms (one near-total, like the event type; two +// selective) plus a long-tail sparse term below the mirror's +// promotion threshold, so the sparse read path is on the plan. +var newBenchIndex = sync.OnceValue(func() *benchIndex { + var contractA xdr.ContractId + contractA[0] = 0xA1 + topic := xdr.ScSymbol("bench-topic") + topicVal := xdr.ScVal{Type: xdr.ScValTypeScvSymbol, Sym: &topic} + topicRaw, err := topicVal.MarshalBinary() + if err != nil { + panic(err) + } + ev := xdr.ContractEvent{ + ContractId: &contractA, + Type: xdr.ContractEventTypeContract, + Body: xdr.ContractEventBody{ + V: 0, + V0: &xdr.ContractEventV0{Topics: []xdr.ScVal{topicVal}, Data: topicVal}, + }, + } + raw, err := ev.MarshalBinary() + if err != nil { + panic(err) + } + + // Dense terms go in through the frozen-Bitmaps constructor (roaring + // mode); the sparse one goes in through AddTo so it stays under the + // promotion threshold and is stored as a plain id list. + bms := NewBitmaps() + typeKey := EventTypeTermKey(xdr.ContractEventTypeContract) + contractKey := ComputeTermKey(contractA[:], FieldContractID) + topic1Key := ComputeTermKey(topicRaw, FieldTopic1) + everything := make([]uint32, 0, benchEvents) + contractIDs := make([]uint32, 0, benchEvents/3+1) + topic1IDs := make([]uint32, 0, benchEvents/7+1) + for id := range uint32(benchEvents) { + everything = append(everything, id) + if id%3 == 0 { + contractIDs = append(contractIDs, id) + } + if id%7 == 0 { + topic1IDs = append(topic1IDs, id) + } + } + bms.AddTo(typeKey, everything...) + bms.AddTo(contractKey, contractIDs...) + bms.AddTo(topic1Key, topic1IDs...) + mirror := NewConcurrentBitmapsFromBitmaps(bms) + + topic0Key := ComputeTermKey(topicRaw, FieldTopic0) + sparse := make([]uint32, 0, promotionThreshold-1) + for i := range uint32(promotionThreshold - 1) { + sparse = append(sparse, i*(benchEvents/promotionThreshold)) + } + mirror.AddTo(topic0Key, sparse...) + + eventType := xdr.ContractEventTypeContract + var topics [protocol.MaxTopicCount][]byte + topics[0] = topicRaw + return &benchIndex{ + reader: &hotLikeIndex{&stubIndex{ + mirror: mirror, count: benchEvents, raw: raw, + }}, + filters: []Filter{ + // Two dense groups AND-ed: the intersect arm. + {ContractID: contractA[:], EventType: &eventType}, + // One long-tail sparse group: the arm Get used to + // materialize a bitmap for on every request. + {Topics: topics}, + }, + // A sub-window, so both window edges are live. + window: IDRange{Start: benchEvents / 4, End: benchEvents * 3 / 4}, + } +}) + +// benchMatches drives one page-sized request and stops, the shape a +// getEvents page actually has. +func benchMatches(b *testing.B, descending bool) { + b.Helper() + fx := newBenchIndex() + ctx := context.Background() + b.ReportAllocs() + b.ResetTimer() + for b.Loop() { + n := 0 + for _, err := range Matches(ctx, fx.reader, fx.filters, fx.window, descending, benchPage) { + if err != nil { + b.Fatal(err) + } + n++ + if n == benchPage { + break + } + } + if n != benchPage { + b.Fatalf("fixture sanity: want %d matches, got %d", benchPage, n) + } + } +} + +// BenchmarkMatchesAscending measures the un-materialized iterator tree +// and BenchmarkMatchesDescending its materialized twin — the same +// query, the same page size, the same fetch work, differing only in +// which candidate path Matches takes. The pair is the in-tree A/B for +// the un-materialized path; it stays honest as long as descending +// keeps the bitmap algebra. +func BenchmarkMatchesAscending(b *testing.B) { benchMatches(b, false) } +func BenchmarkMatchesDescending(b *testing.B) { benchMatches(b, true) } diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go new file mode 100644 index 000000000..5ff4d3e9b --- /dev/null +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go @@ -0,0 +1,360 @@ +package event + +// Full-Matches differential: the ascending iterator tree and the +// descending materialized union must select the same events, in +// mirrored order, over randomized corpora, filters and windows. The +// iterator-level differential in match_iter_test.go stops at +// candidateIter; this one drives the whole call, so it also covers +// term planning, the postings-vs-LookupKeys seam, the window cap and +// the batch loop. + +import ( + "context" + "errors" + "iter" + "math/rand" + "slices" + "testing" + + "github.com/RoaringBitmap/roaring/v2" + "github.com/stretchr/testify/require" + + protocol "github.com/stellar/go-stellar-sdk/protocols/rpc" + "github.com/stellar/go-stellar-sdk/xdr" + + "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/chunk" +) + +// diffCorpus is an in-memory chunk carrying one distinct marshaled +// event per id, so the post-filter has real bytes to verify against. +type diffCorpus struct { + raw [][]byte + mirror *ConcurrentBitmaps +} + +// diffReader serves the corpus through LookupKeys only — the +// materializing seam ColdReader and every out-of-package Reader use. +type diffReader struct{ c *diffCorpus } + +// diffPostingsReader adds the no-materialize seam HotStore carries, so +// the same query also runs over sparse ids read in place. +type diffPostingsReader struct{ diffReader } + +func (r diffReader) ChunkID() chunk.ID { return chunk.ID(0) } +func (r diffReader) EventCount() (uint32, error) { return uint32(len(r.c.raw)), nil } + +func (r diffReader) Offsets() (*LedgerOffsets, error) { + return nil, errors.New("diffReader: Offsets is not part of the match path") +} + +func (r diffReader) LookupKeys(_ context.Context, keys []TermKey) ([]*roaring.Bitmap, error) { + out := make([]*roaring.Bitmap, len(keys)) + for i, k := range keys { + bm, err := r.c.mirror.Get(k) + if err != nil { + return nil, err + } + out[i] = bm + } + return out, nil +} + +func (r diffPostingsReader) lookupPostings(_ context.Context, keys []TermKey) ([]postings, error) { + out := make([]postings, len(keys)) + for i, k := range keys { + out[i] = r.c.mirror.lookupPostings(k) + } + return out, nil +} + +func (r diffReader) FetchEvents(_ context.Context, ids []uint32) ([]Payload, error) { + // The precondition is the point: a dedup bug in the union would + // surface here rather than as a silently doubled result. + if err := validateSortedEventIDs(ids); err != nil { + return nil, err + } + out := make([]Payload, len(ids)) + for i, id := range ids { + out[i] = Payload{ContractEventBytes: r.c.raw[id]} + } + return out, nil +} + +func (r diffReader) FetchRange(_ context.Context, start, count uint32) iter.Seq2[Payload, error] { + return func(yield func(Payload, error) bool) { + total, _ := r.EventCount() + if err := validateFetchRange(start, count, total, r.ChunkID()); err != nil { + yield(Payload{}, err) + return + } + for id := start; id < start+count; id++ { + if !yield(Payload{ContractEventBytes: r.c.raw[id]}, nil) { + return + } + } + } +} + +func (r diffReader) All(ctx context.Context) iter.Seq2[Payload, error] { + total, _ := r.EventCount() + return r.FetchRange(ctx, 0, total) +} + +var ( + _ Reader = diffReader{} + _ Reader = diffPostingsReader{} + _ postingReader = diffPostingsReader{} +) + +// diffVocab is the small closed vocabulary the corpus and the random +// filters share, so a generated filter has a real chance of matching. +type diffVocab struct { + contracts [][]byte + topics []xdr.ScVal + topicRaw [][]byte + types []xdr.ContractEventType +} + +func newDiffVocab(t *testing.T) *diffVocab { + t.Helper() + v := &diffVocab{types: []xdr.ContractEventType{ + xdr.ContractEventTypeSystem, + xdr.ContractEventTypeContract, + xdr.ContractEventTypeDiagnostic, + }} + for i := range 4 { + var cid xdr.ContractId + cid[0] = byte(0xC0 + i) + v.contracts = append(v.contracts, cid[:]) + } + for _, name := range []string{"alpha", "beta", "gamma", "delta", "epsilon"} { + sym := xdr.ScSymbol(name) + val := xdr.ScVal{Type: xdr.ScValTypeScvSymbol, Sym: &sym} + raw, err := val.MarshalBinary() + require.NoError(t, err) + v.topics = append(v.topics, val) + v.topicRaw = append(v.topicRaw, raw) + } + return v +} + +func newDiffCorpus(t *testing.T, rng *rand.Rand, v *diffVocab, n int) *diffCorpus { + t.Helper() + c := &diffCorpus{mirror: NewConcurrentBitmapsFromBitmaps(NewBitmaps())} + for id := range n { + var cid xdr.ContractId + copy(cid[:], v.contracts[rng.Intn(len(v.contracts))]) + nTopics := rng.Intn(protocol.MaxTopicCount + 2) + topics := make([]xdr.ScVal, 0, nTopics) + for range nTopics { + topics = append(topics, v.topics[rng.Intn(len(v.topics))]) + } + sym := xdr.ScSymbol("data") + ev := xdr.ContractEvent{ + ContractId: &cid, + Type: v.types[rng.Intn(len(v.types))], + Body: xdr.ContractEventBody{ + V: 0, + V0: &xdr.ContractEventV0{ + Topics: topics, + Data: xdr.ScVal{Type: xdr.ScValTypeScvSymbol, Sym: &sym}, + }, + }, + } + raw, err := ev.MarshalBinary() + require.NoError(t, err) + c.raw = append(c.raw, raw) + keys, err := TermsForBytes(raw) + require.NoError(t, err) + for _, k := range keys { + c.mirror.AddTo(k, uint32(id)) + } + } + return c +} + +// randomFilters builds a filter list over the shared vocabulary, +// including the unconstrained shape that routes to the match-all path. +func randomFilters(rng *rand.Rand, v *diffVocab) []Filter { + filters := make([]Filter, 0, 3) + for range 1 + rng.Intn(3) { + var f Filter + if rng.Intn(3) > 0 { + f.ContractID = v.contracts[rng.Intn(len(v.contracts))] + } + if rng.Intn(3) == 0 { + et := xdr.ContractEventTypeContract + if rng.Intn(2) == 0 { + et = xdr.ContractEventTypeSystem + } + f.EventType = &et + } + for pos := range min(3, protocol.MaxTopicCount) { + if rng.Intn(4) == 0 { + f.Topics[pos] = v.topicRaw[rng.Intn(len(v.topicRaw))] + } + } + if rng.Intn(3) == 0 { + f.TopicCount = TopicCountFilter{ + Count: rng.Intn(protocol.MaxTopicCount + 1), + Exact: rng.Intn(2) == 0, + } + } + filters = append(filters, f) + } + if rng.Intn(20) == 0 { + return nil // the empty-slice match-all shape + } + return filters +} + +func collectOrdinals(t *testing.T, r Reader, filters []Filter, w IDRange, desc bool) []uint32 { + t.Helper() + var out []uint32 + for m, err := range Matches(context.Background(), r, filters, w, desc, 0) { + require.NoError(t, err) + out = append(out, m.Ordinal) + } + return out +} + +// TestMatches_AscendingDescendingDifferential drives 400 randomized +// queries through both candidate paths and both index seams. The +// ascending stream reversed must equal the descending stream exactly. +func TestMatches_AscendingDescendingDifferential(t *testing.T) { + rng := rand.New(rand.NewSource(20260829)) + v := newDiffVocab(t) + const corpusSize = 300 + corpus := newDiffCorpus(t, rng, v, corpusSize) + + // Shrink the batch so multi-batch seams are exercised on a small + // corpus; the stream's contents must not depend on it. + defer func(n int) { matchBatchSize = n }(matchBatchSize) + matchBatchSize = 7 + + readers := []struct { + name string + r Reader + }{ + {"lookupKeys", diffReader{corpus}}, + {"postings", diffPostingsReader{diffReader{corpus}}}, + } + for _, seam := range readers { + name, r := seam.name, seam.r + t.Run(name, func(t *testing.T) { + matched := 0 + for trial := range 400 { + filters := randomFilters(rng, v) + start := uint32(rng.Intn(corpusSize + 1)) + end := start + uint32(rng.Intn(corpusSize+1-int(start))) + w := IDRange{Start: start, End: end} + + asc := collectOrdinals(t, r, filters, w, false) + desc := collectOrdinals(t, r, filters, w, true) + + for i := 1; i < len(asc); i++ { + require.Less(t, asc[i-1], asc[i], + "trial %d: ascending ordinals must be strictly increasing "+ + "(an equal pair is a union dedup bug)", trial) + } + matched += len(asc) + for _, id := range asc { + require.GreaterOrEqual(t, id, w.Start, "trial %d: below window", trial) + require.Less(t, id, w.End, "trial %d: End must be exclusive", trial) + } + slices.Reverse(desc) + require.Equal(t, asc, desc, + "trial %d: window %v filters %+v", trial, w, filters) + } + // Guard against a vacuous pass: the generated queries must + // actually select events, not agree on emptiness. + require.Greater(t, matched, 5000, + "fixture sanity: randomized queries selected too little") + }) + } +} + +// TestMatches_ConcurrentIngestBorrowSafety turns the borrow contract +// into a race-detector gate. The ascending tree's cursors read mirror +// snapshots in place — no clone anywhere — while AddTo publishes new +// termStates on the same keys, including the sparse→dense promotion +// that swaps a term's whole representation. Under -race, any write +// that reached a borrowed snapshot (a COW slip in roaring, a shared +// container mutation) fails the run; without -race the identity check +// still pins that a pinned window's results are immune to concurrent +// ingest past its End. +func TestMatches_ConcurrentIngestBorrowSafety(t *testing.T) { + rng := rand.New(rand.NewSource(20260830)) + v := newDiffVocab(t) + const corpusSize = 400 + const pinned = corpusSize / 2 + + corpus := &diffCorpus{mirror: NewConcurrentBitmapsFromBitmaps(NewBitmaps())} + keysByID := make([][]TermKey, corpusSize) + for id := range corpusSize { + var cid xdr.ContractId + copy(cid[:], v.contracts[rng.Intn(len(v.contracts))]) + topics := make([]xdr.ScVal, 0, 3) + for range 1 + rng.Intn(3) { + topics = append(topics, v.topics[rng.Intn(len(v.topics))]) + } + sym := xdr.ScSymbol("data") + ev := xdr.ContractEvent{ + ContractId: &cid, + Type: v.types[rng.Intn(len(v.types))], + Body: xdr.ContractEventBody{ + V: 0, + V0: &xdr.ContractEventV0{ + Topics: topics, + Data: xdr.ScVal{Type: xdr.ScValTypeScvSymbol, Sym: &sym}, + }, + }, + } + raw, err := ev.MarshalBinary() + require.NoError(t, err) + corpus.raw = append(corpus.raw, raw) + keys, err := TermsForBytes(raw) + require.NoError(t, err) + keysByID[id] = keys + } + // Only the pinned window is indexed up front; the writer feeds the + // rest live. The shared vocabulary is small, so most keys cross the + // sparse→dense promotion threshold mid-run. + for id := range pinned { + for _, k := range keysByID[id] { + corpus.mirror.AddTo(k, uint32(id)) + } + } + + r := diffPostingsReader{diffReader{corpus}} + et := xdr.ContractEventTypeContract + filters := []Filter{ + {ContractID: v.contracts[0]}, + {Topics: [protocol.MaxTopicCount][]byte{0: v.topicRaw[1]}, EventType: &et}, + {TopicCount: TopicCountFilter{Count: 2}}, + } + window := IDRange{Start: 0, End: pinned} + want := collectOrdinals(t, r, filters, window, false) + require.NotEmpty(t, want, "fixture sanity: the pinned window must match something") + + done := make(chan struct{}) + go func() { + defer close(done) + for id := pinned; id < corpusSize; id++ { + for _, k := range keysByID[id] { + corpus.mirror.AddTo(k, uint32(id)) + } + } + }() + for { + select { + case <-done: + require.Equal(t, want, collectOrdinals(t, r, filters, window, false), + "pinned window changed after ingest completed") + return + default: + require.Equal(t, want, collectOrdinals(t, r, filters, window, false), + "pinned window changed mid-ingest") + } + } +} From 4d35a2873cfbf62279f541f9fbd3207772e0e0fe Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Sat, 29 Aug 2026 11:41:31 +0000 Subject: [PATCH 04/41] =?UTF-8?q?events:=20bound=20the=20intersect=20gallo?= =?UTF-8?q?p=20=E2=80=94=20rarest=20group=20leads,=20overrun=20spills=20to?= =?UTF-8?q?=20bulk=20AND?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An AND of chunk-spanning terms with a thin intersection is the cursor tree's worst case: every alignment round raises the floor a little, per-element and branchy, across a window the page never reaches — where roaring.FastAnd answered the same question in word-level bulk. Measured 5-48x over the materialized twin on fat 2-6-term ANDs at 2-5% overlap, the very shape of a typical contract+topic filter over dense pubnet terms; the deep-AND case is the worst, not the best, since uniformly interleaved fat terms give the floor no leaps. No build-time trigger can pick the loser: two shapes with identical group cardinalities differ 13x in walk cost — selectivity of the intersection is what decides, and it is unknowable before walking. So the filter's AND now carries its bulk twin unevaluated and a runtime budget: groups are ordered rarest-first (making the walk's cost bound the rarest group, and fixing the skewed shape outright), alignment rounds are counted, and past alignBudget the iterator hands [floor, End) to the FastOr/FastAnd answer — exact, since the floor only ever rises past ids no child holds. A wrong-fire costs about twice the bulk answer; a right-fire saves an unbounded walk. Budget swept over 2k-32k; knee at 8192. Quiet-box medians: the bad shapes drop 2-fat 1.03ms to 0.49, 3-fat -49%, 6-fat 9.5ms to 0.98, tiny-overlap 6.9ms to 0.43; skew -24% from ordering alone; union-heavy and selective shapes unchanged, the tree still 4-25x ahead of materialized there. End-to-end ascending pays +2% and 3 allocs; descending untouched. A 300-trial randomized spill-vs-walk-vs-materialized differential and a shrunk-budget whole-Matches stream-identity test pin the spill's exactness; alignBudget is a var only as a test seam, like matchBatchSize. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../rpcv2/stores/event/concurrent_bitmaps.go | 33 + .../internal/rpcv2/stores/event/match_iter.go | 200 +++++- .../rpcv2/stores/event/match_iter_test.go | 601 ++++++++++++++++++ 3 files changed, 829 insertions(+), 5 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go index aab5a930b..c9529c2e8 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go @@ -135,6 +135,20 @@ func (d *denseState) snapshot() *roaring.Bitmap { return bm } +// cardinality is the term's id count without materializing a +// snapshot for it. A live pub already is the term, exactly, so it +// answers lock-free; otherwise the count comes off wbm under mu, +// which costs a walk of the writer's containers rather than a clone +// of them. +func (d *denseState) cardinality() uint64 { + if bm := d.pub.Load(); bm != nil { + return bm.GetCardinality() + } + d.mu.Lock() + defer d.mu.Unlock() + return d.wbm.GetCardinality() +} + // postings is an iterable view of ONE term's event IDs, in whichever // representation the index already holds: the sparse mode's sorted // []uint32, a dense term's live denseState, or a bitmap that came @@ -188,6 +202,25 @@ func (p postings) bitmap() *roaring.Bitmap { return p.bm } +// estimate is the term's cardinality over the whole chunk, the weight +// the ascending path's query plan orders an intersection by. It +// ignores the caller's window, so it ranks terms rather than counting +// a query's candidates. The zero postings weighs 0. +// +// A dense term is counted off the writer's own bitmap +// (denseState.cardinality), not off a snapshot: planning wants a +// number, and snapshot would clone a term written since its last read +// to hand back a bitmap the plan may never walk. +func (p postings) estimate() uint64 { + if p.dense != nil { + return p.dense.cardinality() + } + if p.bm != nil { + return p.bm.GetCardinality() + } + return uint64(len(p.ids)) +} + // AddTo records each eventID under key. Callers feed events in // event-ID order relative to the chunk, so a duplicate is a retry of an // already-added prefix and is skipped. diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go index e9d94fda9..a7e46b097 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go @@ -21,6 +21,7 @@ package event // unchanged. import ( + "cmp" "slices" "github.com/RoaringBitmap/roaring/v2" @@ -213,28 +214,63 @@ func (u *unionIter) advance(floor uint32) { // every child is advanced to the running maximum of the peeks until // they all agree on one id. A child that exhausts ends the // intersection — permanently, since cursors only move forward. +// +// Alignment is bounded when bulk is set. Every round that does not +// settle the AND ends with the leading child stepped past one of its +// ids, so a walk costs up to one round per id of that child — and +// with chunk-sized groups on both sides of a thin overlap, the walk +// spends all of them on a window the consumer's page never reaches. +// Once the budget is gone the AND hands the rest of its window to +// bulk, whose cost is bounded by the containers the terms span rather +// than by the ids inside them. A nil bulk leaves the alignment +// unbounded, which is what an AND with no bulk twin gets. type intersectIter struct { children []idIter cur uint32 ok bool primed bool + + bulk *bulkAnd + budget uint64 + rounds uint64 + spilled idIter } func (n *intersectIter) peek() (uint32, bool) { + if n.spilled != nil { + return n.spilled.peek() + } if !n.primed { n.cur, n.ok = n.align() + if n.spilled != nil { + return n.spilled.peek() + } n.primed = true } return n.cur, n.ok } -// align raises every child to the smallest id all of them hold. +// align raises every child to the smallest id all of them hold, or +// spills to the bulk answer when it runs out of rounds first. func (n *intersectIter) align() (uint32, bool) { cand, ok := n.children[0].peek() if !ok { return 0, false } for { + if n.bulk != nil { + n.rounds++ + if n.rounds > n.budget { + // A child raises the floor only past ids it does not + // hold, and cand only rises, so no id of the + // intersection was skipped between the last one + // yielded and cand: the bulk answer from cand up is + // exactly the rest of this AND. + n.spilled = n.bulk.iter(cand) + n.bulk = nil + return 0, false + } + } raised := false for _, c := range n.children { c.advance(cand) @@ -261,6 +297,10 @@ func (n *intersectIter) next() { if !ok { return } + if n.spilled != nil { + n.spilled.next() + return + } // Post-align every child sits on v; step them all past it. for _, c := range n.children { if cv, cok := c.peek(); cok && cv == v { @@ -274,6 +314,10 @@ func (n *intersectIter) advance(floor uint32) { if v, ok := n.peek(); !ok || v >= floor { return } + if n.spilled != nil { + n.spilled.advance(floor) + return + } for _, c := range n.children { c.advance(floor) } @@ -323,11 +367,11 @@ func intersectOf(children []idIter) idIter { func candidateIter(plans []termPlan, sources []postings, window IDRange) idIter { perFilter := make([]idIter, 0, len(plans)) for _, plan := range plans { - groups := make([]idIter, 0, len(plan)) + groups := make([]candidateGroup, 0, len(plan)) missed := false for _, slots := range plan { - g := groupIter(sources, slots, window) - if g == nil { + g, ok := resolveGroup(sources, slots) + if !ok { missed = true break } @@ -336,11 +380,157 @@ func candidateIter(plans []termPlan, sources []postings, window IDRange) idIter if missed { continue } - perFilter = append(perFilter, intersectOf(groups)) + perFilter = append(perFilter, filterIter(sources, groups, window)) } return unionOf(perFilter) } +// candidateGroup is one of a filter's resolved groups: where its terms +// live in the batched lookup, and the weight that orders the filter's +// AND. +type candidateGroup struct { + slots []int + est uint64 +} + +// resolveGroup weighs the postings at slots, reporting false when +// every one of them is absent from the index — the mirror of +// unionSlots' nil return, and the signal that the owning filter can +// match nothing. +// +// The weight sums the present terms' cardinalities, an upper bound the +// OR's dedup can only lower. It orders an intersection; nothing reads +// it as a count. +func resolveGroup(sources []postings, slots []int) (candidateGroup, bool) { + g := candidateGroup{slots: slots} + present := false + for _, slot := range slots { + p := sources[slot] + if !p.present() { + continue + } + present = true + g.est += p.estimate() + } + return g, present +} + +// alignBudget is how many alignment rounds one filter's AND may spend +// before it spills to the bulk answer. +// +// A round is a handful of galloping advances — tens of nanoseconds +// over chunk-spanning terms — so the budget buys a few hundred +// microseconds of walking, about what the bulk answer costs for a +// filter whose terms span a whole chunk. That is the shape of the +// trade at every value: a spill pays for the walk it abandoned on top +// of the bulk answer, so it costs about twice the bulk on a filter it +// fires on wrongly, and saves the difference between the bulk and an +// unbounded walk on one it fires on rightly. +// +// A filter selective enough to fill a page out of the window's first +// fraction settles in a round or two per id it yields and never comes +// near the budget, whatever it is set to. What the value decides is +// the boundary between filters that yield steadily but slowly — a few +// dozen rounds per id, where the walk still finishes — and the ones +// whose alignment crosses a chunk to find a handful of matches. +// +// A var, not a const, so in-package tests can shrink it to force the +// spill; it never changes what a stream yields. +// +//nolint:gochecknoglobals // test seam; production never writes it +var alignBudget uint64 = 8192 + +// filterIter builds one filter's candidate cursor: the AND of its +// groups, rarest first, bounded by alignBudget. +// +// Rarest first because intersectIter's alignment seeds its floor from +// the leading child and raises the others to it in order, so the +// leading cursor is the one every barren round steps forward, and the +// trailing ones are the ones spared a wasted advance. Intersection is +// commutative, so the order changes only how fast the gallop +// converges — and it is what makes the budget's bound the rarest +// group's cardinality rather than the fattest's. +func filterIter(sources []postings, groups []candidateGroup, window IDRange) idIter { + slices.SortStableFunc(groups, func(a, b candidateGroup) int { + return cmp.Compare(a.est, b.est) + }) + children := make([]idIter, len(groups)) + for i := range groups { + children[i] = groupIter(sources, groups[i].slots, window) + } + if len(children) < 2 { + return intersectOf(children) + } + return &intersectIter{ + children: children, + bulk: &bulkAnd{sources: sources, groups: groups, end: window.End}, + budget: alignBudget, + } +} + +// bulkAnd is a filter's AND as roaring's aggregation would answer it, +// held unevaluated beside the walk that normally answers it instead. +// It is the walk's bound: the cost of the aggregation is set by the +// containers the terms span, so it does not grow with a window the +// walk would have to cross id by id. +type bulkAnd struct { + sources []postings + groups []candidateGroup + end uint32 +} + +// iter computes the filter's candidate set — OR each group, AND the +// groups smallest first — and returns a cursor over the part of it at +// or above floor. +// +// The inputs may be borrowed mirror snapshots: FastAnd and FastOr read +// them without writing through, and never see the single-element slice +// that would make them Clone, so what comes back is this call's own +// bitmap, and only the intersection large. The window lands at the +// leaf, exactly as it does on a borrowed term. +func (b *bulkAnd) iter(floor uint32) idIter { + inputs := make([]*roaring.Bitmap, len(b.groups)) + for i := range b.groups { + inputs[i] = orGroup(b.sources, b.groups[i].slots) + } + // FastAnd intersects left to right, so the smallest input first + // shrinks the accumulator fastest — the caller-side prep roaring's + // own docs call for, and the one unionForFilters does. A group's + // weight only bounds its OR, so the inputs are ranked again once + // they exist. + slices.SortFunc(inputs, func(x, y *roaring.Bitmap) int { + return cmp.Compare(x.GetCardinality(), y.GetCardinality()) + }) + return postings{bm: roaring.FastAnd(inputs...)}.iter( + IDRange{Start: floor, End: b.end}) +} + +// orGroup ORs a group's present terms into one bulkAnd input. A group +// holding one of them is that term's bitmap: FastOr has historically +// Cloned a single-element slice. A sparse term is inflated here, the +// one place the ascending path does what Get does — it costs the ids +// it holds, and only a filter whose walk already overran its budget +// ever pays it. +func orGroup(sources []postings, slots []int) *roaring.Bitmap { + present := make([]*roaring.Bitmap, 0, len(slots)) + for _, slot := range slots { + p := sources[slot] + if bm := p.bitmap(); bm != nil { + present = append(present, bm) + continue + } + if p.ids != nil { + bm := roaring.New() + bm.AddMany(p.ids) + present = append(present, bm) + } + } + if len(present) == 1 { + return present[0] + } + return roaring.FastOr(present...) +} + // groupIter ORs the postings at slots into one cursor, returning nil // when every one of them is absent from the index — the mirror of // unionSlots' nil return, and the signal that the owning filter can diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go index 93eb06dcf..6636eeeb7 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go @@ -299,6 +299,229 @@ func TestGroupIterAbsentGroup(t *testing.T) { assert.Equal(t, []uint32{1, 2, 3}, drain(groupIter(sources, []int{0, 3}, wholeWindow))) } +// TestPostingsEstimate pins the ordering weight on both +// representations: the term's whole-chunk cardinality, window and all. +func TestPostingsEstimate(t *testing.T) { + assert.Equal(t, uint64(0), postings{}.estimate(), "the absent term weighs nothing") + assert.Equal(t, uint64(0), postings{bm: roaring.New()}.estimate()) + assert.Equal(t, uint64(3), sparseSource(1, 2, 3).estimate()) + assert.Equal(t, uint64(3), denseSource(1, 2, 3).estimate()) + assert.Equal(t, uint64(4), denseSource(1, 2, 3, 1<<20).estimate(), + "cardinality spans containers") +} + +// TestResolveGroup pins what a group reports about itself: presence, +// and the summed weight that orders its filter's AND. +func TestResolveGroup(t *testing.T) { + sources := []postings{ + sparseSource(1, 2), + {}, // absent + denseSource(2, 3, 4), + } + + g, ok := resolveGroup(sources, []int{0}) + require.True(t, ok) + assert.Equal(t, uint64(2), g.est) + assert.Equal(t, []int{0}, g.slots) + + g, ok = resolveGroup(sources, []int{0, 2}) + require.True(t, ok) + assert.Equal(t, uint64(5), g.est, "a group's terms sum, overlaps double-counted") + + g, ok = resolveGroup(sources, []int{1, 2}) + require.True(t, ok) + assert.Equal(t, uint64(3), g.est, "an absent term adds nothing") + + g, ok = resolveGroup(sources, []int{1}) + assert.False(t, ok, "a group of absent terms drops its filter") + assert.Equal(t, uint64(0), g.est) +} + +// TestFilterIterOrdersRarestFirst pins the driver choice: the rarest +// group leads the AND however the plan named its groups, which is what +// bounds the walk at one round per id of that group. +func TestFilterIterOrdersRarestFirst(t *testing.T) { + sources := []postings{ + denseSource(1, 2, 3, 4, 5, 6, 7, 8), // 0: the fat group + denseSource(2, 4, 6, 8), // 1 + denseSource(4, 8), // 2: the rare group + } + slotSets := [][]int{{0}, {2}, {1}} + groups := make([]candidateGroup, 0, len(slotSets)) + for _, slots := range slotSets { + g, ok := resolveGroup(sources, slots) + require.True(t, ok) + groups = append(groups, g) + } + it := filterIter(sources, groups, wholeWindow) + n, isIntersect := it.(*intersectIter) + require.True(t, isIntersect) + assert.Equal(t, [][]int{{2}, {1}, {0}}, + [][]int{groups[0].slots, groups[1].slots, groups[2].slots}, + "the groups are reordered rarest first") + assert.Equal(t, alignBudget, n.budget) + assert.Equal(t, []uint32{4, 8}, drain(it)) + + // A one-group filter is the group itself: no AND, so no budget. + one, ok := resolveGroup(sources, []int{2}) + require.True(t, ok) + assert.Equal(t, []uint32{4, 8}, + drain(filterIter(sources, []candidateGroup{one}, wholeWindow))) +} + +// planGroups resolves a whole filter, for the tests that drive +// filterIter directly. +func planGroups(t *testing.T, sources []postings, plan termPlan) []candidateGroup { + t.Helper() + groups := make([]candidateGroup, 0, len(plan)) + for _, slots := range plan { + g, ok := resolveGroup(sources, slots) + require.True(t, ok) + groups = append(groups, g) + } + return groups +} + +// TestIntersectIterSpills pins the fallback: an AND that overruns its +// budget answers the rest of its window out of the bulk bitmap, +// yielding exactly what the walk would have — no id repeated across +// the seam, none dropped at it — at every budget the seam can fall on. +func TestIntersectIterSpills(t *testing.T) { + sources := []postings{ + denseSource(1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12), + denseSource(2, 4, 6, 8, 10, 12), + sparseSource(3, 4, 8, 12, 20), + } + plan := termPlan{{0}, {1}, {2}} + + for _, budget := range []uint64{0, 1, 2, 3, 5, 100} { + it, ok := filterIter(sources, planGroups(t, sources, plan), wholeWindow).(*intersectIter) + require.True(t, ok) + it.budget = budget + assert.Equal(t, []uint32{4, 8, 12}, drain(it), "budget %d", budget) + } + + // A budget of zero spills on the first round, so the whole answer + // comes from the bulk bitmap — sparse group and all, which the + // aggregation inflates rather than walks — and the window still + // lands at the leaf. + it, ok := filterIter(sources, planGroups(t, sources, plan), + IDRange{Start: 0, End: 12}).(*intersectIter) + require.True(t, ok) + it.budget = 0 + assert.Equal(t, []uint32{4, 8}, drain(it)) + assert.NotNil(t, it.spilled) +} + +// TestIntersectIterSpillAdvance pins the seam under advance: a gallop +// that crosses the spill lands where the walk would have. +func TestIntersectIterSpillAdvance(t *testing.T) { + sources := []postings{ + denseSource(1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12), + denseSource(2, 4, 6, 8, 10, 12), + } + plan := termPlan{{0}, {1}} + for _, budget := range []uint64{0, 1, 2, 100} { + it, ok := filterIter(sources, planGroups(t, sources, plan), wholeWindow).(*intersectIter) + require.True(t, ok) + it.budget = budget + it.advance(7) + v, vok := it.peek() + require.True(t, vok, "budget %d", budget) + assert.Equal(t, uint32(8), v, "budget %d", budget) + it.advance(3) + v, _ = it.peek() + assert.Equal(t, uint32(8), v, "advance backwards is a no-op, budget %d", budget) + assert.Equal(t, []uint32{8, 10, 12}, drain(it), "budget %d", budget) + } +} + +// randomBulkPlan draws a corpus of overlapping terms in both +// representations and one filter's plan over it, the shape family the +// spill has to answer identically to the walk. +func randomBulkPlan(rng *rand.Rand, idSpace int) ([]postings, termPlan) { + sources := make([]postings, 1+rng.Intn(5)) + for i := range sources { + n := rng.Intn(idSpace / 2) + seen := make(map[uint32]struct{}, n) + for range n { + seen[uint32(rng.Intn(idSpace))] = struct{}{} + } + ids := make([]uint32, 0, len(seen)) + for id := range seen { + ids = append(ids, id) + } + slices.Sort(ids) + if rng.Intn(4) == 0 { + sources[i] = postings{ids: ids} + } else { + sources[i] = denseSource(ids...) + } + } + plan := make(termPlan, 1+rng.Intn(4)) + for g := range plan { + slots := make([]int, 1+rng.Intn(2)) + for s := range slots { + slots[s] = rng.Intn(len(sources)) + } + plan[g] = slots + } + return sources, plan +} + +// TestFilterIterBulkMatchesWalk is the spill's equivalence gate: over +// randomized plans, source shapes and windows, an AND that spills must +// yield exactly what the unbounded walk yields, and both must equal the +// materialized algebra. The budget is moved rather than the corpus, so +// the same filter is answered every way. +func TestFilterIterBulkMatchesWalk(t *testing.T) { + rng := rand.New(rand.NewSource(20260901)) + const idSpace = 4000 + spills := 0 + for trial := range 300 { + sources, plan := randomBulkPlan(rng, idSpace) + start := uint32(rng.Intn(idSpace)) + end := start + uint32(rng.Intn(idSpace)) + window := IDRange{Start: start, End: end} + + build := func(budget uint64) (idIter, *intersectIter) { + groups := make([]candidateGroup, 0, len(plan)) + for _, slots := range plan { + g, ok := resolveGroup(sources, slots) + if !ok { + return nil, nil + } + groups = append(groups, g) + } + it := filterIter(sources, groups, window) + n, _ := it.(*intersectIter) + if n != nil { + n.budget = budget + } + return it, n + } + walk, _ := build(^uint64(0)) + if walk == nil { + continue + } + want := referenceCandidates([]termPlan{plan}, sources, window) + require.Equal(t, want, drain(walk), + "trial %d: window %v plan %v", trial, window, plan) + // Budgets in the low single digits put the seam at the start of + // a plan's answer and partway into it, so the join is under test + // and not just its ends. + for _, budget := range []uint64{0, 1, 3} { + it, n := build(budget) + require.Equal(t, want, drain(it), + "trial %d budget %d: window %v plan %v", trial, budget, window, plan) + if n != nil && n.spilled != nil { + spills++ + } + } + } + require.Greater(t, spills, 100, "fixture sanity: the spill must actually fire") +} + func TestCandidateIterDropsFilterWithAbsentGroup(t *testing.T) { sources := []postings{ sparseSource(1, 2, 3), @@ -365,6 +588,43 @@ func referenceCandidates(plans []termPlan, sources []postings, window IDRange) [ return union.ToArray() } +// TestMatchesSpillYieldsSameStream drives whole Matches calls with the +// alignment budget shrunk, so every filter's AND spills on its first +// rounds, and requires the stream to be the one the unbounded walk +// yields. It puts the seam where it actually sits: under term +// planning, the window cap, the batch loop and the post-filter. +func TestMatchesSpillYieldsSameStream(t *testing.T) { + rng := rand.New(rand.NewSource(20260902)) + v := newDiffVocab(t) + const corpusSize = 300 + corpus := newDiffCorpus(t, rng, v, corpusSize) + + // Shrink the batch too, so a spill can land mid-page. + defer func(n int) { matchBatchSize = n }(matchBatchSize) + matchBatchSize = 7 + defer func(n uint64) { alignBudget = n }(alignBudget) + + r := diffPostingsReader{diffReader{corpus}} + matched := 0 + for trial := range 200 { + filters := randomFilters(rng, v) + start := uint32(rng.Intn(corpusSize + 1)) + end := start + uint32(rng.Intn(corpusSize+1-int(start))) + w := IDRange{Start: start, End: end} + + alignBudget = ^uint64(0) + want := collectOrdinals(t, r, filters, w, false) + matched += len(want) + for _, budget := range []uint64{0, 1, 4} { + alignBudget = budget + require.Equal(t, want, collectOrdinals(t, r, filters, w, false), + "trial %d budget %d: window %v filters %+v", trial, budget, w, filters) + } + } + require.Greater(t, matched, 2000, + "fixture sanity: randomized queries selected too little") +} + // TestCandidateIterMatchesMaterializedAlgebra is the differential // gate: over randomized plans, source shapes (absent / sparse / dense // / present-but-empty) and windows, the un-materialized tree must @@ -626,3 +886,344 @@ func benchMatches(b *testing.B, descending bool) { // keeps the bitmap algebra. func BenchmarkMatchesAscending(b *testing.B) { benchMatches(b, false) } func BenchmarkMatchesDescending(b *testing.B) { benchMatches(b, true) } + +// ─── the candidate-shape microbench ───────────────────────────────── +// +// The A/B above drives one production query end to end; the matrix +// below isolates the candidate set itself. Both paths answer the same +// synthetic plan over the same mirror — candidateIter's cursor tree +// against unionForFilters' bitmap algebra — with fetch and +// post-filter out of frame, so a shape's number is candidate work +// alone. The shapes are the term geometries the two are expected to +// disagree on: fat partially-overlapping ANDs, a skewed AND, a deep +// AND, and a wide OR. + +// benchFat is one fat term's cardinality against benchEvents: ~7% of +// the domain, the density at which roaring holds a term as bitmap +// containers — the representation FastAnd intersects a word at a +// time and the cursor tree walks a bit at a time. +const benchFat = 300_000 + +// benchRand is a deterministic xorshift. The shapes must be identical +// from run to run, and a fixed stride would hand the gallop a +// regularity real postings do not have. +type benchRand uint64 + +func (r *benchRand) next() uint64 { + x := uint64(*r) + x ^= x << 13 + x ^= x >> 7 + x ^= x << 17 + *r = benchRand(x) + return x +} + +// scatter draws exactly k ascending ids from the residue class +// {i : i ≡ res (mod m), i < domain}, one per fixed stride at a +// jittered offset inside it. Terms built on disjoint residue classes +// interleave at single-id granularity while sharing nothing, so a +// shape's overlap is exactly the class its terms are built to share. +func scatter(rng *benchRand, domain, m, res uint32, k int) []uint32 { + if k == 0 { + return nil + } + class := (domain - res + m - 1) / m + stride := class / uint32(k) + if stride == 0 { + panic("scatter: residue class too small for k") + } + ids := make([]uint32, k) + for t := range k { + ids[t] = (uint32(t)*stride+uint32(rng.next()%uint64(stride)))*m + res + } + return ids +} + +// fatGroup builds n terms of card ids each: term i draws its private +// ids from residue class base+i, and every term also holds the class +// base+n, so the group's joint intersection is exactly that shared +// class and every pairwise overlap is the same set. mod is the total +// number of classes in play, so several groups can be laid over one +// domain without colliding. +func fatGroup(rng *benchRand, domain, mod, base uint32, n, card, shared int) [][]uint32 { + common := scatter(rng, domain, mod, base+uint32(n), shared) + out := make([][]uint32, n) + for i := range out { + ids := scatter(rng, domain, mod, base+uint32(i), card-shared) + ids = append(ids, common...) + slices.Sort(ids) + out[i] = ids + } + return out +} + +// benchShape is one synthetic candidate-set problem: a term corpus in +// the mirror, the plan resolved over it, and the page both paths must +// produce from it. +type benchShape struct { + name string + reader *hotLikeIndex + plans []termPlan + keys []TermKey + window IDRange + // wantCount and wantSum fingerprint the first page. Both paths + // check them every iteration, so a harness that stopped answering + // the query cannot post a fast number. + wantCount int + wantSum uint64 +} + +// newBenchShape indexes terms as one term each, resolves the window to +// most of the domain with both edges live, and fingerprints the first +// page off the materialized algebra — the reference the cursor tree +// must reproduce id for id. +func newBenchShape(name string, domain uint32, terms [][]uint32, plans []termPlan) *benchShape { + bms := NewBitmaps() + keys := make([]TermKey, len(terms)) + for i, ids := range terms { + keys[i] = TermKey{0: byte(i + 1)} + bms.AddTo(keys[i], ids...) + } + s := &benchShape{ + name: name, + reader: &hotLikeIndex{&stubIndex{ + mirror: NewConcurrentBitmapsFromBitmaps(bms), count: domain, + }}, + plans: plans, + keys: keys, + window: IDRange{Start: domain / 32, End: domain - domain/32}, + } + union, err := unionForFilters( + context.Background(), s.reader, s.plans, s.keys, s.window) + if err != nil { + panic(err) + } + it := union.Iterator() + for s.wantCount < benchPage && it.HasNext() { + s.wantCount++ + s.wantSum += uint64(it.Next()) + } + return s +} + +// singleFilterPlan is one filter AND-ing n one-term groups: the +// intersect shapes' plan. +func singleFilterPlan(n int) []termPlan { + plan := make(termPlan, n) + for i := range plan { + plan[i] = []int{i} + } + return []termPlan{plan} +} + +// benchShapes is the shape matrix, each entry built on first use so a +// -bench selecting one shape pays for one shape. domain is a +// parameter so the correctness twin of the matrix can run the same +// geometries small. +func benchShapes(domain uint32) []struct { + name string + build func() *benchShape +} { + scale := func(n int) int { return max(1, n*int(domain)/benchEvents) } + fat := scale(benchFat) + // ~3% of a fat term: the partial overlap that makes an aligning + // AND converge slowly without making it empty. + partial := fat * 3 / 100 + // Just over one page once the window clips it: the intersection + // too small to fill a page early, so the walk spans the window. + tiny := scale(1200) + + shapes := []struct { + name string + build func() *benchShape + }{ + {"a_and2_fat_3pct", func() *benchShape { + rng := benchRand(1) + return newBenchShape("a", domain, + fatGroup(&rng, domain, 3, 0, 2, fat, partial), singleFilterPlan(2)) + }}, + {"b_and3_fat_3pct", func() *benchShape { + rng := benchRand(2) + return newBenchShape("b", domain, + fatGroup(&rng, domain, 4, 0, 3, fat, partial), singleFilterPlan(3)) + }}, + {"c_and2_skew", func() *benchShape { + rng := benchRand(3) + // The small term is a subset of the fat one, spread over + // it, so the AND is entirely decided by the rare side — + // the gallop-friendly control. + big := scatter(&rng, domain, 1, 0, fat) + small := make([]uint32, 0, scale(2000)) + step := len(big) / cap(small) + for i := range cap(small) { + small = append(small, big[i*step]) + } + return newBenchShape("c", domain, + [][]uint32{big, small}, singleFilterPlan(2)) + }}, + {"d_and6_fat_tiny", func() *benchShape { + rng := benchRand(4) + return newBenchShape("d", domain, + fatGroup(&rng, domain, 7, 0, 6, fat, tiny), singleFilterPlan(6)) + }}, + {"e_or10_single_term", func() *benchShape { + rng := benchRand(5) + terms := fatGroup(&rng, domain, 11, 0, 10, scale(30_000), 0) + plans := make([]termPlan, len(terms)) + for i := range plans { + plans[i] = termPlan{{i}} + } + return newBenchShape("e", domain, terms, plans) + }}, + {"f_and2_fat_tiny", func() *benchShape { + rng := benchRand(6) + return newBenchShape("f", domain, + fatGroup(&rng, domain, 3, 0, 2, fat, tiny), singleFilterPlan(2)) + }}, + {"h_and2_fat_overlapping", func() *benchShape { + rng := benchRand(8) + // The serving default: one selective term AND-ed with a + // near-total one (an event type constrains almost + // nothing). The intersection is nearly the selective term + // itself, so a page comes out of the window's first + // fraction — the shape the cursor tree exists to serve, + // and the one any eager rule must leave alone. + selective := scatter(&rng, domain, 3, 0, fat) + nearAll := make([]uint32, 0, domain) + for id := range domain { + if id%50 != 7 { + nearAll = append(nearAll, id) + } + } + return newBenchShape("h", domain, + [][]uint32{selective, nearAll}, singleFilterPlan(2)) + }}, + {"g_and3_x4_filters", func() *benchShape { + rng := benchRand(7) + // The serving shape the tail regression was measured on: + // several filters, each AND-ing a few fat terms. Each + // filter owns four residue classes, so the filters overlap + // only where the union has to dedup them. + terms := make([][]uint32, 0, 12) + plans := make([]termPlan, 0, 4) + for f := range uint32(4) { + group := fatGroup(&rng, domain, 16, f*4, 3, scale(75_000), scale(2250)) + plan := make(termPlan, len(group)) + for i := range group { + plan[i] = []int{len(terms) + i} + } + terms = append(terms, group...) + plans = append(plans, plan) + } + return newBenchShape("g", domain, terms, plans) + }}, + } + return shapes +} + +// benchShapeCache keeps one built corpus per shape name, so the tree +// and materialized runs of a shape share it. Benchmarks run one at a +// time, so a plain map suffices. +var benchShapeCache = map[string]*benchShape{} + +func shapeFor(name string, build func() *benchShape) *benchShape { + s, ok := benchShapeCache[name] + if !ok { + s = build() + benchShapeCache[name] = s + } + return s +} + +// benchCandidatePage pulls one page of candidates through the cursor +// tree — the ascending path's candidate work, with nothing else in +// frame. +func benchCandidatePage(b *testing.B, s *benchShape) { + b.Helper() + ctx := context.Background() + b.ReportAllocs() + for b.Loop() { + sources, err := lookupPostings(ctx, s.reader, s.keys) + if err != nil { + b.Fatal(err) + } + it := candidateIter(s.plans, sources, s.window) + n, sum := 0, uint64(0) + for n < benchPage { + v, ok := it.peek() + if !ok { + break + } + n, sum = n+1, sum+uint64(v) + it.next() + } + if n != s.wantCount || sum != s.wantSum { + b.Fatalf("page mismatch: got (%d, %d), want (%d, %d)", + n, sum, s.wantCount, s.wantSum) + } + } +} + +// benchMaterializedPage is benchCandidatePage's twin over the bitmap +// algebra: build the whole candidate set, then read one page off it. +// The direction of that read is immaterial — the materialization +// dominates and the page is 1000 steps either way — so it reads +// ascending, which makes the two harnesses answer bit for bit. +func benchMaterializedPage(b *testing.B, s *benchShape) { + b.Helper() + ctx := context.Background() + b.ReportAllocs() + for b.Loop() { + union, err := unionForFilters(ctx, s.reader, s.plans, s.keys, s.window) + if err != nil { + b.Fatal(err) + } + it := union.Iterator() + n, sum := 0, uint64(0) + for n < benchPage && it.HasNext() { + n, sum = n+1, sum+uint64(it.Next()) + } + if n != s.wantCount || sum != s.wantSum { + b.Fatalf("page mismatch: got (%d, %d), want (%d, %d)", + n, sum, s.wantCount, s.wantSum) + } + } +} + +// BenchmarkCandidateTree and BenchmarkCandidateMaterialized are the +// per-shape A/B for the candidate set: the same plan, the same +// postings, the same page, differing only in whether the ids are +// pulled through the cursor tree or read off a materialized bitmap. +func BenchmarkCandidateTree(b *testing.B) { + for _, sh := range benchShapes(benchEvents) { + b.Run(sh.name, func(b *testing.B) { + benchCandidatePage(b, shapeFor(sh.name, sh.build)) + }) + } +} + +func BenchmarkCandidateMaterialized(b *testing.B) { + for _, sh := range benchShapes(benchEvents) { + b.Run(sh.name, func(b *testing.B) { + benchMaterializedPage(b, shapeFor(sh.name, sh.build)) + }) + } +} + +// TestBenchShapesAgree runs the whole shape matrix small: every +// geometry the microbench measures must be one both candidate paths +// answer identically, so a shape can never post a number for a query +// the tree gets wrong. +func TestBenchShapesAgree(t *testing.T) { + const domain = 1 << 16 + for _, sh := range benchShapes(domain) { + t.Run(sh.name, func(t *testing.T) { + s := sh.build() + sources, err := lookupPostings(context.Background(), s.reader, s.keys) + require.NoError(t, err) + got := drain(candidateIter(s.plans, sources, s.window)) + require.Equal(t, referenceCandidates(s.plans, sources, s.window), got) + require.NotEmpty(t, got, "shape sanity: the plan must select something") + }) + } +} From ee9d2cd05302f26805f4d108b71ba2ce7a545102 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Sat, 29 Aug 2026 11:42:46 +0000 Subject: [PATCH 05/41] =?UTF-8?q?deps:=20roaring=20v2.18.2=20->=20v2.26.0?= =?UTF-8?q?=20=E2=80=94=20bulk=20aggregation=20without=20the=20pre-OR=20cl?= =?UTF-8?q?one?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The descending events path and the ascending path's gallop-overrun spill both answer filters through roaring.FastOr/FastAnd, and upstream made exactly those cheaper: lazy unions no longer clone their first input (#542) and the container aggregations run as fused single-pass cardinality-slice ops (#559). Measured on our BenchmarkMatchesDescending: -8.5% CPU (FastOr -40%, FastAnd -20%), about two thirds of it with SIMD forced off, allocations unchanged to the byte. Serialized bitmap bytes are identical across the bump, deserialization internals untouched (the pubnet cold unmarshal cost moves with neither version), ingest and the ascending walk are flat. The stricter arrayContainer validation added upstream (#545) is reachable only through Validate/MustReadFrom, which nothing here calls. bits-and-blooms/bitset rides along v1.24.2 -> v1.24.4. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- go.mod | 11 +++++++---- go.sum | 8 ++++---- 2 files changed, 11 insertions(+), 8 deletions(-) diff --git a/go.mod b/go.mod index b1cd757dc..5e5c47e3d 100644 --- a/go.mod +++ b/go.mod @@ -4,9 +4,12 @@ go 1.26 require ( github.com/Masterminds/squirrel v1.5.4 - // Minimum v2.18.2: first upstream release with the FastOr/runContainer16 - // fix (RoaringBitmap/roaring#527) that the fork previously carried. - github.com/RoaringBitmap/roaring/v2 v2.18.2 + // Minimum v2.18.2 (the FastOr/runContainer16 fix, #527, the fork + // previously carried); v2.26.0 for the no-clone lazy union (#542) + // and fused cardinality-slice aggregation (#559) the descending + // events path leans on. SIMD paths are x/sys/cpu-gated and honor + // GODEBUG=cpu.avx512vpopcntdq=off. + github.com/RoaringBitmap/roaring/v2 v2.26.0 github.com/aws/aws-sdk-go-v2 v1.45.1 github.com/aws/aws-sdk-go-v2/config v1.31.16 github.com/aws/aws-sdk-go-v2/service/s3 v1.110.0 @@ -61,7 +64,7 @@ require ( github.com/aws/aws-sdk-go-v2/service/sso v1.30.0 // indirect github.com/aws/aws-sdk-go-v2/service/ssooidc v1.35.4 // indirect github.com/aws/aws-sdk-go-v2/service/sts v1.39.0 // indirect - github.com/bits-and-blooms/bitset v1.24.2 // indirect + github.com/bits-and-blooms/bitset v1.24.4 // indirect github.com/cncf/xds/go v0.0.0-20260202195803-dba9d589def2 // indirect github.com/edsrzf/mmap-go v1.2.0 // indirect github.com/envoyproxy/go-control-plane/envoy v1.37.0 // indirect diff --git a/go.sum b/go.sum index 2828a53be..baafbb3a0 100644 --- a/go.sum +++ b/go.sum @@ -42,8 +42,8 @@ github.com/Masterminds/squirrel v1.5.4 h1:uUcX/aBc8O7Fg9kaISIUsHXdKuqehiXAMQTYX8 github.com/Masterminds/squirrel v1.5.4/go.mod h1:NNaOrjSoIDfDA40n7sr2tPNZRfjzjA400rg+riTZj10= github.com/Microsoft/go-winio v0.6.2 h1:F2VQgta7ecxGYO8k3ZZz3RS8fVIXVxONVUPlNERoyfY= github.com/Microsoft/go-winio v0.6.2/go.mod h1:yd8OoFMLzJbo9gZq8j5qaps8bJ9aShtEA8Ipt1oGCvU= -github.com/RoaringBitmap/roaring/v2 v2.18.2 h1:oPq3Cgx//iDuJQVp6xSInAKW34J9CEwE5GmLI2z+Eic= -github.com/RoaringBitmap/roaring/v2 v2.18.2/go.mod h1:eq4wdNXxtJIS/oikeCzdX1rBzek7ANzbth041hrU8Q4= +github.com/RoaringBitmap/roaring/v2 v2.26.0 h1:K30ZxF4vZcIKvJsbmgfiep2K64f+dILJqkYGoj4xnwU= +github.com/RoaringBitmap/roaring/v2 v2.26.0/go.mod h1:BZufmFbox589n3j5eOmyTaLSGXbRLc2LmQvjKjzSEGU= github.com/ajg/form v0.0.0-20160822230020-523a5da1a92f h1:zvClvFQwU++UpIUBGC8YmDlfhUrweEy1R1Fj1gu5iIM= github.com/ajg/form v0.0.0-20160822230020-523a5da1a92f/go.mod h1:uL1WgH+h2mgNtvBq0339dVnzXdBETtL2LeUXaIv25UY= github.com/andybalholm/brotli v1.0.4 h1:V7DdXeJtZscaqfNuAdSRuRFzuiKlHSC/Zh3zl9qY3JY= @@ -94,8 +94,8 @@ github.com/aws/smithy-go v1.28.1 h1:R/nXH00c8qcfCzQVELtRw+eLQWtzv+VAIEFJ1/xxXlQ= github.com/aws/smithy-go v1.28.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc= github.com/beorn7/perks v1.0.1 h1:VlbKKnNfV8bJzeqoa4cOKqO6bYr3WgKZxO8Z16+hsOM= github.com/beorn7/perks v1.0.1/go.mod h1:G2ZrVWU2WbWT9wwq4/hrbKbnv/1ERSJQ0ibhJ6rlkpw= -github.com/bits-and-blooms/bitset v1.24.2 h1:M7/NzVbsytmtfHbumG+K2bremQPMJuqv1JD3vOaFxp0= -github.com/bits-and-blooms/bitset v1.24.2/go.mod h1:7hO7Gc7Pp1vODcmWvKMRA9BNmbv6a/7QIWpPxHddWR8= +github.com/bits-and-blooms/bitset v1.24.4 h1:95H15Og1clikBrKr/DuzMXkQzECs1M6hhoGXLwLQOZE= +github.com/bits-and-blooms/bitset v1.24.4/go.mod h1:7hO7Gc7Pp1vODcmWvKMRA9BNmbv6a/7QIWpPxHddWR8= github.com/caarlos0/env/v11 v11.4.1 h1:fYwH0sWEsBSMPG7t4e/PEfTFzrWrpjyygXyUnWiSwEw= github.com/caarlos0/env/v11 v11.4.1/go.mod h1:qupehSf/Y0TUTsxKywqRt/vJjN5nz6vauiYEUUr8P4U= github.com/cenkalti/backoff/v4 v4.3.0 h1:MyRJ/UdXutAwSAT+s3wNd7MfTIcy71VQueUuFK343L8= From e1b86cabea37f116de2f01577dea8964b90bfb74 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Sat, 29 Aug 2026 13:28:07 +0000 Subject: [PATCH 06/41] events: lock the cold fetch arena against packfile fan-out MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ColdReader.FetchEvents copies every payload through one byteArena from inside the packfile.ReadItems callback, and ReadItems documents that callback as running on up to ColdReaderOptions.Concurrency goroutines. The arena is one appended buffer — its own doc says not safe for concurrent use — so any configuration with Concurrency > 1 races the copies: measured as a SIGSEGV inside byteArena.copy at 16 workers, 1458 of 3000 requests failing on a pubnet cold cell at 8, and — worst — non-deterministic silently-short results at 4-8 on sac (corrupted payloads fail the exact-match check downstream and vanish). The only value ever wired was 1, so the two contracts had never met; any change that turns on cold read concurrency for events (#772 included) would have detonated this. The lock covers the copy alone — the pread, the record decode and the Unmarshal stay outside it — and previously returned copies are safe to read unlocked because the arena never moves a chunk once handed out. Measured cost at Concurrency=1: none (control cell identical to baseline). The fan-out regression test forces an 8-way read over a multi-record fixture and exact-matches every payload, so a torn copy fails the assertion even on a run the race detector happens not to catch; with the lock removed it reports the data race. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/stores/event/arena.go | 6 ++- .../rpcv2/stores/event/cold_reader.go | 24 ++++++++--- .../stores/event/cold_reader_fanout_test.go | 41 +++++++++++++++++++ 3 files changed, 64 insertions(+), 7 deletions(-) create mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader_fanout_test.go diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go index 4e6e1c14e..60e18effc 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go @@ -4,7 +4,11 @@ package event // chunked allocations, so a fetch that copies hundreds of small payloads // costs a handful of allocations instead of one per payload. Chunks are // only ever appended within capacity, so previously returned copies never -// move. Zero value is ready; not safe for concurrent use. +// move. Zero value is ready. +// +// NOT safe for concurrent use: copy appends to one buffer. A caller whose +// copies come from several goroutines — ColdReader.FetchEvents, once its +// packfile reads fan out — must serialize them. type byteArena struct { buf []byte } diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go index 6d5e21ca0..dd51f56a9 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go @@ -450,7 +450,8 @@ func (c *ColdReader) LookupKeys(ctx context.Context, keys []TermKey) ([]*roaring // records into single ReadAt calls and optionally fans out across // the worker count set via ColdReaderOptions.Concurrency. // result[idx] writes from concurrent workers do not race — each -// idx is unique. +// idx is unique — and the payload arena they share is locked (see +// the callback). func (c *ColdReader) FetchEvents(ctx context.Context, eventIDs []uint32) ([]Payload, error) { if c.closed.Load() { return nil, stores.ErrStoreClosed @@ -474,15 +475,26 @@ func (c *ColdReader) FetchEvents(ctx context.Context, eventIDs []uint32) ([]Payl positions[i] = int(id) } results := make([]Payload, len(eventIDs)) - var arena byteArena + // One arena per call, shared by every worker: a call's payloads live and + // die together, so per-payload clones would only add GC work. The arena + // is one appended buffer and is single-threaded by construction, while + // ReadItems calls the callback from up to Concurrency goroutines — hence + // the lock. It covers the copy alone; the read, the record decode and the + // Unmarshal all stay outside it, so a fan-out still overlaps everything + // that costs. + var ( + arenaMu sync.Mutex + arena byteArena + ) if err := c.events.ReadItems(ctx, positions, func(idx int, data []byte) error { // packfile.ReadItems passes a borrowed data slice valid only for // the duration of fn (see Reader.ReadItems docstring). FetchEvents // returns the Payloads in a slice that outlives fn, so copy before - // Unmarshal aliases the bytes into ContractEventBytes — through an - // arena, since a batch's payloads live and die together and - // per-payload clones only add GC work. - return results[idx].Unmarshal(arena.copy(data)) + // Unmarshal aliases the bytes into ContractEventBytes. + arenaMu.Lock() + owned := arena.copy(data) + arenaMu.Unlock() + return results[idx].Unmarshal(owned) }); err != nil { // packfile.ReadItems also validates sorted positions as defense in // depth; translate its sentinel to ours so callers can errors.Is diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader_fanout_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader_fanout_test.go new file mode 100644 index 000000000..e4f6d8bae --- /dev/null +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader_fanout_test.go @@ -0,0 +1,41 @@ +package event + +import ( + "context" + "testing" + + "github.com/stretchr/testify/require" + + "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/chunk" +) + +// TestColdReader_FetchEventsFansOutSafely pins the payload arena against the +// reader's own fan-out. With Concurrency > 1 packfile.ReadItems splits the +// positions into batches and calls the callback from one goroutine per batch, +// so the arena every callback copies through is shared across goroutines — an +// unsynchronized append there corrupts payloads at best and segfaults at worst. +// The fixture is wide enough (many records, one position each) to force several +// batches, and every payload is checked, so a torn copy fails the assertion even +// on a run where -race happens not to schedule the overlap. +func TestColdReader_FetchEventsFansOutSafely(t *testing.T) { + const ( + chunkID = chunk.ID(0) + events = 4096 // 32 records at eventsPackItemsPerRecord + ) + dir, payloads := buildColdFixture(t, chunkID, events, 2) + + cr, err := OpenColdReader(chunkID, dir, ColdReaderOptions{Concurrency: 8}) + require.NoError(t, err) + t.Cleanup(func() { _ = cr.Close() }) + + ids := make([]uint32, 0, events) + for i := range uint32(events) { + ids = append(ids, i) + } + got, err := cr.FetchEvents(context.Background(), ids) + require.NoError(t, err) + require.Len(t, got, len(ids)) + for i := range ids { + require.Equal(t, dataSym(t, payloads[i]), dataSym(t, got[i]), "payload %d", i) + } +} From cad54c7f3772962cd4df0af07c12158ca26f5f53 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Sat, 29 Aug 2026 13:28:07 +0000 Subject: [PATCH 07/41] =?UTF-8?q?query:=20cold=20events=20reads=20fan=20ou?= =?UTF-8?q?t=20over=20the=20packfiles=20=E2=80=94=20Concurrency=3D8?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A cold getEvents page fetches hundreds of scattered ~90 µs NVMe preads with no ordering between them; ColdReaderOptions{} left packfile ReadItems at Concurrency=1, so a page paid their latencies added together — the dominant term of cold p99. Resolves the TODO(#772) at the tierCold arm with a constant: the fan-out is a property of the storage the daemon reads through, not of the query, and the seam for a deployment that wants a different bound is a Registry field, the way maxScanLedgers is one. Swept 1/4/8/16/32 on the full sac cold chunk: p99 106 / 32.2 / 21.6 / 16.6 / 14.1 ms. Eight removes ~80% of the baseline p99 and captures 92% of what 32 achieves at a quarter of its per-request footprint — the fan-out multiplies goroutines and packfile's pooled 1 MiB coalesced-read buffers by in-flight cold pages, a cost a low-rps cell cannot see, which is the argument against a higher default. Verified on a fresh seed against this exact commit: sac cold service p50 18.3 -> 8.0 ms, p99 98.7 -> 20.9; pubnet p50 5.2 -> 3.0, p99 12.9 -> 5.7; identical item counts on both sides of every cell. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/query/resolve.go | 27 ++++++++++++++++--- 1 file changed, 23 insertions(+), 4 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/query/resolve.go b/cmd/stellar-rpc/internal/rpcv2/query/resolve.go index a74b3ed06..e2e284294 100644 --- a/cmd/stellar-rpc/internal/rpcv2/query/resolve.go +++ b/cmd/stellar-rpc/internal/rpcv2/query/resolve.go @@ -144,6 +144,27 @@ func (a *ReadView) resolveLedgers(c chunk.ID) (LedgerReader, func() error, error } } +// coldEventReadConcurrency is the worker fan-out one cold events read gets over +// its packfiles (ColdReaderOptions.Concurrency → packfile ReadItems). A page's +// payload fetch is hundreds of scattered records, each its own ~90 µs NVMe pread +// after coalescing; the reads have no ordering between them, so serializing them +// only added their latencies together. A constant, not configuration: it is a +// property of the storage the daemon reads through, not of the query the client +// asked for, and the package spells that kind of bound as a constant already +// (defaultMaxScanLedgers). +// +// Why 8: swept 1/4/8/16/32 on a full cold chunk, p99 falls 106 → 32 → 21.6 → +// 16.6 → 14.1 ms — every doubling still helps, by about half as much as the one +// before. The cost side has no such curve: the fan-out is per request, so the +// worker count multiplies both goroutines and packfile's 1 MiB coalesced-read +// buffers by the number of cold pages in flight, and a 50 rps bench with a +// fraction of a request in flight cannot see that. Eight removes 79% of the +// baseline p99 and captures 92% of what 32 achieves, at a quarter of its +// per-request footprint. A +// deployment with headroom to spend can raise it; the seam is a Registry field, +// the way maxScanLedgers is one. +const coldEventReadConcurrency = 8 + // Events resolves chunk c's event store as the common event.Reader the // query engine consumes, uniform across tiers. A cold reader is view-owned — // Release closes it; the hot facade is registry-owned. Returns ErrUnavailable @@ -157,10 +178,8 @@ func (a *ReadView) Events(c chunk.ID) (event.Reader, error) { } switch t { case tierCold: - // TODO(events adapter / #772): thread read concurrency - // (ColdReaderOptions.Concurrency → the packfile ReadItems concurrency) here; - // decide whether it is config-driven or caller-supplied. Default for now. - cr, err := event.OpenColdReader(c, a.catalog.Layout().EventsBucketDir(c), event.ColdReaderOptions{}) + cr, err := event.OpenColdReader(c, a.catalog.Layout().EventsBucketDir(c), + event.ColdReaderOptions{Concurrency: coldEventReadConcurrency}) if err != nil { return nil, err } From 73259ddd5dc10c8ba0a9eeb861ddf22382f492d6 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Sat, 29 Aug 2026 13:58:26 +0000 Subject: [PATCH 08/41] =?UTF-8?q?events:=20mmap=20the=20MPHF=20=E2=80=94?= =?UTF-8?q?=20the=20page=20cache=20is=20the=20cross-request=20index=20cach?= =?UTF-8?q?e?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit openMPHF read index.hash whole into a fresh heap buffer on every open, under two assumptions its own comment carried: that the file is small (hundreds of KB) and that the read amortizes "across the index's lifetime". Neither holds. The reader's lifetime is one request — the read view opens it and Release closes it — and the file's size has no design bound: it scales with the chunk's distinct term count, and terms are caller-controlled on-chain data. A 60M-term chunk makes the file 18.5 MB, which was a whole-file read and heap copy per request — a cardinality cliff an adversary can manufacture. streamhash.Open maps the file instead. Pages fault in on first touch from the kernel page cache, which is keyed by the file rather than the mapping, so every reader of the same chunk shares the resident pages however short its own lifetime, and per-request cost stops scaling with term count at all. No size conditional: at pubnet's 73 KB the two paths differ by microseconds (measured at parity), and the regime where they differ is exactly the one ReadFile loses. A deployment on storage with expensive cold faults and page-cache pressure would want a prefault on open (madvise WILLNEED), not a heap copy. Measured on the sac cold chunk at Concurrency=8, seed-paired cells: service p50 8.0 -> 6.0 ms, p99 20.9 -> 17.1, allocations 34.4 -> 17.2 MB/request — the removed 17.2 MB is the hash read, to the megabyte. The remaining per-request open cost is the packfiles' delta-compressed offset indexes (~10.5 MB and the CPU to decode them), which cannot be used in place in the current format. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../rpcv2/stores/event/cold_format.go | 32 ++++++++++--------- 1 file changed, 17 insertions(+), 15 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go index da09a6b99..087892608 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go @@ -434,24 +434,26 @@ func buildMPHF( // /index.hash produced by an earlier buildMPHF) for // query-time lookups. // -// The file is read into memory up-front via os.ReadFile + -// streamhash.OpenBytes rather than mmapped. Rationale: a typical -// MPHF for a single Chunk is small (~hundreds of KB at production -// term counts), and on storage with expensive random IOPS (e.g. -// EBS, ~1 ms each) mmap page-faults on cold Lookups cost more than -// a single sequential read amortized across the index's lifetime. +// The file is mmapped (streamhash.Open), never read whole: pages +// fault in on first touch from the kernel page cache, which is +// keyed by the file rather than the mapping, so every reader of the +// same chunk shares the resident pages no matter how short its own +// lifetime. A per-request open therefore costs a map/unmap pair and +// the few pages its lookups touch — not a read and heap copy of the +// whole file, whose size has no design bound: it scales with the +// chunk's distinct term count, and terms are caller-controlled +// on-chain data (a 60M-term chunk makes this file 18.5 MB, and a +// whole-file read per request out of that is the cardinality cliff +// this used to be). A deployment on storage where cold faults are +// expensive AND the page cache cannot hold a serving chunk's index +// resident would want a prefault (madvise WILLNEED) on open, not a +// heap copy. // -// Close on the returned handle is a no-op for the OpenBytes path -// (streamhash holds no fd / mmap), but callers should still call it -// for symmetry with other open variants. +// Close unmaps; callers must call it. func openMPHF(path string) (*mphf, error) { - data, err := os.ReadFile(path) + idx, err := streamhash.Open(path) if err != nil { - return nil, fmt.Errorf("events: read %s: %w", path, err) - } - idx, err := streamhash.OpenBytes(data) - if err != nil { - return nil, fmt.Errorf("events: parse %s: %w", path, err) + return nil, fmt.Errorf("events: open %s: %w", path, err) } secret, merr := decodeEventsMeta(idx.UserMetadata()) if merr != nil { From 3d4c91ec15f024b89dc9bce715880e84908fff2b Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Sat, 29 Aug 2026 16:52:21 +0000 Subject: [PATCH 09/41] =?UTF-8?q?packfile:=20pool=20the=20open=20path=20?= =?UTF-8?q?=E2=80=94=20recycle=20offsets,=20decode=20scratch,=20and=20open?= =?UTF-8?q?=20buffers?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A v2 read view opens cold packfiles per request, so every open allocated the decoded offset table, its FOR-decode scratch, and the open-time read buffers afresh: ~12.4 MB per request across a chunk's two events packs at stress scale (468,750 records each), the dominant GC feed of a cold-serving leg — and a shared-process cost, since the collector's work taxes hot traffic in the same daemon. The decode CPU is a couple of milliseconds and is not worth removing; the allocation is. Pools recycle the backing memory only: contents are fully rewritten before every use, puts are capacity-capped so a pathological file cannot pin oversized arrays, and the open path now hands decodeIndex a view into the pooled read buffer instead of copying the index bytes first. The offsets array is reader-private until Close, which recycles it under a closed/inflight handshake: reads increment inflight before checking closed, Close sets closed before reading inflight — so a read racing Close (a caller contract violation that previously failed benignly on the closed fd) backs out with an os.ErrClosed-wrapped error or, if already in flight, degrades the recycle to a leak rather than memory reuse under a live reader. Tests pin byte-exact reads across pool-reuse cycles, the read-after-Close error shape, and the close-during-read no-recycle path under -race. Measured (sac stress cold cell, 50 rps, seed-paired): leg allocations 51.2 GB -> 9.6 GB (17.2 -> 3.2 MB/request), service p50 5.95 -> 4.34 ms, p99 17.1 -> 15.2, identical item counts. The residual is answer-side work (roaring postings, payload copies) plus ~1.3 MB/request of pool misses under concurrency and GC cycling. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/packfile/index.go | 9 +- .../internal/rpcv2/packfile/pools.go | 80 ++++++++++++ .../internal/rpcv2/packfile/pools_test.go | 115 ++++++++++++++++++ .../internal/rpcv2/packfile/reader.go | 63 +++++++++- 4 files changed, 260 insertions(+), 7 deletions(-) create mode 100644 cmd/stellar-rpc/internal/rpcv2/packfile/pools.go create mode 100644 cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/index.go b/cmd/stellar-rpc/internal/rpcv2/packfile/index.go index 93b641c96..874361303 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/index.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/index.go @@ -79,7 +79,8 @@ func decodeIndex(buf []byte, recordCount int, indexSize int, indexBase int64) ([ // (width, min) is at its tail, so intpack.DecodeGroup naturally reads from the end of // its window. Iterating backward lets us shrink the window after each group. groupCount := (recordCount + groupSize - 1) / groupSize - deltas := make([]uint32, recordCount) + deltas := getScratch(recordCount) + defer putScratch(deltas) pos := payloadLen for g := groupCount - 1; g >= 0; g-- { @@ -102,8 +103,10 @@ func decodeIndex(buf []byte, recordCount int, indexSize int, indexBase int64) ([ return nil, fmt.Errorf("%w: index has %d unconsumed bytes after decoding all groups", ErrCorrupt, pos) } - // Forward prefix-sum to build absolute offsets from deltas. - offsets := make([]int64, recordCount+1) + // Forward prefix-sum to build absolute offsets from deltas. The + // pooled array is fully overwritten: every entry below recordCount by + // the loop, the sentinel by the assignment after it. + offsets := getOffsets(recordCount + 1) offset := int64(0) for i, d := range deltas { diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go b/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go new file mode 100644 index 000000000..fb21dbff9 --- /dev/null +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go @@ -0,0 +1,80 @@ +package packfile + +// Open-path allocation pools. The v2 read path opens cold packfiles per +// request, so every open used to allocate the decoded offset index, its +// FOR-decode scratch, and the open-time read buffers afresh — at +// stress-chunk scale (~470K records) about 12 MB per request across a +// chunk's two events packs, the dominant GC feed of a cold-serving leg. +// The decode CPU is not the cost worth removing; the allocation is. +// These pools recycle the backing memory only: contents are always +// fully rewritten before use, and every pooled buffer is either dead +// before its function returns (scratch, open buffers) or reader-private +// until Close hands it back (the offsets). +// +// Puts are capacity-capped so one pathological file cannot pin an +// arbitrarily large array in a pool slot; the caps clear stress-scale +// artifacts (~470K records → 3.75 MB offsets) with 2× headroom, and +// larger buffers just fall to the garbage collector. + +import "sync" + +const ( + maxPooledOffsets = 1 << 20 // entries (8 MiB backing array) + maxPooledScratch = 1 << 20 // entries (4 MiB backing array) + maxPooledOpenBuf = 4 << 20 // bytes +) + +//nolint:gochecknoglobals // process-wide pools, like recordWorkspacePool +var ( + offsetsPool sync.Pool // *[]int64 + scratchPool sync.Pool // *[]uint32 + openBufPool sync.Pool // *[]byte +) + +func getOffsets(n int) []int64 { + if p, _ := offsetsPool.Get().(*[]int64); p != nil && cap(*p) >= n { + return (*p)[:n] + } + return make([]int64, n) +} + +// putOffsets recycles a decoded offset table. The caller must guarantee +// no live reference remains — see Reader.Close for the in-flight +// handshake that establishes this. +func putOffsets(s []int64) { + if cap(s) == 0 || cap(s) > maxPooledOffsets { + return + } + s = s[:0] + offsetsPool.Put(&s) +} + +func getScratch(n int) []uint32 { + if p, _ := scratchPool.Get().(*[]uint32); p != nil && cap(*p) >= n { + return (*p)[:n] + } + return make([]uint32, n) +} + +func putScratch(s []uint32) { + if cap(s) == 0 || cap(s) > maxPooledScratch { + return + } + s = s[:0] + scratchPool.Put(&s) +} + +func getOpenBuf(n int) []byte { + if p, _ := openBufPool.Get().(*[]byte); p != nil && cap(*p) >= n { + return (*p)[:n] + } + return make([]byte, n) +} + +func putOpenBuf(s []byte) { + if cap(s) == 0 || cap(s) > maxPooledOpenBuf { + return + } + s = s[:0] + openBufPool.Put(&s) +} diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go b/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go new file mode 100644 index 000000000..37fa1f9a7 --- /dev/null +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go @@ -0,0 +1,115 @@ +package packfile + +import ( + "bytes" + "context" + "errors" + "fmt" + "os" + "sync" + "testing" + + "github.com/stretchr/testify/require" +) + +// poolItems builds a deterministic item set large enough to span several +// records, so reads exercise the offset table rather than one record. +func poolItems(n int) [][]byte { + items := make([][]byte, n) + for i := range items { + items[i] = bytes.Repeat([]byte{byte(i), byte(i >> 8)}, 64) + } + return items +} + +// TestReaderPoolReuseKeepsReadsCorrect drives several full +// open-read-close cycles over distinct files through the process-wide +// pools. The second and later cycles run on recycled offset tables, +// scratch, and open buffers; byte-exact reads are the proof that a +// recycled buffer never leaks one file's decode into another's. +func TestReaderPoolReuseKeepsReadsCorrect(t *testing.T) { + for cycle := range 4 { + // Vary the item count so recycled arrays are reused at + // different lengths, covering the reslice paths. + items := poolItems(300 + 40*cycle) + path := writeTestPackfile(t, items, WriterOptions{ItemsPerRecord: 16}) + r := Open(path, ReaderOptions{}) + for i, want := range items { + require.NoError(t, r.ReadItem(i, func(got []byte) error { + if !bytes.Equal(got, want) { + return fmt.Errorf("cycle %d item %d mismatch", cycle, i) + } + return nil + })) + } + require.NoError(t, r.Close()) + } +} + +// TestReaderReadAfterCloseFails pins the handshake's fast path: a read +// beginning after Close reports a closed reader (matching os.ErrClosed, +// the shape the closed file descriptor produced before the handshake +// existed) instead of touching recycled memory. +func TestReaderReadAfterCloseFails(t *testing.T) { + items := poolItems(64) + path := writeTestPackfile(t, items, WriterOptions{ItemsPerRecord: 16}) + r := Open(path, ReaderOptions{}) + require.NoError(t, r.ReadItem(0, func([]byte) error { return nil })) + require.NoError(t, r.Close()) + + err := r.ReadItem(0, func([]byte) error { return nil }) + require.ErrorIs(t, err, os.ErrClosed) + err = r.ReadItems(context.Background(), []int{0}, func(int, []byte) error { return nil }) + require.ErrorIs(t, err, os.ErrClosed) + for _, err := range r.ReadRange(0, 1) { + require.ErrorIs(t, err, os.ErrClosed) + } +} + +// TestReaderCloseDuringReadDoesNotRecycle pins the handshake's slow +// path: Close racing an in-flight read (a caller contract violation) +// must leave the offsets to the garbage collector rather than recycle +// them under the reader. The callback parks mid-read while Close runs; +// the bytes it already received must stay intact, and the reader's +// offsets stay live for the rest of the read. Run under -race this also +// proves the handshake's ordering. +func TestReaderCloseDuringReadDoesNotRecycle(t *testing.T) { + items := poolItems(256) + path := writeTestPackfile(t, items, WriterOptions{ItemsPerRecord: 16}) + r := Open(path, ReaderOptions{}) + + parked := make(chan struct{}) + unpark := make(chan struct{}) + var closeErr error + var wg sync.WaitGroup + wg.Add(1) + go func() { + defer wg.Done() + <-parked + closeErr = r.Close() + close(unpark) + }() + + first := true + got := make([]byte, 0, len(items[0])) + err := r.ReadItems(context.Background(), []int{0}, func(_ int, data []byte) error { + got = append(got[:0], data...) + if first { + first = false + close(parked) + <-unpark + } + return nil + }) + wg.Wait() + require.NoError(t, closeErr) + // The read either completed with intact bytes or failed on the + // closed file descriptor — both are the documented outcomes of this + // contract violation; recycled-memory corruption is the one outcome + // the handshake forbids, and the byte check would catch it. + if err == nil { + require.True(t, bytes.Equal(got, items[0]), "payload corrupted by Close during read") + } else { + require.True(t, errors.Is(err, os.ErrClosed) || err != nil) + } +} diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go b/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go index 4503a6a76..16ee0b3a5 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go @@ -161,10 +161,38 @@ type Reader struct { waitOpen func() error // blocks until background open completes + // closed/inflight are the Close-vs-read handshake that lets Close + // recycle the pooled offsets. Reads increment inflight before + // checking closed; Close sets closed before reading inflight — so + // either the read sees closed and backs out, or Close sees the + // reader and leaves the offsets to the garbage collector. A read + // racing Close is a caller contract violation either way; the + // handshake turns its worst case from memory reuse into a leak. + closed atomic.Bool + inflight atomic.Int64 + closeOnce sync.Once closeErr error } +// errReaderClosed is returned by reads that begin after Close. It wraps +// os.ErrClosed so callers matching the pre-existing closed-file error +// shape keep matching. +var errReaderClosed = fmt.Errorf("packfile: read after Close: %w", os.ErrClosed) + +// beginRead registers a read with the Close handshake; endRead must run +// when the read finishes. See the closed/inflight field comment. +func (r *Reader) beginRead() error { + r.inflight.Add(1) + if r.closed.Load() { + r.inflight.Add(-1) + return errReaderClosed + } + return nil +} + +func (r *Reader) endRead() { r.inflight.Add(-1) } + // Open returns a Reader immediately. File I/O runs in a background goroutine; // the first read call blocks until the file is ready. Open itself does not // return an error — option-validation and I/O failures are deferred to the @@ -235,7 +263,8 @@ func doOpen(path string) openResult { // Speculative read: last min(speculativeReadSize, fileSize) bytes. speculativeSize := min(int64(speculativeReadSize), fileSize) speculativeOff := fileSize - speculativeSize - speculativeBuf := make([]byte, speculativeSize) + speculativeBuf := getOpenBuf(int(speculativeSize)) + defer putOpenBuf(speculativeBuf) if _, err := f.ReadAt(speculativeBuf, speculativeOff); err != nil { return openResult{err: fmt.Errorf("packfile: read trailer region: %w", err)} } @@ -262,11 +291,13 @@ func doOpen(path string) openResult { var indexBuf []byte var appData []byte + // Both arms hand decodeIndex a view into a pooled buffer: the index + // bytes are dead once the offsets are built (decodeIndex retains + // nothing), so only appData — which the Reader keeps — is copied out. if tailSize <= speculativeSize { // Index + appData are already inside the speculative read. tailStart := len(speculativeBuf) - int(tailSize) - indexBuf = make([]byte, indexSize) - copy(indexBuf, speculativeBuf[tailStart:tailStart+indexSize]) + indexBuf = speculativeBuf[tailStart : tailStart+indexSize] if appDataSize > 0 { appData = make([]byte, appDataSize) adStart := tailStart + indexSize @@ -275,7 +306,8 @@ func doOpen(path string) openResult { } else { // Single fallback read for index + appData. readSize := indexSize + appDataSize - buf := make([]byte, readSize) + buf := getOpenBuf(readSize) + defer putOpenBuf(buf) if readSize > 0 { if _, err := f.ReadAt(buf, indexBase); err != nil { return openResult{err: fmt.Errorf("packfile: read index region: %w", err)} @@ -417,6 +449,10 @@ func (r *Reader) ReadItem(position int, fn func([]byte) error) error { if err := r.waitOpen(); err != nil { return err } + if err := r.beginRead(); err != nil { + return err + } + defer r.endRead() if position < 0 || position >= r.totalItems { return ErrPositionOutOfRange } @@ -469,6 +505,11 @@ func (r *Reader) ReadRange(start, count int) iter.Seq2[[]byte, error] { yield(nil, err) return } + if err := r.beginRead(); err != nil { + yield(nil, err) + return + } + defer r.endRead() if start < 0 || count < 0 || start > r.totalItems || count > r.totalItems-start { yield(nil, fmt.Errorf("%w: ReadRange(%d, %d) out of [0, %d)", ErrPositionOutOfRange, start, count, r.totalItems)) @@ -577,6 +618,10 @@ func (r *Reader) ReadItems(ctx context.Context, positions []int, fn func(idx int if err := r.waitOpen(); err != nil { return err } + if err := r.beginRead(); err != nil { + return err + } + defer r.endRead() for i, pos := range positions { if pos < 0 || pos >= r.totalItems { @@ -768,11 +813,21 @@ func (r *Reader) Verify(ctx context.Context) error { // Readers); its lifecycle is the caller's responsibility. func (r *Reader) Close() error { r.closeOnce.Do(func() { + r.closed.Store(true) openErr := r.waitOpen() var closeErr error if r.file != nil { closeErr = r.file.Close() } + // Recycle the decoded offsets only when no read is in flight: + // closed is already set, so no new read can begin, and a zero + // inflight count proves no existing one holds the array. A read + // still in flight here is a contract violation; leaving the + // array to the garbage collector keeps it merely a leak. + if r.inflight.Load() == 0 && r.offsets != nil { + putOffsets(r.offsets) + r.offsets = nil + } r.closeErr = errors.Join(openErr, closeErr) }) return r.closeErr From 8f470ef9046dab9724e07aa00cf23238415675bf Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Sun, 30 Aug 2026 04:16:49 +0000 Subject: [PATCH 10/41] review: the cold fan-out seam is a real Registry field; two test gaps MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The fan-out constant's comment pointed at a Registry seam that did not exist; now it does, mirroring maxScanLedgers — zero means the default, copied onto the view at acquisition so one request's fan-out cannot change mid-flight, settable per deployment or per test without a rebuild. batchSizes gains the test its 8x hint cap never had, pinned against the seam so a shrunk batch size scales the cap with it. The close-during-read test's error arm now insists on os.ErrClosed instead of accepting any error. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/packfile/pools_test.go | 3 +- .../internal/rpcv2/query/registry.go | 28 +++++++++++---- .../internal/rpcv2/query/resolve.go | 16 +++++---- .../rpcv2/stores/event/match_iter_test.go | 34 +++++++++++++++++++ 4 files changed, 66 insertions(+), 15 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go b/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go index 37fa1f9a7..09c8124ff 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go @@ -3,7 +3,6 @@ package packfile import ( "bytes" "context" - "errors" "fmt" "os" "sync" @@ -110,6 +109,6 @@ func TestReaderCloseDuringReadDoesNotRecycle(t *testing.T) { if err == nil { require.True(t, bytes.Equal(got, items[0]), "payload corrupted by Close during read") } else { - require.True(t, errors.Is(err, os.ErrClosed) || err != nil) + require.ErrorIs(t, err, os.ErrClosed) } } diff --git a/cmd/stellar-rpc/internal/rpcv2/query/registry.go b/cmd/stellar-rpc/internal/rpcv2/query/registry.go index 8a951f7cf..26fecd41c 100644 --- a/cmd/stellar-rpc/internal/rpcv2/query/registry.go +++ b/cmd/stellar-rpc/internal/rpcv2/query/registry.go @@ -36,6 +36,14 @@ type Registry struct { // cannot leak into another test's pages. maxScanLedgers uint32 + // coldEventReadConcurrency overrides the packfile read fan-out one cold + // events read gets. Zero means defaultColdEventReadConcurrency, resolved + // where the reader is opened (ReadView.Events). Held here, like + // maxScanLedgers, so a deployment with I/O headroom to spend — or a test + // pinning the serialized path — sets it on its Registry rather than + // rebuilding. + coldEventReadConcurrency int + // latest is the newest fully ingested ledger visible to queries, paired with // its close time so both publish atomically. The ingest loop advances it as // the final step of each per-ledger cycle. Queries read a frozen copy @@ -288,6 +296,11 @@ type ReadView struct { // so a view built without it still bounds its pages. maxScanLedgers uint32 + // coldEventReadConcurrency is the registry's cold events fan-out, copied + // at acquisition like maxScanLedgers. Zero means + // defaultColdEventReadConcurrency. + coldEventReadConcurrency int + // closers releases every cold reader this view opened (hot facades are // registry-owned and never appear here). Appended by the resolve methods and // by ScanLedgers' walk backstop, drained by Release — a view's resources live @@ -323,13 +336,14 @@ func (r *Registry) NewReadView() (*ReadView, error) { return nil, err } view := &ReadView{ - latest: *latest, - maxScanLedgers: r.maxScanLedgers, - floor: r.retention.FloorAt(lastComplete), - handles: handles, - snap: snap, - catalog: r.catalog, - oldest: &r.oldest, + latest: *latest, + maxScanLedgers: r.maxScanLedgers, + coldEventReadConcurrency: r.coldEventReadConcurrency, + floor: r.retention.FloorAt(lastComplete), + handles: handles, + snap: snap, + catalog: r.catalog, + oldest: &r.oldest, } // The oldest-close-time cache rides along outside the three-load order: it // is a pure optimization whose staleness the seq check in OldestCloseTime diff --git a/cmd/stellar-rpc/internal/rpcv2/query/resolve.go b/cmd/stellar-rpc/internal/rpcv2/query/resolve.go index e2e284294..3f20e4359 100644 --- a/cmd/stellar-rpc/internal/rpcv2/query/resolve.go +++ b/cmd/stellar-rpc/internal/rpcv2/query/resolve.go @@ -144,7 +144,7 @@ func (a *ReadView) resolveLedgers(c chunk.ID) (LedgerReader, func() error, error } } -// coldEventReadConcurrency is the worker fan-out one cold events read gets over +// defaultColdEventReadConcurrency is the worker fan-out one cold events read gets over // its packfiles (ColdReaderOptions.Concurrency → packfile ReadItems). A page's // payload fetch is hundreds of scattered records, each its own ~90 µs NVMe pread // after coalescing; the reads have no ordering between them, so serializing them @@ -160,10 +160,10 @@ func (a *ReadView) resolveLedgers(c chunk.ID) (LedgerReader, func() error, error // buffers by the number of cold pages in flight, and a 50 rps bench with a // fraction of a request in flight cannot see that. Eight removes 79% of the // baseline p99 and captures 92% of what 32 achieves, at a quarter of its -// per-request footprint. A -// deployment with headroom to spend can raise it; the seam is a Registry field, -// the way maxScanLedgers is one. -const coldEventReadConcurrency = 8 +// per-request footprint. A deployment with headroom to spend raises it via +// the Registry's coldEventReadConcurrency, the way maxScanLedgers is set; +// zero means this default. +const defaultColdEventReadConcurrency = 8 // Events resolves chunk c's event store as the common event.Reader the // query engine consumes, uniform across tiers. A cold reader is view-owned — @@ -178,8 +178,12 @@ func (a *ReadView) Events(c chunk.ID) (event.Reader, error) { } switch t { case tierCold: + conc := a.coldEventReadConcurrency + if conc == 0 { + conc = defaultColdEventReadConcurrency + } cr, err := event.OpenColdReader(c, a.catalog.Layout().EventsBucketDir(c), - event.ColdReaderOptions{Concurrency: coldEventReadConcurrency}) + event.ColdReaderOptions{Concurrency: conc}) if err != nil { return nil, err } diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go index 6636eeeb7..cb24eb014 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go @@ -1227,3 +1227,37 @@ func TestBenchShapesAgree(t *testing.T) { }) } } + +// TestBatchSizes pins the first-batch hint contract: a positive hint +// sizes the first fetch (a page-sized request is one round trip), a +// wild hint is capped at eight default batches, and later batches +// always use the default. The seam interplay matters: the cap scales +// with matchBatchSize so a test-shrunk batch size cannot be blown past +// by a hint. +func TestBatchSizes(t *testing.T) { + first, rest := batchSizes(0) + require.Equal(t, matchBatchSize, first) + require.Equal(t, matchBatchSize, rest) + + first, rest = batchSizes(-3) + require.Equal(t, matchBatchSize, first) + require.Equal(t, matchBatchSize, rest) + + first, rest = batchSizes(7) + require.Equal(t, 7, first) + require.Equal(t, matchBatchSize, rest) + + first, rest = batchSizes(1000) + require.Equal(t, 1000, first, "a page-sized hint is the first fetch size") + require.Equal(t, matchBatchSize, rest) + + first, rest = batchSizes(1 << 20) + require.Equal(t, 8*matchBatchSize, first, "oversized hints are capped") + require.Equal(t, matchBatchSize, rest) + + defer func(n int) { matchBatchSize = n }(matchBatchSize) + matchBatchSize = 7 + first, rest = batchSizes(1000) + require.Equal(t, 56, first, "the cap follows the seam") + require.Equal(t, 7, rest) +} From 14e335ec3bb85e98a8155c253db7ef507f51d88f Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Sun, 30 Aug 2026 04:28:59 +0000 Subject: [PATCH 11/41] =?UTF-8?q?review:=20Matches=20doc=20=E2=80=94=20the?= =?UTF-8?q?=20hint=20cap,=20not=20the=20default,=20handles=20oversized=20f?= =?UTF-8?q?irstBatch?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- cmd/stellar-rpc/internal/rpcv2/stores/event/match.go | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go index 9d8f28d58..33fc6791c 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go @@ -263,8 +263,10 @@ func batchSizes(hint int) (int, int) { // // firstBatch sizes the first internal fetch batch. A consumer that // will stop after N matches passes N, so the first round trip fetches -// no more than it needs; 0 or any out-of-range value uses the -// default. Later batches use the default size. The hint changes I/O +// no more than it needs; zero and negative hints use the default, and +// a positive hint is honored up to eight default batches (a page-sized +// request is one round trip; a wild hint cannot demand an unbounded +// fetch). Later batches use the default size. The hint changes I/O // counts only, never what the stream yields. func Matches( ctx context.Context, r Reader, filters []Filter, window IDRange, From 5d7d9d2f20e41e8fcd1795ffa4992da136823f8e Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Sun, 30 Aug 2026 05:25:52 +0000 Subject: [PATCH 12/41] =?UTF-8?q?lint:=20CI=20findings=20=E2=80=94=20WaitG?= =?UTF-8?q?roup.Go,=20funcorder=20on=20the=20read=20handshake,=20arena=5Ft?= =?UTF-8?q?est=20trio?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/packfile/pools_test.go | 6 ++-- .../internal/rpcv2/packfile/reader.go | 36 +++++++++---------- .../internal/rpcv2/stores/event/arena_test.go | 7 ++-- 3 files changed, 23 insertions(+), 26 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go b/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go index 09c8124ff..9ec6ec348 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go @@ -81,13 +81,11 @@ func TestReaderCloseDuringReadDoesNotRecycle(t *testing.T) { unpark := make(chan struct{}) var closeErr error var wg sync.WaitGroup - wg.Add(1) - go func() { - defer wg.Done() + wg.Go(func() { <-parked closeErr = r.Close() close(unpark) - }() + }) first := true got := make([]byte, 0, len(items[0])) diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go b/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go index 16ee0b3a5..461b95b56 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go @@ -175,24 +175,6 @@ type Reader struct { closeErr error } -// errReaderClosed is returned by reads that begin after Close. It wraps -// os.ErrClosed so callers matching the pre-existing closed-file error -// shape keep matching. -var errReaderClosed = fmt.Errorf("packfile: read after Close: %w", os.ErrClosed) - -// beginRead registers a read with the Close handshake; endRead must run -// when the read finishes. See the closed/inflight field comment. -func (r *Reader) beginRead() error { - r.inflight.Add(1) - if r.closed.Load() { - r.inflight.Add(-1) - return errReaderClosed - } - return nil -} - -func (r *Reader) endRead() { r.inflight.Add(-1) } - // Open returns a Reader immediately. File I/O runs in a background goroutine; // the first read call blocks until the file is ready. Open itself does not // return an error — option-validation and I/O failures are deferred to the @@ -832,3 +814,21 @@ func (r *Reader) Close() error { }) return r.closeErr } + +// errReaderClosed is returned by reads that begin after Close. It wraps +// os.ErrClosed so callers matching the pre-existing closed-file error +// shape keep matching. +var errReaderClosed = fmt.Errorf("packfile: read after Close: %w", os.ErrClosed) + +// beginRead registers a read with the Close handshake; endRead must run +// when the read finishes. See the closed/inflight field comment. +func (r *Reader) beginRead() error { + r.inflight.Add(1) + if r.closed.Load() { + r.inflight.Add(-1) + return errReaderClosed + } + return nil +} + +func (r *Reader) endRead() { r.inflight.Add(-1) } diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go index b37523579..8b53d1f3a 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go @@ -2,7 +2,6 @@ package event import ( "bytes" - "fmt" "testing" "github.com/stretchr/testify/require" @@ -16,7 +15,7 @@ func TestByteArenaCopiesAreStable(t *testing.T) { src := make([]byte, 300) var got [][]byte var want [][]byte - for i := 0; i < 3000; i++ { // ~900KB total: crosses many 64KB chunks + for i := range 3000 { // ~900KB total: crosses many 64KB chunks for j := range src { src[j] = byte(i + j) } @@ -30,12 +29,12 @@ func TestByteArenaCopiesAreStable(t *testing.T) { require.Equal(t, byte(0xAB), hugeCopy[0]) require.Len(t, hugeCopy, 3*arenaChunkSize) for i := range got { - require.True(t, bytes.Equal(got[i], want[i]), fmt.Sprintf("copy %d changed", i)) + require.True(t, bytes.Equal(got[i], want[i]), "copy %d changed", i) } // Appending to a returned copy must not scribble into the arena: the // three-index slice pins capacity to length. c := a.copy([]byte{1, 2, 3}) next := a.copy([]byte{9, 9, 9}) - _ = append(c, 7) //nolint:staticcheck // the append must copy, not extend in place + _ = append(c, 7) // the append must copy, not extend in place require.Equal(t, []byte{9, 9, 9}, next) } From 0c8bfc9ba0ce1c8fa8dd06a6a1ddbdcd46cb16d5 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Sun, 30 Aug 2026 05:35:14 +0000 Subject: [PATCH 13/41] lint: prealloc the arena test's expectation slices Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go index 8b53d1f3a..cf26aaea0 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go @@ -13,8 +13,8 @@ import ( func TestByteArenaCopiesAreStable(t *testing.T) { var a byteArena src := make([]byte, 300) - var got [][]byte - var want [][]byte + got := make([][]byte, 0, 3000) + want := make([][]byte, 0, 3000) for i := range 3000 { // ~900KB total: crosses many 64KB chunks for j := range src { src[j] = byte(i + j) From 484379f154bef3dc16837e89cdeb8f3113cfcbb3 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Sun, 30 Aug 2026 05:44:46 +0000 Subject: [PATCH 14/41] review: the fan-out knob is deployment-reachable; the arena's first chunk ramps MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit SetColdEventReadConcurrency makes the Registry seam the fan-out constant's doc points at real — unexported-field-with-no-path promised a knob nothing could turn. The cold fetch arena's chunks now ramp from 4 KiB, doubling to the 64 KiB unit, so a limit=1 page copying a few hundred bytes no longer allocates 64 KiB for them while a 512-candidate fetch still lands in a handful of chunks. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/query/registry.go | 15 +++++++++++---- .../internal/rpcv2/stores/event/arena.go | 19 ++++++++++++++----- 2 files changed, 25 insertions(+), 9 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/query/registry.go b/cmd/stellar-rpc/internal/rpcv2/query/registry.go index 26fecd41c..4e8a5930f 100644 --- a/cmd/stellar-rpc/internal/rpcv2/query/registry.go +++ b/cmd/stellar-rpc/internal/rpcv2/query/registry.go @@ -38,10 +38,9 @@ type Registry struct { // coldEventReadConcurrency overrides the packfile read fan-out one cold // events read gets. Zero means defaultColdEventReadConcurrency, resolved - // where the reader is opened (ReadView.Events). Held here, like - // maxScanLedgers, so a deployment with I/O headroom to spend — or a test - // pinning the serialized path — sets it on its Registry rather than - // rebuilding. + // where the reader is opened (ReadView.Events). Set through + // SetColdEventReadConcurrency — the deployment-reachable knob the fan-out + // constant's doc points at — or left zero for the swept default. coldEventReadConcurrency int // latest is the newest fully ingested ledger visible to queries, paired with @@ -166,6 +165,14 @@ func NewRegistry(cat *catalog.Catalog, retention geometry.Retention) *Registry { return r } +// SetColdEventReadConcurrency overrides the packfile read fan-out cold +// events reads get; zero restores defaultColdEventReadConcurrency. Call it +// at wiring time, before views are handed out: views copy the value at +// acquisition, so a change never shifts an in-flight request's fan-out. +func (r *Registry) SetColdEventReadConcurrency(n int) { + r.coldEventReadConcurrency = n +} + // SetLatestLedger publishes seq together with its close time, which callers // without one spell UnknownCloseTime(). func (r *Registry) SetLatestLedger(seq uint32, closeTime CloseTime) { diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go index 60e18effc..3dcd89888 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go @@ -13,14 +13,23 @@ type byteArena struct { buf []byte } -// arenaChunkSize is the arena's allocation unit. Big enough that a -// 512-candidate fetch of ~250B payloads fits in one or two chunks, small -// enough that a mostly-idle arena wastes little. -const arenaChunkSize = 64 << 10 +// The arena's allocation unit ramps: the first chunk is small so a +// limit=1 page fetching a few hundred bytes does not pay 64 KiB for +// them, and each subsequent chunk doubles up to arenaChunkSize so a +// 512-candidate fetch of ~250B payloads still lands in a handful of +// allocations. +const ( + arenaFirstChunkSize = 4 << 10 + arenaChunkSize = 64 << 10 +) func (a *byteArena) copy(b []byte) []byte { if len(b) > cap(a.buf)-len(a.buf) { - a.buf = make([]byte, 0, max(arenaChunkSize, len(b))) + next := arenaFirstChunkSize + if c := 2 * cap(a.buf); c > next { + next = min(c, arenaChunkSize) + } + a.buf = make([]byte, 0, max(next, len(b))) } n := len(a.buf) a.buf = append(a.buf, b...) From 78931d0da63239d1924bdea8d865342b198bf97d Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Mon, 31 Aug 2026 14:00:21 +0000 Subject: [PATCH 15/41] events: trim this branch's comments to contracts Measurements and rationale live in the PR body and commit messages; what stays is what a maintainer needs. PR-added comment lines 746 -> 474. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/packfile/index.go | 6 +- .../internal/rpcv2/packfile/pools.go | 25 +- .../internal/rpcv2/packfile/pools_test.go | 39 +-- .../internal/rpcv2/packfile/reader.go | 36 ++- .../internal/rpcv2/query/registry.go | 17 +- .../internal/rpcv2/query/resolve.go | 28 +- .../internal/rpcv2/rocksdb/rocksdb.go | 8 +- .../internal/rpcv2/stores/event/arena.go | 19 +- .../internal/rpcv2/stores/event/arena_test.go | 7 +- .../rpcv2/stores/event/cold_format.go | 20 +- .../rpcv2/stores/event/cold_reader.go | 12 +- .../stores/event/cold_reader_fanout_test.go | 13 +- .../rpcv2/stores/event/concurrent_bitmaps.go | 92 +++--- .../internal/rpcv2/stores/event/hot_store.go | 38 +-- .../internal/rpcv2/stores/event/match.go | 119 +++----- .../internal/rpcv2/stores/event/match_iter.go | 283 +++++++----------- .../rpcv2/stores/event/match_iter_test.go | 204 +++++-------- .../stores/event/matches_differential_test.go | 59 ++-- 18 files changed, 371 insertions(+), 654 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/index.go b/cmd/stellar-rpc/internal/rpcv2/packfile/index.go index 874361303..d3983f318 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/index.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/index.go @@ -103,9 +103,9 @@ func decodeIndex(buf []byte, recordCount int, indexSize int, indexBase int64) ([ return nil, fmt.Errorf("%w: index has %d unconsumed bytes after decoding all groups", ErrCorrupt, pos) } - // Forward prefix-sum to build absolute offsets from deltas. The - // pooled array is fully overwritten: every entry below recordCount by - // the loop, the sentinel by the assignment after it. + // Forward prefix-sum to build absolute offsets from deltas. The pooled + // array is fully overwritten: the entries by the loop, the sentinel by + // the assignment after it. offsets := getOffsets(recordCount + 1) offset := int64(0) diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go b/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go index fb21dbff9..e2dbc90f5 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go @@ -1,20 +1,14 @@ package packfile // Open-path allocation pools. The v2 read path opens cold packfiles per -// request, so every open used to allocate the decoded offset index, its -// FOR-decode scratch, and the open-time read buffers afresh — at -// stress-chunk scale (~470K records) about 12 MB per request across a -// chunk's two events packs, the dominant GC feed of a cold-serving leg. -// The decode CPU is not the cost worth removing; the allocation is. -// These pools recycle the backing memory only: contents are always -// fully rewritten before use, and every pooled buffer is either dead -// before its function returns (scratch, open buffers) or reader-private -// until Close hands it back (the offsets). +// request, and every open allocates the decoded offset index, its FOR-decode +// scratch and the open-time read buffers. These pools recycle the backing +// memory only: contents are always fully rewritten before use, and every +// pooled buffer is either dead before its function returns or reader-private +// until Close hands it back. // -// Puts are capacity-capped so one pathological file cannot pin an -// arbitrarily large array in a pool slot; the caps clear stress-scale -// artifacts (~470K records → 3.75 MB offsets) with 2× headroom, and -// larger buffers just fall to the garbage collector. +// Puts are capacity-capped so one pathological file cannot pin an arbitrarily +// large array in a pool slot; larger buffers fall to the garbage collector. import "sync" @@ -38,9 +32,8 @@ func getOffsets(n int) []int64 { return make([]int64, n) } -// putOffsets recycles a decoded offset table. The caller must guarantee -// no live reference remains — see Reader.Close for the in-flight -// handshake that establishes this. +// putOffsets recycles a decoded offset table. The caller must guarantee no +// live reference remains; see Reader.Close for the in-flight handshake. func putOffsets(s []int64) { if cap(s) == 0 || cap(s) > maxPooledOffsets { return diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go b/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go index 9ec6ec348..b0e418792 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go @@ -11,8 +11,7 @@ import ( "github.com/stretchr/testify/require" ) -// poolItems builds a deterministic item set large enough to span several -// records, so reads exercise the offset table rather than one record. +// poolItems builds a deterministic item set spanning several records. func poolItems(n int) [][]byte { items := make([][]byte, n) for i := range items { @@ -21,15 +20,13 @@ func poolItems(n int) [][]byte { return items } -// TestReaderPoolReuseKeepsReadsCorrect drives several full -// open-read-close cycles over distinct files through the process-wide -// pools. The second and later cycles run on recycled offset tables, -// scratch, and open buffers; byte-exact reads are the proof that a -// recycled buffer never leaks one file's decode into another's. +// Drives several open-read-close cycles over distinct files through the +// process-wide pools; byte-exact reads prove a recycled buffer never leaks one +// file's decode into another's. func TestReaderPoolReuseKeepsReadsCorrect(t *testing.T) { for cycle := range 4 { - // Vary the item count so recycled arrays are reused at - // different lengths, covering the reslice paths. + // Vary the item count so recycled arrays are reused at different + // lengths, covering the reslice paths. items := poolItems(300 + 40*cycle) path := writeTestPackfile(t, items, WriterOptions{ItemsPerRecord: 16}) r := Open(path, ReaderOptions{}) @@ -45,10 +42,8 @@ func TestReaderPoolReuseKeepsReadsCorrect(t *testing.T) { } } -// TestReaderReadAfterCloseFails pins the handshake's fast path: a read -// beginning after Close reports a closed reader (matching os.ErrClosed, -// the shape the closed file descriptor produced before the handshake -// existed) instead of touching recycled memory. +// The handshake's fast path: a read beginning after Close reports a closed +// reader, matching os.ErrClosed, instead of touching recycled memory. func TestReaderReadAfterCloseFails(t *testing.T) { items := poolItems(64) path := writeTestPackfile(t, items, WriterOptions{ItemsPerRecord: 16}) @@ -65,13 +60,10 @@ func TestReaderReadAfterCloseFails(t *testing.T) { } } -// TestReaderCloseDuringReadDoesNotRecycle pins the handshake's slow -// path: Close racing an in-flight read (a caller contract violation) -// must leave the offsets to the garbage collector rather than recycle -// them under the reader. The callback parks mid-read while Close runs; -// the bytes it already received must stay intact, and the reader's -// offsets stay live for the rest of the read. Run under -race this also -// proves the handshake's ordering. +// The handshake's slow path: Close racing an in-flight read, which is a caller +// contract violation, must leave the offsets to the garbage collector rather +// than recycle them under the reader. Under -race this also proves the +// ordering. func TestReaderCloseDuringReadDoesNotRecycle(t *testing.T) { items := poolItems(256) path := writeTestPackfile(t, items, WriterOptions{ItemsPerRecord: 16}) @@ -100,10 +92,9 @@ func TestReaderCloseDuringReadDoesNotRecycle(t *testing.T) { }) wg.Wait() require.NoError(t, closeErr) - // The read either completed with intact bytes or failed on the - // closed file descriptor — both are the documented outcomes of this - // contract violation; recycled-memory corruption is the one outcome - // the handshake forbids, and the byte check would catch it. + // Either outcome is documented for this contract violation. What the + // handshake forbids is recycled-memory corruption, which the byte + // check would catch. if err == nil { require.True(t, bytes.Equal(got, items[0]), "payload corrupted by Close during read") } else { diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go b/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go index 461b95b56..4079a2877 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go @@ -161,13 +161,13 @@ type Reader struct { waitOpen func() error // blocks until background open completes - // closed/inflight are the Close-vs-read handshake that lets Close - // recycle the pooled offsets. Reads increment inflight before - // checking closed; Close sets closed before reading inflight — so - // either the read sees closed and backs out, or Close sees the - // reader and leaves the offsets to the garbage collector. A read - // racing Close is a caller contract violation either way; the - // handshake turns its worst case from memory reuse into a leak. + // closed and inflight are the Close-vs-read handshake that lets Close + // recycle the pooled offsets. Reads increment inflight before checking + // closed; Close sets closed before reading inflight. Either the read + // sees closed and backs out, or Close sees the reader and leaves the + // offsets to the garbage collector. A read racing Close is a caller + // contract violation either way; this turns its worst case from memory + // reuse into a leak. closed atomic.Bool inflight atomic.Int64 @@ -273,9 +273,9 @@ func doOpen(path string) openResult { var indexBuf []byte var appData []byte - // Both arms hand decodeIndex a view into a pooled buffer: the index - // bytes are dead once the offsets are built (decodeIndex retains - // nothing), so only appData — which the Reader keeps — is copied out. + // Both arms hand decodeIndex a view into a pooled buffer. The index bytes + // are dead once the offsets are built, so only appData, which the Reader + // keeps, is copied out. if tailSize <= speculativeSize { // Index + appData are already inside the speculative read. tailStart := len(speculativeBuf) - int(tailSize) @@ -801,11 +801,10 @@ func (r *Reader) Close() error { if r.file != nil { closeErr = r.file.Close() } - // Recycle the decoded offsets only when no read is in flight: - // closed is already set, so no new read can begin, and a zero - // inflight count proves no existing one holds the array. A read - // still in flight here is a contract violation; leaving the - // array to the garbage collector keeps it merely a leak. + // Recycle only when no read is in flight: closed is already set, so + // no new read can begin, and a zero count proves no existing one + // holds the array. A read still in flight is a contract violation; + // leaving the array to the collector keeps it merely a leak. if r.inflight.Load() == 0 && r.offsets != nil { putOffsets(r.offsets) r.offsets = nil @@ -816,12 +815,11 @@ func (r *Reader) Close() error { } // errReaderClosed is returned by reads that begin after Close. It wraps -// os.ErrClosed so callers matching the pre-existing closed-file error -// shape keep matching. +// os.ErrClosed so callers matching the closed-file error shape keep matching. var errReaderClosed = fmt.Errorf("packfile: read after Close: %w", os.ErrClosed) -// beginRead registers a read with the Close handshake; endRead must run -// when the read finishes. See the closed/inflight field comment. +// beginRead registers a read with the Close handshake; endRead must run when +// the read finishes. See the closed and inflight field comment. func (r *Reader) beginRead() error { r.inflight.Add(1) if r.closed.Load() { diff --git a/cmd/stellar-rpc/internal/rpcv2/query/registry.go b/cmd/stellar-rpc/internal/rpcv2/query/registry.go index 4e8a5930f..633de5726 100644 --- a/cmd/stellar-rpc/internal/rpcv2/query/registry.go +++ b/cmd/stellar-rpc/internal/rpcv2/query/registry.go @@ -38,9 +38,7 @@ type Registry struct { // coldEventReadConcurrency overrides the packfile read fan-out one cold // events read gets. Zero means defaultColdEventReadConcurrency, resolved - // where the reader is opened (ReadView.Events). Set through - // SetColdEventReadConcurrency — the deployment-reachable knob the fan-out - // constant's doc points at — or left zero for the swept default. + // where the reader is opened. Set through SetColdEventReadConcurrency. coldEventReadConcurrency int // latest is the newest fully ingested ledger visible to queries, paired with @@ -165,10 +163,10 @@ func NewRegistry(cat *catalog.Catalog, retention geometry.Retention) *Registry { return r } -// SetColdEventReadConcurrency overrides the packfile read fan-out cold -// events reads get; zero restores defaultColdEventReadConcurrency. Call it -// at wiring time, before views are handed out: views copy the value at -// acquisition, so a change never shifts an in-flight request's fan-out. +// SetColdEventReadConcurrency overrides the packfile read fan-out cold events +// reads get; zero restores the default. Call it at wiring time, before views +// are handed out: a view copies the value at acquisition, so a change never +// shifts an in-flight request's fan-out. func (r *Registry) SetColdEventReadConcurrency(n int) { r.coldEventReadConcurrency = n } @@ -303,9 +301,8 @@ type ReadView struct { // so a view built without it still bounds its pages. maxScanLedgers uint32 - // coldEventReadConcurrency is the registry's cold events fan-out, copied - // at acquisition like maxScanLedgers. Zero means - // defaultColdEventReadConcurrency. + // coldEventReadConcurrency is the registry's cold events fan-out, copied at + // acquisition like maxScanLedgers. Zero means the default. coldEventReadConcurrency int // closers releases every cold reader this view opened (hot facades are diff --git a/cmd/stellar-rpc/internal/rpcv2/query/resolve.go b/cmd/stellar-rpc/internal/rpcv2/query/resolve.go index 3f20e4359..b973c1f31 100644 --- a/cmd/stellar-rpc/internal/rpcv2/query/resolve.go +++ b/cmd/stellar-rpc/internal/rpcv2/query/resolve.go @@ -144,25 +144,17 @@ func (a *ReadView) resolveLedgers(c chunk.ID) (LedgerReader, func() error, error } } -// defaultColdEventReadConcurrency is the worker fan-out one cold events read gets over -// its packfiles (ColdReaderOptions.Concurrency → packfile ReadItems). A page's -// payload fetch is hundreds of scattered records, each its own ~90 µs NVMe pread -// after coalescing; the reads have no ordering between them, so serializing them -// only added their latencies together. A constant, not configuration: it is a -// property of the storage the daemon reads through, not of the query the client -// asked for, and the package spells that kind of bound as a constant already -// (defaultMaxScanLedgers). +// defaultColdEventReadConcurrency is the worker fan-out one cold events read +// gets over its packfiles. A page's payload fetch is hundreds of scattered +// records with no ordering between them, so serializing them only added their +// latencies together. A constant rather than configuration: it is a property +// of the storage the daemon reads through, not of the query the client asked +// for, which is how the package already spells defaultMaxScanLedgers. // -// Why 8: swept 1/4/8/16/32 on a full cold chunk, p99 falls 106 → 32 → 21.6 → -// 16.6 → 14.1 ms — every doubling still helps, by about half as much as the one -// before. The cost side has no such curve: the fan-out is per request, so the -// worker count multiplies both goroutines and packfile's 1 MiB coalesced-read -// buffers by the number of cold pages in flight, and a 50 rps bench with a -// fraction of a request in flight cannot see that. Eight removes 79% of the -// baseline p99 and captures 92% of what 32 achieves, at a quarter of its -// per-request footprint. A deployment with headroom to spend raises it via -// the Registry's coldEventReadConcurrency, the way maxScanLedgers is set; -// zero means this default. +// The fan-out is per request, so the worker count multiplies both goroutines +// and packfile's coalesced-read buffers by the number of cold pages in flight. +// A deployment with headroom raises it through the Registry's +// coldEventReadConcurrency; zero means this default. const defaultColdEventReadConcurrency = 8 // Events resolves chunk c's event store as the common event.Reader the diff --git a/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go b/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go index ea1e1e703..1b9acbe6b 100644 --- a/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go +++ b/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go @@ -275,11 +275,9 @@ func (s *Store) BatchMultiGet(cf string, keys [][]byte) ([][]byte, error) { } defer pinned.Destroy() - // Copy out of the pinned cache pages (Destroy invalidates them) — - // through one arena, not a clone per value: a batch is fetched, - // decoded, and dropped as a unit, so per-value allocations only add - // GC work. The returned slices therefore share one backing array: - // they are read-only, and retaining any of them retains the batch. + // Copy out of the pinned cache pages, which Destroy invalidates, through + // one arena rather than a clone per value. The returned slices share one + // backing array: they are read-only, and retaining one retains the batch. total := 0 for _, p := range pinned { total += len(p.Data()) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go index 3dcd89888..df9ae8851 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go @@ -1,23 +1,18 @@ package event // byteArena hands out stable copies of transient byte slices from large -// chunked allocations, so a fetch that copies hundreds of small payloads -// costs a handful of allocations instead of one per payload. Chunks are -// only ever appended within capacity, so previously returned copies never -// move. Zero value is ready. +// chunked allocations, so a fetch copying hundreds of small payloads costs a +// handful of allocations rather than one each. Chunks are only appended within +// capacity, so previously returned copies never move. Zero value is ready. // -// NOT safe for concurrent use: copy appends to one buffer. A caller whose -// copies come from several goroutines — ColdReader.FetchEvents, once its -// packfile reads fan out — must serialize them. +// NOT safe for concurrent use: copy appends to one buffer, so a caller whose +// copies come from several goroutines must serialize them. type byteArena struct { buf []byte } -// The arena's allocation unit ramps: the first chunk is small so a -// limit=1 page fetching a few hundred bytes does not pay 64 KiB for -// them, and each subsequent chunk doubles up to arenaChunkSize so a -// 512-candidate fetch of ~250B payloads still lands in a handful of -// allocations. +// The allocation unit ramps: the first chunk is small so a one-event page does +// not pay 64 KiB, and each subsequent chunk doubles up to arenaChunkSize. const ( arenaFirstChunkSize = 4 << 10 arenaChunkSize = 64 << 10 diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go index cf26aaea0..3a537d860 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena_test.go @@ -7,9 +7,7 @@ import ( "github.com/stretchr/testify/require" ) -// TestByteArenaCopiesAreStable pins the arena's one load-bearing property: -// a returned copy never moves or changes, however much is copied after it — -// including copies that force new chunks and copies larger than a chunk. +// A returned copy never moves or changes, however much is copied after it. func TestByteArenaCopiesAreStable(t *testing.T) { var a byteArena src := make([]byte, 300) @@ -31,8 +29,7 @@ func TestByteArenaCopiesAreStable(t *testing.T) { for i := range got { require.True(t, bytes.Equal(got[i], want[i]), "copy %d changed", i) } - // Appending to a returned copy must not scribble into the arena: the - // three-index slice pins capacity to length. + // Appending to a returned copy must not scribble into the arena. c := a.copy([]byte{1, 2, 3}) next := a.copy([]byte{9, 9, 9}) _ = append(c, 7) // the append must copy, not extend in place diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go index 087892608..c93b125cc 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go @@ -434,20 +434,12 @@ func buildMPHF( // /index.hash produced by an earlier buildMPHF) for // query-time lookups. // -// The file is mmapped (streamhash.Open), never read whole: pages -// fault in on first touch from the kernel page cache, which is -// keyed by the file rather than the mapping, so every reader of the -// same chunk shares the resident pages no matter how short its own -// lifetime. A per-request open therefore costs a map/unmap pair and -// the few pages its lookups touch — not a read and heap copy of the -// whole file, whose size has no design bound: it scales with the -// chunk's distinct term count, and terms are caller-controlled -// on-chain data (a 60M-term chunk makes this file 18.5 MB, and a -// whole-file read per request out of that is the cardinality cliff -// this used to be). A deployment on storage where cold faults are -// expensive AND the page cache cannot hold a serving chunk's index -// resident would want a prefault (madvise WILLNEED) on open, not a -// heap copy. +// The file is mmapped, never read whole. Pages fault in from the kernel page +// cache, which is keyed by the file rather than the mapping, so every reader +// of the same chunk shares them however short its own lifetime. A per-request +// open therefore costs a map/unmap pair and the pages its lookups touch, not a +// heap copy of a file whose size has no design bound: it scales with the +// chunk's distinct term count, which is caller-controlled on-chain data. // // Close unmaps; callers must call it. func openMPHF(path string) (*mphf, error) { diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go index dd51f56a9..a835ac7d2 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go @@ -475,13 +475,11 @@ func (c *ColdReader) FetchEvents(ctx context.Context, eventIDs []uint32) ([]Payl positions[i] = int(id) } results := make([]Payload, len(eventIDs)) - // One arena per call, shared by every worker: a call's payloads live and - // die together, so per-payload clones would only add GC work. The arena - // is one appended buffer and is single-threaded by construction, while - // ReadItems calls the callback from up to Concurrency goroutines — hence - // the lock. It covers the copy alone; the read, the record decode and the - // Unmarshal all stay outside it, so a fan-out still overlaps everything - // that costs. + // One arena per call, shared by every worker, because a call's payloads + // live and die together. The arena is a single appended buffer while + // ReadItems calls back from up to Concurrency goroutines, hence the lock. + // It covers the copy alone, so a fan-out still overlaps the read, the + // record decode and the Unmarshal. var ( arenaMu sync.Mutex arena byteArena diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader_fanout_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader_fanout_test.go index e4f6d8bae..9816aa11c 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader_fanout_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader_fanout_test.go @@ -9,14 +9,11 @@ import ( "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/chunk" ) -// TestColdReader_FetchEventsFansOutSafely pins the payload arena against the -// reader's own fan-out. With Concurrency > 1 packfile.ReadItems splits the -// positions into batches and calls the callback from one goroutine per batch, -// so the arena every callback copies through is shared across goroutines — an -// unsynchronized append there corrupts payloads at best and segfaults at worst. -// The fixture is wide enough (many records, one position each) to force several -// batches, and every payload is checked, so a torn copy fails the assertion even -// on a run where -race happens not to schedule the overlap. +// Pins the payload arena against the reader's own fan-out: with Concurrency +// above one, ReadItems calls the callback from one goroutine per batch, so an +// unsynchronized append into the shared arena corrupts payloads or segfaults. +// Every payload is checked, so a torn copy fails even when -race does not +// schedule the overlap. func TestColdReader_FetchEventsFansOutSafely(t *testing.T) { const ( chunkID = chunk.ID(0) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go index c9529c2e8..c3e34f875 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go @@ -136,10 +136,9 @@ func (d *denseState) snapshot() *roaring.Bitmap { } // cardinality is the term's id count without materializing a -// snapshot for it. A live pub already is the term, exactly, so it -// answers lock-free; otherwise the count comes off wbm under mu, -// which costs a walk of the writer's containers rather than a clone -// of them. +// snapshot for it. A live pub is the term exactly, so it answers +// lock-free; otherwise the count comes off wbm under mu, a walk of +// the writer's containers rather than a clone of them. func (d *denseState) cardinality() uint64 { if bm := d.pub.Load(); bm != nil { return bm.GetCardinality() @@ -149,52 +148,37 @@ func (d *denseState) cardinality() uint64 { return d.wbm.GetCardinality() } -// postings is an iterable view of ONE term's event IDs, in whichever -// representation the index already holds: the sparse mode's sorted -// []uint32, a dense term's live denseState, or a bitmap that came -// from outside the mirror (the cold tier's, and the ascending path's -// own bulk answers). Exactly one of the three is set; the zero value -// means the term is absent. +// postings is an iterable view of one term's event IDs in whichever +// representation the index holds: the sparse mode's sorted []uint32, a dense +// term's live denseState, or a bitmap from outside the mirror (the cold tier's, +// and the ascending path's own bulk answers). Exactly one is set; the zero value +// means the term is absent. It exists so the ascending match path can iterate a +// sparse term in place, where Get has to promise a bitmap and so pays a +// roaring.New plus AddMany on every sparse lookup. // -// It exists so the ascending match path can iterate a sparse term in -// place. Get has to promise a *roaring.Bitmap, so it pays a -// roaring.New + AddMany over the id list on every sparse lookup — -// measurably the bulk of the hot read path's container churn, and -// pure waste when the consumer only walks the ids in order. -// -// Ownership: ids borrows the atomically published termState's slice, -// which no writer ever mutates (a sparse AddTo publishes a NEW -// termState), so it may be held indefinitely; it must be read, never -// written or appended to. dense is the live term, and every bitmap it -// yields comes from denseState.snapshot — the writer's own wbm and -// the raw pub pointer are never handed out — so what bitmap() -// returns obeys Get's read-only contract verbatim, forbidden and safe -// method lists included. -// -// A postings over a dense term is a view, not a frozen copy: two -// materializations of it can straddle an ingest and the later one -// hold more ids. Callers pin a window before the lookup and clip -// every cursor to it at the leaf (see bitmapIter.end), so ids a write -// added after the pin sit above the window and are never yielded. +// Ownership matches Get's: ids borrows the atomically published termState, which +// no writer ever mutates, and every bitmap a dense term yields comes from +// denseState.snapshot — never wbm, never the raw pub pointer — so Get's +// read-only contract applies verbatim, and the id slice must likewise be read, +// never written or appended to. A dense postings is a view, not a frozen copy: +// a later materialization can hold ids a write added since. Callers pin a window +// before the lookup and clip every cursor to it at the leaf (bitmapIter.end), so +// those ids sit above the window and are never yielded. type postings struct { ids []uint32 bm *roaring.Bitmap dense *denseState } -// present reports whether the term is in the index at all. A term -// that is present but holds no ids (possible only for an empty bitmap -// handed to NewConcurrentBitmapsFromBitmaps) is present: it yields an -// exhausted cursor, which intersects and unions to the same result an -// absent term's caller-side skip would produce. +// present reports whether the term is in the index at all. A term present but +// holding no ids still counts as present: it yields an exhausted cursor, which +// intersects and unions to what an absent term's caller-side skip produces. func (p postings) present() bool { return p.bm != nil || p.ids != nil || p.dense != nil } -// bitmap is the term's ids as a roaring bitmap, or nil when the term -// is sparse or absent — sparse callers walk ids instead. A dense term -// is snapshotted here, the only place outside Get that materializes -// one, which is what keeps the writer's wbm off every read path. -// Repeat calls cost nothing while no write lands: denseState caches -// the snapshot it published. +// bitmap is the term's ids as a roaring bitmap, or nil when the term is sparse +// or absent — sparse callers walk ids instead. A dense term is snapshotted here, +// the only place outside Get that materializes one, and repeat calls cost +// nothing while no write lands: denseState caches the snapshot it published. func (p postings) bitmap() *roaring.Bitmap { if p.dense != nil { return p.dense.snapshot() @@ -202,15 +186,12 @@ func (p postings) bitmap() *roaring.Bitmap { return p.bm } -// estimate is the term's cardinality over the whole chunk, the weight -// the ascending path's query plan orders an intersection by. It -// ignores the caller's window, so it ranks terms rather than counting -// a query's candidates. The zero postings weighs 0. -// -// A dense term is counted off the writer's own bitmap -// (denseState.cardinality), not off a snapshot: planning wants a -// number, and snapshot would clone a term written since its last read -// to hand back a bitmap the plan may never walk. +// estimate is the term's cardinality over the whole chunk, the weight the +// ascending path orders an intersection by. It ignores the caller's window, so +// it ranks terms rather than counting a query's candidates. A dense term is +// counted off denseState.cardinality, not a snapshot: planning wants a number, +// and snapshot would clone a term written since its last read for a bitmap the +// plan may never walk. func (p postings) estimate() uint64 { if p.dense != nil { return p.dense.cardinality() @@ -258,13 +239,10 @@ func (s *ConcurrentBitmaps) AddTo(key TermKey, eventIDs ...uint32) { p.Store(termStateFromIDs(appendSorted(ids, eventIDs))) } -// lookupPostings is Get without the sparse-mode materialization: it -// hands back the term's live representation rather than converting it -// to a bitmap. Same concurrency story as Get — the RLock covers the -// map lookup only, and the entry load is lock-free. A dense term is -// handed back as its denseState, so the bitmap a caller eventually -// reads is denseState.snapshot's, immutable and current as of the -// call that asks for it. A miss returns the zero postings. +// lookupPostings is Get without the sparse-mode materialization: it hands back +// the term's live representation rather than converting it to a bitmap. Same +// concurrency story as Get; a dense term comes back as its denseState, so the +// bitmap a caller reads is snapshot's. A miss returns the zero postings. func (s *ConcurrentBitmaps) lookupPostings(key TermKey) postings { s.rwmu.RLock() p := s.terms[key] diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go index 23696f806..c515a3849 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go @@ -29,15 +29,10 @@ const ( // // - DataCF holds XDR-encoded event payloads: compressible (zstd // typically 2-3× on XDR) and read in batches via -// BatchedMultiGetCF. The block is the decompression unit of a -// point read, so its size trades compression context against -// per-miss work: getEvents fetches scattered ~250B events, and a -// 32 KiB block made every cache miss decompress ~128 events to -// serve one. Measured under the K-stratified events corpus at -// limit=1000, 8 KiB vs 32 KiB: service p50 −53% / p99 −59% and -// −41% peak RSS at stress density (sac-6000), neutral on pubnet; -// +0.8% on disk; ingest wall unchanged; 4 KiB adds nothing more -// (the curve is flat below 8 KiB). +// BatchedMultiGetCF. The block is the decompression unit of a point +// read, so its size trades compression context against per-miss work: +// getEvents fetches scattered ~250B events, and a 32 KiB block made +// every cache miss decompress ~128 of them to serve one. // - IndexCF stores 20-byte (term_hash || event_id) keys with // empty values — nothing in the values to compress, and small // blocks reduce wasted I/O per random Lookup miss (each Lookup @@ -112,8 +107,8 @@ type HotStore struct { offsets *ConcurrentLedgerOffsets } -// Compile-time guards: *HotStore satisfies Reader, and the optional -// postingReader seam match.go's ascending path probes for. +// Compile-time guards: *HotStore satisfies Reader and the optional +// postingReader seam. var ( _ Reader = (*HotStore)(nil) _ postingReader = (*HotStore)(nil) @@ -455,23 +450,14 @@ func (h *HotStore) IngestLedgerToBatch( return func() { h.applyLedger(startID, termKeys) }, nil } -// lookupPostings is the no-materialize half of LookupKeys, and the -// hot store's implementation of the optional postingReader seam the -// ascending match path probes for (see match.go). It returns each -// term's live mirror representation — sorted ids for a sparse term, -// the roaring bitmap for a dense one — so a query that only walks ids -// in ascending order never pays Get's roaring.New + AddMany per -// sparse term. +// lookupPostings is the no-materialize half of LookupKeys, and the hot store's +// implementation of the optional postingReader seam. It returns each term's +// live mirror representation, so a query that only walks ids in ascending +// order never pays Get's roaring.New plus AddMany per sparse term. // -// Results are positionally aligned with keys; a miss is the zero -// postings, which postings.present() reports as absent. Same -// borrowed-snapshot contract as LookupKeys: read-only, valid +// Results are positionally aligned with keys; a miss is the zero postings. +// Same borrowed-snapshot contract as LookupKeys: read-only, valid // indefinitely. -// -// The cold reader deliberately does NOT implement this. Its postings -// arrive as freshly-unmarshaled bitmaps out of index.pack, so there -// is no un-materialized representation to expose; match.go's fallback -// wraps its LookupKeys bitmaps in the same cursor type. func (h *HotStore) lookupPostings(ctx context.Context, keys []TermKey) ([]postings, error) { if h.chunkStore.IsClosed() { return nil, stores.ErrStoreClosed diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go index 33fc6791c..2599938b1 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go @@ -11,12 +11,10 @@ package event // then stream in internal batches. On the cold path this is one // MPHF+index.pack round trip per Matches call, not per batch. // -// The candidate set is then built one of two ways. ASCENDING pulls -// ids through the un-materialized iterator tree in match_iter.go: no -// intermediate bitmap, and no index work past what the consumer -// pulls. DESCENDING materializes a union bitmap through roaring's -// aggregation, because roaring's reverse iterator offers no gallop to -// build ascending-style combinators on. +// The candidate set is built one of two ways. Ascending pulls ids through the +// un-materialized iterator tree in match_iter.go. Descending materializes a +// union bitmap, because roaring's reverse iterator offers no gallop to build +// ascending-style combinators on. import ( "bytes" @@ -225,13 +223,10 @@ type Match struct { // LookupKeys result. type termPlan [][]int -// batchSizes resolves the first and following internal batch sizes -// from the caller's hint. A positive hint sizes the first batch in both -// directions — smaller for a small page, larger so a page-sized request -// is one storage round trip instead of several — capped at eight default -// batches so a wild hint cannot demand an unbounded fetch. Both results -// are clamped positive, so a zero test seam cannot stall a stream (a -// zero step never advances). +// batchSizes resolves the first and following internal batch sizes from the +// caller's hint, capped at eight default batches so a wild hint cannot demand +// an unbounded fetch. Both are clamped positive: a zero step never advances, +// so a zero test seam would stall the stream. func batchSizes(hint int) (int, int) { rest := max(1, matchBatchSize) first := rest @@ -261,12 +256,9 @@ func batchSizes(hint int) (int, int) { // drops are invisible: the iterator advances past them internally, so // consumers never see or reason about resume state. // -// firstBatch sizes the first internal fetch batch. A consumer that -// will stop after N matches passes N, so the first round trip fetches -// no more than it needs; zero and negative hints use the default, and -// a positive hint is honored up to eight default batches (a page-sized -// request is one round trip; a wild hint cannot demand an unbounded -// fetch). Later batches use the default size. The hint changes I/O +// firstBatch sizes the first internal fetch batch: a consumer that will stop +// after N matches passes N. Zero and negative hints use the default, and a +// positive one is honored up to eight default batches. The hint changes I/O // counts only, never what the stream yields. func Matches( ctx context.Context, r Reader, filters []Filter, window IDRange, @@ -288,13 +280,7 @@ func Matches( streamRange(ctx, r, window, descending, firstBatch, yield) return } - // Direction splits the plan from here. Ascending pulls - // candidates through the un-materialized iterator tree - // (match_iter.go): no intermediate bitmap, and the walk stops - // the moment the consumer does. Descending materializes the - // union the old way — roaring's reverse iterator has no - // AdvanceIfNeeded, so there is no reverse gallop to build the - // combinators on. + // Direction splits the plan from here; see the file header. if descending { union, err := unionForFilters(ctx, r, plans, uniqueKeys, window) if err != nil { @@ -313,9 +299,8 @@ func Matches( return } candidates := candidateIter(plans, sources, window) - // The structural twin of the descending path's - // union.IsEmpty(): one peek settles whether anything matches, - // without reading a single id further than that. + // The twin of the descending path's union.IsEmpty(): one peek + // settles whether anything matches, reading no further. if _, ok := candidates.peek(); !ok { return } @@ -355,11 +340,10 @@ func validateMatchCall(ctx context.Context, r Reader, filters []Filter, window I // holds the slots filter i needs; the terms within a group are OR-ed // and the groups AND-ed. // -// matchAll reports that some filter (or the empty slice) constrains -// nothing, so the caller streams the window directly. Reading that -// condition off the term groups themselves is what keeps an -// unconstrained filter from intersecting nothing and coming back -// empty instead. +// matchAll reports that some filter, or the empty slice, constrains nothing, +// so the caller streams the window directly. Reading that off the term groups +// is what keeps an unconstrained filter from intersecting nothing and coming +// back empty. func planIndexTerms(filters []Filter) ([]termPlan, []TermKey, bool) { if len(filters) == 0 { return nil, nil, true @@ -384,26 +368,20 @@ func planIndexTerms(filters []Filter) ([]termPlan, []TermKey, bool) { return plans, uniqueKeys, false } -// postingReader is the optional, hot-tier half of the index read -// surface: a Reader that can expose a term's postings WITHOUT -// materializing a bitmap for it. +// postingReader is the optional, hot-tier half of the index read surface: a +// Reader that can expose a term's postings without materializing a bitmap. // -// Deliberately not folded into Reader. Cold postings genuinely are -// bitmaps — ColdReader unmarshals a fresh one per term out of -// index.pack — so it has nothing un-materialized to hand back and -// could only implement the method by re-wrapping what LookupKeys -// already returns. Hoisting it into Reader would also impose it on -// every out-of-package implementation (the query package's test -// fakes) for no gain. The assertion in lookupPostings picks the fast -// path where it exists and wraps bitmaps everywhere else. +// Deliberately not folded into Reader. Cold postings genuinely are bitmaps, +// unmarshaled per term out of index.pack, so ColdReader has nothing +// un-materialized to hand back, and hoisting the method would impose it on +// every out-of-package implementation for no gain. type postingReader interface { lookupPostings(ctx context.Context, keys []TermKey) ([]postings, error) } -// lookupPostings resolves keys to per-term postings, positionally -// aligned with keys, in one batched call. It takes the -// no-materialize path when r offers one and otherwise falls back to -// LookupKeys; a nil bitmap stays the zero postings, i.e. absent. +// lookupPostings resolves keys to per-term postings, positionally aligned with +// keys, in one batched call. It takes the no-materialize path when r offers +// one; a nil bitmap stays the zero postings, meaning absent. func lookupPostings(ctx context.Context, r Reader, keys []TermKey) ([]postings, error) { if pr, ok := r.(postingReader); ok { sources, err := pr.lookupPostings(ctx, keys) @@ -425,16 +403,10 @@ func lookupPostings(ctx context.Context, r Reader, keys []TermKey) ([]postings, return sources, nil } -// unionForFilters materializes the DESCENDING path's candidate set: -// steps 2-5 below, over the plan planIndexTerms already resolved. The -// result is empty when no candidate falls in the window, and is never -// a borrowed mirror snapshot (the window AND allocates on the -// borrowing path), so downstream iteration is safe. -// -// The ascending path does not come through here at all — see -// candidateIter, which answers the same question with no intermediate -// bitmap. This shape survives because roaring offers no reverse -// gallop to build ascending-style combinators on. +// unionForFilters materializes the descending path's candidate set over the +// plan planIndexTerms resolved. The result is never a borrowed mirror +// snapshot, because the window AND allocates on the borrowing path, so +// downstream iteration is safe. The ascending path uses candidateIter instead. func unionForFilters( ctx context.Context, r Reader, filterPlans []termPlan, uniqueKeys []TermKey, window IDRange, @@ -527,12 +499,10 @@ func unionForFilters( return union, nil } -// streamUnion walks the DESCENDING path's materialized union bitmap -// in internal batches: collect candidate ordinals up to the batch -// size, fetch, post-filter, yield the survivors. Drops advance the -// walk with no yield. The first batch is sized to firstBatch (see -// Matches); later batches use the default. Stepping one id at a time -// is fine here: the fetch I/O dominates a 512-step loop. +// streamUnion walks the descending path's materialized union bitmap in +// internal batches: collect candidate ordinals up to the batch size, fetch, +// post-filter, yield the survivors. Stepping one id at a time is fine here, +// because the fetch I/O dominates the loop. func streamUnion( ctx context.Context, r Reader, filters []Filter, union *roaring.Bitmap, firstBatch int, yield func(Match, error) bool, @@ -559,11 +529,9 @@ func streamUnion( } } -// streamCandidates is streamUnion's ASCENDING twin over the -// un-materialized iterator tree. Same batch loop, same post-filter, -// same yields; the difference is that candidates are pulled out of -// the tree one at a time as the batch fills, so a consumer that stops -// after one page never touched the postings past it. +// streamCandidates is streamUnion's ascending twin over the un-materialized +// iterator tree. Candidates are pulled from the tree as the batch fills, so a +// consumer that stops after one page never touched the postings past it. func streamCandidates( ctx context.Context, r Reader, filters []Filter, candidates idIter, firstBatch int, yield func(Match, error) bool, @@ -594,13 +562,10 @@ func streamCandidates( } } -// emitBatch fetches one batch of candidate ordinals, drops the -// bitmap-side false positives and yields the survivors, reporting -// whether the stream should continue. -// -// FetchEvents requires ascending ids, so a descending batch — which -// arrives highest-first — is flipped in place before the fetch and -// the surviving matches are flipped back before they are yielded. +// emitBatch fetches one batch of candidate ordinals, drops the bitmap-side +// false positives and yields the survivors, reporting whether the stream +// should continue. FetchEvents requires ascending ids, so a descending batch +// is flipped in place before the fetch and flipped back before yielding. func emitBatch( ctx context.Context, r Reader, filters []Filter, ids []uint32, descending bool, yield func(Match, error) bool, diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go index a7e46b097..5865c5d52 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go @@ -1,24 +1,12 @@ package event -// match_iter.go is the ascending match path's un-materialized query -// plan: a tree of peekable ascending cursors that pulls candidate -// event IDs straight out of the index's postings, in order, and stops -// as soon as the consumer stops. +// match_iter.go is the ascending match path's un-materialized query plan: a +// tree of peekable ascending cursors that pulls candidate event IDs straight +// out of the index's postings, in order, with no intermediate bitmap. // -// Why it exists: unionForFilters' materializing shape — OR every -// group, AND every filter's groups, OR across filters, AND a window -// bitmap — builds several fresh roaring bitmaps over the FULL -// multi-million-posting terms just to serve one 1000-event page, and -// materializes a bitmap per sparse term on top (see -// ConcurrentBitmaps.Get). On the serving allocation profile that -// construction was ~80% of hot getEvents allocations. The tree below -// answers the same question with no intermediate bitmap at all: the -// only per-request allocations are the handful of cursor structs. -// -// Direction: ascending only. roaring's reverse iterator has no -// AdvanceIfNeeded, so there is no reverse gallop to build the -// combinators on; descending keeps the materialized path in match.go -// unchanged. +// Ascending only. roaring's reverse iterator has no AdvanceIfNeeded, so there +// is no reverse gallop to build these combinators on, and descending keeps +// the materialized path in match.go. import ( "cmp" @@ -27,13 +15,10 @@ import ( "github.com/RoaringBitmap/roaring/v2" ) -// idIter is a peekable cursor over a strictly ascending run of -// chunk-relative event IDs. -// -// Structural early exit: nothing in the tree reads further than the -// consumer pulls. peek resolves exactly one id — the combinators -// cache it and do no lookahead — so a batch loop that stops after N -// ids has touched only the postings those N ids needed. +// idIter is a peekable cursor over a strictly ascending run of chunk-relative +// event IDs. Nothing in the tree reads further than the consumer pulls: peek +// resolves exactly one id and the combinators cache it rather than looking +// ahead, so a loop that stops after N ids touched only what those N needed. type idIter interface { // peek returns the id under the cursor without consuming it. ok // is false once the cursor is exhausted, which is permanent. @@ -47,24 +32,20 @@ type idIter interface { advance(floor uint32) } -// emptyIter is the permanently exhausted cursor: an absent term, or a -// query where no filter survived group resolution. Zero-size, so an -// idIter holding it never allocates. +// emptyIter is the permanently exhausted cursor: an absent term, or a query +// where no filter survived group resolution. type emptyIter struct{} func (emptyIter) peek() (uint32, bool) { return 0, false } func (emptyIter) next() {} func (emptyIter) advance(uint32) {} -// sliceIter walks a sorted []uint32 postings list in place — the hot -// mirror's sparse-term representation, read directly instead of being -// inflated into a roaring bitmap per lookup. -// -// The window is applied by reslicing at construction (see -// postings.iter), so the cursor itself carries no bounds check and -// the underlying array is never copied. The mirror publishes a fresh -// slice on every AddTo and never mutates a published one, so the -// borrowed backing array is immutable for the cursor's lifetime. +// sliceIter walks the hot mirror's sparse-term postings list in place, rather +// than inflating it into a roaring bitmap per lookup. The window is applied by +// reslicing at construction, so the cursor carries no bounds check and never +// copies the array. The mirror publishes a fresh slice on every AddTo and +// never mutates a published one, so the borrowed array is immutable for the +// cursor's lifetime. type sliceIter struct { ids []uint32 i int @@ -87,28 +68,21 @@ func (s *sliceIter) advance(floor uint32) { if s.i >= len(s.ids) || s.ids[s.i] >= floor { return } - // Binary search over the unread tail: one gallop can skip most of - // a long postings list. + // Binary search over the unread tail, so one gallop can skip most of it. j, _ := slices.BinarySearch(s.ids[s.i:], floor) s.i += j } -// bitmapIter walks a roaring bitmap's set bits through the library's -// IntPeekable cursor, whose PeekNext/AdvanceIfNeeded are the -// roaring-side gallop. Both read through getContainerAtIndex, not the -// *Writable* accessor, so they are on the COW-safe read list in -// ConcurrentBitmaps.Get's contract and are safe on a borrowed mirror -// snapshot. +// bitmapIter walks a roaring bitmap through the library's IntPeekable cursor. +// PeekNext and AdvanceIfNeeded read through getContainerAtIndex rather than +// the writable accessor, so they are on the COW-safe read list in +// ConcurrentBitmaps.Get's contract and are safe on a borrowed mirror snapshot. // -// end clamps the caller's pinned window at the LEAF rather than at -// the root. Every source stops at the window's upper bound, so no -// combinator above it ever aligns on ids outside the window — which -// is what makes killing the materialized range-AND free rather than a -// cost shifted upward. It is also what clips phantom IDs from a -// concurrent hot-store ingest: the mirror publishes index entries -// before offsets, so a lookup can briefly surface IDs past -// EventCount, and the stream must stay inside the snapshot the caller -// pinned at request entry. +// end clamps the caller's pinned window at the leaf, so no combinator above +// ever aligns on ids outside it. That is also what clips phantom IDs from a +// concurrent ingest: the mirror publishes index entries before offsets, so a +// lookup can briefly surface IDs past EventCount, and the stream must stay +// inside the snapshot the caller pinned at request entry. type bitmapIter struct { it roaring.IntPeekable end uint32 @@ -135,11 +109,9 @@ func (b *bitmapIter) advance(floor uint32) { b.it.AdvanceIfNeeded(floor) } -// iter returns an ascending cursor over the postings clipped to -// window, without materializing anything: a sparse term is resliced -// in place, a dense one is walked through roaring's own iterator. -// Callers must check present() first — absent postings are the -// group-missed signal, not an empty cursor. +// iter returns an ascending cursor over the postings clipped to window, +// materializing nothing. Callers must check present() first: absent postings +// are the group-missed signal, not an empty cursor. func (p postings) iter(window IDRange) idIter { if bm := p.bitmap(); bm != nil { it := bm.Iterator() @@ -149,20 +121,16 @@ func (p postings) iter(window IDRange) idIter { ids := p.ids lo, _ := slices.BinarySearch(ids, window.Start) ids = ids[lo:] - // BinarySearch returns the first index at or above the target, so - // this drops exactly the ids at or past the window's exclusive End. + // BinarySearch lands at or above the target, dropping exactly the ids at + // or past the window's exclusive End. hi, _ := slices.BinarySearch(ids, window.End) return &sliceIter{ids: ids[:hi]} } -// unionIter is the OR of its children: the ascending merge of their -// ids with equal ids across children collapsed to one. -// -// Children are scanned linearly for the minimum rather than kept in a -// heap. K is the number of terms in one group (at most the -// topic-count bucket family) or the number of filters in a query — -// single digits in practice — and advance has to move all K children -// regardless, which would force a full re-heapify on every gallop. +// unionIter is the OR of its children: the ascending merge of their ids, with +// equal ids across children collapsed to one. Children are scanned linearly +// for the minimum rather than heaped, because advance has to move all of them +// regardless and K is single digits. type unionIter struct { children []idIter cur uint32 @@ -188,10 +156,9 @@ func (u *unionIter) next() { if !ok { return } - // Dedup: EVERY child sitting on the winning id steps past it, so - // an id several children hold is yielded once. Stepping only the - // winner would re-emit it from each of the others in turn — and - // FetchEvents rejects a duplicate id outright. + // Every child sitting on the winning id steps past it, so an id several + // children hold is yielded once. Stepping only the winner would re-emit + // it from each of the others, and FetchEvents rejects a duplicate. for _, c := range u.children { if cv, cok := c.peek(); cok && cv == v { c.next() @@ -210,20 +177,15 @@ func (u *unionIter) advance(floor uint32) { u.primed = false } -// intersectIter is the AND of its children, by galloping alignment: -// every child is advanced to the running maximum of the peeks until -// they all agree on one id. A child that exhausts ends the -// intersection — permanently, since cursors only move forward. +// intersectIter is the AND of its children by galloping alignment: every child +// is advanced to the running maximum of the peeks until they all agree on one +// id. A child that exhausts ends the intersection permanently. // -// Alignment is bounded when bulk is set. Every round that does not -// settle the AND ends with the leading child stepped past one of its -// ids, so a walk costs up to one round per id of that child — and -// with chunk-sized groups on both sides of a thin overlap, the walk -// spends all of them on a window the consumer's page never reaches. -// Once the budget is gone the AND hands the rest of its window to -// bulk, whose cost is bounded by the containers the terms span rather -// than by the ids inside them. A nil bulk leaves the alignment -// unbounded, which is what an AND with no bulk twin gets. +// Alignment is bounded when bulk is set. A walk costs up to one round per id +// of the leading child, which on a thin overlap between chunk-sized groups is +// spent on a window the consumer never reaches. Once the budget is gone the +// AND hands the rest of its window to bulk, whose cost is set by the +// containers the terms span. A nil bulk leaves the alignment unbounded. type intersectIter struct { children []idIter cur uint32 @@ -261,11 +223,9 @@ func (n *intersectIter) align() (uint32, bool) { if n.bulk != nil { n.rounds++ if n.rounds > n.budget { - // A child raises the floor only past ids it does not - // hold, and cand only rises, so no id of the - // intersection was skipped between the last one - // yielded and cand: the bulk answer from cand up is - // exactly the rest of this AND. + // A child raises the floor only past ids it does not hold, + // and cand only rises, so nothing in the intersection was + // skipped: the bulk answer from cand up is the rest of it. n.spilled = n.bulk.iter(cand) n.bulk = nil return 0, false @@ -279,10 +239,9 @@ func (n *intersectIter) align() (uint32, bool) { return 0, false } if v > cand { - // This child overshot the floor. Raise it and repeat - // the pass so the children already visited are pulled - // up to the new floor too. cand strictly increases - // per pass, so the loop terminates. + // Raise the floor and repeat the pass, so children already + // visited are pulled up too. cand strictly increases per + // pass, which is what terminates the loop. cand, raised = v, true } } @@ -324,12 +283,10 @@ func (n *intersectIter) advance(floor uint32) { n.primed = false } -// unionOf and intersectOf collapse a single input to the input -// itself, rather than wrapping it in a combinator that would re-scan -// a one-element slice on every step. This is the iterator-side twin -// of the singleton guards the materialized path needs for a harder -// reason: roaring's FastAnd/FastOr have historically Cloned a -// single-input slice, so unionForFilters must never hand them one. +// unionOf and intersectOf collapse a single input to the input itself, rather +// than wrapping it in a combinator that re-scans a one-element slice per step. +// The materialized path needs the same guard for a harder reason: roaring's +// FastAnd and FastOr have historically Cloned a single-input slice. func unionOf(children []idIter) idIter { switch len(children) { case 0: @@ -344,8 +301,7 @@ func unionOf(children []idIter) idIter { func intersectOf(children []idIter) idIter { switch len(children) { case 0: - // Unreachable: a filter that names no term group takes the - // match-all path before any index I/O. + // Unreachable: a filter naming no term group takes the match-all path. return emptyIter{} case 1: return children[0] @@ -354,16 +310,11 @@ func intersectOf(children []idIter) idIter { } } -// candidateIter assembles the ascending candidate cursor for one -// Matches call out of the batched lookup's postings: terms within a -// group OR, a filter's groups AND, filters OR — the same three steps -// unionForFilters performs on bitmaps, and clamped to the same -// window, but with nothing materialized in between. -// -// A filter with an entirely absent group contributes nothing, exactly -// as in the materialized path; if that leaves no filter at all the -// result is the exhausted cursor, the un-materialized form of -// "union.IsEmpty()". +// candidateIter assembles one Matches call's ascending candidate cursor out of +// the batched lookup's postings: terms within a group OR, a filter's groups +// AND, filters OR, clamped to window. A filter with an entirely absent group +// contributes nothing; if that leaves no filter the result is the exhausted +// cursor. func candidateIter(plans []termPlan, sources []postings, window IDRange) idIter { perFilter := make([]idIter, 0, len(plans)) for _, plan := range plans { @@ -385,22 +336,18 @@ func candidateIter(plans []termPlan, sources []postings, window IDRange) idIter return unionOf(perFilter) } -// candidateGroup is one of a filter's resolved groups: where its terms -// live in the batched lookup, and the weight that orders the filter's -// AND. +// candidateGroup is one of a filter's resolved groups: where its terms live in +// the batched lookup, and the weight that orders the filter's AND. type candidateGroup struct { slots []int est uint64 } -// resolveGroup weighs the postings at slots, reporting false when -// every one of them is absent from the index — the mirror of -// unionSlots' nil return, and the signal that the owning filter can -// match nothing. -// -// The weight sums the present terms' cardinalities, an upper bound the -// OR's dedup can only lower. It orders an intersection; nothing reads -// it as a count. +// resolveGroup weighs the postings at slots, reporting false when every one is +// absent from the index, which is the signal that the owning filter can match +// nothing. The weight sums the present terms' cardinalities, an upper bound +// the OR's dedup can only lower; it orders an intersection and is never read +// as a count. func resolveGroup(sources []postings, slots []int) (candidateGroup, bool) { g := candidateGroup{slots: slots} present := false @@ -415,41 +362,23 @@ func resolveGroup(sources []postings, slots []int) (candidateGroup, bool) { return g, present } -// alignBudget is how many alignment rounds one filter's AND may spend -// before it spills to the bulk answer. -// -// A round is a handful of galloping advances — tens of nanoseconds -// over chunk-spanning terms — so the budget buys a few hundred -// microseconds of walking, about what the bulk answer costs for a -// filter whose terms span a whole chunk. That is the shape of the -// trade at every value: a spill pays for the walk it abandoned on top -// of the bulk answer, so it costs about twice the bulk on a filter it -// fires on wrongly, and saves the difference between the bulk and an -// unbounded walk on one it fires on rightly. -// -// A filter selective enough to fill a page out of the window's first -// fraction settles in a round or two per id it yields and never comes -// near the budget, whatever it is set to. What the value decides is -// the boundary between filters that yield steadily but slowly — a few -// dozen rounds per id, where the walk still finishes — and the ones -// whose alignment crosses a chunk to find a handful of matches. +// alignBudget is how many alignment rounds one filter's AND may spend before +// it spills to the bulk answer. It separates filters that yield slowly but +// steadily, where the walk still finishes, from those whose alignment crosses +// a chunk to find a handful of matches. // -// A var, not a const, so in-package tests can shrink it to force the -// spill; it never changes what a stream yields. +// A var rather than a const so in-package tests can shrink it to force the +// spill. It never changes what a stream yields. // //nolint:gochecknoglobals // test seam; production never writes it var alignBudget uint64 = 8192 -// filterIter builds one filter's candidate cursor: the AND of its -// groups, rarest first, bounded by alignBudget. -// -// Rarest first because intersectIter's alignment seeds its floor from -// the leading child and raises the others to it in order, so the -// leading cursor is the one every barren round steps forward, and the -// trailing ones are the ones spared a wasted advance. Intersection is -// commutative, so the order changes only how fast the gallop -// converges — and it is what makes the budget's bound the rarest -// group's cardinality rather than the fattest's. +// filterIter builds one filter's candidate cursor: the AND of its groups, +// rarest first, bounded by alignBudget. Alignment seeds its floor from the +// leading child, so that cursor is the one every barren round steps forward. +// Intersection is commutative, so the order only changes how fast the gallop +// converges — and it makes the budget's bound the rarest group's cardinality +// rather than the fattest's. func filterIter(sources []postings, groups []candidateGroup, window IDRange) idIter { slices.SortStableFunc(groups, func(a, b candidateGroup) int { return cmp.Compare(a.est, b.est) @@ -468,36 +397,28 @@ func filterIter(sources []postings, groups []candidateGroup, window IDRange) idI } } -// bulkAnd is a filter's AND as roaring's aggregation would answer it, -// held unevaluated beside the walk that normally answers it instead. -// It is the walk's bound: the cost of the aggregation is set by the -// containers the terms span, so it does not grow with a window the -// walk would have to cross id by id. +// bulkAnd is a filter's AND as roaring's aggregation would answer it, held +// unevaluated beside the walk. It is the walk's bound: the aggregation's cost +// is set by the containers the terms span, not by the ids inside them. type bulkAnd struct { sources []postings groups []candidateGroup end uint32 } -// iter computes the filter's candidate set — OR each group, AND the -// groups smallest first — and returns a cursor over the part of it at -// or above floor. -// -// The inputs may be borrowed mirror snapshots: FastAnd and FastOr read -// them without writing through, and never see the single-element slice -// that would make them Clone, so what comes back is this call's own -// bitmap, and only the intersection large. The window lands at the -// leaf, exactly as it does on a borrowed term. +// iter computes the filter's candidate set and returns a cursor over the part +// of it at or above floor. The inputs may be borrowed mirror snapshots: +// FastAnd and FastOr read them without writing through, and never see the +// single-element slice that would make them Clone, so what comes back is this +// call's own bitmap. func (b *bulkAnd) iter(floor uint32) idIter { inputs := make([]*roaring.Bitmap, len(b.groups)) for i := range b.groups { inputs[i] = orGroup(b.sources, b.groups[i].slots) } - // FastAnd intersects left to right, so the smallest input first - // shrinks the accumulator fastest — the caller-side prep roaring's - // own docs call for, and the one unionForFilters does. A group's - // weight only bounds its OR, so the inputs are ranked again once - // they exist. + // FastAnd intersects left to right, so the smallest input first shrinks + // the accumulator fastest. A group's weight only bounds its OR, so the + // inputs are ranked again once they exist. slices.SortFunc(inputs, func(x, y *roaring.Bitmap) int { return cmp.Compare(x.GetCardinality(), y.GetCardinality()) }) @@ -505,12 +426,10 @@ func (b *bulkAnd) iter(floor uint32) idIter { IDRange{Start: floor, End: b.end}) } -// orGroup ORs a group's present terms into one bulkAnd input. A group -// holding one of them is that term's bitmap: FastOr has historically -// Cloned a single-element slice. A sparse term is inflated here, the -// one place the ascending path does what Get does — it costs the ids -// it holds, and only a filter whose walk already overran its budget -// ever pays it. +// orGroup ORs a group's present terms into one bulkAnd input. A group holding +// one is that term's bitmap, because FastOr has historically Cloned a +// single-element slice. A sparse term is inflated here, the one place the +// ascending path materializes, and only a filter that overran its budget pays. func orGroup(sources []postings, slots []int) *roaring.Bitmap { present := make([]*roaring.Bitmap, 0, len(slots)) for _, slot := range slots { @@ -531,10 +450,8 @@ func orGroup(sources []postings, slots []int) *roaring.Bitmap { return roaring.FastOr(present...) } -// groupIter ORs the postings at slots into one cursor, returning nil -// when every one of them is absent from the index — the mirror of -// unionSlots' nil return, and the signal that the owning filter can -// match nothing. +// groupIter ORs the postings at slots into one cursor, returning nil when +// every one is absent, which drops the owning filter. func groupIter(sources []postings, slots []int, window IDRange) idIter { present := make([]idIter, 0, len(slots)) for _, slot := range slots { diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go index cb24eb014..cc14538b9 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go @@ -1,12 +1,8 @@ package event -// match_iter_test.go covers the ascending path's un-materialized -// query plan: the cursor sources, the union/intersect combinators, -// and the tree candidateIter assembles from them. The end-to-end -// semantics stay pinned black-box by match_test.go; what is pinned -// here is the machinery underneath it, plus a randomized differential -// check that the tree and the descending path's materialized -// bitmap algebra answer identically. +// match_iter_test.go covers the ascending path's un-materialized query plan: +// the cursor sources, the combinators, and the tree candidateIter assembles, +// plus a randomized differential against the materialized bitmap algebra. import ( "context" @@ -27,12 +23,11 @@ import ( "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/chunk" ) -// wholeWindow is the no-op window: every cursor test that is not about -// clamping uses it. +// wholeWindow is the no-op window, for every test not about clamping. var wholeWindow = IDRange{Start: 0, End: ^uint32(0)} -// drain pulls a cursor dry. Always returns a non-nil slice so an empty -// result compares equal to a materialized bitmap's ToArray(). +// drain pulls a cursor dry, never returning nil so an empty result compares +// equal to a materialized bitmap's ToArray(). func drain(it idIter) []uint32 { out := []uint32{} for { @@ -45,9 +40,7 @@ func drain(it idIter) []uint32 { } } -// sparseSource / denseSource build the two representations the index -// actually holds, so a test can pin that both cursor sources behave -// identically. +// sparseSource and denseSource build the two representations the index holds. func sparseSource(ids ...uint32) postings { return postings{ids: ids} } func denseSource(ids ...uint32) postings { @@ -56,8 +49,7 @@ func denseSource(ids ...uint32) postings { return postings{bm: bm} } -// sourceKinds runs fn against both representations of the same id set, -// so every cursor-level assertion is made twice. +// sourceKinds runs fn against both representations of the same id set. func sourceKinds(ids ...uint32) map[string]func() postings { return map[string]func() postings{ "sparse": func() postings { return sparseSource(ids...) }, @@ -74,10 +66,8 @@ func TestPostingsPresent(t *testing.T) { assert.Empty(t, drain(postings{bm: roaring.New()}.iter(wholeWindow))) } -// TestIDIterSources pins peek/next/advance on both leaf sources: peek -// does not consume, advance lands on the first id at or above min, -// advance never moves backwards, and both are idempotent at -// exhaustion. +// peek does not consume, advance lands on the first id at or above the floor +// and never moves backwards, and both are idempotent at exhaustion. func TestIDIterSources(t *testing.T) { for name, mk := range sourceKinds(3, 7, 8, 20, 100) { t.Run(name, func(t *testing.T) { @@ -115,9 +105,8 @@ func TestIDIterSources(t *testing.T) { } } -// TestIDIterSourcesWindowClampsBothEnds pins the window applied at the -// leaf: ids below Start are skipped at construction and ids at or -// above the exclusive End exhaust the cursor. +// The window applied at the leaf: ids below Start are skipped at construction +// and ids at or above the exclusive End exhaust the cursor. func TestIDIterSourcesWindowClampsBothEnds(t *testing.T) { for name, mk := range sourceKinds(1, 4, 5, 6, 9, 10, 11) { t.Run(name, func(t *testing.T) { @@ -154,9 +143,8 @@ func TestEmptyIter(t *testing.T) { assert.Empty(t, drain(it)) } -// TestUnionIterDedups is the combinator's load-bearing property: an id -// several children hold is yielded once. Emitting it per child would -// hand FetchEvents a duplicate, which it rejects outright. +// An id several children hold is yielded once; emitting it per child would +// hand FetchEvents a duplicate, which it rejects. func TestUnionIterDedups(t *testing.T) { u := &unionIter{children: []idIter{ sparseSource(1, 3, 5, 7).iter(wholeWindow), @@ -196,9 +184,8 @@ func TestUnionIterAdvanceAndEdges(t *testing.T) { assert.Empty(t, drain(allEmpty)) } -// TestIntersectIterGallops covers the AND: a plain overlap, a -// three-way overlap that forces several alignment passes, disjoint -// children, and an empty child short-circuiting the whole thing. +// The AND: a plain overlap, a three-way overlap forcing several alignment +// passes, disjoint children, and an empty child short-circuiting. func TestIntersectIterGallops(t *testing.T) { t.Run("overlap", func(t *testing.T) { n := &intersectIter{children: []idIter{ @@ -209,8 +196,8 @@ func TestIntersectIterGallops(t *testing.T) { }) t.Run("three way with long gallops", func(t *testing.T) { - // Each child holds a long run the others skip, so alignment - // has to gallop repeatedly and in both orders. + // Each child holds a long run the others skip, so alignment gallops + // repeatedly and in both orders. a := make([]uint32, 0, 400) b := make([]uint32, 0, 400) c := make([]uint32, 0, 400) @@ -266,11 +253,8 @@ func TestIntersectIterAdvance(t *testing.T) { assert.Equal(t, []uint32{6, 8}, drain(n)) } -// TestSingleChildCollapse pins that a one-input union or intersect is -// the input itself, not a wrapper. The materialized path needs the -// same guard for a harder reason (roaring's FastAnd/FastOr Clone a -// singleton input); here it just keeps a one-constraint filter from -// re-scanning a one-element slice per step. +// A one-input union or intersect is the input itself, not a wrapper, which +// keeps a one-constraint filter from re-scanning a one-element slice per step. func TestSingleChildCollapse(t *testing.T) { leaf := sparseSource(1, 2).iter(wholeWindow) assert.Same(t, leaf, unionOf([]idIter{leaf})) @@ -281,9 +265,7 @@ func TestSingleChildCollapse(t *testing.T) { assert.IsType(t, &intersectIter{}, intersectOf([]idIter{leaf, leaf})) } -// TestGroupIterAbsentGroup pins the absent-group signal: a group whose -// every term is missing from the index returns nil, which drops the -// owning filter from the union entirely. +// A group whose every term is missing returns nil, dropping the owning filter. func TestGroupIterAbsentGroup(t *testing.T) { sources := []postings{ sparseSource(1, 2), @@ -299,8 +281,7 @@ func TestGroupIterAbsentGroup(t *testing.T) { assert.Equal(t, []uint32{1, 2, 3}, drain(groupIter(sources, []int{0, 3}, wholeWindow))) } -// TestPostingsEstimate pins the ordering weight on both -// representations: the term's whole-chunk cardinality, window and all. +// The ordering weight is the term's whole-chunk cardinality, window and all. func TestPostingsEstimate(t *testing.T) { assert.Equal(t, uint64(0), postings{}.estimate(), "the absent term weighs nothing") assert.Equal(t, uint64(0), postings{bm: roaring.New()}.estimate()) @@ -310,8 +291,7 @@ func TestPostingsEstimate(t *testing.T) { "cardinality spans containers") } -// TestResolveGroup pins what a group reports about itself: presence, -// and the summed weight that orders its filter's AND. +// What a group reports about itself: presence, and its summed weight. func TestResolveGroup(t *testing.T) { sources := []postings{ sparseSource(1, 2), @@ -337,9 +317,8 @@ func TestResolveGroup(t *testing.T) { assert.Equal(t, uint64(0), g.est) } -// TestFilterIterOrdersRarestFirst pins the driver choice: the rarest -// group leads the AND however the plan named its groups, which is what -// bounds the walk at one round per id of that group. +// The rarest group leads the AND however the plan named its groups, which is +// what bounds the walk at one round per id of that group. func TestFilterIterOrdersRarestFirst(t *testing.T) { sources := []postings{ denseSource(1, 2, 3, 4, 5, 6, 7, 8), // 0: the fat group @@ -369,8 +348,7 @@ func TestFilterIterOrdersRarestFirst(t *testing.T) { drain(filterIter(sources, []candidateGroup{one}, wholeWindow))) } -// planGroups resolves a whole filter, for the tests that drive -// filterIter directly. +// planGroups resolves a whole filter, for tests driving filterIter directly. func planGroups(t *testing.T, sources []postings, plan termPlan) []candidateGroup { t.Helper() groups := make([]candidateGroup, 0, len(plan)) @@ -382,10 +360,8 @@ func planGroups(t *testing.T, sources []postings, plan termPlan) []candidateGrou return groups } -// TestIntersectIterSpills pins the fallback: an AND that overruns its -// budget answers the rest of its window out of the bulk bitmap, -// yielding exactly what the walk would have — no id repeated across -// the seam, none dropped at it — at every budget the seam can fall on. +// An AND that overruns its budget answers the rest of its window from the bulk +// bitmap, yielding exactly what the walk would have at every seam position. func TestIntersectIterSpills(t *testing.T) { sources := []postings{ denseSource(1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12), @@ -401,10 +377,8 @@ func TestIntersectIterSpills(t *testing.T) { assert.Equal(t, []uint32{4, 8, 12}, drain(it), "budget %d", budget) } - // A budget of zero spills on the first round, so the whole answer - // comes from the bulk bitmap — sparse group and all, which the - // aggregation inflates rather than walks — and the window still - // lands at the leaf. + // A budget of zero spills on the first round, so the whole answer comes + // from the bulk bitmap and the window still lands at the leaf. it, ok := filterIter(sources, planGroups(t, sources, plan), IDRange{Start: 0, End: 12}).(*intersectIter) require.True(t, ok) @@ -413,8 +387,7 @@ func TestIntersectIterSpills(t *testing.T) { assert.NotNil(t, it.spilled) } -// TestIntersectIterSpillAdvance pins the seam under advance: a gallop -// that crosses the spill lands where the walk would have. +// A gallop that crosses the spill lands where the walk would have. func TestIntersectIterSpillAdvance(t *testing.T) { sources := []postings{ denseSource(1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12), @@ -436,9 +409,8 @@ func TestIntersectIterSpillAdvance(t *testing.T) { } } -// randomBulkPlan draws a corpus of overlapping terms in both -// representations and one filter's plan over it, the shape family the -// spill has to answer identically to the walk. +// randomBulkPlan draws overlapping terms in both representations and one +// filter's plan over them. func randomBulkPlan(rng *rand.Rand, idSpace int) ([]postings, termPlan) { sources := make([]postings, 1+rng.Intn(5)) for i := range sources { @@ -469,11 +441,10 @@ func randomBulkPlan(rng *rand.Rand, idSpace int) ([]postings, termPlan) { return sources, plan } -// TestFilterIterBulkMatchesWalk is the spill's equivalence gate: over -// randomized plans, source shapes and windows, an AND that spills must -// yield exactly what the unbounded walk yields, and both must equal the -// materialized algebra. The budget is moved rather than the corpus, so -// the same filter is answered every way. +// Over randomized plans, shapes and windows, an AND that spills must yield +// what the unbounded walk yields, and both must equal the materialized +// algebra. The budget moves rather than the corpus, so one filter is answered +// every way. func TestFilterIterBulkMatchesWalk(t *testing.T) { rng := rand.New(rand.NewSource(20260901)) const idSpace = 4000 @@ -507,9 +478,8 @@ func TestFilterIterBulkMatchesWalk(t *testing.T) { want := referenceCandidates([]termPlan{plan}, sources, window) require.Equal(t, want, drain(walk), "trial %d: window %v plan %v", trial, window, plan) - // Budgets in the low single digits put the seam at the start of - // a plan's answer and partway into it, so the join is under test - // and not just its ends. + // Low budgets put the seam at the start of a plan's answer and + // partway into it, so the join is under test and not just its ends. for _, budget := range []uint64{0, 1, 3} { it, n := build(budget) require.Equal(t, want, drain(it), @@ -533,18 +503,15 @@ func TestCandidateIterDropsFilterWithAbsentGroup(t *testing.T) { plans := []termPlan{{{0}, {1}}, {{2}}} assert.Equal(t, []uint32{9}, drain(candidateIter(plans, sources, wholeWindow))) - // Every filter dropped → the exhausted cursor, the - // un-materialized form of union.IsEmpty(). + // Every filter dropped, so the cursor is exhausted. allMissed := []termPlan{{{0}, {1}}} it := candidateIter(allMissed, sources, wholeWindow) _, ok := it.peek() assert.False(t, ok) } -// referenceCandidates is an independent, deliberately naive -// materialized implementation of the same query algebra -// candidateIter answers: OR within a group, AND across a filter's -// groups, OR across filters, AND the window. +// referenceCandidates is an independent, naive materialized implementation of +// the algebra candidateIter answers. func referenceCandidates(plans []termPlan, sources []postings, window IDRange) []uint32 { materialize := func(p postings) *roaring.Bitmap { if bm := p.bitmap(); bm != nil { @@ -588,10 +555,8 @@ func referenceCandidates(plans []termPlan, sources []postings, window IDRange) [ return union.ToArray() } -// TestMatchesSpillYieldsSameStream drives whole Matches calls with the -// alignment budget shrunk, so every filter's AND spills on its first -// rounds, and requires the stream to be the one the unbounded walk -// yields. It puts the seam where it actually sits: under term +// Drives whole Matches calls with the budget shrunk so every AND spills, and +// requires the stream the unbounded walk yields — with the seam under term // planning, the window cap, the batch loop and the post-filter. func TestMatchesSpillYieldsSameStream(t *testing.T) { rng := rand.New(rand.NewSource(20260902)) @@ -625,10 +590,8 @@ func TestMatchesSpillYieldsSameStream(t *testing.T) { "fixture sanity: randomized queries selected too little") } -// TestCandidateIterMatchesMaterializedAlgebra is the differential -// gate: over randomized plans, source shapes (absent / sparse / dense -// / present-but-empty) and windows, the un-materialized tree must -// yield exactly what the bitmap algebra does. +// Over randomized plans, source shapes and windows, the un-materialized tree +// must yield exactly what the bitmap algebra does. func TestCandidateIterMatchesMaterializedAlgebra(t *testing.T) { rng := rand.New(rand.NewSource(20260829)) const idSpace = 400 @@ -684,11 +647,9 @@ func TestCandidateIterMatchesMaterializedAlgebra(t *testing.T) { // ─── the A/B benchmark ────────────────────────────────────────────── -// stubIndex is a Reader over an in-memory mirror and one shared event -// payload: enough for Matches to run end to end without RocksDB, so a -// benchmark measures the match layer rather than the storage tier. -// FetchEvents reuses its result buffer, keeping the per-batch fetch -// cost identical in both directions. +// stubIndex is a Reader over an in-memory mirror and one shared payload, so a +// benchmark measures the match layer rather than the storage tier. FetchEvents +// reuses its buffer, keeping the per-batch fetch cost equal in both directions. type stubIndex struct { mirror *ConcurrentBitmaps count uint32 @@ -878,25 +839,15 @@ func benchMatches(b *testing.B, descending bool) { } } -// BenchmarkMatchesAscending measures the un-materialized iterator tree -// and BenchmarkMatchesDescending its materialized twin — the same -// query, the same page size, the same fetch work, differing only in -// which candidate path Matches takes. The pair is the in-tree A/B for -// the un-materialized path; it stays honest as long as descending -// keeps the bitmap algebra. +// The in-tree A/B for the un-materialized path: the same query, page size and +// fetch work, differing only in which candidate path Matches takes. func BenchmarkMatchesAscending(b *testing.B) { benchMatches(b, false) } func BenchmarkMatchesDescending(b *testing.B) { benchMatches(b, true) } -// ─── the candidate-shape microbench ───────────────────────────────── -// -// The A/B above drives one production query end to end; the matrix -// below isolates the candidate set itself. Both paths answer the same -// synthetic plan over the same mirror — candidateIter's cursor tree -// against unionForFilters' bitmap algebra — with fetch and -// post-filter out of frame, so a shape's number is candidate work -// alone. The shapes are the term geometries the two are expected to -// disagree on: fat partially-overlapping ANDs, a skewed AND, a deep -// AND, and a wide OR. +// The candidate-shape microbench isolates the candidate set itself: both paths +// answer the same synthetic plan over the same mirror, with fetch and +// post-filter out of frame. The shapes are the term geometries the two are +// expected to disagree on. // benchFat is one fat term's cardinality against benchEvents: ~7% of // the domain, the density at which roaring holds a term as bitmap @@ -918,11 +869,10 @@ func (r *benchRand) next() uint64 { return x } -// scatter draws exactly k ascending ids from the residue class -// {i : i ≡ res (mod m), i < domain}, one per fixed stride at a -// jittered offset inside it. Terms built on disjoint residue classes -// interleave at single-id granularity while sharing nothing, so a -// shape's overlap is exactly the class its terms are built to share. +// scatter draws k ascending ids from one residue class, one per stride at a +// jittered offset. Terms on disjoint classes interleave at single-id +// granularity while sharing nothing, so a shape's overlap is exactly the class +// its terms share. func scatter(rng *benchRand, domain, m, res uint32, k int) []uint32 { if k == 0 { return nil @@ -939,12 +889,9 @@ func scatter(rng *benchRand, domain, m, res uint32, k int) []uint32 { return ids } -// fatGroup builds n terms of card ids each: term i draws its private -// ids from residue class base+i, and every term also holds the class -// base+n, so the group's joint intersection is exactly that shared -// class and every pairwise overlap is the same set. mod is the total -// number of classes in play, so several groups can be laid over one -// domain without colliding. +// fatGroup builds n terms of card ids each, drawing private ids from one +// residue class per term plus one class every term holds, so the joint +// intersection is exactly that shared class. func fatGroup(rng *benchRand, domain, mod, base uint32, n, card, shared int) [][]uint32 { common := scatter(rng, domain, mod, base+uint32(n), shared) out := make([][]uint32, n) @@ -1082,12 +1029,9 @@ func benchShapes(domain uint32) []struct { }}, {"h_and2_fat_overlapping", func() *benchShape { rng := benchRand(8) - // The serving default: one selective term AND-ed with a - // near-total one (an event type constrains almost - // nothing). The intersection is nearly the selective term - // itself, so a page comes out of the window's first - // fraction — the shape the cursor tree exists to serve, - // and the one any eager rule must leave alone. + // The serving default: one selective term AND-ed with a near-total + // one, so a page comes out of the window's first fraction. This + // is the shape any eager rule must leave alone. selective := scatter(&rng, domain, 3, 0, fat) nearAll := make([]uint32, 0, domain) for id := range domain { @@ -1164,11 +1108,9 @@ func benchCandidatePage(b *testing.B, s *benchShape) { } } -// benchMaterializedPage is benchCandidatePage's twin over the bitmap -// algebra: build the whole candidate set, then read one page off it. -// The direction of that read is immaterial — the materialization -// dominates and the page is 1000 steps either way — so it reads -// ascending, which makes the two harnesses answer bit for bit. +// benchMaterializedPage is benchCandidatePage's twin over the bitmap algebra: +// build the whole candidate set, then read one page off it. It reads ascending +// so the two harnesses answer bit for bit. func benchMaterializedPage(b *testing.B, s *benchShape) { b.Helper() ctx := context.Background() @@ -1228,12 +1170,10 @@ func TestBenchShapesAgree(t *testing.T) { } } -// TestBatchSizes pins the first-batch hint contract: a positive hint -// sizes the first fetch (a page-sized request is one round trip), a -// wild hint is capped at eight default batches, and later batches -// always use the default. The seam interplay matters: the cap scales -// with matchBatchSize so a test-shrunk batch size cannot be blown past -// by a hint. +// The first-batch hint contract: a positive hint sizes the first fetch, a wild +// one is capped at eight default batches, and later batches use the default. +// The cap scales with matchBatchSize, so a test-shrunk batch cannot be blown +// past by a hint. func TestBatchSizes(t *testing.T) { first, rest := batchSizes(0) require.Equal(t, matchBatchSize, first) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go index 5ff4d3e9b..488b377c2 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go @@ -1,12 +1,8 @@ package event -// Full-Matches differential: the ascending iterator tree and the -// descending materialized union must select the same events, in -// mirrored order, over randomized corpora, filters and windows. The -// iterator-level differential in match_iter_test.go stops at -// candidateIter; this one drives the whole call, so it also covers -// term planning, the postings-vs-LookupKeys seam, the window cap and -// the batch loop. +// Full-Matches differential: the ascending iterator tree and the descending +// materialized union must select the same events, in mirrored order, over +// randomized corpora, filters and windows. import ( "context" @@ -25,19 +21,16 @@ import ( "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/chunk" ) -// diffCorpus is an in-memory chunk carrying one distinct marshaled -// event per id, so the post-filter has real bytes to verify against. +// diffCorpus is an in-memory chunk with one distinct event per id. type diffCorpus struct { raw [][]byte mirror *ConcurrentBitmaps } -// diffReader serves the corpus through LookupKeys only — the -// materializing seam ColdReader and every out-of-package Reader use. +// diffReader serves the corpus through LookupKeys only, the materializing seam. type diffReader struct{ c *diffCorpus } -// diffPostingsReader adds the no-materialize seam HotStore carries, so -// the same query also runs over sparse ids read in place. +// diffPostingsReader adds the no-materialize seam HotStore carries. type diffPostingsReader struct{ diffReader } func (r diffReader) ChunkID() chunk.ID { return chunk.ID(0) } @@ -68,8 +61,7 @@ func (r diffPostingsReader) lookupPostings(_ context.Context, keys []TermKey) ([ } func (r diffReader) FetchEvents(_ context.Context, ids []uint32) ([]Payload, error) { - // The precondition is the point: a dedup bug in the union would - // surface here rather than as a silently doubled result. + // A dedup bug in the union surfaces here, not as a doubled result. if err := validateSortedEventIDs(ids); err != nil { return nil, err } @@ -106,8 +98,7 @@ var ( _ postingReader = diffPostingsReader{} ) -// diffVocab is the small closed vocabulary the corpus and the random -// filters share, so a generated filter has a real chance of matching. +// diffVocab is the closed vocabulary the corpus and the random filters share. type diffVocab struct { contracts [][]byte topics []xdr.ScVal @@ -173,8 +164,8 @@ func newDiffCorpus(t *testing.T, rng *rand.Rand, v *diffVocab, n int) *diffCorpu return c } -// randomFilters builds a filter list over the shared vocabulary, -// including the unconstrained shape that routes to the match-all path. +// randomFilters builds a filter list over the shared vocabulary, including the +// unconstrained shape that routes to the match-all path. func randomFilters(rng *rand.Rand, v *diffVocab) []Filter { filters := make([]Filter, 0, 3) for range 1 + rng.Intn(3) { @@ -218,17 +209,15 @@ func collectOrdinals(t *testing.T, r Reader, filters []Filter, w IDRange, desc b return out } -// TestMatches_AscendingDescendingDifferential drives 400 randomized -// queries through both candidate paths and both index seams. The -// ascending stream reversed must equal the descending stream exactly. +// Drives randomized queries through both candidate paths and both index seams; +// the ascending stream reversed must equal the descending stream. func TestMatches_AscendingDescendingDifferential(t *testing.T) { rng := rand.New(rand.NewSource(20260829)) v := newDiffVocab(t) const corpusSize = 300 corpus := newDiffCorpus(t, rng, v, corpusSize) - // Shrink the batch so multi-batch seams are exercised on a small - // corpus; the stream's contents must not depend on it. + // Shrink the batch so multi-batch seams are exercised on a small corpus. defer func(n int) { matchBatchSize = n }(matchBatchSize) matchBatchSize = 7 @@ -266,23 +255,18 @@ func TestMatches_AscendingDescendingDifferential(t *testing.T) { require.Equal(t, asc, desc, "trial %d: window %v filters %+v", trial, w, filters) } - // Guard against a vacuous pass: the generated queries must - // actually select events, not agree on emptiness. + // Guard against a vacuous pass: the queries must select events. require.Greater(t, matched, 5000, "fixture sanity: randomized queries selected too little") }) } } -// TestMatches_ConcurrentIngestBorrowSafety turns the borrow contract -// into a race-detector gate. The ascending tree's cursors read mirror -// snapshots in place — no clone anywhere — while AddTo publishes new -// termStates on the same keys, including the sparse→dense promotion -// that swaps a term's whole representation. Under -race, any write -// that reached a borrowed snapshot (a COW slip in roaring, a shared -// container mutation) fails the run; without -race the identity check -// still pins that a pinned window's results are immune to concurrent -// ingest past its End. +// Turns the borrow contract into a race-detector gate: the ascending cursors +// read mirror snapshots in place while AddTo publishes new termStates on the +// same keys, including the sparse-to-dense promotion. Under -race any write +// reaching a borrowed snapshot fails the run; without it, the identity check +// still pins that a pinned window is immune to ingest past its End. func TestMatches_ConcurrentIngestBorrowSafety(t *testing.T) { rng := rand.New(rand.NewSource(20260830)) v := newDiffVocab(t) @@ -317,9 +301,8 @@ func TestMatches_ConcurrentIngestBorrowSafety(t *testing.T) { require.NoError(t, err) keysByID[id] = keys } - // Only the pinned window is indexed up front; the writer feeds the - // rest live. The shared vocabulary is small, so most keys cross the - // sparse→dense promotion threshold mid-run. + // Only the pinned window is indexed up front; the writer feeds the rest + // live, and most keys cross the promotion threshold mid-run. for id := range pinned { for _, k := range keysByID[id] { corpus.mirror.AddTo(k, uint32(id)) From e742b01001ccf9451e40dbc150af294c73c865c7 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Mon, 31 Aug 2026 14:45:42 +0000 Subject: [PATCH 16/41] query: the fan-out doc names its override Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- cmd/stellar-rpc/internal/rpcv2/query/resolve.go | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/query/resolve.go b/cmd/stellar-rpc/internal/rpcv2/query/resolve.go index b973c1f31..a150bb0b1 100644 --- a/cmd/stellar-rpc/internal/rpcv2/query/resolve.go +++ b/cmd/stellar-rpc/internal/rpcv2/query/resolve.go @@ -147,9 +147,10 @@ func (a *ReadView) resolveLedgers(c chunk.ID) (LedgerReader, func() error, error // defaultColdEventReadConcurrency is the worker fan-out one cold events read // gets over its packfiles. A page's payload fetch is hundreds of scattered // records with no ordering between them, so serializing them only added their -// latencies together. A constant rather than configuration: it is a property -// of the storage the daemon reads through, not of the query the client asked -// for, which is how the package already spells defaultMaxScanLedgers. +// latencies together. The right value is a property of the storage the daemon +// reads through, never of the query the client asked for: this default is the +// NVMe-measured choice, and a deployment on different storage overrides it at +// wiring time via Registry.SetColdEventReadConcurrency. // // The fan-out is per request, so the worker count multiplies both goroutines // and packfile's coalesced-read buffers by the number of cold pages in flight. From 958a0c5a42bee8af46543740e9d0fa601ec25cf4 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 10:39:07 +0000 Subject: [PATCH 17/41] events: pin the postings accessors' freshness against the lazy dense snapshot MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit #973 made a dense term's reader-visible bitmap a lazily published snapshot: the writer mutates wbm in place under the term mutex and drops pub, and a reader clones once when it finds pub nil. This branch's no-materialize seam — lookupPostings, postings.bitmap, postings.estimate — is now built on that, and its failure mode is invisible to every test the branch already has: sources are resolved once up front and only read, so an accessor handing back a stale or nil pub agrees with the materialized twin on all of them. So the accessors get their own write-then-read-through-the-accessor tests, the only shape that separates the two. They cover a sparse term, the AddTo that promotes one, a dense term whose snapshot the writer has just dropped, the estimate path (fresh count, and still no clone), and the same race the index's own freshness stress runs, but driven through lookupPostings. Each fails against a lookupPostings that returns pub directly, an AddTo that forgets to drop pub, and — for the estimate case — an estimate that reaches for snapshot(). HotStore.lookupPostings' doc said "borrowed snapshot", true of the representation #968 was written against; it now names snapshot's shared clone, matching what #973 put in Reader.LookupKeys. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/stores/event/hot_store.go | 5 +- .../stores/event/postings_freshness_test.go | 329 ++++++++++++++++++ 2 files changed, 332 insertions(+), 2 deletions(-) create mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/postings_freshness_test.go diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go index c515a3849..16fdf22fd 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go @@ -456,8 +456,9 @@ func (h *HotStore) IngestLedgerToBatch( // order never pays Get's roaring.New plus AddMany per sparse term. // // Results are positionally aligned with keys; a miss is the zero postings. -// Same borrowed-snapshot contract as LookupKeys: read-only, valid -// indefinitely. +// Same read-only contract as LookupKeys: a sparse term borrows the mirror's +// published id slice, and a dense one materializes through +// denseState.snapshot, the shared immutable bitmap LookupKeys hands out. func (h *HotStore) lookupPostings(ctx context.Context, keys []TermKey) ([]postings, error) { if h.chunkStore.IsClosed() { return nil, stores.ErrStoreClosed diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/postings_freshness_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/postings_freshness_test.go new file mode 100644 index 000000000..ae29147d7 --- /dev/null +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/postings_freshness_test.go @@ -0,0 +1,329 @@ +package event + +import ( + "runtime" + "sync" + "sync/atomic" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/chunk" +) + +// This file pins ONE property of the no-materialize read seam +// (lookupPostings, postings.bitmap, postings.estimate): a read that +// starts after an AddTo returns observes that AddTo's ids. +// +// It needs its own tests because the differentials cannot see the bug +// it guards. matches_differential_test.go and match_iter_test.go build +// their sources up front and then only read, so a postings accessor +// that returned a stale dense snapshot — the raw pub pointer, nil-or- +// behind after every write — would agree with the materialized twin on +// every one of them. Only write-then-read-through-the-accessor +// separates the two. + +// postingsIDs drains a term's postings through the accessor the query +// engine uses, so a test asserting on it exercises the same path a +// real read takes rather than the representation underneath. +func postingsIDs(t *testing.T, s *ConcurrentBitmaps, key TermKey) []uint32 { + t.Helper() + p := s.lookupPostings(key) + require.True(t, p.present(), "the term must be present in the index") + return drain(p.iter(wholeWindow)) +} + +// denseOf returns the term's denseState, failing the test when the +// term has not promoted. Tests use it only to observe pub, the field +// whose nil-ness is what makes a read take the un-snapshotted path. +func denseOf(t *testing.T, s *ConcurrentBitmaps, key TermKey) *denseState { + t.Helper() + st := s.terms[key].Load() + require.NotNil(t, st.dense, "term must be dense") + return st.dense +} + +// ascending is the id list a term holds after n writes of stride ids. +func ascending(n, stride uint32) []uint32 { + out := make([]uint32, n) + for i := range out { + out[i] = uint32(i) * stride + } + return out +} + +// TestPostings_SparseLookupSeesTheWriteJustMade: on a sparse term, +// every accessor reads the termState the last AddTo published — the +// ids, the count, and the cursor all include the write that just +// returned. +func TestPostings_SparseLookupSeesTheWriteJustMade(t *testing.T) { + s := newTestConcurrentBitmaps() + key := ComputeTermKey([]byte("sparse-freshness"), FieldTopic0) + + want := make([]uint32, 0, promotionThreshold-1) + for i := range uint32(promotionThreshold - 1) { + id := i * 3 + s.AddTo(key, id) + want = append(want, id) + + p := s.lookupPostings(key) + require.True(t, p.present(), "a term written to is present") + require.Nil(t, p.bitmap(), "a sub-threshold term stays sparse") + assert.Equal(t, want, drain(p.iter(wholeWindow)), + "the cursor must yield every id added so far, the last one included") + assert.Equal(t, uint64(len(want)), p.estimate(), + "estimate must count the write that just returned") + } +} + +// TestPostings_PromotionIsVisibleThroughLookupPostings: the AddTo that +// crosses promotionThreshold rebuilds the term as a bitmap, and a +// lookup right after it sees the whole term — the promoting batch +// included — through the dense representation. +func TestPostings_PromotionIsVisibleThroughLookupPostings(t *testing.T) { + s := newTestConcurrentBitmaps() + key := ComputeTermKey([]byte("promotion-freshness"), FieldTopic0) + + const stride = 5 + // One short of the threshold: still a list, no bitmap. + below := ascending(promotionThreshold-1, stride) + s.AddTo(key, below...) + require.Nil(t, s.terms[key].Load().dense, "one below the threshold stays sparse") + require.Nil(t, s.lookupPostings(key).bitmap(), "a sparse term has no bitmap") + require.Equal(t, uint64(len(below)), s.lookupPostings(key).estimate()) + + // The id that promotes. Read the accessors before anything else + // touches the term, so a promotion the lookup missed shows up as a + // short drain rather than being papered over by a later snapshot. + promoting := uint32(promotionThreshold-1) * stride + s.AddTo(key, promoting) + want := append(append([]uint32{}, below...), promoting) + + p := s.lookupPostings(key) + require.True(t, p.present()) + require.NotNil(t, p.bitmap(), "crossing the threshold promotes to a bitmap") + assert.True(t, p.bitmap().Contains(promoting), + "the promoting id must be in the bitmap the promotion built") + assert.Equal(t, want, drain(p.iter(wholeWindow)), + "a lookup straight after promotion yields every id the term holds") + assert.Equal(t, uint64(len(want)), p.estimate(), + "estimate must count the promoting write") + + // And the writes after promotion, which take the in-place dense path. + next := promoting + stride + s.AddTo(key, next) + want = append(want, next) + assert.Equal(t, want, postingsIDs(t, s, key), + "the first dense-mode write must be visible to the next lookup") + assert.Equal(t, uint64(len(want)), s.lookupPostings(key).estimate()) +} + +// TestPostings_DenseLookupSeesWritesSinceLastSnapshot: the case a +// stale read would pass every differential on. A reader publishes a +// snapshot, the writer invalidates it, and the next lookup must clone +// again rather than hand back the pub pointer it finds nil. +func TestPostings_DenseLookupSeesWritesSinceLastSnapshot(t *testing.T) { + s := newTestConcurrentBitmaps() + key := ComputeTermKey([]byte("dense-freshness"), FieldTopic0) + + seed := ascending(promotionThreshold, 4) + s.AddTo(key, seed...) + d := denseOf(t, s, key) + + // A first read publishes the snapshot every later read would + // wrongly reuse. + held := s.lookupPostings(key).bitmap() + require.NotNil(t, held) + require.Same(t, held, d.pub.Load(), "the first read publishes what it returns") + heldCard := held.GetCardinality() + + // Writes spread over fresh containers, so a stale read is wrong by + // more than a bit inside a container the snapshot already shares. + fresh := []uint32{1_000_000, 1_000_000 + 65_536, 1_000_000 + 3*65_536} + s.AddTo(key, fresh...) + require.Nil(t, d.pub.Load(), + "AddTo drops the snapshot, so the next read takes the un-snapshotted path") + + p := s.lookupPostings(key) + bm := p.bitmap() + require.NotNil(t, bm) + for _, id := range fresh { + assert.True(t, bm.Contains(id), + "a lookup after AddTo must observe id %d", id) + assert.False(t, held.Contains(id), + "the snapshot taken before the write stays frozen") + } + assert.Equal(t, heldCard+uint64(len(fresh)), bm.GetCardinality()) + assert.Equal(t, heldCard+uint64(len(fresh)), p.estimate(), + "estimate must count the writes made since the last snapshot") + assert.Subset(t, drain(p.iter(wholeWindow)), fresh, + "the cursor the query engine walks must yield the fresh ids too") +} + +// TestPostings_EstimateCountsUnsnapshottedWritesWithoutCloning pins +// both halves of the planner's cost rule at once: weighing a dense +// term written since its last read reports the CURRENT count, and +// does it without publishing a snapshot — the clone only a read that +// actually walks the term should pay. +func TestPostings_EstimateCountsUnsnapshottedWritesWithoutCloning(t *testing.T) { + s := newTestConcurrentBitmaps() + key := ComputeTermKey([]byte("estimate-freshness"), FieldTopic0) + + seed := ascending(promotionThreshold, 4) + s.AddTo(key, seed...) + d := denseOf(t, s, key) + + // Never read: pub is nil from promotion onwards. + require.Nil(t, d.pub.Load()) + assert.Equal(t, uint64(len(seed)), s.lookupPostings(key).estimate(), + "a term nobody has read yet weighs what it holds") + assert.Nil(t, d.pub.Load(), "estimate must not publish a snapshot") + + // Read once to publish, then invalidate and weigh again. + require.NotNil(t, s.lookupPostings(key).bitmap()) + require.NotNil(t, d.pub.Load()) + + s.AddTo(key, 2_000_000, 2_000_000+65_536) + require.Nil(t, d.pub.Load()) + assert.Equal(t, uint64(len(seed)+2), s.lookupPostings(key).estimate(), + "estimate reads through to the writer's bitmap, never a dropped snapshot") + assert.Nil(t, d.pub.Load(), + "weighing a written-since term must still not clone it") +} + +// TestPostings_FreshnessUnderConcurrentPublishers is +// TestConcurrentBitmaps_FreshnessUnderConcurrentPublishers aimed at +// the accessors #968 added: readers race the writer through +// lookupPostings instead of Get, and each read must observe the id +// whose AddTo returned before the read began. Run with -race. +func TestPostings_FreshnessUnderConcurrentPublishers(t *testing.T) { + s := newTestConcurrentBitmaps() + key := ComputeTermKey([]byte("postings-freshness-stress"), FieldTopic0) + + // ~200 containers, so a republish Clone is long enough for a + // concurrent reader to interleave with it. + seed := make([]uint32, 0, 200*8) + for c := range uint32(200) { + for j := range uint32(8) { + seed = append(seed, c*65_536+j) + } + } + s.AddTo(key, seed...) + seedCard := uint64(len(seed)) + + numReaders := max(16, 2*runtime.GOMAXPROCS(0)) + // firstID + numBatches*idStride + 65_536 stays below MaxUint32. + const ( + numBatches = 500 + firstID = uint32(20_000_000) + idStride = uint32(131_072) + perBatch = 3 // freshnessWriter adds three ids per batch + ) + + var committed, observed, batches atomic.Uint32 + var done atomic.Bool + var reads atomic.Uint64 + var wg sync.WaitGroup + + wg.Go(func() { + defer done.Store(true) + freshnessWriter(t, s, key, freshnessWriterCounters{ + committed: &committed, observed: &observed, batches: &batches, + }, numBatches, firstID, idStride) + }) + + for range numReaders { + wg.Go(func() { + for !done.Load() { + // Sample the batch count BEFORE the read: whatever it + // says is already durable in the writer's bitmap, so + // the read cannot legally weigh less. + finished := batches.Load() + want := committed.Load() + if want == 0 { + runtime.Gosched() + continue + } + p := s.lookupPostings(key) + if !p.present() { + t.Errorf("lookupPostings lost a term that has been written") + return + } + bm := p.bitmap() + if bm == nil { + t.Errorf("lookupPostings returned no bitmap for a dense term") + return + } + reads.Add(1) + // Yield: a read after a write takes the term mutex, + // and unyielding readers starve the writer. + runtime.Gosched() + if !bm.Contains(want) { + t.Errorf("lookupPostings returned a bitmap missing id %d, "+ + "committed before the lookup started (cardinality %d)", + want, bm.GetCardinality()) + return + } + // estimate runs after bitmap, so it can only have + // grown: a stale read of the dropped snapshot would + // come back short. + if est := p.estimate(); est < seedCard+uint64(finished)*perBatch { + t.Errorf("estimate = %d, below the %d ids committed before the read", + est, seedCard+uint64(finished)*perBatch) + return + } else if est < bm.GetCardinality() { + t.Errorf("estimate = %d, below the %d of the bitmap it just handed out", + est, bm.GetCardinality()) + return + } + storeMax(&observed, want) + } + }) + } + + wg.Wait() + t.Logf("postings freshness stress: %d reads, %d batches", reads.Load(), batches.Load()) + assert.Equal(t, uint32(numBatches), batches.Load(), "the writer must finish every batch") + assert.GreaterOrEqual(t, observed.Load(), committed.Load(), + "a reader must observe the final committed batch") + require.Positive(t, reads.Load(), "the stress loop must have done real reads") +} + +// TestHotStore_LookupPostingsSeesTheWriteJustMade carries the same +// property one layer up, through the seam the query planner actually +// calls: HotStore.lookupPostings must reflect an applyLedger that has +// returned, for a sparse term and for a dense one alike. +func TestHotStore_LookupPostingsSeesTheWriteJustMade(t *testing.T) { + h := openHotStoreForTest(t, chunk.ID(0)).store + // index() is the store's documented test-only write hook, which is + // what lets this drive the mirror without an ingest whose term + // derivation would decide the representations for us. + mirror := h.index() + sparseKey := ComputeTermKey([]byte("hot-sparse"), FieldTopic0) + denseKey := ComputeTermKey([]byte("hot-dense"), FieldTopic0) + + mirror.AddTo(denseKey, ascending(promotionThreshold, 4)...) + mirror.AddTo(sparseKey, 1, 2, 3) + + // Publish snapshots, then invalidate the dense one. + first, err := h.lookupPostings(t.Context(), []TermKey{sparseKey, denseKey}) + require.NoError(t, err) + require.Len(t, first, 2) + require.NotNil(t, first[1].bitmap()) + require.NotNil(t, denseOf(t, mirror, denseKey).pub.Load()) + + mirror.AddTo(sparseKey, 4) + mirror.AddTo(denseKey, 3_000_000) + require.Nil(t, denseOf(t, mirror, denseKey).pub.Load()) + + got, err := h.lookupPostings(t.Context(), []TermKey{sparseKey, denseKey}) + require.NoError(t, err) + require.Len(t, got, 2) + assert.Equal(t, []uint32{1, 2, 3, 4}, drain(got[0].iter(wholeWindow)), + "the sparse term must carry the id added since the last lookup") + assert.Equal(t, uint64(4), got[0].estimate()) + assert.True(t, got[1].bitmap().Contains(uint32(3_000_000)), + "the dense term must carry the id added since its snapshot was dropped") + assert.Equal(t, uint64(promotionThreshold+1), got[1].estimate()) +} From 8424908e3691cadb2009f62b07d6e3a5a2977d0c Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 14:17:07 +0000 Subject: [PATCH 18/41] =?UTF-8?q?query:=20drop=20the=20cold=20fan-out=20ov?= =?UTF-8?q?erride=20=E2=80=94=20the=20constant=20is=20the=20contract?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit SetColdEventReadConcurrency, the Registry field behind it and the ReadView copy had no caller anywhere: the "knob" was a setter shape, not a deployment path, and Events resolved a zero back to the constant on every cold read. The cold reader now takes defaultColdEventReadConcurrency directly, and the constant's doc says what it is — the storage-derived fan-out, whose footprint is workers x in-flight cold pages, compiled in until a deployment needs a config knob. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/query/registry.go | 32 ++++--------------- .../internal/rpcv2/query/resolve.go | 19 +++++------ 2 files changed, 15 insertions(+), 36 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/query/registry.go b/cmd/stellar-rpc/internal/rpcv2/query/registry.go index 633de5726..8a951f7cf 100644 --- a/cmd/stellar-rpc/internal/rpcv2/query/registry.go +++ b/cmd/stellar-rpc/internal/rpcv2/query/registry.go @@ -36,11 +36,6 @@ type Registry struct { // cannot leak into another test's pages. maxScanLedgers uint32 - // coldEventReadConcurrency overrides the packfile read fan-out one cold - // events read gets. Zero means defaultColdEventReadConcurrency, resolved - // where the reader is opened. Set through SetColdEventReadConcurrency. - coldEventReadConcurrency int - // latest is the newest fully ingested ledger visible to queries, paired with // its close time so both publish atomically. The ingest loop advances it as // the final step of each per-ledger cycle. Queries read a frozen copy @@ -163,14 +158,6 @@ func NewRegistry(cat *catalog.Catalog, retention geometry.Retention) *Registry { return r } -// SetColdEventReadConcurrency overrides the packfile read fan-out cold events -// reads get; zero restores the default. Call it at wiring time, before views -// are handed out: a view copies the value at acquisition, so a change never -// shifts an in-flight request's fan-out. -func (r *Registry) SetColdEventReadConcurrency(n int) { - r.coldEventReadConcurrency = n -} - // SetLatestLedger publishes seq together with its close time, which callers // without one spell UnknownCloseTime(). func (r *Registry) SetLatestLedger(seq uint32, closeTime CloseTime) { @@ -301,10 +288,6 @@ type ReadView struct { // so a view built without it still bounds its pages. maxScanLedgers uint32 - // coldEventReadConcurrency is the registry's cold events fan-out, copied at - // acquisition like maxScanLedgers. Zero means the default. - coldEventReadConcurrency int - // closers releases every cold reader this view opened (hot facades are // registry-owned and never appear here). Appended by the resolve methods and // by ScanLedgers' walk backstop, drained by Release — a view's resources live @@ -340,14 +323,13 @@ func (r *Registry) NewReadView() (*ReadView, error) { return nil, err } view := &ReadView{ - latest: *latest, - maxScanLedgers: r.maxScanLedgers, - coldEventReadConcurrency: r.coldEventReadConcurrency, - floor: r.retention.FloorAt(lastComplete), - handles: handles, - snap: snap, - catalog: r.catalog, - oldest: &r.oldest, + latest: *latest, + maxScanLedgers: r.maxScanLedgers, + floor: r.retention.FloorAt(lastComplete), + handles: handles, + snap: snap, + catalog: r.catalog, + oldest: &r.oldest, } // The oldest-close-time cache rides along outside the three-load order: it // is a pure optimization whose staleness the seq check in OldestCloseTime diff --git a/cmd/stellar-rpc/internal/rpcv2/query/resolve.go b/cmd/stellar-rpc/internal/rpcv2/query/resolve.go index a150bb0b1..8f6d1543b 100644 --- a/cmd/stellar-rpc/internal/rpcv2/query/resolve.go +++ b/cmd/stellar-rpc/internal/rpcv2/query/resolve.go @@ -148,14 +148,15 @@ func (a *ReadView) resolveLedgers(c chunk.ID) (LedgerReader, func() error, error // gets over its packfiles. A page's payload fetch is hundreds of scattered // records with no ordering between them, so serializing them only added their // latencies together. The right value is a property of the storage the daemon -// reads through, never of the query the client asked for: this default is the -// NVMe-measured choice, and a deployment on different storage overrides it at -// wiring time via Registry.SetColdEventReadConcurrency. +// reads through, never of the query the client asked for; this is the +// NVMe-measured choice. // // The fan-out is per request, so the worker count multiplies both goroutines -// and packfile's coalesced-read buffers by the number of cold pages in flight. -// A deployment with headroom raises it through the Registry's -// coldEventReadConcurrency; zero means this default. +// and packfile's coalesced-read buffers by the number of cold pages in flight +// — the footprint is workers × in-flight cold pages, not workers alone. +// +// The value is a compiled-in constant: changing it is a code change, and a +// config knob gets added when a deployment on different storage needs one. const defaultColdEventReadConcurrency = 8 // Events resolves chunk c's event store as the common event.Reader the @@ -171,12 +172,8 @@ func (a *ReadView) Events(c chunk.ID) (event.Reader, error) { } switch t { case tierCold: - conc := a.coldEventReadConcurrency - if conc == 0 { - conc = defaultColdEventReadConcurrency - } cr, err := event.OpenColdReader(c, a.catalog.Layout().EventsBucketDir(c), - event.ColdReaderOptions{Concurrency: conc}) + event.ColdReaderOptions{Concurrency: defaultColdEventReadConcurrency}) if err != nil { return nil, err } From 72bb3c09a9dbb60062787a816b06cfb647f89ff3 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 14:18:46 +0000 Subject: [PATCH 19/41] packfile: a pool size miss returns the buffer; the offsets cap states its bound MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit sync.Pool.Get removes the buffer it hands back, so the three getters were draining the pool one buffer per undersized hit: a run of opens over growing files emptied it and every later open allocated. A miss now Puts the buffer back before allocating the larger one. maxPooledOffsets said 1<<20 and nothing else. It documents what it has to exceed — recordCount+1 for the largest pack the daemon opens, which chunk geometry fixes and this package cannot import — and what happens silently when a bound like it is crossed: the Put is skipped, the pool drains, and the allocation win reverts. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/packfile/pools.go | 48 ++++++++++++++++--- 1 file changed, 41 insertions(+), 7 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go b/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go index e2dbc90f5..bdc3a5538 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go @@ -9,11 +9,32 @@ package packfile // // Puts are capacity-capped so one pathological file cannot pin an arbitrarily // large array in a pool slot; larger buffers fall to the garbage collector. +// Every cap must therefore sit above what a legitimate packfile needs, because +// crossing one is silent: the Put is skipped, the pool drains, and every open +// allocates afresh — still correct, but without the allocation win these pools +// exist for. import "sync" +// maxPooledOffsets caps the decoded offset table (recordCount+1 int64s, so an +// 8 MiB array at the cap) a Put may retain. What it has to exceed is fixed by +// chunk geometry, which this package cannot reach — packfile is a container +// format with no domain dependencies — so the bound is a constant carrying +// headroom rather than a derived one: +// +// - the ledger cold pack stores one ledger per record, so its table is +// chunk.LedgersPerChunk+1 entries: 10,001; +// - events.pack and index.pack store 128 items per record, so 1<<20 entries +// covers ~134M events, or ~134M distinct index terms, within a single +// chunk — against the ~600K terms a production chunk carries today. +// +// Raise it if chunk geometry grows past that; see the file comment for what +// happens silently if it isn't raised. +const maxPooledOffsets = 1 << 20 // entries (8 MiB backing array) + const ( - maxPooledOffsets = 1 << 20 // entries (8 MiB backing array) + // maxPooledScratch tracks maxPooledOffsets: the FOR-decode scratch is + // recordCount uint32s to the offset table's recordCount+1 int64s. maxPooledScratch = 1 << 20 // entries (4 MiB backing array) maxPooledOpenBuf = 4 << 20 // bytes ) @@ -25,9 +46,16 @@ var ( openBufPool sync.Pool // *[]byte ) +// A size miss hands the pooled buffer back before allocating: Get has already +// removed it from the pool, so returning it is the only thing that keeps a run +// of growing opens from draining the pool one buffer per open. + func getOffsets(n int) []int64 { - if p, _ := offsetsPool.Get().(*[]int64); p != nil && cap(*p) >= n { - return (*p)[:n] + if p, _ := offsetsPool.Get().(*[]int64); p != nil { + if cap(*p) >= n { + return (*p)[:n] + } + putOffsets(*p) } return make([]int64, n) } @@ -43,8 +71,11 @@ func putOffsets(s []int64) { } func getScratch(n int) []uint32 { - if p, _ := scratchPool.Get().(*[]uint32); p != nil && cap(*p) >= n { - return (*p)[:n] + if p, _ := scratchPool.Get().(*[]uint32); p != nil { + if cap(*p) >= n { + return (*p)[:n] + } + putScratch(*p) } return make([]uint32, n) } @@ -58,8 +89,11 @@ func putScratch(s []uint32) { } func getOpenBuf(n int) []byte { - if p, _ := openBufPool.Get().(*[]byte); p != nil && cap(*p) >= n { - return (*p)[:n] + if p, _ := openBufPool.Get().(*[]byte); p != nil { + if cap(*p) >= n { + return (*p)[:n] + } + putOpenBuf(*p) } return make([]byte, n) } From 6409e4f9a91255cfd115781827601756be59eefb Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 14:20:32 +0000 Subject: [PATCH 20/41] events, deps: comments that describe what the code actually does MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four claims this PR left behind or never checked: - Payload's buffer-lifetime contract called the hot FetchEvents bytes "freshly allocated, caller-owned". BatchMultiGet copies the whole batch into one arena, so retaining one Payload pins every payload fetched with it — worth knowing before a caller holds one. - mphf.Close called itself a no-op for OpenBytes. The MPHF is mmapped and Close unmaps; there is no OpenBytes path here. - The roaring pin justified v2.26.0 with two aggregation claims. fastaggregation.go and parallel.go are byte-identical to v2.18.2; what the bump ships is vectorized container kernels, assembly and all, behind the x/sys/cpu and GODEBUG gates. - The block-size block did not say when a size change lands: SSTs get it as they are written, so existing chunks pick 8 KiB up on natural compaction or rotation and a restart moves nothing. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/stores/event/cold_format.go | 2 +- cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go | 5 +++++ cmd/stellar-rpc/internal/rpcv2/stores/event/payload.go | 8 +++++--- go.mod | 9 +++++---- 4 files changed, 16 insertions(+), 8 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go index c93b125cc..7aea0d4c9 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go @@ -482,7 +482,7 @@ func (m *mphf) Lookup(key TermKey) (uint32, error) { return uint32(slot), nil } -// Close releases the index; a no-op for the in-memory OpenBytes path. +// Close unmaps the index file; callers must call it (see openMPHF). func (m *mphf) Close() error { return m.idx.Close() } diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go index 16fdf22fd..fc8d67768 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go @@ -39,6 +39,11 @@ const ( // reads one block to find one key). // - OffsetsCF stores 8-byte (ledger_seq -> event_count) rows in // the tens-of-thousands per chunk — same shape as IndexCF. +// +// A block size takes effect as SSTs are written, so chunks already on disk +// keep whatever size they were built with: changing one of these constants +// reaches a running deployment only as natural compaction or chunk rotation +// rewrites those SSTs, never at restart. const ( dataCFBlockSize = 8 * 1024 indexCFBlockSize = 4 * 1024 diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/payload.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/payload.go index 475ac5e09..1af815eb8 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/payload.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/payload.go @@ -168,9 +168,11 @@ func (p *Payload) MarshalInto(dst []byte) ([]byte, error) { // the event store read paths apply this two ways: // // - FetchEvents passes data that outlives the returned slice — hot from -// rocksdb.BatchMultiGet (freshly allocated, caller-owned), cold by -// cloning the borrowed packfile.ReadItems buffer — so its Payloads are -// safe to retain. +// rocksdb.BatchMultiGet, cold by cloning the borrowed +// packfile.ReadItems buffer — so its Payloads are safe to retain. The +// hot bytes are batch-shared, not per-payload: BatchMultiGet copies one +// batch into one backing array, so retaining a single Payload pins every +// payload fetched with it. // - FetchRange / All pass the iterator's borrowed buffer directly // (rocksdb.IterateRange / packfile.ReadRange, valid only for the // current step), so each yielded Payload is borrowed; a consumer that diff --git a/go.mod b/go.mod index 5e5c47e3d..99666382f 100644 --- a/go.mod +++ b/go.mod @@ -5,10 +5,11 @@ go 1.26 require ( github.com/Masterminds/squirrel v1.5.4 // Minimum v2.18.2 (the FastOr/runContainer16 fix, #527, the fork - // previously carried); v2.26.0 for the no-clone lazy union (#542) - // and fused cardinality-slice aggregation (#559) the descending - // events path leans on. SIMD paths are x/sys/cpu-gated and honor - // GODEBUG=cpu.avx512vpopcntdq=off. + // previously carried). v2.26.0 ships vectorized container kernels, + // hand-written assembly included, gated by x/sys/cpu and GODEBUG + // (cpu.avx512vpopcntdq=off and friends); the aggregation layer this + // package's contracts rest on — fastaggregation.go, parallel.go — is + // byte-identical to v2.18. github.com/RoaringBitmap/roaring/v2 v2.26.0 github.com/aws/aws-sdk-go-v2 v1.45.1 github.com/aws/aws-sdk-go-v2/config v1.31.16 From b75de99a771ec8077b6aa8fc4a715eb48a5931be Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 14:12:31 +0000 Subject: [PATCH 21/41] events: pin roaring's read-only argument contract for shared index bitmaps Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../stores/event/roaring_contract_test.go | 280 ++++++++++++++++++ 1 file changed, 280 insertions(+) create mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go new file mode 100644 index 000000000..7844c91a4 --- /dev/null +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go @@ -0,0 +1,280 @@ +package event + +// roaring_contract_test.go pins the library-level property the event index is +// built on: the aggregation entry points this package calls with shared +// bitmaps must treat those bitmaps as read-only. +// +// ConcurrentBitmaps.Get and denseState.snapshot publish one bitmap to every +// concurrent reader at once and never mutate it afterwards. Readers therefore +// hand the same *roaring.Bitmap to roaring.FastAnd, roaring.FastOr and +// Bitmap.AndAny from many goroutines at once. That is sound only while roaring +// writes exclusively through the receiver (And, AndAny) or into a freshly +// allocated answer (FastAnd, FastOr) — a property roaring documents by +// implication and this file pins by observation. +// +// The pin is on the dependency, not on any caller: it is written against +// roaring alone, so it holds however the match path is built, and it is the +// first thing to run on a roaring version bump. A failure here means the new +// version is unsafe to take, not that a caller is wrong. +// +// Pinned version: github.com/RoaringBitmap/roaring/v2 v2.26.0. + +import ( + "math/rand" + "sync" + "testing" + + "github.com/RoaringBitmap/roaring/v2" + "github.com/stretchr/testify/require" +) + +// sharedBitmaps returns bitmaps in the shape the index publishes: a +// writer-private bitmap marked copy-on-write, and the Clone of it that +// denseState.snapshot hands to readers. The clone shares its containers with +// the writer's copy, so a write reaching either is visible in the other, which +// is what makes an unnoticed mutation here a live corruption bug rather than a +// style violation. +// +// The set spans all three container kinds and several high keys, because the +// aggregation paths branch on container type and on key advance: only a mixed +// corpus reaches all of them. +func sharedBitmaps(t *testing.T) []*roaring.Bitmap { + t.Helper() + rng := rand.New(rand.NewSource(20260909)) + + // Array container: a few hundred scattered ids inside one high key. + sparse := roaring.New() + for range 400 { + sparse.Add(uint32(rng.Intn(1 << 16))) + } + // Bitmap container: dense enough that roaring stores it as a bitset. + dense := roaring.New() + for range 40_000 { + dense.Add(uint32(rng.Intn(1 << 16))) + } + // Run container: contiguous stretches, RunOptimize'd so the container + // really is a run and not an array or a bitmap. + runs := roaring.New() + for i := range uint32(50) { + runs.AddRange(uint64(i*1000), uint64(i*1000+600)) + } + runs.RunOptimize() + // Containers under several high keys, so the aggregation's key-advance + // paths run too. + wide := roaring.New() + for key := range uint32(4) { + for i := range uint32(5000) { + wide.Add(key<<16 + i*7) + } + } + + out := make([]*roaring.Bitmap, 0, 4) + for _, wbm := range []*roaring.Bitmap{sparse, dense, runs, wide} { + wbm.SetCopyOnWrite(true) + snap := wbm.Clone() + require.True(t, snap.GetCopyOnWrite(), + "fixture: the published snapshot must carry the copy-on-write mark") + out = append(out, snap) + } + return out +} + +func bitmapBytes(t *testing.T, bm *roaring.Bitmap) []byte { + t.Helper() + b, err := bm.ToBytes() + require.NoError(t, err) + return b +} + +func bitmapImages(t *testing.T, bms []*roaring.Bitmap) [][]byte { + t.Helper() + out := make([][]byte, len(bms)) + for i, bm := range bms { + out[i] = bitmapBytes(t, bm) + } + return out +} + +func requireUnchanged(t *testing.T, bms []*roaring.Bitmap, before [][]byte, op string) { + t.Helper() + for i, bm := range bms { + require.Equal(t, before[i], bitmapBytes(t, bm), + "%s mutated shared argument %d: roaring can no longer be handed "+ + "denseState.snapshot bitmaps at this version", op, i) + } +} + +// freshRange is the receiver shape the match path hands to AndAny: a bitmap +// this call allocated, holding one contiguous id range, shared with nobody. +func freshRange(lo, hi uint64) *roaring.Bitmap { + acc := roaring.New() + acc.AddRange(lo, hi) + return acc +} + +// TestRoaringContract_AggregationDoesNotMutateInputs is the core pin: every +// aggregation this package calls with shared bitmaps leaves those bitmaps +// byte-identical. +func TestRoaringContract_AggregationDoesNotMutateInputs(t *testing.T) { + t.Parallel() + + // Ranges chosen to hit the shapes that differ inside roaring: a whole + // high key (AddRange produces a full run container, whose iand takes the + // clone-the-other-side branch), a partial one, one the inputs do not + // reach at all, and a span crossing several. + ranges := []struct { + name string + lo, hi uint64 + }{ + {"whole key", 0, 1 << 16}, + {"partial key", 1000, 40_000}, + {"key boundary", 65_000, 66_000}, + {"several keys", 0, 4 << 16}, + {"disjoint key", 40 << 16, 41 << 16}, + } + + for _, w := range ranges { + t.Run("AndAny/"+w.name, func(t *testing.T) { + t.Parallel() + for n := 1; n <= 4; n++ { + shared := sharedBitmaps(t)[:n] + before := bitmapImages(t, shared) + acc := freshRange(w.lo, w.hi) + acc.AndAny(shared...) + requireUnchanged(t, shared, before, "AndAny") + + // The accumulator must equal the definition AndAny documents, + // so a silently wrong answer cannot pass as an unmutated one. + // The leading empty bitmap keeps FastOr off its single-input + // Clone shortcut, whose result would share containers with a + // copy-on-write argument. + want := roaring.FastOr(append([]*roaring.Bitmap{roaring.New()}, shared...)...) + want.And(freshRange(w.lo, w.hi)) + require.Equal(t, bitmapBytes(t, want), bitmapBytes(t, acc), + "AndAny must equal x.And(FastOr(args)) for %d args", n) + } + }) + } + + t.Run("FastAnd", func(t *testing.T) { + t.Parallel() + for n := 2; n <= 4; n++ { + shared := sharedBitmaps(t)[:n] + before := bitmapImages(t, shared) + require.NotNil(t, roaring.FastAnd(shared...)) + requireUnchanged(t, shared, before, "FastAnd") + } + }) + + t.Run("FastOr", func(t *testing.T) { + t.Parallel() + for n := 2; n <= 4; n++ { + shared := sharedBitmaps(t)[:n] + before := bitmapImages(t, shared) + require.NotNil(t, roaring.FastOr(shared...)) + requireUnchanged(t, shared, before, "FastOr") + } + }) + + // AndAny delegates a single argument to And, so And carries the same + // obligation and is pinned separately. + t.Run("And", func(t *testing.T) { + t.Parallel() + for _, shared := range sharedBitmaps(t) { + before := bitmapBytes(t, shared) + acc := freshRange(0, 1<<16) + acc.And(shared) + require.Equal(t, before, bitmapBytes(t, shared), + "Bitmap.And mutated its argument") + } + }) +} + +// TestRoaringContract_AndAnyDoesNotRetainArguments pins the other half of the +// contract: AndAny copies out of its arguments rather than aliasing their +// storage into the receiver. A caller that hands it a reusable scratch bitmap +// depends on the answer surviving that scratch's next Clear. +func TestRoaringContract_AndAnyDoesNotRetainArguments(t *testing.T) { + t.Parallel() + + for _, shared := range sharedBitmaps(t) { + for _, nArgs := range []int{1, 2} { + scratch := roaring.New() + for i := range uint32(3000) { + scratch.Add(i * 3) + } + acc := freshRange(0, 1<<16) + if nArgs == 1 { + acc.AndAny(scratch) + } else { + acc.AndAny(shared, scratch) + } + answer := bitmapBytes(t, acc) + + scratch.Clear() + require.Equal(t, answer, bitmapBytes(t, acc), + "AndAny with %d args aliased a scratch argument's storage into "+ + "the receiver: reusing a scratch bitmap is unsafe", nArgs) + } + } +} + +// TestRoaringContract_ConcurrentReadersShareArguments is the race-detector +// gate. Eight goroutines aggregate over the same shared, copy-on-write-marked +// bitmaps at once, the way concurrent getEvents requests do against one +// denseState snapshot. Under -race any write reaching a shared bitmap fails +// the run; without it, the byte-identity check and the agreement between +// goroutines still catch a mutation. +func TestRoaringContract_ConcurrentReadersShareArguments(t *testing.T) { + t.Parallel() + + const goroutines = 8 + const rounds = 32 + + shared := sharedBitmaps(t) + before := bitmapImages(t, shared) + + // The single-threaded answer every goroutine must reproduce. + seq := freshRange(0, 4<<16) + seq.AndAny(shared...) + wantAndAny := bitmapBytes(t, seq) + wantFastAnd := bitmapBytes(t, roaring.FastAnd(shared...)) + wantFastOr := bitmapBytes(t, roaring.FastOr(shared...)) + + results := make([][3][]byte, goroutines) + var wg sync.WaitGroup + start := make(chan struct{}) + for g := range goroutines { + wg.Go(func() { + <-start + var last [3][]byte + for range rounds { + acc := freshRange(0, 4<<16) + acc.AndAny(shared...) + b, err := acc.ToBytes() + if err != nil { + panic(err) + } + last[0] = b + if b, err = roaring.FastAnd(shared...).ToBytes(); err != nil { + panic(err) + } + last[1] = b + if b, err = roaring.FastOr(shared...).ToBytes(); err != nil { + panic(err) + } + last[2] = b + } + results[g] = last + }) + } + close(start) + wg.Wait() + + requireUnchanged(t, shared, before, "concurrent aggregation") + for g := range goroutines { + require.Equal(t, wantAndAny, results[g][0], "goroutine %d disagreed on AndAny", g) + require.Equal(t, wantFastAnd, results[g][1], "goroutine %d disagreed on FastAnd", g) + require.Equal(t, wantFastOr, results[g][2], "goroutine %d disagreed on FastOr", g) + } +} From 32eee609672f0cc7aff1e7a48fd9c665baa6e7c7 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 14:23:51 +0000 Subject: [PATCH 22/41] events: a slab-stepped match engine beside the cursor tree, proved byte-identical Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/stores/event/slab_match.go | 438 +++++++++++ .../stores/event/slab_match_bench_test.go | 327 ++++++++ .../event/slab_match_differential_test.go | 706 ++++++++++++++++++ 3 files changed, 1471 insertions(+) create mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go create mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_bench_test.go create mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_differential_test.go diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go new file mode 100644 index 000000000..4c8c6e349 --- /dev/null +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go @@ -0,0 +1,438 @@ +package event + +// slab_match.go is a second evaluation engine for the query Matches serves: +// slabMatches. It steps the window one roaring slab at a time — 65536 ids, the +// span of exactly one container — and answers the whole filter algebra inside +// that slab, where match_iter.go pulls candidates through a lazy cursor tree +// (ascending) and match.go materializes a whole-window union (descending). +// +// Clip-early slab evaluation, per slab, per filter: +// +// acc := roaring.New() // ours, shared with nobody +// acc.AddRange(slabLo, slabHi) // the caller's window, clipped to this slab +// acc.AndAny(group0Terms...) // acc ∩ (t1 ∪ t2 ∪ …), in place on acc +// acc.AndAny(group1Terms...) // …AND the next group, still in place +// +// The window is applied first rather than last, so no id outside it is ever +// read. The successive in-place AndAny is the AND across a filter's groups, +// and roaring.FastOr unions the surviving filters' per-slab results. Every +// input to a slab's evaluation is a single container and every intermediate +// the engine allocates holds at most one. +// +// Direction is the slab walk order and nothing else: ascending walks slabs low +// to high and reads each result forward, descending walks them high to low and +// reads each result backward. One code path serves both, with no gallop, no +// alignment budget, no spill and no separate descending machinery. +// +// Laziness is per slab rather than per id. A consumer that stops after one +// page has evaluated only the slabs that page spans, and the cost inside a +// slab is bounded by the containers its inputs hold there rather than by the +// width of the window. The granularity is coarser than the cursor tree's — a +// page ending mid-slab has paid for the whole slab — and bounded above by one +// container's worth of work per input per filter. +// +// The trade is descending over a whole window: the materialized path ANDs the +// chunk-sized terms once with roaring's bulk aggregation, while this one +// re-enters the algebra per slab. On the benchmarks in +// slab_match_bench_test.go, over a 300k-event corpus, that is +8µs on a +// descending page of the two-fat-term AND (9.9µs → 18.1µs) and +170µs on a +// descending full scan of one chunk-sized term measured on candidates alone +// (1.20ms → 1.37ms), which the fetch swallows end to end (10.88ms → 10.93ms). +// Descending pages that stop early are faster (148µs → 128µs), because the +// materialized path pays for the whole chunk before yielding anything. +// Seeding the walk at the slab holding the window's high bound, rather than +// re-clipping the accumulator there, would recover part of the scan cost. +// +// Ownership. acc is built by this call and is the only bitmap ever mutated. +// The term bitmaps handed to AndAny may be shared, copy-on-write-marked mirror +// snapshots (denseState.snapshot), which the index's contract forbids mutating +// or Cloning: AndAny reads them through roaring's read-only container +// accessors and writes only through the receiver, so passing them is safe, and +// roaring_contract_test.go pins that property against the pinned roaring +// version. Sparse terms never become bitmaps beyond the ids that land inside +// the slab under evaluation. + +import ( + "cmp" + "context" + "iter" + "slices" + + "github.com/RoaringBitmap/roaring/v2" +) + +// slabShift sets the slab width as a power of two: 1<<16 is one roaring +// container, so a slab's evaluation touches exactly one container per input +// and every result the engine builds is single-container. +// +// A var rather than a const so in-package tests can shrink it and drive slab +// seams over a small corpus. It never changes what a stream yields. +// +//nolint:gochecknoglobals // test seam; production never writes it +var slabShift uint = 16 + +// slabMatches is Matches over the slab-stepped engine: same arguments, same +// validation, same yields, same order, same match-all and absent-term +// handling. It is a drop-in alternative to Matches on the index-served paths, +// and the match-all path is shared verbatim. +func slabMatches( + ctx context.Context, r Reader, filters []Filter, window IDRange, + descending bool, firstBatch int, +) iter.Seq2[Match, error] { + return func(yield func(Match, error) bool) { + if err := validateMatchCall(ctx, r, filters, window); err != nil { + yield(Match{}, err) + return + } + if window.isEmpty() { + return + } + plans, uniqueKeys, matchAll := planIndexTerms(filters) + // Match-all path: identical to Matches'. The window is dense, so it + // streams Reader.FetchRange without touching the index. + if matchAll { + streamRange(ctx, r, window, descending, firstBatch, yield) + return + } + sources, err := lookupPostings(ctx, r, uniqueKeys) + if err != nil { + yield(Match{}, err) + return + } + st := newSlabStepper(plans, sources, window, descending) + // The twin of Matches' peek/IsEmpty early out: no filter survived + // group resolution, so nothing can match and no slab is worth + // evaluating. + if len(st.filters) == 0 { + return + } + streamSlabs(ctx, r, filters, st, descending, firstBatch, yield) + } +} + +// slabTerms is one of a filter's term groups resolved out of the batched +// lookup and held in whichever representation the index gave it: bitmaps for +// dense (and cold) terms, borrowed id lists for sparse ones. The group's value +// is the union of the two halves. +// +// est is the summed cardinality of the present terms over the whole chunk — +// the same weight match_iter.go's resolveGroup computes, used the same way, to +// order a filter's AND. +type slabTerms struct { + bitmaps []*roaring.Bitmap + lists [][]uint32 + est uint64 +} + +// slabFilter is one filter's groups, ordered rarest first. +type slabFilter struct { + groups []slabTerms +} + +// resolveSlabTerms collects the postings at slots, reporting false when every +// one of them is absent from the index — the signal that the owning filter can +// match nothing, exactly as in resolveGroup. +// +// A present term holding no ids contributes nothing to the union but still +// keeps the group alive, which an absent term's caller-side skip would not do. +func resolveSlabTerms(sources []postings, slots []int) (slabTerms, bool) { + var g slabTerms + present := false + for _, slot := range slots { + p := sources[slot] + if !p.present() { + continue + } + present = true + g.est += p.estimate() + // A dense term is snapshotted once here, for the whole query, so every + // slab reads the same immutable bitmap. + if bm := p.bitmap(); bm != nil { + g.bitmaps = append(g.bitmaps, bm) + continue + } + if len(p.ids) > 0 { + g.lists = append(g.lists, p.ids) + } + } + return g, present +} + +// resolveSlabFilters is the whole planning step: resolve every filter's +// groups, drop the filters that named an entirely absent group, and order each +// survivor's groups rarest first so the accumulator shrinks fastest and a +// group that empties it ends the slab before the fat groups are read. +func resolveSlabFilters(plans []termPlan, sources []postings) []slabFilter { + out := make([]slabFilter, 0, len(plans)) + for _, plan := range plans { + groups := make([]slabTerms, 0, len(plan)) + missed := false + for _, slots := range plan { + g, ok := resolveSlabTerms(sources, slots) + if !ok { + missed = true + break + } + groups = append(groups, g) + } + if missed { + continue + } + slices.SortStableFunc(groups, func(a, b slabTerms) int { + return cmp.Compare(a.est, b.est) + }) + out = append(out, slabFilter{groups: groups}) + } + return out +} + +// slabScratch is the per-query reusable working set of the slab loop: the +// AndAny argument slice, and one bitmap holding whichever sparse ids land in +// the slab under evaluation. +// +// Reusing sparse across groups is safe because AndAny is done with its +// arguments when it returns — it copies out of them and never retains a +// container — which roaring_contract_test.go pins alongside the read-only +// property. +type slabScratch struct { + args []*roaring.Bitmap + sparse *roaring.Bitmap +} + +// inputs returns the AndAny arguments for g over [lo, hi): the group's term +// bitmaps, plus a scratch bitmap for the sparse ids inside the slab when the +// group has any. An empty result means the group holds nothing in this slab, +// so the owning filter matches nothing here. +// +// A group with no sparse terms hands back its own slice with no copy. +func (sc *slabScratch) inputs(g *slabTerms, lo, hi uint32) []*roaring.Bitmap { + if len(g.lists) == 0 { + return g.bitmaps + } + if sc.sparse == nil { + sc.sparse = roaring.New() + } else { + sc.sparse.Clear() + } + hit := false + for _, ids := range g.lists { + // Both bounds are found by binary search, so the borrowed list is + // never copied and never scanned outside the slab. + lower, _ := slices.BinarySearch(ids, lo) + tail := ids[lower:] + upper, _ := slices.BinarySearch(tail, hi) + if sub := tail[:upper]; len(sub) > 0 { + sc.sparse.AddMany(sub) + hit = true + } + } + sc.args = append(sc.args[:0], g.bitmaps...) + if hit { + sc.args = append(sc.args, sc.sparse) + } + return sc.args +} + +// eval returns f's matches inside [lo, hi) as a freshly built bitmap the +// caller owns, or nil when f matches nothing there. +// +// The accumulator starts as the slab window itself and is narrowed group by +// group in place. AndAny is x.And(FastOr(args)) without the intermediate +// union, so one call is a whole group; a single-group filter is therefore one +// AddRange plus one AndAny, with no FastAnd and no clone-the-input shortcut to +// guard against. +func (f *slabFilter) eval(lo, hi uint32, sc *slabScratch) *roaring.Bitmap { + var acc *roaring.Bitmap + for i := range f.groups { + inputs := sc.inputs(&f.groups[i], lo, hi) + if len(inputs) == 0 { + return nil + } + if acc == nil { + acc = roaring.New() + acc.AddRange(uint64(lo), uint64(hi)) + } + acc.AndAny(inputs...) + if acc.IsEmpty() { + return nil + } + } + // acc is nil only for a filter that named no group at all, which takes the + // match-all path upstream and never reaches here. Matching intersectOf, + // the unreachable case is the empty candidate set. + return acc +} + +// slabStepper walks one query's slabs in emission order, evaluating a slab +// only when the consumer has drained the previous one. +type slabStepper struct { + filters []slabFilter + window IDRange + desc bool + + // cursor is the next unevaluated boundary: the inclusive low bound + // ascending, the exclusive high bound descending. + cursor uint32 + done bool + + scratch slabScratch + perFilter []*roaring.Bitmap + + // cur is the current slab's result, held only for its iterator. + cur *roaring.Bitmap + asc roaring.ManyIntIterable + rev roaring.IntIterable +} + +func newSlabStepper( + plans []termPlan, sources []postings, window IDRange, descending bool, +) *slabStepper { + s := &slabStepper{ + filters: resolveSlabFilters(plans, sources), + window: window, + desc: descending, + } + if descending { + s.cursor = window.End + } else { + s.cursor = window.Start + } + return s +} + +// nextBounds returns the next slab's [lo, hi) clipped to the window, walking +// away from the cursor in the query's direction. The first slab is clipped at +// the cursor by these bounds alone — there is no seek. +func (s *slabStepper) nextBounds() (uint32, uint32, bool) { + if s.done { + return 0, 0, false + } + if s.desc { + hi := s.cursor + lo := s.window.Start + // The base of the slab holding hi-1. hi > window.Start >= 0 here, + // because an empty window never reaches the stepper and the walk stops + // at window.Start. + if base := ((uint64(hi) - 1) >> slabShift) << slabShift; base > uint64(lo) { + lo = uint32(base) //nolint:gosec // base < hi <= MaxUint32 + } else { + s.done = true + } + s.cursor = lo + return lo, hi, true + } + lo := s.cursor + hi := s.window.End + // The base of the slab above the one holding lo. + if next := (uint64(lo)>>slabShift + 1) << slabShift; next < uint64(hi) { + hi = uint32(next) //nolint:gosec // next < hi <= MaxUint32 + } else { + s.done = true + } + s.cursor = hi + return lo, hi, true +} + +// evalSlab is the union across filters of their per-slab results, or nil when +// the slab holds nothing. Every input is this call's own bitmap, so FastOr's +// single-input clone shortcut is unreachable and would be harmless anyway. +func (s *slabStepper) evalSlab(lo, hi uint32) *roaring.Bitmap { + s.perFilter = s.perFilter[:0] + for i := range s.filters { + if bm := s.filters[i].eval(lo, hi, &s.scratch); bm != nil { + s.perFilter = append(s.perFilter, bm) + } + } + switch len(s.perFilter) { + case 0: + return nil + case 1: + return s.perFilter[0] + default: + return roaring.FastOr(s.perFilter...) + } +} + +// ensureSlab advances to the next slab that holds a match, reporting false +// once the window is exhausted. +func (s *slabStepper) ensureSlab() bool { + for s.cur == nil { + lo, hi, ok := s.nextBounds() + if !ok { + return false + } + bm := s.evalSlab(lo, hi) + if bm == nil { + continue + } + s.cur = bm + if s.desc { + s.rev = bm.ReverseIterator() + } else { + s.asc = bm.ManyIterator() + } + } + return true +} + +func (s *slabStepper) dropSlab() { + s.cur, s.asc, s.rev = nil, nil, nil +} + +// appendUpTo appends at most n ids in emission order to dst, evaluating slabs +// as it fills, and returns the extended slice. A result shorter than n means +// the query is exhausted. +func (s *slabStepper) appendUpTo(dst []uint32, n int) []uint32 { + for len(dst) < n { + if !s.ensureSlab() { + return dst + } + if s.desc { + for len(dst) < n && s.rev.HasNext() { + dst = append(dst, s.rev.Next()) + } + if !s.rev.HasNext() { + s.dropSlab() + } + continue + } + // NextMany fills the tail directly, so an ascending page is copied out + // of the slab's containers in bulk rather than id by id. It returns + // short only at the end of the bitmap. + want := n - len(dst) + base := len(dst) + dst = slices.Grow(dst, want)[:base+want] + got := s.asc.NextMany(dst[base:]) + dst = dst[:base+got] + if got < want { + s.dropSlab() + } + } + return dst +} + +// streamSlabs is the single streaming loop, shared by both directions: fill +// one internal batch of candidate ordinals out of the stepper, fetch, +// post-filter, yield the survivors. It is streamUnion and streamCandidates +// collapsed into one, because the stepper already hides the direction. +func streamSlabs( + ctx context.Context, r Reader, filters []Filter, st *slabStepper, + descending bool, firstBatch int, yield func(Match, error) bool, +) { + batch, rest := batchSizes(firstBatch) + ids := make([]uint32, 0, batch) + for { + if err := ctx.Err(); err != nil { + yield(Match{}, err) + return + } + ids = st.appendUpTo(ids[:0], batch) + batch = rest + if len(ids) == 0 { + return + } + if !emitBatch(ctx, r, filters, ids, descending, yield) { + return + } + } +} diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_bench_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_bench_test.go new file mode 100644 index 000000000..70b2f2767 --- /dev/null +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_bench_test.go @@ -0,0 +1,327 @@ +package event + +// slab_match_bench_test.go measures the slab engine against the cursor tree at +// two levels, because they answer different questions: +// +// - BenchmarkMatchPage is what a getEvents page costs end to end — candidate +// generation plus the fetch and post-filter both engines share. It is the +// number a request sees, and the shared half dilutes the engine difference +// exactly as production does. +// - BenchmarkMatchCandidates strips the shared half and times only the +// machinery being replaced: the cursor tree and the descending union +// against the slab stepper. +// +// Both run over the shaped corpus from slab_match_differential_test.go, sized +// to span several real 65536-id slabs. + +import ( + "context" + "iter" + "slices" + "testing" + + "github.com/stretchr/testify/require" + + protocol "github.com/stellar/go-stellar-sdk/protocols/rpc" +) + +// matchEngine is the shape both engines share, so a benchmark can name one +// and drive it without branching in the timed loop. +type matchEngine = func( + context.Context, Reader, []Filter, IDRange, bool, int, +) iter.Seq2[Match, error] + +// benchDir is one direction arm of every case. +type benchDir struct { + name string + desc bool +} + +func benchDirs() []benchDir { return []benchDir{{"asc", false}, {"desc", true}} } + +// benchArms pairs each engine with the name its benchmark reports under. +func benchArms() []struct { + name string + engine matchEngine +} { + return []struct { + name string + engine matchEngine + }{{"cursor", Matches}, {"slab", slabMatches}} +} + +// benchCorpusSize spans four and a half slabs, so a page served from the head +// of the window leaves most of the corpus untouched and a full scan crosses +// every slab seam. +const benchCorpusSize = 300_000 + +// The corpus is expensive enough to build that every benchmark shares one. +var benchFixtureCache *shapedFixture + +func benchFixture(tb testing.TB) *shapedFixture { + tb.Helper() + if benchFixtureCache == nil { + benchFixtureCache = newShapedFixture(tb, benchCorpusSize) + } + return benchFixtureCache +} + +// filterCommonPage is the everyday getEvents shape: a chunk-sized contract +// term ANDed with a chunk-sized topic term, beside a second filter naming a +// rare contract whose postings are still a sparse id list. Matches are common +// enough that a 1000-item page fills from the head of the window. +func (f *shapedFixture) filterCommonPage() []Filter { + return []Filter{ + { + ContractID: f.vocab.contracts[1], + Topics: [protocol.MaxTopicCount][]byte{0: f.vocab.topicRaw[1]}, + }, + {ContractID: f.vocab.contracts[2]}, + } +} + +// benchCase is one query shape plus the page size a consumer stops at. A zero +// limit is a full-window scan. +type benchCase struct { + name string + filters []Filter + limit int +} + +func benchCases(f *shapedFixture) []benchCase { + return []benchCase{ + {"common", f.filterCommonPage(), 1000}, + {"spill", f.filterThinOverlap(), 1000}, + {"fullscan", f.filterDenseOnly(), 0}, + } +} + +// BenchmarkMatchPage times one served page, fetch and post-filter included. +func BenchmarkMatchPage(b *testing.B) { + f := benchFixture(b) + r := diffPostingsReader{diffReader{f.corpus}} + for _, c := range benchCases(f) { + for _, dir := range benchDirs() { + for _, arm := range benchArms() { + b.Run(c.name+"/"+dir.name+"/"+arm.name, func(b *testing.B) { + benchPageRun(b, r, arm.engine, c, dir.desc) + }) + } + } + } +} + +func benchPageRun(b *testing.B, r Reader, engine matchEngine, c benchCase, desc bool) { + b.Helper() + ctx := context.Background() + window := IDRange{0, benchCorpusSize} + b.ReportAllocs() + var sink uint32 + for b.Loop() { + n := 0 + for m, err := range engine(ctx, r, c.filters, window, desc, c.limit) { + if err != nil { + b.Fatal(err) + } + sink, n = m.Ordinal, n+1 + if c.limit > 0 && n == c.limit { + break + } + } + } + _ = sink +} + +// BenchmarkMatchCandidates times candidate generation alone: the cursor tree +// and the descending materialized union against the slab stepper, drained in +// the same batch sizes the streaming loops use, with no fetch and no +// post-filter in the way. +func BenchmarkMatchCandidates(b *testing.B) { + f := benchFixture(b) + r := diffPostingsReader{diffReader{f.corpus}} + window := IDRange{0, benchCorpusSize} + + for _, c := range benchCases(f) { + for _, dir := range benchDirs() { + for _, name := range []string{"cursor", "slab"} { + b.Run(c.name+"/"+dir.name+"/"+name, func(b *testing.B) { + ctx := context.Background() + b.ReportAllocs() + var sink int + for b.Loop() { + if name == "slab" { + sink = drainSlabCandidates(ctx, b, r, c, window, dir.desc) + } else { + sink = drainCursorCandidates(ctx, b, r, c, window, dir.desc) + } + } + _ = sink + }) + } + } + } +} + +// drainCursorCandidates mirrors streamCandidates and streamUnion without the +// fetch: same batch sizing, same per-batch id collection, same early stop. +func drainCursorCandidates( + ctx context.Context, b *testing.B, r Reader, c benchCase, window IDRange, desc bool, +) int { + b.Helper() + plans, keys, matchAll := planIndexTerms(c.filters) + if matchAll { + b.Fatal("benchmark filters must reach the index") + } + if desc { + union, err := unionForFilters(ctx, r, plans, keys, window) + if err != nil { + b.Fatal(err) + } + it := union.ReverseIterator() + return drainBatches(c, func(ids []uint32, batch int) []uint32 { + for it.HasNext() && len(ids) < batch { + ids = append(ids, it.Next()) + } + return ids + }) + } + sources, err := lookupPostings(ctx, r, keys) + if err != nil { + b.Fatal(err) + } + cand := candidateIter(plans, sources, window) + return drainBatches(c, func(ids []uint32, batch int) []uint32 { + for len(ids) < batch { + id, ok := cand.peek() + if !ok { + return ids + } + ids = append(ids, id) + cand.next() + } + return ids + }) +} + +// drainSlabCandidates is the same drain over the slab stepper. +func drainSlabCandidates( + ctx context.Context, b *testing.B, r Reader, c benchCase, window IDRange, desc bool, +) int { + b.Helper() + plans, keys, matchAll := planIndexTerms(c.filters) + if matchAll { + b.Fatal("benchmark filters must reach the index") + } + sources, err := lookupPostings(ctx, r, keys) + if err != nil { + b.Fatal(err) + } + st := newSlabStepper(plans, sources, window, desc) + return drainBatches(c, func(ids []uint32, batch int) []uint32 { + return st.appendUpTo(ids, batch) + }) +} + +// drainBatches is the streaming loops' batch cadence with the fetch removed: +// fill up to the batch size, stop when a fill comes back empty or the page is +// full. Both arms share it so the harness cannot favor either. +func drainBatches(c benchCase, fill func(ids []uint32, batch int) []uint32) int { + batch, rest := batchSizes(c.limit) + ids := make([]uint32, 0, batch) + total := 0 + for { + ids = fill(ids[:0], batch) + batch = rest + if len(ids) == 0 { + return total + } + total += len(ids) + if c.limit > 0 && total >= c.limit { + return total + } + } +} + +// ───────────── the in-tree shape matrix, with a slab arm ───────────── + +// BenchmarkCandidateSlab is the third arm of match_iter_test.go's per-shape +// A/B: the same plan, the same postings, the same page fingerprint, answered +// by the slab stepper instead of the cursor tree (BenchmarkCandidateTree) or +// the whole-window union (BenchmarkCandidateMaterialized). The shape matrix +// already holds the fat/thin-overlap geometries the alignment budget was built +// for, so this is the directly comparable number. +func BenchmarkCandidateSlab(b *testing.B) { + for _, sh := range benchShapes(benchEvents) { + b.Run(sh.name, func(b *testing.B) { + benchSlabPage(b, shapeFor(sh.name, sh.build)) + }) + } +} + +func benchSlabPage(b *testing.B, s *benchShape) { + b.Helper() + ctx := context.Background() + b.ReportAllocs() + for b.Loop() { + sources, err := lookupPostings(ctx, s.reader, s.keys) + if err != nil { + b.Fatal(err) + } + st := newSlabStepper(s.plans, sources, s.window, false) + ids := make([]uint32, 0, benchPage) + n, sum := 0, uint64(0) + for n < benchPage { + ids = st.appendUpTo(ids[:0], benchPage-n) + if len(ids) == 0 { + break + } + for _, v := range ids { + n, sum = n+1, sum+uint64(v) + } + } + if n != s.wantCount || sum != s.wantSum { + b.Fatalf("page mismatch: got (%d, %d), want (%d, %d)", + n, sum, s.wantCount, s.wantSum) + } + } +} + +// TestBenchShapesAgreeSlab is TestBenchShapesAgree's slab twin: every geometry +// the microbench above measures must be one the slab stepper answers exactly, +// so a shape can never post a number for a query it gets wrong. +func TestBenchShapesAgreeSlab(t *testing.T) { + const domain = 1 << 16 + for _, sh := range benchShapes(domain) { + t.Run(sh.name, func(t *testing.T) { + s := sh.build() + sources, err := lookupPostings(context.Background(), s.reader, s.keys) + require.NoError(t, err) + want := referenceCandidates(s.plans, sources, s.window) + + st := newSlabStepper(s.plans, sources, s.window, false) + got := []uint32{} + for { + before := len(got) + got = st.appendUpTo(got, before+512) + if len(got) == before { + break + } + } + require.Equal(t, want, got) + require.NotEmpty(t, got, "shape sanity: the plan must select something") + + // The descending arm reads the same set backwards. + rev := newSlabStepper(s.plans, sources, s.window, true) + gotDesc := []uint32{} + for { + before := len(gotDesc) + gotDesc = rev.appendUpTo(gotDesc, before+512) + if len(gotDesc) == before { + break + } + } + slices.Reverse(gotDesc) + require.Equal(t, want, gotDesc) + }) + } +} diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_differential_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_differential_test.go new file mode 100644 index 000000000..279b52584 --- /dev/null +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_differential_test.go @@ -0,0 +1,706 @@ +package event + +// slab_match_differential_test.go holds slabMatches to being +// indistinguishable from Matches. Not "selects the same events" — +// byte-identical Match streams, same order, same ordinals, including every +// truncated prefix a paged consumer would stop on. +// +// Two corpora carry the matrix. The randomized one is the shape +// matches_differential_test.go drives the two existing paths with (300 events, +// 400 trials, both index seams, both directions), re-run at several slab +// widths so a small corpus still crosses many slab seams. The shaped one is +// large enough to span real 65536-id slabs and is built so each named query +// shape — dense-only, sparse-only, mixed, absent, match-all, and the fat/fat +// thin-overlap shape that makes the cursor tree spend its alignment budget — +// is present by construction rather than by luck. + +import ( + "context" + "iter" + "math/rand" + "slices" + "strings" + "testing" + + "github.com/stretchr/testify/require" + + protocol "github.com/stellar/go-stellar-sdk/protocols/rpc" + "github.com/stellar/go-stellar-sdk/xdr" +) + +// drainMatches drains seq into a slice, stopping after limit items when +// limit is positive. The result is compared whole, so it pins the payload +// bytes, the ordinals and their order in one assertion. +func drainMatches(tb testing.TB, seq iter.Seq2[Match, error], limit int) []Match { + tb.Helper() + out := []Match{} + for m, err := range seq { + require.NoError(tb, err) + out = append(out, m) + if limit > 0 && len(out) == limit { + break + } + } + return out +} + +// diffCase is one (filters, window, direction, limit) query the two engines +// must answer identically. +type diffCase struct { + name string + filters []Filter + window IDRange + desc bool + limit int +} + +func requireSameStream(tb testing.TB, r Reader, c diffCase) []Match { + tb.Helper() + ctx := context.Background() + want := drainMatches(tb, Matches(ctx, r, c.filters, c.window, c.desc, c.limit), c.limit) + got := drainMatches(tb, slabMatches(ctx, r, c.filters, c.window, c.desc, c.limit), c.limit) + require.Equal(tb, want, got, + "slabMatches diverged from Matches: case %q window %v desc=%v limit=%d", + c.name, c.window, c.desc, c.limit) + return want +} + +// ───────────────────────── the shaped corpus ───────────────────────── + +// slabShape is one distinct event in the shaped corpus. The corpus assigns a +// shape to every id by rule, so a corpus spanning several slabs costs a couple +// of dozen XDR marshals instead of one per event. +type slabShape struct { + contract int + topic0 int + topic1 int // -1 for an event carrying one topic + evType int +} + +// shapedFixture is the shaped corpus plus the vocabulary its filters are +// written against. +type shapedFixture struct { + corpus *diffCorpus + vocab *diffVocab + n uint32 + // thin holds the ids where the two fat terms of the spill shape overlap. + thin []uint32 + // rareContract and rareTopic hold the ids carrying the sparse terms. + rareContract []uint32 + rareTopic []uint32 +} + +// shapeFor is the corpus's id → event rule. It is written so that: +// +// - contract 0 and contract 1 split the corpus in half: two dense terms. +// - contract 2 is carried by a handful of ids: a term that stays sparse. +// - topic 0 tracks the id's parity, except on the thin-overlap ids, so +// {contract 0} ∧ {topic0 = odd-topic} is two chunk-sized terms meeting on +// a few ids — the shape whose alignment the cursor tree gives up on. +// - a second topic appears on a handful of ids, so the topic-count bucket +// family has one dense bucket and one sparse bucket, making the +// "at least one topic" group a mixed dense/sparse OR. +// - the system event type is rare enough to stay sparse. +func (f *shapedFixture) shapeFor(id uint32) slabShape { + sh := slabShape{topic1: -1} + switch { + case slices.Contains(f.rareContract, id): + sh.contract = 2 + case id%2 == 0: + sh.contract = 0 + default: + sh.contract = 1 + } + if id%2 == 1 || slices.Contains(f.thin, id) { + sh.topic0 = 1 + } + if slices.Contains(f.rareTopic, id) { + sh.topic1 = 2 + } + if id%2000 == 0 { + sh.evType = 0 // system + } else { + sh.evType = 1 // contract + } + return sh +} + +// newShapedFixture builds the shaped corpus over n ids. n is chosen by the +// caller to cross at least one 65536-id slab boundary. +func newShapedFixture(tb testing.TB, n uint32) *shapedFixture { + tb.Helper() + v := newDiffVocabTB(tb) + f := &shapedFixture{vocab: v, n: n} + // Thin-overlap ids: even, spread across the window, and deliberately + // sitting on both sides of a slab boundary. + for _, id := range []uint32{4, 30_000, 65_534, 65_536, 65_538, n - 2} { + if id < n { + f.thin = append(f.thin, id) + } + } + for _, id := range []uint32{1, 12_345, 65_535, 66_000, n - 1} { + if id < n { + f.rareContract = append(f.rareContract, id) + } + } + for id := uint32(0); id < n; id += n/37 + 1 { + f.rareTopic = append(f.rareTopic, id) + } + slices.Sort(f.thin) + slices.Sort(f.rareContract) + slices.Sort(f.rareTopic) + + raws := make(map[slabShape][]byte) + keys := make(map[slabShape][]TermKey) + c := &diffCorpus{ + raw: make([][]byte, n), + mirror: NewConcurrentBitmapsFromBitmaps(NewBitmaps()), + } + idsByKey := make(map[TermKey][]uint32) + for id := range n { + sh := f.shapeFor(id) + raw, ok := raws[sh] + if !ok { + raw = f.marshalShape(tb, sh) + ks, err := TermsForBytes(raw) + require.NoError(tb, err) + raws[sh], keys[sh] = raw, ks + } + c.raw[id] = raw + for _, k := range keys[sh] { + idsByKey[k] = append(idsByKey[k], id) + } + } + for k, ids := range idsByKey { + c.mirror.AddTo(k, ids...) + } + f.corpus = c + return f +} + +func (f *shapedFixture) marshalShape(tb testing.TB, sh slabShape) []byte { + tb.Helper() + var cid xdr.ContractId + copy(cid[:], f.vocab.contracts[sh.contract]) + topics := []xdr.ScVal{f.vocab.topics[sh.topic0]} + if sh.topic1 >= 0 { + topics = append(topics, f.vocab.topics[sh.topic1]) + } + sym := xdr.ScSymbol("data") + ev := xdr.ContractEvent{ + ContractId: &cid, + Type: f.vocab.types[sh.evType], + Body: xdr.ContractEventBody{ + V: 0, + V0: &xdr.ContractEventV0{ + Topics: topics, + Data: xdr.ScVal{Type: xdr.ScValTypeScvSymbol, Sym: &sym}, + }, + }, + } + raw, err := ev.MarshalBinary() + require.NoError(tb, err) + return raw +} + +// newDiffVocabTB is newDiffVocab widened to testing.TB so benchmarks can build +// the same vocabulary. +func newDiffVocabTB(tb testing.TB) *diffVocab { + tb.Helper() + v := &diffVocab{types: []xdr.ContractEventType{ + xdr.ContractEventTypeSystem, + xdr.ContractEventTypeContract, + xdr.ContractEventTypeDiagnostic, + }} + for i := range 4 { + cid := xdr.ContractId{0: byte(0xC0 + i)} + v.contracts = append(v.contracts, cid[:]) + } + for name := range strings.FieldsSeq("alpha beta gamma delta epsilon") { + sym := xdr.ScSymbol(name) + val := xdr.ScVal{Type: xdr.ScValTypeScvSymbol, Sym: &sym} + raw, err := val.MarshalBinary() + require.NoError(tb, err) + v.topics, v.topicRaw = append(v.topics, val), append(v.topicRaw, raw) + } + return v +} + +// named query shapes over the shaped corpus. +func (f *shapedFixture) filterDenseOnly() []Filter { + return []Filter{{ContractID: f.vocab.contracts[0]}} +} + +func (f *shapedFixture) filterSparseOnly() []Filter { + return []Filter{{ContractID: f.vocab.contracts[2]}} +} + +func (f *shapedFixture) filterMixedGroups() []Filter { + // A sparse group AND a dense group inside one filter. + return []Filter{{ + ContractID: f.vocab.contracts[2], + Topics: [protocol.MaxTopicCount][]byte{0: f.vocab.topicRaw[1]}, + }} +} + +func (f *shapedFixture) filterMixedOrGroup() []Filter { + // One group that ORs a dense bucket with a sparse one: the topic-count + // family, where "one topic" is the whole corpus and "two topics" is the + // handful of ids carrying the second topic. + return []Filter{{TopicCount: TopicCountFilter{Count: 1}}} +} + +func (f *shapedFixture) filterThinOverlap() []Filter { + // Two chunk-sized terms meeting on |thin| ids. + return []Filter{{ + ContractID: f.vocab.contracts[0], + Topics: [protocol.MaxTopicCount][]byte{0: f.vocab.topicRaw[1]}, + }} +} + +func (f *shapedFixture) filterAbsentTerm() []Filter { + // contracts[3] is in the vocabulary but never used by the corpus. + return []Filter{{ContractID: f.vocab.contracts[3]}} +} + +func (f *shapedFixture) filterAbsentPlusPresent() []Filter { + return []Filter{ + {ContractID: f.vocab.contracts[3]}, + {ContractID: f.vocab.contracts[2]}, + } +} + +// The absent group's position inside a filter matters: an engine that stops +// resolving groups at the first miss but keeps the filter would still answer +// the leading-miss shape correctly and get the trailing-miss shape wrong. Both +// orders are named, since termGroups emits contract before topics. +func (f *shapedFixture) filterAbsentGroupTrailing() []Filter { + // topicRaw[4] is in the vocabulary but no corpus event carries it. + return []Filter{{ + ContractID: f.vocab.contracts[0], + Topics: [protocol.MaxTopicCount][]byte{0: f.vocab.topicRaw[4]}, + }} +} + +func (f *shapedFixture) filterAbsentGroupLeading() []Filter { + return []Filter{{ + ContractID: f.vocab.contracts[3], + Topics: [protocol.MaxTopicCount][]byte{0: f.vocab.topicRaw[0]}, + }} +} + +// A group whose terms are only partly absent: the corpus carries one and two +// topics, so "at least two" ORs one sparse bucket with three absent ones, +// while "at least three" is a group that is absent in full. +func (f *shapedFixture) filterPartlyAbsentGroup() []Filter { + return []Filter{{TopicCount: TopicCountFilter{Count: 2}}} +} + +func (f *shapedFixture) filterWhollyAbsentGroup() []Filter { + return []Filter{{TopicCount: TopicCountFilter{Count: 3}}} +} + +func (f *shapedFixture) filterUnion() []Filter { + sysType := xdr.ContractEventTypeSystem + return []Filter{ + {ContractID: f.vocab.contracts[2]}, + {EventType: &sysType}, + { + ContractID: f.vocab.contracts[0], + Topics: [protocol.MaxTopicCount][]byte{0: f.vocab.topicRaw[1]}, + }, + } +} + +// ───────────────────────── the shaped matrix ───────────────────────── + +// shapedCorpusSize spans slab 0 whole and part of slab 1, so every window +// bound below is a real slab-relative position rather than a synthetic one. +const shapedCorpusSize = 70_000 + +func TestSlabMatches_ShapedDifferential(t *testing.T) { + f := newShapedFixture(t, shapedCorpusSize) + readers := []struct { + name string + r Reader + }{ + {"lookupKeys", diffReader{f.corpus}}, + {"postings", diffPostingsReader{diffReader{f.corpus}}}, + } + + sysType := xdr.ContractEventTypeSystem + filterShapes := []struct { + name string + filters []Filter + }{ + {"dense only", f.filterDenseOnly()}, + {"sparse only", f.filterSparseOnly()}, + {"mixed groups", f.filterMixedGroups()}, + {"mixed or group", f.filterMixedOrGroup()}, + {"thin overlap", f.filterThinOverlap()}, + {"absent term", f.filterAbsentTerm()}, + {"absent plus present", f.filterAbsentPlusPresent()}, + {"absent group trailing", f.filterAbsentGroupTrailing()}, + {"absent group leading", f.filterAbsentGroupLeading()}, + {"partly absent group", f.filterPartlyAbsentGroup()}, + {"wholly absent group", f.filterWhollyAbsentGroup()}, + {"union of filters", f.filterUnion()}, + {"match all empty slice", nil}, + {"match all wildcard filter", []Filter{{}}}, + {"match all beside constrained", []Filter{{EventType: &sysType}, {}}}, + {"exact topic count", []Filter{{TopicCount: TopicCountFilter{Count: 2, Exact: true}}}}, + } + + const slab = 1 << 16 + windows := []struct { + name string + w IDRange + }{ + {"whole corpus", IDRange{0, shapedCorpusSize}}, + {"empty at zero", IDRange{0, 0}}, + {"empty mid slab", IDRange{12_345, 12_345}}, + {"empty at boundary", IDRange{slab, slab}}, + {"first slab exactly", IDRange{0, slab}}, + {"second slab only", IDRange{slab, shapedCorpusSize}}, + {"cursor mid slab", IDRange{33_333, shapedCorpusSize}}, + {"cursor one below boundary", IDRange{slab - 1, shapedCorpusSize}}, + {"cursor on boundary", IDRange{slab, shapedCorpusSize}}, + {"cursor one above boundary", IDRange{slab + 1, shapedCorpusSize}}, + {"end one below boundary", IDRange{0, slab - 1}}, + {"end on boundary", IDRange{0, slab}}, + {"end one above boundary", IDRange{0, slab + 1}}, + {"single id at boundary", IDRange{slab, slab + 1}}, + {"straddles boundary", IDRange{slab - 3, slab + 3}}, + {"tail", IDRange{shapedCorpusSize - 5, shapedCorpusSize}}, + } + + // Limits chosen against slab 0's yield for the fat filters (~35k): one + // page's worth, and a single item, so both a page that ends mid-slab and + // one that ends on a slab's last id are covered at every window bound. + limits := []int{1, 1000} + + for _, seam := range readers { + t.Run(seam.name, func(t *testing.T) { + for _, fs := range filterShapes { + for _, w := range windows { + for _, desc := range []bool{false, true} { + for _, limit := range limits { + c := diffCase{ + name: fs.name, + filters: fs.filters, + window: w.w, + desc: desc, + limit: limit, + } + requireSameStream(t, seam.r, c) + } + } + } + } + }) + } + + // The bounded matrix above never reaches the end of a fat stream. This pass + // does: whole streams, unlimited, over the windows where "the end" is a + // different thing — the corpus end, a slab boundary, and a window living + // entirely inside the second slab. + t.Run("whole streams", func(t *testing.T) { + r := diffPostingsReader{diffReader{f.corpus}} + for _, fs := range filterShapes { + for _, w := range []IDRange{ + {0, shapedCorpusSize}, + {slab, shapedCorpusSize}, + {slab - 3, slab + 3}, + } { + for _, desc := range []bool{false, true} { + requireSameStream(t, r, diffCase{ + name: fs.name, + filters: fs.filters, + window: w, + desc: desc, + }) + } + } + } + }) +} + +// The shaped corpus must hold the shapes its filters are named for: +// chunk-sized terms, sparse terms below the promotion threshold, and a thin +// overlap. Drift in the corpus rules would otherwise turn the matrix above +// into a weaker test without failing it. +func TestSlabMatches_ShapedFixtureIsWhatItClaims(t *testing.T) { + f := newShapedFixture(t, shapedCorpusSize) + r := diffPostingsReader{diffReader{f.corpus}} + ctx := context.Background() + window := IDRange{0, shapedCorpusSize} + + card := func(filters []Filter) int { + return len(drainMatches(t, Matches(ctx, r, filters, window, false, 0), 0)) + } + + require.Greater(t, card(f.filterDenseOnly()), 30_000, "dense term must be chunk-sized") + require.Len(t, drainMatches(t, + Matches(ctx, r, f.filterSparseOnly(), window, false, 0), 0), len(f.rareContract), + "sparse term must hold exactly the rare ids") + require.Less(t, len(f.rareContract), promotionThreshold, + "the sparse term must stay under the promotion threshold") + require.Len(t, drainMatches(t, + Matches(ctx, r, f.filterThinOverlap(), window, false, 0), 0), len(f.thin), + "the thin overlap must be exactly the constructed ids") + require.Greater(t, len(f.thin), 1) + require.Less(t, len(f.thin), 10, "the overlap must be thin") + require.Zero(t, card(f.filterAbsentTerm()), "the absent term must select nothing") + require.Zero(t, card(f.filterAbsentGroupTrailing()), + "a filter whose second group is absent must select nothing, even though "+ + "its first group is chunk-sized") + require.Zero(t, card(f.filterAbsentGroupLeading()), + "a filter whose first group is absent must select nothing") + require.Zero(t, card(f.filterWhollyAbsentGroup()), + "a group whose every bucket is absent must select nothing") + require.Len(t, drainMatches(t, + Matches(ctx, r, f.filterPartlyAbsentGroup(), window, false, 0), 0), len(f.rareTopic), + "the partly-absent group must select exactly its one present bucket") + + // The spill shape is only the spill shape if the walk really would + // overrun its budget: both sides chunk-sized, the meeting point rare. + fat, err := r.lookupPostings(ctx, []TermKey{ + ComputeTermKey(f.vocab.contracts[0], FieldContractID), + ComputeTermKey(f.vocab.topicRaw[1], topicField(0)), + }) + require.NoError(t, err) + for i, p := range fat { + require.Greater(t, p.estimate(), alignBudget, + "spill-shape term %d must be fatter than the alignment budget", i) + } +} + +// The cursor tree reaches its spill path by budget, so the shaped matrix above +// exercises it only at the default budget. Re-running the thin-overlap shape +// with the budget shrunk forces every AND in the reference engine through +// bulkAnd, which is the other answer slabMatches has to agree with. +func TestSlabMatches_SpilledReferenceAgrees(t *testing.T) { + f := newShapedFixture(t, shapedCorpusSize) + r := diffPostingsReader{diffReader{f.corpus}} + + defer func(n uint64) { alignBudget = n }(alignBudget) + for _, budget := range []uint64{0, 1, 7, 8192} { + alignBudget = budget + for _, w := range []IDRange{ + {0, shapedCorpusSize}, + {1 << 16, shapedCorpusSize}, + {(1 << 16) - 3, (1 << 16) + 3}, + } { + for _, desc := range []bool{false, true} { + for _, limit := range []int{0, 1, 3} { + requireSameStream(t, r, diffCase{ + name: "thin overlap spilled", + filters: f.filterThinOverlap(), + window: w, + desc: desc, + limit: limit, + }) + } + } + } + } +} + +// ───────────────────────── the randomized matrix ───────────────────────── + +// TestSlabMatches_RandomizedDifferential re-runs matches_differential_test.go's +// matrix — same seed shape, same corpus size, same trial count, both index +// seams — asserting byte-identity against Matches rather than internal +// consistency, at several slab widths so a 300-event corpus still crosses +// dozens of slab seams. +func TestSlabMatches_RandomizedDifferential(t *testing.T) { + v := newDiffVocab(t) + const corpusSize = 300 + corpus := newDiffCorpus(t, rand.New(rand.NewSource(20260829)), v, corpusSize) + + // Shrink the batch so multi-batch seams are exercised on a small corpus. + defer func(n int) { matchBatchSize = n }(matchBatchSize) + matchBatchSize = 7 + defer func(s uint) { slabShift = s }(slabShift) + + readers := []struct { + name string + r Reader + }{ + {"lookupKeys", diffReader{corpus}}, + {"postings", diffPostingsReader{diffReader{corpus}}}, + } + // slabShift 2 and 4 put 75 and 19 slab seams inside the corpus; 16 is the + // production width, where the whole corpus is one slab. + for _, shift := range []uint{2, 4, 16} { + slabShift = shift + for _, seam := range readers { + r := seam.r + t.Run(seam.name, func(t *testing.T) { + rng := rand.New(rand.NewSource(int64(20260909 + shift))) + matched := 0 + for trial := range 400 { + matched += randomizedTrial(t, r, v, rng, corpusSize, trial) + } + require.Greater(t, matched, 2000, + "fixture sanity: randomized queries selected too little") + }) + } + } +} + +// randomizedTrial runs one random query through both engines in both +// directions and returns how many matches the ascending run selected. +func randomizedTrial( + t *testing.T, r Reader, v *diffVocab, rng *rand.Rand, corpusSize, trial int, +) int { + t.Helper() + filters := randomFilters(rng, v) + start := uint32(rng.Intn(corpusSize + 1)) + end := start + uint32(rng.Intn(corpusSize+1-int(start))) + limit := []int{0, 0, 1, 3, 17, 200}[rng.Intn(6)] + + matched := 0 + for _, desc := range []bool{false, true} { + got := requireSameStream(t, r, diffCase{ + name: "randomized", + filters: filters, + window: IDRange{Start: start, End: end}, + desc: desc, + limit: limit, + }) + if !desc { + matched = len(got) + } + requireStrictOrder(t, got, desc, trial) + } + return matched +} + +func requireStrictOrder(t *testing.T, got []Match, desc bool, trial int) { + t.Helper() + for i := 1; i < len(got); i++ { + if desc { + require.Greater(t, got[i-1].Ordinal, got[i].Ordinal, + "trial %d: descending ordinals must strictly decrease", trial) + continue + } + require.Less(t, got[i-1].Ordinal, got[i].Ordinal, + "trial %d: ascending ordinals must strictly increase", trial) + } +} + +// The slabShift seam must be invisible in the output: every width reproduces +// the production width's stream exactly. +func TestSlabMatches_SlabWidthIsInvisible(t *testing.T) { + f := newShapedFixture(t, shapedCorpusSize) + r := diffPostingsReader{diffReader{f.corpus}} + ctx := context.Background() + + defer func(s uint) { slabShift = s }(slabShift) + cases := [][]Filter{ + f.filterDenseOnly(), + f.filterSparseOnly(), + f.filterMixedOrGroup(), + f.filterThinOverlap(), + f.filterUnion(), + } + windows := []IDRange{ + {0, shapedCorpusSize}, + {65_000, 67_000}, + {65_536, shapedCorpusSize}, + } + for ci, filters := range cases { + for _, w := range windows { + for _, desc := range []bool{false, true} { + slabShift = 16 + want := drainMatches(t, slabMatches(ctx, r, filters, w, desc, 0), 250) + for _, shift := range []uint{3, 8, 13, 17, 20, 31} { + slabShift = shift + got := drainMatches(t, slabMatches(ctx, r, filters, w, desc, 0), 250) + require.Equal(t, want, got, + "case %d window %v desc=%v: slabShift %d changed the stream", + ci, w, desc, shift) + } + } + } + } +} + +// ───────────────────────── the candidate-set pin ───────────────────────── + +// fetchTracer records every ordinal the engine fetches. +// +// Output equality alone cannot see a candidate-set bug that only widens the +// set: postFilter re-verifies every fetched event against the filters, so a +// superset of the true matches still yields the right stream and only costs +// more I/O. Recording the fetches turns "same answer" into "same work", which +// is the claim a replacement engine has to make. +type fetchTracer struct { + diffPostingsReader + + fetched *[]uint32 +} + +func (r fetchTracer) FetchEvents(ctx context.Context, ids []uint32) ([]Payload, error) { + *r.fetched = append(*r.fetched, ids...) + return r.diffPostingsReader.FetchEvents(ctx, ids) +} + +var ( + _ Reader = fetchTracer{} + _ postingReader = fetchTracer{} +) + +// TestSlabMatches_SameCandidatesFetched pins that the two engines resolve the +// same candidate set, batch for batch and in the same order — not merely the +// same surviving matches. +func TestSlabMatches_SameCandidatesFetched(t *testing.T) { + f := newShapedFixture(t, shapedCorpusSize) + ctx := context.Background() + + const slab = 1 << 16 + shapes := [][]Filter{ + f.filterDenseOnly(), + f.filterSparseOnly(), + f.filterMixedGroups(), + f.filterMixedOrGroup(), + f.filterThinOverlap(), + f.filterAbsentTerm(), + f.filterAbsentGroupTrailing(), + f.filterAbsentGroupLeading(), + f.filterPartlyAbsentGroup(), + f.filterWhollyAbsentGroup(), + f.filterUnion(), + } + windows := []IDRange{ + {0, shapedCorpusSize}, + {slab, shapedCorpusSize}, + {slab - 3, slab + 3}, + {33_333, shapedCorpusSize}, + } + + trace := func( + engine func(context.Context, Reader, []Filter, IDRange, bool, int) iter.Seq2[Match, error], + filters []Filter, w IDRange, desc bool, limit int, + ) []uint32 { + fetched := []uint32{} + r := fetchTracer{diffPostingsReader{diffReader{f.corpus}}, &fetched} + drainMatches(t, engine(ctx, r, filters, w, desc, limit), limit) + return fetched + } + + for si, filters := range shapes { + for _, w := range windows { + for _, desc := range []bool{false, true} { + for _, limit := range []int{0, 1, 1000} { + want := trace(Matches, filters, w, desc, limit) + got := trace(slabMatches, filters, w, desc, limit) + require.Equal(t, want, got, + "shape %d window %v desc=%v limit=%d: the engines fetched "+ + "different candidates", si, w, desc, limit) + } + } + } + } +} From d503ce463ee24ea16acdbc10715565142081e92d Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 14:33:27 +0000 Subject: [PATCH 23/41] =?UTF-8?q?events:=20the=20slab=20walk=20is=20the=20?= =?UTF-8?q?match=20engine=20=E2=80=94=20drop=20the=20cursor=20tree=20and?= =?UTF-8?q?=20the=20descending=20union?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../rpcv2/stores/event/concurrent_bitmaps.go | 4 +- .../stores/event/concurrent_bitmaps_test.go | 43 + .../internal/rpcv2/stores/event/match.go | 217 +-- .../internal/rpcv2/stores/event/match_iter.go | 466 ------- .../rpcv2/stores/event/match_iter_test.go | 1203 ----------------- .../internal/rpcv2/stores/event/match_test.go | 70 +- .../stores/event/matches_differential_test.go | 18 +- .../stores/event/postings_freshness_test.go | 23 +- .../internal/rpcv2/stores/event/slab_match.go | 101 +- .../stores/event/slab_match_bench_test.go | 661 +++++++-- ...ifferential_test.go => slab_match_test.go} | 429 +++--- 11 files changed, 944 insertions(+), 2291 deletions(-) delete mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go delete mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go rename cmd/stellar-rpc/internal/rpcv2/stores/event/{slab_match_differential_test.go => slab_match_test.go} (62%) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go index c3e34f875..da4ff03f3 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go @@ -162,8 +162,8 @@ func (d *denseState) cardinality() uint64 { // read-only contract applies verbatim, and the id slice must likewise be read, // never written or appended to. A dense postings is a view, not a frozen copy: // a later materialization can hold ids a write added since. Callers pin a window -// before the lookup and clip every cursor to it at the leaf (bitmapIter.end), so -// those ids sit above the window and are never yielded. +// before the lookup and intersect every read with it (the slab accumulator's +// range), so those ids sit above the window and are never yielded. type postings struct { ids []uint32 bm *roaring.Bitmap diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps_test.go index 2234c5d33..85a7bfc1b 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps_test.go @@ -4,10 +4,53 @@ import ( "sync" "testing" + "github.com/RoaringBitmap/roaring/v2" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) +// sparseSource and denseSource build the two representations the index holds +// for a term, for tests that need postings without a store behind them. +func sparseSource(ids ...uint32) postings { return postings{ids: ids} } + +func denseSource(ids ...uint32) postings { + bm := roaring.New() + bm.AddMany(ids) + return postings{bm: bm} +} + +// postingIDs reads a term's ids out of whichever representation it holds, +// through the accessor the match path uses. It never returns nil, so an empty +// term compares equal to a materialized bitmap's ToArray(). +func postingIDs(p postings) []uint32 { + if bm := p.bitmap(); bm != nil { + return bm.ToArray() + } + return append([]uint32{}, p.ids...) +} + +// A term present but holding no ids is still present: it contributes nothing +// to a union and does not drop its group, which is what distinguishes it from +// a term absent from the index. +func TestPostingsPresent(t *testing.T) { + assert.False(t, postings{}.present(), "the zero postings is the absent term") + assert.True(t, sparseSource(1).present()) + assert.True(t, denseSource(1).present()) + assert.True(t, postings{bm: roaring.New()}.present(), + "a present-but-empty bitmap is present; it just holds nothing") + assert.Empty(t, postingIDs(postings{bm: roaring.New()})) +} + +// The ordering weight is the term's whole-chunk cardinality, window and all. +func TestPostingsEstimate(t *testing.T) { + assert.Equal(t, uint64(0), postings{}.estimate(), "the absent term weighs nothing") + assert.Equal(t, uint64(0), postings{bm: roaring.New()}.estimate()) + assert.Equal(t, uint64(3), sparseSource(1, 2, 3).estimate()) + assert.Equal(t, uint64(3), denseSource(1, 2, 3).estimate()) + assert.Equal(t, uint64(4), denseSource(1, 2, 3, 1<<20).estimate(), + "cardinality spans containers") +} + // newTestConcurrentBitmaps builds an empty ConcurrentBitmaps via the // only remaining constructor (production always converts from a // warmup/backfill-built Bitmaps). diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go index 2599938b1..a682653b9 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go @@ -11,21 +11,16 @@ package event // then stream in internal batches. On the cold path this is one // MPHF+index.pack round trip per Matches call, not per batch. // -// The candidate set is built one of two ways. Ascending pulls ids through the -// un-materialized iterator tree in match_iter.go. Descending materializes a -// union bitmap, because roaring's reverse iterator offers no gallop to build -// ascending-style combinators on. +// The candidate set comes from the slab-stepped engine in slab_match.go, which +// answers both directions from one walk over the window. import ( "bytes" - "cmp" "context" "fmt" "iter" "slices" - "github.com/RoaringBitmap/roaring/v2" - protocol "github.com/stellar/go-stellar-sdk/protocols/rpc" "github.com/stellar/go-stellar-sdk/xdr" ) @@ -220,7 +215,7 @@ type Match struct { } // termPlan is a filter's termGroups resolved to slots in the batched -// LookupKeys result. +// term lookup's result. type termPlan [][]int // batchSizes resolves the first and following internal batch sizes from the @@ -280,31 +275,18 @@ func Matches( streamRange(ctx, r, window, descending, firstBatch, yield) return } - // Direction splits the plan from here; see the file header. - if descending { - union, err := unionForFilters(ctx, r, plans, uniqueKeys, window) - if err != nil { - yield(Match{}, err) - return - } - if union.IsEmpty() { - return - } - streamUnion(ctx, r, filters, union, firstBatch, yield) - return - } sources, err := lookupPostings(ctx, r, uniqueKeys) if err != nil { yield(Match{}, err) return } - candidates := candidateIter(plans, sources, window) - // The twin of the descending path's union.IsEmpty(): one peek - // settles whether anything matches, reading no further. - if _, ok := candidates.peek(); !ok { + st := newSlabStepper(plans, sources, window, descending) + // No filter survived group resolution, so nothing can match and no + // slab is worth evaluating. + if len(st.filters) == 0 { return } - streamCandidates(ctx, r, filters, candidates, firstBatch, yield) + streamSlabs(ctx, r, filters, st, descending, firstBatch, yield) } } @@ -403,165 +385,6 @@ func lookupPostings(ctx context.Context, r Reader, keys []TermKey) ([]postings, return sources, nil } -// unionForFilters materializes the descending path's candidate set over the -// plan planIndexTerms resolved. The result is never a borrowed mirror -// snapshot, because the window AND allocates on the borrowing path, so -// downstream iteration is safe. The ascending path uses candidateIter instead. -func unionForFilters( - ctx context.Context, r Reader, filterPlans []termPlan, uniqueKeys []TermKey, - window IDRange, -) (*roaring.Bitmap, error) { - // ───── 2. Single batched lookup for all unique terms ───── - bitmaps, err := r.LookupKeys(ctx, uniqueKeys) - if err != nil { - return nil, fmt.Errorf("events: query lookup: %w", err) - } - - // ───── 3. Per-filter intersect ───── - // - // If a whole group is absent from the index (every bitmap in it is - // nil), that filter's intersection is empty — skip it without - // contributing to the union. - // - // Bitmap ownership in perFilter is mixed: - // - Single-constraint filter: we borrow bitmaps[s] directly (a - // mirror snapshot from LookupKeys), skipping FastAnd's Clone. - // - Multi-constraint filter: FastAnd allocates a fresh result. - // Either way the downstream union (FastOr) and the window AND never - // mutate their inputs, so a borrowed entry stays valid through - // the rest of the function. FastAnd never mutates its inputs - // either, so the same bitmap may appear across multiple filters - // safely. - perFilter := make([]*roaring.Bitmap, 0, len(filterPlans)) - for _, plan := range filterPlans { - inputs := make([]*roaring.Bitmap, 0, len(plan)) - missed := false - for _, slots := range plan { - group := unionSlots(bitmaps, slots) - if group == nil { - missed = true - break - } - inputs = append(inputs, group) - } - if missed { - continue - } - if len(inputs) == 1 { - perFilter = append(perFilter, inputs[0]) - continue - } - // FastAnd intersects left-to-right — putting the smallest - // bitmap first shrinks the accumulator fastest. roaring's own - // docs call this out as the recommended caller-side prep. - slices.SortFunc(inputs, func(a, b *roaring.Bitmap) int { - return cmp.Compare(a.GetCardinality(), b.GetCardinality()) - }) - perFilter = append(perFilter, roaring.FastAnd(inputs...)) - } - - if len(perFilter) == 0 { - return roaring.New(), nil - } - - // ───── 4. Union across filters ───── - // Single-filter case: FastOr would Clone — skip it and use the - // already-computed bitmap directly. That bitmap may be borrowed - // (from LookupKeys), so step 5's window And uses the fresh-result - // variant on that path to avoid mutating shared state. - var union *roaring.Bitmap - singleFilter := len(perFilter) == 1 - if singleFilter { - union = perFilter[0] - } else { - union = roaring.FastOr(perFilter...) - } - - // ───── 5. Apply the event-ID window ───── - // - // The window AND enforces the caller's pinned range. It also clips - // phantom IDs from a concurrent hot-store ingest: the mirror - // publishes entries before offsets, so LookupKeys can briefly - // surface IDs past EventCount. The AND keeps the stream strictly - // within the snapshot the caller pinned at request entry. - // - // This covers the multi-term group too. Its bitmaps are separate - // mirror snapshots taken at different instants, but an event never - // moves between the terms of one group once ingested, so a torn read - // across them can only surface IDs past the pinned End. - rangeBM := roaring.New() - rangeBM.AddRange(uint64(window.Start), uint64(window.End)) - if singleFilter { - union = roaring.And(union, rangeBM) // fresh result; union may be borrowed - } else { - union.And(rangeBM) // FastOr output is owned; in-place is fine - } - return union, nil -} - -// streamUnion walks the descending path's materialized union bitmap in -// internal batches: collect candidate ordinals up to the batch size, fetch, -// post-filter, yield the survivors. Stepping one id at a time is fine here, -// because the fetch I/O dominates the loop. -func streamUnion( - ctx context.Context, r Reader, filters []Filter, union *roaring.Bitmap, - firstBatch int, yield func(Match, error) bool, -) { - it := union.ReverseIterator() - batch, rest := batchSizes(firstBatch) - ids := make([]uint32, 0, batch) - for { - if err := ctx.Err(); err != nil { - yield(Match{}, err) - return - } - ids = ids[:0] - for it.HasNext() && len(ids) < batch { - ids = append(ids, it.Next()) - } - batch = rest - if len(ids) == 0 { - return - } - if !emitBatch(ctx, r, filters, ids, true, yield) { - return - } - } -} - -// streamCandidates is streamUnion's ascending twin over the un-materialized -// iterator tree. Candidates are pulled from the tree as the batch fills, so a -// consumer that stops after one page never touched the postings past it. -func streamCandidates( - ctx context.Context, r Reader, filters []Filter, candidates idIter, - firstBatch int, yield func(Match, error) bool, -) { - batch, rest := batchSizes(firstBatch) - ids := make([]uint32, 0, batch) - for { - if err := ctx.Err(); err != nil { - yield(Match{}, err) - return - } - ids = ids[:0] - for len(ids) < batch { - id, ok := candidates.peek() - if !ok { - break - } - ids = append(ids, id) - candidates.next() - } - batch = rest - if len(ids) == 0 { - return - } - if !emitBatch(ctx, r, filters, ids, false, yield) { - return - } - } -} - // emitBatch fetches one batch of candidate ordinals, drops the bitmap-side // false positives and yields the survivors, reporting whether the stream // should continue. FetchEvents requires ascending ids, so a descending batch @@ -666,30 +489,6 @@ func CountDistinctTerms(filters []Filter) int { return len(unique) } -// unionSlots ORs the bitmaps at slots, and returns nil when every one -// of them is absent from the index. A lone present bitmap is borrowed -// rather than cloned, like the single-constraint path in -// unionForFilters. -func unionSlots(bitmaps []*roaring.Bitmap, slots []int) *roaring.Bitmap { - if len(slots) == 1 { - return bitmaps[slots[0]] - } - present := make([]*roaring.Bitmap, 0, len(slots)) - for _, s := range slots { - if bitmaps[s] != nil { - present = append(present, bitmaps[s]) - } - } - switch len(present) { - case 0: - return nil - case 1: - return present[0] - default: - return roaring.FastOr(present...) - } -} - // indexOfOrAddTerm returns the index of key inside *keys, appending // it first if absent. func indexOfOrAddTerm(keys *[]TermKey, key TermKey) int { diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go deleted file mode 100644 index 5865c5d52..000000000 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter.go +++ /dev/null @@ -1,466 +0,0 @@ -package event - -// match_iter.go is the ascending match path's un-materialized query plan: a -// tree of peekable ascending cursors that pulls candidate event IDs straight -// out of the index's postings, in order, with no intermediate bitmap. -// -// Ascending only. roaring's reverse iterator has no AdvanceIfNeeded, so there -// is no reverse gallop to build these combinators on, and descending keeps -// the materialized path in match.go. - -import ( - "cmp" - "slices" - - "github.com/RoaringBitmap/roaring/v2" -) - -// idIter is a peekable cursor over a strictly ascending run of chunk-relative -// event IDs. Nothing in the tree reads further than the consumer pulls: peek -// resolves exactly one id and the combinators cache it rather than looking -// ahead, so a loop that stops after N ids touched only what those N needed. -type idIter interface { - // peek returns the id under the cursor without consuming it. ok - // is false once the cursor is exhausted, which is permanent. - peek() (id uint32, ok bool) - // next steps past the id peek returns. A no-op on an exhausted - // cursor. - next() - // advance moves the cursor to the first id >= floor, exhausting - // it when there is none. Never moves backwards, so a floor at or - // below the current id is a no-op. - advance(floor uint32) -} - -// emptyIter is the permanently exhausted cursor: an absent term, or a query -// where no filter survived group resolution. -type emptyIter struct{} - -func (emptyIter) peek() (uint32, bool) { return 0, false } -func (emptyIter) next() {} -func (emptyIter) advance(uint32) {} - -// sliceIter walks the hot mirror's sparse-term postings list in place, rather -// than inflating it into a roaring bitmap per lookup. The window is applied by -// reslicing at construction, so the cursor carries no bounds check and never -// copies the array. The mirror publishes a fresh slice on every AddTo and -// never mutates a published one, so the borrowed array is immutable for the -// cursor's lifetime. -type sliceIter struct { - ids []uint32 - i int -} - -func (s *sliceIter) peek() (uint32, bool) { - if s.i >= len(s.ids) { - return 0, false - } - return s.ids[s.i], true -} - -func (s *sliceIter) next() { - if s.i < len(s.ids) { - s.i++ - } -} - -func (s *sliceIter) advance(floor uint32) { - if s.i >= len(s.ids) || s.ids[s.i] >= floor { - return - } - // Binary search over the unread tail, so one gallop can skip most of it. - j, _ := slices.BinarySearch(s.ids[s.i:], floor) - s.i += j -} - -// bitmapIter walks a roaring bitmap through the library's IntPeekable cursor. -// PeekNext and AdvanceIfNeeded read through getContainerAtIndex rather than -// the writable accessor, so they are on the COW-safe read list in -// ConcurrentBitmaps.Get's contract and are safe on a borrowed mirror snapshot. -// -// end clamps the caller's pinned window at the leaf, so no combinator above -// ever aligns on ids outside it. That is also what clips phantom IDs from a -// concurrent ingest: the mirror publishes index entries before offsets, so a -// lookup can briefly surface IDs past EventCount, and the stream must stay -// inside the snapshot the caller pinned at request entry. -type bitmapIter struct { - it roaring.IntPeekable - end uint32 -} - -func (b *bitmapIter) peek() (uint32, bool) { - // PeekNext is only defined while HasNext holds. - if !b.it.HasNext() { - return 0, false - } - if v := b.it.PeekNext(); v < b.end { - return v, true - } - return 0, false -} - -func (b *bitmapIter) next() { - if _, ok := b.peek(); ok { - b.it.Next() - } -} - -func (b *bitmapIter) advance(floor uint32) { - b.it.AdvanceIfNeeded(floor) -} - -// iter returns an ascending cursor over the postings clipped to window, -// materializing nothing. Callers must check present() first: absent postings -// are the group-missed signal, not an empty cursor. -func (p postings) iter(window IDRange) idIter { - if bm := p.bitmap(); bm != nil { - it := bm.Iterator() - it.AdvanceIfNeeded(window.Start) - return &bitmapIter{it: it, end: window.End} - } - ids := p.ids - lo, _ := slices.BinarySearch(ids, window.Start) - ids = ids[lo:] - // BinarySearch lands at or above the target, dropping exactly the ids at - // or past the window's exclusive End. - hi, _ := slices.BinarySearch(ids, window.End) - return &sliceIter{ids: ids[:hi]} -} - -// unionIter is the OR of its children: the ascending merge of their ids, with -// equal ids across children collapsed to one. Children are scanned linearly -// for the minimum rather than heaped, because advance has to move all of them -// regardless and K is single digits. -type unionIter struct { - children []idIter - cur uint32 - ok bool - primed bool -} - -func (u *unionIter) peek() (uint32, bool) { - if !u.primed { - u.cur, u.ok = 0, false - for _, c := range u.children { - if v, ok := c.peek(); ok && (!u.ok || v < u.cur) { - u.cur, u.ok = v, true - } - } - u.primed = true - } - return u.cur, u.ok -} - -func (u *unionIter) next() { - v, ok := u.peek() - if !ok { - return - } - // Every child sitting on the winning id steps past it, so an id several - // children hold is yielded once. Stepping only the winner would re-emit - // it from each of the others, and FetchEvents rejects a duplicate. - for _, c := range u.children { - if cv, cok := c.peek(); cok && cv == v { - c.next() - } - } - u.primed = false -} - -func (u *unionIter) advance(floor uint32) { - if v, ok := u.peek(); !ok || v >= floor { - return - } - for _, c := range u.children { - c.advance(floor) - } - u.primed = false -} - -// intersectIter is the AND of its children by galloping alignment: every child -// is advanced to the running maximum of the peeks until they all agree on one -// id. A child that exhausts ends the intersection permanently. -// -// Alignment is bounded when bulk is set. A walk costs up to one round per id -// of the leading child, which on a thin overlap between chunk-sized groups is -// spent on a window the consumer never reaches. Once the budget is gone the -// AND hands the rest of its window to bulk, whose cost is set by the -// containers the terms span. A nil bulk leaves the alignment unbounded. -type intersectIter struct { - children []idIter - cur uint32 - ok bool - primed bool - - bulk *bulkAnd - budget uint64 - rounds uint64 - spilled idIter -} - -func (n *intersectIter) peek() (uint32, bool) { - if n.spilled != nil { - return n.spilled.peek() - } - if !n.primed { - n.cur, n.ok = n.align() - if n.spilled != nil { - return n.spilled.peek() - } - n.primed = true - } - return n.cur, n.ok -} - -// align raises every child to the smallest id all of them hold, or -// spills to the bulk answer when it runs out of rounds first. -func (n *intersectIter) align() (uint32, bool) { - cand, ok := n.children[0].peek() - if !ok { - return 0, false - } - for { - if n.bulk != nil { - n.rounds++ - if n.rounds > n.budget { - // A child raises the floor only past ids it does not hold, - // and cand only rises, so nothing in the intersection was - // skipped: the bulk answer from cand up is the rest of it. - n.spilled = n.bulk.iter(cand) - n.bulk = nil - return 0, false - } - } - raised := false - for _, c := range n.children { - c.advance(cand) - v, cok := c.peek() - if !cok { - return 0, false - } - if v > cand { - // Raise the floor and repeat the pass, so children already - // visited are pulled up too. cand strictly increases per - // pass, which is what terminates the loop. - cand, raised = v, true - } - } - if !raised { - return cand, true - } - } -} - -func (n *intersectIter) next() { - v, ok := n.peek() - if !ok { - return - } - if n.spilled != nil { - n.spilled.next() - return - } - // Post-align every child sits on v; step them all past it. - for _, c := range n.children { - if cv, cok := c.peek(); cok && cv == v { - c.next() - } - } - n.primed = false -} - -func (n *intersectIter) advance(floor uint32) { - if v, ok := n.peek(); !ok || v >= floor { - return - } - if n.spilled != nil { - n.spilled.advance(floor) - return - } - for _, c := range n.children { - c.advance(floor) - } - n.primed = false -} - -// unionOf and intersectOf collapse a single input to the input itself, rather -// than wrapping it in a combinator that re-scans a one-element slice per step. -// The materialized path needs the same guard for a harder reason: roaring's -// FastAnd and FastOr have historically Cloned a single-input slice. -func unionOf(children []idIter) idIter { - switch len(children) { - case 0: - return emptyIter{} - case 1: - return children[0] - default: - return &unionIter{children: children} - } -} - -func intersectOf(children []idIter) idIter { - switch len(children) { - case 0: - // Unreachable: a filter naming no term group takes the match-all path. - return emptyIter{} - case 1: - return children[0] - default: - return &intersectIter{children: children} - } -} - -// candidateIter assembles one Matches call's ascending candidate cursor out of -// the batched lookup's postings: terms within a group OR, a filter's groups -// AND, filters OR, clamped to window. A filter with an entirely absent group -// contributes nothing; if that leaves no filter the result is the exhausted -// cursor. -func candidateIter(plans []termPlan, sources []postings, window IDRange) idIter { - perFilter := make([]idIter, 0, len(plans)) - for _, plan := range plans { - groups := make([]candidateGroup, 0, len(plan)) - missed := false - for _, slots := range plan { - g, ok := resolveGroup(sources, slots) - if !ok { - missed = true - break - } - groups = append(groups, g) - } - if missed { - continue - } - perFilter = append(perFilter, filterIter(sources, groups, window)) - } - return unionOf(perFilter) -} - -// candidateGroup is one of a filter's resolved groups: where its terms live in -// the batched lookup, and the weight that orders the filter's AND. -type candidateGroup struct { - slots []int - est uint64 -} - -// resolveGroup weighs the postings at slots, reporting false when every one is -// absent from the index, which is the signal that the owning filter can match -// nothing. The weight sums the present terms' cardinalities, an upper bound -// the OR's dedup can only lower; it orders an intersection and is never read -// as a count. -func resolveGroup(sources []postings, slots []int) (candidateGroup, bool) { - g := candidateGroup{slots: slots} - present := false - for _, slot := range slots { - p := sources[slot] - if !p.present() { - continue - } - present = true - g.est += p.estimate() - } - return g, present -} - -// alignBudget is how many alignment rounds one filter's AND may spend before -// it spills to the bulk answer. It separates filters that yield slowly but -// steadily, where the walk still finishes, from those whose alignment crosses -// a chunk to find a handful of matches. -// -// A var rather than a const so in-package tests can shrink it to force the -// spill. It never changes what a stream yields. -// -//nolint:gochecknoglobals // test seam; production never writes it -var alignBudget uint64 = 8192 - -// filterIter builds one filter's candidate cursor: the AND of its groups, -// rarest first, bounded by alignBudget. Alignment seeds its floor from the -// leading child, so that cursor is the one every barren round steps forward. -// Intersection is commutative, so the order only changes how fast the gallop -// converges — and it makes the budget's bound the rarest group's cardinality -// rather than the fattest's. -func filterIter(sources []postings, groups []candidateGroup, window IDRange) idIter { - slices.SortStableFunc(groups, func(a, b candidateGroup) int { - return cmp.Compare(a.est, b.est) - }) - children := make([]idIter, len(groups)) - for i := range groups { - children[i] = groupIter(sources, groups[i].slots, window) - } - if len(children) < 2 { - return intersectOf(children) - } - return &intersectIter{ - children: children, - bulk: &bulkAnd{sources: sources, groups: groups, end: window.End}, - budget: alignBudget, - } -} - -// bulkAnd is a filter's AND as roaring's aggregation would answer it, held -// unevaluated beside the walk. It is the walk's bound: the aggregation's cost -// is set by the containers the terms span, not by the ids inside them. -type bulkAnd struct { - sources []postings - groups []candidateGroup - end uint32 -} - -// iter computes the filter's candidate set and returns a cursor over the part -// of it at or above floor. The inputs may be borrowed mirror snapshots: -// FastAnd and FastOr read them without writing through, and never see the -// single-element slice that would make them Clone, so what comes back is this -// call's own bitmap. -func (b *bulkAnd) iter(floor uint32) idIter { - inputs := make([]*roaring.Bitmap, len(b.groups)) - for i := range b.groups { - inputs[i] = orGroup(b.sources, b.groups[i].slots) - } - // FastAnd intersects left to right, so the smallest input first shrinks - // the accumulator fastest. A group's weight only bounds its OR, so the - // inputs are ranked again once they exist. - slices.SortFunc(inputs, func(x, y *roaring.Bitmap) int { - return cmp.Compare(x.GetCardinality(), y.GetCardinality()) - }) - return postings{bm: roaring.FastAnd(inputs...)}.iter( - IDRange{Start: floor, End: b.end}) -} - -// orGroup ORs a group's present terms into one bulkAnd input. A group holding -// one is that term's bitmap, because FastOr has historically Cloned a -// single-element slice. A sparse term is inflated here, the one place the -// ascending path materializes, and only a filter that overran its budget pays. -func orGroup(sources []postings, slots []int) *roaring.Bitmap { - present := make([]*roaring.Bitmap, 0, len(slots)) - for _, slot := range slots { - p := sources[slot] - if bm := p.bitmap(); bm != nil { - present = append(present, bm) - continue - } - if p.ids != nil { - bm := roaring.New() - bm.AddMany(p.ids) - present = append(present, bm) - } - } - if len(present) == 1 { - return present[0] - } - return roaring.FastOr(present...) -} - -// groupIter ORs the postings at slots into one cursor, returning nil when -// every one is absent, which drops the owning filter. -func groupIter(sources []postings, slots []int, window IDRange) idIter { - present := make([]idIter, 0, len(slots)) - for _, slot := range slots { - if p := sources[slot]; p.present() { - present = append(present, p.iter(window)) - } - } - if len(present) == 0 { - return nil - } - return unionOf(present) -} diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go deleted file mode 100644 index cc14538b9..000000000 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_iter_test.go +++ /dev/null @@ -1,1203 +0,0 @@ -package event - -// match_iter_test.go covers the ascending path's un-materialized query plan: -// the cursor sources, the combinators, and the tree candidateIter assembles, -// plus a randomized differential against the materialized bitmap algebra. - -import ( - "context" - "errors" - "iter" - "math/rand" - "slices" - "sync" - "testing" - - "github.com/RoaringBitmap/roaring/v2" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - protocol "github.com/stellar/go-stellar-sdk/protocols/rpc" - "github.com/stellar/go-stellar-sdk/xdr" - - "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/chunk" -) - -// wholeWindow is the no-op window, for every test not about clamping. -var wholeWindow = IDRange{Start: 0, End: ^uint32(0)} - -// drain pulls a cursor dry, never returning nil so an empty result compares -// equal to a materialized bitmap's ToArray(). -func drain(it idIter) []uint32 { - out := []uint32{} - for { - v, ok := it.peek() - if !ok { - return out - } - out = append(out, v) - it.next() - } -} - -// sparseSource and denseSource build the two representations the index holds. -func sparseSource(ids ...uint32) postings { return postings{ids: ids} } - -func denseSource(ids ...uint32) postings { - bm := roaring.New() - bm.AddMany(ids) - return postings{bm: bm} -} - -// sourceKinds runs fn against both representations of the same id set. -func sourceKinds(ids ...uint32) map[string]func() postings { - return map[string]func() postings{ - "sparse": func() postings { return sparseSource(ids...) }, - "dense": func() postings { return denseSource(ids...) }, - } -} - -func TestPostingsPresent(t *testing.T) { - assert.False(t, postings{}.present(), "the zero postings is the absent term") - assert.True(t, sparseSource(1).present()) - assert.True(t, denseSource(1).present()) - assert.True(t, postings{bm: roaring.New()}.present(), - "a present-but-empty bitmap is present; it just yields nothing") - assert.Empty(t, drain(postings{bm: roaring.New()}.iter(wholeWindow))) -} - -// peek does not consume, advance lands on the first id at or above the floor -// and never moves backwards, and both are idempotent at exhaustion. -func TestIDIterSources(t *testing.T) { - for name, mk := range sourceKinds(3, 7, 8, 20, 100) { - t.Run(name, func(t *testing.T) { - it := mk().iter(wholeWindow) - assert.Equal(t, []uint32{3, 7, 8, 20, 100}, drain(mk().iter(wholeWindow))) - - v, ok := it.peek() - require.True(t, ok) - assert.Equal(t, uint32(3), v) - v2, _ := it.peek() - assert.Equal(t, v, v2, "peek must not consume") - - it.advance(7) - v, ok = it.peek() - require.True(t, ok) - assert.Equal(t, uint32(7), v, "advance lands on an id equal to min") - - it.advance(4) - v, _ = it.peek() - assert.Equal(t, uint32(7), v, "advance must never move backwards") - - it.advance(9) - v, ok = it.peek() - require.True(t, ok) - assert.Equal(t, uint32(20), v, "advance skips the gap to the next id above min") - - it.advance(1000) - _, ok = it.peek() - assert.False(t, ok, "advance past the last id exhausts the cursor") - it.next() // must not panic - it.advance(0) - _, ok = it.peek() - assert.False(t, ok, "exhaustion is permanent") - }) - } -} - -// The window applied at the leaf: ids below Start are skipped at construction -// and ids at or above the exclusive End exhaust the cursor. -func TestIDIterSourcesWindowClampsBothEnds(t *testing.T) { - for name, mk := range sourceKinds(1, 4, 5, 6, 9, 10, 11) { - t.Run(name, func(t *testing.T) { - assert.Equal(t, []uint32{4, 5, 6, 9}, - drain(mk().iter(IDRange{Start: 4, End: 10})), - "Start is inclusive, End exclusive") - assert.Equal(t, []uint32{1, 4, 5, 6, 9, 10, 11}, - drain(mk().iter(IDRange{Start: 0, End: 12}))) - assert.Empty(t, drain(mk().iter(IDRange{Start: 12, End: 20})), - "a window entirely above the postings yields nothing") - assert.Empty(t, drain(mk().iter(IDRange{Start: 0, End: 1})), - "a window entirely below the postings yields nothing") - assert.Equal(t, []uint32{11}, - drain(mk().iter(IDRange{Start: 11, End: 12})), - "a one-id window ending just past the last id keeps it") - - // advance must not resurrect ids the End clamp removed. - it := mk().iter(IDRange{Start: 0, End: 7}) - it.advance(9) - _, ok := it.peek() - assert.False(t, ok, "advancing past End must not escape the window") - }) - } -} - -func TestEmptyIter(t *testing.T) { - it := idIter(emptyIter{}) - _, ok := it.peek() - assert.False(t, ok) - it.next() - it.advance(5) - _, ok = it.peek() - assert.False(t, ok) - assert.Empty(t, drain(it)) -} - -// An id several children hold is yielded once; emitting it per child would -// hand FetchEvents a duplicate, which it rejects. -func TestUnionIterDedups(t *testing.T) { - u := &unionIter{children: []idIter{ - sparseSource(1, 3, 5, 7).iter(wholeWindow), - denseSource(3, 4, 5).iter(wholeWindow), - sparseSource(5).iter(wholeWindow), - }} - assert.Equal(t, []uint32{1, 3, 4, 5, 7}, drain(u)) -} - -func TestUnionIterAdvanceAndEdges(t *testing.T) { - mk := func() idIter { - return &unionIter{children: []idIter{ - sparseSource(2, 6, 10).iter(wholeWindow), - denseSource(4, 6, 12).iter(wholeWindow), - emptyIter{}, - }} - } - assert.Equal(t, []uint32{2, 4, 6, 10, 12}, drain(mk()), - "an exhausted child contributes nothing and does not stop the union") - - u := mk() - u.advance(5) - v, ok := u.peek() - require.True(t, ok) - assert.Equal(t, uint32(6), v) - u.advance(3) - v, _ = u.peek() - assert.Equal(t, uint32(6), v, "advance backwards is a no-op") - assert.Equal(t, []uint32{6, 10, 12}, drain(u)) - - u = mk() - u.advance(100) - _, ok = u.peek() - assert.False(t, ok) - - allEmpty := &unionIter{children: []idIter{emptyIter{}, emptyIter{}}} - assert.Empty(t, drain(allEmpty)) -} - -// The AND: a plain overlap, a three-way overlap forcing several alignment -// passes, disjoint children, and an empty child short-circuiting. -func TestIntersectIterGallops(t *testing.T) { - t.Run("overlap", func(t *testing.T) { - n := &intersectIter{children: []idIter{ - sparseSource(1, 2, 3, 4, 5, 6).iter(wholeWindow), - denseSource(2, 4, 6, 8).iter(wholeWindow), - }} - assert.Equal(t, []uint32{2, 4, 6}, drain(n)) - }) - - t.Run("three way with long gallops", func(t *testing.T) { - // Each child holds a long run the others skip, so alignment gallops - // repeatedly and in both orders. - a := make([]uint32, 0, 400) - b := make([]uint32, 0, 400) - c := make([]uint32, 0, 400) - for i := range uint32(400) { - a = append(a, i*2) // even - b = append(b, i*3) // multiples of 3 - c = append(c, i*10) // multiples of 10 - } - n := &intersectIter{children: []idIter{ - denseSource(a...).iter(wholeWindow), - denseSource(b...).iter(wholeWindow), - sparseSource(c...).iter(wholeWindow), - }} - want := []uint32{} - for i := uint32(0); i <= 780; i += 30 { // lcm(2,3,10) = 30, capped by c's max - want = append(want, i) - } - assert.Equal(t, want, drain(n)) - }) - - t.Run("disjoint", func(t *testing.T) { - n := &intersectIter{children: []idIter{ - sparseSource(1, 3, 5).iter(wholeWindow), - sparseSource(2, 4, 6).iter(wholeWindow), - }} - assert.Empty(t, drain(n)) - }) - - t.Run("empty child", func(t *testing.T) { - n := &intersectIter{children: []idIter{ - sparseSource(1, 2, 3).iter(wholeWindow), - emptyIter{}, - sparseSource(2).iter(wholeWindow), - }} - assert.Empty(t, drain(n)) - _, ok := n.peek() - assert.False(t, ok, "an exhausted child ends the intersection permanently") - }) -} - -func TestIntersectIterAdvance(t *testing.T) { - n := &intersectIter{children: []idIter{ - denseSource(1, 2, 3, 4, 5, 6, 7, 8).iter(wholeWindow), - sparseSource(2, 4, 6, 8).iter(wholeWindow), - }} - n.advance(5) - v, ok := n.peek() - require.True(t, ok) - assert.Equal(t, uint32(6), v) - n.advance(1) - v, _ = n.peek() - assert.Equal(t, uint32(6), v, "advance backwards is a no-op") - assert.Equal(t, []uint32{6, 8}, drain(n)) -} - -// A one-input union or intersect is the input itself, not a wrapper, which -// keeps a one-constraint filter from re-scanning a one-element slice per step. -func TestSingleChildCollapse(t *testing.T) { - leaf := sparseSource(1, 2).iter(wholeWindow) - assert.Same(t, leaf, unionOf([]idIter{leaf})) - assert.Same(t, leaf, intersectOf([]idIter{leaf})) - assert.Equal(t, emptyIter{}, unionOf(nil)) - assert.Equal(t, emptyIter{}, intersectOf(nil)) - assert.IsType(t, &unionIter{}, unionOf([]idIter{leaf, leaf})) - assert.IsType(t, &intersectIter{}, intersectOf([]idIter{leaf, leaf})) -} - -// A group whose every term is missing returns nil, dropping the owning filter. -func TestGroupIterAbsentGroup(t *testing.T) { - sources := []postings{ - sparseSource(1, 2), - {}, // absent - {}, // absent - denseSource(2, 3), - } - assert.Nil(t, groupIter(sources, []int{1}, wholeWindow)) - assert.Nil(t, groupIter(sources, []int{1, 2}, wholeWindow), - "a group is absent only when every one of its terms is") - assert.Equal(t, []uint32{1, 2}, drain(groupIter(sources, []int{0, 1}, wholeWindow)), - "a partly-present group ORs only the present terms") - assert.Equal(t, []uint32{1, 2, 3}, drain(groupIter(sources, []int{0, 3}, wholeWindow))) -} - -// The ordering weight is the term's whole-chunk cardinality, window and all. -func TestPostingsEstimate(t *testing.T) { - assert.Equal(t, uint64(0), postings{}.estimate(), "the absent term weighs nothing") - assert.Equal(t, uint64(0), postings{bm: roaring.New()}.estimate()) - assert.Equal(t, uint64(3), sparseSource(1, 2, 3).estimate()) - assert.Equal(t, uint64(3), denseSource(1, 2, 3).estimate()) - assert.Equal(t, uint64(4), denseSource(1, 2, 3, 1<<20).estimate(), - "cardinality spans containers") -} - -// What a group reports about itself: presence, and its summed weight. -func TestResolveGroup(t *testing.T) { - sources := []postings{ - sparseSource(1, 2), - {}, // absent - denseSource(2, 3, 4), - } - - g, ok := resolveGroup(sources, []int{0}) - require.True(t, ok) - assert.Equal(t, uint64(2), g.est) - assert.Equal(t, []int{0}, g.slots) - - g, ok = resolveGroup(sources, []int{0, 2}) - require.True(t, ok) - assert.Equal(t, uint64(5), g.est, "a group's terms sum, overlaps double-counted") - - g, ok = resolveGroup(sources, []int{1, 2}) - require.True(t, ok) - assert.Equal(t, uint64(3), g.est, "an absent term adds nothing") - - g, ok = resolveGroup(sources, []int{1}) - assert.False(t, ok, "a group of absent terms drops its filter") - assert.Equal(t, uint64(0), g.est) -} - -// The rarest group leads the AND however the plan named its groups, which is -// what bounds the walk at one round per id of that group. -func TestFilterIterOrdersRarestFirst(t *testing.T) { - sources := []postings{ - denseSource(1, 2, 3, 4, 5, 6, 7, 8), // 0: the fat group - denseSource(2, 4, 6, 8), // 1 - denseSource(4, 8), // 2: the rare group - } - slotSets := [][]int{{0}, {2}, {1}} - groups := make([]candidateGroup, 0, len(slotSets)) - for _, slots := range slotSets { - g, ok := resolveGroup(sources, slots) - require.True(t, ok) - groups = append(groups, g) - } - it := filterIter(sources, groups, wholeWindow) - n, isIntersect := it.(*intersectIter) - require.True(t, isIntersect) - assert.Equal(t, [][]int{{2}, {1}, {0}}, - [][]int{groups[0].slots, groups[1].slots, groups[2].slots}, - "the groups are reordered rarest first") - assert.Equal(t, alignBudget, n.budget) - assert.Equal(t, []uint32{4, 8}, drain(it)) - - // A one-group filter is the group itself: no AND, so no budget. - one, ok := resolveGroup(sources, []int{2}) - require.True(t, ok) - assert.Equal(t, []uint32{4, 8}, - drain(filterIter(sources, []candidateGroup{one}, wholeWindow))) -} - -// planGroups resolves a whole filter, for tests driving filterIter directly. -func planGroups(t *testing.T, sources []postings, plan termPlan) []candidateGroup { - t.Helper() - groups := make([]candidateGroup, 0, len(plan)) - for _, slots := range plan { - g, ok := resolveGroup(sources, slots) - require.True(t, ok) - groups = append(groups, g) - } - return groups -} - -// An AND that overruns its budget answers the rest of its window from the bulk -// bitmap, yielding exactly what the walk would have at every seam position. -func TestIntersectIterSpills(t *testing.T) { - sources := []postings{ - denseSource(1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12), - denseSource(2, 4, 6, 8, 10, 12), - sparseSource(3, 4, 8, 12, 20), - } - plan := termPlan{{0}, {1}, {2}} - - for _, budget := range []uint64{0, 1, 2, 3, 5, 100} { - it, ok := filterIter(sources, planGroups(t, sources, plan), wholeWindow).(*intersectIter) - require.True(t, ok) - it.budget = budget - assert.Equal(t, []uint32{4, 8, 12}, drain(it), "budget %d", budget) - } - - // A budget of zero spills on the first round, so the whole answer comes - // from the bulk bitmap and the window still lands at the leaf. - it, ok := filterIter(sources, planGroups(t, sources, plan), - IDRange{Start: 0, End: 12}).(*intersectIter) - require.True(t, ok) - it.budget = 0 - assert.Equal(t, []uint32{4, 8}, drain(it)) - assert.NotNil(t, it.spilled) -} - -// A gallop that crosses the spill lands where the walk would have. -func TestIntersectIterSpillAdvance(t *testing.T) { - sources := []postings{ - denseSource(1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12), - denseSource(2, 4, 6, 8, 10, 12), - } - plan := termPlan{{0}, {1}} - for _, budget := range []uint64{0, 1, 2, 100} { - it, ok := filterIter(sources, planGroups(t, sources, plan), wholeWindow).(*intersectIter) - require.True(t, ok) - it.budget = budget - it.advance(7) - v, vok := it.peek() - require.True(t, vok, "budget %d", budget) - assert.Equal(t, uint32(8), v, "budget %d", budget) - it.advance(3) - v, _ = it.peek() - assert.Equal(t, uint32(8), v, "advance backwards is a no-op, budget %d", budget) - assert.Equal(t, []uint32{8, 10, 12}, drain(it), "budget %d", budget) - } -} - -// randomBulkPlan draws overlapping terms in both representations and one -// filter's plan over them. -func randomBulkPlan(rng *rand.Rand, idSpace int) ([]postings, termPlan) { - sources := make([]postings, 1+rng.Intn(5)) - for i := range sources { - n := rng.Intn(idSpace / 2) - seen := make(map[uint32]struct{}, n) - for range n { - seen[uint32(rng.Intn(idSpace))] = struct{}{} - } - ids := make([]uint32, 0, len(seen)) - for id := range seen { - ids = append(ids, id) - } - slices.Sort(ids) - if rng.Intn(4) == 0 { - sources[i] = postings{ids: ids} - } else { - sources[i] = denseSource(ids...) - } - } - plan := make(termPlan, 1+rng.Intn(4)) - for g := range plan { - slots := make([]int, 1+rng.Intn(2)) - for s := range slots { - slots[s] = rng.Intn(len(sources)) - } - plan[g] = slots - } - return sources, plan -} - -// Over randomized plans, shapes and windows, an AND that spills must yield -// what the unbounded walk yields, and both must equal the materialized -// algebra. The budget moves rather than the corpus, so one filter is answered -// every way. -func TestFilterIterBulkMatchesWalk(t *testing.T) { - rng := rand.New(rand.NewSource(20260901)) - const idSpace = 4000 - spills := 0 - for trial := range 300 { - sources, plan := randomBulkPlan(rng, idSpace) - start := uint32(rng.Intn(idSpace)) - end := start + uint32(rng.Intn(idSpace)) - window := IDRange{Start: start, End: end} - - build := func(budget uint64) (idIter, *intersectIter) { - groups := make([]candidateGroup, 0, len(plan)) - for _, slots := range plan { - g, ok := resolveGroup(sources, slots) - if !ok { - return nil, nil - } - groups = append(groups, g) - } - it := filterIter(sources, groups, window) - n, _ := it.(*intersectIter) - if n != nil { - n.budget = budget - } - return it, n - } - walk, _ := build(^uint64(0)) - if walk == nil { - continue - } - want := referenceCandidates([]termPlan{plan}, sources, window) - require.Equal(t, want, drain(walk), - "trial %d: window %v plan %v", trial, window, plan) - // Low budgets put the seam at the start of a plan's answer and - // partway into it, so the join is under test and not just its ends. - for _, budget := range []uint64{0, 1, 3} { - it, n := build(budget) - require.Equal(t, want, drain(it), - "trial %d budget %d: window %v plan %v", trial, budget, window, plan) - if n != nil && n.spilled != nil { - spills++ - } - } - } - require.Greater(t, spills, 100, "fixture sanity: the spill must actually fire") -} - -func TestCandidateIterDropsFilterWithAbsentGroup(t *testing.T) { - sources := []postings{ - sparseSource(1, 2, 3), - {}, // absent - sparseSource(9), - } - // Filter 0 needs slot 1, which is absent → contributes nothing. - // Filter 1 is slot 2 alone → survives. - plans := []termPlan{{{0}, {1}}, {{2}}} - assert.Equal(t, []uint32{9}, drain(candidateIter(plans, sources, wholeWindow))) - - // Every filter dropped, so the cursor is exhausted. - allMissed := []termPlan{{{0}, {1}}} - it := candidateIter(allMissed, sources, wholeWindow) - _, ok := it.peek() - assert.False(t, ok) -} - -// referenceCandidates is an independent, naive materialized implementation of -// the algebra candidateIter answers. -func referenceCandidates(plans []termPlan, sources []postings, window IDRange) []uint32 { - materialize := func(p postings) *roaring.Bitmap { - if bm := p.bitmap(); bm != nil { - return bm - } - bm := roaring.New() - bm.AddMany(p.ids) - return bm - } - union := roaring.New() - for _, plan := range plans { - var acc *roaring.Bitmap - missed := false - for _, slots := range plan { - group := roaring.New() - present := false - for _, s := range slots { - if sources[s].present() { - present = true - group.Or(materialize(sources[s])) - } - } - if !present { - missed = true - break - } - if acc == nil { - acc = group - } else { - acc.And(group) - } - } - if missed { - continue - } - union.Or(acc) - } - windowBM := roaring.New() - windowBM.AddRange(uint64(window.Start), uint64(window.End)) - union.And(windowBM) - return union.ToArray() -} - -// Drives whole Matches calls with the budget shrunk so every AND spills, and -// requires the stream the unbounded walk yields — with the seam under term -// planning, the window cap, the batch loop and the post-filter. -func TestMatchesSpillYieldsSameStream(t *testing.T) { - rng := rand.New(rand.NewSource(20260902)) - v := newDiffVocab(t) - const corpusSize = 300 - corpus := newDiffCorpus(t, rng, v, corpusSize) - - // Shrink the batch too, so a spill can land mid-page. - defer func(n int) { matchBatchSize = n }(matchBatchSize) - matchBatchSize = 7 - defer func(n uint64) { alignBudget = n }(alignBudget) - - r := diffPostingsReader{diffReader{corpus}} - matched := 0 - for trial := range 200 { - filters := randomFilters(rng, v) - start := uint32(rng.Intn(corpusSize + 1)) - end := start + uint32(rng.Intn(corpusSize+1-int(start))) - w := IDRange{Start: start, End: end} - - alignBudget = ^uint64(0) - want := collectOrdinals(t, r, filters, w, false) - matched += len(want) - for _, budget := range []uint64{0, 1, 4} { - alignBudget = budget - require.Equal(t, want, collectOrdinals(t, r, filters, w, false), - "trial %d budget %d: window %v filters %+v", trial, budget, w, filters) - } - } - require.Greater(t, matched, 2000, - "fixture sanity: randomized queries selected too little") -} - -// Over randomized plans, source shapes and windows, the un-materialized tree -// must yield exactly what the bitmap algebra does. -func TestCandidateIterMatchesMaterializedAlgebra(t *testing.T) { - rng := rand.New(rand.NewSource(20260829)) - const idSpace = 400 - for trial := range 500 { - nSources := 1 + rng.Intn(6) - sources := make([]postings, nSources) - for i := range sources { - switch rng.Intn(5) { - case 0: - // absent - case 1: - sources[i] = postings{bm: roaring.New()} // present, empty - default: - n := rng.Intn(40) - seen := make(map[uint32]struct{}, n) - for range n { - seen[uint32(rng.Intn(idSpace))] = struct{}{} - } - ids := make([]uint32, 0, len(seen)) - for id := range seen { - ids = append(ids, id) - } - slices.Sort(ids) - if rng.Intn(2) == 0 { - sources[i] = postings{ids: ids} - } else { - sources[i] = denseSource(ids...) - } - } - } - plans := make([]termPlan, 1+rng.Intn(3)) - for f := range plans { - plan := make(termPlan, 1+rng.Intn(3)) - for g := range plan { - slots := make([]int, 1+rng.Intn(3)) - for s := range slots { - slots[s] = rng.Intn(nSources) - } - plan[g] = slots - } - plans[f] = plan - } - start := uint32(rng.Intn(idSpace)) - end := start + uint32(rng.Intn(idSpace)) - window := IDRange{Start: start, End: end} - - want := referenceCandidates(plans, sources, window) - got := drain(candidateIter(plans, sources, window)) - require.Equal(t, want, got, - "trial %d: window %v plans %v", trial, window, plans) - } -} - -// ─── the A/B benchmark ────────────────────────────────────────────── - -// stubIndex is a Reader over an in-memory mirror and one shared payload, so a -// benchmark measures the match layer rather than the storage tier. FetchEvents -// reuses its buffer, keeping the per-batch fetch cost equal in both directions. -type stubIndex struct { - mirror *ConcurrentBitmaps - count uint32 - raw []byte - buf []Payload -} - -func (s *stubIndex) ChunkID() chunk.ID { return chunk.ID(0) } -func (s *stubIndex) EventCount() (uint32, error) { return s.count, nil } - -func (s *stubIndex) Offsets() (*LedgerOffsets, error) { - return nil, errors.New("stubIndex: Offsets is not part of the match path") -} - -func (s *stubIndex) LookupKeys(_ context.Context, keys []TermKey) ([]*roaring.Bitmap, error) { - out := make([]*roaring.Bitmap, len(keys)) - for i, k := range keys { - bm, err := s.mirror.Get(k) - if err != nil { - return nil, err - } - out[i] = bm - } - return out, nil -} - -func (s *stubIndex) FetchEvents(_ context.Context, ids []uint32) ([]Payload, error) { - if err := validateSortedEventIDs(ids); err != nil { - return nil, err - } - s.buf = s.buf[:0] - for range ids { - s.buf = append(s.buf, Payload{ContractEventBytes: s.raw}) - } - return s.buf, nil -} - -func (s *stubIndex) FetchRange(_ context.Context, start, count uint32) iter.Seq2[Payload, error] { - return func(yield func(Payload, error) bool) { - if err := validateFetchRange(start, count, s.count, s.ChunkID()); err != nil { - yield(Payload{}, err) - return - } - for range count { - if !yield(Payload{ContractEventBytes: s.raw}, nil) { - return - } - } - } -} - -func (s *stubIndex) All(ctx context.Context) iter.Seq2[Payload, error] { - return s.FetchRange(ctx, 0, s.count) -} - -// hotLikeIndex carries the same optional no-materialize seam HotStore -// does, so the ascending benchmark exercises the production fast path -// (sparse terms read in place) rather than the bitmap fallback. -type hotLikeIndex struct{ *stubIndex } - -func (h *hotLikeIndex) lookupPostings(_ context.Context, keys []TermKey) ([]postings, error) { - out := make([]postings, len(keys)) - for i, k := range keys { - out[i] = h.mirror.lookupPostings(k) - } - return out, nil -} - -var ( - _ Reader = (*stubIndex)(nil) - _ Reader = (*hotLikeIndex)(nil) - _ postingReader = (*hotLikeIndex)(nil) -) - -const ( - // ~4M events: half a production chunk (~9M), enough that the - // materialized path's intermediates are the multi-container - // bitmaps the real one builds. - benchEvents = 1 << 22 - benchPage = 1000 // getEvents' max page size -) - -type benchIndex struct { - reader *hotLikeIndex - filters []Filter - window IDRange -} - -// newBenchIndex builds the synthetic chunk once for both directions: -// three dense terms (one near-total, like the event type; two -// selective) plus a long-tail sparse term below the mirror's -// promotion threshold, so the sparse read path is on the plan. -var newBenchIndex = sync.OnceValue(func() *benchIndex { - var contractA xdr.ContractId - contractA[0] = 0xA1 - topic := xdr.ScSymbol("bench-topic") - topicVal := xdr.ScVal{Type: xdr.ScValTypeScvSymbol, Sym: &topic} - topicRaw, err := topicVal.MarshalBinary() - if err != nil { - panic(err) - } - ev := xdr.ContractEvent{ - ContractId: &contractA, - Type: xdr.ContractEventTypeContract, - Body: xdr.ContractEventBody{ - V: 0, - V0: &xdr.ContractEventV0{Topics: []xdr.ScVal{topicVal}, Data: topicVal}, - }, - } - raw, err := ev.MarshalBinary() - if err != nil { - panic(err) - } - - // Dense terms go in through the frozen-Bitmaps constructor (roaring - // mode); the sparse one goes in through AddTo so it stays under the - // promotion threshold and is stored as a plain id list. - bms := NewBitmaps() - typeKey := EventTypeTermKey(xdr.ContractEventTypeContract) - contractKey := ComputeTermKey(contractA[:], FieldContractID) - topic1Key := ComputeTermKey(topicRaw, FieldTopic1) - everything := make([]uint32, 0, benchEvents) - contractIDs := make([]uint32, 0, benchEvents/3+1) - topic1IDs := make([]uint32, 0, benchEvents/7+1) - for id := range uint32(benchEvents) { - everything = append(everything, id) - if id%3 == 0 { - contractIDs = append(contractIDs, id) - } - if id%7 == 0 { - topic1IDs = append(topic1IDs, id) - } - } - bms.AddTo(typeKey, everything...) - bms.AddTo(contractKey, contractIDs...) - bms.AddTo(topic1Key, topic1IDs...) - mirror := NewConcurrentBitmapsFromBitmaps(bms) - - topic0Key := ComputeTermKey(topicRaw, FieldTopic0) - sparse := make([]uint32, 0, promotionThreshold-1) - for i := range uint32(promotionThreshold - 1) { - sparse = append(sparse, i*(benchEvents/promotionThreshold)) - } - mirror.AddTo(topic0Key, sparse...) - - eventType := xdr.ContractEventTypeContract - var topics [protocol.MaxTopicCount][]byte - topics[0] = topicRaw - return &benchIndex{ - reader: &hotLikeIndex{&stubIndex{ - mirror: mirror, count: benchEvents, raw: raw, - }}, - filters: []Filter{ - // Two dense groups AND-ed: the intersect arm. - {ContractID: contractA[:], EventType: &eventType}, - // One long-tail sparse group: the arm Get used to - // materialize a bitmap for on every request. - {Topics: topics}, - }, - // A sub-window, so both window edges are live. - window: IDRange{Start: benchEvents / 4, End: benchEvents * 3 / 4}, - } -}) - -// benchMatches drives one page-sized request and stops, the shape a -// getEvents page actually has. -func benchMatches(b *testing.B, descending bool) { - b.Helper() - fx := newBenchIndex() - ctx := context.Background() - b.ReportAllocs() - b.ResetTimer() - for b.Loop() { - n := 0 - for _, err := range Matches(ctx, fx.reader, fx.filters, fx.window, descending, benchPage) { - if err != nil { - b.Fatal(err) - } - n++ - if n == benchPage { - break - } - } - if n != benchPage { - b.Fatalf("fixture sanity: want %d matches, got %d", benchPage, n) - } - } -} - -// The in-tree A/B for the un-materialized path: the same query, page size and -// fetch work, differing only in which candidate path Matches takes. -func BenchmarkMatchesAscending(b *testing.B) { benchMatches(b, false) } -func BenchmarkMatchesDescending(b *testing.B) { benchMatches(b, true) } - -// The candidate-shape microbench isolates the candidate set itself: both paths -// answer the same synthetic plan over the same mirror, with fetch and -// post-filter out of frame. The shapes are the term geometries the two are -// expected to disagree on. - -// benchFat is one fat term's cardinality against benchEvents: ~7% of -// the domain, the density at which roaring holds a term as bitmap -// containers — the representation FastAnd intersects a word at a -// time and the cursor tree walks a bit at a time. -const benchFat = 300_000 - -// benchRand is a deterministic xorshift. The shapes must be identical -// from run to run, and a fixed stride would hand the gallop a -// regularity real postings do not have. -type benchRand uint64 - -func (r *benchRand) next() uint64 { - x := uint64(*r) - x ^= x << 13 - x ^= x >> 7 - x ^= x << 17 - *r = benchRand(x) - return x -} - -// scatter draws k ascending ids from one residue class, one per stride at a -// jittered offset. Terms on disjoint classes interleave at single-id -// granularity while sharing nothing, so a shape's overlap is exactly the class -// its terms share. -func scatter(rng *benchRand, domain, m, res uint32, k int) []uint32 { - if k == 0 { - return nil - } - class := (domain - res + m - 1) / m - stride := class / uint32(k) - if stride == 0 { - panic("scatter: residue class too small for k") - } - ids := make([]uint32, k) - for t := range k { - ids[t] = (uint32(t)*stride+uint32(rng.next()%uint64(stride)))*m + res - } - return ids -} - -// fatGroup builds n terms of card ids each, drawing private ids from one -// residue class per term plus one class every term holds, so the joint -// intersection is exactly that shared class. -func fatGroup(rng *benchRand, domain, mod, base uint32, n, card, shared int) [][]uint32 { - common := scatter(rng, domain, mod, base+uint32(n), shared) - out := make([][]uint32, n) - for i := range out { - ids := scatter(rng, domain, mod, base+uint32(i), card-shared) - ids = append(ids, common...) - slices.Sort(ids) - out[i] = ids - } - return out -} - -// benchShape is one synthetic candidate-set problem: a term corpus in -// the mirror, the plan resolved over it, and the page both paths must -// produce from it. -type benchShape struct { - name string - reader *hotLikeIndex - plans []termPlan - keys []TermKey - window IDRange - // wantCount and wantSum fingerprint the first page. Both paths - // check them every iteration, so a harness that stopped answering - // the query cannot post a fast number. - wantCount int - wantSum uint64 -} - -// newBenchShape indexes terms as one term each, resolves the window to -// most of the domain with both edges live, and fingerprints the first -// page off the materialized algebra — the reference the cursor tree -// must reproduce id for id. -func newBenchShape(name string, domain uint32, terms [][]uint32, plans []termPlan) *benchShape { - bms := NewBitmaps() - keys := make([]TermKey, len(terms)) - for i, ids := range terms { - keys[i] = TermKey{0: byte(i + 1)} - bms.AddTo(keys[i], ids...) - } - s := &benchShape{ - name: name, - reader: &hotLikeIndex{&stubIndex{ - mirror: NewConcurrentBitmapsFromBitmaps(bms), count: domain, - }}, - plans: plans, - keys: keys, - window: IDRange{Start: domain / 32, End: domain - domain/32}, - } - union, err := unionForFilters( - context.Background(), s.reader, s.plans, s.keys, s.window) - if err != nil { - panic(err) - } - it := union.Iterator() - for s.wantCount < benchPage && it.HasNext() { - s.wantCount++ - s.wantSum += uint64(it.Next()) - } - return s -} - -// singleFilterPlan is one filter AND-ing n one-term groups: the -// intersect shapes' plan. -func singleFilterPlan(n int) []termPlan { - plan := make(termPlan, n) - for i := range plan { - plan[i] = []int{i} - } - return []termPlan{plan} -} - -// benchShapes is the shape matrix, each entry built on first use so a -// -bench selecting one shape pays for one shape. domain is a -// parameter so the correctness twin of the matrix can run the same -// geometries small. -func benchShapes(domain uint32) []struct { - name string - build func() *benchShape -} { - scale := func(n int) int { return max(1, n*int(domain)/benchEvents) } - fat := scale(benchFat) - // ~3% of a fat term: the partial overlap that makes an aligning - // AND converge slowly without making it empty. - partial := fat * 3 / 100 - // Just over one page once the window clips it: the intersection - // too small to fill a page early, so the walk spans the window. - tiny := scale(1200) - - shapes := []struct { - name string - build func() *benchShape - }{ - {"a_and2_fat_3pct", func() *benchShape { - rng := benchRand(1) - return newBenchShape("a", domain, - fatGroup(&rng, domain, 3, 0, 2, fat, partial), singleFilterPlan(2)) - }}, - {"b_and3_fat_3pct", func() *benchShape { - rng := benchRand(2) - return newBenchShape("b", domain, - fatGroup(&rng, domain, 4, 0, 3, fat, partial), singleFilterPlan(3)) - }}, - {"c_and2_skew", func() *benchShape { - rng := benchRand(3) - // The small term is a subset of the fat one, spread over - // it, so the AND is entirely decided by the rare side — - // the gallop-friendly control. - big := scatter(&rng, domain, 1, 0, fat) - small := make([]uint32, 0, scale(2000)) - step := len(big) / cap(small) - for i := range cap(small) { - small = append(small, big[i*step]) - } - return newBenchShape("c", domain, - [][]uint32{big, small}, singleFilterPlan(2)) - }}, - {"d_and6_fat_tiny", func() *benchShape { - rng := benchRand(4) - return newBenchShape("d", domain, - fatGroup(&rng, domain, 7, 0, 6, fat, tiny), singleFilterPlan(6)) - }}, - {"e_or10_single_term", func() *benchShape { - rng := benchRand(5) - terms := fatGroup(&rng, domain, 11, 0, 10, scale(30_000), 0) - plans := make([]termPlan, len(terms)) - for i := range plans { - plans[i] = termPlan{{i}} - } - return newBenchShape("e", domain, terms, plans) - }}, - {"f_and2_fat_tiny", func() *benchShape { - rng := benchRand(6) - return newBenchShape("f", domain, - fatGroup(&rng, domain, 3, 0, 2, fat, tiny), singleFilterPlan(2)) - }}, - {"h_and2_fat_overlapping", func() *benchShape { - rng := benchRand(8) - // The serving default: one selective term AND-ed with a near-total - // one, so a page comes out of the window's first fraction. This - // is the shape any eager rule must leave alone. - selective := scatter(&rng, domain, 3, 0, fat) - nearAll := make([]uint32, 0, domain) - for id := range domain { - if id%50 != 7 { - nearAll = append(nearAll, id) - } - } - return newBenchShape("h", domain, - [][]uint32{selective, nearAll}, singleFilterPlan(2)) - }}, - {"g_and3_x4_filters", func() *benchShape { - rng := benchRand(7) - // The serving shape the tail regression was measured on: - // several filters, each AND-ing a few fat terms. Each - // filter owns four residue classes, so the filters overlap - // only where the union has to dedup them. - terms := make([][]uint32, 0, 12) - plans := make([]termPlan, 0, 4) - for f := range uint32(4) { - group := fatGroup(&rng, domain, 16, f*4, 3, scale(75_000), scale(2250)) - plan := make(termPlan, len(group)) - for i := range group { - plan[i] = []int{len(terms) + i} - } - terms = append(terms, group...) - plans = append(plans, plan) - } - return newBenchShape("g", domain, terms, plans) - }}, - } - return shapes -} - -// benchShapeCache keeps one built corpus per shape name, so the tree -// and materialized runs of a shape share it. Benchmarks run one at a -// time, so a plain map suffices. -var benchShapeCache = map[string]*benchShape{} - -func shapeFor(name string, build func() *benchShape) *benchShape { - s, ok := benchShapeCache[name] - if !ok { - s = build() - benchShapeCache[name] = s - } - return s -} - -// benchCandidatePage pulls one page of candidates through the cursor -// tree — the ascending path's candidate work, with nothing else in -// frame. -func benchCandidatePage(b *testing.B, s *benchShape) { - b.Helper() - ctx := context.Background() - b.ReportAllocs() - for b.Loop() { - sources, err := lookupPostings(ctx, s.reader, s.keys) - if err != nil { - b.Fatal(err) - } - it := candidateIter(s.plans, sources, s.window) - n, sum := 0, uint64(0) - for n < benchPage { - v, ok := it.peek() - if !ok { - break - } - n, sum = n+1, sum+uint64(v) - it.next() - } - if n != s.wantCount || sum != s.wantSum { - b.Fatalf("page mismatch: got (%d, %d), want (%d, %d)", - n, sum, s.wantCount, s.wantSum) - } - } -} - -// benchMaterializedPage is benchCandidatePage's twin over the bitmap algebra: -// build the whole candidate set, then read one page off it. It reads ascending -// so the two harnesses answer bit for bit. -func benchMaterializedPage(b *testing.B, s *benchShape) { - b.Helper() - ctx := context.Background() - b.ReportAllocs() - for b.Loop() { - union, err := unionForFilters(ctx, s.reader, s.plans, s.keys, s.window) - if err != nil { - b.Fatal(err) - } - it := union.Iterator() - n, sum := 0, uint64(0) - for n < benchPage && it.HasNext() { - n, sum = n+1, sum+uint64(it.Next()) - } - if n != s.wantCount || sum != s.wantSum { - b.Fatalf("page mismatch: got (%d, %d), want (%d, %d)", - n, sum, s.wantCount, s.wantSum) - } - } -} - -// BenchmarkCandidateTree and BenchmarkCandidateMaterialized are the -// per-shape A/B for the candidate set: the same plan, the same -// postings, the same page, differing only in whether the ids are -// pulled through the cursor tree or read off a materialized bitmap. -func BenchmarkCandidateTree(b *testing.B) { - for _, sh := range benchShapes(benchEvents) { - b.Run(sh.name, func(b *testing.B) { - benchCandidatePage(b, shapeFor(sh.name, sh.build)) - }) - } -} - -func BenchmarkCandidateMaterialized(b *testing.B) { - for _, sh := range benchShapes(benchEvents) { - b.Run(sh.name, func(b *testing.B) { - benchMaterializedPage(b, shapeFor(sh.name, sh.build)) - }) - } -} - -// TestBenchShapesAgree runs the whole shape matrix small: every -// geometry the microbench measures must be one both candidate paths -// answer identically, so a shape can never post a number for a query -// the tree gets wrong. -func TestBenchShapesAgree(t *testing.T) { - const domain = 1 << 16 - for _, sh := range benchShapes(domain) { - t.Run(sh.name, func(t *testing.T) { - s := sh.build() - sources, err := lookupPostings(context.Background(), s.reader, s.keys) - require.NoError(t, err) - got := drain(candidateIter(s.plans, sources, s.window)) - require.Equal(t, referenceCandidates(s.plans, sources, s.window), got) - require.NotEmpty(t, got, "shape sanity: the plan must select something") - }) - } -} - -// The first-batch hint contract: a positive hint sizes the first fetch, a wild -// one is capped at eight default batches, and later batches use the default. -// The cap scales with matchBatchSize, so a test-shrunk batch cannot be blown -// past by a hint. -func TestBatchSizes(t *testing.T) { - first, rest := batchSizes(0) - require.Equal(t, matchBatchSize, first) - require.Equal(t, matchBatchSize, rest) - - first, rest = batchSizes(-3) - require.Equal(t, matchBatchSize, first) - require.Equal(t, matchBatchSize, rest) - - first, rest = batchSizes(7) - require.Equal(t, 7, first) - require.Equal(t, matchBatchSize, rest) - - first, rest = batchSizes(1000) - require.Equal(t, 1000, first, "a page-sized hint is the first fetch size") - require.Equal(t, matchBatchSize, rest) - - first, rest = batchSizes(1 << 20) - require.Equal(t, 8*matchBatchSize, first, "oversized hints are capped") - require.Equal(t, matchBatchSize, rest) - - defer func(n int) { matchBatchSize = n }(matchBatchSize) - matchBatchSize = 7 - first, rest = batchSizes(1000) - require.Equal(t, 56, first, "the cap follows the seam") - require.Equal(t, 7, rest) -} diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go index 671bc2718..fa14a0493 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go @@ -620,9 +620,9 @@ func TestQuery_ChunkWithLedgersButZeroEvents(t *testing.T) { // TestQuery_DescendingWithRangeAndMaxEvents covers the // three-way combination — order × range × cap — that no other test -// hits together. Forces the descending branch of streamUnion -// (ReverseIterator with a per-batch flip) over a range-narrowed -// union, then the shim's MaxEvents truncation. +// hits together. Forces the descending slab walk (each slab's result +// read backwards, with a per-batch flip for the fetch) over a +// range-narrowed window, then the shim's MaxEvents truncation. func TestQuery_DescendingWithRangeAndMaxEvents(t *testing.T) { fx := newMultiLedgerQueryFixture(t) first := chunk.ID(0).FirstLedger() @@ -1126,26 +1126,6 @@ func TestQuery_InvalidFilterRejected(t *testing.T) { } } -// TestUnionSlots covers the OR-within-a-group step directly, including the -// all-absent case a fixture cannot reach: the topic-count buckets are the only -// multi-term group, and the overflow bucket is populated in any chunk holding -// an event with topics. -func TestUnionSlots(t *testing.T) { - first := roaring.BitmapOf(1, 2) - second := roaring.BitmapOf(3) - bitmaps := []*roaring.Bitmap{first, nil, second, nil} - - assert.Same(t, first, unionSlots(bitmaps, []int{0}), - "a lone bitmap is borrowed, not cloned") - assert.Nil(t, unionSlots(bitmaps, []int{1})) - assert.Nil(t, unionSlots(bitmaps, []int{1, 3}), - "a group absent from the index empties the filter") - assert.Same(t, second, unionSlots(bitmaps, []int{1, 2}), - "the one present bitmap in a group is borrowed too") - assert.Equal(t, []uint32{1, 2, 3}, unionSlots(bitmaps, []int{0, 2}).ToArray()) - assert.Equal(t, []uint32{1, 2}, first.ToArray(), "inputs must not be mutated") -} - // ─── Cold-reader parity coverage ──────────────────────────────────────── // // The hot tests above prove Query works against *HotStore. The whole @@ -1162,12 +1142,14 @@ func TestUnionSlots(t *testing.T) { // // - match-all asc → streamRange + cold FetchRange // - match-all desc + cap → streamRange top-down, slices.Backward -// - single-filter (contractID) → LookupKeys + streamUnion asc -// - multi-term filter (AND) → FastAnd over multiple cold bitmaps +// - single-filter (contractID) → lookupPostings, which a ColdReader +// serves off LookupKeys, then the +// ascending slab walk +// - multi-term filter (AND) → AndAny per group over cold bitmaps // - cross-filter (OR) → FastOr across filters -// - ledger range + filter → roaring.And with the range bitmap -// - descending + range + cap → ReverseIterator on cold-derived -// union, single-filter And path +// - ledger range + filter → the window clipped into each slab's +// accumulator +// - descending + range + cap → the slab walk run high to low // // What we don't replay against cold: // - The mirror-poisoning collision test (mutating an mmap'd cold @@ -1779,3 +1761,35 @@ func TestCountDistinctTerms(t *testing.T) { {ContractID: cid, TopicCount: TopicCountFilter{Count: 1}}, }), "topic-count buckets are not value terms and are not counted") } + +// The first-batch hint contract: a positive hint sizes the first fetch, a wild +// one is capped at eight default batches, and later batches use the default. +// The cap scales with matchBatchSize, so a test-shrunk batch cannot be blown +// past by a hint. +func TestBatchSizes(t *testing.T) { + first, rest := batchSizes(0) + require.Equal(t, matchBatchSize, first) + require.Equal(t, matchBatchSize, rest) + + first, rest = batchSizes(-3) + require.Equal(t, matchBatchSize, first) + require.Equal(t, matchBatchSize, rest) + + first, rest = batchSizes(7) + require.Equal(t, 7, first) + require.Equal(t, matchBatchSize, rest) + + first, rest = batchSizes(1000) + require.Equal(t, 1000, first, "a page-sized hint is the first fetch size") + require.Equal(t, matchBatchSize, rest) + + first, rest = batchSizes(1 << 20) + require.Equal(t, 8*matchBatchSize, first, "oversized hints are capped") + require.Equal(t, matchBatchSize, rest) + + defer func(n int) { matchBatchSize = n }(matchBatchSize) + matchBatchSize = 7 + first, rest = batchSizes(1000) + require.Equal(t, 56, first, "the cap follows the seam") + require.Equal(t, 7, rest) +} diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go index 488b377c2..67d02e6c5 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go @@ -1,8 +1,10 @@ package event -// Full-Matches differential: the ascending iterator tree and the descending -// materialized union must select the same events, in mirrored order, over -// randomized corpora, filters and windows. +// Full-Matches differential across the index's two read seams: LookupKeys, +// which materializes every term as a bitmap, and lookupPostings, which hands +// sparse terms back as borrowed id lists. Both must select the same events +// over randomized corpora, filters and windows, and the ascending stream +// reversed must equal the descending one. import ( "context" @@ -209,8 +211,8 @@ func collectOrdinals(t *testing.T, r Reader, filters []Filter, w IDRange, desc b return out } -// Drives randomized queries through both candidate paths and both index seams; -// the ascending stream reversed must equal the descending stream. +// Drives randomized queries through both index seams in both directions; the +// ascending stream reversed must equal the descending stream. func TestMatches_AscendingDescendingDifferential(t *testing.T) { rng := rand.New(rand.NewSource(20260829)) v := newDiffVocab(t) @@ -262,9 +264,9 @@ func TestMatches_AscendingDescendingDifferential(t *testing.T) { } } -// Turns the borrow contract into a race-detector gate: the ascending cursors -// read mirror snapshots in place while AddTo publishes new termStates on the -// same keys, including the sparse-to-dense promotion. Under -race any write +// Turns the borrow contract into a race-detector gate: the match path reads +// mirror snapshots in place while AddTo publishes new termStates on the same +// keys, including the sparse-to-dense promotion. Under -race any write // reaching a borrowed snapshot fails the run; without it, the identity check // still pins that a pinned window is immune to ingest past its End. func TestMatches_ConcurrentIngestBorrowSafety(t *testing.T) { diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/postings_freshness_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/postings_freshness_test.go index ae29147d7..6f78b5c0b 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/postings_freshness_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/postings_freshness_test.go @@ -16,13 +16,12 @@ import ( // (lookupPostings, postings.bitmap, postings.estimate): a read that // starts after an AddTo returns observes that AddTo's ids. // -// It needs its own tests because the differentials cannot see the bug -// it guards. matches_differential_test.go and match_iter_test.go build -// their sources up front and then only read, so a postings accessor -// that returned a stale dense snapshot — the raw pub pointer, nil-or- -// behind after every write — would agree with the materialized twin on -// every one of them. Only write-then-read-through-the-accessor -// separates the two. +// It needs its own tests because the match-path tests cannot see the +// bug it guards. They build their sources up front and then only read, +// so a postings accessor that returned a stale dense snapshot — the raw +// pub pointer, nil-or-behind after every write — would agree with them +// on every case. Only write-then-read-through-the-accessor separates +// the two. // postingsIDs drains a term's postings through the accessor the query // engine uses, so a test asserting on it exercises the same path a @@ -31,7 +30,7 @@ func postingsIDs(t *testing.T, s *ConcurrentBitmaps, key TermKey) []uint32 { t.Helper() p := s.lookupPostings(key) require.True(t, p.present(), "the term must be present in the index") - return drain(p.iter(wholeWindow)) + return postingIDs(p) } // denseOf returns the term's denseState, failing the test when the @@ -70,7 +69,7 @@ func TestPostings_SparseLookupSeesTheWriteJustMade(t *testing.T) { p := s.lookupPostings(key) require.True(t, p.present(), "a term written to is present") require.Nil(t, p.bitmap(), "a sub-threshold term stays sparse") - assert.Equal(t, want, drain(p.iter(wholeWindow)), + assert.Equal(t, want, postingIDs(p), "the cursor must yield every id added so far, the last one included") assert.Equal(t, uint64(len(want)), p.estimate(), "estimate must count the write that just returned") @@ -105,7 +104,7 @@ func TestPostings_PromotionIsVisibleThroughLookupPostings(t *testing.T) { require.NotNil(t, p.bitmap(), "crossing the threshold promotes to a bitmap") assert.True(t, p.bitmap().Contains(promoting), "the promoting id must be in the bitmap the promotion built") - assert.Equal(t, want, drain(p.iter(wholeWindow)), + assert.Equal(t, want, postingIDs(p), "a lookup straight after promotion yields every id the term holds") assert.Equal(t, uint64(len(want)), p.estimate(), "estimate must count the promoting write") @@ -157,7 +156,7 @@ func TestPostings_DenseLookupSeesWritesSinceLastSnapshot(t *testing.T) { assert.Equal(t, heldCard+uint64(len(fresh)), bm.GetCardinality()) assert.Equal(t, heldCard+uint64(len(fresh)), p.estimate(), "estimate must count the writes made since the last snapshot") - assert.Subset(t, drain(p.iter(wholeWindow)), fresh, + assert.Subset(t, postingIDs(p), fresh, "the cursor the query engine walks must yield the fresh ids too") } @@ -320,7 +319,7 @@ func TestHotStore_LookupPostingsSeesTheWriteJustMade(t *testing.T) { got, err := h.lookupPostings(t.Context(), []TermKey{sparseKey, denseKey}) require.NoError(t, err) require.Len(t, got, 2) - assert.Equal(t, []uint32{1, 2, 3, 4}, drain(got[0].iter(wholeWindow)), + assert.Equal(t, []uint32{1, 2, 3, 4}, postingIDs(got[0]), "the sparse term must carry the id added since the last lookup") assert.Equal(t, uint64(4), got[0].estimate()) assert.True(t, got[1].bitmap().Contains(uint32(3_000_000)), diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go index 4c8c6e349..df2b7c802 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go @@ -1,10 +1,8 @@ package event -// slab_match.go is a second evaluation engine for the query Matches serves: -// slabMatches. It steps the window one roaring slab at a time — 65536 ids, the -// span of exactly one container — and answers the whole filter algebra inside -// that slab, where match_iter.go pulls candidates through a lazy cursor tree -// (ascending) and match.go materializes a whole-window union (descending). +// slab_match.go is the candidate machinery behind Matches. It steps the window +// one roaring slab at a time — 65536 ids, the span of exactly one container — +// and answers the whole filter algebra inside that slab. // // Clip-early slab evaluation, per slab, per filter: // @@ -24,24 +22,23 @@ package event // reads each result backward. One code path serves both, with no gallop, no // alignment budget, no spill and no separate descending machinery. // -// Laziness is per slab rather than per id. A consumer that stops after one -// page has evaluated only the slabs that page spans, and the cost inside a -// slab is bounded by the containers its inputs hold there rather than by the -// width of the window. The granularity is coarser than the cursor tree's — a -// page ending mid-slab has paid for the whole slab — and bounded above by one -// container's worth of work per input per filter. +// Laziness is per slab, not per id. A consumer that stops after one page has +// evaluated only the slabs that page spans, and the cost inside a slab is +// bounded by the containers its inputs hold there rather than by the width of +// the window. The unit is the slab, so a page ending mid-slab has paid for +// that whole slab: one container's worth of work per input per filter. // -// The trade is descending over a whole window: the materialized path ANDs the -// chunk-sized terms once with roaring's bulk aggregation, while this one -// re-enters the algebra per slab. On the benchmarks in -// slab_match_bench_test.go, over a 300k-event corpus, that is +8µs on a -// descending page of the two-fat-term AND (9.9µs → 18.1µs) and +170µs on a -// descending full scan of one chunk-sized term measured on candidates alone -// (1.20ms → 1.37ms), which the fetch swallows end to end (10.88ms → 10.93ms). -// Descending pages that stop early are faster (148µs → 128µs), because the -// materialized path pays for the whole chunk before yielding anything. -// Seeding the walk at the slab holding the window's high bound, rather than -// re-clipping the accumulator there, would recover part of the scan cost. +// The trade is descending over a whole window. The whole-chunk union this +// replaced ANDed the chunk-sized terms once with roaring's bulk aggregation, +// where the walk re-enters the algebra per slab. Measured over a 300k-event +// corpus, that cost +8µs on a descending page of a two-fat-term AND (9.9µs → +// 18.1µs) and +170µs on a descending full scan of one chunk-sized term timed +// on candidates alone (1.20ms → 1.37ms), which the fetch swallows end to end +// (10.88ms → 10.93ms). Descending pages that stop early got faster (148µs → +// 128µs), because the union paid for the whole chunk before yielding +// anything. Seeding the walk at the slab holding the window's high bound, +// rather than re-clipping the accumulator there, would recover part of the +// scan cost. // // Ownership. acc is built by this call and is the only bitmap ever mutated. // The term bitmaps handed to AndAny may be shared, copy-on-write-marked mirror @@ -55,7 +52,6 @@ package event import ( "cmp" "context" - "iter" "slices" "github.com/RoaringBitmap/roaring/v2" @@ -71,53 +67,14 @@ import ( //nolint:gochecknoglobals // test seam; production never writes it var slabShift uint = 16 -// slabMatches is Matches over the slab-stepped engine: same arguments, same -// validation, same yields, same order, same match-all and absent-term -// handling. It is a drop-in alternative to Matches on the index-served paths, -// and the match-all path is shared verbatim. -func slabMatches( - ctx context.Context, r Reader, filters []Filter, window IDRange, - descending bool, firstBatch int, -) iter.Seq2[Match, error] { - return func(yield func(Match, error) bool) { - if err := validateMatchCall(ctx, r, filters, window); err != nil { - yield(Match{}, err) - return - } - if window.isEmpty() { - return - } - plans, uniqueKeys, matchAll := planIndexTerms(filters) - // Match-all path: identical to Matches'. The window is dense, so it - // streams Reader.FetchRange without touching the index. - if matchAll { - streamRange(ctx, r, window, descending, firstBatch, yield) - return - } - sources, err := lookupPostings(ctx, r, uniqueKeys) - if err != nil { - yield(Match{}, err) - return - } - st := newSlabStepper(plans, sources, window, descending) - // The twin of Matches' peek/IsEmpty early out: no filter survived - // group resolution, so nothing can match and no slab is worth - // evaluating. - if len(st.filters) == 0 { - return - } - streamSlabs(ctx, r, filters, st, descending, firstBatch, yield) - } -} - // slabTerms is one of a filter's term groups resolved out of the batched // lookup and held in whichever representation the index gave it: bitmaps for // dense (and cold) terms, borrowed id lists for sparse ones. The group's value // is the union of the two halves. // -// est is the summed cardinality of the present terms over the whole chunk — -// the same weight match_iter.go's resolveGroup computes, used the same way, to -// order a filter's AND. +// est is the summed cardinality of the present terms over the whole chunk. It +// ignores the window, so it ranks a filter's groups rather than counting a +// query's candidates, and it is what orders the AND. type slabTerms struct { bitmaps []*roaring.Bitmap lists [][]uint32 @@ -131,7 +88,7 @@ type slabFilter struct { // resolveSlabTerms collects the postings at slots, reporting false when every // one of them is absent from the index — the signal that the owning filter can -// match nothing, exactly as in resolveGroup. +// match nothing. // // A present term holding no ids contributes nothing to the union but still // keeps the group alive, which an absent term's caller-side skip would not do. @@ -258,8 +215,8 @@ func (f *slabFilter) eval(lo, hi uint32, sc *slabScratch) *roaring.Bitmap { } } // acc is nil only for a filter that named no group at all, which takes the - // match-all path upstream and never reaches here. Matching intersectOf, - // the unreachable case is the empty candidate set. + // match-all path upstream and never reaches here; the nil is read as the + // empty candidate set either way. return acc } @@ -411,10 +368,10 @@ func (s *slabStepper) appendUpTo(dst []uint32, n int) []uint32 { return dst } -// streamSlabs is the single streaming loop, shared by both directions: fill -// one internal batch of candidate ordinals out of the stepper, fetch, -// post-filter, yield the survivors. It is streamUnion and streamCandidates -// collapsed into one, because the stepper already hides the direction. +// streamSlabs is the streaming loop, shared by both directions: fill one +// internal batch of candidate ordinals out of the stepper, fetch, post-filter, +// yield the survivors. One loop serves both because the stepper already hides +// the direction. func streamSlabs( ctx context.Context, r Reader, filters []Filter, st *slabStepper, descending bool, firstBatch int, yield func(Match, error) bool, diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_bench_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_bench_test.go index 70b2f2767..ccde8d2e4 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_bench_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_bench_test.go @@ -1,35 +1,36 @@ package event -// slab_match_bench_test.go measures the slab engine against the cursor tree at -// two levels, because they answer different questions: +// slab_match_bench_test.go measures the match path at three levels, because +// they answer different questions: // -// - BenchmarkMatchPage is what a getEvents page costs end to end — candidate -// generation plus the fetch and post-filter both engines share. It is the -// number a request sees, and the shared half dilutes the engine difference -// exactly as production does. -// - BenchmarkMatchCandidates strips the shared half and times only the -// machinery being replaced: the cursor tree and the descending union -// against the slab stepper. +// - BenchmarkMatchPage is what a getEvents page costs end to end: candidate +// generation plus the fetch and post-filter around it. It is the number a +// request sees. +// - BenchmarkMatchCandidates strips the fetch and times candidate generation +// alone, drained in the batch sizes the streaming loop uses. +// - BenchmarkCandidateSlab times candidate generation over synthetic term +// geometries instead of a corpus, so a shape can be posed directly: two +// fat terms overlapping thinly, six fat terms, ten single-term filters. // -// Both run over the shaped corpus from slab_match_differential_test.go, sized -// to span several real 65536-id slabs. +// The first two run over the shaped corpus from slab_match_test.go, sized to +// span several real 65536-id slabs. import ( "context" + "errors" "iter" "slices" + "sync" "testing" + "github.com/RoaringBitmap/roaring/v2" "github.com/stretchr/testify/require" protocol "github.com/stellar/go-stellar-sdk/protocols/rpc" -) + "github.com/stellar/go-stellar-sdk/xdr" -// matchEngine is the shape both engines share, so a benchmark can name one -// and drive it without branching in the timed loop. -type matchEngine = func( - context.Context, Reader, []Filter, IDRange, bool, int, -) iter.Seq2[Match, error] + "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/chunk" +) // benchDir is one direction arm of every case. type benchDir struct { @@ -39,17 +40,6 @@ type benchDir struct { func benchDirs() []benchDir { return []benchDir{{"asc", false}, {"desc", true}} } -// benchArms pairs each engine with the name its benchmark reports under. -func benchArms() []struct { - name string - engine matchEngine -} { - return []struct { - name string - engine matchEngine - }{{"cursor", Matches}, {"slab", slabMatches}} -} - // benchCorpusSize spans four and a half slabs, so a page served from the head // of the window leaves most of the corpus untouched and a full scan crosses // every slab seam. @@ -91,7 +81,7 @@ type benchCase struct { func benchCases(f *shapedFixture) []benchCase { return []benchCase{ {"common", f.filterCommonPage(), 1000}, - {"spill", f.filterThinOverlap(), 1000}, + {"overlap", f.filterThinOverlap(), 1000}, {"fullscan", f.filterDenseOnly(), 0}, } } @@ -102,16 +92,14 @@ func BenchmarkMatchPage(b *testing.B) { r := diffPostingsReader{diffReader{f.corpus}} for _, c := range benchCases(f) { for _, dir := range benchDirs() { - for _, arm := range benchArms() { - b.Run(c.name+"/"+dir.name+"/"+arm.name, func(b *testing.B) { - benchPageRun(b, r, arm.engine, c, dir.desc) - }) - } + b.Run(c.name+"/"+dir.name, func(b *testing.B) { + benchPageRun(b, r, c, dir.desc) + }) } } } -func benchPageRun(b *testing.B, r Reader, engine matchEngine, c benchCase, desc bool) { +func benchPageRun(b *testing.B, r Reader, c benchCase, desc bool) { b.Helper() ctx := context.Background() window := IDRange{0, benchCorpusSize} @@ -119,7 +107,7 @@ func benchPageRun(b *testing.B, r Reader, engine matchEngine, c benchCase, desc var sink uint32 for b.Loop() { n := 0 - for m, err := range engine(ctx, r, c.filters, window, desc, c.limit) { + for m, err := range Matches(ctx, r, c.filters, window, desc, c.limit) { if err != nil { b.Fatal(err) } @@ -132,9 +120,8 @@ func benchPageRun(b *testing.B, r Reader, engine matchEngine, c benchCase, desc _ = sink } -// BenchmarkMatchCandidates times candidate generation alone: the cursor tree -// and the descending materialized union against the slab stepper, drained in -// the same batch sizes the streaming loops use, with no fetch and no +// BenchmarkMatchCandidates times candidate generation alone: the slab stepper +// drained in the batch sizes the streaming loop uses, with no fetch and no // post-filter in the way. func BenchmarkMatchCandidates(b *testing.B) { f := benchFixture(b) @@ -143,67 +130,22 @@ func BenchmarkMatchCandidates(b *testing.B) { for _, c := range benchCases(f) { for _, dir := range benchDirs() { - for _, name := range []string{"cursor", "slab"} { - b.Run(c.name+"/"+dir.name+"/"+name, func(b *testing.B) { - ctx := context.Background() - b.ReportAllocs() - var sink int - for b.Loop() { - if name == "slab" { - sink = drainSlabCandidates(ctx, b, r, c, window, dir.desc) - } else { - sink = drainCursorCandidates(ctx, b, r, c, window, dir.desc) - } - } - _ = sink - }) - } - } - } -} - -// drainCursorCandidates mirrors streamCandidates and streamUnion without the -// fetch: same batch sizing, same per-batch id collection, same early stop. -func drainCursorCandidates( - ctx context.Context, b *testing.B, r Reader, c benchCase, window IDRange, desc bool, -) int { - b.Helper() - plans, keys, matchAll := planIndexTerms(c.filters) - if matchAll { - b.Fatal("benchmark filters must reach the index") - } - if desc { - union, err := unionForFilters(ctx, r, plans, keys, window) - if err != nil { - b.Fatal(err) + b.Run(c.name+"/"+dir.name, func(b *testing.B) { + ctx := context.Background() + b.ReportAllocs() + var sink int + for b.Loop() { + sink = drainSlabCandidates(ctx, b, r, c, window, dir.desc) + } + _ = sink + }) } - it := union.ReverseIterator() - return drainBatches(c, func(ids []uint32, batch int) []uint32 { - for it.HasNext() && len(ids) < batch { - ids = append(ids, it.Next()) - } - return ids - }) - } - sources, err := lookupPostings(ctx, r, keys) - if err != nil { - b.Fatal(err) } - cand := candidateIter(plans, sources, window) - return drainBatches(c, func(ids []uint32, batch int) []uint32 { - for len(ids) < batch { - id, ok := cand.peek() - if !ok { - return ids - } - ids = append(ids, id) - cand.next() - } - return ids - }) } -// drainSlabCandidates is the same drain over the slab stepper. +// drainSlabCandidates is the streaming loop's batch cadence with the fetch +// removed: fill up to the batch size, stop when a fill comes back empty or the +// page is full. func drainSlabCandidates( ctx context.Context, b *testing.B, r Reader, c benchCase, window IDRange, desc bool, ) int { @@ -217,20 +159,12 @@ func drainSlabCandidates( b.Fatal(err) } st := newSlabStepper(plans, sources, window, desc) - return drainBatches(c, func(ids []uint32, batch int) []uint32 { - return st.appendUpTo(ids, batch) - }) -} -// drainBatches is the streaming loops' batch cadence with the fetch removed: -// fill up to the batch size, stop when a fill comes back empty or the page is -// full. Both arms share it so the harness cannot favor either. -func drainBatches(c benchCase, fill func(ids []uint32, batch int) []uint32) int { batch, rest := batchSizes(c.limit) ids := make([]uint32, 0, batch) total := 0 for { - ids = fill(ids[:0], batch) + ids = st.appendUpTo(ids[:0], batch) batch = rest if len(ids) == 0 { return total @@ -242,14 +176,428 @@ func drainBatches(c benchCase, fill func(ids []uint32, batch int) []uint32) int } } -// ───────────── the in-tree shape matrix, with a slab arm ───────────── +// ───────────── the synthetic index ───────────── + +// stubIndex is a Reader over an in-memory mirror and one shared payload, so a +// benchmark measures the match layer rather than the storage tier. FetchEvents +// reuses its buffer, keeping the per-batch fetch cost equal in both directions. +type stubIndex struct { + mirror *ConcurrentBitmaps + count uint32 + raw []byte + buf []Payload +} + +func (s *stubIndex) ChunkID() chunk.ID { return chunk.ID(0) } +func (s *stubIndex) EventCount() (uint32, error) { return s.count, nil } + +func (s *stubIndex) Offsets() (*LedgerOffsets, error) { + return nil, errors.New("stubIndex: Offsets is not part of the match path") +} + +func (s *stubIndex) LookupKeys(_ context.Context, keys []TermKey) ([]*roaring.Bitmap, error) { + out := make([]*roaring.Bitmap, len(keys)) + for i, k := range keys { + bm, err := s.mirror.Get(k) + if err != nil { + return nil, err + } + out[i] = bm + } + return out, nil +} + +func (s *stubIndex) FetchEvents(_ context.Context, ids []uint32) ([]Payload, error) { + if err := validateSortedEventIDs(ids); err != nil { + return nil, err + } + s.buf = s.buf[:0] + for range ids { + s.buf = append(s.buf, Payload{ContractEventBytes: s.raw}) + } + return s.buf, nil +} + +func (s *stubIndex) FetchRange(_ context.Context, start, count uint32) iter.Seq2[Payload, error] { + return func(yield func(Payload, error) bool) { + if err := validateFetchRange(start, count, s.count, s.ChunkID()); err != nil { + yield(Payload{}, err) + return + } + for range count { + if !yield(Payload{ContractEventBytes: s.raw}, nil) { + return + } + } + } +} + +func (s *stubIndex) All(ctx context.Context) iter.Seq2[Payload, error] { + return s.FetchRange(ctx, 0, s.count) +} + +// hotLikeIndex carries the same optional no-materialize seam HotStore does, so +// the benchmarks exercise the production fast path (sparse terms read in +// place) rather than the bitmap fallback. +type hotLikeIndex struct{ *stubIndex } + +func (h *hotLikeIndex) lookupPostings(_ context.Context, keys []TermKey) ([]postings, error) { + out := make([]postings, len(keys)) + for i, k := range keys { + out[i] = h.mirror.lookupPostings(k) + } + return out, nil +} + +var ( + _ Reader = (*stubIndex)(nil) + _ Reader = (*hotLikeIndex)(nil) + _ postingReader = (*hotLikeIndex)(nil) +) + +const ( + // ~4M events: half a production chunk (~9M), enough that the intermediates + // are the multi-container bitmaps a real chunk builds. + benchEvents = 1 << 22 + benchPage = 1000 // getEvents' max page size +) + +type benchIndex struct { + reader *hotLikeIndex + filters []Filter + window IDRange +} + +// newBenchIndex builds the synthetic chunk once for both directions: three +// dense terms (one near-total, like the event type; two selective) plus a +// long-tail sparse term below the mirror's promotion threshold, so the sparse +// read path is on the plan. +var newBenchIndex = sync.OnceValue(func() *benchIndex { + var contractA xdr.ContractId + contractA[0] = 0xA1 + topic := xdr.ScSymbol("bench-topic") + topicVal := xdr.ScVal{Type: xdr.ScValTypeScvSymbol, Sym: &topic} + topicRaw, err := topicVal.MarshalBinary() + if err != nil { + panic(err) + } + ev := xdr.ContractEvent{ + ContractId: &contractA, + Type: xdr.ContractEventTypeContract, + Body: xdr.ContractEventBody{ + V: 0, + V0: &xdr.ContractEventV0{Topics: []xdr.ScVal{topicVal}, Data: topicVal}, + }, + } + raw, err := ev.MarshalBinary() + if err != nil { + panic(err) + } + + // Dense terms go in through the frozen-Bitmaps constructor (roaring mode); + // the sparse one goes in through AddTo so it stays under the promotion + // threshold and is stored as a plain id list. + bms := NewBitmaps() + typeKey := EventTypeTermKey(xdr.ContractEventTypeContract) + contractKey := ComputeTermKey(contractA[:], FieldContractID) + topic1Key := ComputeTermKey(topicRaw, FieldTopic1) + everything := make([]uint32, 0, benchEvents) + contractIDs := make([]uint32, 0, benchEvents/3+1) + topic1IDs := make([]uint32, 0, benchEvents/7+1) + for id := range uint32(benchEvents) { + everything = append(everything, id) + if id%3 == 0 { + contractIDs = append(contractIDs, id) + } + if id%7 == 0 { + topic1IDs = append(topic1IDs, id) + } + } + bms.AddTo(typeKey, everything...) + bms.AddTo(contractKey, contractIDs...) + bms.AddTo(topic1Key, topic1IDs...) + mirror := NewConcurrentBitmapsFromBitmaps(bms) + + topic0Key := ComputeTermKey(topicRaw, FieldTopic0) + sparse := make([]uint32, 0, promotionThreshold-1) + for i := range uint32(promotionThreshold - 1) { + sparse = append(sparse, i*(benchEvents/promotionThreshold)) + } + mirror.AddTo(topic0Key, sparse...) + + eventType := xdr.ContractEventTypeContract + var topics [protocol.MaxTopicCount][]byte + topics[0] = topicRaw + return &benchIndex{ + reader: &hotLikeIndex{&stubIndex{ + mirror: mirror, count: benchEvents, raw: raw, + }}, + filters: []Filter{ + // Two dense groups AND-ed: the intersect arm. + {ContractID: contractA[:], EventType: &eventType}, + // One long-tail sparse group: the arm Get used to materialize a + // bitmap for on every request. + {Topics: topics}, + }, + // A sub-window, so both window edges are live. + window: IDRange{Start: benchEvents / 4, End: benchEvents * 3 / 4}, + } +}) -// BenchmarkCandidateSlab is the third arm of match_iter_test.go's per-shape -// A/B: the same plan, the same postings, the same page fingerprint, answered -// by the slab stepper instead of the cursor tree (BenchmarkCandidateTree) or -// the whole-window union (BenchmarkCandidateMaterialized). The shape matrix -// already holds the fat/thin-overlap geometries the alignment budget was built -// for, so this is the directly comparable number. +// benchMatches drives one page-sized request and stops, the shape a getEvents +// page actually has. +func benchMatches(b *testing.B, descending bool) { + b.Helper() + fx := newBenchIndex() + ctx := context.Background() + b.ReportAllocs() + b.ResetTimer() + for b.Loop() { + n := 0 + for _, err := range Matches(ctx, fx.reader, fx.filters, fx.window, descending, benchPage) { + if err != nil { + b.Fatal(err) + } + n++ + if n == benchPage { + break + } + } + if n != benchPage { + b.Fatalf("fixture sanity: want %d matches, got %d", benchPage, n) + } + } +} + +func BenchmarkMatchesAscending(b *testing.B) { benchMatches(b, false) } +func BenchmarkMatchesDescending(b *testing.B) { benchMatches(b, true) } + +// ───────────── the synthetic shape matrix ───────────── + +// benchFat is one fat term's cardinality against benchEvents: ~7% of the +// domain, the density at which roaring holds a term as bitmap containers. +const benchFat = 300_000 + +// benchRand is a deterministic xorshift. The shapes must be identical from run +// to run, and a fixed stride would hand a walking AND a regularity real +// postings do not have. +type benchRand uint64 + +func (r *benchRand) next() uint64 { + x := uint64(*r) + x ^= x << 13 + x ^= x >> 7 + x ^= x << 17 + *r = benchRand(x) + return x +} + +// scatter draws k ascending ids from one residue class, one per stride at a +// jittered offset. Terms on disjoint classes interleave at single-id +// granularity while sharing nothing, so a shape's overlap is exactly the class +// its terms share. +func scatter(rng *benchRand, domain, m, res uint32, k int) []uint32 { + if k == 0 { + return nil + } + class := (domain - res + m - 1) / m + stride := class / uint32(k) + if stride == 0 { + panic("scatter: residue class too small for k") + } + ids := make([]uint32, k) + for t := range k { + ids[t] = (uint32(t)*stride+uint32(rng.next()%uint64(stride)))*m + res + } + return ids +} + +// fatGroup builds n terms of card ids each, drawing private ids from one +// residue class per term plus one class every term holds, so the joint +// intersection is exactly that shared class. +func fatGroup(rng *benchRand, domain, mod, base uint32, n, card, shared int) [][]uint32 { + common := scatter(rng, domain, mod, base+uint32(n), shared) + out := make([][]uint32, n) + for i := range out { + ids := scatter(rng, domain, mod, base+uint32(i), card-shared) + ids = append(ids, common...) + slices.Sort(ids) + out[i] = ids + } + return out +} + +// benchShape is one synthetic candidate-set problem: a term corpus in the +// mirror, the plan resolved over it, and the page the engine must produce. +type benchShape struct { + name string + reader *hotLikeIndex + plans []termPlan + keys []TermKey + window IDRange + // wantCount and wantSum fingerprint the first page. The benchmark checks + // them every iteration, so a harness that stopped answering the query + // cannot post a fast number. + wantCount int + wantSum uint64 +} + +// newBenchShape indexes terms as one term each, resolves the window to most of +// the domain with both edges live, and fingerprints the first page off the +// naive materialized algebra in referenceCandidates. +func newBenchShape(name string, domain uint32, terms [][]uint32, plans []termPlan) *benchShape { + bms := NewBitmaps() + keys := make([]TermKey, len(terms)) + for i, ids := range terms { + keys[i] = TermKey{0: byte(i + 1)} + bms.AddTo(keys[i], ids...) + } + s := &benchShape{ + name: name, + reader: &hotLikeIndex{&stubIndex{ + mirror: NewConcurrentBitmapsFromBitmaps(bms), count: domain, + }}, + plans: plans, + keys: keys, + window: IDRange{Start: domain / 32, End: domain - domain/32}, + } + sources, err := lookupPostings(context.Background(), s.reader, s.keys) + if err != nil { + panic(err) + } + for _, id := range referenceCandidates(s.plans, sources, s.window) { + if s.wantCount == benchPage { + break + } + s.wantCount++ + s.wantSum += uint64(id) + } + return s +} + +// singleFilterPlan is one filter AND-ing n one-term groups: the intersect +// shapes' plan. +func singleFilterPlan(n int) []termPlan { + plan := make(termPlan, n) + for i := range plan { + plan[i] = []int{i} + } + return []termPlan{plan} +} + +// benchShapes is the shape matrix, each entry built on first use so a -bench +// selecting one shape pays for one shape. domain is a parameter so the +// correctness twin of the matrix can run the same geometries small. +func benchShapes(domain uint32) []struct { + name string + build func() *benchShape +} { + scale := func(n int) int { return max(1, n*int(domain)/benchEvents) } + fat := scale(benchFat) + // ~3% of a fat term: the partial overlap that makes an AND converge slowly + // without making it empty. + partial := fat * 3 / 100 + // Just over one page once the window clips it: the intersection too small + // to fill a page early, so the walk spans the window. + tiny := scale(1200) + + shapes := []struct { + name string + build func() *benchShape + }{ + {"a_and2_fat_3pct", func() *benchShape { + rng := benchRand(1) + return newBenchShape("a", domain, + fatGroup(&rng, domain, 3, 0, 2, fat, partial), singleFilterPlan(2)) + }}, + {"b_and3_fat_3pct", func() *benchShape { + rng := benchRand(2) + return newBenchShape("b", domain, + fatGroup(&rng, domain, 4, 0, 3, fat, partial), singleFilterPlan(3)) + }}, + {"c_and2_skew", func() *benchShape { + rng := benchRand(3) + // The small term is a subset of the fat one, spread over it, so the + // AND is entirely decided by the rare side. + big := scatter(&rng, domain, 1, 0, fat) + small := make([]uint32, 0, scale(2000)) + step := len(big) / cap(small) + for i := range cap(small) { + small = append(small, big[i*step]) + } + return newBenchShape("c", domain, + [][]uint32{big, small}, singleFilterPlan(2)) + }}, + {"d_and6_fat_tiny", func() *benchShape { + rng := benchRand(4) + return newBenchShape("d", domain, + fatGroup(&rng, domain, 7, 0, 6, fat, tiny), singleFilterPlan(6)) + }}, + {"e_or10_single_term", func() *benchShape { + rng := benchRand(5) + terms := fatGroup(&rng, domain, 11, 0, 10, scale(30_000), 0) + plans := make([]termPlan, len(terms)) + for i := range plans { + plans[i] = termPlan{{i}} + } + return newBenchShape("e", domain, terms, plans) + }}, + {"f_and2_fat_tiny", func() *benchShape { + rng := benchRand(6) + return newBenchShape("f", domain, + fatGroup(&rng, domain, 3, 0, 2, fat, tiny), singleFilterPlan(2)) + }}, + {"h_and2_fat_overlapping", func() *benchShape { + rng := benchRand(8) + // The serving default: one selective term AND-ed with a near-total + // one, so a page comes out of the window's first fraction. + selective := scatter(&rng, domain, 3, 0, fat) + nearAll := make([]uint32, 0, domain) + for id := range domain { + if id%50 != 7 { + nearAll = append(nearAll, id) + } + } + return newBenchShape("h", domain, + [][]uint32{selective, nearAll}, singleFilterPlan(2)) + }}, + {"g_and3_x4_filters", func() *benchShape { + rng := benchRand(7) + // Several filters, each AND-ing a few fat terms. Each filter owns + // four residue classes, so the filters overlap only where the union + // has to dedup them. + terms := make([][]uint32, 0, 12) + plans := make([]termPlan, 0, 4) + for f := range uint32(4) { + group := fatGroup(&rng, domain, 16, f*4, 3, scale(75_000), scale(2250)) + plan := make(termPlan, len(group)) + for i := range group { + plan[i] = []int{len(terms) + i} + } + terms = append(terms, group...) + plans = append(plans, plan) + } + return newBenchShape("g", domain, terms, plans) + }}, + } + return shapes +} + +// benchShapeCache keeps one built corpus per shape name. Benchmarks run one at +// a time, so a plain map suffices. +var benchShapeCache = map[string]*benchShape{} + +func shapeFor(name string, build func() *benchShape) *benchShape { + s, ok := benchShapeCache[name] + if !ok { + s = build() + benchShapeCache[name] = s + } + return s +} + +// BenchmarkCandidateSlab pulls one page of candidates per shape, with the +// fetch and the post-filter out of frame. func BenchmarkCandidateSlab(b *testing.B) { for _, sh := range benchShapes(benchEvents) { b.Run(sh.name, func(b *testing.B) { @@ -286,10 +634,58 @@ func benchSlabPage(b *testing.B, s *benchShape) { } } -// TestBenchShapesAgreeSlab is TestBenchShapesAgree's slab twin: every geometry -// the microbench above measures must be one the slab stepper answers exactly, -// so a shape can never post a number for a query it gets wrong. -func TestBenchShapesAgreeSlab(t *testing.T) { +// referenceCandidates is an independent, naive materialized implementation of +// the algebra the slab stepper answers: OR each group whole, AND the groups, +// OR across filters, then clip to the window. It builds every intermediate at +// full chunk width, which is what the stepper exists not to do, so agreement +// between the two is a real check rather than a restatement. +func referenceCandidates(plans []termPlan, sources []postings, window IDRange) []uint32 { + materialize := func(p postings) *roaring.Bitmap { + if bm := p.bitmap(); bm != nil { + return bm + } + bm := roaring.New() + bm.AddMany(p.ids) + return bm + } + union := roaring.New() + for _, plan := range plans { + var acc *roaring.Bitmap + missed := false + for _, slots := range plan { + group := roaring.New() + present := false + for _, s := range slots { + if sources[s].present() { + present = true + group.Or(materialize(sources[s])) + } + } + if !present { + missed = true + break + } + if acc == nil { + acc = group + } else { + acc.And(group) + } + } + if missed { + continue + } + union.Or(acc) + } + windowBM := roaring.New() + windowBM.AddRange(uint64(window.Start), uint64(window.End)) + union.And(windowBM) + return union.ToArray() +} + +// TestBenchShapesAgree runs the whole shape matrix small: every geometry the +// microbench measures must be one the stepper answers exactly, in both +// directions, so a shape can never post a number for a query it gets wrong. +func TestBenchShapesAgree(t *testing.T) { const domain = 1 << 16 for _, sh := range benchShapes(domain) { t.Run(sh.name, func(t *testing.T) { @@ -298,30 +694,27 @@ func TestBenchShapesAgreeSlab(t *testing.T) { require.NoError(t, err) want := referenceCandidates(s.plans, sources, s.window) - st := newSlabStepper(s.plans, sources, s.window, false) - got := []uint32{} - for { - before := len(got) - got = st.appendUpTo(got, before+512) - if len(got) == before { - break - } - } + got := drainStepper(newSlabStepper(s.plans, sources, s.window, false)) require.Equal(t, want, got) require.NotEmpty(t, got, "shape sanity: the plan must select something") // The descending arm reads the same set backwards. - rev := newSlabStepper(s.plans, sources, s.window, true) - gotDesc := []uint32{} - for { - before := len(gotDesc) - gotDesc = rev.appendUpTo(gotDesc, before+512) - if len(gotDesc) == before { - break - } - } + gotDesc := drainStepper(newSlabStepper(s.plans, sources, s.window, true)) slices.Reverse(gotDesc) require.Equal(t, want, gotDesc) }) } } + +// drainStepper pulls a stepper dry in 512-id fills, never returning nil so an +// empty result compares equal to a materialized bitmap's ToArray(). +func drainStepper(st *slabStepper) []uint32 { + out := []uint32{} + for { + before := len(out) + out = st.appendUpTo(out, before+512) + if len(out) == before { + return out + } + } +} diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_differential_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go similarity index 62% rename from cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_differential_test.go rename to cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go index 279b52584..9a385a3a4 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_differential_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go @@ -1,20 +1,18 @@ package event -// slab_match_differential_test.go holds slabMatches to being -// indistinguishable from Matches. Not "selects the same events" — -// byte-identical Match streams, same order, same ordinals, including every -// truncated prefix a paged consumer would stop on. +// slab_match_test.go covers the slab engine over a corpus built to hold each +// named query shape by construction: dense-only, sparse-only, mixed, absent, +// match-all, and the fat/fat thin-overlap shape whose two chunk-sized terms +// meet on a handful of ids. // -// Two corpora carry the matrix. The randomized one is the shape -// matches_differential_test.go drives the two existing paths with (300 events, -// 400 trials, both index seams, both directions), re-run at several slab -// widths so a small corpus still crosses many slab seams. The shaped one is -// large enough to span real 65536-id slabs and is built so each named query -// shape — dense-only, sparse-only, mixed, absent, match-all, and the fat/fat -// thin-overlap shape that makes the cursor tree spend its alignment budget — -// is present by construction rather than by luck. +// The answer every case is checked against is computed without the index — +// postFilter run over every ordinal in the corpus, then clipped to the window, +// the direction and the page. It shares no code with the term planning, the +// slab walk or the batching, so an engine that drops an id at a slab seam, +// yields one twice or emits out of order disagrees with it. import ( + "cmp" "context" "iter" "math/rand" @@ -22,15 +20,16 @@ import ( "strings" "testing" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" protocol "github.com/stellar/go-stellar-sdk/protocols/rpc" "github.com/stellar/go-stellar-sdk/xdr" ) -// drainMatches drains seq into a slice, stopping after limit items when -// limit is positive. The result is compared whole, so it pins the payload -// bytes, the ordinals and their order in one assertion. +// drainMatches drains seq into a slice, stopping after limit items when limit +// is positive. The result is compared whole, so it pins the payload bytes, the +// ordinals and their order in one assertion. func drainMatches(tb testing.TB, seq iter.Seq2[Match, error], limit int) []Match { tb.Helper() out := []Match{} @@ -44,9 +43,58 @@ func drainMatches(tb testing.TB, seq iter.Seq2[Match, error], limit int) []Match return out } -// diffCase is one (filters, window, direction, limit) query the two engines -// must answer identically. -type diffCase struct { +// ───────────────────────── the answer without the index ───────────────────── + +// matchingEvents is the corpus's whole answer to filters, computed by running +// the post-filter over every ordinal in it rather than by asking the index. +// +// An empty filter slice is the match-all shape, which postFilter is never +// reached with: it selects everything. +func matchingEvents(tb testing.TB, c *diffCorpus, filters []Filter) []Match { + tb.Helper() + ids := make([]uint32, len(c.raw)) + payloads := make([]Payload, len(c.raw)) + for id := range c.raw { + ids[id] = uint32(id) + payloads[id] = Payload{ContractEventBytes: c.raw[id]} + } + if len(filters) == 0 { + out := make([]Match, len(ids)) + for i := range ids { + out[i] = Match{Payload: payloads[i], Ordinal: ids[i]} + } + return out + } + out, err := postFilter(payloads, ids, filters) + require.NoError(tb, err) + return out +} + +// expectedStream clips the whole-corpus answer to one query: the window, the +// direction, and the page the consumer stops on. all is ascending by ordinal, +// so the window is a slice of it. +func expectedStream(all []Match, w IDRange, desc bool, limit int) []Match { + byOrdinal := func(m Match, id uint32) int { return cmp.Compare(m.Ordinal, id) } + lo, _ := slices.BinarySearchFunc(all, w.Start, byOrdinal) + hi, _ := slices.BinarySearchFunc(all, w.End, byOrdinal) + in := all[lo:hi] + + n := len(in) + if limit > 0 && n > limit { + n = limit + } + out := make([]Match, 0, n) + if !desc { + return append(out, in[:n]...) + } + for i := len(in) - 1; len(out) < n; i-- { + out = append(out, in[i]) + } + return out +} + +// queryCase is one (filters, window, direction, limit) query. +type queryCase struct { name string filters []Filter window IDRange @@ -54,15 +102,15 @@ type diffCase struct { limit int } -func requireSameStream(tb testing.TB, r Reader, c diffCase) []Match { +// requireStream drives one case through Matches and requires the stream the +// corpus says it must be. +func requireStream(tb testing.TB, r Reader, all []Match, c queryCase) []Match { tb.Helper() - ctx := context.Background() - want := drainMatches(tb, Matches(ctx, r, c.filters, c.window, c.desc, c.limit), c.limit) - got := drainMatches(tb, slabMatches(ctx, r, c.filters, c.window, c.desc, c.limit), c.limit) - require.Equal(tb, want, got, - "slabMatches diverged from Matches: case %q window %v desc=%v limit=%d", - c.name, c.window, c.desc, c.limit) - return want + got := drainMatches(tb, + Matches(context.Background(), r, c.filters, c.window, c.desc, c.limit), c.limit) + require.Equal(tb, expectedStream(all, c.window, c.desc, c.limit), got, + "case %q window %v desc=%v limit=%d", c.name, c.window, c.desc, c.limit) + return got } // ───────────────────────── the shaped corpus ───────────────────────── @@ -83,7 +131,7 @@ type shapedFixture struct { corpus *diffCorpus vocab *diffVocab n uint32 - // thin holds the ids where the two fat terms of the spill shape overlap. + // thin holds the ids where the two fat terms of the overlap shape meet. thin []uint32 // rareContract and rareTopic hold the ids carrying the sparse terms. rareContract []uint32 @@ -96,7 +144,7 @@ type shapedFixture struct { // - contract 2 is carried by a handful of ids: a term that stays sparse. // - topic 0 tracks the id's parity, except on the thin-overlap ids, so // {contract 0} ∧ {topic0 = odd-topic} is two chunk-sized terms meeting on -// a few ids — the shape whose alignment the cursor tree gives up on. +// a few ids. // - a second topic appears on a handful of ids, so the topic-count bucket // family has one dense bucket and one sparse bucket, making the // "at least one topic" group a mixed dense/sparse OR. @@ -312,27 +360,19 @@ func (f *shapedFixture) filterUnion() []Filter { } } -// ───────────────────────── the shaped matrix ───────────────────────── - // shapedCorpusSize spans slab 0 whole and part of slab 1, so every window // bound below is a real slab-relative position rather than a synthetic one. const shapedCorpusSize = 70_000 -func TestSlabMatches_ShapedDifferential(t *testing.T) { - f := newShapedFixture(t, shapedCorpusSize) - readers := []struct { - name string - r Reader - }{ - {"lookupKeys", diffReader{f.corpus}}, - {"postings", diffPostingsReader{diffReader{f.corpus}}}, - } +// namedShape is one query shape with the name its failures report under. +type namedShape struct { + name string + filters []Filter +} +func (f *shapedFixture) namedShapes() []namedShape { sysType := xdr.ContractEventTypeSystem - filterShapes := []struct { - name string - filters []Filter - }{ + return []namedShape{ {"dense only", f.filterDenseOnly()}, {"sparse only", f.filterSparseOnly()}, {"mixed groups", f.filterMixedGroups()}, @@ -350,6 +390,19 @@ func TestSlabMatches_ShapedDifferential(t *testing.T) { {"match all beside constrained", []Filter{{EventType: &sysType}, {}}}, {"exact topic count", []Filter{{TopicCount: TopicCountFilter{Count: 2, Exact: true}}}}, } +} + +// ───────────────────────── the shaped matrix ───────────────────────── + +func TestMatches_ShapedMatrix(t *testing.T) { + f := newShapedFixture(t, shapedCorpusSize) + readers := []struct { + name string + r Reader + }{ + {"lookupKeys", diffReader{f.corpus}}, + {"postings", diffPostingsReader{diffReader{f.corpus}}}, + } const slab = 1 << 16 windows := []struct { @@ -379,57 +432,56 @@ func TestSlabMatches_ShapedDifferential(t *testing.T) { // one that ends on a slab's last id are covered at every window bound. limits := []int{1, 1000} - for _, seam := range readers { - t.Run(seam.name, func(t *testing.T) { - for _, fs := range filterShapes { - for _, w := range windows { - for _, desc := range []bool{false, true} { - for _, limit := range limits { - c := diffCase{ - name: fs.name, - filters: fs.filters, - window: w.w, - desc: desc, - limit: limit, - } - requireSameStream(t, seam.r, c) - } + for _, sh := range f.namedShapes() { + all := matchingEvents(t, f.corpus, sh.filters) + for _, seam := range readers { + for _, w := range windows { + for _, desc := range []bool{false, true} { + for _, limit := range limits { + requireStream(t, seam.r, all, queryCase{ + name: sh.name + "/" + seam.name + "/" + w.name, + filters: sh.filters, + window: w.w, + desc: desc, + limit: limit, + }) } } } - }) + } } +} - // The bounded matrix above never reaches the end of a fat stream. This pass - // does: whole streams, unlimited, over the windows where "the end" is a - // different thing — the corpus end, a slab boundary, and a window living - // entirely inside the second slab. - t.Run("whole streams", func(t *testing.T) { - r := diffPostingsReader{diffReader{f.corpus}} - for _, fs := range filterShapes { - for _, w := range []IDRange{ - {0, shapedCorpusSize}, - {slab, shapedCorpusSize}, - {slab - 3, slab + 3}, - } { - for _, desc := range []bool{false, true} { - requireSameStream(t, r, diffCase{ - name: fs.name, - filters: fs.filters, - window: w, - desc: desc, - }) - } +// The bounded matrix above never reaches the end of a fat stream. This one +// does: whole streams, unlimited, over the windows where "the end" is a +// different thing — the corpus end, a slab boundary, and a window living +// entirely inside the second slab. +func TestMatches_WholeStreams(t *testing.T) { + f := newShapedFixture(t, shapedCorpusSize) + r := diffPostingsReader{diffReader{f.corpus}} + const slab = 1 << 16 + + for _, sh := range f.namedShapes() { + all := matchingEvents(t, f.corpus, sh.filters) + for _, w := range []IDRange{ + {0, shapedCorpusSize}, + {slab, shapedCorpusSize}, + {slab - 3, slab + 3}, + } { + for _, desc := range []bool{false, true} { + requireStream(t, r, all, queryCase{ + name: sh.name, filters: sh.filters, window: w, desc: desc, + }) } } - }) + } } // The shaped corpus must hold the shapes its filters are named for: // chunk-sized terms, sparse terms below the promotion threshold, and a thin // overlap. Drift in the corpus rules would otherwise turn the matrix above // into a weaker test without failing it. -func TestSlabMatches_ShapedFixtureIsWhatItClaims(t *testing.T) { +func TestMatches_ShapedFixtureIsWhatItClaims(t *testing.T) { f := newShapedFixture(t, shapedCorpusSize) r := diffPostingsReader{diffReader{f.corpus}} ctx := context.Background() @@ -462,58 +514,26 @@ func TestSlabMatches_ShapedFixtureIsWhatItClaims(t *testing.T) { Matches(ctx, r, f.filterPartlyAbsentGroup(), window, false, 0), 0), len(f.rareTopic), "the partly-absent group must select exactly its one present bucket") - // The spill shape is only the spill shape if the walk really would - // overrun its budget: both sides chunk-sized, the meeting point rare. + // The overlap shape is only the overlap shape if both sides are + // chunk-sized and their meeting point is rare. fat, err := r.lookupPostings(ctx, []TermKey{ ComputeTermKey(f.vocab.contracts[0], FieldContractID), ComputeTermKey(f.vocab.topicRaw[1], topicField(0)), }) require.NoError(t, err) for i, p := range fat { - require.Greater(t, p.estimate(), alignBudget, - "spill-shape term %d must be fatter than the alignment budget", i) - } -} - -// The cursor tree reaches its spill path by budget, so the shaped matrix above -// exercises it only at the default budget. Re-running the thin-overlap shape -// with the budget shrunk forces every AND in the reference engine through -// bulkAnd, which is the other answer slabMatches has to agree with. -func TestSlabMatches_SpilledReferenceAgrees(t *testing.T) { - f := newShapedFixture(t, shapedCorpusSize) - r := diffPostingsReader{diffReader{f.corpus}} - - defer func(n uint64) { alignBudget = n }(alignBudget) - for _, budget := range []uint64{0, 1, 7, 8192} { - alignBudget = budget - for _, w := range []IDRange{ - {0, shapedCorpusSize}, - {1 << 16, shapedCorpusSize}, - {(1 << 16) - 3, (1 << 16) + 3}, - } { - for _, desc := range []bool{false, true} { - for _, limit := range []int{0, 1, 3} { - requireSameStream(t, r, diffCase{ - name: "thin overlap spilled", - filters: f.filterThinOverlap(), - window: w, - desc: desc, - limit: limit, - }) - } - } - } + require.Greater(t, p.estimate(), uint64(shapedCorpusSize/3), + "thin-overlap term %d must be chunk-sized", i) } } // ───────────────────────── the randomized matrix ───────────────────────── -// TestSlabMatches_RandomizedDifferential re-runs matches_differential_test.go's -// matrix — same seed shape, same corpus size, same trial count, both index -// seams — asserting byte-identity against Matches rather than internal -// consistency, at several slab widths so a 300-event corpus still crosses -// dozens of slab seams. -func TestSlabMatches_RandomizedDifferential(t *testing.T) { +// TestMatches_RandomizedAgainstPostFilter drives random filters, windows and +// page sizes over a random corpus at several slab widths, so a 300-event +// corpus still crosses dozens of slab seams, and requires the corpus's own +// answer every time. +func TestMatches_RandomizedAgainstPostFilter(t *testing.T) { v := newDiffVocab(t) const corpusSize = 300 corpus := newDiffCorpus(t, rand.New(rand.NewSource(20260829)), v, corpusSize) @@ -540,7 +560,7 @@ func TestSlabMatches_RandomizedDifferential(t *testing.T) { rng := rand.New(rand.NewSource(int64(20260909 + shift))) matched := 0 for trial := range 400 { - matched += randomizedTrial(t, r, v, rng, corpusSize, trial) + matched += randomizedTrial(t, r, corpus, v, rng, corpusSize, trial) } require.Greater(t, matched, 2000, "fixture sanity: randomized queries selected too little") @@ -549,20 +569,22 @@ func TestSlabMatches_RandomizedDifferential(t *testing.T) { } } -// randomizedTrial runs one random query through both engines in both -// directions and returns how many matches the ascending run selected. +// randomizedTrial runs one random query in both directions and returns how +// many matches the ascending run selected. func randomizedTrial( - t *testing.T, r Reader, v *diffVocab, rng *rand.Rand, corpusSize, trial int, + t *testing.T, r Reader, corpus *diffCorpus, v *diffVocab, rng *rand.Rand, + corpusSize, trial int, ) int { t.Helper() filters := randomFilters(rng, v) start := uint32(rng.Intn(corpusSize + 1)) end := start + uint32(rng.Intn(corpusSize+1-int(start))) limit := []int{0, 0, 1, 3, 17, 200}[rng.Intn(6)] + all := matchingEvents(t, corpus, filters) matched := 0 for _, desc := range []bool{false, true} { - got := requireSameStream(t, r, diffCase{ + got := requireStream(t, r, all, queryCase{ name: "randomized", filters: filters, window: IDRange{Start: start, End: end}, @@ -592,7 +614,7 @@ func requireStrictOrder(t *testing.T, got []Match, desc bool, trial int) { // The slabShift seam must be invisible in the output: every width reproduces // the production width's stream exactly. -func TestSlabMatches_SlabWidthIsInvisible(t *testing.T) { +func TestMatches_SlabWidthIsInvisible(t *testing.T) { f := newShapedFixture(t, shapedCorpusSize) r := diffPostingsReader{diffReader{f.corpus}} ctx := context.Background() @@ -614,10 +636,10 @@ func TestSlabMatches_SlabWidthIsInvisible(t *testing.T) { for _, w := range windows { for _, desc := range []bool{false, true} { slabShift = 16 - want := drainMatches(t, slabMatches(ctx, r, filters, w, desc, 0), 250) + want := drainMatches(t, Matches(ctx, r, filters, w, desc, 0), 250) for _, shift := range []uint{3, 8, 13, 17, 20, 31} { slabShift = shift - got := drainMatches(t, slabMatches(ctx, r, filters, w, desc, 0), 250) + got := drainMatches(t, Matches(ctx, r, filters, w, desc, 0), 250) require.Equal(t, want, got, "case %d window %v desc=%v: slabShift %d changed the stream", ci, w, desc, shift) @@ -629,21 +651,22 @@ func TestSlabMatches_SlabWidthIsInvisible(t *testing.T) { // ───────────────────────── the candidate-set pin ───────────────────────── -// fetchTracer records every ordinal the engine fetches. +// fetchTracer records the ordinals of every FetchEvents call, one entry per +// call, in call order. // -// Output equality alone cannot see a candidate-set bug that only widens the -// set: postFilter re-verifies every fetched event against the filters, so a +// Output equality alone cannot see a candidate set that is merely too wide: +// postFilter re-verifies every fetched event against the filters, so a // superset of the true matches still yields the right stream and only costs -// more I/O. Recording the fetches turns "same answer" into "same work", which -// is the claim a replacement engine has to make. +// more I/O. Recording the fetches turns "the right answer" into "the right +// work". type fetchTracer struct { diffPostingsReader - fetched *[]uint32 + batches *[][]uint32 } func (r fetchTracer) FetchEvents(ctx context.Context, ids []uint32) ([]Payload, error) { - *r.fetched = append(*r.fetched, ids...) + *r.batches = append(*r.batches, slices.Clone(ids)) return r.diffPostingsReader.FetchEvents(ctx, ids) } @@ -652,12 +675,15 @@ var ( _ postingReader = fetchTracer{} ) -// TestSlabMatches_SameCandidatesFetched pins that the two engines resolve the -// same candidate set, batch for batch and in the same order — not merely the -// same surviving matches. -func TestSlabMatches_SameCandidatesFetched(t *testing.T) { +// TestMatches_FetchesOnlyTrueCandidates pins the candidate set itself: the +// ordinals the engine fetches are the query's true matches, in emission order, +// with nothing extra read and nothing skipped. A consumer that stops after a +// page has fetched only the batches that page spans. +// +// The match-all shapes are excluded because they never reach the index: they +// stream FetchRange instead. +func TestMatches_FetchesOnlyTrueCandidates(t *testing.T) { f := newShapedFixture(t, shapedCorpusSize) - ctx := context.Background() const slab = 1 << 16 shapes := [][]Filter{ @@ -680,27 +706,116 @@ func TestSlabMatches_SameCandidatesFetched(t *testing.T) { {33_333, shapedCorpusSize}, } - trace := func( - engine func(context.Context, Reader, []Filter, IDRange, bool, int) iter.Seq2[Match, error], - filters []Filter, w IDRange, desc bool, limit int, - ) []uint32 { - fetched := []uint32{} - r := fetchTracer{diffPostingsReader{diffReader{f.corpus}}, &fetched} - drainMatches(t, engine(ctx, r, filters, w, desc, limit), limit) - return fetched - } - for si, filters := range shapes { + all := matchingEvents(t, f.corpus, filters) for _, w := range windows { for _, desc := range []bool{false, true} { for _, limit := range []int{0, 1, 1000} { - want := trace(Matches, filters, w, desc, limit) - got := trace(slabMatches, filters, w, desc, limit) - require.Equal(t, want, got, - "shape %d window %v desc=%v limit=%d: the engines fetched "+ - "different candidates", si, w, desc, limit) + want := expectedStream(all, w, desc, 0) + fetched := traceFetches(t, f, filters, w, desc, limit) + + wantIDs := make([]uint32, 0, len(want)) + for _, m := range want { + wantIDs = append(wantIDs, m.Ordinal) + } + require.LessOrEqual(t, len(fetched), len(wantIDs), + "shape %d window %v desc=%v limit=%d: fetched an ordinal "+ + "outside the answer", si, w, desc, limit) + require.Equal(t, wantIDs[:len(fetched)], fetched, + "shape %d window %v desc=%v limit=%d: the fetched "+ + "candidates are not the answer's leading run", + si, w, desc, limit) + require.GreaterOrEqual(t, len(fetched), pageFloor(limit, len(wantIDs)), + "shape %d window %v desc=%v limit=%d: the page was served "+ + "without fetching enough candidates", si, w, desc, limit) } } } } } + +// traceFetches drives one query and returns the ordinals it fetched in +// emission order. FetchEvents takes ascending ids, so a descending batch is +// fetched flipped; flipping it back recovers the order the stream emits in. +func traceFetches( + t *testing.T, f *shapedFixture, filters []Filter, w IDRange, desc bool, limit int, +) []uint32 { + t.Helper() + batches := [][]uint32{} + r := fetchTracer{diffPostingsReader{diffReader{f.corpus}}, &batches} + drainMatches(t, Matches(context.Background(), r, filters, w, desc, limit), limit) + + out := []uint32{} + for _, b := range batches { + if desc { + slices.Reverse(b) + } + out = append(out, b...) + } + return out +} + +// pageFloor is how many candidates a query must have fetched to have served +// its page: the page itself, or the whole answer when it is shorter. +func pageFloor(limit, answer int) int { + if limit <= 0 { + return answer + } + return min(limit, answer) +} + +// ───────────────────────── the planning step ───────────────────────── + +// What a group reports about itself: presence, and its summed weight. +func TestResolveSlabTerms(t *testing.T) { + sources := []postings{ + sparseSource(1, 2), + {}, // absent + denseSource(2, 3, 4), + } + + g, ok := resolveSlabTerms(sources, []int{0}) + require.True(t, ok) + assert.Equal(t, uint64(2), g.est) + assert.Equal(t, [][]uint32{{1, 2}}, g.lists) + assert.Empty(t, g.bitmaps, "a sparse term stays an id list") + + g, ok = resolveSlabTerms(sources, []int{0, 2}) + require.True(t, ok) + assert.Equal(t, uint64(5), g.est, "a group's terms sum, overlaps double-counted") + assert.Len(t, g.bitmaps, 1) + assert.Len(t, g.lists, 1, "a mixed group keeps both representations") + + g, ok = resolveSlabTerms(sources, []int{1, 2}) + require.True(t, ok) + assert.Equal(t, uint64(3), g.est, "an absent term adds nothing") + + g, ok = resolveSlabTerms(sources, []int{1}) + assert.False(t, ok, "a group of absent terms drops its filter") + assert.Equal(t, uint64(0), g.est) +} + +// The rarest group leads the AND however the plan named its groups, so the +// accumulator shrinks fastest and a group that empties it ends the slab before +// the fat groups are read. +func TestResolveSlabFiltersOrdersRarestFirst(t *testing.T) { + sources := []postings{ + denseSource(1, 2, 3, 4, 5, 6, 7, 8), // 0: the fat group + denseSource(2, 4, 6, 8), // 1 + denseSource(4, 8), // 2: the rare group + {}, // 3: absent + } + + got := resolveSlabFilters([]termPlan{{{0}, {2}, {1}}}, sources) + require.Len(t, got, 1) + ests := make([]uint64, 0, len(got[0].groups)) + for _, g := range got[0].groups { + ests = append(ests, g.est) + } + assert.Equal(t, []uint64{2, 4, 8}, ests, "the groups are reordered rarest first") + + assert.Empty(t, resolveSlabFilters([]termPlan{{{0}, {3}}}, sources), + "a filter naming a wholly absent group is dropped") + assert.Len(t, resolveSlabFilters([]termPlan{{{0}, {3}}, {{1}}}, sources), 1, + "the drop takes only its own filter") +} From f011769f5c219c6a6ef3695f63b5c20e792eeaf2 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 15:46:46 +0000 Subject: [PATCH 24/41] =?UTF-8?q?events:=20one=20oracle=20for=20the=20matc?= =?UTF-8?q?h=20path=20=E2=80=94=20drop=20the=20benchmarks=20and=20the=20tw?= =?UTF-8?q?in=20suites?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The benchmark harness lives out of tree; its numbers are already in the PR body and the commit messages that measured them. slab_match_bench_test.go goes whole, TestBenchShapesAgree included: the geometries it checked come back as two case lines in the shaped matrix, a five-group AND and an eight-filter union, checked against the index-free answer rather than against a materialized restatement of the same algebra. Two test suites were saying what the oracle suites already say. TestMatches_AscendingDescendingDifferential required the ascending stream reversed to equal the descending one over both index seams; TestMatches_RandomizedAgainstPostFilter runs the same corpus, seams and directions against postFilter, which is the stronger claim. TestMatches_SlabWidthIsInvisible required every slab width to reproduce the production width's stream; the randomized suite now sweeps shifts 1, 2, 4, 8 and 16 against the oracle instead, and the shaped matrix already covers the single-slab window at the production width. The shared fixture stays, with its testing.TB vocabulary twin collapsed into one. slab_match.go's comments lose the pseudo-code walkthrough of eval, the alternatives it does not implement, and the sentences the function docs below already carry. The laziness contract, the descending trade and the ownership rules are untouched. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../stores/event/matches_differential_test.go | 76 +- .../internal/rpcv2/stores/event/slab_match.go | 53 +- .../stores/event/slab_match_bench_test.go | 720 ------------------ .../rpcv2/stores/event/slab_match_test.go | 146 ++-- 4 files changed, 97 insertions(+), 898 deletions(-) delete mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_bench_test.go diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go index 67d02e6c5..0e83f7086 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go @@ -1,17 +1,18 @@ package event -// Full-Matches differential across the index's two read seams: LookupKeys, +// The in-memory chunk the match-path tests run against, and the borrow-safety +// gate over it. +// +// The corpus is served through both of the index's read seams: LookupKeys, // which materializes every term as a bitmap, and lookupPostings, which hands -// sparse terms back as borrowed id lists. Both must select the same events -// over randomized corpora, filters and windows, and the ascending stream -// reversed must equal the descending one. +// sparse terms back as borrowed id lists. slab_match_test.go drives both +// against an answer computed without the index. import ( "context" "errors" "iter" "math/rand" - "slices" "testing" "github.com/RoaringBitmap/roaring/v2" @@ -108,25 +109,23 @@ type diffVocab struct { types []xdr.ContractEventType } -func newDiffVocab(t *testing.T) *diffVocab { - t.Helper() +func newDiffVocab(tb testing.TB) *diffVocab { + tb.Helper() v := &diffVocab{types: []xdr.ContractEventType{ xdr.ContractEventTypeSystem, xdr.ContractEventTypeContract, xdr.ContractEventTypeDiagnostic, }} for i := range 4 { - var cid xdr.ContractId - cid[0] = byte(0xC0 + i) + cid := xdr.ContractId{0: byte(0xC0 + i)} v.contracts = append(v.contracts, cid[:]) } for _, name := range []string{"alpha", "beta", "gamma", "delta", "epsilon"} { sym := xdr.ScSymbol(name) val := xdr.ScVal{Type: xdr.ScValTypeScvSymbol, Sym: &sym} raw, err := val.MarshalBinary() - require.NoError(t, err) - v.topics = append(v.topics, val) - v.topicRaw = append(v.topicRaw, raw) + require.NoError(tb, err) + v.topics, v.topicRaw = append(v.topics, val), append(v.topicRaw, raw) } return v } @@ -211,59 +210,6 @@ func collectOrdinals(t *testing.T, r Reader, filters []Filter, w IDRange, desc b return out } -// Drives randomized queries through both index seams in both directions; the -// ascending stream reversed must equal the descending stream. -func TestMatches_AscendingDescendingDifferential(t *testing.T) { - rng := rand.New(rand.NewSource(20260829)) - v := newDiffVocab(t) - const corpusSize = 300 - corpus := newDiffCorpus(t, rng, v, corpusSize) - - // Shrink the batch so multi-batch seams are exercised on a small corpus. - defer func(n int) { matchBatchSize = n }(matchBatchSize) - matchBatchSize = 7 - - readers := []struct { - name string - r Reader - }{ - {"lookupKeys", diffReader{corpus}}, - {"postings", diffPostingsReader{diffReader{corpus}}}, - } - for _, seam := range readers { - name, r := seam.name, seam.r - t.Run(name, func(t *testing.T) { - matched := 0 - for trial := range 400 { - filters := randomFilters(rng, v) - start := uint32(rng.Intn(corpusSize + 1)) - end := start + uint32(rng.Intn(corpusSize+1-int(start))) - w := IDRange{Start: start, End: end} - - asc := collectOrdinals(t, r, filters, w, false) - desc := collectOrdinals(t, r, filters, w, true) - - for i := 1; i < len(asc); i++ { - require.Less(t, asc[i-1], asc[i], - "trial %d: ascending ordinals must be strictly increasing "+ - "(an equal pair is a union dedup bug)", trial) - } - matched += len(asc) - for _, id := range asc { - require.GreaterOrEqual(t, id, w.Start, "trial %d: below window", trial) - require.Less(t, id, w.End, "trial %d: End must be exclusive", trial) - } - slices.Reverse(desc) - require.Equal(t, asc, desc, - "trial %d: window %v filters %+v", trial, w, filters) - } - // Guard against a vacuous pass: the queries must select events. - require.Greater(t, matched, 5000, - "fixture sanity: randomized queries selected too little") - }) - } -} - // Turns the borrow contract into a race-detector gate: the match path reads // mirror snapshots in place while AddTo publishes new termStates on the same // keys, including the sparse-to-dense promotion. Under -race any write diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go index df2b7c802..0cd099966 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go @@ -4,23 +4,13 @@ package event // one roaring slab at a time — 65536 ids, the span of exactly one container — // and answers the whole filter algebra inside that slab. // -// Clip-early slab evaluation, per slab, per filter: -// -// acc := roaring.New() // ours, shared with nobody -// acc.AddRange(slabLo, slabHi) // the caller's window, clipped to this slab -// acc.AndAny(group0Terms...) // acc ∩ (t1 ∪ t2 ∪ …), in place on acc -// acc.AndAny(group1Terms...) // …AND the next group, still in place -// // The window is applied first rather than last, so no id outside it is ever -// read. The successive in-place AndAny is the AND across a filter's groups, -// and roaring.FastOr unions the surviving filters' per-slab results. Every -// input to a slab's evaluation is a single container and every intermediate -// the engine allocates holds at most one. +// read. Every input to a slab's evaluation is a single container and every +// intermediate the engine allocates holds at most one. // // Direction is the slab walk order and nothing else: ascending walks slabs low // to high and reads each result forward, descending walks them high to low and -// reads each result backward. One code path serves both, with no gallop, no -// alignment budget, no spill and no separate descending machinery. +// reads each result backward. One code path serves both. // // Laziness is per slab, not per id. A consumer that stops after one page has // evaluated only the slabs that page spans, and the cost inside a slab is @@ -57,9 +47,8 @@ import ( "github.com/RoaringBitmap/roaring/v2" ) -// slabShift sets the slab width as a power of two: 1<<16 is one roaring -// container, so a slab's evaluation touches exactly one container per input -// and every result the engine builds is single-container. +// slabShift sets the slab width as a power of two: 1<<16 is exactly one +// roaring container. // // A var rather than a const so in-package tests can shrink it and drive slab // seams over a small corpus. It never changes what a stream yields. @@ -115,10 +104,10 @@ func resolveSlabTerms(sources []postings, slots []int) (slabTerms, bool) { return g, present } -// resolveSlabFilters is the whole planning step: resolve every filter's -// groups, drop the filters that named an entirely absent group, and order each -// survivor's groups rarest first so the accumulator shrinks fastest and a -// group that empties it ends the slab before the fat groups are read. +// resolveSlabFilters is the planning step: resolve every filter's groups, drop +// the filters that named an entirely absent group, and order each survivor's +// groups rarest first so the accumulator shrinks fastest and a group that +// empties it ends the slab before the fat groups are read. func resolveSlabFilters(plans []termPlan, sources []postings) []slabFilter { out := make([]slabFilter, 0, len(plans)) for _, plan := range plans { @@ -194,10 +183,8 @@ func (sc *slabScratch) inputs(g *slabTerms, lo, hi uint32) []*roaring.Bitmap { // caller owns, or nil when f matches nothing there. // // The accumulator starts as the slab window itself and is narrowed group by -// group in place. AndAny is x.And(FastOr(args)) without the intermediate -// union, so one call is a whole group; a single-group filter is therefore one -// AddRange plus one AndAny, with no FastAnd and no clone-the-input shortcut to -// guard against. +// group in place: AndAny is x.And(FastOr(args)) without the intermediate +// union, so one call is a whole group. func (f *slabFilter) eval(lo, hi uint32, sc *slabScratch) *roaring.Bitmap { var acc *roaring.Bitmap for i := range f.groups { @@ -259,7 +246,7 @@ func newSlabStepper( // nextBounds returns the next slab's [lo, hi) clipped to the window, walking // away from the cursor in the query's direction. The first slab is clipped at -// the cursor by these bounds alone — there is no seek. +// the cursor by these bounds alone. func (s *slabStepper) nextBounds() (uint32, uint32, bool) { if s.done { return 0, 0, false @@ -291,8 +278,8 @@ func (s *slabStepper) nextBounds() (uint32, uint32, bool) { } // evalSlab is the union across filters of their per-slab results, or nil when -// the slab holds nothing. Every input is this call's own bitmap, so FastOr's -// single-input clone shortcut is unreachable and would be harmless anyway. +// the slab holds nothing. Every input is a bitmap this call owns, so the +// caller owns the union whichever path FastOr takes. func (s *slabStepper) evalSlab(lo, hi uint32) *roaring.Bitmap { s.perFilter = s.perFilter[:0] for i := range s.filters { @@ -353,9 +340,8 @@ func (s *slabStepper) appendUpTo(dst []uint32, n int) []uint32 { } continue } - // NextMany fills the tail directly, so an ascending page is copied out - // of the slab's containers in bulk rather than id by id. It returns - // short only at the end of the bitmap. + // NextMany fills the tail in bulk and returns short only at the end + // of the bitmap, so a short fill is this slab's last id. want := n - len(dst) base := len(dst) dst = slices.Grow(dst, want)[:base+want] @@ -368,10 +354,9 @@ func (s *slabStepper) appendUpTo(dst []uint32, n int) []uint32 { return dst } -// streamSlabs is the streaming loop, shared by both directions: fill one -// internal batch of candidate ordinals out of the stepper, fetch, post-filter, -// yield the survivors. One loop serves both because the stepper already hides -// the direction. +// streamSlabs is the streaming loop, shared by both directions because the +// stepper already hides direction: fill one internal batch of candidate +// ordinals out of the stepper, fetch, post-filter, yield the survivors. func streamSlabs( ctx context.Context, r Reader, filters []Filter, st *slabStepper, descending bool, firstBatch int, yield func(Match, error) bool, diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_bench_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_bench_test.go deleted file mode 100644 index ccde8d2e4..000000000 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_bench_test.go +++ /dev/null @@ -1,720 +0,0 @@ -package event - -// slab_match_bench_test.go measures the match path at three levels, because -// they answer different questions: -// -// - BenchmarkMatchPage is what a getEvents page costs end to end: candidate -// generation plus the fetch and post-filter around it. It is the number a -// request sees. -// - BenchmarkMatchCandidates strips the fetch and times candidate generation -// alone, drained in the batch sizes the streaming loop uses. -// - BenchmarkCandidateSlab times candidate generation over synthetic term -// geometries instead of a corpus, so a shape can be posed directly: two -// fat terms overlapping thinly, six fat terms, ten single-term filters. -// -// The first two run over the shaped corpus from slab_match_test.go, sized to -// span several real 65536-id slabs. - -import ( - "context" - "errors" - "iter" - "slices" - "sync" - "testing" - - "github.com/RoaringBitmap/roaring/v2" - "github.com/stretchr/testify/require" - - protocol "github.com/stellar/go-stellar-sdk/protocols/rpc" - "github.com/stellar/go-stellar-sdk/xdr" - - "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/chunk" -) - -// benchDir is one direction arm of every case. -type benchDir struct { - name string - desc bool -} - -func benchDirs() []benchDir { return []benchDir{{"asc", false}, {"desc", true}} } - -// benchCorpusSize spans four and a half slabs, so a page served from the head -// of the window leaves most of the corpus untouched and a full scan crosses -// every slab seam. -const benchCorpusSize = 300_000 - -// The corpus is expensive enough to build that every benchmark shares one. -var benchFixtureCache *shapedFixture - -func benchFixture(tb testing.TB) *shapedFixture { - tb.Helper() - if benchFixtureCache == nil { - benchFixtureCache = newShapedFixture(tb, benchCorpusSize) - } - return benchFixtureCache -} - -// filterCommonPage is the everyday getEvents shape: a chunk-sized contract -// term ANDed with a chunk-sized topic term, beside a second filter naming a -// rare contract whose postings are still a sparse id list. Matches are common -// enough that a 1000-item page fills from the head of the window. -func (f *shapedFixture) filterCommonPage() []Filter { - return []Filter{ - { - ContractID: f.vocab.contracts[1], - Topics: [protocol.MaxTopicCount][]byte{0: f.vocab.topicRaw[1]}, - }, - {ContractID: f.vocab.contracts[2]}, - } -} - -// benchCase is one query shape plus the page size a consumer stops at. A zero -// limit is a full-window scan. -type benchCase struct { - name string - filters []Filter - limit int -} - -func benchCases(f *shapedFixture) []benchCase { - return []benchCase{ - {"common", f.filterCommonPage(), 1000}, - {"overlap", f.filterThinOverlap(), 1000}, - {"fullscan", f.filterDenseOnly(), 0}, - } -} - -// BenchmarkMatchPage times one served page, fetch and post-filter included. -func BenchmarkMatchPage(b *testing.B) { - f := benchFixture(b) - r := diffPostingsReader{diffReader{f.corpus}} - for _, c := range benchCases(f) { - for _, dir := range benchDirs() { - b.Run(c.name+"/"+dir.name, func(b *testing.B) { - benchPageRun(b, r, c, dir.desc) - }) - } - } -} - -func benchPageRun(b *testing.B, r Reader, c benchCase, desc bool) { - b.Helper() - ctx := context.Background() - window := IDRange{0, benchCorpusSize} - b.ReportAllocs() - var sink uint32 - for b.Loop() { - n := 0 - for m, err := range Matches(ctx, r, c.filters, window, desc, c.limit) { - if err != nil { - b.Fatal(err) - } - sink, n = m.Ordinal, n+1 - if c.limit > 0 && n == c.limit { - break - } - } - } - _ = sink -} - -// BenchmarkMatchCandidates times candidate generation alone: the slab stepper -// drained in the batch sizes the streaming loop uses, with no fetch and no -// post-filter in the way. -func BenchmarkMatchCandidates(b *testing.B) { - f := benchFixture(b) - r := diffPostingsReader{diffReader{f.corpus}} - window := IDRange{0, benchCorpusSize} - - for _, c := range benchCases(f) { - for _, dir := range benchDirs() { - b.Run(c.name+"/"+dir.name, func(b *testing.B) { - ctx := context.Background() - b.ReportAllocs() - var sink int - for b.Loop() { - sink = drainSlabCandidates(ctx, b, r, c, window, dir.desc) - } - _ = sink - }) - } - } -} - -// drainSlabCandidates is the streaming loop's batch cadence with the fetch -// removed: fill up to the batch size, stop when a fill comes back empty or the -// page is full. -func drainSlabCandidates( - ctx context.Context, b *testing.B, r Reader, c benchCase, window IDRange, desc bool, -) int { - b.Helper() - plans, keys, matchAll := planIndexTerms(c.filters) - if matchAll { - b.Fatal("benchmark filters must reach the index") - } - sources, err := lookupPostings(ctx, r, keys) - if err != nil { - b.Fatal(err) - } - st := newSlabStepper(plans, sources, window, desc) - - batch, rest := batchSizes(c.limit) - ids := make([]uint32, 0, batch) - total := 0 - for { - ids = st.appendUpTo(ids[:0], batch) - batch = rest - if len(ids) == 0 { - return total - } - total += len(ids) - if c.limit > 0 && total >= c.limit { - return total - } - } -} - -// ───────────── the synthetic index ───────────── - -// stubIndex is a Reader over an in-memory mirror and one shared payload, so a -// benchmark measures the match layer rather than the storage tier. FetchEvents -// reuses its buffer, keeping the per-batch fetch cost equal in both directions. -type stubIndex struct { - mirror *ConcurrentBitmaps - count uint32 - raw []byte - buf []Payload -} - -func (s *stubIndex) ChunkID() chunk.ID { return chunk.ID(0) } -func (s *stubIndex) EventCount() (uint32, error) { return s.count, nil } - -func (s *stubIndex) Offsets() (*LedgerOffsets, error) { - return nil, errors.New("stubIndex: Offsets is not part of the match path") -} - -func (s *stubIndex) LookupKeys(_ context.Context, keys []TermKey) ([]*roaring.Bitmap, error) { - out := make([]*roaring.Bitmap, len(keys)) - for i, k := range keys { - bm, err := s.mirror.Get(k) - if err != nil { - return nil, err - } - out[i] = bm - } - return out, nil -} - -func (s *stubIndex) FetchEvents(_ context.Context, ids []uint32) ([]Payload, error) { - if err := validateSortedEventIDs(ids); err != nil { - return nil, err - } - s.buf = s.buf[:0] - for range ids { - s.buf = append(s.buf, Payload{ContractEventBytes: s.raw}) - } - return s.buf, nil -} - -func (s *stubIndex) FetchRange(_ context.Context, start, count uint32) iter.Seq2[Payload, error] { - return func(yield func(Payload, error) bool) { - if err := validateFetchRange(start, count, s.count, s.ChunkID()); err != nil { - yield(Payload{}, err) - return - } - for range count { - if !yield(Payload{ContractEventBytes: s.raw}, nil) { - return - } - } - } -} - -func (s *stubIndex) All(ctx context.Context) iter.Seq2[Payload, error] { - return s.FetchRange(ctx, 0, s.count) -} - -// hotLikeIndex carries the same optional no-materialize seam HotStore does, so -// the benchmarks exercise the production fast path (sparse terms read in -// place) rather than the bitmap fallback. -type hotLikeIndex struct{ *stubIndex } - -func (h *hotLikeIndex) lookupPostings(_ context.Context, keys []TermKey) ([]postings, error) { - out := make([]postings, len(keys)) - for i, k := range keys { - out[i] = h.mirror.lookupPostings(k) - } - return out, nil -} - -var ( - _ Reader = (*stubIndex)(nil) - _ Reader = (*hotLikeIndex)(nil) - _ postingReader = (*hotLikeIndex)(nil) -) - -const ( - // ~4M events: half a production chunk (~9M), enough that the intermediates - // are the multi-container bitmaps a real chunk builds. - benchEvents = 1 << 22 - benchPage = 1000 // getEvents' max page size -) - -type benchIndex struct { - reader *hotLikeIndex - filters []Filter - window IDRange -} - -// newBenchIndex builds the synthetic chunk once for both directions: three -// dense terms (one near-total, like the event type; two selective) plus a -// long-tail sparse term below the mirror's promotion threshold, so the sparse -// read path is on the plan. -var newBenchIndex = sync.OnceValue(func() *benchIndex { - var contractA xdr.ContractId - contractA[0] = 0xA1 - topic := xdr.ScSymbol("bench-topic") - topicVal := xdr.ScVal{Type: xdr.ScValTypeScvSymbol, Sym: &topic} - topicRaw, err := topicVal.MarshalBinary() - if err != nil { - panic(err) - } - ev := xdr.ContractEvent{ - ContractId: &contractA, - Type: xdr.ContractEventTypeContract, - Body: xdr.ContractEventBody{ - V: 0, - V0: &xdr.ContractEventV0{Topics: []xdr.ScVal{topicVal}, Data: topicVal}, - }, - } - raw, err := ev.MarshalBinary() - if err != nil { - panic(err) - } - - // Dense terms go in through the frozen-Bitmaps constructor (roaring mode); - // the sparse one goes in through AddTo so it stays under the promotion - // threshold and is stored as a plain id list. - bms := NewBitmaps() - typeKey := EventTypeTermKey(xdr.ContractEventTypeContract) - contractKey := ComputeTermKey(contractA[:], FieldContractID) - topic1Key := ComputeTermKey(topicRaw, FieldTopic1) - everything := make([]uint32, 0, benchEvents) - contractIDs := make([]uint32, 0, benchEvents/3+1) - topic1IDs := make([]uint32, 0, benchEvents/7+1) - for id := range uint32(benchEvents) { - everything = append(everything, id) - if id%3 == 0 { - contractIDs = append(contractIDs, id) - } - if id%7 == 0 { - topic1IDs = append(topic1IDs, id) - } - } - bms.AddTo(typeKey, everything...) - bms.AddTo(contractKey, contractIDs...) - bms.AddTo(topic1Key, topic1IDs...) - mirror := NewConcurrentBitmapsFromBitmaps(bms) - - topic0Key := ComputeTermKey(topicRaw, FieldTopic0) - sparse := make([]uint32, 0, promotionThreshold-1) - for i := range uint32(promotionThreshold - 1) { - sparse = append(sparse, i*(benchEvents/promotionThreshold)) - } - mirror.AddTo(topic0Key, sparse...) - - eventType := xdr.ContractEventTypeContract - var topics [protocol.MaxTopicCount][]byte - topics[0] = topicRaw - return &benchIndex{ - reader: &hotLikeIndex{&stubIndex{ - mirror: mirror, count: benchEvents, raw: raw, - }}, - filters: []Filter{ - // Two dense groups AND-ed: the intersect arm. - {ContractID: contractA[:], EventType: &eventType}, - // One long-tail sparse group: the arm Get used to materialize a - // bitmap for on every request. - {Topics: topics}, - }, - // A sub-window, so both window edges are live. - window: IDRange{Start: benchEvents / 4, End: benchEvents * 3 / 4}, - } -}) - -// benchMatches drives one page-sized request and stops, the shape a getEvents -// page actually has. -func benchMatches(b *testing.B, descending bool) { - b.Helper() - fx := newBenchIndex() - ctx := context.Background() - b.ReportAllocs() - b.ResetTimer() - for b.Loop() { - n := 0 - for _, err := range Matches(ctx, fx.reader, fx.filters, fx.window, descending, benchPage) { - if err != nil { - b.Fatal(err) - } - n++ - if n == benchPage { - break - } - } - if n != benchPage { - b.Fatalf("fixture sanity: want %d matches, got %d", benchPage, n) - } - } -} - -func BenchmarkMatchesAscending(b *testing.B) { benchMatches(b, false) } -func BenchmarkMatchesDescending(b *testing.B) { benchMatches(b, true) } - -// ───────────── the synthetic shape matrix ───────────── - -// benchFat is one fat term's cardinality against benchEvents: ~7% of the -// domain, the density at which roaring holds a term as bitmap containers. -const benchFat = 300_000 - -// benchRand is a deterministic xorshift. The shapes must be identical from run -// to run, and a fixed stride would hand a walking AND a regularity real -// postings do not have. -type benchRand uint64 - -func (r *benchRand) next() uint64 { - x := uint64(*r) - x ^= x << 13 - x ^= x >> 7 - x ^= x << 17 - *r = benchRand(x) - return x -} - -// scatter draws k ascending ids from one residue class, one per stride at a -// jittered offset. Terms on disjoint classes interleave at single-id -// granularity while sharing nothing, so a shape's overlap is exactly the class -// its terms share. -func scatter(rng *benchRand, domain, m, res uint32, k int) []uint32 { - if k == 0 { - return nil - } - class := (domain - res + m - 1) / m - stride := class / uint32(k) - if stride == 0 { - panic("scatter: residue class too small for k") - } - ids := make([]uint32, k) - for t := range k { - ids[t] = (uint32(t)*stride+uint32(rng.next()%uint64(stride)))*m + res - } - return ids -} - -// fatGroup builds n terms of card ids each, drawing private ids from one -// residue class per term plus one class every term holds, so the joint -// intersection is exactly that shared class. -func fatGroup(rng *benchRand, domain, mod, base uint32, n, card, shared int) [][]uint32 { - common := scatter(rng, domain, mod, base+uint32(n), shared) - out := make([][]uint32, n) - for i := range out { - ids := scatter(rng, domain, mod, base+uint32(i), card-shared) - ids = append(ids, common...) - slices.Sort(ids) - out[i] = ids - } - return out -} - -// benchShape is one synthetic candidate-set problem: a term corpus in the -// mirror, the plan resolved over it, and the page the engine must produce. -type benchShape struct { - name string - reader *hotLikeIndex - plans []termPlan - keys []TermKey - window IDRange - // wantCount and wantSum fingerprint the first page. The benchmark checks - // them every iteration, so a harness that stopped answering the query - // cannot post a fast number. - wantCount int - wantSum uint64 -} - -// newBenchShape indexes terms as one term each, resolves the window to most of -// the domain with both edges live, and fingerprints the first page off the -// naive materialized algebra in referenceCandidates. -func newBenchShape(name string, domain uint32, terms [][]uint32, plans []termPlan) *benchShape { - bms := NewBitmaps() - keys := make([]TermKey, len(terms)) - for i, ids := range terms { - keys[i] = TermKey{0: byte(i + 1)} - bms.AddTo(keys[i], ids...) - } - s := &benchShape{ - name: name, - reader: &hotLikeIndex{&stubIndex{ - mirror: NewConcurrentBitmapsFromBitmaps(bms), count: domain, - }}, - plans: plans, - keys: keys, - window: IDRange{Start: domain / 32, End: domain - domain/32}, - } - sources, err := lookupPostings(context.Background(), s.reader, s.keys) - if err != nil { - panic(err) - } - for _, id := range referenceCandidates(s.plans, sources, s.window) { - if s.wantCount == benchPage { - break - } - s.wantCount++ - s.wantSum += uint64(id) - } - return s -} - -// singleFilterPlan is one filter AND-ing n one-term groups: the intersect -// shapes' plan. -func singleFilterPlan(n int) []termPlan { - plan := make(termPlan, n) - for i := range plan { - plan[i] = []int{i} - } - return []termPlan{plan} -} - -// benchShapes is the shape matrix, each entry built on first use so a -bench -// selecting one shape pays for one shape. domain is a parameter so the -// correctness twin of the matrix can run the same geometries small. -func benchShapes(domain uint32) []struct { - name string - build func() *benchShape -} { - scale := func(n int) int { return max(1, n*int(domain)/benchEvents) } - fat := scale(benchFat) - // ~3% of a fat term: the partial overlap that makes an AND converge slowly - // without making it empty. - partial := fat * 3 / 100 - // Just over one page once the window clips it: the intersection too small - // to fill a page early, so the walk spans the window. - tiny := scale(1200) - - shapes := []struct { - name string - build func() *benchShape - }{ - {"a_and2_fat_3pct", func() *benchShape { - rng := benchRand(1) - return newBenchShape("a", domain, - fatGroup(&rng, domain, 3, 0, 2, fat, partial), singleFilterPlan(2)) - }}, - {"b_and3_fat_3pct", func() *benchShape { - rng := benchRand(2) - return newBenchShape("b", domain, - fatGroup(&rng, domain, 4, 0, 3, fat, partial), singleFilterPlan(3)) - }}, - {"c_and2_skew", func() *benchShape { - rng := benchRand(3) - // The small term is a subset of the fat one, spread over it, so the - // AND is entirely decided by the rare side. - big := scatter(&rng, domain, 1, 0, fat) - small := make([]uint32, 0, scale(2000)) - step := len(big) / cap(small) - for i := range cap(small) { - small = append(small, big[i*step]) - } - return newBenchShape("c", domain, - [][]uint32{big, small}, singleFilterPlan(2)) - }}, - {"d_and6_fat_tiny", func() *benchShape { - rng := benchRand(4) - return newBenchShape("d", domain, - fatGroup(&rng, domain, 7, 0, 6, fat, tiny), singleFilterPlan(6)) - }}, - {"e_or10_single_term", func() *benchShape { - rng := benchRand(5) - terms := fatGroup(&rng, domain, 11, 0, 10, scale(30_000), 0) - plans := make([]termPlan, len(terms)) - for i := range plans { - plans[i] = termPlan{{i}} - } - return newBenchShape("e", domain, terms, plans) - }}, - {"f_and2_fat_tiny", func() *benchShape { - rng := benchRand(6) - return newBenchShape("f", domain, - fatGroup(&rng, domain, 3, 0, 2, fat, tiny), singleFilterPlan(2)) - }}, - {"h_and2_fat_overlapping", func() *benchShape { - rng := benchRand(8) - // The serving default: one selective term AND-ed with a near-total - // one, so a page comes out of the window's first fraction. - selective := scatter(&rng, domain, 3, 0, fat) - nearAll := make([]uint32, 0, domain) - for id := range domain { - if id%50 != 7 { - nearAll = append(nearAll, id) - } - } - return newBenchShape("h", domain, - [][]uint32{selective, nearAll}, singleFilterPlan(2)) - }}, - {"g_and3_x4_filters", func() *benchShape { - rng := benchRand(7) - // Several filters, each AND-ing a few fat terms. Each filter owns - // four residue classes, so the filters overlap only where the union - // has to dedup them. - terms := make([][]uint32, 0, 12) - plans := make([]termPlan, 0, 4) - for f := range uint32(4) { - group := fatGroup(&rng, domain, 16, f*4, 3, scale(75_000), scale(2250)) - plan := make(termPlan, len(group)) - for i := range group { - plan[i] = []int{len(terms) + i} - } - terms = append(terms, group...) - plans = append(plans, plan) - } - return newBenchShape("g", domain, terms, plans) - }}, - } - return shapes -} - -// benchShapeCache keeps one built corpus per shape name. Benchmarks run one at -// a time, so a plain map suffices. -var benchShapeCache = map[string]*benchShape{} - -func shapeFor(name string, build func() *benchShape) *benchShape { - s, ok := benchShapeCache[name] - if !ok { - s = build() - benchShapeCache[name] = s - } - return s -} - -// BenchmarkCandidateSlab pulls one page of candidates per shape, with the -// fetch and the post-filter out of frame. -func BenchmarkCandidateSlab(b *testing.B) { - for _, sh := range benchShapes(benchEvents) { - b.Run(sh.name, func(b *testing.B) { - benchSlabPage(b, shapeFor(sh.name, sh.build)) - }) - } -} - -func benchSlabPage(b *testing.B, s *benchShape) { - b.Helper() - ctx := context.Background() - b.ReportAllocs() - for b.Loop() { - sources, err := lookupPostings(ctx, s.reader, s.keys) - if err != nil { - b.Fatal(err) - } - st := newSlabStepper(s.plans, sources, s.window, false) - ids := make([]uint32, 0, benchPage) - n, sum := 0, uint64(0) - for n < benchPage { - ids = st.appendUpTo(ids[:0], benchPage-n) - if len(ids) == 0 { - break - } - for _, v := range ids { - n, sum = n+1, sum+uint64(v) - } - } - if n != s.wantCount || sum != s.wantSum { - b.Fatalf("page mismatch: got (%d, %d), want (%d, %d)", - n, sum, s.wantCount, s.wantSum) - } - } -} - -// referenceCandidates is an independent, naive materialized implementation of -// the algebra the slab stepper answers: OR each group whole, AND the groups, -// OR across filters, then clip to the window. It builds every intermediate at -// full chunk width, which is what the stepper exists not to do, so agreement -// between the two is a real check rather than a restatement. -func referenceCandidates(plans []termPlan, sources []postings, window IDRange) []uint32 { - materialize := func(p postings) *roaring.Bitmap { - if bm := p.bitmap(); bm != nil { - return bm - } - bm := roaring.New() - bm.AddMany(p.ids) - return bm - } - union := roaring.New() - for _, plan := range plans { - var acc *roaring.Bitmap - missed := false - for _, slots := range plan { - group := roaring.New() - present := false - for _, s := range slots { - if sources[s].present() { - present = true - group.Or(materialize(sources[s])) - } - } - if !present { - missed = true - break - } - if acc == nil { - acc = group - } else { - acc.And(group) - } - } - if missed { - continue - } - union.Or(acc) - } - windowBM := roaring.New() - windowBM.AddRange(uint64(window.Start), uint64(window.End)) - union.And(windowBM) - return union.ToArray() -} - -// TestBenchShapesAgree runs the whole shape matrix small: every geometry the -// microbench measures must be one the stepper answers exactly, in both -// directions, so a shape can never post a number for a query it gets wrong. -func TestBenchShapesAgree(t *testing.T) { - const domain = 1 << 16 - for _, sh := range benchShapes(domain) { - t.Run(sh.name, func(t *testing.T) { - s := sh.build() - sources, err := lookupPostings(context.Background(), s.reader, s.keys) - require.NoError(t, err) - want := referenceCandidates(s.plans, sources, s.window) - - got := drainStepper(newSlabStepper(s.plans, sources, s.window, false)) - require.Equal(t, want, got) - require.NotEmpty(t, got, "shape sanity: the plan must select something") - - // The descending arm reads the same set backwards. - gotDesc := drainStepper(newSlabStepper(s.plans, sources, s.window, true)) - slices.Reverse(gotDesc) - require.Equal(t, want, gotDesc) - }) - } -} - -// drainStepper pulls a stepper dry in 512-id fills, never returning nil so an -// empty result compares equal to a materialized bitmap's ToArray(). -func drainStepper(st *slabStepper) []uint32 { - out := []uint32{} - for { - before := len(out) - out = st.appendUpTo(out, before+512) - if len(out) == before { - return out - } - } -} diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go index 9a385a3a4..3a4ffab45 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go @@ -2,8 +2,8 @@ package event // slab_match_test.go covers the slab engine over a corpus built to hold each // named query shape by construction: dense-only, sparse-only, mixed, absent, -// match-all, and the fat/fat thin-overlap shape whose two chunk-sized terms -// meet on a handful of ids. +// match-all, the fat/fat thin-overlap shape whose two chunk-sized terms meet +// on a handful of ids, and the high-arity AND and union. // // The answer every case is checked against is computed without the index — // postFilter run over every ordinal in the corpus, then clipped to the window, @@ -17,7 +17,6 @@ import ( "iter" "math/rand" "slices" - "strings" "testing" "github.com/stretchr/testify/assert" @@ -130,7 +129,6 @@ type slabShape struct { type shapedFixture struct { corpus *diffCorpus vocab *diffVocab - n uint32 // thin holds the ids where the two fat terms of the overlap shape meet. thin []uint32 // rareContract and rareTopic hold the ids carrying the sparse terms. @@ -173,12 +171,16 @@ func (f *shapedFixture) shapeFor(id uint32) slabShape { return sh } -// newShapedFixture builds the shaped corpus over n ids. n is chosen by the -// caller to cross at least one 65536-id slab boundary. -func newShapedFixture(tb testing.TB, n uint32) *shapedFixture { +// shapedCorpusSize spans slab 0 whole and part of slab 1, so every window +// bound below is a real slab-relative position rather than a synthetic one. +const shapedCorpusSize uint32 = 70_000 + +// newShapedFixture builds the shaped corpus. +func newShapedFixture(tb testing.TB) *shapedFixture { tb.Helper() - v := newDiffVocabTB(tb) - f := &shapedFixture{vocab: v, n: n} + const n = shapedCorpusSize + v := newDiffVocab(tb) + f := &shapedFixture{vocab: v} // Thin-overlap ids: even, spread across the window, and deliberately // sitting on both sides of a slab boundary. for _, id := range []uint32{4, 30_000, 65_534, 65_536, 65_538, n - 2} { @@ -251,29 +253,6 @@ func (f *shapedFixture) marshalShape(tb testing.TB, sh slabShape) []byte { return raw } -// newDiffVocabTB is newDiffVocab widened to testing.TB so benchmarks can build -// the same vocabulary. -func newDiffVocabTB(tb testing.TB) *diffVocab { - tb.Helper() - v := &diffVocab{types: []xdr.ContractEventType{ - xdr.ContractEventTypeSystem, - xdr.ContractEventTypeContract, - xdr.ContractEventTypeDiagnostic, - }} - for i := range 4 { - cid := xdr.ContractId{0: byte(0xC0 + i)} - v.contracts = append(v.contracts, cid[:]) - } - for name := range strings.FieldsSeq("alpha beta gamma delta epsilon") { - sym := xdr.ScSymbol(name) - val := xdr.ScVal{Type: xdr.ScValTypeScvSymbol, Sym: &sym} - raw, err := val.MarshalBinary() - require.NoError(tb, err) - v.topics, v.topicRaw = append(v.topics, val), append(v.topicRaw, raw) - } - return v -} - // named query shapes over the shaped corpus. func (f *shapedFixture) filterDenseOnly() []Filter { return []Filter{{ContractID: f.vocab.contracts[0]}} @@ -360,9 +339,43 @@ func (f *shapedFixture) filterUnion() []Filter { } } -// shapedCorpusSize spans slab 0 whole and part of slab 1, so every window -// bound below is a real slab-relative position rather than a synthetic one. -const shapedCorpusSize = 70_000 +// filterDeepAND names every constrainable field at once, so the AND runs five +// groups deep: a near-total event-type term, two chunk-sized terms and two +// sparse ones, meeting on the handful of ids carrying a second topic. Exact +// keeps the count group in the plan, which an "at least" bound the constrained +// positions already imply would not. +func (f *shapedFixture) filterDeepAND() []Filter { + evType := xdr.ContractEventTypeContract + return []Filter{{ + ContractID: f.vocab.contracts[0], + EventType: &evType, + Topics: [protocol.MaxTopicCount][]byte{ + 0: f.vocab.topicRaw[0], + 1: f.vocab.topicRaw[2], + }, + TopicCount: TopicCountFilter{Count: 2, Exact: true}, + }} +} + +// filterWideUnion drives eight filters into one FastOr, the widest of the +// named shapes, mixing dense, sparse, absent and multi-group filters so the +// union also has to drop a filter and dedup ids several of them select. +func (f *shapedFixture) filterWideUnion() []Filter { + sysType := xdr.ContractEventTypeSystem + return []Filter{ + { + ContractID: f.vocab.contracts[0], + Topics: [protocol.MaxTopicCount][]byte{0: f.vocab.topicRaw[1]}, + }, + {ContractID: f.vocab.contracts[1]}, + {ContractID: f.vocab.contracts[2]}, + {ContractID: f.vocab.contracts[3]}, + {EventType: &sysType}, + {Topics: [protocol.MaxTopicCount][]byte{0: f.vocab.topicRaw[0]}}, + {Topics: [protocol.MaxTopicCount][]byte{1: f.vocab.topicRaw[2]}}, + {TopicCount: TopicCountFilter{Count: 2, Exact: true}}, + } +} // namedShape is one query shape with the name its failures report under. type namedShape struct { @@ -385,6 +398,8 @@ func (f *shapedFixture) namedShapes() []namedShape { {"partly absent group", f.filterPartlyAbsentGroup()}, {"wholly absent group", f.filterWhollyAbsentGroup()}, {"union of filters", f.filterUnion()}, + {"deep and", f.filterDeepAND()}, + {"wide union", f.filterWideUnion()}, {"match all empty slice", nil}, {"match all wildcard filter", []Filter{{}}}, {"match all beside constrained", []Filter{{EventType: &sysType}, {}}}, @@ -395,7 +410,7 @@ func (f *shapedFixture) namedShapes() []namedShape { // ───────────────────────── the shaped matrix ───────────────────────── func TestMatches_ShapedMatrix(t *testing.T) { - f := newShapedFixture(t, shapedCorpusSize) + f := newShapedFixture(t) readers := []struct { name string r Reader @@ -457,7 +472,7 @@ func TestMatches_ShapedMatrix(t *testing.T) { // different thing — the corpus end, a slab boundary, and a window living // entirely inside the second slab. func TestMatches_WholeStreams(t *testing.T) { - f := newShapedFixture(t, shapedCorpusSize) + f := newShapedFixture(t) r := diffPostingsReader{diffReader{f.corpus}} const slab = 1 << 16 @@ -482,7 +497,7 @@ func TestMatches_WholeStreams(t *testing.T) { // overlap. Drift in the corpus rules would otherwise turn the matrix above // into a weaker test without failing it. func TestMatches_ShapedFixtureIsWhatItClaims(t *testing.T) { - f := newShapedFixture(t, shapedCorpusSize) + f := newShapedFixture(t) r := diffPostingsReader{diffReader{f.corpus}} ctx := context.Background() window := IDRange{0, shapedCorpusSize} @@ -514,6 +529,14 @@ func TestMatches_ShapedFixtureIsWhatItClaims(t *testing.T) { Matches(ctx, r, f.filterPartlyAbsentGroup(), window, false, 0), 0), len(f.rareTopic), "the partly-absent group must select exactly its one present bucket") + // The high-arity shapes must reach the corpus. A five-group AND that + // intersected to nothing, or a union that selected a corner of it, would + // pass the matrix vacuously. + require.Greater(t, card(f.filterDeepAND()), 10, + "the five-group AND must still select something") + require.Greater(t, card(f.filterWideUnion()), 30_000, + "the wide union must span the corpus") + // The overlap shape is only the overlap shape if both sides are // chunk-sized and their meeting point is rare. fat, err := r.lookupPostings(ctx, []TermKey{ @@ -550,9 +573,11 @@ func TestMatches_RandomizedAgainstPostFilter(t *testing.T) { {"lookupKeys", diffReader{corpus}}, {"postings", diffPostingsReader{diffReader{corpus}}}, } - // slabShift 2 and 4 put 75 and 19 slab seams inside the corpus; 16 is the - // production width, where the whole corpus is one slab. - for _, shift := range []uint{2, 4, 16} { + // The slab width is a seam, not a behavior: every width must reproduce the + // same stream. 1, 2 and 4 put 150, 75 and 19 slab seams inside the corpus, + // 8 leaves a single seam, and 16 is the production width, where the whole + // corpus is one slab. + for _, shift := range []uint{1, 2, 4, 8, 16} { slabShift = shift for _, seam := range readers { r := seam.r @@ -612,43 +637,6 @@ func requireStrictOrder(t *testing.T, got []Match, desc bool, trial int) { } } -// The slabShift seam must be invisible in the output: every width reproduces -// the production width's stream exactly. -func TestMatches_SlabWidthIsInvisible(t *testing.T) { - f := newShapedFixture(t, shapedCorpusSize) - r := diffPostingsReader{diffReader{f.corpus}} - ctx := context.Background() - - defer func(s uint) { slabShift = s }(slabShift) - cases := [][]Filter{ - f.filterDenseOnly(), - f.filterSparseOnly(), - f.filterMixedOrGroup(), - f.filterThinOverlap(), - f.filterUnion(), - } - windows := []IDRange{ - {0, shapedCorpusSize}, - {65_000, 67_000}, - {65_536, shapedCorpusSize}, - } - for ci, filters := range cases { - for _, w := range windows { - for _, desc := range []bool{false, true} { - slabShift = 16 - want := drainMatches(t, Matches(ctx, r, filters, w, desc, 0), 250) - for _, shift := range []uint{3, 8, 13, 17, 20, 31} { - slabShift = shift - got := drainMatches(t, Matches(ctx, r, filters, w, desc, 0), 250) - require.Equal(t, want, got, - "case %d window %v desc=%v: slabShift %d changed the stream", - ci, w, desc, shift) - } - } - } - } -} - // ───────────────────────── the candidate-set pin ───────────────────────── // fetchTracer records the ordinals of every FetchEvents call, one entry per @@ -683,7 +671,7 @@ var ( // The match-all shapes are excluded because they never reach the index: they // stream FetchRange instead. func TestMatches_FetchesOnlyTrueCandidates(t *testing.T) { - f := newShapedFixture(t, shapedCorpusSize) + f := newShapedFixture(t) const slab = 1 << 16 shapes := [][]Filter{ From 96aa40dfe0d00f85c049886d2018a2e72fa5276e Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 16:07:32 +0000 Subject: [PATCH 25/41] =?UTF-8?q?events:=20LookupKeys=20is=20the=20only=20?= =?UTF-8?q?index=20path=20=E2=80=94=20drop=20the=20postings=20seam?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The postings seam existed so the cursor tree could walk a sparse term's id list without materializing a bitmap for it. That tree is gone. Under slab evaluation a sparse term is at most 63 ids materialized once per query, and the engine already snapshots every dense term up front, so the seam bought nothing and cost a second lookup path, a second representation inside every term group, and a per-slab binary search over borrowed id slices. Deleted: the postings struct and its present/bitmap/estimate accessors, the postingReader interface and both lookupPostings implementations, ConcurrentBitmaps.lookupPostings, denseState.cardinality, slabTerms.lists with the sparse arm of the group union, and slabScratch whole — with the sparse ids gone, a group's AndAny arguments are just the bitmaps it holds, so there is nothing left to keep between slabs. resolveSlabTerms now holds what Reader.LookupKeys returned, and group ordering reads GetCardinality off those held bitmaps rather than a lock-free count off the writer's. Presence is a non-nil bitmap; a present-but-empty one still keeps its group alive. Freshness is unchanged and now stated where the bitmaps are held: the lookup is a point-in-time image of every term the query names. A sparse hot term is copied out of the mirror's atomically published id list, a dense one is denseState's published snapshot, and neither grows under its holder — so a walk sees no id ingested after it started, which is what IDRange's pinned window already promises. postings_freshness_test.go guarded the deleted seam and goes with it, as do TestPostingsPresent, TestPostingsEstimate and the sparseSource / denseSource / postingIDs helpers. The suites that drove the corpus through both seams now drive the one that remains, and the planning tests take LookupKeys' []*roaring.Bitmap. TestMatches_ConcurrentIngestBorrowSafety stays: its subject is the match path under concurrent AddTo, which survives. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../rpcv2/stores/event/concurrent_bitmaps.go | 91 +---- .../stores/event/concurrent_bitmaps_test.go | 43 --- .../internal/rpcv2/stores/event/hot_store.go | 41 +-- .../internal/rpcv2/stores/event/match.go | 46 +-- .../internal/rpcv2/stores/event/match_test.go | 3 +- .../stores/event/matches_differential_test.go | 37 +- .../stores/event/postings_freshness_test.go | 328 ------------------ .../internal/rpcv2/stores/event/slab_match.go | 141 +++----- .../rpcv2/stores/event/slab_match_test.go | 117 +++---- 9 files changed, 130 insertions(+), 717 deletions(-) delete mode 100644 cmd/stellar-rpc/internal/rpcv2/stores/event/postings_freshness_test.go diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go index da4ff03f3..e8795bf97 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go @@ -102,6 +102,12 @@ func NewConcurrentBitmapsFromBitmaps(b Bitmaps) *ConcurrentBitmaps { // // A Get that starts after an AddTo returns sees that AddTo's IDs, and // the pointer stays valid for as long as the caller holds it. +// +// The result is a point-in-time image either way, which is what lets a +// query hold it for a whole walk: a sparse term is copied out of the +// atomically published id list, and a dense term's snapshot is never +// mutated once published. Neither grows under a holder as ingest +// continues. func (s *ConcurrentBitmaps) Get(key TermKey) (*roaring.Bitmap, error) { s.rwmu.RLock() p := s.terms[key] @@ -135,73 +141,6 @@ func (d *denseState) snapshot() *roaring.Bitmap { return bm } -// cardinality is the term's id count without materializing a -// snapshot for it. A live pub is the term exactly, so it answers -// lock-free; otherwise the count comes off wbm under mu, a walk of -// the writer's containers rather than a clone of them. -func (d *denseState) cardinality() uint64 { - if bm := d.pub.Load(); bm != nil { - return bm.GetCardinality() - } - d.mu.Lock() - defer d.mu.Unlock() - return d.wbm.GetCardinality() -} - -// postings is an iterable view of one term's event IDs in whichever -// representation the index holds: the sparse mode's sorted []uint32, a dense -// term's live denseState, or a bitmap from outside the mirror (the cold tier's, -// and the ascending path's own bulk answers). Exactly one is set; the zero value -// means the term is absent. It exists so the ascending match path can iterate a -// sparse term in place, where Get has to promise a bitmap and so pays a -// roaring.New plus AddMany on every sparse lookup. -// -// Ownership matches Get's: ids borrows the atomically published termState, which -// no writer ever mutates, and every bitmap a dense term yields comes from -// denseState.snapshot — never wbm, never the raw pub pointer — so Get's -// read-only contract applies verbatim, and the id slice must likewise be read, -// never written or appended to. A dense postings is a view, not a frozen copy: -// a later materialization can hold ids a write added since. Callers pin a window -// before the lookup and intersect every read with it (the slab accumulator's -// range), so those ids sit above the window and are never yielded. -type postings struct { - ids []uint32 - bm *roaring.Bitmap - dense *denseState -} - -// present reports whether the term is in the index at all. A term present but -// holding no ids still counts as present: it yields an exhausted cursor, which -// intersects and unions to what an absent term's caller-side skip produces. -func (p postings) present() bool { return p.bm != nil || p.ids != nil || p.dense != nil } - -// bitmap is the term's ids as a roaring bitmap, or nil when the term is sparse -// or absent — sparse callers walk ids instead. A dense term is snapshotted here, -// the only place outside Get that materializes one, and repeat calls cost -// nothing while no write lands: denseState caches the snapshot it published. -func (p postings) bitmap() *roaring.Bitmap { - if p.dense != nil { - return p.dense.snapshot() - } - return p.bm -} - -// estimate is the term's cardinality over the whole chunk, the weight the -// ascending path orders an intersection by. It ignores the caller's window, so -// it ranks terms rather than counting a query's candidates. A dense term is -// counted off denseState.cardinality, not a snapshot: planning wants a number, -// and snapshot would clone a term written since its last read for a bitmap the -// plan may never walk. -func (p postings) estimate() uint64 { - if p.dense != nil { - return p.dense.cardinality() - } - if p.bm != nil { - return p.bm.GetCardinality() - } - return uint64(len(p.ids)) -} - // AddTo records each eventID under key. Callers feed events in // event-ID order relative to the chunk, so a duplicate is a retry of an // already-added prefix and is skipped. @@ -239,24 +178,6 @@ func (s *ConcurrentBitmaps) AddTo(key TermKey, eventIDs ...uint32) { p.Store(termStateFromIDs(appendSorted(ids, eventIDs))) } -// lookupPostings is Get without the sparse-mode materialization: it hands back -// the term's live representation rather than converting it to a bitmap. Same -// concurrency story as Get; a dense term comes back as its denseState, so the -// bitmap a caller reads is snapshot's. A miss returns the zero postings. -func (s *ConcurrentBitmaps) lookupPostings(key TermKey) postings { - s.rwmu.RLock() - p := s.terms[key] - s.rwmu.RUnlock() - if p == nil { - return postings{} - } - st := p.Load() - if st.dense != nil { - return postings{dense: st.dense} - } - return postings{ids: st.ids} -} - // appendSorted appends the ids in src that are greater than dst's // last element. func appendSorted(dst, src []uint32) []uint32 { diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps_test.go index 85a7bfc1b..2234c5d33 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps_test.go @@ -4,53 +4,10 @@ import ( "sync" "testing" - "github.com/RoaringBitmap/roaring/v2" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) -// sparseSource and denseSource build the two representations the index holds -// for a term, for tests that need postings without a store behind them. -func sparseSource(ids ...uint32) postings { return postings{ids: ids} } - -func denseSource(ids ...uint32) postings { - bm := roaring.New() - bm.AddMany(ids) - return postings{bm: bm} -} - -// postingIDs reads a term's ids out of whichever representation it holds, -// through the accessor the match path uses. It never returns nil, so an empty -// term compares equal to a materialized bitmap's ToArray(). -func postingIDs(p postings) []uint32 { - if bm := p.bitmap(); bm != nil { - return bm.ToArray() - } - return append([]uint32{}, p.ids...) -} - -// A term present but holding no ids is still present: it contributes nothing -// to a union and does not drop its group, which is what distinguishes it from -// a term absent from the index. -func TestPostingsPresent(t *testing.T) { - assert.False(t, postings{}.present(), "the zero postings is the absent term") - assert.True(t, sparseSource(1).present()) - assert.True(t, denseSource(1).present()) - assert.True(t, postings{bm: roaring.New()}.present(), - "a present-but-empty bitmap is present; it just holds nothing") - assert.Empty(t, postingIDs(postings{bm: roaring.New()})) -} - -// The ordering weight is the term's whole-chunk cardinality, window and all. -func TestPostingsEstimate(t *testing.T) { - assert.Equal(t, uint64(0), postings{}.estimate(), "the absent term weighs nothing") - assert.Equal(t, uint64(0), postings{bm: roaring.New()}.estimate()) - assert.Equal(t, uint64(3), sparseSource(1, 2, 3).estimate()) - assert.Equal(t, uint64(3), denseSource(1, 2, 3).estimate()) - assert.Equal(t, uint64(4), denseSource(1, 2, 3, 1<<20).estimate(), - "cardinality spans containers") -} - // newTestConcurrentBitmaps builds an empty ConcurrentBitmaps via the // only remaining constructor (production always converts from a // warmup/backfill-built Bitmaps). diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go index fc8d67768..eae2a9c4b 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go @@ -112,12 +112,8 @@ type HotStore struct { offsets *ConcurrentLedgerOffsets } -// Compile-time guards: *HotStore satisfies Reader and the optional -// postingReader seam. -var ( - _ Reader = (*HotStore)(nil) - _ postingReader = (*HotStore)(nil) -) +// Compile-time guard: *HotStore satisfies Reader. +var _ Reader = (*HotStore)(nil) // NewWithStore wraps an ALREADY-OPEN rocksdb.Store as an events HotStore on the // three events CFs (CFNames()), running the mandatory warmup to rebuild the @@ -188,6 +184,13 @@ func (h *HotStore) Offsets() (*LedgerOffsets, error) { // to batch — but exposing this method satisfies the Reader // interface so callers can program against batched lookups // uniformly. +// +// Freshness, which the match path holds these bitmaps across a whole +// query on: each is a point-in-time image of its term. A sparse term +// is copied out of the mirror's atomically published id list; a dense +// one is denseState.snapshot, the immutable clone shared with every +// other reader of that term. Neither grows under its holder as ingest +// continues, so a walk started now never sees an id written later. func (h *HotStore) LookupKeys(ctx context.Context, keys []TermKey) ([]*roaring.Bitmap, error) { if h.chunkStore.IsClosed() { return nil, stores.ErrStoreClosed @@ -455,32 +458,6 @@ func (h *HotStore) IngestLedgerToBatch( return func() { h.applyLedger(startID, termKeys) }, nil } -// lookupPostings is the no-materialize half of LookupKeys, and the hot store's -// implementation of the optional postingReader seam. It returns each term's -// live mirror representation, so a query that only walks ids in ascending -// order never pays Get's roaring.New plus AddMany per sparse term. -// -// Results are positionally aligned with keys; a miss is the zero postings. -// Same read-only contract as LookupKeys: a sparse term borrows the mirror's -// published id slice, and a dense one materializes through -// denseState.snapshot, the shared immutable bitmap LookupKeys hands out. -func (h *HotStore) lookupPostings(ctx context.Context, keys []TermKey) ([]postings, error) { - if h.chunkStore.IsClosed() { - return nil, stores.ErrStoreClosed - } - if err := ctx.Err(); err != nil { - return nil, err - } - if len(keys) == 0 { - return nil, nil - } - results := make([]postings, len(keys)) - for i, key := range keys { - results[i] = h.mirror.lookupPostings(key) - } - return results, nil -} - // index returns the in-memory term mirror. Test-only write hook: no production // path reads it. Kept unexported until #772 decides whether the v2 read path // hooks into it. diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go index a682653b9..8f30fa0bd 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go @@ -7,9 +7,10 @@ package event // Matches. // // Optimization shape: terms are deduped across filters and issued as -// a single batched index lookup at iteration start; payload fetches -// then stream in internal batches. On the cold path this is one -// MPHF+index.pack round trip per Matches call, not per batch. +// a single batched Reader.LookupKeys at iteration start, whose bitmaps +// the walk then holds for the whole query; payload fetches stream in +// internal batches. On the cold path the lookup is one MPHF+index.pack +// round trip per Matches call, not per batch. // // The candidate set comes from the slab-stepped engine in slab_match.go, which // answers both directions from one walk over the window. @@ -275,9 +276,9 @@ func Matches( streamRange(ctx, r, window, descending, firstBatch, yield) return } - sources, err := lookupPostings(ctx, r, uniqueKeys) + sources, err := r.LookupKeys(ctx, uniqueKeys) if err != nil { - yield(Match{}, err) + yield(Match{}, fmt.Errorf("events: query lookup: %w", err)) return } st := newSlabStepper(plans, sources, window, descending) @@ -350,41 +351,6 @@ func planIndexTerms(filters []Filter) ([]termPlan, []TermKey, bool) { return plans, uniqueKeys, false } -// postingReader is the optional, hot-tier half of the index read surface: a -// Reader that can expose a term's postings without materializing a bitmap. -// -// Deliberately not folded into Reader. Cold postings genuinely are bitmaps, -// unmarshaled per term out of index.pack, so ColdReader has nothing -// un-materialized to hand back, and hoisting the method would impose it on -// every out-of-package implementation for no gain. -type postingReader interface { - lookupPostings(ctx context.Context, keys []TermKey) ([]postings, error) -} - -// lookupPostings resolves keys to per-term postings, positionally aligned with -// keys, in one batched call. It takes the no-materialize path when r offers -// one; a nil bitmap stays the zero postings, meaning absent. -func lookupPostings(ctx context.Context, r Reader, keys []TermKey) ([]postings, error) { - if pr, ok := r.(postingReader); ok { - sources, err := pr.lookupPostings(ctx, keys) - if err != nil { - return nil, fmt.Errorf("events: query lookup: %w", err) - } - return sources, nil - } - bitmaps, err := r.LookupKeys(ctx, keys) - if err != nil { - return nil, fmt.Errorf("events: query lookup: %w", err) - } - sources := make([]postings, len(bitmaps)) - for i, bm := range bitmaps { - if bm != nil { - sources[i] = postings{bm: bm} - } - } - return sources, nil -} - // emitBatch fetches one batch of candidate ordinals, drops the bitmap-side // false positives and yields the survivors, reporting whether the stream // should continue. FetchEvents requires ascending ids, so a descending batch diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go index fa14a0493..dc0888de8 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go @@ -1142,8 +1142,7 @@ func TestQuery_InvalidFilterRejected(t *testing.T) { // // - match-all asc → streamRange + cold FetchRange // - match-all desc + cap → streamRange top-down, slices.Backward -// - single-filter (contractID) → lookupPostings, which a ColdReader -// serves off LookupKeys, then the +// - single-filter (contractID) → one LookupKeys term, then the // ascending slab walk // - multi-term filter (AND) → AndAny per group over cold bitmaps // - cross-filter (OR) → FastOr across filters diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go index 0e83f7086..a55d9cb84 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go @@ -3,10 +3,9 @@ package event // The in-memory chunk the match-path tests run against, and the borrow-safety // gate over it. // -// The corpus is served through both of the index's read seams: LookupKeys, -// which materializes every term as a bitmap, and lookupPostings, which hands -// sparse terms back as borrowed id lists. slab_match_test.go drives both -// against an answer computed without the index. +// The corpus is served through the index's one read seam, LookupKeys, which +// materializes every term as a bitmap. slab_match_test.go drives it against an +// answer computed without the index. import ( "context" @@ -30,12 +29,10 @@ type diffCorpus struct { mirror *ConcurrentBitmaps } -// diffReader serves the corpus through LookupKeys only, the materializing seam. +// diffReader serves the corpus through LookupKeys, the seam Matches reads the +// index by. type diffReader struct{ c *diffCorpus } -// diffPostingsReader adds the no-materialize seam HotStore carries. -type diffPostingsReader struct{ diffReader } - func (r diffReader) ChunkID() chunk.ID { return chunk.ID(0) } func (r diffReader) EventCount() (uint32, error) { return uint32(len(r.c.raw)), nil } @@ -55,14 +52,6 @@ func (r diffReader) LookupKeys(_ context.Context, keys []TermKey) ([]*roaring.Bi return out, nil } -func (r diffPostingsReader) lookupPostings(_ context.Context, keys []TermKey) ([]postings, error) { - out := make([]postings, len(keys)) - for i, k := range keys { - out[i] = r.c.mirror.lookupPostings(k) - } - return out, nil -} - func (r diffReader) FetchEvents(_ context.Context, ids []uint32) ([]Payload, error) { // A dedup bug in the union surfaces here, not as a doubled result. if err := validateSortedEventIDs(ids); err != nil { @@ -95,11 +84,7 @@ func (r diffReader) All(ctx context.Context) iter.Seq2[Payload, error] { return r.FetchRange(ctx, 0, total) } -var ( - _ Reader = diffReader{} - _ Reader = diffPostingsReader{} - _ postingReader = diffPostingsReader{} -) +var _ Reader = diffReader{} // diffVocab is the closed vocabulary the corpus and the random filters share. type diffVocab struct { @@ -210,10 +195,10 @@ func collectOrdinals(t *testing.T, r Reader, filters []Filter, w IDRange, desc b return out } -// Turns the borrow contract into a race-detector gate: the match path reads -// mirror snapshots in place while AddTo publishes new termStates on the same -// keys, including the sparse-to-dense promotion. Under -race any write -// reaching a borrowed snapshot fails the run; without it, the identity check +// Turns the borrow contract into a race-detector gate: the match path holds +// mirror snapshots across a whole walk while AddTo publishes new termStates on +// the same keys, including the sparse-to-dense promotion. Under -race any +// write reaching a held snapshot fails the run; without it, the identity check // still pins that a pinned window is immune to ingest past its End. func TestMatches_ConcurrentIngestBorrowSafety(t *testing.T) { rng := rand.New(rand.NewSource(20260830)) @@ -257,7 +242,7 @@ func TestMatches_ConcurrentIngestBorrowSafety(t *testing.T) { } } - r := diffPostingsReader{diffReader{corpus}} + r := diffReader{corpus} et := xdr.ContractEventTypeContract filters := []Filter{ {ContractID: v.contracts[0]}, diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/postings_freshness_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/postings_freshness_test.go deleted file mode 100644 index 6f78b5c0b..000000000 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/postings_freshness_test.go +++ /dev/null @@ -1,328 +0,0 @@ -package event - -import ( - "runtime" - "sync" - "sync/atomic" - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/chunk" -) - -// This file pins ONE property of the no-materialize read seam -// (lookupPostings, postings.bitmap, postings.estimate): a read that -// starts after an AddTo returns observes that AddTo's ids. -// -// It needs its own tests because the match-path tests cannot see the -// bug it guards. They build their sources up front and then only read, -// so a postings accessor that returned a stale dense snapshot — the raw -// pub pointer, nil-or-behind after every write — would agree with them -// on every case. Only write-then-read-through-the-accessor separates -// the two. - -// postingsIDs drains a term's postings through the accessor the query -// engine uses, so a test asserting on it exercises the same path a -// real read takes rather than the representation underneath. -func postingsIDs(t *testing.T, s *ConcurrentBitmaps, key TermKey) []uint32 { - t.Helper() - p := s.lookupPostings(key) - require.True(t, p.present(), "the term must be present in the index") - return postingIDs(p) -} - -// denseOf returns the term's denseState, failing the test when the -// term has not promoted. Tests use it only to observe pub, the field -// whose nil-ness is what makes a read take the un-snapshotted path. -func denseOf(t *testing.T, s *ConcurrentBitmaps, key TermKey) *denseState { - t.Helper() - st := s.terms[key].Load() - require.NotNil(t, st.dense, "term must be dense") - return st.dense -} - -// ascending is the id list a term holds after n writes of stride ids. -func ascending(n, stride uint32) []uint32 { - out := make([]uint32, n) - for i := range out { - out[i] = uint32(i) * stride - } - return out -} - -// TestPostings_SparseLookupSeesTheWriteJustMade: on a sparse term, -// every accessor reads the termState the last AddTo published — the -// ids, the count, and the cursor all include the write that just -// returned. -func TestPostings_SparseLookupSeesTheWriteJustMade(t *testing.T) { - s := newTestConcurrentBitmaps() - key := ComputeTermKey([]byte("sparse-freshness"), FieldTopic0) - - want := make([]uint32, 0, promotionThreshold-1) - for i := range uint32(promotionThreshold - 1) { - id := i * 3 - s.AddTo(key, id) - want = append(want, id) - - p := s.lookupPostings(key) - require.True(t, p.present(), "a term written to is present") - require.Nil(t, p.bitmap(), "a sub-threshold term stays sparse") - assert.Equal(t, want, postingIDs(p), - "the cursor must yield every id added so far, the last one included") - assert.Equal(t, uint64(len(want)), p.estimate(), - "estimate must count the write that just returned") - } -} - -// TestPostings_PromotionIsVisibleThroughLookupPostings: the AddTo that -// crosses promotionThreshold rebuilds the term as a bitmap, and a -// lookup right after it sees the whole term — the promoting batch -// included — through the dense representation. -func TestPostings_PromotionIsVisibleThroughLookupPostings(t *testing.T) { - s := newTestConcurrentBitmaps() - key := ComputeTermKey([]byte("promotion-freshness"), FieldTopic0) - - const stride = 5 - // One short of the threshold: still a list, no bitmap. - below := ascending(promotionThreshold-1, stride) - s.AddTo(key, below...) - require.Nil(t, s.terms[key].Load().dense, "one below the threshold stays sparse") - require.Nil(t, s.lookupPostings(key).bitmap(), "a sparse term has no bitmap") - require.Equal(t, uint64(len(below)), s.lookupPostings(key).estimate()) - - // The id that promotes. Read the accessors before anything else - // touches the term, so a promotion the lookup missed shows up as a - // short drain rather than being papered over by a later snapshot. - promoting := uint32(promotionThreshold-1) * stride - s.AddTo(key, promoting) - want := append(append([]uint32{}, below...), promoting) - - p := s.lookupPostings(key) - require.True(t, p.present()) - require.NotNil(t, p.bitmap(), "crossing the threshold promotes to a bitmap") - assert.True(t, p.bitmap().Contains(promoting), - "the promoting id must be in the bitmap the promotion built") - assert.Equal(t, want, postingIDs(p), - "a lookup straight after promotion yields every id the term holds") - assert.Equal(t, uint64(len(want)), p.estimate(), - "estimate must count the promoting write") - - // And the writes after promotion, which take the in-place dense path. - next := promoting + stride - s.AddTo(key, next) - want = append(want, next) - assert.Equal(t, want, postingsIDs(t, s, key), - "the first dense-mode write must be visible to the next lookup") - assert.Equal(t, uint64(len(want)), s.lookupPostings(key).estimate()) -} - -// TestPostings_DenseLookupSeesWritesSinceLastSnapshot: the case a -// stale read would pass every differential on. A reader publishes a -// snapshot, the writer invalidates it, and the next lookup must clone -// again rather than hand back the pub pointer it finds nil. -func TestPostings_DenseLookupSeesWritesSinceLastSnapshot(t *testing.T) { - s := newTestConcurrentBitmaps() - key := ComputeTermKey([]byte("dense-freshness"), FieldTopic0) - - seed := ascending(promotionThreshold, 4) - s.AddTo(key, seed...) - d := denseOf(t, s, key) - - // A first read publishes the snapshot every later read would - // wrongly reuse. - held := s.lookupPostings(key).bitmap() - require.NotNil(t, held) - require.Same(t, held, d.pub.Load(), "the first read publishes what it returns") - heldCard := held.GetCardinality() - - // Writes spread over fresh containers, so a stale read is wrong by - // more than a bit inside a container the snapshot already shares. - fresh := []uint32{1_000_000, 1_000_000 + 65_536, 1_000_000 + 3*65_536} - s.AddTo(key, fresh...) - require.Nil(t, d.pub.Load(), - "AddTo drops the snapshot, so the next read takes the un-snapshotted path") - - p := s.lookupPostings(key) - bm := p.bitmap() - require.NotNil(t, bm) - for _, id := range fresh { - assert.True(t, bm.Contains(id), - "a lookup after AddTo must observe id %d", id) - assert.False(t, held.Contains(id), - "the snapshot taken before the write stays frozen") - } - assert.Equal(t, heldCard+uint64(len(fresh)), bm.GetCardinality()) - assert.Equal(t, heldCard+uint64(len(fresh)), p.estimate(), - "estimate must count the writes made since the last snapshot") - assert.Subset(t, postingIDs(p), fresh, - "the cursor the query engine walks must yield the fresh ids too") -} - -// TestPostings_EstimateCountsUnsnapshottedWritesWithoutCloning pins -// both halves of the planner's cost rule at once: weighing a dense -// term written since its last read reports the CURRENT count, and -// does it without publishing a snapshot — the clone only a read that -// actually walks the term should pay. -func TestPostings_EstimateCountsUnsnapshottedWritesWithoutCloning(t *testing.T) { - s := newTestConcurrentBitmaps() - key := ComputeTermKey([]byte("estimate-freshness"), FieldTopic0) - - seed := ascending(promotionThreshold, 4) - s.AddTo(key, seed...) - d := denseOf(t, s, key) - - // Never read: pub is nil from promotion onwards. - require.Nil(t, d.pub.Load()) - assert.Equal(t, uint64(len(seed)), s.lookupPostings(key).estimate(), - "a term nobody has read yet weighs what it holds") - assert.Nil(t, d.pub.Load(), "estimate must not publish a snapshot") - - // Read once to publish, then invalidate and weigh again. - require.NotNil(t, s.lookupPostings(key).bitmap()) - require.NotNil(t, d.pub.Load()) - - s.AddTo(key, 2_000_000, 2_000_000+65_536) - require.Nil(t, d.pub.Load()) - assert.Equal(t, uint64(len(seed)+2), s.lookupPostings(key).estimate(), - "estimate reads through to the writer's bitmap, never a dropped snapshot") - assert.Nil(t, d.pub.Load(), - "weighing a written-since term must still not clone it") -} - -// TestPostings_FreshnessUnderConcurrentPublishers is -// TestConcurrentBitmaps_FreshnessUnderConcurrentPublishers aimed at -// the accessors #968 added: readers race the writer through -// lookupPostings instead of Get, and each read must observe the id -// whose AddTo returned before the read began. Run with -race. -func TestPostings_FreshnessUnderConcurrentPublishers(t *testing.T) { - s := newTestConcurrentBitmaps() - key := ComputeTermKey([]byte("postings-freshness-stress"), FieldTopic0) - - // ~200 containers, so a republish Clone is long enough for a - // concurrent reader to interleave with it. - seed := make([]uint32, 0, 200*8) - for c := range uint32(200) { - for j := range uint32(8) { - seed = append(seed, c*65_536+j) - } - } - s.AddTo(key, seed...) - seedCard := uint64(len(seed)) - - numReaders := max(16, 2*runtime.GOMAXPROCS(0)) - // firstID + numBatches*idStride + 65_536 stays below MaxUint32. - const ( - numBatches = 500 - firstID = uint32(20_000_000) - idStride = uint32(131_072) - perBatch = 3 // freshnessWriter adds three ids per batch - ) - - var committed, observed, batches atomic.Uint32 - var done atomic.Bool - var reads atomic.Uint64 - var wg sync.WaitGroup - - wg.Go(func() { - defer done.Store(true) - freshnessWriter(t, s, key, freshnessWriterCounters{ - committed: &committed, observed: &observed, batches: &batches, - }, numBatches, firstID, idStride) - }) - - for range numReaders { - wg.Go(func() { - for !done.Load() { - // Sample the batch count BEFORE the read: whatever it - // says is already durable in the writer's bitmap, so - // the read cannot legally weigh less. - finished := batches.Load() - want := committed.Load() - if want == 0 { - runtime.Gosched() - continue - } - p := s.lookupPostings(key) - if !p.present() { - t.Errorf("lookupPostings lost a term that has been written") - return - } - bm := p.bitmap() - if bm == nil { - t.Errorf("lookupPostings returned no bitmap for a dense term") - return - } - reads.Add(1) - // Yield: a read after a write takes the term mutex, - // and unyielding readers starve the writer. - runtime.Gosched() - if !bm.Contains(want) { - t.Errorf("lookupPostings returned a bitmap missing id %d, "+ - "committed before the lookup started (cardinality %d)", - want, bm.GetCardinality()) - return - } - // estimate runs after bitmap, so it can only have - // grown: a stale read of the dropped snapshot would - // come back short. - if est := p.estimate(); est < seedCard+uint64(finished)*perBatch { - t.Errorf("estimate = %d, below the %d ids committed before the read", - est, seedCard+uint64(finished)*perBatch) - return - } else if est < bm.GetCardinality() { - t.Errorf("estimate = %d, below the %d of the bitmap it just handed out", - est, bm.GetCardinality()) - return - } - storeMax(&observed, want) - } - }) - } - - wg.Wait() - t.Logf("postings freshness stress: %d reads, %d batches", reads.Load(), batches.Load()) - assert.Equal(t, uint32(numBatches), batches.Load(), "the writer must finish every batch") - assert.GreaterOrEqual(t, observed.Load(), committed.Load(), - "a reader must observe the final committed batch") - require.Positive(t, reads.Load(), "the stress loop must have done real reads") -} - -// TestHotStore_LookupPostingsSeesTheWriteJustMade carries the same -// property one layer up, through the seam the query planner actually -// calls: HotStore.lookupPostings must reflect an applyLedger that has -// returned, for a sparse term and for a dense one alike. -func TestHotStore_LookupPostingsSeesTheWriteJustMade(t *testing.T) { - h := openHotStoreForTest(t, chunk.ID(0)).store - // index() is the store's documented test-only write hook, which is - // what lets this drive the mirror without an ingest whose term - // derivation would decide the representations for us. - mirror := h.index() - sparseKey := ComputeTermKey([]byte("hot-sparse"), FieldTopic0) - denseKey := ComputeTermKey([]byte("hot-dense"), FieldTopic0) - - mirror.AddTo(denseKey, ascending(promotionThreshold, 4)...) - mirror.AddTo(sparseKey, 1, 2, 3) - - // Publish snapshots, then invalidate the dense one. - first, err := h.lookupPostings(t.Context(), []TermKey{sparseKey, denseKey}) - require.NoError(t, err) - require.Len(t, first, 2) - require.NotNil(t, first[1].bitmap()) - require.NotNil(t, denseOf(t, mirror, denseKey).pub.Load()) - - mirror.AddTo(sparseKey, 4) - mirror.AddTo(denseKey, 3_000_000) - require.Nil(t, denseOf(t, mirror, denseKey).pub.Load()) - - got, err := h.lookupPostings(t.Context(), []TermKey{sparseKey, denseKey}) - require.NoError(t, err) - require.Len(t, got, 2) - assert.Equal(t, []uint32{1, 2, 3, 4}, postingIDs(got[0]), - "the sparse term must carry the id added since the last lookup") - assert.Equal(t, uint64(4), got[0].estimate()) - assert.True(t, got[1].bitmap().Contains(uint32(3_000_000)), - "the dense term must carry the id added since its snapshot was dropped") - assert.Equal(t, uint64(promotionThreshold+1), got[1].estimate()) -} diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go index 0cd099966..528310cf6 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go @@ -30,14 +30,23 @@ package event // rather than re-clipping the accumulator there, would recover part of the // scan cost. // -// Ownership. acc is built by this call and is the only bitmap ever mutated. -// The term bitmaps handed to AndAny may be shared, copy-on-write-marked mirror -// snapshots (denseState.snapshot), which the index's contract forbids mutating -// or Cloning: AndAny reads them through roaring's read-only container -// accessors and writes only through the receiver, so passing them is safe, and -// roaring_contract_test.go pins that property against the pinned roaring -// version. Sparse terms never become bitmaps beyond the ids that land inside -// the slab under evaluation. +// Ownership. Every term the query names is materialized once, by the single +// batched Reader.LookupKeys call in Matches, and the resulting bitmaps are +// held for the whole walk. They are read-only: a hot dense term's bitmap is +// the denseState.snapshot every concurrent reader shares, which the index's +// contract forbids mutating or Cloning. AndAny reads them through roaring's +// read-only container accessors and writes only through the receiver, so +// passing them is safe, and roaring_contract_test.go pins that property +// against the pinned roaring version. acc is built by this call and is the +// only bitmap ever mutated. +// +// Freshness follows from holding them. The bitmaps are a point-in-time image +// of the index taken at query start — a sparse hot term is copied out of the +// atomically published id list, a dense one is denseState's published +// snapshot — so an id ingested while the walk runs is invisible to it, in +// either direction and at every slab. That is what the pinned window already +// promises: see IDRange's snapshot-isolation contract, under which no event +// past the pinned End is visible to the request anyway. import ( "cmp" @@ -57,16 +66,14 @@ import ( var slabShift uint = 16 // slabTerms is one of a filter's term groups resolved out of the batched -// lookup and held in whichever representation the index gave it: bitmaps for -// dense (and cold) terms, borrowed id lists for sparse ones. The group's value -// is the union of the two halves. +// lookup: the bitmaps the index returned for the group's present terms, held +// for the whole query. The group's value is their union. // -// est is the summed cardinality of the present terms over the whole chunk. It -// ignores the window, so it ranks a filter's groups rather than counting a -// query's candidates, and it is what orders the AND. +// est is the summed cardinality of those bitmaps. It ignores the window, so it +// ranks a filter's groups rather than counting a query's candidates, and it is +// what orders the AND. type slabTerms struct { bitmaps []*roaring.Bitmap - lists [][]uint32 est uint64 } @@ -75,40 +82,31 @@ type slabFilter struct { groups []slabTerms } -// resolveSlabTerms collects the postings at slots, reporting false when every +// resolveSlabTerms collects the bitmaps at slots, reporting false when every // one of them is absent from the index — the signal that the owning filter can // match nothing. // -// A present term holding no ids contributes nothing to the union but still -// keeps the group alive, which an absent term's caller-side skip would not do. -func resolveSlabTerms(sources []postings, slots []int) (slabTerms, bool) { +// A present term is a non-nil bitmap, empty or not: an empty one contributes +// nothing to the union but still keeps the group alive, which an absent term's +// caller-side skip would not do. +func resolveSlabTerms(sources []*roaring.Bitmap, slots []int) (slabTerms, bool) { var g slabTerms - present := false for _, slot := range slots { - p := sources[slot] - if !p.present() { - continue - } - present = true - g.est += p.estimate() - // A dense term is snapshotted once here, for the whole query, so every - // slab reads the same immutable bitmap. - if bm := p.bitmap(); bm != nil { - g.bitmaps = append(g.bitmaps, bm) + bm := sources[slot] + if bm == nil { continue } - if len(p.ids) > 0 { - g.lists = append(g.lists, p.ids) - } + g.bitmaps = append(g.bitmaps, bm) + g.est += bm.GetCardinality() } - return g, present + return g, len(g.bitmaps) > 0 } // resolveSlabFilters is the planning step: resolve every filter's groups, drop // the filters that named an entirely absent group, and order each survivor's // groups rarest first so the accumulator shrinks fastest and a group that // empties it ends the slab before the fat groups are read. -func resolveSlabFilters(plans []termPlan, sources []postings) []slabFilter { +func resolveSlabFilters(plans []termPlan, sources []*roaring.Bitmap) []slabFilter { out := make([]slabFilter, 0, len(plans)) for _, plan := range plans { groups := make([]slabTerms, 0, len(plan)) @@ -132,78 +130,24 @@ func resolveSlabFilters(plans []termPlan, sources []postings) []slabFilter { return out } -// slabScratch is the per-query reusable working set of the slab loop: the -// AndAny argument slice, and one bitmap holding whichever sparse ids land in -// the slab under evaluation. -// -// Reusing sparse across groups is safe because AndAny is done with its -// arguments when it returns — it copies out of them and never retains a -// container — which roaring_contract_test.go pins alongside the read-only -// property. -type slabScratch struct { - args []*roaring.Bitmap - sparse *roaring.Bitmap -} - -// inputs returns the AndAny arguments for g over [lo, hi): the group's term -// bitmaps, plus a scratch bitmap for the sparse ids inside the slab when the -// group has any. An empty result means the group holds nothing in this slab, -// so the owning filter matches nothing here. -// -// A group with no sparse terms hands back its own slice with no copy. -func (sc *slabScratch) inputs(g *slabTerms, lo, hi uint32) []*roaring.Bitmap { - if len(g.lists) == 0 { - return g.bitmaps - } - if sc.sparse == nil { - sc.sparse = roaring.New() - } else { - sc.sparse.Clear() - } - hit := false - for _, ids := range g.lists { - // Both bounds are found by binary search, so the borrowed list is - // never copied and never scanned outside the slab. - lower, _ := slices.BinarySearch(ids, lo) - tail := ids[lower:] - upper, _ := slices.BinarySearch(tail, hi) - if sub := tail[:upper]; len(sub) > 0 { - sc.sparse.AddMany(sub) - hit = true - } - } - sc.args = append(sc.args[:0], g.bitmaps...) - if hit { - sc.args = append(sc.args, sc.sparse) - } - return sc.args -} - // eval returns f's matches inside [lo, hi) as a freshly built bitmap the // caller owns, or nil when f matches nothing there. // // The accumulator starts as the slab window itself and is narrowed group by // group in place: AndAny is x.And(FastOr(args)) without the intermediate // union, so one call is a whole group. -func (f *slabFilter) eval(lo, hi uint32, sc *slabScratch) *roaring.Bitmap { - var acc *roaring.Bitmap +// +// A filter reaching here always names at least one group: one that names none +// matches everything and takes the match-all path upstream. +func (f *slabFilter) eval(lo, hi uint32) *roaring.Bitmap { + acc := roaring.New() + acc.AddRange(uint64(lo), uint64(hi)) for i := range f.groups { - inputs := sc.inputs(&f.groups[i], lo, hi) - if len(inputs) == 0 { - return nil - } - if acc == nil { - acc = roaring.New() - acc.AddRange(uint64(lo), uint64(hi)) - } - acc.AndAny(inputs...) + acc.AndAny(f.groups[i].bitmaps...) if acc.IsEmpty() { return nil } } - // acc is nil only for a filter that named no group at all, which takes the - // match-all path upstream and never reaches here; the nil is read as the - // empty candidate set either way. return acc } @@ -219,7 +163,6 @@ type slabStepper struct { cursor uint32 done bool - scratch slabScratch perFilter []*roaring.Bitmap // cur is the current slab's result, held only for its iterator. @@ -229,7 +172,7 @@ type slabStepper struct { } func newSlabStepper( - plans []termPlan, sources []postings, window IDRange, descending bool, + plans []termPlan, sources []*roaring.Bitmap, window IDRange, descending bool, ) *slabStepper { s := &slabStepper{ filters: resolveSlabFilters(plans, sources), @@ -283,7 +226,7 @@ func (s *slabStepper) nextBounds() (uint32, uint32, bool) { func (s *slabStepper) evalSlab(lo, hi uint32) *roaring.Bitmap { s.perFilter = s.perFilter[:0] for i := range s.filters { - if bm := s.filters[i].eval(lo, hi, &s.scratch); bm != nil { + if bm := s.filters[i].eval(lo, hi); bm != nil { s.perFilter = append(s.perFilter, bm) } } diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go index 3a4ffab45..dab8dc802 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go @@ -19,6 +19,7 @@ import ( "slices" "testing" + "github.com/RoaringBitmap/roaring/v2" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -411,13 +412,7 @@ func (f *shapedFixture) namedShapes() []namedShape { func TestMatches_ShapedMatrix(t *testing.T) { f := newShapedFixture(t) - readers := []struct { - name string - r Reader - }{ - {"lookupKeys", diffReader{f.corpus}}, - {"postings", diffPostingsReader{diffReader{f.corpus}}}, - } + r := diffReader{f.corpus} const slab = 1 << 16 windows := []struct { @@ -449,18 +444,16 @@ func TestMatches_ShapedMatrix(t *testing.T) { for _, sh := range f.namedShapes() { all := matchingEvents(t, f.corpus, sh.filters) - for _, seam := range readers { - for _, w := range windows { - for _, desc := range []bool{false, true} { - for _, limit := range limits { - requireStream(t, seam.r, all, queryCase{ - name: sh.name + "/" + seam.name + "/" + w.name, - filters: sh.filters, - window: w.w, - desc: desc, - limit: limit, - }) - } + for _, w := range windows { + for _, desc := range []bool{false, true} { + for _, limit := range limits { + requireStream(t, r, all, queryCase{ + name: sh.name + "/" + w.name, + filters: sh.filters, + window: w.w, + desc: desc, + limit: limit, + }) } } } @@ -473,7 +466,7 @@ func TestMatches_ShapedMatrix(t *testing.T) { // entirely inside the second slab. func TestMatches_WholeStreams(t *testing.T) { f := newShapedFixture(t) - r := diffPostingsReader{diffReader{f.corpus}} + r := diffReader{f.corpus} const slab = 1 << 16 for _, sh := range f.namedShapes() { @@ -498,7 +491,7 @@ func TestMatches_WholeStreams(t *testing.T) { // into a weaker test without failing it. func TestMatches_ShapedFixtureIsWhatItClaims(t *testing.T) { f := newShapedFixture(t) - r := diffPostingsReader{diffReader{f.corpus}} + r := diffReader{f.corpus} ctx := context.Background() window := IDRange{0, shapedCorpusSize} @@ -539,13 +532,14 @@ func TestMatches_ShapedFixtureIsWhatItClaims(t *testing.T) { // The overlap shape is only the overlap shape if both sides are // chunk-sized and their meeting point is rare. - fat, err := r.lookupPostings(ctx, []TermKey{ + fat, err := r.LookupKeys(ctx, []TermKey{ ComputeTermKey(f.vocab.contracts[0], FieldContractID), ComputeTermKey(f.vocab.topicRaw[1], topicField(0)), }) require.NoError(t, err) - for i, p := range fat { - require.Greater(t, p.estimate(), uint64(shapedCorpusSize/3), + for i, bm := range fat { + require.NotNil(t, bm, "thin-overlap term %d must be indexed", i) + require.Greater(t, bm.GetCardinality(), uint64(shapedCorpusSize/3), "thin-overlap term %d must be chunk-sized", i) } } @@ -566,31 +560,20 @@ func TestMatches_RandomizedAgainstPostFilter(t *testing.T) { matchBatchSize = 7 defer func(s uint) { slabShift = s }(slabShift) - readers := []struct { - name string - r Reader - }{ - {"lookupKeys", diffReader{corpus}}, - {"postings", diffPostingsReader{diffReader{corpus}}}, - } + r := diffReader{corpus} // The slab width is a seam, not a behavior: every width must reproduce the // same stream. 1, 2 and 4 put 150, 75 and 19 slab seams inside the corpus, // 8 leaves a single seam, and 16 is the production width, where the whole // corpus is one slab. for _, shift := range []uint{1, 2, 4, 8, 16} { slabShift = shift - for _, seam := range readers { - r := seam.r - t.Run(seam.name, func(t *testing.T) { - rng := rand.New(rand.NewSource(int64(20260909 + shift))) - matched := 0 - for trial := range 400 { - matched += randomizedTrial(t, r, corpus, v, rng, corpusSize, trial) - } - require.Greater(t, matched, 2000, - "fixture sanity: randomized queries selected too little") - }) + rng := rand.New(rand.NewSource(int64(20260909 + shift))) + matched := 0 + for trial := range 400 { + matched += randomizedTrial(t, r, corpus, v, rng, corpusSize, trial) } + require.Greater(t, matched, 2000, + "fixture sanity: randomized queries selected too little") } } @@ -648,20 +631,17 @@ func requireStrictOrder(t *testing.T, got []Match, desc bool, trial int) { // more I/O. Recording the fetches turns "the right answer" into "the right // work". type fetchTracer struct { - diffPostingsReader + diffReader batches *[][]uint32 } func (r fetchTracer) FetchEvents(ctx context.Context, ids []uint32) ([]Payload, error) { *r.batches = append(*r.batches, slices.Clone(ids)) - return r.diffPostingsReader.FetchEvents(ctx, ids) + return r.diffReader.FetchEvents(ctx, ids) } -var ( - _ Reader = fetchTracer{} - _ postingReader = fetchTracer{} -) +var _ Reader = fetchTracer{} // TestMatches_FetchesOnlyTrueCandidates pins the candidate set itself: the // ordinals the engine fetches are the query's true matches, in emission order, @@ -730,7 +710,7 @@ func traceFetches( ) []uint32 { t.Helper() batches := [][]uint32{} - r := fetchTracer{diffPostingsReader{diffReader{f.corpus}}, &batches} + r := fetchTracer{diffReader{f.corpus}, &batches} drainMatches(t, Matches(context.Background(), r, filters, w, desc, limit), limit) out := []uint32{} @@ -754,29 +734,42 @@ func pageFloor(limit, answer int) int { // ───────────────────────── the planning step ───────────────────────── +// termBitmap is the shape LookupKeys hands the planner: one materialized +// bitmap per present term, nil for an absent one. +func termBitmap(ids ...uint32) *roaring.Bitmap { + bm := roaring.New() + bm.AddMany(ids) + return bm +} + // What a group reports about itself: presence, and its summed weight. func TestResolveSlabTerms(t *testing.T) { - sources := []postings{ - sparseSource(1, 2), - {}, // absent - denseSource(2, 3, 4), + sources := []*roaring.Bitmap{ + termBitmap(1, 2), + nil, // absent + termBitmap(2, 3, 4), + termBitmap(), // present, holding nothing } g, ok := resolveSlabTerms(sources, []int{0}) require.True(t, ok) assert.Equal(t, uint64(2), g.est) - assert.Equal(t, [][]uint32{{1, 2}}, g.lists) - assert.Empty(t, g.bitmaps, "a sparse term stays an id list") + assert.Len(t, g.bitmaps, 1, "the group holds the term's bitmap itself") + assert.Same(t, sources[0], g.bitmaps[0], "the lookup's bitmap is held, not copied") g, ok = resolveSlabTerms(sources, []int{0, 2}) require.True(t, ok) assert.Equal(t, uint64(5), g.est, "a group's terms sum, overlaps double-counted") - assert.Len(t, g.bitmaps, 1) - assert.Len(t, g.lists, 1, "a mixed group keeps both representations") + assert.Len(t, g.bitmaps, 2) g, ok = resolveSlabTerms(sources, []int{1, 2}) require.True(t, ok) assert.Equal(t, uint64(3), g.est, "an absent term adds nothing") + assert.Len(t, g.bitmaps, 1, "an absent term is not held") + + g, ok = resolveSlabTerms(sources, []int{3}) + require.True(t, ok, "a present-but-empty term keeps its group alive") + assert.Equal(t, uint64(0), g.est) g, ok = resolveSlabTerms(sources, []int{1}) assert.False(t, ok, "a group of absent terms drops its filter") @@ -787,11 +780,11 @@ func TestResolveSlabTerms(t *testing.T) { // accumulator shrinks fastest and a group that empties it ends the slab before // the fat groups are read. func TestResolveSlabFiltersOrdersRarestFirst(t *testing.T) { - sources := []postings{ - denseSource(1, 2, 3, 4, 5, 6, 7, 8), // 0: the fat group - denseSource(2, 4, 6, 8), // 1 - denseSource(4, 8), // 2: the rare group - {}, // 3: absent + sources := []*roaring.Bitmap{ + termBitmap(1, 2, 3, 4, 5, 6, 7, 8), // 0: the fat group + termBitmap(2, 4, 6, 8), // 1 + termBitmap(4, 8), // 2: the rare group + nil, // 3: absent } got := resolveSlabFilters([]termPlan{{{0}, {2}, {1}}}, sources) From 709514a91479be4170d7f1608f4d49e1310807ad Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 16:18:52 +0000 Subject: [PATCH 26/41] events: the slab walk skips slabs it can prove hold no candidate MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The walk stepped to the slab adjacent to the last one it evaluated, so a term holding five ids spread over a chunk paid for every slab between them: one accumulator, one AddRange and one AndAny per group per empty slab. It now asks the held term bitmaps where the next candidate could be and seeks there. The bound is proved from the same bitmaps the slab would have been evaluated with, through roaring's NextValue ascending and PreviousValue descending: - within a group the terms are OR-ed, so a candidate is at or above the smallest of their NextValue(pos), and no term holding one proves the group empty from pos on; - within a filter the groups are AND-ed, so the filter's bound is the largest of its groups', and one empty group ends the filter; - across filters the results are OR-ed, so the query's bound is the smallest of the live filters', and all filters ended ends the walk. Descending mirrors it with the min/max swapped. The post-filter only drops candidates, so a bound proved on the index bounds the stream. A group naming no term proves no bound and returns the cursor unchanged: unreachable today, but the answer that skips nothing is the safe one. The bound also clips the accumulator inside its own slab, so a slab is entered at the candidate rather than at its base — which is what the descending trade in the file header wanted for the first slab of a descending walk, and now gets for every slab in both directions. Gates. roaring_contract_test.go grows the pin the skip rests on: both searches answer inclusive of the target and -1 for none, checked against the bitmap's own ids over all three container kinds, and neither writes to the bitmap it reads — they run on denseState.snapshot bitmaps shared with every other in-flight query, under -race in the concurrent case. TestSlabStepperSkipsCandidateFreeSlabs drives nextBounds directly and requires the exact slabs opened, since a stream cannot see the difference: the rare term opens 3 slabs of 10 in both directions, ANDing a chunk-sized term with it opens the same 3, that chunk-sized term alone opens all 10, and two OR-ed rare filters open the union of theirs. TestMatches_RareTermsSpanSlabs is the oracle gate over the shaped corpus, running the rare shapes at 64-, 1024- and 8192-wide slabs, where the walk skips hundreds of slabs per query and an over-skip loses matches. Inverting either combination rule, or the direction, fails the shaped matrix, the randomized matrix and the candidate-set pin. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../stores/event/roaring_contract_test.go | 125 ++++++++++++- .../internal/rpcv2/stores/event/slab_match.go | 177 +++++++++++++++++- .../rpcv2/stores/event/slab_match_test.go | 134 +++++++++++++ 3 files changed, 424 insertions(+), 12 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go index 7844c91a4..c3c69c424 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go @@ -1,8 +1,10 @@ package event -// roaring_contract_test.go pins the library-level property the event index is +// roaring_contract_test.go pins the library-level properties the event index is // built on: the aggregation entry points this package calls with shared -// bitmaps must treat those bitmaps as read-only. +// bitmaps must treat those bitmaps as read-only, and the value searches the +// slab walk proves its skips with must answer inclusively, read-only, and -1 +// for none. // // ConcurrentBitmaps.Get and denseState.snapshot publish one bitmap to every // concurrent reader at once and never mutate it afterwards. Readers therefore @@ -21,6 +23,7 @@ package event import ( "math/rand" + "slices" "sync" "testing" @@ -278,3 +281,121 @@ func TestRoaringContract_ConcurrentReadersShareArguments(t *testing.T) { require.Equal(t, wantFastOr, results[g][2], "goroutine %d disagreed on FastOr", g) } } + +// TestRoaringContract_ValueSearchIsInclusiveAndReadOnly pins NextValue and +// PreviousValue, the searches the slab walk proves a skip with. The walk skips +// every slab between the cursor and the answer, so an answer that overshot the +// target in the walk's direction — or a search that reported -1 with ids still +// to come — would silently drop matches. +// +// The definition is checked against the bitmap's own ids rather than against +// hand-written expectations, over targets that land on an id, between two, on +// and beside a container boundary, and past the last id. +func TestRoaringContract_ValueSearchIsInclusiveAndReadOnly(t *testing.T) { + t.Parallel() + + for i, shared := range sharedBitmaps(t) { + ids := shared.ToArray() + require.NotEmpty(t, ids, "fixture: bitmap %d holds nothing", i) + before := bitmapBytes(t, shared) + + for _, target := range searchTargets(ids) { + // The definition, read off the ids: the first id at or after the + // target, and the last id at or before it. + at, _ := slices.BinarySearch(ids, target) + wantNext, wantPrev := int64(-1), int64(-1) + if at < len(ids) { + wantNext = int64(ids[at]) + } + if at < len(ids) && ids[at] == target { + wantPrev = int64(target) + } else if at > 0 { + wantPrev = int64(ids[at-1]) + } + + require.Equal(t, wantNext, shared.NextValue(target), + "bitmap %d: NextValue(%d) must be the first id at or above the target", + i, target) + require.Equal(t, wantPrev, shared.PreviousValue(target), + "bitmap %d: PreviousValue(%d) must be the last id at or below the target", + i, target) + } + + require.Equal(t, before, bitmapBytes(t, shared), + "bitmap %d: a value search mutated the bitmap it searched: the slab "+ + "walk cannot run them on denseState.snapshot bitmaps at this version", i) + } + + empty := roaring.New() + require.Equal(t, int64(-1), empty.NextValue(0), + "an empty bitmap must report no next value") + require.Equal(t, int64(-1), empty.PreviousValue(1<<20), + "an empty bitmap must report no previous value") +} + +// searchTargets returns the positions worth asking about for a bitmap holding +// ids: each id, its neighbors, every container boundary the ids span, and the +// ends of the uint32 range. +func searchTargets(ids []uint32) []uint32 { + out := []uint32{0, 1<<32 - 1} + for _, id := range ids { + out = append(out, id) + if id > 0 { + out = append(out, id-1) + } + if id < 1<<32-1 { + out = append(out, id+1) + } + } + for key := range uint32(6) { + out = append(out, key<<16) + if key > 0 { + out = append(out, key<<16-1) + } + } + slices.Sort(out) + return slices.Compact(out) +} + +// TestRoaringContract_ConcurrentValueSearch is the race-detector gate for the +// searches: the slab walk runs them on the same denseState snapshot from every +// in-flight query at once, so they must not lazily materialize anything inside +// the bitmap they read. +func TestRoaringContract_ConcurrentValueSearch(t *testing.T) { + t.Parallel() + + const goroutines = 8 + shared := sharedBitmaps(t) + before := bitmapImages(t, shared) + + targets := []uint32{0, 1, 1 << 15, 1 << 16, 1<<16 + 1, 3 << 16, 1<<20 - 1} + want := make([][]int64, len(shared)) + for i, bm := range shared { + for _, target := range targets { + want[i] = append(want[i], bm.NextValue(target), bm.PreviousValue(target)) + } + } + + got := make([][][]int64, goroutines) + var wg sync.WaitGroup + start := make(chan struct{}) + for g := range goroutines { + wg.Go(func() { + <-start + out := make([][]int64, len(shared)) + for i, bm := range shared { + for _, target := range targets { + out[i] = append(out[i], bm.NextValue(target), bm.PreviousValue(target)) + } + } + got[g] = out + }) + } + close(start) + wg.Wait() + + requireUnchanged(t, shared, before, "concurrent value search") + for g := range goroutines { + require.Equal(t, want, got[g], "goroutine %d disagreed on the value searches", g) + } +} diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go index 528310cf6..61d781a1d 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go @@ -18,6 +18,12 @@ package event // the window. The unit is the slab, so a page ending mid-slab has paid for // that whole slab: one container's worth of work per input per filter. // +// Slabs that can hold no candidate are not paid for at all. Before evaluating +// one the walk asks the held term bitmaps where the next candidate could be +// and jumps there when the answer is past this slab, so a rare term spread +// over a wide window costs its own slabs and no others. See "the skip" below +// for the bound and why it is sound. +// // The trade is descending over a whole window. The whole-chunk union this // replaced ANDed the chunk-sized terms once with roaring's bulk aggregation, // where the walk re-enters the algebra per slab. Measured over a 300k-event @@ -26,9 +32,9 @@ package event // on candidates alone (1.20ms → 1.37ms), which the fetch swallows end to end // (10.88ms → 10.93ms). Descending pages that stop early got faster (148µs → // 128µs), because the union paid for the whole chunk before yielding -// anything. Seeding the walk at the slab holding the window's high bound, -// rather than re-clipping the accumulator there, would recover part of the -// scan cost. +// anything. The skip closes the rest of that gap where the terms are sparse +// enough to prove a jump, and seeds the first descending slab at the highest +// candidate rather than at the window's high bound. // // Ownership. Every term the query names is materialized once, by the single // batched Reader.LookupKeys call in Matches, and the resulting bitmaps are @@ -151,6 +157,100 @@ func (f *slabFilter) eval(lo, hi uint32) *roaring.Bitmap { return acc } +// ──────────────────────── the skip ──────────────────────────── +// +// The walk skips a slab only on a bound it has proved, from the same bitmaps +// it would have evaluated the slab with. Ascending, at position pos: +// +// - a candidate lies in at least one of a group's terms, so it is at or +// above the smallest id at or above pos that any of them holds — the +// minimum of their NextValue(pos). No term holding one proves the owning +// filter matches nothing from pos on; +// - a candidate satisfies every group of its filter, so the filter's bound +// is the largest of its groups' bounds; +// - a candidate belongs to some filter, so the query's bound is the +// smallest of the surviving filters' bounds, and all filters exhausted +// means the walk is over. +// +// Descending is the mirror: PreviousValue, maximum within a group, minimum +// across a filter's groups, maximum across filters. The post-filter only ever +// drops candidates, so a bound proved on the index bounds the stream. +// +// roaring's NextValue and PreviousValue answer inclusive of the target and +// -1 for none, and read the bitmap without writing it — they are called on +// snapshots shared with every other reader. roaring_contract_test.go pins +// both properties against the pinned version. + +// nextBound is the group's bound: the smallest id at or above pos that any of +// its terms holds. ok is false when none does, which proves the group, and so +// the filter owning it, matches nothing from pos on. +func (g *slabTerms) nextBound(pos uint32) (uint32, bool) { + if len(g.bitmaps) == 0 { + // A group naming no term constrains nothing and proves no bound. + // resolveSlabTerms never builds one and a filter that would take + // the match-all path never reaches the stepper, but the answer that + // skips nothing is the safe one to give. + return pos, true + } + best := int64(-1) + for _, bm := range g.bitmaps { + v := bm.NextValue(pos) + if v >= 0 && (best < 0 || v < best) { + best = v + } + } + if best < 0 { + return 0, false + } + return uint32(best), true +} + +// prevBound is nextBound descending: the largest id at or below pos. +func (g *slabTerms) prevBound(pos uint32) (uint32, bool) { + if len(g.bitmaps) == 0 { + return pos, true + } + best := int64(-1) + for _, bm := range g.bitmaps { + v := bm.PreviousValue(pos) + if v > best { + best = v + } + } + if best < 0 { + return 0, false + } + return uint32(best), true +} + +// nextBound is the filter's bound: a candidate satisfies every group, so the +// strongest of the groups' bounds holds. ok is false as soon as one group +// proves the filter is done. +func (f *slabFilter) nextBound(pos uint32) (uint32, bool) { + bound := pos + for i := range f.groups { + b, ok := f.groups[i].nextBound(pos) + if !ok { + return 0, false + } + bound = max(bound, b) + } + return bound, true +} + +// prevBound is nextBound descending: the smallest of the groups' bounds. +func (f *slabFilter) prevBound(pos uint32) (uint32, bool) { + bound := pos + for i := range f.groups { + b, ok := f.groups[i].prevBound(pos) + if !ok { + return 0, false + } + bound = min(bound, b) + } + return bound, true +} + // slabStepper walks one query's slabs in emission order, evaluating a slab // only when the consumer has drained the previous one. type slabStepper struct { @@ -187,19 +287,72 @@ func newSlabStepper( return s } +// seekAsc is the lowest position at or above pos that any filter can still +// match at, and false when none can inside the window: a candidate belongs to +// some filter, so the smallest of their bounds holds for the union. Filters that +// are done are dropped from the minimum rather than ending the walk, since a +// live one may still match. +func (s *slabStepper) seekAsc(pos uint32) (uint32, bool) { + var best uint32 + found := false + for i := range s.filters { + b, ok := s.filters[i].nextBound(pos) + if !ok { + continue + } + if !found || b < best { + best, found = b, true + } + } + if !found || best >= s.window.End { + return 0, false + } + return best, true +} + +// seekDesc is seekAsc mirrored: the largest of the filters' bounds at or below +// hi-1, returned as the exclusive high bound the walk resumes at. +func (s *slabStepper) seekDesc(hi uint32) (uint32, bool) { + var best uint32 + found := false + for i := range s.filters { + b, ok := s.filters[i].prevBound(hi - 1) + if !ok { + continue + } + if !found || b > best { + best, found = b, true + } + } + if !found || best < s.window.Start { + return 0, false + } + return best + 1, true +} + // nextBounds returns the next slab's [lo, hi) clipped to the window, walking -// away from the cursor in the query's direction. The first slab is clipped at -// the cursor by these bounds alone. +// away from the cursor in the query's direction. +// +// The cursor moves to the proved bound first, so the slab returned is the one +// holding the next possible candidate rather than the one adjacent to the +// last: every slab between is candidate-free for every filter. The bound also +// clips the accumulator inside its own slab, so a slab entered part-way is +// entered at the candidate and not at its base. func (s *slabStepper) nextBounds() (uint32, uint32, bool) { if s.done { return 0, 0, false } if s.desc { - hi := s.cursor + // hi-1 is read inside seekDesc. hi > window.Start >= 0 here, because + // an empty window never reaches the stepper and the walk stops at + // window.Start. + hi, ok := s.seekDesc(s.cursor) + if !ok { + s.done = true + return 0, 0, false + } lo := s.window.Start - // The base of the slab holding hi-1. hi > window.Start >= 0 here, - // because an empty window never reaches the stepper and the walk stops - // at window.Start. + // The base of the slab holding hi-1. if base := ((uint64(hi) - 1) >> slabShift) << slabShift; base > uint64(lo) { lo = uint32(base) //nolint:gosec // base < hi <= MaxUint32 } else { @@ -208,7 +361,11 @@ func (s *slabStepper) nextBounds() (uint32, uint32, bool) { s.cursor = lo return lo, hi, true } - lo := s.cursor + lo, ok := s.seekAsc(s.cursor) + if !ok { + s.done = true + return 0, 0, false + } hi := s.window.End // The base of the slab above the one holding lo. if next := (uint64(lo)>>slabShift + 1) << slabShift; next < uint64(hi) { diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go index dab8dc802..71a25e613 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go @@ -5,6 +5,10 @@ package event // match-all, the fat/fat thin-overlap shape whose two chunk-sized terms meet // on a handful of ids, and the high-arity AND and union. // +// The walk skips slabs it can prove hold no candidate, so those same shapes +// run again at slab widths narrow enough to put dozens of empty slabs between +// consecutive ids, where a bound that over-skipped would drop matches. +// // The answer every case is checked against is computed without the index — // postFilter run over every ordinal in the corpus, then clipped to the window, // the direction and the page. It shares no code with the term planning, the @@ -800,3 +804,133 @@ func TestResolveSlabFiltersOrdersRarestFirst(t *testing.T) { assert.Len(t, resolveSlabFilters([]termPlan{{{0}, {3}}, {{1}}}, sources), 1, "the drop takes only its own filter") } + +// ───────────────────────── the skip ───────────────────────── + +// The skip is invisible in a stream: a walk that evaluated every slab in the +// window yields exactly the same ids. Driving nextBounds directly reports the +// slabs the walk actually opens, which is the claim — and, in the fat case, +// the claim that a provable bound never skips a slab that holds something. +func TestSlabStepperSkipsCandidateFreeSlabs(t *testing.T) { + defer func(s uint) { slabShift = s }(slabShift) + slabShift = 16 + const slab = 1 << 16 + const slabs = 10 + window := IDRange{0, slabs * slab} + + // A rare term whose three ids sit in slabs 0, 3 and 9; a second rare term + // in slabs 1 and 6; a chunk-sized term holding every id in the window; and + // a term present but empty. + fat := roaring.New() + fat.AddRange(uint64(window.Start), uint64(window.End)) + sources := []*roaring.Bitmap{ + termBitmap(5, 3*slab+7, 9*slab+1), + termBitmap(slab+1, 6*slab+3), + fat, + termBitmap(), + } + + walk := func(plans []termPlan, desc bool) [][2]uint32 { + st := newSlabStepper(plans, sources, window, desc) + out := [][2]uint32{} + for { + lo, hi, ok := st.nextBounds() + if !ok { + return out + } + out = append(out, [2]uint32{lo, hi}) + } + } + + // The rare term alone opens its own three slabs and no others, in either + // direction, and each one is entered at the candidate rather than at the + // slab's base. + rareAsc := [][2]uint32{{5, slab}, {3*slab + 7, 4 * slab}, {9*slab + 1, 10 * slab}} + rareDesc := [][2]uint32{ + {9 * slab, 9*slab + 2}, {3 * slab, 3*slab + 8}, {0, 6}, + } + assert.Equal(t, rareAsc, walk([]termPlan{{{0}}}, false)) + assert.Equal(t, rareDesc, walk([]termPlan{{{0}}}, true)) + + // ANDing it with a chunk-sized term changes nothing: the AND's bound is + // the strongest of its groups', and the fat group proves none. + assert.Equal(t, rareAsc, walk([]termPlan{{{0}, {2}}}, false), + "a chunk-sized group must not weaken the rare group's bound") + assert.Equal(t, rareDesc, walk([]termPlan{{{0}, {2}}}, true)) + + // The chunk-sized term alone opens every slab: a bound is a bound, and + // this one proves nothing to skip. + full := make([][2]uint32, 0, slabs) + for i := range uint32(slabs) { + full = append(full, [2]uint32{i * slab, (i + 1) * slab}) + } + assert.Len(t, full, slabs) + assert.Equal(t, full, walk([]termPlan{{{2}}}, false), + "a term holding every id must not skip a slab") + + // Across OR-ed filters the union of their slabs is opened, and nothing + // else: slabs 0, 1, 3, 6, 9. + assert.Equal(t, [][2]uint32{ + {5, slab}, + {slab + 1, 2 * slab}, + {3*slab + 7, 4 * slab}, + {6*slab + 3, 7 * slab}, + {9*slab + 1, 10 * slab}, + }, walk([]termPlan{{{0}}, {{1}}}, false)) + + // A filter whose only term is present but empty ends the walk before the + // first slab, rather than reading all ten. + assert.Empty(t, walk([]termPlan{{{3}}}, false)) + assert.Empty(t, walk([]termPlan{{{3}}}, true)) +} + +// TestMatches_RareTermsSpanSlabs is the oracle gate on the skip. The shapes a +// rare term dominates — five ids spread over the whole corpus, alone and ANDed +// with a chunk-sized term — run at slab widths that put hundreds of +// candidate-free slabs between consecutive ids, where the walk skips almost +// everything and a bound that over-skipped would silently lose matches. +// +// The windows start and end between rare ids, so the first and last slab the +// walk seeks to are clipped by the window rather than by a candidate. +func TestMatches_RareTermsSpanSlabs(t *testing.T) { + f := newShapedFixture(t) + r := diffReader{f.corpus} + defer func(s uint) { slabShift = s }(slabShift) + + require.LessOrEqual(t, len(f.rareContract), 5, + "fixture: the rare term must stay rare for the skip to matter") + + shapes := []namedShape{ + {"rare alone", f.filterSparseOnly()}, + {"rare and chunk-sized", f.filterMixedGroups()}, + {"rare or chunk-sized", f.filterUnion()}, + } + windows := []IDRange{ + {0, shapedCorpusSize}, + {20_000, shapedCorpusSize}, + {0, 60_000}, + {20_000, 60_000}, + } + + for _, sh := range shapes { + all := matchingEvents(t, f.corpus, sh.filters) + // 64-, 1024- and 8192-wide slabs put 1093, 68 and 8 seams inside the + // corpus against the 5 ids the rare term holds. + for _, shift := range []uint{6, 10, 13} { + slabShift = shift + for _, w := range windows { + for _, desc := range []bool{false, true} { + for _, limit := range []int{0, 1, 3} { + requireStream(t, r, all, queryCase{ + name: sh.name, + filters: sh.filters, + window: w, + desc: desc, + limit: limit, + }) + } + } + } + } + } +} From ae923c41014588fdaa1ff8833a28dd8e028efc35 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 16:22:17 +0000 Subject: [PATCH 27/41] events: union the per-filter slab results in place MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit evalSlab handed its per-filter results to roaring.FastOr, which allocates a fresh answer and copies every input into it. The inputs are bitmaps eval built for this call and nobody else holds — no shared snapshot reaches here, so there is no read-only argument to respect — and the union can run in place into the first of them. The per-filter slice existed only to hold FastOr's variadic arguments and goes with it: the results are OR-ed as they are produced. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/stores/event/slab_match.go | 34 +++++++++---------- 1 file changed, 17 insertions(+), 17 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go index 61d781a1d..bdebb700b 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go @@ -43,8 +43,9 @@ package event // contract forbids mutating or Cloning. AndAny reads them through roaring's // read-only container accessors and writes only through the receiver, so // passing them is safe, and roaring_contract_test.go pins that property -// against the pinned roaring version. acc is built by this call and is the -// only bitmap ever mutated. +// against the pinned roaring version. The only bitmaps this file mutates are +// the per-filter accumulators it built itself, which is also why the union +// across filters can run in place into the first of them. // // Freshness follows from holding them. The bitmaps are a point-in-time image // of the index taken at query start — a sparse hot term is copied out of the @@ -263,8 +264,6 @@ type slabStepper struct { cursor uint32 done bool - perFilter []*roaring.Bitmap - // cur is the current slab's result, held only for its iterator. cur *roaring.Bitmap asc roaring.ManyIntIterable @@ -378,23 +377,24 @@ func (s *slabStepper) nextBounds() (uint32, uint32, bool) { } // evalSlab is the union across filters of their per-slab results, or nil when -// the slab holds nothing. Every input is a bitmap this call owns, so the -// caller owns the union whichever path FastOr takes. +// the slab holds nothing. +// +// Every per-filter result is a bitmap eval built for this call and nobody else +// holds, so the union runs in place into the first of them rather than +// allocating a separate answer to copy them all into. func (s *slabStepper) evalSlab(lo, hi uint32) *roaring.Bitmap { - s.perFilter = s.perFilter[:0] + var acc *roaring.Bitmap for i := range s.filters { - if bm := s.filters[i].eval(lo, hi); bm != nil { - s.perFilter = append(s.perFilter, bm) + bm := s.filters[i].eval(lo, hi) + switch { + case bm == nil: + case acc == nil: + acc = bm + default: + acc.Or(bm) } } - switch len(s.perFilter) { - case 0: - return nil - case 1: - return s.perFilter[0] - default: - return roaring.FastOr(s.perFilter...) - } + return acc } // ensureSlab advances to the next slab that holds a match, reporting false From 3c5a9042c884f0a22bc384f4d6395a64fc14a6a4 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 16:26:13 +0000 Subject: [PATCH 28/41] packfile: count the pooled buffers a capacity cap drops MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The file comment says what happens when a cap is crossed — the Put is skipped, the pool drains, every open allocates afresh — and calls it silent. It no longer has to be: each dropped Put now increments a process-wide counter, read through PoolCapSkips. Zero is the state the caps are chosen for. Any count means some packfile crossed one; a count climbing with the open rate means the pools have stopped recycling and the cap the file comment sizes against its geometry wants raising. Only the cap arm counts. An empty slice was never poolable, and the buffer a size-miss Get hands back is under its cap by construction — the pool held it — so the two are split out of the one condition they shared. Counter and accessor only, in the shape rocksdb.DeferredCloseOps and ledger.MissingPackOpens already have. Wiring it is one counterFunc line in observability.NewPrometheusMetrics, beside those two. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/packfile/pools.go | 51 ++++++++++++++++--- .../internal/rpcv2/packfile/pools_test.go | 24 +++++++++ 2 files changed, 67 insertions(+), 8 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go b/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go index bdc3a5538..7ac799a0a 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go @@ -10,11 +10,15 @@ package packfile // Puts are capacity-capped so one pathological file cannot pin an arbitrarily // large array in a pool slot; larger buffers fall to the garbage collector. // Every cap must therefore sit above what a legitimate packfile needs, because -// crossing one is silent: the Put is skipped, the pool drains, and every open -// allocates afresh — still correct, but without the allocation win these pools -// exist for. +// crossing one is otherwise silent: the Put is skipped, the pool drains, and +// every open allocates afresh — still correct, but without the allocation win +// these pools exist for. capSkips counts those skipped Puts so the drain is +// visible from outside the process instead. -import "sync" +import ( + "sync" + "sync/atomic" +) // maxPooledOffsets caps the decoded offset table (recordCount+1 int64s, so an // 8 MiB array at the cap) a Put may retain. What it has to exceed is fixed by @@ -46,9 +50,28 @@ var ( openBufPool sync.Pool // *[]byte ) +// capSkips counts the Puts dropped for exceeding their pool's cap, across all +// three pools. Each one is a buffer the pool did not get back, so a count that +// climbs with the open rate is the file comment's silent drain in progress. +// Process-wide by design — the metrics exporter reads it via PoolCapSkips. +// +//nolint:gochecknoglobals // one tally across process-wide pools; read-only outside this file +var capSkips atomic.Uint64 + +// PoolCapSkips returns the process-wide count of pooled buffers dropped for +// exceeding a capacity cap. +// +// Zero is the state the caps are chosen for, and the only healthy one: any +// count means some packfile crossed a cap, and a climbing count means the +// pools have stopped recycling. Raise the cap the file comment sizes against +// the geometry that grew. See capSkips. +func PoolCapSkips() uint64 { return capSkips.Load() } + // A size miss hands the pooled buffer back before allocating: Get has already // removed it from the pool, so returning it is the only thing that keeps a run -// of growing opens from draining the pool one buffer per open. +// of growing opens from draining the pool one buffer per open. That buffer is +// under its cap by construction — the pool held it — so it never counts as a +// cap skip. func getOffsets(n int) []int64 { if p, _ := offsetsPool.Get().(*[]int64); p != nil { @@ -63,7 +86,11 @@ func getOffsets(n int) []int64 { // putOffsets recycles a decoded offset table. The caller must guarantee no // live reference remains; see Reader.Close for the in-flight handshake. func putOffsets(s []int64) { - if cap(s) == 0 || cap(s) > maxPooledOffsets { + if cap(s) == 0 { + return + } + if cap(s) > maxPooledOffsets { + capSkips.Add(1) return } s = s[:0] @@ -81,7 +108,11 @@ func getScratch(n int) []uint32 { } func putScratch(s []uint32) { - if cap(s) == 0 || cap(s) > maxPooledScratch { + if cap(s) == 0 { + return + } + if cap(s) > maxPooledScratch { + capSkips.Add(1) return } s = s[:0] @@ -99,7 +130,11 @@ func getOpenBuf(n int) []byte { } func putOpenBuf(s []byte) { - if cap(s) == 0 || cap(s) > maxPooledOpenBuf { + if cap(s) == 0 { + return + } + if cap(s) > maxPooledOpenBuf { + capSkips.Add(1) return } s = s[:0] diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go b/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go index b0e418792..8cf045b15 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go @@ -101,3 +101,27 @@ func TestReaderCloseDuringReadDoesNotRecycle(t *testing.T) { require.ErrorIs(t, err, os.ErrClosed) } } + +// The counter is the only signal the caps have been crossed, so it must count +// exactly the Puts a cap dropped: an over-cap buffer on every pool, and +// nothing for the buffers a Put keeps or for the empty slices it ignores. +// +// The count is process-wide, so the assertions are on the delta. +func TestPoolCapSkipsCountsDroppedPuts(t *testing.T) { + before := PoolCapSkips() + + putOffsets(make([]int64, 0, maxPooledOffsets)) + putScratch(make([]uint32, 0, maxPooledScratch)) + putOpenBuf(make([]byte, 0, maxPooledOpenBuf)) + putOffsets(nil) + putScratch(nil) + putOpenBuf(nil) + require.Equal(t, before, PoolCapSkips(), + "a Put at the cap is pooled and an empty one is ignored; neither is a cap skip") + + putOffsets(make([]int64, 0, maxPooledOffsets+1)) + putScratch(make([]uint32, 0, maxPooledScratch+1)) + putOpenBuf(make([]byte, 0, maxPooledOpenBuf+1)) + require.Equal(t, before+3, PoolCapSkips(), + "each pool must count the Put its cap dropped") +} From 3586a7c056f72a14cb0308c0762e7073dbc9f2b6 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 17:06:59 +0000 Subject: [PATCH 29/41] events: the slab walk holds the bounds it proves MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every step proved every filter's bound from scratch, and every slab was evaluated for every filter. A filter whose next candidate sat ten slabs away was asked again at each of those slabs — one NextValue per term of every group — and evaluated in each of them too, spending a roaring.New, an AddRange and an AndAny per group to rediscover it matches nothing there. A filter whose terms had run out was dropped from the minimum and then asked and evaluated again for the rest of the window. Bounds are monotone in the walk direction, so a bound proved at one cursor position still proves the same emptiness at every position the cursor reaches up to it. Each filter's bound is now held: only the ones the cursor has reached are proved again, a filter whose bound lies past the current slab is not evaluated in it at all, and a filter whose bound is exhausted is retired from the walk for good. What a slab costs is the filters live in it rather than every filter the query named. A held bound is one the walk itself proved, so it never over-skips; it can be weaker than one proved afresh, which costs a slab that could have been skipped and never a match. The stream is unchanged, and the oracle suites are the proof: ShapedMatrix, RandomizedAgainstPostFilter at every slab width, RareTermsSpanSlabs, FetchesOnlyTrueCandidates and SlabStepperSkipsCandidateFreeSlabs — which pins the opened slabs themselves — all pass unchanged under -race. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/stores/event/slab_match.go | 91 +++++++++++++++---- 1 file changed, 75 insertions(+), 16 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go index bdebb700b..d4266333e 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go @@ -181,6 +181,16 @@ func (f *slabFilter) eval(lo, hi uint32) *roaring.Bitmap { // -1 for none, and read the bitmap without writing it — they are called on // snapshots shared with every other reader. roaring_contract_test.go pins // both properties against the pinned version. +// +// A bound the walk has proved is held rather than proved again at every slab, +// and it decides which filters a slab is evaluated for: see slabStepper.bounds. + +// boundRetired marks a filter that has proved it can match nothing more in the +// window: its terms ran out ahead of the cursor, which only ever moves further +// from them, so the walk drops it for good rather than asking again at every +// remaining slab. Below every id, so the test for a bound outside the current +// slab covers it too. +const boundRetired = int64(-1) // nextBound is the group's bound: the smallest id at or above pos that any of // its terms holds. ok is false when none does, which proves the group, and so @@ -259,6 +269,25 @@ type slabStepper struct { window IDRange desc bool + // bounds holds what the walk has proved about each filter, parallel to + // filters: the id it proved its next candidate lies at or past — at or + // before, descending — or boundRetired once it proved it has none left. + // + // The invariant that makes holding one sound: a filter's bound is monotone + // in the walk direction, so one proved at a cursor position still proves + // the same emptiness at every position the cursor reaches up to it. A held + // bound can be weaker than one proved afresh, because a filter's groups + // can pull apart as the cursor advances; weaker costs a slab the walk + // could have skipped and never a match. + // + // So a bound is proved again only once the cursor has reached it, and a + // bound past the current slab excuses its filter from being evaluated + // there at all — the work a slab costs is the filters live in it rather + // than every filter the query named. Seeding each bound at the cursor's + // own start, where it proves nothing, is what makes the first step prove + // them all. + bounds []int64 + // cursor is the next unevaluated boundary: the inclusive low bound // ascending, the exclusive high bound descending. cursor uint32 @@ -283,50 +312,72 @@ func newSlabStepper( } else { s.cursor = window.Start } + s.bounds = make([]int64, len(s.filters)) + for i := range s.bounds { + s.bounds[i] = int64(s.cursor) + } return s } // seekAsc is the lowest position at or above pos that any filter can still // match at, and false when none can inside the window: a candidate belongs to // some filter, so the smallest of their bounds holds for the union. Filters that -// are done are dropped from the minimum rather than ending the walk, since a -// live one may still match. +// are done are retired rather than ending the walk, since a live one may still +// match. +// +// Only a filter whose stored bound pos has reached is asked again; the rest +// answer from the bound they proved earlier, which pos has not yet passed. func (s *slabStepper) seekAsc(pos uint32) (uint32, bool) { - var best uint32 + var best int64 found := false for i := range s.filters { - b, ok := s.filters[i].nextBound(pos) - if !ok { + if s.bounds[i] == boundRetired { continue } - if !found || b < best { - best, found = b, true + if s.bounds[i] <= int64(pos) { + b, ok := s.filters[i].nextBound(pos) + if !ok { + s.bounds[i] = boundRetired + continue + } + s.bounds[i] = int64(b) + } + if !found || s.bounds[i] < best { + best, found = s.bounds[i], true } } - if !found || best >= s.window.End { + if !found || best >= int64(s.window.End) { return 0, false } - return best, true + return uint32(best), true } // seekDesc is seekAsc mirrored: the largest of the filters' bounds at or below // hi-1, returned as the exclusive high bound the walk resumes at. func (s *slabStepper) seekDesc(hi uint32) (uint32, bool) { - var best uint32 + pos := hi - 1 + var best int64 found := false for i := range s.filters { - b, ok := s.filters[i].prevBound(hi - 1) - if !ok { + if s.bounds[i] == boundRetired { continue } - if !found || b > best { - best, found = b, true + if s.bounds[i] >= int64(pos) { + b, ok := s.filters[i].prevBound(pos) + if !ok { + s.bounds[i] = boundRetired + continue + } + s.bounds[i] = int64(b) + } + if !found || s.bounds[i] > best { + best, found = s.bounds[i], true } } - if !found || best < s.window.Start { + if !found || best < int64(s.window.Start) { return 0, false } - return best + 1, true + return uint32(best) + 1, true } // nextBounds returns the next slab's [lo, hi) clipped to the window, walking @@ -379,12 +430,20 @@ func (s *slabStepper) nextBounds() (uint32, uint32, bool) { // evalSlab is the union across filters of their per-slab results, or nil when // the slab holds nothing. // +// Only the filters this slab is about are evaluated. The bound the seek just +// proved for each one says where its next candidate can be, and one lying +// outside [lo, hi) — a retired filter's included — already proves the filter +// matches nothing here, which eval would spend a bitmap to rediscover. +// // Every per-filter result is a bitmap eval built for this call and nobody else // holds, so the union runs in place into the first of them rather than // allocating a separate answer to copy them all into. func (s *slabStepper) evalSlab(lo, hi uint32) *roaring.Bitmap { var acc *roaring.Bitmap for i := range s.filters { + if b := s.bounds[i]; b < int64(lo) || b >= int64(hi) { + continue + } bm := s.filters[i].eval(lo, hi) switch { case bm == nil: From 27fdd8020a423496a8d2ff90773a2f01704e98b6 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 17:08:02 +0000 Subject: [PATCH 30/41] observability: export the packfile pool cap-skip counter MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The counter the previous commit added had no way out of the process. It is now the fourth CounterFunc in NewPrometheusMetrics, beside rocksdb.DeferredCloseOps and ledger.MissingPackOpens, and it belongs to their family exactly: tallied where its condition is detected, below any metrics plumbing, and healthy only at zero — so it takes the same "rate > 0" alert rule the other three do. pools.go's "the metrics exporter reads it via PoolCapSkips" is true as of this commit. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/observability/observability.go | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/observability/observability.go b/cmd/stellar-rpc/internal/rpcv2/observability/observability.go index 697972728..88ef02172 100644 --- a/cmd/stellar-rpc/internal/rpcv2/observability/observability.go +++ b/cmd/stellar-rpc/internal/rpcv2/observability/observability.go @@ -5,6 +5,7 @@ import ( "github.com/prometheus/client_golang/prometheus" + "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/packfile" "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/query" "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/rocksdb" "github.com/stellar/stellar-rpc/cmd/stellar-rpc/internal/rpcv2/stores/ledger" @@ -153,10 +154,10 @@ func NewPrometheusMetrics(registry *prometheus.Registry, namespace string) *Prom }, []string{"phase"}), } - // Three serving-invariant counters are tallied where their condition is - // detected — deep in the store and routing packages, below any metrics - // plumbing — as package-level atomics. CounterFuncs export them; each - // should flatline at zero, so one alert rule per family is "rate > 0". + // Four serving-invariant counters are tallied where their condition is + // detected — deep in the store, routing and packfile packages, below any + // metrics plumbing — as package-level atomics. CounterFuncs export them; + // each should flatline at zero, so one alert rule per family is "rate > 0". counterFunc := func(name, help string, read func() uint64) prometheus.CounterFunc { return prometheus.NewCounterFunc(prometheus.CounterOpts{ Namespace: namespace, Subsystem: subsystem, Name: name, Help: help, @@ -180,6 +181,10 @@ func NewPrometheusMetrics(registry *prometheus.Registry, namespace string) *Prom "cold ledger packs whose file was gone on first read "+ "(routing only opens packs the catalog snapshot holds; any count is an alarm)", ledger.MissingPackOpens), + counterFunc("pooled_buffer_cap_skips_total", + "pooled packfile buffers dropped for exceeding a pool's capacity cap "+ + "(the pool drains and every open allocates afresh; any count means a cap wants raising)", + packfile.PoolCapSkips), prometheus.NewGaugeFunc(prometheus.GaugeOpts{ Namespace: namespace, Subsystem: subsystem, Name: "open_snapshots", Help: "RocksDB snapshots currently held, across all stores " + From 9f1db00db22d3dc995c9825371b9ddd8f9bc6d73 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 17:41:14 +0000 Subject: [PATCH 31/41] events: one buffer for the hot fetch's key list MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit FetchEvents built its RocksDB key list with encodeDataKey per event ID. That helper returns a slice of a stack array, so the array escapes through the return and the compiler heap-allocates it — "hot_store.go:251:26: moved to heap: key" under -gcflags=-m — once per ID. A limit=1000 page paid a thousand four-byte allocations just to name the rows it was about to read. encodeDataKeys carves the whole list out of one buffer instead: two allocations for the batch rather than one per ID. Each key is a full slice expression over its own window, so the keys are byte-identical to encodeDataKey's and appending to one cannot reach the next. They are handed straight to BatchedMultiGetCF, which copies every key into C memory and frees that copy before it returns, so nothing of ours outlives the call. encodeDataKey stays for the five single-key callers, where its allocation is the only one there is. The measured cost of a 512-ID fetch drops from 2057 allocations to 1546. The 1546 are not ours: grocksdb spends three per key inside the batched get — one *PinnableSlice, plus the two size out-params PinnableSlice.Data escapes because BatchMultiGet calls Data twice per key. So the new budget test asserts three allocations per ID rather than a flat constant, and names where the three come from; the fourth is what it catches. Reverting this commit trips it (2057 > 1600). Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/rocksdb/rocksdb.go | 5 ++ .../internal/rpcv2/stores/event/hot_store.go | 31 ++++++- .../rpcv2/stores/event/hot_store_test.go | 83 +++++++++++++++++++ 3 files changed, 115 insertions(+), 4 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go b/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go index 1b9acbe6b..b3dedbea9 100644 --- a/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go +++ b/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go @@ -244,6 +244,11 @@ func (s *Store) GetPinned(cf string, key []byte, fn func(value []byte) error) (b // merge adjacent SST seeks across the input set. Behavior on // unsorted input is undefined per RocksDB semantics. // +// keys are not retained past the call: grocksdb copies each into C +// memory and frees that copy before the batched get returns. A caller +// may therefore carve the whole list out of one buffer rather than +// allocating a key at a time (see event.encodeDataKeys). +// // Uses async_io read options so the kernel can issue overlapping // I/Os under the hood (notable on EBS / high random-latency // storage). The batched call is a single CGO crossing; callers diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go index eae2a9c4b..07291ea63 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go @@ -246,10 +246,7 @@ func (h *HotStore) FetchEvents(ctx context.Context, eventIDs []uint32) ([]Payloa return nil, err } - keys := make([][]byte, len(eventIDs)) - for i, id := range eventIDs { - keys[i] = encodeDataKey(id) - } + keys := encodeDataKeys(eventIDs) values, err := h.chunkStore.BatchMultiGet(DataCF, keys) if err != nil { return nil, fmt.Errorf("events: batch fetch from chunk %s: %w", h.chunkID, err) @@ -689,6 +686,32 @@ func encodeDataKey(eventID uint32) []byte { return key[:] } +// encodeDataKeys encodes every id into ONE backing buffer and returns +// per-id sub-slices of it, in input order: two allocations for the +// whole batch (the buffer plus the slice headers) instead of one per +// id. encodeDataKey cannot serve this loop — its array escapes through +// the returned slice, so each 4-byte key is heap-allocated +// ("moved to heap: key" under -gcflags=-m), and a limit=1000 page pays +// 1000 of them just to name its rows. +// +// The returned slices alias one array and are read-only. A consumer +// must not retain them past the call it passes them to. +// rocksdb.Store.BatchMultiGet qualifies: grocksdb copies every key into +// C memory and frees that copy before the batched get returns, so +// nothing outlives the call. +func encodeDataKeys(eventIDs []uint32) [][]byte { + buf := make([]byte, dataKeyLen*len(eventIDs)) + keys := make([][]byte, len(eventIDs)) + for i, id := range eventIDs { + lo := i * dataKeyLen + hi := lo + dataKeyLen + key := buf[lo:hi:hi] + binary.BigEndian.PutUint32(key, id) + keys[i] = key + } + return keys +} + func encodeIndexKey(term TermKey, eventID uint32) []byte { var key [indexKeyLen]byte copy(key[:16], term[:]) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go index 2a2a2180d..975edb61d 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go @@ -346,6 +346,89 @@ func TestHotStore_FetchEventsRejectsUnsortedInput(t *testing.T) { require.ErrorIs(t, err, ErrUnsortedEventIDs, "duplicate input must error") } +// fetchEventsPerIDAllocBudget is the number of heap allocations +// FetchEvents may spend per event ID. All three are grocksdb's, spent +// inside the batched multi-get: one *PinnableSlice per key, plus the +// two size out-params PinnableSlice.Data escapes to the heap (BatchMultiGet +// calls Data twice per key, once to size the arena and once to fill it). +// Our own per-batch work — the key list, the value arena, the Payload +// slice — is a fixed handful of allocations regardless of batch size. +// +// If a grocksdb bump moves this number, raise it deliberately after +// checking where the new allocation comes from; do not widen it to +// absorb a regression on our side of the boundary. +const fetchEventsPerIDAllocBudget = 3 + +// TestHotStore_FetchEventsAllocationBudget pins that FetchEvents spends +// no per-ID allocation of its own. The regression it catches is building +// the RocksDB key list with encodeDataKey per ID: that helper returns a +// slice of a stack array, so every key escapes to the heap +// ("moved to heap: key" under -gcflags=-m) and a limit=1000 page pays +// 1000 extra allocations just to name its rows. encodeDataKeys carves +// them all out of one buffer instead, which drops the measured cost from +// 4 allocations per ID to the 3 grocksdb charges. +func TestHotStore_FetchEventsAllocationBudget(t *testing.T) { + const chunkID = chunk.ID(0) + const n = 512 + h := openHotStoreForTest(t, chunkID) + + payloads := make([]Payload, n) + for i := range n { + p, _ := makePayload(fmt.Sprintf("evt-%03d", i)) + payloads[i] = p + } + require.NoError(t, ingestLedgerEvents(h.store, 2, payloads)) + + ids := make([]uint32, n) + for i := range n { + ids[i] = uint32(i) + } + ctx := context.Background() + // Warm the block cache first: a cold read allocates in RocksDB's own + // Go-side plumbing on the way to the SSTs and would skew run one. + _, err := h.store.FetchEvents(ctx, ids) + require.NoError(t, err) + + allocs := testing.AllocsPerRun(20, func() { + if _, err := h.store.FetchEvents(ctx, ids); err != nil { + t.Error(err) + } + }) + + // Per-ID budget plus generous room for the fixed per-batch handful. + const fixedAllocSlack = 64 + budget := float64(n*fetchEventsPerIDAllocBudget + fixedAllocSlack) + assert.LessOrEqual(t, allocs, budget, + "FetchEvents of %d IDs allocated %.0f times (budget %.0f): a per-ID "+ + "allocation crept back into the fetch path", n, allocs, budget) +} + +// TestEncodeDataKeys pins encodeDataKeys itself: the keys are +// byte-identical to encodeDataKey's, they are distinct windows onto one +// buffer, and the allocation count does not grow with the batch — the +// property the FetchEvents budget above rests on. +func TestEncodeDataKeys(t *testing.T) { + ids := []uint32{0, 1, 7, 1 << 20, ^uint32(0)} + keys := encodeDataKeys(ids) + require.Len(t, keys, len(ids)) + for i, id := range ids { + assert.Equal(t, encodeDataKey(id), keys[i], "key %d", i) + assert.Len(t, keys[i], dataKeyLen, "key %d", i) + // Full slice expression: appending to one key must not scribble + // over the next one. + assert.Equal(t, dataKeyLen, cap(keys[i]), "key %d capacity", i) + } + + // Same allocation count two orders of magnitude apart. + small := testing.AllocsPerRun(100, func() { _ = encodeDataKeys(make([]uint32, 8)) }) + large := testing.AllocsPerRun(100, func() { _ = encodeDataKeys(make([]uint32, 4096)) }) + // One for the key buffer, one for the slice headers, one for the + // make([]uint32) the closure itself does. + const encodeDataKeysAllocs = 3 + assert.LessOrEqual(t, small, float64(encodeDataKeysAllocs), "8 IDs") + assert.LessOrEqual(t, large, float64(encodeDataKeysAllocs), "4096 IDs") +} + func TestHotStore_AllStreamsInEventIDOrder(t *testing.T) { const chunkID = chunk.ID(0) h := openHotStoreForTest(t, chunkID) From a0691ad3a75359b59dc2e337e26e1b3094faee9f Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 17:43:09 +0000 Subject: [PATCH 32/41] rocksdb: build the batched read's options once, at open MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every BatchMultiGet built a ReadOptions, set async IO on it and destroyed it — three CGO crossings per batch to configure a read that is itself one crossing. The comment explained why it could not just use s.ro: async IO on the shared options would reach every concurrent Get, GetPinned and iterator on the Store, which is not what the batched path is asking for. Both facts hold with a second options object. roAsync is built beside ro at open with the one setting that separates them, destroyed beside it in teardown, and never mutated in between — so it is as safe to share across concurrent batched reads as ro already is across concurrent Gets, and the shared ro still carries nothing the batched path wanted. The settings are identical to what the per-batch options carried: NewDefaultReadOptions plus SetAsyncIO(true), nothing else. No read changes shape. The same options reach BatchedMultiGetCF with the same values; what goes away is rebuilding them. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/rocksdb/rocksdb.go | 32 ++++++++++++------- 1 file changed, 20 insertions(+), 12 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go b/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go index b3dedbea9..59d4176dd 100644 --- a/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go +++ b/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go @@ -130,6 +130,14 @@ type Store struct { ro *grocksdb.ReadOptions wo *grocksdb.WriteOptions + // roAsync is ro plus async IO, for BatchMultiGet. A second options + // object rather than a flag on ro: async IO belongs to the batched + // read alone, and setting it on the shared ro would hand it to every + // concurrent Get, GetPinned and iterator on this Store. Built once at + // open and never mutated after, so a batched read crosses into C for + // the read itself and not to configure it. + roAsync *grocksdb.ReadOptions + // cache is the block cache shared across every CF in this store, // created in applyTuning when BlockCacheMB is set. bbtos are the // per-CF block-based-table options (one per CF, carrying the pinned @@ -249,11 +257,12 @@ func (s *Store) GetPinned(cf string, key []byte, fn func(value []byte) error) (b // may therefore carve the whole list out of one buffer rather than // allocating a key at a time (see event.encodeDataKeys). // -// Uses async_io read options so the kernel can issue overlapping -// I/Os under the hood (notable on EBS / high random-latency -// storage). The batched call is a single CGO crossing; callers -// needing cancellation between individual key reads should not use -// this API — split into multiple calls or use Get in a loop. +// Reads through the store's async_io read options (roAsync, built at +// open) so the kernel can issue overlapping I/Os under the hood +// (notable on EBS / high random-latency storage). The batched call is +// a single CGO crossing; callers needing cancellation between +// individual key reads should not use this API — split into multiple +// calls or use Get in a loop. func (s *Store) BatchMultiGet(cf string, keys [][]byte) ([][]byte, error) { if len(keys) == 0 { return nil, nil @@ -268,13 +277,7 @@ func (s *Store) BatchMultiGet(cf string, keys [][]byte) ([][]byte, error) { return nil, err } - // Fresh ReadOptions: mutating s.ro would surface async_io to - // every concurrent reader on this Store. - ro := grocksdb.NewDefaultReadOptions() - ro.SetAsyncIO(true) - defer ro.Destroy() - - pinned, err := s.db.BatchedMultiGetCF(ro, cfh, true /* sortedInput */, keys...) + pinned, err := s.db.BatchedMultiGetCF(s.roAsync, cfh, true /* sortedInput */, keys...) if err != nil { return nil, fmt.Errorf("rocksdb: batched multi get on %q: %w", cf, err) } @@ -555,6 +558,7 @@ func (s *Store) teardownLocked() { cfh.Destroy() } s.ro.Destroy() + s.roAsync.Destroy() s.wo.Destroy() s.db.Close() s.opts.Destroy() @@ -771,8 +775,12 @@ func (s *Store) constructAndOpen() error { s.cfOpts = cfOpts s.cfHandles = cfMap s.ro = grocksdb.NewDefaultReadOptions() + s.roAsync = grocksdb.NewDefaultReadOptions() s.wo = grocksdb.NewDefaultWriteOptions() + // The one setting that separates roAsync from ro. See the field. + s.roAsync.SetAsyncIO(true) + // WAL on + per-write Sync on — non-negotiable across every // rpcv2 store, so pinned here on the shared wo rather // than exposed via Tuning. The ingestion contract From 6789545b1820ae972b6cad5eed76ac462e0ff825 Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 17:55:52 +0000 Subject: [PATCH 33/41] rocksdb: read each pinned value once in the batched get Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- .../internal/rpcv2/rocksdb/rocksdb.go | 22 ++++++++++++------- .../rpcv2/stores/event/hot_store_test.go | 15 +++++++------ 2 files changed, 22 insertions(+), 15 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go b/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go index 59d4176dd..7b0aece42 100644 --- a/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go +++ b/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go @@ -284,20 +284,26 @@ func (s *Store) BatchMultiGet(cf string, keys [][]byte) ([][]byte, error) { defer pinned.Destroy() // Copy out of the pinned cache pages, which Destroy invalidates, through - // one arena rather than a clone per value. The returned slices share one - // backing array: they are read-only, and retaining one retains the batch. + // one arena rather than a clone per value. Each pinned value is read ONCE: + // Data heap-allocates a cgo out-param per call, so the sizing pass keeps + // the C-backed slice (valid until pinned.Destroy) and the copy pass reads + // from it. The returned slices share one backing array: they are + // read-only, and retaining one retains the batch. + results := make([][]byte, len(keys)) total := 0 - for _, p := range pinned { - total += len(p.Data()) + for i, p := range pinned { + if p.Exists() { + results[i] = p.Data() + total += len(results[i]) + } } arena := make([]byte, 0, total) - results := make([][]byte, len(keys)) - for i, p := range pinned { - if !p.Exists() { + for i, v := range results { + if v == nil { continue } n := len(arena) - arena = append(arena, p.Data()...) + arena = append(arena, v...) results[i] = arena[n:len(arena):len(arena)] } return results, nil diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go index 975edb61d..be438413b 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go @@ -347,17 +347,18 @@ func TestHotStore_FetchEventsRejectsUnsortedInput(t *testing.T) { } // fetchEventsPerIDAllocBudget is the number of heap allocations -// FetchEvents may spend per event ID. All three are grocksdb's, spent -// inside the batched multi-get: one *PinnableSlice per key, plus the -// two size out-params PinnableSlice.Data escapes to the heap (BatchMultiGet -// calls Data twice per key, once to size the arena and once to fill it). -// Our own per-batch work — the key list, the value arena, the Payload -// slice — is a fixed handful of allocations regardless of batch size. +// FetchEvents may spend per event ID. Both are grocksdb's, spent inside +// the batched multi-get: one *PinnableSlice per key, plus the size +// out-param the single PinnableSlice.Data call per key escapes to the +// heap (BatchMultiGet reads each pinned value once and copies from the +// kept C-backed slice). Our own per-batch work — the key list, the +// value arena, the Payload slice — is a fixed handful of allocations +// regardless of batch size. // // If a grocksdb bump moves this number, raise it deliberately after // checking where the new allocation comes from; do not widen it to // absorb a regression on our side of the boundary. -const fetchEventsPerIDAllocBudget = 3 +const fetchEventsPerIDAllocBudget = 2 // TestHotStore_FetchEventsAllocationBudget pins that FetchEvents spends // no per-ID allocation of its own. The regression it catches is building From 1ba88a1b6d141b32ce2887a5ce47fd32e81e53cb Mon Sep 17 00:00:00 2001 From: Tamir Sen Date: Wed, 9 Sep 2026 18:24:40 +0000 Subject: [PATCH 34/41] events: the fetch hint is a validated page size, honored in full Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01Y7opQUW9UZPKzE6B3tz9b5 --- cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go | 6 +++--- .../internal/rpcv2/stores/event/match.go | 15 ++++++++------- .../internal/rpcv2/stores/event/match_test.go | 10 ++-------- 3 files changed, 13 insertions(+), 18 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go b/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go index 7b0aece42..4ce963506 100644 --- a/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go +++ b/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go @@ -298,12 +298,12 @@ func (s *Store) BatchMultiGet(cf string, keys [][]byte) ([][]byte, error) { } } arena := make([]byte, 0, total) - for i, v := range results { - if v == nil { + for i := range results { + if !pinned[i].Exists() { continue } n := len(arena) - arena = append(arena, v...) + arena = append(arena, results[i]...) results[i] = arena[n:len(arena):len(arena)] } return results, nil diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go index 8f30fa0bd..c3273ed30 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go @@ -220,14 +220,15 @@ type Match struct { type termPlan [][]int // batchSizes resolves the first and following internal batch sizes from the -// caller's hint, capped at eight default batches so a wild hint cannot demand -// an unbounded fetch. Both are clamped positive: a zero step never advances, -// so a zero test seam would stall the stream. +// caller's hint. The hint is a validated page size — every handler clamps it +// to its protocol limit before it reaches Matches — so it is honored in full +// and a page arrives in one fetch. Both sizes are clamped positive: a zero +// step never advances, so a zero test seam would stall the stream. func batchSizes(hint int) (int, int) { rest := max(1, matchBatchSize) first := rest if hint > 0 { - first = min(hint, 8*rest) + first = hint } return first, rest } @@ -253,9 +254,9 @@ func batchSizes(hint int) (int, int) { // consumers never see or reason about resume state. // // firstBatch sizes the first internal fetch batch: a consumer that will stop -// after N matches passes N. Zero and negative hints use the default, and a -// positive one is honored up to eight default batches. The hint changes I/O -// counts only, never what the stream yields. +// after N matches passes N. Zero and negative hints use the default; a +// positive one is a validated page size and is honored in full. The hint +// changes I/O counts only, never what the stream yields. func Matches( ctx context.Context, r Reader, filters []Filter, window IDRange, descending bool, firstBatch int, diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go index dc0888de8..b6234fe35 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go @@ -1782,13 +1782,7 @@ func TestBatchSizes(t *testing.T) { require.Equal(t, 1000, first, "a page-sized hint is the first fetch size") require.Equal(t, matchBatchSize, rest) - first, rest = batchSizes(1 << 20) - require.Equal(t, 8*matchBatchSize, first, "oversized hints are capped") + first, rest = batchSizes(10_000) + require.Equal(t, 10_000, first, "a v1 page-sized hint is honored in full") require.Equal(t, matchBatchSize, rest) - - defer func(n int) { matchBatchSize = n }(matchBatchSize) - matchBatchSize = 7 - first, rest = batchSizes(1000) - require.Equal(t, 56, first, "the cap follows the seam") - require.Equal(t, 7, rest) } From 149449355c3940b64c621bcf6df71a04fcd433c0 Mon Sep 17 00:00:00 2001 From: tamirms Date: Wed, 9 Sep 2026 19:54:44 +0100 Subject: [PATCH 35/41] packfile: return the pooled offsets on every failed open decodeIndex took the offsets table from the pool and then dropped it on its final structural check, and doOpen ran its two trailer cross-checks after decodeIndex had succeeded, so those error paths dropped it too. A corrupt pack re-opened by every request drained the pool one table per open, up to 8 MiB each, with the cap-skip counter at zero. The mismatch path now puts the table back, and the trailer checks run before the decode, so the Reader is the only owner a successful open leaves behind. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01M23ZvEkyrFjyUUbm7zLwob --- cmd/stellar-rpc/internal/rpcv2/packfile/index.go | 1 + cmd/stellar-rpc/internal/rpcv2/packfile/reader.go | 12 +++++++----- 2 files changed, 8 insertions(+), 5 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/index.go b/cmd/stellar-rpc/internal/rpcv2/packfile/index.go index d3983f318..afdcf81a8 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/index.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/index.go @@ -116,6 +116,7 @@ func decodeIndex(buf []byte, recordCount int, indexSize int, indexBase int64) ([ // Structural sanity check: running sum must arrive at indexBase. if offset != indexBase { + putOffsets(offsets) return nil, fmt.Errorf("%w: final offset %d != indexBase %d", ErrCorrupt, offset, indexBase) } diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go b/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go index 4079a2877..282b8cf14 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go @@ -311,11 +311,6 @@ func doOpen(path string) openResult { ErrChecksum, trailer.AppDataCRC, computed)} } - offsets, err := decodeIndex(indexBuf, recordCount, indexSize, indexBase) - if err != nil { - return openResult{err: err} - } - // Not gated on recordCount: a trailer claiming items but no itemsPerRecord // would otherwise pass Open and index past the offsets slice on first read. if itemsPerRecord <= 0 && (recordCount > 0 || totalItems > 0) { @@ -336,6 +331,13 @@ func doOpen(path string) openResult { } } + // Last, so that no error path below holds a pooled offsets table: on + // success the Reader owns it until Close recycles it. + offsets, err := decodeIndex(indexBuf, recordCount, indexSize, indexBase) + if err != nil { + return openResult{err: err} + } + // Empty packfiles may legitimately have itemsPerRecord==0 on disk; // default the *internal* int to 1 so modulo math is well-defined. The // Trailer view keeps the on-disk value verbatim. From a84845b808f0cbd3a392a4ecb5a40a30dc70f5c9 Mon Sep 17 00:00:00 2001 From: tamirms Date: Wed, 9 Sep 2026 19:54:45 +0100 Subject: [PATCH 36/41] events: compare whole payloads in the fan-out test; fix two stale test comments The fan-out regression test compared only each event's data symbol, so a torn copy in the header or elsewhere in the XDR could pass. It now compares the whole Payload. The allocation-budget comment still counted three grocksdb allocations per ID against a budget of two, and the batchSizes comment still described the eight-batch cap that batchSizes no longer applies. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01M23ZvEkyrFjyUUbm7zLwob --- .../rpcv2/stores/event/cold_reader_fanout_test.go | 6 +++--- .../internal/rpcv2/stores/event/hot_store_test.go | 9 +++------ .../internal/rpcv2/stores/event/match_test.go | 7 +++---- 3 files changed, 9 insertions(+), 13 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader_fanout_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader_fanout_test.go index 9816aa11c..bbe18b157 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader_fanout_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader_fanout_test.go @@ -12,8 +12,8 @@ import ( // Pins the payload arena against the reader's own fan-out: with Concurrency // above one, ReadItems calls the callback from one goroutine per batch, so an // unsynchronized append into the shared arena corrupts payloads or segfaults. -// Every payload is checked, so a torn copy fails even when -race does not -// schedule the overlap. +// Every payload is compared whole, so a torn copy fails even when -race does +// not schedule the overlap. func TestColdReader_FetchEventsFansOutSafely(t *testing.T) { const ( chunkID = chunk.ID(0) @@ -33,6 +33,6 @@ func TestColdReader_FetchEventsFansOutSafely(t *testing.T) { require.NoError(t, err) require.Len(t, got, len(ids)) for i := range ids { - require.Equal(t, dataSym(t, payloads[i]), dataSym(t, got[i]), "payload %d", i) + require.Equal(t, payloads[i], got[i], "payload %d", i) } } diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go index be438413b..bf3fe31d1 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go @@ -362,12 +362,9 @@ const fetchEventsPerIDAllocBudget = 2 // TestHotStore_FetchEventsAllocationBudget pins that FetchEvents spends // no per-ID allocation of its own. The regression it catches is building -// the RocksDB key list with encodeDataKey per ID: that helper returns a -// slice of a stack array, so every key escapes to the heap -// ("moved to heap: key" under -gcflags=-m) and a limit=1000 page pays -// 1000 extra allocations just to name its rows. encodeDataKeys carves -// them all out of one buffer instead, which drops the measured cost from -// 4 allocations per ID to the 3 grocksdb charges. +// the RocksDB key list one heap-allocated key at a time, which cost a +// limit=1000 page a thousand extra allocations; encodeDataKeys carves +// them out of one buffer instead, leaving only grocksdb's two per ID. func TestHotStore_FetchEventsAllocationBudget(t *testing.T) { const chunkID = chunk.ID(0) const n = 512 diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go index b6234fe35..5ffec2445 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go @@ -1761,10 +1761,9 @@ func TestCountDistinctTerms(t *testing.T) { }), "topic-count buckets are not value terms and are not counted") } -// The first-batch hint contract: a positive hint sizes the first fetch, a wild -// one is capped at eight default batches, and later batches use the default. -// The cap scales with matchBatchSize, so a test-shrunk batch cannot be blown -// past by a hint. +// The first-batch hint contract: a positive hint sizes the first fetch in +// full, since every caller passes a validated page size, and later batches +// use the default. func TestBatchSizes(t *testing.T) { first, rest := batchSizes(0) require.Equal(t, matchBatchSize, first) From 32f544ac2046109f9e8789a2bb6f6786ee187f6a Mon Sep 17 00:00:00 2001 From: tamirms Date: Wed, 9 Sep 2026 20:00:55 +0100 Subject: [PATCH 37/41] rpcv2: trim the comments this branch added Rewrite the comments the branch added so they state contracts and hazards in plain sentences: the measurements, the alternatives not taken and the narration go, and what stays reads without the diff open. slab_match.go's header and bound helpers carry most of the cut; the pool, arena, packfile, rocksdb and store comments follow the same rule. Get's doc now lists the read-only operations the engine actually uses on shared bitmaps. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01M23ZvEkyrFjyUUbm7zLwob --- .../internal/rpcv2/packfile/index.go | 5 +- .../internal/rpcv2/packfile/pools.go | 66 ++--- .../internal/rpcv2/packfile/reader.go | 28 +- .../internal/rpcv2/query/resolve.go | 21 +- .../internal/rpcv2/rocksdb/rocksdb.go | 36 ++- .../internal/rpcv2/stores/event/arena.go | 15 +- .../rpcv2/stores/event/cold_format.go | 11 +- .../rpcv2/stores/event/cold_reader.go | 9 +- .../rpcv2/stores/event/concurrent_bitmaps.go | 15 +- .../internal/rpcv2/stores/event/hot_store.go | 42 ++- .../internal/rpcv2/stores/event/match.go | 38 ++- .../internal/rpcv2/stores/event/payload.go | 11 +- .../internal/rpcv2/stores/event/slab_match.go | 255 ++++++------------ go.mod | 9 +- 14 files changed, 208 insertions(+), 353 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/index.go b/cmd/stellar-rpc/internal/rpcv2/packfile/index.go index afdcf81a8..aa981faf2 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/index.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/index.go @@ -103,9 +103,8 @@ func decodeIndex(buf []byte, recordCount int, indexSize int, indexBase int64) ([ return nil, fmt.Errorf("%w: index has %d unconsumed bytes after decoding all groups", ErrCorrupt, pos) } - // Forward prefix-sum to build absolute offsets from deltas. The pooled - // array is fully overwritten: the entries by the loop, the sentinel by - // the assignment after it. + // Forward prefix-sum from deltas to absolute offsets. Every entry of the + // pooled table is overwritten below, the sentinel included. offsets := getOffsets(recordCount + 1) offset := int64(0) diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go b/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go index 7ac799a0a..bc1ca6b4d 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/pools.go @@ -1,39 +1,27 @@ package packfile -// Open-path allocation pools. The v2 read path opens cold packfiles per -// request, and every open allocates the decoded offset index, its FOR-decode -// scratch and the open-time read buffers. These pools recycle the backing -// memory only: contents are always fully rewritten before use, and every -// pooled buffer is either dead before its function returns or reader-private -// until Close hands it back. +// Pools for the open path. Cold packfiles are opened per request, and every +// open decodes an offset table, with FOR-decode scratch, out of read buffers; +// these pools recycle that memory. Contents are fully rewritten before use, +// and a pooled buffer is either dead before its function returns or +// reader-private until Close returns it. // -// Puts are capacity-capped so one pathological file cannot pin an arbitrarily -// large array in a pool slot; larger buffers fall to the garbage collector. -// Every cap must therefore sit above what a legitimate packfile needs, because -// crossing one is otherwise silent: the Put is skipped, the pool drains, and -// every open allocates afresh — still correct, but without the allocation win -// these pools exist for. capSkips counts those skipped Puts so the drain is -// visible from outside the process instead. +// Puts are capped so one pathological file cannot pin a huge array. A cap +// must sit above what a legitimate packfile needs: crossing it skips the Put, +// the pool drains, and every open allocates afresh. capSkips counts those +// skips so the drain is visible. import ( "sync" "sync/atomic" ) -// maxPooledOffsets caps the decoded offset table (recordCount+1 int64s, so an -// 8 MiB array at the cap) a Put may retain. What it has to exceed is fixed by -// chunk geometry, which this package cannot reach — packfile is a container -// format with no domain dependencies — so the bound is a constant carrying -// headroom rather than a derived one: -// -// - the ledger cold pack stores one ledger per record, so its table is -// chunk.LedgersPerChunk+1 entries: 10,001; -// - events.pack and index.pack store 128 items per record, so 1<<20 entries -// covers ~134M events, or ~134M distinct index terms, within a single -// chunk — against the ~600K terms a production chunk carries today. -// -// Raise it if chunk geometry grows past that; see the file comment for what -// happens silently if it isn't raised. +// maxPooledOffsets caps the decoded offset table (recordCount+1 int64s) a Put +// may retain. Chunk geometry, which this package cannot import, sets what it +// must exceed: the ledger pack holds one ledger per record, 10,001 entries; +// events.pack and index.pack hold 128 items per record, so 1<<20 entries +// covers about 134M events or index terms per chunk, against roughly 600K +// terms in a production chunk today. Raise it if the geometry grows. const maxPooledOffsets = 1 << 20 // entries (8 MiB backing array) const ( @@ -50,28 +38,22 @@ var ( openBufPool sync.Pool // *[]byte ) -// capSkips counts the Puts dropped for exceeding their pool's cap, across all -// three pools. Each one is a buffer the pool did not get back, so a count that -// climbs with the open rate is the file comment's silent drain in progress. -// Process-wide by design — the metrics exporter reads it via PoolCapSkips. +// capSkips counts Puts dropped for exceeding a cap, across all three pools. +// A count that climbs with the open rate means the pools have stopped +// recycling. Exported through PoolCapSkips. // //nolint:gochecknoglobals // one tally across process-wide pools; read-only outside this file var capSkips atomic.Uint64 // PoolCapSkips returns the process-wide count of pooled buffers dropped for -// exceeding a capacity cap. -// -// Zero is the state the caps are chosen for, and the only healthy one: any -// count means some packfile crossed a cap, and a climbing count means the -// pools have stopped recycling. Raise the cap the file comment sizes against -// the geometry that grew. See capSkips. +// exceeding a capacity cap. Zero is the only healthy value; a climbing count +// means a cap wants raising. func PoolCapSkips() uint64 { return capSkips.Load() } -// A size miss hands the pooled buffer back before allocating: Get has already -// removed it from the pool, so returning it is the only thing that keeps a run -// of growing opens from draining the pool one buffer per open. That buffer is -// under its cap by construction — the pool held it — so it never counts as a -// cap skip. +// On a size miss the pooled buffer is returned before a larger one is +// allocated: Get already removed it, and returning it is what keeps a run of +// growing opens from draining the pool. It is under its cap, so it never +// counts as a skip. func getOffsets(n int) []int64 { if p, _ := offsetsPool.Get().(*[]int64); p != nil { diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go b/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go index 282b8cf14..3d63edb30 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/reader.go @@ -161,13 +161,11 @@ type Reader struct { waitOpen func() error // blocks until background open completes - // closed and inflight are the Close-vs-read handshake that lets Close - // recycle the pooled offsets. Reads increment inflight before checking - // closed; Close sets closed before reading inflight. Either the read - // sees closed and backs out, or Close sees the reader and leaves the - // offsets to the garbage collector. A read racing Close is a caller - // contract violation either way; this turns its worst case from memory - // reuse into a leak. + // closed and inflight let Close recycle the pooled offsets safely. A read + // increments inflight before checking closed; Close sets closed before + // reading inflight. So either the read backs out, or Close sees it and + // leaves the offsets to the garbage collector. A read racing Close is a + // caller bug either way; this makes its worst case a leak, not reuse. closed atomic.Bool inflight atomic.Int64 @@ -273,9 +271,8 @@ func doOpen(path string) openResult { var indexBuf []byte var appData []byte - // Both arms hand decodeIndex a view into a pooled buffer. The index bytes - // are dead once the offsets are built, so only appData, which the Reader - // keeps, is copied out. + // Both arms hand decodeIndex a view into a pooled buffer; only appData, + // which the Reader keeps, is copied out. if tailSize <= speculativeSize { // Index + appData are already inside the speculative read. tailStart := len(speculativeBuf) - int(tailSize) @@ -331,8 +328,8 @@ func doOpen(path string) openResult { } } - // Last, so that no error path below holds a pooled offsets table: on - // success the Reader owns it until Close recycles it. + // Decoded last: the table is pooled, and every check that could still + // fail has run, so on success the Reader owns it until Close. offsets, err := decodeIndex(indexBuf, recordCount, indexSize, indexBase) if err != nil { return openResult{err: err} @@ -803,10 +800,9 @@ func (r *Reader) Close() error { if r.file != nil { closeErr = r.file.Close() } - // Recycle only when no read is in flight: closed is already set, so - // no new read can begin, and a zero count proves no existing one - // holds the array. A read still in flight is a contract violation; - // leaving the array to the collector keeps it merely a leak. + // Recycle only when no read is in flight. closed is already set, so + // no new read can begin; a read still in flight is a caller bug, and + // leaving the array to the collector keeps that a leak. if r.inflight.Load() == 0 && r.offsets != nil { putOffsets(r.offsets) r.offsets = nil diff --git a/cmd/stellar-rpc/internal/rpcv2/query/resolve.go b/cmd/stellar-rpc/internal/rpcv2/query/resolve.go index 8f6d1543b..d738d9489 100644 --- a/cmd/stellar-rpc/internal/rpcv2/query/resolve.go +++ b/cmd/stellar-rpc/internal/rpcv2/query/resolve.go @@ -144,19 +144,14 @@ func (a *ReadView) resolveLedgers(c chunk.ID) (LedgerReader, func() error, error } } -// defaultColdEventReadConcurrency is the worker fan-out one cold events read -// gets over its packfiles. A page's payload fetch is hundreds of scattered -// records with no ordering between them, so serializing them only added their -// latencies together. The right value is a property of the storage the daemon -// reads through, never of the query the client asked for; this is the -// NVMe-measured choice. -// -// The fan-out is per request, so the worker count multiplies both goroutines -// and packfile's coalesced-read buffers by the number of cold pages in flight -// — the footprint is workers × in-flight cold pages, not workers alone. -// -// The value is a compiled-in constant: changing it is a code change, and a -// config knob gets added when a deployment on different storage needs one. +// defaultColdEventReadConcurrency is the worker fan-out for one cold events +// read over its packfiles. A page's payload fetch is hundreds of scattered +// records with no ordering between them, and serial reads add their latencies +// together. The value depends on the storage the daemon reads through, not on +// the query, and was measured on NVMe. Its footprint is workers times +// in-flight cold pages, in goroutines and in packfile's coalesced-read +// buffers. It is compiled in; a config knob can follow if a deployment needs +// a different value. const defaultColdEventReadConcurrency = 8 // Events resolves chunk c's event store as the common event.Reader the diff --git a/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go b/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go index 4ce963506..d4bd7ec4b 100644 --- a/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go +++ b/cmd/stellar-rpc/internal/rpcv2/rocksdb/rocksdb.go @@ -130,12 +130,10 @@ type Store struct { ro *grocksdb.ReadOptions wo *grocksdb.WriteOptions - // roAsync is ro plus async IO, for BatchMultiGet. A second options - // object rather than a flag on ro: async IO belongs to the batched - // read alone, and setting it on the shared ro would hand it to every - // concurrent Get, GetPinned and iterator on this Store. Built once at - // open and never mutated after, so a batched read crosses into C for - // the read itself and not to configure it. + // roAsync is ro plus async IO, used by BatchMultiGet. A separate object + // keeps the shared ro, and every Get and iterator on it, synchronous. + // Built at open and never mutated, so it is safe to share across + // concurrent batched reads. roAsync *grocksdb.ReadOptions // cache is the block cache shared across every CF in this store, @@ -253,16 +251,13 @@ func (s *Store) GetPinned(cf string, key []byte, fn func(value []byte) error) (b // unsorted input is undefined per RocksDB semantics. // // keys are not retained past the call: grocksdb copies each into C -// memory and frees that copy before the batched get returns. A caller -// may therefore carve the whole list out of one buffer rather than -// allocating a key at a time (see event.encodeDataKeys). +// memory and frees the copy before returning, so a caller may carve +// the whole list out of one buffer (see event.encodeDataKeys). // -// Reads through the store's async_io read options (roAsync, built at -// open) so the kernel can issue overlapping I/Os under the hood -// (notable on EBS / high random-latency storage). The batched call is -// a single CGO crossing; callers needing cancellation between -// individual key reads should not use this API — split into multiple -// calls or use Get in a loop. +// Reads use roAsync, the store's async_io read options, so the kernel +// can overlap the I/Os. The batched call is a single CGO crossing; +// callers needing cancellation between individual keys should split +// into multiple calls or use Get in a loop. func (s *Store) BatchMultiGet(cf string, keys [][]byte) ([][]byte, error) { if len(keys) == 0 { return nil, nil @@ -284,11 +279,10 @@ func (s *Store) BatchMultiGet(cf string, keys [][]byte) ([][]byte, error) { defer pinned.Destroy() // Copy out of the pinned cache pages, which Destroy invalidates, through - // one arena rather than a clone per value. Each pinned value is read ONCE: - // Data heap-allocates a cgo out-param per call, so the sizing pass keeps - // the C-backed slice (valid until pinned.Destroy) and the copy pass reads - // from it. The returned slices share one backing array: they are - // read-only, and retaining one retains the batch. + // one arena rather than a clone per value. Each pinned value is read once, + // since Data heap-allocates an out-param per call: the sizing pass keeps + // the C-backed slice, valid until Destroy, and the copy pass reads it. The + // returned slices share one backing array; retaining one retains the batch. results := make([][]byte, len(keys)) total := 0 for i, p := range pinned { @@ -784,7 +778,7 @@ func (s *Store) constructAndOpen() error { s.roAsync = grocksdb.NewDefaultReadOptions() s.wo = grocksdb.NewDefaultWriteOptions() - // The one setting that separates roAsync from ro. See the field. + // Async IO is the only difference from ro; see the field. s.roAsync.SetAsyncIO(true) // WAL on + per-write Sync on — non-negotiable across every diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go index df9ae8851..7861e273e 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/arena.go @@ -1,18 +1,17 @@ package event -// byteArena hands out stable copies of transient byte slices from large -// chunked allocations, so a fetch copying hundreds of small payloads costs a -// handful of allocations rather than one each. Chunks are only appended within -// capacity, so previously returned copies never move. Zero value is ready. +// byteArena hands out stable copies of transient byte slices, carved from +// larger chunks so a fetch copying hundreds of payloads costs a few +// allocations rather than one each. A chunk is only appended within its +// capacity, so returned copies never move. The zero value is ready. // -// NOT safe for concurrent use: copy appends to one buffer, so a caller whose -// copies come from several goroutines must serialize them. +// Not safe for concurrent use. type byteArena struct { buf []byte } -// The allocation unit ramps: the first chunk is small so a one-event page does -// not pay 64 KiB, and each subsequent chunk doubles up to arenaChunkSize. +// The first chunk is small so a one-event page does not pay 64 KiB; each +// later chunk doubles up to arenaChunkSize. const ( arenaFirstChunkSize = 4 << 10 arenaChunkSize = 64 << 10 diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go index 7aea0d4c9..071494e65 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_format.go @@ -434,12 +434,11 @@ func buildMPHF( // /index.hash produced by an earlier buildMPHF) for // query-time lookups. // -// The file is mmapped, never read whole. Pages fault in from the kernel page -// cache, which is keyed by the file rather than the mapping, so every reader -// of the same chunk shares them however short its own lifetime. A per-request -// open therefore costs a map/unmap pair and the pages its lookups touch, not a -// heap copy of a file whose size has no design bound: it scales with the -// chunk's distinct term count, which is caller-controlled on-chain data. +// The file is mmapped rather than read whole. Pages fault in from the kernel +// page cache, which every reader of the same chunk shares regardless of its +// own lifetime, so a per-request open costs a map and unmap plus the pages +// its lookups touch, not a copy of a file whose size scales with the chunk's +// term count. // // Close unmaps; callers must call it. func openMPHF(path string) (*mphf, error) { diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go index a835ac7d2..3edbf1933 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/cold_reader.go @@ -475,11 +475,10 @@ func (c *ColdReader) FetchEvents(ctx context.Context, eventIDs []uint32) ([]Payl positions[i] = int(id) } results := make([]Payload, len(eventIDs)) - // One arena per call, shared by every worker, because a call's payloads - // live and die together. The arena is a single appended buffer while - // ReadItems calls back from up to Concurrency goroutines, hence the lock. - // It covers the copy alone, so a fan-out still overlaps the read, the - // record decode and the Unmarshal. + // One arena per call, shared by every worker, since a call's payloads + // live and die together. ReadItems calls back from up to Concurrency + // goroutines and the arena is a single appended buffer, hence the lock. + // It covers the copy alone, so the read, decode and Unmarshal overlap. var ( arenaMu sync.Mutex arena byteArena diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go index e8795bf97..9f990cb1e 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go @@ -97,17 +97,16 @@ func NewConcurrentBitmapsFromBitmaps(b Bitmaps) *ConcurrentBitmaps { // Clone-the-input shortcut when there is only one input) // // Safe: any non-mutating read (Contains, GetCardinality, Iterator, -// ToArray, IsEmpty, Minimum, Maximum) plus roaring.And / FastAnd / +// NextValue, PreviousValue, ToArray, IsEmpty, Minimum, Maximum), +// passing it as an argument to AndAny, and roaring.And / FastAnd / // FastOr with 2+ inputs. // // A Get that starts after an AddTo returns sees that AddTo's IDs, and -// the pointer stays valid for as long as the caller holds it. -// -// The result is a point-in-time image either way, which is what lets a -// query hold it for a whole walk: a sparse term is copied out of the -// atomically published id list, and a dense term's snapshot is never -// mutated once published. Neither grows under a holder as ingest -// continues. +// the pointer stays valid for as long as the caller holds it. The +// result is a point-in-time image: a sparse term is copied out of the +// published id list, and a dense term's snapshot is never mutated once +// published. Neither grows as ingest continues, so a query can hold it +// for a whole walk. func (s *ConcurrentBitmaps) Get(key TermKey) (*roaring.Bitmap, error) { s.rwmu.RLock() p := s.terms[key] diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go index 07291ea63..771ee0a7f 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store.go @@ -29,10 +29,9 @@ const ( // // - DataCF holds XDR-encoded event payloads: compressible (zstd // typically 2-3× on XDR) and read in batches via -// BatchedMultiGetCF. The block is the decompression unit of a point -// read, so its size trades compression context against per-miss work: -// getEvents fetches scattered ~250B events, and a 32 KiB block made -// every cache miss decompress ~128 of them to serve one. +// BatchedMultiGetCF. A point read decompresses one block to serve +// one ~250B event, so a 32 KiB block paid for ~128 events per cache +// miss. // - IndexCF stores 20-byte (term_hash || event_id) keys with // empty values — nothing in the values to compress, and small // blocks reduce wasted I/O per random Lookup miss (each Lookup @@ -40,10 +39,9 @@ const ( // - OffsetsCF stores 8-byte (ledger_seq -> event_count) rows in // the tens-of-thousands per chunk — same shape as IndexCF. // -// A block size takes effect as SSTs are written, so chunks already on disk -// keep whatever size they were built with: changing one of these constants -// reaches a running deployment only as natural compaction or chunk rotation -// rewrites those SSTs, never at restart. +// A block size applies to SSTs as they are written. Chunks already on disk +// keep theirs until compaction or rotation rewrites them; a restart changes +// nothing. const ( dataCFBlockSize = 8 * 1024 indexCFBlockSize = 4 * 1024 @@ -185,12 +183,11 @@ func (h *HotStore) Offsets() (*LedgerOffsets, error) { // interface so callers can program against batched lookups // uniformly. // -// Freshness, which the match path holds these bitmaps across a whole -// query on: each is a point-in-time image of its term. A sparse term -// is copied out of the mirror's atomically published id list; a dense -// one is denseState.snapshot, the immutable clone shared with every -// other reader of that term. Neither grows under its holder as ingest -// continues, so a walk started now never sees an id written later. +// Each bitmap is a point-in-time image of its term: a sparse term is +// copied out of the mirror's published id list, a dense one is +// denseState.snapshot, the immutable clone shared with every other +// reader. Neither grows under its holder, so a walk never sees an id +// written after its lookup. func (h *HotStore) LookupKeys(ctx context.Context, keys []TermKey) ([]*roaring.Bitmap, error) { if h.chunkStore.IsClosed() { return nil, stores.ErrStoreClosed @@ -686,19 +683,14 @@ func encodeDataKey(eventID uint32) []byte { return key[:] } -// encodeDataKeys encodes every id into ONE backing buffer and returns +// encodeDataKeys encodes every id into one backing buffer and returns // per-id sub-slices of it, in input order: two allocations for the -// whole batch (the buffer plus the slice headers) instead of one per -// id. encodeDataKey cannot serve this loop — its array escapes through -// the returned slice, so each 4-byte key is heap-allocated -// ("moved to heap: key" under -gcflags=-m), and a limit=1000 page pays -// 1000 of them just to name its rows. +// batch rather than one per id, which is what encodeDataKey costs +// because its array escapes through the returned slice. // -// The returned slices alias one array and are read-only. A consumer -// must not retain them past the call it passes them to. -// rocksdb.Store.BatchMultiGet qualifies: grocksdb copies every key into -// C memory and frees that copy before the batched get returns, so -// nothing outlives the call. +// The slices alias one array and must not be retained past the call +// they are passed to. BatchMultiGet qualifies: grocksdb copies each key +// into C memory and frees the copy before returning. func encodeDataKeys(eventIDs []uint32) [][]byte { buf := make([]byte, dataKeyLen*len(eventIDs)) keys := make([][]byte, len(eventIDs)) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go index c3273ed30..cebc2bee2 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go @@ -8,12 +8,11 @@ package event // // Optimization shape: terms are deduped across filters and issued as // a single batched Reader.LookupKeys at iteration start, whose bitmaps -// the walk then holds for the whole query; payload fetches stream in +// the walk holds for the whole query; payload fetches stream in // internal batches. On the cold path the lookup is one MPHF+index.pack -// round trip per Matches call, not per batch. -// -// The candidate set comes from the slab-stepped engine in slab_match.go, which -// answers both directions from one walk over the window. +// round trip per Matches call. The candidate set comes from the slab +// engine in slab_match.go, which serves both directions from one walk +// over the window. import ( "bytes" @@ -220,10 +219,9 @@ type Match struct { type termPlan [][]int // batchSizes resolves the first and following internal batch sizes from the -// caller's hint. The hint is a validated page size — every handler clamps it -// to its protocol limit before it reaches Matches — so it is honored in full -// and a page arrives in one fetch. Both sizes are clamped positive: a zero -// step never advances, so a zero test seam would stall the stream. +// caller's hint. The hint is a page size the handler has already validated, +// so it is honored in full and a page arrives in one fetch. Both sizes are +// clamped positive: a zero step would stall the stream. func batchSizes(hint int) (int, int) { rest := max(1, matchBatchSize) first := rest @@ -253,9 +251,8 @@ func batchSizes(hint int) (int, int) { // drops are invisible: the iterator advances past them internally, so // consumers never see or reason about resume state. // -// firstBatch sizes the first internal fetch batch: a consumer that will stop -// after N matches passes N. Zero and negative hints use the default; a -// positive one is a validated page size and is honored in full. The hint +// firstBatch sizes the first internal fetch: a consumer that will stop after +// N matches passes N. Zero and negative hints use the default. The hint // changes I/O counts only, never what the stream yields. func Matches( ctx context.Context, r Reader, filters []Filter, window IDRange, @@ -317,17 +314,14 @@ func validateMatchCall(ctx context.Context, r Reader, filters []Filter, window I return nil } -// planIndexTerms is step 1 of the index side, shared by both -// directions and run before any index I/O: dedupe the terms the -// filters name across the whole query, and resolve each filter's -// groups to slots in the single batched lookup that follows. plans[i] -// holds the slots filter i needs; the terms within a group are OR-ed -// and the groups AND-ed. +// planIndexTerms dedupes the terms the filters name and resolves each +// filter's groups to slots in the single batched lookup that follows; +// it runs before any index I/O. plans[i] holds the slots filter i +// needs: terms within a group are OR-ed and groups are AND-ed. // -// matchAll reports that some filter, or the empty slice, constrains nothing, -// so the caller streams the window directly. Reading that off the term groups -// is what keeps an unconstrained filter from intersecting nothing and coming -// back empty. +// matchAll reports that some filter, or the empty slice, constrains +// nothing, so the caller streams the window directly rather than +// intersecting nothing and returning empty. func planIndexTerms(filters []Filter) ([]termPlan, []TermKey, bool) { if len(filters) == 0 { return nil, nil, true diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/payload.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/payload.go index 1af815eb8..a41745648 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/payload.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/payload.go @@ -167,12 +167,11 @@ func (p *Payload) MarshalInto(dst []byte) ([]byte, error) { // slice ALIASES into data and is valid only as long as data is. The // the event store read paths apply this two ways: // -// - FetchEvents passes data that outlives the returned slice — hot from -// rocksdb.BatchMultiGet, cold by cloning the borrowed -// packfile.ReadItems buffer — so its Payloads are safe to retain. The -// hot bytes are batch-shared, not per-payload: BatchMultiGet copies one -// batch into one backing array, so retaining a single Payload pins every -// payload fetched with it. +// - FetchEvents passes data that outlives the returned slice, hot from +// rocksdb.BatchMultiGet, cold copied out of the borrowed +// packfile.ReadItems buffer, so its Payloads are safe to retain. Hot +// bytes are shared by the whole batch: retaining one Payload pins +// every payload fetched with it. // - FetchRange / All pass the iterator's borrowed buffer directly // (rocksdb.IterateRange / packfile.ReadRange, valid only for the // current step), so each yielded Payload is borrowed; a consumer that diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go index d4266333e..8e8330704 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go @@ -1,59 +1,25 @@ package event -// slab_match.go is the candidate machinery behind Matches. It steps the window -// one roaring slab at a time — 65536 ids, the span of exactly one container — -// and answers the whole filter algebra inside that slab. +// slab_match.go produces the candidate ids behind Matches. The window is +// walked one slab at a time, 65536 ids, the span of one roaring container, +// and the whole filter algebra is evaluated inside each slab. Direction is +// only the walk order: ascending walks slabs low to high and reads each +// result forward, descending walks high to low and reads backward. // -// The window is applied first rather than last, so no id outside it is ever -// read. Every input to a slab's evaluation is a single container and every -// intermediate the engine allocates holds at most one. +// Work is lazy per slab. A consumer that stops after one page has paid for +// the slabs that page spans, one container per input per filter each. Slabs +// that can hold no candidate are skipped: before evaluating a slab the walk +// asks the term bitmaps where the next candidate can be and jumps there (see +// the bound helpers below). // -// Direction is the slab walk order and nothing else: ascending walks slabs low -// to high and reads each result forward, descending walks them high to low and -// reads each result backward. One code path serves both. -// -// Laziness is per slab, not per id. A consumer that stops after one page has -// evaluated only the slabs that page spans, and the cost inside a slab is -// bounded by the containers its inputs hold there rather than by the width of -// the window. The unit is the slab, so a page ending mid-slab has paid for -// that whole slab: one container's worth of work per input per filter. -// -// Slabs that can hold no candidate are not paid for at all. Before evaluating -// one the walk asks the held term bitmaps where the next candidate could be -// and jumps there when the answer is past this slab, so a rare term spread -// over a wide window costs its own slabs and no others. See "the skip" below -// for the bound and why it is sound. -// -// The trade is descending over a whole window. The whole-chunk union this -// replaced ANDed the chunk-sized terms once with roaring's bulk aggregation, -// where the walk re-enters the algebra per slab. Measured over a 300k-event -// corpus, that cost +8µs on a descending page of a two-fat-term AND (9.9µs → -// 18.1µs) and +170µs on a descending full scan of one chunk-sized term timed -// on candidates alone (1.20ms → 1.37ms), which the fetch swallows end to end -// (10.88ms → 10.93ms). Descending pages that stop early got faster (148µs → -// 128µs), because the union paid for the whole chunk before yielding -// anything. The skip closes the rest of that gap where the terms are sparse -// enough to prove a jump, and seeds the first descending slab at the highest -// candidate rather than at the window's high bound. -// -// Ownership. Every term the query names is materialized once, by the single -// batched Reader.LookupKeys call in Matches, and the resulting bitmaps are -// held for the whole walk. They are read-only: a hot dense term's bitmap is -// the denseState.snapshot every concurrent reader shares, which the index's -// contract forbids mutating or Cloning. AndAny reads them through roaring's -// read-only container accessors and writes only through the receiver, so -// passing them is safe, and roaring_contract_test.go pins that property -// against the pinned roaring version. The only bitmaps this file mutates are -// the per-filter accumulators it built itself, which is also why the union -// across filters can run in place into the first of them. -// -// Freshness follows from holding them. The bitmaps are a point-in-time image -// of the index taken at query start — a sparse hot term is copied out of the -// atomically published id list, a dense one is denseState's published -// snapshot — so an id ingested while the walk runs is invisible to it, in -// either direction and at every slab. That is what the pinned window already -// promises: see IDRange's snapshot-isolation contract, under which no event -// past the pinned End is visible to the request anyway. +// The term bitmaps come from the single Reader.LookupKeys call at query start +// and are held for the whole walk. They are read-only and may be snapshots +// shared with other readers. AndAny reads its arguments and writes only its +// receiver, which roaring_contract_test.go pins against the pinned roaring +// version; the only bitmaps this file mutates are the per-filter accumulators +// it builds. Because the lookup is a point-in-time image, ids ingested during +// the walk are invisible to it, as IDRange's snapshot-isolation contract +// already requires. import ( "cmp" @@ -63,22 +29,17 @@ import ( "github.com/RoaringBitmap/roaring/v2" ) -// slabShift sets the slab width as a power of two: 1<<16 is exactly one -// roaring container. -// -// A var rather than a const so in-package tests can shrink it and drive slab -// seams over a small corpus. It never changes what a stream yields. +// slabShift is the slab width as a power of two; 1<<16 is one roaring +// container. A var so tests can shrink it. It never changes what a stream +// yields. // //nolint:gochecknoglobals // test seam; production never writes it var slabShift uint = 16 -// slabTerms is one of a filter's term groups resolved out of the batched -// lookup: the bitmaps the index returned for the group's present terms, held -// for the whole query. The group's value is their union. -// -// est is the summed cardinality of those bitmaps. It ignores the window, so it -// ranks a filter's groups rather than counting a query's candidates, and it is -// what orders the AND. +// slabTerms is one term group of a filter: the bitmaps the index returned for +// its present terms, held for the whole query. The group's value is their +// union. est is their summed cardinality over the whole chunk and orders a +// filter's groups, rarest first. type slabTerms struct { bitmaps []*roaring.Bitmap est uint64 @@ -89,13 +50,9 @@ type slabFilter struct { groups []slabTerms } -// resolveSlabTerms collects the bitmaps at slots, reporting false when every -// one of them is absent from the index — the signal that the owning filter can -// match nothing. -// -// A present term is a non-nil bitmap, empty or not: an empty one contributes -// nothing to the union but still keeps the group alive, which an absent term's -// caller-side skip would not do. +// resolveSlabTerms collects the bitmaps at slots. ok is false when every one +// is absent from the index, in which case the owning filter can match nothing. +// A present but empty bitmap keeps the group alive. func resolveSlabTerms(sources []*roaring.Bitmap, slots []int) (slabTerms, bool) { var g slabTerms for _, slot := range slots { @@ -109,10 +66,9 @@ func resolveSlabTerms(sources []*roaring.Bitmap, slots []int) (slabTerms, bool) return g, len(g.bitmaps) > 0 } -// resolveSlabFilters is the planning step: resolve every filter's groups, drop -// the filters that named an entirely absent group, and order each survivor's -// groups rarest first so the accumulator shrinks fastest and a group that -// empties it ends the slab before the fat groups are read. +// resolveSlabFilters resolves every filter's groups, drops the filters that +// named an entirely absent group, and orders each survivor's groups rarest +// first so the accumulator shrinks fastest. func resolveSlabFilters(plans []termPlan, sources []*roaring.Bitmap) []slabFilter { out := make([]slabFilter, 0, len(plans)) for _, plan := range plans { @@ -137,15 +93,11 @@ func resolveSlabFilters(plans []termPlan, sources []*roaring.Bitmap) []slabFilte return out } -// eval returns f's matches inside [lo, hi) as a freshly built bitmap the -// caller owns, or nil when f matches nothing there. -// -// The accumulator starts as the slab window itself and is narrowed group by -// group in place: AndAny is x.And(FastOr(args)) without the intermediate -// union, so one call is a whole group. -// -// A filter reaching here always names at least one group: one that names none -// matches everything and takes the match-all path upstream. +// eval returns f's matches inside [lo, hi) as a fresh bitmap the caller owns, +// or nil when there are none. The accumulator starts as the slab range and is +// narrowed in place by one AndAny per group, which is x.And(FastOr(args)) +// without the intermediate union. A filter always names at least one group; +// one that names none takes the match-all path upstream. func (f *slabFilter) eval(lo, hi uint32) *roaring.Bitmap { acc := roaring.New() acc.AddRange(uint64(lo), uint64(hi)) @@ -158,49 +110,33 @@ func (f *slabFilter) eval(lo, hi uint32) *roaring.Bitmap { return acc } -// ──────────────────────── the skip ──────────────────────────── -// -// The walk skips a slab only on a bound it has proved, from the same bitmaps -// it would have evaluated the slab with. Ascending, at position pos: -// -// - a candidate lies in at least one of a group's terms, so it is at or -// above the smallest id at or above pos that any of them holds — the -// minimum of their NextValue(pos). No term holding one proves the owning -// filter matches nothing from pos on; -// - a candidate satisfies every group of its filter, so the filter's bound -// is the largest of its groups' bounds; -// - a candidate belongs to some filter, so the query's bound is the -// smallest of the surviving filters' bounds, and all filters exhausted -// means the walk is over. -// -// Descending is the mirror: PreviousValue, maximum within a group, minimum -// across a filter's groups, maximum across filters. The post-filter only ever -// drops candidates, so a bound proved on the index bounds the stream. +// Bounds. A slab is skipped only on a bound proved from the term bitmaps +// themselves. Ascending, at position pos: // -// roaring's NextValue and PreviousValue answer inclusive of the target and -// -1 for none, and read the bitmap without writing it — they are called on -// snapshots shared with every other reader. roaring_contract_test.go pins -// both properties against the pinned version. +// - a group's bound is the smallest NextValue(pos) over its terms, since a +// candidate lies in at least one of them; no term holding one proves the +// group, and so its filter, matches nothing from pos on; +// - a filter's bound is the largest of its groups' bounds, since a +// candidate satisfies every group; +// - the query's bound is the smallest of the live filters' bounds. // -// A bound the walk has proved is held rather than proved again at every slab, -// and it decides which filters a slab is evaluated for: see slabStepper.bounds. +// Descending mirrors this with PreviousValue and the min and max swapped. +// The post-filter only drops candidates, so a bound proved on the index +// bounds the stream. NextValue and PreviousValue are inclusive of the target, +// return -1 for none, and do not write the bitmap they search; +// roaring_contract_test.go pins all three properties. -// boundRetired marks a filter that has proved it can match nothing more in the -// window: its terms ran out ahead of the cursor, which only ever moves further -// from them, so the walk drops it for good rather than asking again at every -// remaining slab. Below every id, so the test for a bound outside the current -// slab covers it too. +// boundRetired marks a filter whose terms ran out ahead of the cursor. The +// cursor never comes back, so the filter is dropped for the rest of the walk. +// It sorts below every id, so the "outside this slab" test covers it. const boundRetired = int64(-1) // nextBound is the group's bound: the smallest id at or above pos that any of -// its terms holds. ok is false when none does, which proves the group, and so -// the filter owning it, matches nothing from pos on. +// its terms holds. ok is false when none does. func (g *slabTerms) nextBound(pos uint32) (uint32, bool) { if len(g.bitmaps) == 0 { - // A group naming no term constrains nothing and proves no bound. - // resolveSlabTerms never builds one and a filter that would take - // the match-all path never reaches the stepper, but the answer that - // skips nothing is the safe one to give. + // Unreachable, since resolveSlabTerms never builds an empty group; + // an empty group constrains nothing and so proves no bound. return pos, true } best := int64(-1) @@ -234,9 +170,8 @@ func (g *slabTerms) prevBound(pos uint32) (uint32, bool) { return uint32(best), true } -// nextBound is the filter's bound: a candidate satisfies every group, so the -// strongest of the groups' bounds holds. ok is false as soon as one group -// proves the filter is done. +// nextBound is the filter's bound: the largest of its groups' bounds. ok is +// false as soon as one group proves the filter is done. func (f *slabFilter) nextBound(pos uint32) (uint32, bool) { bound := pos for i := range f.groups { @@ -269,23 +204,14 @@ type slabStepper struct { window IDRange desc bool - // bounds holds what the walk has proved about each filter, parallel to - // filters: the id it proved its next candidate lies at or past — at or - // before, descending — or boundRetired once it proved it has none left. - // - // The invariant that makes holding one sound: a filter's bound is monotone - // in the walk direction, so one proved at a cursor position still proves - // the same emptiness at every position the cursor reaches up to it. A held - // bound can be weaker than one proved afresh, because a filter's groups - // can pull apart as the cursor advances; weaker costs a slab the walk - // could have skipped and never a match. - // - // So a bound is proved again only once the cursor has reached it, and a - // bound past the current slab excuses its filter from being evaluated - // there at all — the work a slab costs is the filters live in it rather - // than every filter the query named. Seeding each bound at the cursor's - // own start, where it proves nothing, is what makes the first step prove - // them all. + // bounds holds, per filter, the id its next candidate is proved to lie at + // or past (at or before, descending), or boundRetired once it has none + // left. A bound is monotone in the walk direction, so one proved earlier + // still holds at every position up to it: it is re-proved only once the + // cursor reaches it, and a filter whose bound lies past the current slab + // is not evaluated there. A held bound can be weaker than a fresh one, + // which costs an evaluated slab, never a match. Bounds start at the + // cursor so the first step proves them all. bounds []int64 // cursor is the next unevaluated boundary: the inclusive low bound @@ -319,14 +245,10 @@ func newSlabStepper( return s } -// seekAsc is the lowest position at or above pos that any filter can still -// match at, and false when none can inside the window: a candidate belongs to -// some filter, so the smallest of their bounds holds for the union. Filters that -// are done are retired rather than ending the walk, since a live one may still -// match. -// -// Only a filter whose stored bound pos has reached is asked again; the rest -// answer from the bound they proved earlier, which pos has not yet passed. +// seekAsc returns the lowest position at or above pos where some filter can +// still match, or false when none can inside the window. A filter is asked +// again only once pos has reached its held bound; a filter with no bound +// left is retired rather than ending the walk. func (s *slabStepper) seekAsc(pos uint32) (uint32, bool) { var best int64 found := false @@ -380,22 +302,18 @@ func (s *slabStepper) seekDesc(hi uint32) (uint32, bool) { return uint32(best) + 1, true } -// nextBounds returns the next slab's [lo, hi) clipped to the window, walking -// away from the cursor in the query's direction. -// -// The cursor moves to the proved bound first, so the slab returned is the one -// holding the next possible candidate rather than the one adjacent to the -// last: every slab between is candidate-free for every filter. The bound also -// clips the accumulator inside its own slab, so a slab entered part-way is -// entered at the candidate and not at its base. +// nextBounds returns the next slab's [lo, hi), clipped to the window, in the +// walk's direction. The cursor first moves to the proved bound, so the slab +// returned holds the next possible candidate and is entered at that candidate +// rather than at its base. func (s *slabStepper) nextBounds() (uint32, uint32, bool) { if s.done { return 0, 0, false } if s.desc { - // hi-1 is read inside seekDesc. hi > window.Start >= 0 here, because - // an empty window never reaches the stepper and the walk stops at - // window.Start. + // hi > window.Start here, because an empty window never reaches the + // stepper and the walk stops at window.Start, so seekDesc's hi-1 + // cannot underflow. hi, ok := s.seekDesc(s.cursor) if !ok { s.done = true @@ -427,17 +345,11 @@ func (s *slabStepper) nextBounds() (uint32, uint32, bool) { return lo, hi, true } -// evalSlab is the union across filters of their per-slab results, or nil when -// the slab holds nothing. -// -// Only the filters this slab is about are evaluated. The bound the seek just -// proved for each one says where its next candidate can be, and one lying -// outside [lo, hi) — a retired filter's included — already proves the filter -// matches nothing here, which eval would spend a bitmap to rediscover. -// -// Every per-filter result is a bitmap eval built for this call and nobody else -// holds, so the union runs in place into the first of them rather than -// allocating a separate answer to copy them all into. +// evalSlab unions the per-filter results for [lo, hi), or returns nil when +// the slab holds nothing. A filter whose bound lies outside the slab is +// skipped, since the bound already proves it matches nothing here. The +// results are this call's own bitmaps, so the union runs in place into the +// first of them. func (s *slabStepper) evalSlab(lo, hi uint32) *roaring.Bitmap { var acc *roaring.Bitmap for i := range s.filters { @@ -513,9 +425,8 @@ func (s *slabStepper) appendUpTo(dst []uint32, n int) []uint32 { return dst } -// streamSlabs is the streaming loop, shared by both directions because the -// stepper already hides direction: fill one internal batch of candidate -// ordinals out of the stepper, fetch, post-filter, yield the survivors. +// streamSlabs is the streaming loop for both directions: fill one batch of +// candidate ordinals from the stepper, fetch, post-filter, yield. func streamSlabs( ctx context.Context, r Reader, filters []Filter, st *slabStepper, descending bool, firstBatch int, yield func(Match, error) bool, diff --git a/go.mod b/go.mod index 99666382f..66a2d5bbe 100644 --- a/go.mod +++ b/go.mod @@ -4,12 +4,9 @@ go 1.26 require ( github.com/Masterminds/squirrel v1.5.4 - // Minimum v2.18.2 (the FastOr/runContainer16 fix, #527, the fork - // previously carried). v2.26.0 ships vectorized container kernels, - // hand-written assembly included, gated by x/sys/cpu and GODEBUG - // (cpu.avx512vpopcntdq=off and friends); the aggregation layer this - // package's contracts rest on — fastaggregation.go, parallel.go — is - // byte-identical to v2.18. + // v2.18.2 is the minimum (the FastOr/runContainer16 fix, #527). + // v2.26.0 adds vectorized container kernels behind x/sys/cpu and + // GODEBUG gates; the aggregation layer is unchanged since v2.18. github.com/RoaringBitmap/roaring/v2 v2.26.0 github.com/aws/aws-sdk-go-v2 v1.45.1 github.com/aws/aws-sdk-go-v2/config v1.31.16 From 17f89bafcf67d7dc91fb9ec0f9b9dc5076a875fe Mon Sep 17 00:00:00 2001 From: tamirms Date: Wed, 9 Sep 2026 20:10:34 +0100 Subject: [PATCH 38/41] event: trim the test suite and its comments Drop the tests that duplicate what the whole-stream comparisons already prove or pin an internal heuristic rather than behavior: - requireStrictOrder, redundant with the whole-stream equality every randomized trial already asserts - TestResolveSlabFiltersOrdersRarestFirst, which pinned the rarest-first ordering heuristic; the absent-group drop it also covered is exercised by TestResolveSlabTerms and the shaped matrix - the FastAnd and FastOr subtests of the roaring contract, since the match path only calls AndAny, And, NextValue and PreviousValue on shared bitmaps - TestRoaringContract_ConcurrentValueSearch, folded into the aggregation race gate as TestRoaringContract_ConcurrentReaders Rename TestMatches_WindowANDLeavesBorrowedBitmapUntouched, whose doc described a branch that no longer exists, and shorten the remaining test comments. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01M23ZvEkyrFjyUUbm7zLwob --- .../internal/rpcv2/packfile/pools_test.go | 31 +-- .../rpcv2/stores/event/concurrent_bitmaps.go | 9 +- .../rpcv2/stores/event/hot_store_test.go | 32 +-- .../internal/rpcv2/stores/event/match_test.go | 30 +-- .../stores/event/matches_differential_test.go | 17 +- .../stores/event/roaring_contract_test.go | 193 ++++----------- .../rpcv2/stores/event/slab_match_test.go | 229 ++++++------------ 7 files changed, 163 insertions(+), 378 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go b/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go index 8cf045b15..a2892c8e7 100644 --- a/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/packfile/pools_test.go @@ -20,13 +20,11 @@ func poolItems(n int) [][]byte { return items } -// Drives several open-read-close cycles over distinct files through the -// process-wide pools; byte-exact reads prove a recycled buffer never leaks one -// file's decode into another's. +// Several open-read-close cycles over distinct files, so a recycled buffer +// that leaked one file's decode into another's would fail the byte checks. func TestReaderPoolReuseKeepsReadsCorrect(t *testing.T) { for cycle := range 4 { - // Vary the item count so recycled arrays are reused at different - // lengths, covering the reslice paths. + // Different item counts reuse recycled arrays at different lengths. items := poolItems(300 + 40*cycle) path := writeTestPackfile(t, items, WriterOptions{ItemsPerRecord: 16}) r := Open(path, ReaderOptions{}) @@ -42,8 +40,8 @@ func TestReaderPoolReuseKeepsReadsCorrect(t *testing.T) { } } -// The handshake's fast path: a read beginning after Close reports a closed -// reader, matching os.ErrClosed, instead of touching recycled memory. +// A read beginning after Close reports os.ErrClosed rather than touching +// recycled memory. func TestReaderReadAfterCloseFails(t *testing.T) { items := poolItems(64) path := writeTestPackfile(t, items, WriterOptions{ItemsPerRecord: 16}) @@ -60,10 +58,8 @@ func TestReaderReadAfterCloseFails(t *testing.T) { } } -// The handshake's slow path: Close racing an in-flight read, which is a caller -// contract violation, must leave the offsets to the garbage collector rather -// than recycle them under the reader. Under -race this also proves the -// ordering. +// Close racing an in-flight read, a caller contract violation, must leave the +// offsets to the garbage collector rather than recycle them under the reader. func TestReaderCloseDuringReadDoesNotRecycle(t *testing.T) { items := poolItems(256) path := writeTestPackfile(t, items, WriterOptions{ItemsPerRecord: 16}) @@ -92,9 +88,8 @@ func TestReaderCloseDuringReadDoesNotRecycle(t *testing.T) { }) wg.Wait() require.NoError(t, closeErr) - // Either outcome is documented for this contract violation. What the - // handshake forbids is recycled-memory corruption, which the byte - // check would catch. + // Either outcome is allowed; what the handshake forbids is recycled-memory + // corruption, which the byte check would catch. if err == nil { require.True(t, bytes.Equal(got, items[0]), "payload corrupted by Close during read") } else { @@ -102,11 +97,9 @@ func TestReaderCloseDuringReadDoesNotRecycle(t *testing.T) { } } -// The counter is the only signal the caps have been crossed, so it must count -// exactly the Puts a cap dropped: an over-cap buffer on every pool, and -// nothing for the buffers a Put keeps or for the empty slices it ignores. -// -// The count is process-wide, so the assertions are on the delta. +// The counter must count exactly the Puts a cap dropped, on every pool, and +// nothing for kept buffers or ignored empty slices. It is process-wide, so +// the assertions are on the delta. func TestPoolCapSkipsCountsDroppedPuts(t *testing.T) { before := PoolCapSkips() diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go index 9f990cb1e..3c54ee804 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go @@ -93,13 +93,12 @@ func NewConcurrentBitmapsFromBitmaps(b Bitmaps) *ConcurrentBitmaps { // - RunOptimize, AddRange, RemoveRange, FlipInt // - Add, AddMany, Remove, CheckedAdd, CheckedRemove, AddInt // - SetCopyOnWrite -// - single-input roaring.FastAnd / roaring.FastOr (roaring takes a -// Clone-the-input shortcut when there is only one input) +// - roaring.FastAnd / roaring.FastOr with a single input, which Clone it // // Safe: any non-mutating read (Contains, GetCardinality, Iterator, -// NextValue, PreviousValue, ToArray, IsEmpty, Minimum, Maximum), -// passing it as an argument to AndAny, and roaring.And / FastAnd / -// FastOr with 2+ inputs. +// NextValue, PreviousValue, ToArray, IsEmpty, Minimum, Maximum) and +// passing it as an argument to AndAny or And; roaring_contract_test.go +// pins these against the pinned roaring version. // // A Get that starts after an AddTo returns sees that AddTo's IDs, and // the pointer stays valid for as long as the caller holds it. The diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go index bf3fe31d1..aa3343bd6 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/hot_store_test.go @@ -346,25 +346,17 @@ func TestHotStore_FetchEventsRejectsUnsortedInput(t *testing.T) { require.ErrorIs(t, err, ErrUnsortedEventIDs, "duplicate input must error") } -// fetchEventsPerIDAllocBudget is the number of heap allocations -// FetchEvents may spend per event ID. Both are grocksdb's, spent inside -// the batched multi-get: one *PinnableSlice per key, plus the size -// out-param the single PinnableSlice.Data call per key escapes to the -// heap (BatchMultiGet reads each pinned value once and copies from the -// kept C-backed slice). Our own per-batch work — the key list, the -// value arena, the Payload slice — is a fixed handful of allocations -// regardless of batch size. -// -// If a grocksdb bump moves this number, raise it deliberately after -// checking where the new allocation comes from; do not widen it to -// absorb a regression on our side of the boundary. +// fetchEventsPerIDAllocBudget is the number of heap allocations FetchEvents +// may spend per event ID. Both are grocksdb's: one *PinnableSlice per key +// and the size out-param of the one PinnableSlice.Data call per key. Our +// own work per batch is a fixed handful. If a grocksdb bump moves this +// number, raise it after checking where the new allocation comes from; do +// not widen it to absorb a regression on our side. const fetchEventsPerIDAllocBudget = 2 -// TestHotStore_FetchEventsAllocationBudget pins that FetchEvents spends -// no per-ID allocation of its own. The regression it catches is building -// the RocksDB key list one heap-allocated key at a time, which cost a -// limit=1000 page a thousand extra allocations; encodeDataKeys carves -// them out of one buffer instead, leaving only grocksdb's two per ID. +// TestHotStore_FetchEventsAllocationBudget pins that FetchEvents spends no +// per-ID allocation of its own; the regression it catches is a +// heap-allocated key per ID. func TestHotStore_FetchEventsAllocationBudget(t *testing.T) { const chunkID = chunk.ID(0) const n = 512 @@ -401,10 +393,8 @@ func TestHotStore_FetchEventsAllocationBudget(t *testing.T) { "allocation crept back into the fetch path", n, allocs, budget) } -// TestEncodeDataKeys pins encodeDataKeys itself: the keys are -// byte-identical to encodeDataKey's, they are distinct windows onto one -// buffer, and the allocation count does not grow with the batch — the -// property the FetchEvents budget above rests on. +// TestEncodeDataKeys pins that the keys equal encodeDataKey's, are distinct +// windows onto one buffer, and cost a fixed number of allocations. func TestEncodeDataKeys(t *testing.T) { ids := []uint32{0, 1, 7, 1 << 20, ^uint32(0)} keys := encodeDataKeys(ids) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go index 5ffec2445..2fe5a62bf 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go @@ -1145,7 +1145,7 @@ func TestQuery_InvalidFilterRejected(t *testing.T) { // - single-filter (contractID) → one LookupKeys term, then the // ascending slab walk // - multi-term filter (AND) → AndAny per group over cold bitmaps -// - cross-filter (OR) → FastOr across filters +// - cross-filter (OR) → the in-place Or across filters // - ledger range + filter → the window clipped into each slab's // accumulator // - descending + range + cap → the slab walk run high to low @@ -1606,21 +1606,11 @@ func TestMatches_EmptyStreams(t *testing.T) { wholeChunk(t, fx.store), false)) } -// TestMatches_WindowANDLeavesBorrowedBitmapUntouched pins the -// singleFilter branch of the window AND: a single-constraint filter -// borrows the hot mirror's bitmap directly from LookupKeys, and the -// narrowing AND must allocate a fresh result rather than shrink the -// mirror's live state in place. -// -// The borrow only exists for DENSE terms (the mirror's sparse mode -// materializes a fresh bitmap per Get, which no mutation can corrupt), -// so the term is first promoted past the mirror's promotion threshold -// with injected ids outside the query window (never fetched). -// TestQuery_DoesNotMutateMirrorBitmaps cannot catch the mutation: -// its filters carry two constraints (FastAnd-owned inputs) and it -// compares only cardinality over whole-chunk ranges, where the AND is -// a no-op. -func TestMatches_WindowANDLeavesBorrowedBitmapUntouched(t *testing.T) { +// TestMatches_LeavesSharedSnapshotUntouched pins that a narrowing window over +// a single-term filter leaves the hot mirror's shared bitmap as it was. Only +// dense terms are shared, so the term is first promoted with injected ids +// above the query window. +func TestMatches_LeavesSharedSnapshotUntouched(t *testing.T) { fx := newQueryFixture(t) key := ComputeTermKey(fx.contractA[:], FieldContractID) // Promote contract A's term (real matches: ids 0, 1, 4) to dense @@ -1634,18 +1624,16 @@ func TestMatches_WindowANDLeavesBorrowedBitmapUntouched(t *testing.T) { "fixture sanity: the term must be dense so LookupKeys borrows") snapshot := before.Clone() - // Single filter, single constraint, narrowing range: the borrowed - // path with an AND that actually removes ids. + // A narrowing window over a single-term filter. got := collectMatches(t, fx.store, []Filter{{ContractID: fx.contractA[:]}}, IDRange{Start: 0, End: 2}, false) assert.Equal(t, []uint32{0, 1}, matchOrdinals(got)) after := lookupOne(t, fx.store, key) assert.True(t, snapshot.Equals(after), - "the window AND must not mutate the mirror's term bitmap in place") + "Matches must not mutate the mirror's shared term bitmap") - // End to end: the same filter over the whole chunk still sees the - // ids an in-place AND would have destroyed. + // The same filter over the whole chunk still sees every id. full := collectMatches(t, fx.store, []Filter{{ContractID: fx.contractA[:]}}, wholeChunk(t, fx.store), false) assert.Equal(t, []uint32{0, 1, 4}, matchOrdinals(full)) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go index a55d9cb84..677f5b24c 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/matches_differential_test.go @@ -1,11 +1,7 @@ package event -// The in-memory chunk the match-path tests run against, and the borrow-safety -// gate over it. -// -// The corpus is served through the index's one read seam, LookupKeys, which -// materializes every term as a bitmap. slab_match_test.go drives it against an -// answer computed without the index. +// The in-memory chunk the match-path tests run against, served through +// LookupKeys, and the borrow-safety gate over it. import ( "context" @@ -195,11 +191,10 @@ func collectOrdinals(t *testing.T, r Reader, filters []Filter, w IDRange, desc b return out } -// Turns the borrow contract into a race-detector gate: the match path holds -// mirror snapshots across a whole walk while AddTo publishes new termStates on -// the same keys, including the sparse-to-dense promotion. Under -race any -// write reaching a held snapshot fails the run; without it, the identity check -// still pins that a pinned window is immune to ingest past its End. +// The match path holds mirror snapshots across a whole walk while AddTo +// publishes new termStates on the same keys, sparse-to-dense promotion +// included. Under -race any write reaching a held snapshot fails the run; +// without it, the identity check pins that a pinned window ignores later ingest. func TestMatches_ConcurrentIngestBorrowSafety(t *testing.T) { rng := rand.New(rand.NewSource(20260830)) v := newDiffVocab(t) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go index c3c69c424..0bab7e101 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go @@ -1,23 +1,13 @@ package event -// roaring_contract_test.go pins the library-level properties the event index is -// built on: the aggregation entry points this package calls with shared -// bitmaps must treat those bitmaps as read-only, and the value searches the -// slab walk proves its skips with must answer inclusively, read-only, and -1 -// for none. -// -// ConcurrentBitmaps.Get and denseState.snapshot publish one bitmap to every -// concurrent reader at once and never mutate it afterwards. Readers therefore -// hand the same *roaring.Bitmap to roaring.FastAnd, roaring.FastOr and -// Bitmap.AndAny from many goroutines at once. That is sound only while roaring -// writes exclusively through the receiver (And, AndAny) or into a freshly -// allocated answer (FastAnd, FastOr) — a property roaring documents by -// implication and this file pins by observation. -// -// The pin is on the dependency, not on any caller: it is written against -// roaring alone, so it holds however the match path is built, and it is the -// first thing to run on a roaring version bump. A failure here means the new -// version is unsafe to take, not that a caller is wrong. +// roaring_contract_test.go pins the roaring properties the event index rests +// on. ConcurrentBitmaps.Get publishes one bitmap to every concurrent reader +// and never mutates it, and the match path hands those bitmaps to AndAny, +// NextValue and PreviousValue from many goroutines at once. That is sound +// only while AndAny writes nothing but its receiver and shares no storage +// with its arguments, and while the searches are read-only, inclusive of the +// target, and return -1 for none. The tests are written against roaring +// alone, so they are the first thing to run on a version bump. // // Pinned version: github.com/RoaringBitmap/roaring/v2 v2.26.0. @@ -31,16 +21,10 @@ import ( "github.com/stretchr/testify/require" ) -// sharedBitmaps returns bitmaps in the shape the index publishes: a -// writer-private bitmap marked copy-on-write, and the Clone of it that -// denseState.snapshot hands to readers. The clone shares its containers with -// the writer's copy, so a write reaching either is visible in the other, which -// is what makes an unnoticed mutation here a live corruption bug rather than a -// style violation. -// -// The set spans all three container kinds and several high keys, because the -// aggregation paths branch on container type and on key advance: only a mixed -// corpus reaches all of them. +// sharedBitmaps returns bitmaps in the shape the index publishes: the +// copy-on-write Clone that denseState.snapshot hands to readers, sharing its +// containers with the writer's bitmap. The set spans all three container +// kinds and several high keys, since the roaring paths branch on both. func sharedBitmaps(t *testing.T) []*roaring.Bitmap { t.Helper() rng := rand.New(rand.NewSource(20260909)) @@ -115,16 +99,14 @@ func freshRange(lo, hi uint64) *roaring.Bitmap { return acc } -// TestRoaringContract_AggregationDoesNotMutateInputs is the core pin: every -// aggregation this package calls with shared bitmaps leaves those bitmaps +// TestRoaringContract_AggregationDoesNotMutateInputs pins that AndAny, and +// And, which AndAny delegates a single argument to, leave shared arguments // byte-identical. func TestRoaringContract_AggregationDoesNotMutateInputs(t *testing.T) { t.Parallel() - // Ranges chosen to hit the shapes that differ inside roaring: a whole - // high key (AddRange produces a full run container, whose iand takes the - // clone-the-other-side branch), a partial one, one the inputs do not - // reach at all, and a span crossing several. + // Ranges that take different paths inside roaring: a whole high key, a + // partial one, one the inputs do not reach, and a span across several. ranges := []struct { name string lo, hi uint64 @@ -159,28 +141,6 @@ func TestRoaringContract_AggregationDoesNotMutateInputs(t *testing.T) { }) } - t.Run("FastAnd", func(t *testing.T) { - t.Parallel() - for n := 2; n <= 4; n++ { - shared := sharedBitmaps(t)[:n] - before := bitmapImages(t, shared) - require.NotNil(t, roaring.FastAnd(shared...)) - requireUnchanged(t, shared, before, "FastAnd") - } - }) - - t.Run("FastOr", func(t *testing.T) { - t.Parallel() - for n := 2; n <= 4; n++ { - shared := sharedBitmaps(t)[:n] - before := bitmapImages(t, shared) - require.NotNil(t, roaring.FastOr(shared...)) - requireUnchanged(t, shared, before, "FastOr") - } - }) - - // AndAny delegates a single argument to And, so And carries the same - // obligation and is pinned separately. t.Run("And", func(t *testing.T) { t.Parallel() for _, shared := range sharedBitmaps(t) { @@ -193,10 +153,10 @@ func TestRoaringContract_AggregationDoesNotMutateInputs(t *testing.T) { }) } -// TestRoaringContract_AndAnyDoesNotRetainArguments pins the other half of the -// contract: AndAny copies out of its arguments rather than aliasing their -// storage into the receiver. A caller that hands it a reusable scratch bitmap -// depends on the answer surviving that scratch's next Clear. +// TestRoaringContract_AndAnyDoesNotRetainArguments pins that AndAny copies +// out of its arguments rather than aliasing their storage into the receiver: +// the slab walk goes on to write the receiver, with further AndAny groups and +// the in-place union across filters. func TestRoaringContract_AndAnyDoesNotRetainArguments(t *testing.T) { t.Parallel() @@ -216,41 +176,44 @@ func TestRoaringContract_AndAnyDoesNotRetainArguments(t *testing.T) { scratch.Clear() require.Equal(t, answer, bitmapBytes(t, acc), - "AndAny with %d args aliased a scratch argument's storage into "+ - "the receiver: reusing a scratch bitmap is unsafe", nArgs) + "AndAny with %d args aliased an argument's storage into the receiver", nArgs) } } } -// TestRoaringContract_ConcurrentReadersShareArguments is the race-detector -// gate. Eight goroutines aggregate over the same shared, copy-on-write-marked -// bitmaps at once, the way concurrent getEvents requests do against one -// denseState snapshot. Under -race any write reaching a shared bitmap fails -// the run; without it, the byte-identity check and the agreement between -// goroutines still catch a mutation. -func TestRoaringContract_ConcurrentReadersShareArguments(t *testing.T) { +// TestRoaringContract_ConcurrentReaders is the race-detector gate: eight +// goroutines run AndAny, NextValue and PreviousValue over the same shared +// copy-on-write snapshots at once, as concurrent queries do. Under -race any +// write to a shared bitmap fails the run; without it, the byte-identity check +// and the agreement between goroutines still catch one. +func TestRoaringContract_ConcurrentReaders(t *testing.T) { t.Parallel() const goroutines = 8 const rounds = 32 + targets := []uint32{0, 1, 1 << 15, 1 << 16, 1<<16 + 1, 3 << 16, 1<<20 - 1} shared := sharedBitmaps(t) before := bitmapImages(t, shared) - // The single-threaded answer every goroutine must reproduce. + // The single-threaded answers every goroutine must reproduce. seq := freshRange(0, 4<<16) seq.AndAny(shared...) wantAndAny := bitmapBytes(t, seq) - wantFastAnd := bitmapBytes(t, roaring.FastAnd(shared...)) - wantFastOr := bitmapBytes(t, roaring.FastOr(shared...)) + var wantSearch []int64 + for _, bm := range shared { + for _, target := range targets { + wantSearch = append(wantSearch, bm.NextValue(target), bm.PreviousValue(target)) + } + } - results := make([][3][]byte, goroutines) + gotAndAny := make([][]byte, goroutines) + gotSearch := make([][]int64, goroutines) var wg sync.WaitGroup start := make(chan struct{}) for g := range goroutines { wg.Go(func() { <-start - var last [3][]byte for range rounds { acc := freshRange(0, 4<<16) acc.AndAny(shared...) @@ -258,39 +221,31 @@ func TestRoaringContract_ConcurrentReadersShareArguments(t *testing.T) { if err != nil { panic(err) } - last[0] = b - if b, err = roaring.FastAnd(shared...).ToBytes(); err != nil { - panic(err) - } - last[1] = b - if b, err = roaring.FastOr(shared...).ToBytes(); err != nil { - panic(err) + gotAndAny[g] = b + var search []int64 + for _, bm := range shared { + for _, target := range targets { + search = append(search, bm.NextValue(target), bm.PreviousValue(target)) + } } - last[2] = b + gotSearch[g] = search } - results[g] = last }) } close(start) wg.Wait() - requireUnchanged(t, shared, before, "concurrent aggregation") + requireUnchanged(t, shared, before, "concurrent reads") for g := range goroutines { - require.Equal(t, wantAndAny, results[g][0], "goroutine %d disagreed on AndAny", g) - require.Equal(t, wantFastAnd, results[g][1], "goroutine %d disagreed on FastAnd", g) - require.Equal(t, wantFastOr, results[g][2], "goroutine %d disagreed on FastOr", g) + require.Equal(t, wantAndAny, gotAndAny[g], "goroutine %d disagreed on AndAny", g) + require.Equal(t, wantSearch, gotSearch[g], "goroutine %d disagreed on the value searches", g) } } // TestRoaringContract_ValueSearchIsInclusiveAndReadOnly pins NextValue and -// PreviousValue, the searches the slab walk proves a skip with. The walk skips -// every slab between the cursor and the answer, so an answer that overshot the -// target in the walk's direction — or a search that reported -1 with ids still -// to come — would silently drop matches. -// -// The definition is checked against the bitmap's own ids rather than against -// hand-written expectations, over targets that land on an id, between two, on -// and beside a container boundary, and past the last id. +// PreviousValue, which the slab walk proves its skips with: an answer past +// the target, or a -1 with ids still to come, would silently drop matches. +// The definition is checked against the bitmap's own ids. func TestRoaringContract_ValueSearchIsInclusiveAndReadOnly(t *testing.T) { t.Parallel() @@ -333,9 +288,8 @@ func TestRoaringContract_ValueSearchIsInclusiveAndReadOnly(t *testing.T) { "an empty bitmap must report no previous value") } -// searchTargets returns the positions worth asking about for a bitmap holding -// ids: each id, its neighbors, every container boundary the ids span, and the -// ends of the uint32 range. +// searchTargets returns each id, its neighbors, the container boundaries the +// ids span, and the ends of the uint32 range. func searchTargets(ids []uint32) []uint32 { out := []uint32{0, 1<<32 - 1} for _, id := range ids { @@ -356,46 +310,3 @@ func searchTargets(ids []uint32) []uint32 { slices.Sort(out) return slices.Compact(out) } - -// TestRoaringContract_ConcurrentValueSearch is the race-detector gate for the -// searches: the slab walk runs them on the same denseState snapshot from every -// in-flight query at once, so they must not lazily materialize anything inside -// the bitmap they read. -func TestRoaringContract_ConcurrentValueSearch(t *testing.T) { - t.Parallel() - - const goroutines = 8 - shared := sharedBitmaps(t) - before := bitmapImages(t, shared) - - targets := []uint32{0, 1, 1 << 15, 1 << 16, 1<<16 + 1, 3 << 16, 1<<20 - 1} - want := make([][]int64, len(shared)) - for i, bm := range shared { - for _, target := range targets { - want[i] = append(want[i], bm.NextValue(target), bm.PreviousValue(target)) - } - } - - got := make([][][]int64, goroutines) - var wg sync.WaitGroup - start := make(chan struct{}) - for g := range goroutines { - wg.Go(func() { - <-start - out := make([][]int64, len(shared)) - for i, bm := range shared { - for _, target := range targets { - out[i] = append(out[i], bm.NextValue(target), bm.PreviousValue(target)) - } - } - got[g] = out - }) - } - close(start) - wg.Wait() - - requireUnchanged(t, shared, before, "concurrent value search") - for g := range goroutines { - require.Equal(t, want, got[g], "goroutine %d disagreed on the value searches", g) - } -} diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go index 71a25e613..dff37d37f 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go @@ -1,19 +1,10 @@ package event // slab_match_test.go covers the slab engine over a corpus built to hold each -// named query shape by construction: dense-only, sparse-only, mixed, absent, -// match-all, the fat/fat thin-overlap shape whose two chunk-sized terms meet -// on a handful of ids, and the high-arity AND and union. -// -// The walk skips slabs it can prove hold no candidate, so those same shapes -// run again at slab widths narrow enough to put dozens of empty slabs between -// consecutive ids, where a bound that over-skipped would drop matches. -// -// The answer every case is checked against is computed without the index — -// postFilter run over every ordinal in the corpus, then clipped to the window, -// the direction and the page. It shares no code with the term planning, the -// slab walk or the batching, so an engine that drops an id at a slab seam, -// yields one twice or emits out of order disagrees with it. +// named query shape by construction, and again at slab widths narrow enough +// to put many empty slabs between consecutive ids. Every case is checked +// against an answer computed without the index: postFilter over every +// ordinal in the corpus, clipped to the window, the direction and the page. import ( "cmp" @@ -32,8 +23,7 @@ import ( ) // drainMatches drains seq into a slice, stopping after limit items when limit -// is positive. The result is compared whole, so it pins the payload bytes, the -// ordinals and their order in one assertion. +// is positive. func drainMatches(tb testing.TB, seq iter.Seq2[Match, error], limit int) []Match { tb.Helper() out := []Match{} @@ -50,10 +40,8 @@ func drainMatches(tb testing.TB, seq iter.Seq2[Match, error], limit int) []Match // ───────────────────────── the answer without the index ───────────────────── // matchingEvents is the corpus's whole answer to filters, computed by running -// the post-filter over every ordinal in it rather than by asking the index. -// -// An empty filter slice is the match-all shape, which postFilter is never -// reached with: it selects everything. +// the post-filter over every ordinal rather than by asking the index. An +// empty filter slice selects everything. func matchingEvents(tb testing.TB, c *diffCorpus, filters []Filter) []Match { tb.Helper() ids := make([]uint32, len(c.raw)) @@ -119,9 +107,8 @@ func requireStream(tb testing.TB, r Reader, all []Match, c queryCase) []Match { // ───────────────────────── the shaped corpus ───────────────────────── -// slabShape is one distinct event in the shaped corpus. The corpus assigns a -// shape to every id by rule, so a corpus spanning several slabs costs a couple -// of dozen XDR marshals instead of one per event. +// slabShape is one distinct event in the shaped corpus; ids map to shapes by +// rule, so the corpus costs a few dozen XDR marshals. type slabShape struct { contract int topic0 int @@ -141,17 +128,15 @@ type shapedFixture struct { rareTopic []uint32 } -// shapeFor is the corpus's id → event rule. It is written so that: +// shapeFor is the corpus's id to event rule: // -// - contract 0 and contract 1 split the corpus in half: two dense terms. -// - contract 2 is carried by a handful of ids: a term that stays sparse. -// - topic 0 tracks the id's parity, except on the thin-overlap ids, so -// {contract 0} ∧ {topic0 = odd-topic} is two chunk-sized terms meeting on -// a few ids. -// - a second topic appears on a handful of ids, so the topic-count bucket -// family has one dense bucket and one sparse bucket, making the -// "at least one topic" group a mixed dense/sparse OR. -// - the system event type is rare enough to stay sparse. +// - contracts 0 and 1 split the corpus in half: two dense terms; +// - contract 2 is carried by a handful of ids: a sparse term; +// - topic 0 tracks the id's parity except on the thin-overlap ids, so +// contract 0 AND the odd topic is two chunk-sized terms meeting on a few; +// - a second topic appears on a handful of ids, so the topic-count buckets +// are one dense and one sparse; +// - the system event type stays sparse. func (f *shapedFixture) shapeFor(id uint32) slabShape { sh := slabShape{topic1: -1} switch { @@ -302,10 +287,9 @@ func (f *shapedFixture) filterAbsentPlusPresent() []Filter { } } -// The absent group's position inside a filter matters: an engine that stops -// resolving groups at the first miss but keeps the filter would still answer -// the leading-miss shape correctly and get the trailing-miss shape wrong. Both -// orders are named, since termGroups emits contract before topics. +// Both positions of an absent group are named: an engine that stopped +// resolving at the first miss but kept the filter would only get the trailing +// shape wrong. func (f *shapedFixture) filterAbsentGroupTrailing() []Filter { // topicRaw[4] is in the vocabulary but no corpus event carries it. return []Filter{{ @@ -321,9 +305,8 @@ func (f *shapedFixture) filterAbsentGroupLeading() []Filter { }} } -// A group whose terms are only partly absent: the corpus carries one and two -// topics, so "at least two" ORs one sparse bucket with three absent ones, -// while "at least three" is a group that is absent in full. +// The corpus carries one and two topics, so "at least two" ORs one sparse +// bucket with three absent ones and "at least three" is absent in full. func (f *shapedFixture) filterPartlyAbsentGroup() []Filter { return []Filter{{TopicCount: TopicCountFilter{Count: 2}}} } @@ -344,11 +327,9 @@ func (f *shapedFixture) filterUnion() []Filter { } } -// filterDeepAND names every constrainable field at once, so the AND runs five -// groups deep: a near-total event-type term, two chunk-sized terms and two -// sparse ones, meeting on the handful of ids carrying a second topic. Exact -// keeps the count group in the plan, which an "at least" bound the constrained -// positions already imply would not. +// filterDeepAND names every constrainable field, so the AND runs five groups +// deep and meets on the handful of ids carrying a second topic. Exact keeps +// the count group in the plan. func (f *shapedFixture) filterDeepAND() []Filter { evType := xdr.ContractEventTypeContract return []Filter{{ @@ -362,9 +343,8 @@ func (f *shapedFixture) filterDeepAND() []Filter { }} } -// filterWideUnion drives eight filters into one FastOr, the widest of the -// named shapes, mixing dense, sparse, absent and multi-group filters so the -// union also has to drop a filter and dedup ids several of them select. +// filterWideUnion unions eight filters, mixing dense, sparse, absent and +// multi-group ones, so the union drops a filter and dedups shared ids. func (f *shapedFixture) filterWideUnion() []Filter { sysType := xdr.ContractEventTypeSystem return []Filter{ @@ -441,9 +421,8 @@ func TestMatches_ShapedMatrix(t *testing.T) { {"tail", IDRange{shapedCorpusSize - 5, shapedCorpusSize}}, } - // Limits chosen against slab 0's yield for the fat filters (~35k): one - // page's worth, and a single item, so both a page that ends mid-slab and - // one that ends on a slab's last id are covered at every window bound. + // One page and a single item: both a page ending mid-slab and one ending + // on a slab's last id, at every window bound. limits := []int{1, 1000} for _, sh := range f.namedShapes() { @@ -464,10 +443,8 @@ func TestMatches_ShapedMatrix(t *testing.T) { } } -// The bounded matrix above never reaches the end of a fat stream. This one -// does: whole streams, unlimited, over the windows where "the end" is a -// different thing — the corpus end, a slab boundary, and a window living -// entirely inside the second slab. +// The bounded matrix never reaches the end of a fat stream; this runs whole +// streams over the windows where the end differs. func TestMatches_WholeStreams(t *testing.T) { f := newShapedFixture(t) r := diffReader{f.corpus} @@ -489,10 +466,8 @@ func TestMatches_WholeStreams(t *testing.T) { } } -// The shaped corpus must hold the shapes its filters are named for: -// chunk-sized terms, sparse terms below the promotion threshold, and a thin -// overlap. Drift in the corpus rules would otherwise turn the matrix above -// into a weaker test without failing it. +// The corpus must hold the shapes its filters are named for; drift in its +// rules would weaken the matrix without failing it. func TestMatches_ShapedFixtureIsWhatItClaims(t *testing.T) { f := newShapedFixture(t) r := diffReader{f.corpus} @@ -526,16 +501,14 @@ func TestMatches_ShapedFixtureIsWhatItClaims(t *testing.T) { Matches(ctx, r, f.filterPartlyAbsentGroup(), window, false, 0), 0), len(f.rareTopic), "the partly-absent group must select exactly its one present bucket") - // The high-arity shapes must reach the corpus. A five-group AND that - // intersected to nothing, or a union that selected a corner of it, would - // pass the matrix vacuously. + // The high-arity shapes must select something, or the matrix passes + // vacuously. require.Greater(t, card(f.filterDeepAND()), 10, "the five-group AND must still select something") require.Greater(t, card(f.filterWideUnion()), 30_000, "the wide union must span the corpus") - // The overlap shape is only the overlap shape if both sides are - // chunk-sized and their meeting point is rare. + // Both sides of the overlap must be chunk-sized. fat, err := r.LookupKeys(ctx, []TermKey{ ComputeTermKey(f.vocab.contracts[0], FieldContractID), ComputeTermKey(f.vocab.topicRaw[1], topicField(0)), @@ -551,30 +524,26 @@ func TestMatches_ShapedFixtureIsWhatItClaims(t *testing.T) { // ───────────────────────── the randomized matrix ───────────────────────── // TestMatches_RandomizedAgainstPostFilter drives random filters, windows and -// page sizes over a random corpus at several slab widths, so a 300-event -// corpus still crosses dozens of slab seams, and requires the corpus's own -// answer every time. +// page sizes over a random corpus at several slab widths. func TestMatches_RandomizedAgainstPostFilter(t *testing.T) { v := newDiffVocab(t) const corpusSize = 300 corpus := newDiffCorpus(t, rand.New(rand.NewSource(20260829)), v, corpusSize) - // Shrink the batch so multi-batch seams are exercised on a small corpus. + // A small batch exercises batch seams on a small corpus. defer func(n int) { matchBatchSize = n }(matchBatchSize) matchBatchSize = 7 defer func(s uint) { slabShift = s }(slabShift) r := diffReader{corpus} - // The slab width is a seam, not a behavior: every width must reproduce the - // same stream. 1, 2 and 4 put 150, 75 and 19 slab seams inside the corpus, - // 8 leaves a single seam, and 16 is the production width, where the whole - // corpus is one slab. + // Every width must yield the same stream: 1, 2 and 4 put many slab seams + // inside the corpus, 8 leaves one, and 16 is the production width. for _, shift := range []uint{1, 2, 4, 8, 16} { slabShift = shift rng := rand.New(rand.NewSource(int64(20260909 + shift))) matched := 0 - for trial := range 400 { - matched += randomizedTrial(t, r, corpus, v, rng, corpusSize, trial) + for range 400 { + matched += randomizedTrial(t, r, corpus, v, rng, corpusSize) } require.Greater(t, matched, 2000, "fixture sanity: randomized queries selected too little") @@ -585,7 +554,7 @@ func TestMatches_RandomizedAgainstPostFilter(t *testing.T) { // many matches the ascending run selected. func randomizedTrial( t *testing.T, r Reader, corpus *diffCorpus, v *diffVocab, rng *rand.Rand, - corpusSize, trial int, + corpusSize int, ) int { t.Helper() filters := randomFilters(rng, v) @@ -606,34 +575,15 @@ func randomizedTrial( if !desc { matched = len(got) } - requireStrictOrder(t, got, desc, trial) } return matched } -func requireStrictOrder(t *testing.T, got []Match, desc bool, trial int) { - t.Helper() - for i := 1; i < len(got); i++ { - if desc { - require.Greater(t, got[i-1].Ordinal, got[i].Ordinal, - "trial %d: descending ordinals must strictly decrease", trial) - continue - } - require.Less(t, got[i-1].Ordinal, got[i].Ordinal, - "trial %d: ascending ordinals must strictly increase", trial) - } -} - // ───────────────────────── the candidate-set pin ───────────────────────── -// fetchTracer records the ordinals of every FetchEvents call, one entry per -// call, in call order. -// -// Output equality alone cannot see a candidate set that is merely too wide: -// postFilter re-verifies every fetched event against the filters, so a -// superset of the true matches still yields the right stream and only costs -// more I/O. Recording the fetches turns "the right answer" into "the right -// work". +// fetchTracer records the ordinals of every FetchEvents call, in call order. +// Output equality cannot see an over-wide candidate set, since postFilter +// drops the extra fetches; recording them can. type fetchTracer struct { diffReader @@ -647,13 +597,10 @@ func (r fetchTracer) FetchEvents(ctx context.Context, ids []uint32) ([]Payload, var _ Reader = fetchTracer{} -// TestMatches_FetchesOnlyTrueCandidates pins the candidate set itself: the -// ordinals the engine fetches are the query's true matches, in emission order, -// with nothing extra read and nothing skipped. A consumer that stops after a -// page has fetched only the batches that page spans. -// -// The match-all shapes are excluded because they never reach the index: they -// stream FetchRange instead. +// TestMatches_FetchesOnlyTrueCandidates pins that the ordinals the engine +// fetches are exactly the query's matches in emission order, and that a +// consumer stopping after a page fetched only the batches it spans. Match-all +// shapes stream FetchRange and never reach the index. func TestMatches_FetchesOnlyTrueCandidates(t *testing.T) { f := newShapedFixture(t) @@ -707,8 +654,8 @@ func TestMatches_FetchesOnlyTrueCandidates(t *testing.T) { } // traceFetches drives one query and returns the ordinals it fetched in -// emission order. FetchEvents takes ascending ids, so a descending batch is -// fetched flipped; flipping it back recovers the order the stream emits in. +// emission order; a descending batch is fetched flipped, so it is flipped +// back. func traceFetches( t *testing.T, f *shapedFixture, filters []Filter, w IDRange, desc bool, limit int, ) []uint32 { @@ -738,8 +685,7 @@ func pageFloor(limit, answer int) int { // ───────────────────────── the planning step ───────────────────────── -// termBitmap is the shape LookupKeys hands the planner: one materialized -// bitmap per present term, nil for an absent one. +// termBitmap is a materialized term, the shape LookupKeys hands the planner. func termBitmap(ids ...uint32) *roaring.Bitmap { bm := roaring.New() bm.AddMany(ids) @@ -780,37 +726,10 @@ func TestResolveSlabTerms(t *testing.T) { assert.Equal(t, uint64(0), g.est) } -// The rarest group leads the AND however the plan named its groups, so the -// accumulator shrinks fastest and a group that empties it ends the slab before -// the fat groups are read. -func TestResolveSlabFiltersOrdersRarestFirst(t *testing.T) { - sources := []*roaring.Bitmap{ - termBitmap(1, 2, 3, 4, 5, 6, 7, 8), // 0: the fat group - termBitmap(2, 4, 6, 8), // 1 - termBitmap(4, 8), // 2: the rare group - nil, // 3: absent - } - - got := resolveSlabFilters([]termPlan{{{0}, {2}, {1}}}, sources) - require.Len(t, got, 1) - ests := make([]uint64, 0, len(got[0].groups)) - for _, g := range got[0].groups { - ests = append(ests, g.est) - } - assert.Equal(t, []uint64{2, 4, 8}, ests, "the groups are reordered rarest first") - - assert.Empty(t, resolveSlabFilters([]termPlan{{{0}, {3}}}, sources), - "a filter naming a wholly absent group is dropped") - assert.Len(t, resolveSlabFilters([]termPlan{{{0}, {3}}, {{1}}}, sources), 1, - "the drop takes only its own filter") -} - // ───────────────────────── the skip ───────────────────────── -// The skip is invisible in a stream: a walk that evaluated every slab in the -// window yields exactly the same ids. Driving nextBounds directly reports the -// slabs the walk actually opens, which is the claim — and, in the fat case, -// the claim that a provable bound never skips a slab that holds something. +// The skip is invisible in a stream, so this drives nextBounds directly and +// requires the exact slabs the walk opens. func TestSlabStepperSkipsCandidateFreeSlabs(t *testing.T) { defer func(s uint) { slabShift = s }(slabShift) slabShift = 16 @@ -818,9 +737,8 @@ func TestSlabStepperSkipsCandidateFreeSlabs(t *testing.T) { const slabs = 10 window := IDRange{0, slabs * slab} - // A rare term whose three ids sit in slabs 0, 3 and 9; a second rare term - // in slabs 1 and 6; a chunk-sized term holding every id in the window; and - // a term present but empty. + // A rare term in slabs 0, 3 and 9; another in slabs 1 and 6; a term + // holding every id in the window; and a term present but empty. fat := roaring.New() fat.AddRange(uint64(window.Start), uint64(window.End)) sources := []*roaring.Bitmap{ @@ -842,9 +760,8 @@ func TestSlabStepperSkipsCandidateFreeSlabs(t *testing.T) { } } - // The rare term alone opens its own three slabs and no others, in either - // direction, and each one is entered at the candidate rather than at the - // slab's base. + // The rare term alone opens its three slabs and no others, in either + // direction, each entered at the candidate rather than at the slab base. rareAsc := [][2]uint32{{5, slab}, {3*slab + 7, 4 * slab}, {9*slab + 1, 10 * slab}} rareDesc := [][2]uint32{ {9 * slab, 9*slab + 2}, {3 * slab, 3*slab + 8}, {0, 6}, @@ -852,14 +769,13 @@ func TestSlabStepperSkipsCandidateFreeSlabs(t *testing.T) { assert.Equal(t, rareAsc, walk([]termPlan{{{0}}}, false)) assert.Equal(t, rareDesc, walk([]termPlan{{{0}}}, true)) - // ANDing it with a chunk-sized term changes nothing: the AND's bound is - // the strongest of its groups', and the fat group proves none. + // ANDing it with a chunk-sized term changes nothing: the filter's bound is + // the strongest of its groups'. assert.Equal(t, rareAsc, walk([]termPlan{{{0}, {2}}}, false), "a chunk-sized group must not weaken the rare group's bound") assert.Equal(t, rareDesc, walk([]termPlan{{{0}, {2}}}, true)) - // The chunk-sized term alone opens every slab: a bound is a bound, and - // this one proves nothing to skip. + // The chunk-sized term alone opens every slab. full := make([][2]uint32, 0, slabs) for i := range uint32(slabs) { full = append(full, [2]uint32{i * slab, (i + 1) * slab}) @@ -868,8 +784,7 @@ func TestSlabStepperSkipsCandidateFreeSlabs(t *testing.T) { assert.Equal(t, full, walk([]termPlan{{{2}}}, false), "a term holding every id must not skip a slab") - // Across OR-ed filters the union of their slabs is opened, and nothing - // else: slabs 0, 1, 3, 6, 9. + // OR-ed filters open the union of their slabs: 0, 1, 3, 6, 9. assert.Equal(t, [][2]uint32{ {5, slab}, {slab + 1, 2 * slab}, @@ -878,20 +793,15 @@ func TestSlabStepperSkipsCandidateFreeSlabs(t *testing.T) { {9*slab + 1, 10 * slab}, }, walk([]termPlan{{{0}}, {{1}}}, false)) - // A filter whose only term is present but empty ends the walk before the - // first slab, rather than reading all ten. + // A present but empty term ends the walk before the first slab. assert.Empty(t, walk([]termPlan{{{3}}}, false)) assert.Empty(t, walk([]termPlan{{{3}}}, true)) } -// TestMatches_RareTermsSpanSlabs is the oracle gate on the skip. The shapes a -// rare term dominates — five ids spread over the whole corpus, alone and ANDed -// with a chunk-sized term — run at slab widths that put hundreds of -// candidate-free slabs between consecutive ids, where the walk skips almost -// everything and a bound that over-skipped would silently lose matches. -// -// The windows start and end between rare ids, so the first and last slab the -// walk seeks to are clipped by the window rather than by a candidate. +// TestMatches_RareTermsSpanSlabs is the oracle gate on the skip: the shapes +// a rare term dominates run at slab widths that put hundreds of empty slabs +// between consecutive ids. The windows start and end between rare ids, so +// the first and last slab are clipped by the window rather than a candidate. func TestMatches_RareTermsSpanSlabs(t *testing.T) { f := newShapedFixture(t) r := diffReader{f.corpus} @@ -914,8 +824,7 @@ func TestMatches_RareTermsSpanSlabs(t *testing.T) { for _, sh := range shapes { all := matchingEvents(t, f.corpus, sh.filters) - // 64-, 1024- and 8192-wide slabs put 1093, 68 and 8 seams inside the - // corpus against the 5 ids the rare term holds. + // 64-, 1024- and 8192-wide slabs. for _, shift := range []uint{6, 10, 13} { slabShift = shift for _, w := range windows { From 9f032725652eba2e4dd607d41315790fe9edbf6a Mon Sep 17 00:00:00 2001 From: tamirms Date: Wed, 9 Sep 2026 20:20:03 +0100 Subject: [PATCH 39/41] event: preallocate the value-search results in the roaring race gate Fixes the two prealloc findings by moving the NextValue/PreviousValue loop into one helper that sizes its result up front. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01M23ZvEkyrFjyUUbm7zLwob --- .../stores/event/roaring_contract_test.go | 27 ++++++++++--------- 1 file changed, 14 insertions(+), 13 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go index 0bab7e101..17adc62ff 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go @@ -200,12 +200,7 @@ func TestRoaringContract_ConcurrentReaders(t *testing.T) { seq := freshRange(0, 4<<16) seq.AndAny(shared...) wantAndAny := bitmapBytes(t, seq) - var wantSearch []int64 - for _, bm := range shared { - for _, target := range targets { - wantSearch = append(wantSearch, bm.NextValue(target), bm.PreviousValue(target)) - } - } + wantSearch := valueSearches(shared, targets) gotAndAny := make([][]byte, goroutines) gotSearch := make([][]int64, goroutines) @@ -222,13 +217,7 @@ func TestRoaringContract_ConcurrentReaders(t *testing.T) { panic(err) } gotAndAny[g] = b - var search []int64 - for _, bm := range shared { - for _, target := range targets { - search = append(search, bm.NextValue(target), bm.PreviousValue(target)) - } - } - gotSearch[g] = search + gotSearch[g] = valueSearches(shared, targets) } }) } @@ -242,6 +231,18 @@ func TestRoaringContract_ConcurrentReaders(t *testing.T) { } } +// valueSearches runs NextValue and PreviousValue for every target on every +// bitmap. +func valueSearches(bms []*roaring.Bitmap, targets []uint32) []int64 { + out := make([]int64, 0, 2*len(bms)*len(targets)) + for _, bm := range bms { + for _, target := range targets { + out = append(out, bm.NextValue(target), bm.PreviousValue(target)) + } + } + return out +} + // TestRoaringContract_ValueSearchIsInclusiveAndReadOnly pins NextValue and // PreviousValue, which the slab walk proves its skips with: an answer past // the target, or a -1 with ids still to come, would silently drop matches. From 39a0c94e7112e11076be0113ba7652e8677bc4ac Mon Sep 17 00:00:00 2001 From: tamirms Date: Fri, 11 Sep 2026 21:56:10 +0100 Subject: [PATCH 40/41] event: evaluate each slab with one FastAnd per plan eval narrowed a per-filter accumulator through the filter's groups one AndAny at a time, so every filter paid an 8 KiB accumulator on every slab even when its groups had nothing in common there, and a topic-count range paid its union on top. A filter now yields plans that are conjunctions of single terms, one per bucket of a range, since A and (b or c) is (A and b) or (A and c) and the union across plans keeps the or; the planner drops a plan that repeats an earlier one. The slab engine then holds one bitmap per term, runs one FastAnd over the slab's id range and the terms per plan, keeps its bounds as the searches return them, and counts each term's cardinality once. The slab bitmap is built once per slab and shared by every plan. With the pinned roaring, FastAnd intersects pairwise and the cost is unchanged or lower. With roaring's count-first intersections (RoaringBitmap/roaring#566 and the FastAnd follow-up built on it) an empty result allocates nothing, which is what bounds the slab walk on filter lists that match nothing; the version bump will add the contract test for that. roaring_contract_test.go now pins that FastAnd leaves its inputs untouched, returns fresh containers on the two- and many-input paths over whole and partial slab ranges, and agrees with the pairwise And chain for every subset of the container kinds; that a slab range built with AddRange is stored as run containers; and runs FastAnd in the concurrent-readers race gate. The AndAny pins go, since the engine no longer calls it. The shaped differential matrix gains a filter with a term and a count range, and a planner test covers the fan-out and the dedup. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01M23ZvEkyrFjyUUbm7zLwob --- .../rpcv2/stores/event/concurrent_bitmaps.go | 4 +- .../internal/rpcv2/stores/event/match.go | 93 ++++--- .../internal/rpcv2/stores/event/match_test.go | 8 +- .../stores/event/roaring_contract_test.go | 200 ++++++++------ .../internal/rpcv2/stores/event/slab_match.go | 259 +++++++----------- .../rpcv2/stores/event/slab_match_test.go | 81 +++--- 6 files changed, 313 insertions(+), 332 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go index 3c54ee804..102e2f841 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/concurrent_bitmaps.go @@ -97,8 +97,8 @@ func NewConcurrentBitmapsFromBitmaps(b Bitmaps) *ConcurrentBitmaps { // // Safe: any non-mutating read (Contains, GetCardinality, Iterator, // NextValue, PreviousValue, ToArray, IsEmpty, Minimum, Maximum) and -// passing it as an argument to AndAny or And; roaring_contract_test.go -// pins these against the pinned roaring version. +// passing it to FastAnd beside at least one other input; +// roaring_contract_test.go pins these against the pinned roaring version. // // A Get that starts after an AddTo returns sees that AddTo's IDs, and // the pointer stays valid for as long as the caller holds it. The diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go index cebc2bee2..4a471dd95 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match.go @@ -85,9 +85,9 @@ func (f TopicCountFilter) termKeys() []TermKey { // valueTermKeys returns one term per constrained value field // (contract ID, event type, topics): the single enumeration -// termGroups and CountDistinctTerms share, so the two cannot drift +// termPlans and CountDistinctTerms share, so the two cannot drift // over which values a filter names. The topic-count buckets are not -// value terms; termGroups adds them separately and the budget does +// value terms; termPlans adds them separately and the budget does // not count them. func (f *Filter) valueTermKeys() []TermKey { var keys []TermKey @@ -106,25 +106,30 @@ func (f *Filter) valueTermKeys() []TermKey { return keys } -// termGroups returns the indexed terms this filter constrains, grouped -// by field: the bitmaps within a group are OR-ed and the groups are -// AND-ed. Only the topic-count group ever holds more than one term. -func (f *Filter) termGroups() [][]TermKey { - var groups [][]TermKey - for _, key := range f.valueTermKeys() { - groups = append(groups, []TermKey{key}) - } - // A constrained topic position already implies an "at least" count at - // or below it, since a topic term is only indexed for events carrying - // that position. Skipping the group there keeps the common - // ["a", "**"] shape from OR-ing chunk-sized bucket bitmaps; the - // post-filter enforces the count either way. +// termPlans returns the index terms a candidate must carry, one +// conjunction per plan. A filter yields one plan, its value terms, unless +// it constrains the topic count, when it yields one plan per bucket, a +// single one for an exact count: A and (b or c) is (A and b) or (A and +// c), and the union across plans keeps the or. A constrained topic +// position already implies an "at least" count at or below it, since a +// topic term is only indexed for events carrying that position. Skipping +// the buckets there keeps the common ["a", "**"] shape from fanning out +// over chunk-sized bucket bitmaps; the post-filter enforces the count +// either way. +func (f *Filter) termPlans() [][]TermKey { + values := f.valueTermKeys() + var buckets []TermKey if !f.impliesTopicCount() { - if keys := f.TopicCount.termKeys(); len(keys) > 0 { - groups = append(groups, keys) - } + buckets = f.TopicCount.termKeys() + } + if len(buckets) == 0 { + return [][]TermKey{values} } - return groups + plans := make([][]TermKey, 0, len(buckets)) + for _, bucket := range buckets { + plans = append(plans, append(slices.Clone(values), bucket)) + } + return plans } // impliesTopicCount reports whether f's constrained topic positions @@ -214,9 +219,9 @@ type Match struct { Ordinal uint32 } -// termPlan is a filter's termGroups resolved to slots in the batched -// term lookup's result. -type termPlan [][]int +// termPlan is one of a filter's termPlans, as slots in the batched term +// lookup's result. +type termPlan []int // batchSizes resolves the first and following internal batch sizes from the // caller's hint. The hint is a page size the handler has already validated, @@ -280,9 +285,9 @@ func Matches( return } st := newSlabStepper(plans, sources, window, descending) - // No filter survived group resolution, so nothing can match and no - // slab is worth evaluating. - if len(st.filters) == 0 { + // No plan survived term resolution, so nothing can match and no slab + // is worth evaluating. + if len(st.plans) == 0 { return } streamSlabs(ctx, r, filters, st, descending, firstBatch, yield) @@ -314,10 +319,10 @@ func validateMatchCall(ctx context.Context, r Reader, filters []Filter, window I return nil } -// planIndexTerms dedupes the terms the filters name and resolves each -// filter's groups to slots in the single batched lookup that follows; -// it runs before any index I/O. plans[i] holds the slots filter i -// needs: terms within a group are OR-ed and groups are AND-ed. +// planIndexTerms maps every filter's plans to slots in the single batched +// lookup that follows; it runs before any index I/O. A plan that repeats +// an earlier one is dropped: plans only pick candidates, and the +// post-filter still runs every filter. // // matchAll reports that some filter, or the empty slice, constrains // nothing, so the caller streams the window directly rather than @@ -327,21 +332,25 @@ func planIndexTerms(filters []Filter) ([]termPlan, []TermKey, bool) { return nil, nil, true } var uniqueKeys []TermKey - plans := make([]termPlan, len(filters)) + plans := make([]termPlan, 0, len(filters)) for i := range filters { - groups := filters[i].termGroups() - if len(groups) == 0 { - return nil, nil, true - } - plan := make(termPlan, len(groups)) - for g, keys := range groups { - slots := make([]int, len(keys)) + for _, keys := range filters[i].termPlans() { + if len(keys) == 0 { + return nil, nil, true + } + plan := make(termPlan, len(keys)) for j, key := range keys { - slots[j] = indexOfOrAddTerm(&uniqueKeys, key) + plan[j] = indexOfOrAddTerm(&uniqueKeys, key) + } + // Slots follow field order, so equal plans are equal slices. + dup := slices.ContainsFunc(plans, func(p termPlan) bool { + return slices.Equal(p, plan) + }) + if dup { + continue } - plan[g] = slots + plans = append(plans, plan) } - plans[i] = plan } return plans, uniqueKeys, false } @@ -435,9 +444,9 @@ func ValidateFilters(filters []Filter) error { // filters name, deduped by field and value together: one contract ID // in five filters counts once, the same bytes in two topic positions // count twice. Topic-count buckets are excluded: they are an -// implementation detail of the engine's grouping, not a value the +// implementation detail of the engine's plans, not a value the // client named. Exported for the v2 handler's term-budget check. It -// lives here, beside termGroups, so the budget and the engine's +// lives here, beside termPlans, so the budget and the engine's // lookups agree on what a value term is: TermKey over the store's // canonical bytes. func CountDistinctTerms(filters []Filter) int { diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go index 2fe5a62bf..496374604 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/match_test.go @@ -1144,10 +1144,10 @@ func TestQuery_InvalidFilterRejected(t *testing.T) { // - match-all desc + cap → streamRange top-down, slices.Backward // - single-filter (contractID) → one LookupKeys term, then the // ascending slab walk -// - multi-term filter (AND) → AndAny per group over cold bitmaps -// - cross-filter (OR) → the in-place Or across filters -// - ledger range + filter → the window clipped into each slab's -// accumulator +// - multi-term filter (AND) → one FastAnd per plan over cold bitmaps +// - cross-filter (OR) → the in-place Or across plans +// - ledger range + filter → the window clipped to each slab's +// id range // - descending + range + cap → the slab walk run high to low // // What we don't replay against cold: diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go index 17adc62ff..25d637d37 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/roaring_contract_test.go @@ -2,10 +2,10 @@ package event // roaring_contract_test.go pins the roaring properties the event index rests // on. ConcurrentBitmaps.Get publishes one bitmap to every concurrent reader -// and never mutates it, and the match path hands those bitmaps to AndAny, +// and never mutates it, and the match path hands those bitmaps to FastAnd, // NextValue and PreviousValue from many goroutines at once. That is sound -// only while AndAny writes nothing but its receiver and shares no storage -// with its arguments, and while the searches are read-only, inclusive of the +// only while FastAnd reads its arguments and returns containers that share +// no storage with them, and the searches are read-only, inclusive of the // target, and return -1 for none. The tests are written against roaring // alone, so they are the first thing to run on a version bump. // @@ -91,101 +91,131 @@ func requireUnchanged(t *testing.T, bms []*roaring.Bitmap, before [][]byte, op s } } -// freshRange is the receiver shape the match path hands to AndAny: a bitmap -// this call allocated, holding one contiguous id range, shared with nobody. +// freshRange is the slab window the match path hands to FastAnd: one +// contiguous id range, in a bitmap the test may write. func freshRange(lo, hi uint64) *roaring.Bitmap { - acc := roaring.New() - acc.AddRange(lo, hi) - return acc + bm := roaring.New() + bm.AddRange(lo, hi) + return bm } -// TestRoaringContract_AggregationDoesNotMutateInputs pins that AndAny, and -// And, which AndAny delegates a single argument to, leave shared arguments -// byte-identical. +// scratchBitmap is a bitmap the test owns outright and may write: one array +// container of scattered ids. +func scratchBitmap() *roaring.Bitmap { + scratch := roaring.New() + for i := range uint32(3000) { + scratch.Add(i * 3) + } + return scratch +} + +// windowRanges are slab windows that take different paths inside roaring: a +// whole high key, a partial one, one across a key boundary, a span over +// several keys, one the inputs do not reach, and the four-id span that is the +// smallest range roaring keeps as a run. +var windowRanges = []struct { + name string + lo, hi uint64 +}{ + {"whole key", 0, 1 << 16}, + {"partial key", 1000, 40_000}, + {"key boundary", 65_000, 66_000}, + {"several keys", 0, 4 << 16}, + {"disjoint key", 40 << 16, 41 << 16}, + {"four ids", 8000, 8004}, +} + +// TestRoaringContract_AggregationDoesNotMutateInputs pins that FastAnd +// leaves shared arguments byte-identical and equals the pairwise And chain, +// for every non-empty subset of the shared bitmaps beside the window: each +// container kind meets the window alone, in pairs and all together, which +// is every kernel a plan of one to seven terms can reach. func TestRoaringContract_AggregationDoesNotMutateInputs(t *testing.T) { t.Parallel() - // Ranges that take different paths inside roaring: a whole high key, a - // partial one, one the inputs do not reach, and a span across several. - ranges := []struct { - name string - lo, hi uint64 - }{ - {"whole key", 0, 1 << 16}, - {"partial key", 1000, 40_000}, - {"key boundary", 65_000, 66_000}, - {"several keys", 0, 4 << 16}, - {"disjoint key", 40 << 16, 41 << 16}, - } - - for _, w := range ranges { - t.Run("AndAny/"+w.name, func(t *testing.T) { + for _, w := range windowRanges { + t.Run(w.name, func(t *testing.T) { t.Parallel() - for n := 1; n <= 4; n++ { - shared := sharedBitmaps(t)[:n] - before := bitmapImages(t, shared) - acc := freshRange(w.lo, w.hi) - acc.AndAny(shared...) - requireUnchanged(t, shared, before, "AndAny") - - // The accumulator must equal the definition AndAny documents, - // so a silently wrong answer cannot pass as an unmutated one. - // The leading empty bitmap keeps FastOr off its single-input - // Clone shortcut, whose result would share containers with a - // copy-on-write argument. - want := roaring.FastOr(append([]*roaring.Bitmap{roaring.New()}, shared...)...) - want.And(freshRange(w.lo, w.hi)) - require.Equal(t, bitmapBytes(t, want), bitmapBytes(t, acc), - "AndAny must equal x.And(FastOr(args)) for %d args", n) + shared := sharedBitmaps(t) + for pick := 1; pick < 1< 0 -} - -// resolveSlabFilters resolves every filter's groups, drops the filters that -// named an entirely absent group, and orders each survivor's groups rarest -// first so the accumulator shrinks fastest. -func resolveSlabFilters(plans []termPlan, sources []*roaring.Bitmap) []slabFilter { - out := make([]slabFilter, 0, len(plans)) + absent := func(slot int) bool { return sources[slot] == nil } + out := make([]slabPlan, 0, len(plans)) for _, plan := range plans { - groups := make([]slabTerms, 0, len(plan)) - missed := false - for _, slots := range plan { - g, ok := resolveSlabTerms(sources, slots) - if !ok { - missed = true - break - } - groups = append(groups, g) - } - if missed { + if slices.ContainsFunc(plan, absent) { continue } - slices.SortStableFunc(groups, func(a, b slabTerms) int { - return cmp.Compare(a.est, b.est) + slots := slices.Clone(plan) + slices.SortStableFunc(slots, func(a, b int) int { + return cmp.Compare(cards[a], cards[b]) }) - out = append(out, slabFilter{groups: groups}) + p := make(slabPlan, 0, len(slots)) + for _, slot := range slots { + p = append(p, sources[slot]) + } + out = append(out, p) } return out } -// eval returns f's matches inside [lo, hi) as a fresh bitmap the caller owns, -// or nil when there are none. The accumulator starts as the slab range and is -// narrowed in place by one AndAny per group, which is x.And(FastOr(args)) -// without the intermediate union. A filter always names at least one group; -// one that names none takes the match-all path upstream. -func (f *slabFilter) eval(lo, hi uint32) *roaring.Bitmap { - acc := roaring.New() - acc.AddRange(uint64(lo), uint64(hi)) - for i := range f.groups { - acc.AndAny(f.groups[i].bitmaps...) - if acc.IsEmpty() { - return nil - } +// eval returns p's matches inside slab, the bitmap of one slab's ids, as a +// fresh bitmap the caller owns, or nil when there are none. The slab and +// the plan's terms go to roaring in one FastAnd call, so that it can find +// an empty intersection before allocating a container for it. +func (p slabPlan) eval(slab *roaring.Bitmap) *roaring.Bitmap { + ops := make([]*roaring.Bitmap, 0, len(p)+1) + ops = append(ops, slab) + ops = append(ops, p...) + res := roaring.FastAnd(ops...) + if res.IsEmpty() { + return nil } - return acc + return res } // Bounds. A slab is skipped only on a bound proved from the term bitmaps // themselves. Ascending, at position pos: // -// - a group's bound is the smallest NextValue(pos) over its terms, since a -// candidate lies in at least one of them; no term holding one proves the -// group, and so its filter, matches nothing from pos on; -// - a filter's bound is the largest of its groups' bounds, since a -// candidate satisfies every group; -// - the query's bound is the smallest of the live filters' bounds. +// - a term's bound is NextValue(pos), the first id at or above pos it +// holds; a term holding none proves its plan matches nothing from pos on; +// - a plan's bound is the largest of its terms' bounds, since a candidate +// satisfies every term; +// - the query's bound is the smallest of the live plans' bounds. // // Descending mirrors this with PreviousValue and the min and max swapped. // The post-filter only drops candidates, so a bound proved on the index @@ -126,89 +103,53 @@ func (f *slabFilter) eval(lo, hi uint32) *roaring.Bitmap { // return -1 for none, and do not write the bitmap they search; // roaring_contract_test.go pins all three properties. -// boundRetired marks a filter whose terms ran out ahead of the cursor. The -// cursor never comes back, so the filter is dropped for the rest of the walk. -// It sorts below every id, so the "outside this slab" test covers it. +// boundRetired is the bound of a plan whose terms ran out ahead of the +// cursor: the cursor never comes back, so the plan is dropped for the rest +// of the walk. It is roaring's own "none" and sorts below every id, so the +// "outside this slab" test covers it. const boundRetired = int64(-1) -// nextBound is the group's bound: the smallest id at or above pos that any of -// its terms holds. ok is false when none does. -func (g *slabTerms) nextBound(pos uint32) (uint32, bool) { - if len(g.bitmaps) == 0 { - // Unreachable, since resolveSlabTerms never builds an empty group; - // an empty group constrains nothing and so proves no bound. - return pos, true - } - best := int64(-1) - for _, bm := range g.bitmaps { +// nextBound is the plan's bound: the largest of its terms' first ids at or +// above pos, or boundRetired as soon as one term holds none, which proves +// the plan is done. +func (p slabPlan) nextBound(pos uint32) int64 { + bound := int64(pos) + for _, bm := range p { v := bm.NextValue(pos) - if v >= 0 && (best < 0 || v < best) { - best = v + if v < 0 { + return boundRetired } + bound = max(bound, v) } - if best < 0 { - return 0, false - } - return uint32(best), true + return bound } -// prevBound is nextBound descending: the largest id at or below pos. -func (g *slabTerms) prevBound(pos uint32) (uint32, bool) { - if len(g.bitmaps) == 0 { - return pos, true - } - best := int64(-1) - for _, bm := range g.bitmaps { +// prevBound is nextBound descending: the smallest of the terms' last ids at +// or below pos. +func (p slabPlan) prevBound(pos uint32) int64 { + bound := int64(pos) + for _, bm := range p { v := bm.PreviousValue(pos) - if v > best { - best = v - } - } - if best < 0 { - return 0, false - } - return uint32(best), true -} - -// nextBound is the filter's bound: the largest of its groups' bounds. ok is -// false as soon as one group proves the filter is done. -func (f *slabFilter) nextBound(pos uint32) (uint32, bool) { - bound := pos - for i := range f.groups { - b, ok := f.groups[i].nextBound(pos) - if !ok { - return 0, false - } - bound = max(bound, b) - } - return bound, true -} - -// prevBound is nextBound descending: the smallest of the groups' bounds. -func (f *slabFilter) prevBound(pos uint32) (uint32, bool) { - bound := pos - for i := range f.groups { - b, ok := f.groups[i].prevBound(pos) - if !ok { - return 0, false + if v < 0 { + return boundRetired } - bound = min(bound, b) + bound = min(bound, v) } - return bound, true + return bound } // slabStepper walks one query's slabs in emission order, evaluating a slab // only when the consumer has drained the previous one. type slabStepper struct { - filters []slabFilter - window IDRange - desc bool + plans []slabPlan + window IDRange + desc bool - // bounds holds, per filter, the id its next candidate is proved to lie at + // bounds holds, per plan, the id its next candidate is proved to lie at // or past (at or before, descending), or boundRetired once it has none // left. A bound is monotone in the walk direction, so one proved earlier // still holds at every position up to it: it is re-proved only once the - // cursor reaches it, and a filter whose bound lies past the current slab + // cursor reaches it, and a plan whose bound lies past the current slab // is not evaluated there. A held bound can be weaker than a fresh one, // which costs an evaluated slab, never a match. Bounds start at the // cursor so the first step proves them all. @@ -229,40 +170,38 @@ func newSlabStepper( plans []termPlan, sources []*roaring.Bitmap, window IDRange, descending bool, ) *slabStepper { s := &slabStepper{ - filters: resolveSlabFilters(plans, sources), - window: window, - desc: descending, + plans: resolveSlabPlans(plans, sources), + window: window, + desc: descending, } if descending { s.cursor = window.End } else { s.cursor = window.Start } - s.bounds = make([]int64, len(s.filters)) + s.bounds = make([]int64, len(s.plans)) for i := range s.bounds { s.bounds[i] = int64(s.cursor) } return s } -// seekAsc returns the lowest position at or above pos where some filter can -// still match, or false when none can inside the window. A filter is asked -// again only once pos has reached its held bound; a filter with no bound -// left is retired rather than ending the walk. +// seekAsc returns the lowest position at or above pos where some plan can +// still match, or false when none can inside the window. A plan is asked +// again only once pos has reached its held bound; a plan with no bound left +// is retired rather than ending the walk. func (s *slabStepper) seekAsc(pos uint32) (uint32, bool) { var best int64 found := false - for i := range s.filters { + for i := range s.plans { if s.bounds[i] == boundRetired { continue } if s.bounds[i] <= int64(pos) { - b, ok := s.filters[i].nextBound(pos) - if !ok { - s.bounds[i] = boundRetired + s.bounds[i] = s.plans[i].nextBound(pos) + if s.bounds[i] == boundRetired { continue } - s.bounds[i] = int64(b) } if !found || s.bounds[i] < best { best, found = s.bounds[i], true @@ -274,23 +213,21 @@ func (s *slabStepper) seekAsc(pos uint32) (uint32, bool) { return uint32(best), true } -// seekDesc is seekAsc mirrored: the largest of the filters' bounds at or below +// seekDesc is seekAsc mirrored: the largest of the plans' bounds at or below // hi-1, returned as the exclusive high bound the walk resumes at. func (s *slabStepper) seekDesc(hi uint32) (uint32, bool) { pos := hi - 1 var best int64 found := false - for i := range s.filters { + for i := range s.plans { if s.bounds[i] == boundRetired { continue } if s.bounds[i] >= int64(pos) { - b, ok := s.filters[i].prevBound(pos) - if !ok { - s.bounds[i] = boundRetired + s.bounds[i] = s.plans[i].prevBound(pos) + if s.bounds[i] == boundRetired { continue } - s.bounds[i] = int64(b) } if !found || s.bounds[i] > best { best, found = s.bounds[i], true @@ -345,18 +282,20 @@ func (s *slabStepper) nextBounds() (uint32, uint32, bool) { return lo, hi, true } -// evalSlab unions the per-filter results for [lo, hi), or returns nil when -// the slab holds nothing. A filter whose bound lies outside the slab is +// evalSlab unions the per-plan results for [lo, hi), or returns nil when +// the slab holds nothing. A plan whose bound lies outside the slab is // skipped, since the bound already proves it matches nothing here. The // results are this call's own bitmaps, so the union runs in place into the // first of them. func (s *slabStepper) evalSlab(lo, hi uint32) *roaring.Bitmap { var acc *roaring.Bitmap - for i := range s.filters { + slab := roaring.New() + slab.AddRange(uint64(lo), uint64(hi)) + for i := range s.plans { if b := s.bounds[i]; b < int64(lo) || b >= int64(hi) { continue } - bm := s.filters[i].eval(lo, hi) + bm := s.plans[i].eval(slab) switch { case bm == nil: case acc == nil: diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go index dff37d37f..f54040b2a 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match_test.go @@ -389,6 +389,7 @@ func (f *shapedFixture) namedShapes() []namedShape { {"match all wildcard filter", []Filter{{}}}, {"match all beside constrained", []Filter{{EventType: &sysType}, {}}}, {"exact topic count", []Filter{{TopicCount: TopicCountFilter{Count: 2, Exact: true}}}}, + {"term and count range", []Filter{{ContractID: f.vocab.contracts[0], TopicCount: TopicCountFilter{Count: 1}}}}, } } @@ -692,38 +693,42 @@ func termBitmap(ids ...uint32) *roaring.Bitmap { return bm } -// What a group reports about itself: presence, and its summed weight. -func TestResolveSlabTerms(t *testing.T) { +// resolveSlabPlans keeps the plans whose every term is present, orders each +// one's terms rarest first, and holds the lookup's bitmaps rather than +// copying them. +func TestResolveSlabPlans(t *testing.T) { sources := []*roaring.Bitmap{ termBitmap(1, 2), nil, // absent termBitmap(2, 3, 4), termBitmap(), // present, holding nothing } - - g, ok := resolveSlabTerms(sources, []int{0}) - require.True(t, ok) - assert.Equal(t, uint64(2), g.est) - assert.Len(t, g.bitmaps, 1, "the group holds the term's bitmap itself") - assert.Same(t, sources[0], g.bitmaps[0], "the lookup's bitmap is held, not copied") - - g, ok = resolveSlabTerms(sources, []int{0, 2}) - require.True(t, ok) - assert.Equal(t, uint64(5), g.est, "a group's terms sum, overlaps double-counted") - assert.Len(t, g.bitmaps, 2) - - g, ok = resolveSlabTerms(sources, []int{1, 2}) - require.True(t, ok) - assert.Equal(t, uint64(3), g.est, "an absent term adds nothing") - assert.Len(t, g.bitmaps, 1, "an absent term is not held") - - g, ok = resolveSlabTerms(sources, []int{3}) - require.True(t, ok, "a present-but-empty term keeps its group alive") - assert.Equal(t, uint64(0), g.est) - - g, ok = resolveSlabTerms(sources, []int{1}) - assert.False(t, ok, "a group of absent terms drops its filter") - assert.Equal(t, uint64(0), g.est) + got := resolveSlabPlans([]termPlan{{0}, {2, 0}, {1, 2}, {3}, {1}}, sources) + require.Len(t, got, 3, "a plan naming an absent term matches nothing and is dropped") + require.Len(t, got[0], 1) + assert.Same(t, sources[0], got[0][0], "the lookup's bitmap is held, not copied") + require.Len(t, got[1], 2) + assert.Same(t, sources[0], got[1][0], "terms are ordered rarest first") + assert.Same(t, sources[2], got[1][1]) + require.Len(t, got[2], 1) + assert.Same(t, sources[3], got[2][0], "a present but empty term keeps its plan") +} + +// A topic-count range fans out to one plan per bucket, and a repeated filter +// adds no plan. +func TestPlanIndexTermsSplitsRanges(t *testing.T) { + cid := []byte{0xAB} + ranged := Filter{ContractID: cid, TopicCount: TopicCountFilter{Count: 2}} + plans, keys, matchAll := planIndexTerms([]Filter{ranged, ranged, {ContractID: cid}}) + require.False(t, matchAll) + buckets := len(TopicCountTermKeysAtLeast(2)) + require.Len(t, plans, buckets+1) + assert.Len(t, keys, buckets+1) + for _, p := range plans[:buckets] { + assert.Equal(t, 0, p[0], "the contract slot leads every plan of the range") + assert.Len(t, p, 2) + } + assert.Equal(t, termPlan{0}, plans[buckets]) } // ───────────────────────── the skip ───────────────────────── @@ -766,14 +771,14 @@ func TestSlabStepperSkipsCandidateFreeSlabs(t *testing.T) { rareDesc := [][2]uint32{ {9 * slab, 9*slab + 2}, {3 * slab, 3*slab + 8}, {0, 6}, } - assert.Equal(t, rareAsc, walk([]termPlan{{{0}}}, false)) - assert.Equal(t, rareDesc, walk([]termPlan{{{0}}}, true)) + assert.Equal(t, rareAsc, walk([]termPlan{{0}}, false)) + assert.Equal(t, rareDesc, walk([]termPlan{{0}}, true)) - // ANDing it with a chunk-sized term changes nothing: the filter's bound is - // the strongest of its groups'. - assert.Equal(t, rareAsc, walk([]termPlan{{{0}, {2}}}, false), - "a chunk-sized group must not weaken the rare group's bound") - assert.Equal(t, rareDesc, walk([]termPlan{{{0}, {2}}}, true)) + // ANDing it with a chunk-sized term changes nothing: the plan's bound is + // the strongest of its terms'. + assert.Equal(t, rareAsc, walk([]termPlan{{0, 2}}, false), + "a chunk-sized term must not weaken the rare term's bound") + assert.Equal(t, rareDesc, walk([]termPlan{{0, 2}}, true)) // The chunk-sized term alone opens every slab. full := make([][2]uint32, 0, slabs) @@ -781,21 +786,21 @@ func TestSlabStepperSkipsCandidateFreeSlabs(t *testing.T) { full = append(full, [2]uint32{i * slab, (i + 1) * slab}) } assert.Len(t, full, slabs) - assert.Equal(t, full, walk([]termPlan{{{2}}}, false), + assert.Equal(t, full, walk([]termPlan{{2}}, false), "a term holding every id must not skip a slab") - // OR-ed filters open the union of their slabs: 0, 1, 3, 6, 9. + // OR-ed plans open the union of their slabs: 0, 1, 3, 6, 9. assert.Equal(t, [][2]uint32{ {5, slab}, {slab + 1, 2 * slab}, {3*slab + 7, 4 * slab}, {6*slab + 3, 7 * slab}, {9*slab + 1, 10 * slab}, - }, walk([]termPlan{{{0}}, {{1}}}, false)) + }, walk([]termPlan{{0}, {1}}, false)) // A present but empty term ends the walk before the first slab. - assert.Empty(t, walk([]termPlan{{{3}}}, false)) - assert.Empty(t, walk([]termPlan{{{3}}}, true)) + assert.Empty(t, walk([]termPlan{{3}}, false)) + assert.Empty(t, walk([]termPlan{{3}}, true)) } // TestMatches_RareTermsSpanSlabs is the oracle gate on the skip: the shapes From c9b4053d58e8ab681f71bf4256b7f78060e2264f Mon Sep 17 00:00:00 2001 From: tamirms Date: Sat, 12 Sep 2026 07:33:14 +0100 Subject: [PATCH 41/41] event: say what the pinned FastAnd allocates in eval's doc The sentence read as if an empty plan cost no container today; at roaring v2.26.0 FastAnd intersects pairwise and allocates the first intermediate, and the count-first behaviour arrives with the roaring bump. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01M23ZvEkyrFjyUUbm7zLwob --- cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go index eb297c903..51aec3a04 100644 --- a/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go +++ b/cmd/stellar-rpc/internal/rpcv2/stores/event/slab_match.go @@ -75,8 +75,10 @@ func resolveSlabPlans(plans []termPlan, sources []*roaring.Bitmap) []slabPlan { // eval returns p's matches inside slab, the bitmap of one slab's ids, as a // fresh bitmap the caller owns, or nil when there are none. The slab and -// the plan's terms go to roaring in one FastAnd call, so that it can find -// an empty intersection before allocating a container for it. +// the plan's terms go to roaring in one FastAnd call, the shape in which a +// count-first FastAnd can reject an empty intersection without allocating +// for it; the pinned v2.26.0 still intersects pairwise and allocates the +// first intermediate, so that saving arrives with the roaring bump. func (p slabPlan) eval(slab *roaring.Bitmap) *roaring.Bitmap { ops := make([]*roaring.Bitmap, 0, len(p)+1) ops = append(ops, slab)