Skip to content

Commit 69fc5ce

Browse files
perf: stream equality and range index scans instead of materializing every id
NitriteIndexer.findByFilter returns a LinkedHashSet of every matching id, so find(k = v).firstOrNull() built the whole match set before handing back one row, and a bounded page paid for the entire result. On a non-unique index over a low-cardinality field that set is a large fraction of the collection on every lookup. The composite layout already keeps its rows in key order, so the two plan shapes that map onto one bounded walk of it, an equality on the indexed field and a two-sided range on it, are now served by a lazy iterator that starts at the first key inside the bounds and stops at the first key outside them. It honours the plan's reverse scan order by visiting the key groups backwards while reading each group forwards, exactly as the materialized scan orders them, skips entries removed in an open transaction, and returns a document indexed under several keys once. NitriteIndex.findNitriteIdStream and NitriteIndexer.findByFilterStream are new default methods returning null, so every other index type, plugin indexer and plan shape keeps the materialized path unchanged. ReadOperations prefers the stream when one is offered; the covered-count shortcut that lets size() answer without fetching documents is kept by counting the streamed ids on demand, so size() still reads the index only. Tests compare the stream with the materialized scan for equality, range and reverse order, check the shapes it declines, show with a spied map that only one key is read for the first row, and exercise counts, paging, descending order, multi-valued fields and removals through the public API. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
1 parent 38caf34 commit 69fc5ce

9 files changed

Lines changed: 487 additions & 30 deletions

File tree

‎nitrite/src/main/java/org/dizitart/no2/collection/operation/ReadOperations.java‎

Lines changed: 41 additions & 26 deletions
Original file line numberDiff line numberDiff line change
@@ -111,35 +111,43 @@ private void prepareLogicalFilter(LogicalFilter logicalFilter) {
111111
}
112112

113113
private DocumentCursor createCursor(FindPlan findPlan) {
114-
// -1 means "not an index scan"; the index branch records the exact id-set size here.
115-
long[] indexedIdCount = { -1 };
116-
RecordStream<Pair<NitriteId, Document>> recordStream = findSuitableStream(findPlan, indexedIdCount);
114+
IndexScan scan = new IndexScan();
115+
RecordStream<Pair<NitriteId, Document>> recordStream = findSuitableStream(findPlan, scan);
117116
DocumentStream cursor = new DocumentStream(recordStream, processorChain);
118117
cursor.setFindPlan(findPlan);
119-
cursor.setCoveredCount(computeCoveredCount(findPlan, indexedIdCount[0]));
118+
if (isCountCovered(findPlan)) {
119+
if (findPlan.getIndexDescriptor() == null) {
120+
// pure full scan over the whole collection
121+
cursor.setCoveredCount(nitriteMap.size());
122+
} else if (scan.lazyStream != null) {
123+
// the ids are read lazily: count them from the index only if size() is asked
124+
cursor.setCoveredCountSupplier(scan.lazyStream::countIds);
125+
} else if (scan.idCount >= 0) {
126+
// the index supplied the exact matching id set
127+
cursor.setCoveredCount(scan.idCount);
128+
}
129+
}
120130
return cursor;
121131
}
122132

133+
/** What an index scan handed back: an exact id count, or the lazy stream, or neither. */
134+
private static final class IndexScan {
135+
private long idCount = -1;
136+
private IndexedStream lazyStream;
137+
}
138+
123139
/**
124140
* Returns the exact match count when the query is fully answered without fetching documents,
125141
* or {@code null} when the cursor must be drained to count. The count is exact only when
126142
* nothing downstream drops or changes cardinality (a post-filter, skip, or limit); sort does
127143
* not change the count, and an OR-union needs de-duplication so its count cannot be derived.
128144
*/
129-
private Long computeCoveredCount(FindPlan findPlan, long indexedIdCount) {
130-
if (!findPlan.getSubPlans().isEmpty()
131-
|| findPlan.getCollectionScanFilter() != null
132-
|| findPlan.getSkip() != null
133-
|| findPlan.getLimit() != null
134-
|| findPlan.getByIdFilter() != null) {
135-
return null;
136-
}
137-
if (findPlan.getIndexDescriptor() != null) {
138-
// the index supplied the exact matching id set
139-
return indexedIdCount >= 0 ? indexedIdCount : null;
140-
}
141-
// pure full scan over the whole collection
142-
return nitriteMap.size();
145+
private boolean isCountCovered(FindPlan findPlan) {
146+
return findPlan.getSubPlans().isEmpty()
147+
&& findPlan.getCollectionScanFilter() == null
148+
&& findPlan.getSkip() == null
149+
&& findPlan.getLimit() == null
150+
&& findPlan.getByIdFilter() == null;
143151
}
144152

145153
/**
@@ -193,7 +201,7 @@ private static Object indexedValue(DBValue dbValue) {
193201
return dbValue == null || dbValue instanceof DBNull ? null : dbValue.getValue();
194202
}
195203

196-
private RecordStream<Pair<NitriteId, Document>> findSuitableStream(FindPlan findPlan, long[] indexedIdCount) {
204+
private RecordStream<Pair<NitriteId, Document>> findSuitableStream(FindPlan findPlan, IndexScan scan) {
197205
RecordStream<Pair<NitriteId, Document>> rawStream;
198206
RecordStream<Pair<NitriteId, Document>> indexSortedStream = null;
199207

@@ -202,7 +210,7 @@ private RecordStream<Pair<NitriteId, Document>> findSuitableStream(FindPlan find
202210
List<RecordStream<Pair<NitriteId, Document>>> subStreams = new ArrayList<>();
203211
for (FindPlan subPlan : findPlan.getSubPlans()) {
204212
// a sub-plan's own id count cannot answer the union's count (dedup), so discard it
205-
RecordStream<Pair<NitriteId, Document>> suitableStream = findSuitableStream(subPlan, new long[]{ -1 });
213+
RecordStream<Pair<NitriteId, Document>> suitableStream = findSuitableStream(subPlan, new IndexScan());
206214
subStreams.add(suitableStream);
207215
}
208216

@@ -233,14 +241,21 @@ private RecordStream<Pair<NitriteId, Document>> findSuitableStream(FindPlan find
233241
if (indexDescriptor != null) {
234242
// get optimized filter
235243
NitriteIndexer indexer = nitriteConfig.findIndexer(indexDescriptor.getIndexType());
236-
LinkedHashSet<NitriteId> nitriteIds = indexer.findByFilter(findPlan, nitriteConfig);
244+
RecordStream<NitriteId> idStream = indexer.findByFilterStream(findPlan, nitriteConfig);
245+
if (idStream != null) {
246+
// the index walks its matches lazily; a size() is counted from it on demand
247+
scan.lazyStream = new IndexedStream(idStream, nitriteMap);
248+
rawStream = scan.lazyStream;
249+
} else {
250+
LinkedHashSet<NitriteId> nitriteIds = indexer.findByFilter(findPlan, nitriteConfig);
237251

238-
// the index supplied the exact matching id set; record its size so a size()
239-
// with no row-dropping step downstream can answer from it without fetching
240-
indexedIdCount[0] = nitriteIds.size();
252+
// the index supplied the exact matching id set; record its size so a size()
253+
// with no row-dropping step downstream can answer from it without fetching
254+
scan.idCount = nitriteIds.size();
241255

242-
// create indexed stream from optimized filter
243-
rawStream = new IndexedStream(nitriteIds, nitriteMap);
256+
// create indexed stream from optimized filter
257+
rawStream = new IndexedStream(nitriteIds, nitriteMap);
258+
}
244259
} else {
245260
indexSortedStream = indexSortedStream(findPlan);
246261
rawStream = indexSortedStream != null ? indexSortedStream : nitriteMap.entries();

‎nitrite/src/main/java/org/dizitart/no2/common/streams/DocumentStream.java‎

Lines changed: 12 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -33,6 +33,7 @@
3333

3434
import java.util.Collections;
3535
import java.util.Iterator;
36+
import java.util.function.LongSupplier;
3637

3738
/**
3839
* @since 4.0
@@ -53,6 +54,13 @@ public class DocumentStream implements DocumentCursor {
5354
@Setter
5455
private Long coveredCount;
5556

57+
/**
58+
* Answers {@link #size()} from the index on demand when the match count is known to be
59+
* covered but the ids are streamed lazily rather than materialized; evaluated once.
60+
*/
61+
@Setter
62+
private LongSupplier coveredCountSupplier;
63+
5664
public DocumentStream(RecordStream<Pair<NitriteId, Document>> recordStream,
5765
ProcessorChain processorChain) {
5866
this.recordStream = recordStream;
@@ -64,6 +72,10 @@ public long size() {
6472
if (coveredCount != null) {
6573
return coveredCount;
6674
}
75+
if (coveredCountSupplier != null) {
76+
coveredCount = coveredCountSupplier.getAsLong();
77+
return coveredCount;
78+
}
6779
return Iterables.size(this);
6880
}
6981

‎nitrite/src/main/java/org/dizitart/no2/common/streams/IndexedStream.java‎

Lines changed: 16 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -24,17 +24,16 @@
2424
import org.dizitart.no2.store.NitriteMap;
2525

2626
import java.util.Iterator;
27-
import java.util.Set;
2827

2928
/**
3029
* @author Anindya Chatterjee
3130
* @since 4.0
3231
*/
3332
public class IndexedStream implements RecordStream<Pair<NitriteId, Document>> {
3433
private final NitriteMap<NitriteId, Document> nitriteMap;
35-
private final Set<NitriteId> nitriteIds;
34+
private final Iterable<NitriteId> nitriteIds;
3635

37-
public IndexedStream(Set<NitriteId> nitriteIds,
36+
public IndexedStream(Iterable<NitriteId> nitriteIds,
3837
NitriteMap<NitriteId, Document> nitriteMap) {
3938
this.nitriteIds = nitriteIds;
4039
this.nitriteMap = nitriteMap;
@@ -45,6 +44,20 @@ public Iterator<Pair<NitriteId, Document>> iterator() {
4544
return new IndexedStreamIterator(nitriteIds.iterator(), nitriteMap);
4645
}
4746

47+
/**
48+
* Counts the ids the index supplied, walking the id source only, without fetching a
49+
* single document.
50+
*
51+
* @return the number of ids
52+
*/
53+
public long countIds() {
54+
long count = 0;
55+
for (NitriteId ignored : nitriteIds) {
56+
count++;
57+
}
58+
return count;
59+
}
60+
4861
private static class IndexedStreamIterator implements Iterator<Pair<NitriteId, Document>>,
4962
SkippableIterator {
5063
private final Iterator<NitriteId> iterator;

‎nitrite/src/main/java/org/dizitart/no2/index/ComparableIndexer.java‎

Lines changed: 7 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -21,6 +21,7 @@
2121
import org.dizitart.no2.collection.NitriteId;
2222
import org.dizitart.no2.common.DBValue;
2323
import org.dizitart.no2.common.FieldValues;
24+
import org.dizitart.no2.common.RecordStream;
2425
import org.dizitart.no2.common.Fields;
2526
import org.dizitart.no2.common.tuples.Pair;
2627
import org.dizitart.no2.exceptions.IndexingException;
@@ -66,6 +67,12 @@ public LinkedHashSet<NitriteId> findByFilter(FindPlan findPlan, NitriteConfig ni
6667
return nitriteIndex.findNitriteIds(findPlan);
6768
}
6869

70+
@Override
71+
public RecordStream<NitriteId> findByFilterStream(FindPlan findPlan, NitriteConfig nitriteConfig) {
72+
NitriteIndex nitriteIndex = findNitriteIndex(findPlan.getIndexDescriptor(), nitriteConfig);
73+
return nitriteIndex.findNitriteIdStream(findPlan);
74+
}
75+
6976
@Override
7077
public List<Pair<DBValue, NitriteId>> readSortKeys(IndexDescriptor indexDescriptor,
7178
NitriteConfig nitriteConfig,

‎nitrite/src/main/java/org/dizitart/no2/index/NitriteIndex.java‎

Lines changed: 14 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -21,6 +21,7 @@
2121
import org.dizitart.no2.collection.NitriteId;
2222
import org.dizitart.no2.common.DBValue;
2323
import org.dizitart.no2.common.FieldValues;
24+
import org.dizitart.no2.common.RecordStream;
2425
import org.dizitart.no2.common.tuples.Pair;
2526
import org.dizitart.no2.exceptions.UniqueConstraintException;
2627
import org.dizitart.no2.exceptions.ValidationException;
@@ -74,6 +75,19 @@ public interface NitriteIndex {
7475
*/
7576
LinkedHashSet<NitriteId> findNitriteIds(FindPlan findPlan);
7677

78+
/**
79+
* Streams the ids matching the plan lazily, in index order and without duplicates, or
80+
* returns {@code null} when this index cannot do so for the given plan, in which case the
81+
* caller falls back to {@link #findNitriteIds(FindPlan)}. A stream lets a query that only
82+
* needs the first rows, or a bounded page, stop reading the index as soon as it has them.
83+
*
84+
* @param findPlan the find plan
85+
* @return a re-iterable stream of ids, or {@code null}
86+
*/
87+
default RecordStream<NitriteId> findNitriteIdStream(FindPlan findPlan) {
88+
return null;
89+
}
90+
7791
/**
7892
* Reads every {@code (indexed value, id)} pair out of the index, so a sorted query can
7993
* decide its order without deserializing a single document.

‎nitrite/src/main/java/org/dizitart/no2/index/NitriteIndexer.java‎

Lines changed: 14 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -21,6 +21,7 @@
2121
import org.dizitart.no2.collection.NitriteId;
2222
import org.dizitart.no2.common.DBValue;
2323
import org.dizitart.no2.common.FieldValues;
24+
import org.dizitart.no2.common.RecordStream;
2425
import org.dizitart.no2.common.Fields;
2526
import org.dizitart.no2.common.module.NitritePlugin;
2627
import org.dizitart.no2.common.tuples.Pair;
@@ -88,6 +89,19 @@ public interface NitriteIndexer extends NitritePlugin {
8889
*/
8990
LinkedHashSet<NitriteId> findByFilter(FindPlan findPlan, NitriteConfig nitriteConfig);
9091

92+
/**
93+
* Streams the ids matching the plan lazily, or returns {@code null} when the indexer has no
94+
* lazy path for it and {@link #findByFilter(FindPlan, NitriteConfig)} must be used. The
95+
* default is {@code null}, so existing indexer plugins are unaffected.
96+
*
97+
* @param findPlan the find plan
98+
* @param nitriteConfig the nitrite config
99+
* @return a re-iterable stream of ids, or {@code null}
100+
*/
101+
default RecordStream<NitriteId> findByFilterStream(FindPlan findPlan, NitriteConfig nitriteConfig) {
102+
return null;
103+
}
104+
91105
/**
92106
* Reads every {@code (indexed value, id)} pair out of the given index, so a sorted query
93107
* can decide its order without deserializing a single document.

0 commit comments

Comments
 (0)