|
@@ -683,6 +683,9 @@ LmdbDocumentStore::IndexPlanStats LmdbDocumentStore::index_plan_stats() const {
|
|
|
declined_unselective_.load(std::memory_order_relaxed);
|
|
declined_unselective_.load(std::memory_order_relaxed);
|
|
|
st.intersected_scans = intersected_scans_.load(std::memory_order_relaxed);
|
|
st.intersected_scans = intersected_scans_.load(std::memory_order_relaxed);
|
|
|
st.range_scans = range_scans_.load(std::memory_order_relaxed);
|
|
st.range_scans = range_scans_.load(std::memory_order_relaxed);
|
|
|
|
|
+ st.ordered_scans = ordered_scans_.load(std::memory_order_relaxed);
|
|
|
|
|
+ st.union_scans = union_scans_.load(std::memory_order_relaxed);
|
|
|
|
|
+ st.exists_scans = exists_scans_.load(std::memory_order_relaxed);
|
|
|
return st;
|
|
return st;
|
|
|
}
|
|
}
|
|
|
|
|
|
|
@@ -692,6 +695,9 @@ void LmdbDocumentStore::reset_index_plan_stats() {
|
|
|
declined_unselective_.store(0, std::memory_order_relaxed);
|
|
declined_unselective_.store(0, std::memory_order_relaxed);
|
|
|
intersected_scans_.store(0, std::memory_order_relaxed);
|
|
intersected_scans_.store(0, std::memory_order_relaxed);
|
|
|
range_scans_.store(0, std::memory_order_relaxed);
|
|
range_scans_.store(0, std::memory_order_relaxed);
|
|
|
|
|
+ ordered_scans_.store(0, std::memory_order_relaxed);
|
|
|
|
|
+ union_scans_.store(0, std::memory_order_relaxed);
|
|
|
|
|
+ exists_scans_.store(0, std::memory_order_relaxed);
|
|
|
}
|
|
}
|
|
|
|
|
|
|
|
std::optional<LmdbDocumentStore::IndexStats>
|
|
std::optional<LmdbDocumentStore::IndexStats>
|
|
@@ -982,6 +988,136 @@ uint64_t LmdbDocumentStore::count(std::string_view collection) {
|
|
|
// status="completed" matches 6,676 of 10,101 rows (66%), so an index on `status`
|
|
// status="completed" matches 6,676 of 10,101 rows (66%), so an index on `status`
|
|
|
// would make that query SLOWER. Declaring an index must not be able to
|
|
// would make that query SLOWER. Declaring an index must not be able to
|
|
|
// pessimise a query, so the plan is re-decided per value, not per field.
|
|
// pessimise a query, so the plan is re-decided per value, not per field.
|
|
|
|
|
+
|
|
|
|
|
+// Rows in a collection sub-db, excluding the v2.4.4 identity sentinel. mdb_stat
|
|
|
|
|
+// counts it, so comparing a raw ms_entries against an index's posting count is
|
|
|
|
|
+// off by exactly one - which silently disabled the ordered plan in every case
|
|
|
|
|
+// until a test asserted the plan was actually taken.
|
|
|
|
|
+uint64_t LmdbDocumentStore::rowCountTxn(ReadTxn& rtxn, unsigned int dbi) {
|
|
|
|
|
+ MDB_stat st{};
|
|
|
|
|
+ if (mdb_stat(rtxn.raw(), dbi, &st) != MDB_SUCCESS) return 0;
|
|
|
|
|
+ uint64_t n = st.ms_entries;
|
|
|
|
|
+ if (!read_subdb_identity(rtxn, dbi).empty() && n > 0) --n;
|
|
|
|
|
+ return n;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+bool LmdbDocumentStore::orderedHitsFromIndex(
|
|
|
|
|
+ ReadTxn& rtxn,
|
|
|
|
|
+ std::string_view collection,
|
|
|
|
|
+ const smartbotic::database::Query& query,
|
|
|
|
|
+ unsigned int coll_dbi,
|
|
|
|
|
+ std::vector<std::string>& out,
|
|
|
|
|
+ uint64_t& exact_total) {
|
|
|
|
|
+
|
|
|
|
|
+ if (!query.filters.empty()) return false;
|
|
|
|
|
+ if (!query.sort || query.sort->field.empty()) return false;
|
|
|
|
|
+ const auto fields = indexed_fields(collection);
|
|
|
|
|
+ if (std::find(fields.begin(), fields.end(), query.sort->field) == fields.end()) {
|
|
|
|
|
+ return false;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ const uint64_t rows = rowCountTxn(rtxn, coll_dbi);
|
|
|
|
|
+ if (rows == 0) return false;
|
|
|
|
|
+
|
|
|
|
|
+ auto dbi_opt = try_open_for_read(rtxn,
|
|
|
|
|
+ index_subdb_name(collection, query.sort->field));
|
|
|
|
|
+ if (!dbi_opt) return false;
|
|
|
|
|
+
|
|
|
|
|
+ // Total order or nothing. entries != rows means some row has no posting (the
|
|
|
|
|
+ // field is absent, or its value is unindexable) or several (an array), and in
|
|
|
|
|
+ // either case the index cannot state where that row sorts.
|
|
|
|
|
+ MDB_stat ist{};
|
|
|
|
|
+ if (mdb_stat(rtxn.raw(), *dbi_opt, &ist) != MDB_SUCCESS) return false;
|
|
|
|
|
+ uint64_t entries = ist.ms_entries;
|
|
|
|
|
+ if (!read_subdb_identity(rtxn, *dbi_opt).empty() && entries > 0) --entries;
|
|
|
|
|
+ if (entries != rows) return false;
|
|
|
|
|
+
|
|
|
|
|
+ MDB_cursor* cur = nullptr;
|
|
|
|
|
+ mdb_check(mdb_cursor_open(rtxn.raw(), *dbi_opt, &cur), "cursor_open (index order)");
|
|
|
|
|
+ struct G { MDB_cursor* c; ~G() { if (c) mdb_cursor_close(c); } } g{cur};
|
|
|
|
|
+
|
|
|
|
|
+ const bool desc = query.sort->descending;
|
|
|
|
|
+ const uint64_t want = static_cast<uint64_t>(query.offset) + query.limit;
|
|
|
|
|
+
|
|
|
|
|
+ // Direction handles ties correctly without extra work. sort_documents breaks
|
|
|
|
|
+ // an equal-key tie on id - ascending by id, descending by reverse id - and
|
|
|
|
|
+ // DUPSORT stores a key's ids ascending. So MDB_FIRST/MDB_NEXT yields
|
|
|
|
|
+ // ascending keys with ascending ids, and MDB_LAST/MDB_PREV yields descending
|
|
|
|
|
+ // keys with descending ids. Both match.
|
|
|
|
|
+ MDB_val k{0, nullptr};
|
|
|
|
|
+ MDB_val v{0, nullptr};
|
|
|
|
|
+ int rc = mdb_cursor_get(cur, &k, &v, desc ? MDB_LAST : MDB_FIRST);
|
|
|
|
|
+ std::vector<std::string> ids;
|
|
|
|
|
+ while (rc == MDB_SUCCESS && ids.size() < want) {
|
|
|
|
|
+ if (!is_identity_key(to_sv(k))) ids.emplace_back(to_sv(v));
|
|
|
|
|
+ rc = mdb_cursor_get(cur, &k, &v, desc ? MDB_PREV : MDB_NEXT);
|
|
|
|
|
+ }
|
|
|
|
|
+ if (rc != MDB_SUCCESS && rc != MDB_NOTFOUND) {
|
|
|
|
|
+ throw_mdb(rc, "cursor step (index order)");
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // A field holding one-element arrays on every row would satisfy entries ==
|
|
|
|
|
+ // rows while sort_documents compares whole arrays. Check the rows actually
|
|
|
|
|
+ // being returned and abandon if any is an array, so the page can never
|
|
|
|
|
+ // disagree with the general path.
|
|
|
|
|
+ //
|
|
|
|
|
+ // Honest residual: a row OUTSIDE this page could still be an array and change
|
|
|
|
|
+ // the true ordering. Sorting by an array-valued field is not meaningful, so
|
|
|
|
|
+ // this is documented rather than defended further.
|
|
|
|
|
+ const uint64_t start = std::min<uint64_t>(query.offset, ids.size());
|
|
|
|
|
+ const uint64_t end = std::min<uint64_t>(want, ids.size());
|
|
|
|
|
+ for (uint64_t i = start; i < end; ++i) {
|
|
|
|
|
+ MDB_val hk = to_val(ids[i]);
|
|
|
|
|
+ MDB_val hv{0, nullptr};
|
|
|
|
|
+ if (mdb_get(rtxn.raw(), coll_dbi, &hk, &hv) != MDB_SUCCESS) continue;
|
|
|
|
|
+ const auto bytes = to_sv(hv);
|
|
|
|
|
+ yyjson_doc* d = yyjson_read(bytes.data(), bytes.size(), 0);
|
|
|
|
|
+ if (!d) continue;
|
|
|
|
|
+ auto val = resolve_from_yyjson(yyjson_doc_get_root(d), query.sort->field);
|
|
|
|
|
+ const bool is_array = val && val->is_array();
|
|
|
|
|
+ yyjson_doc_free(d);
|
|
|
|
|
+ if (is_array) return false;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ out = std::move(ids);
|
|
|
|
|
+ exact_total = rows; // no filters, so every row matches
|
|
|
|
|
+ ordered_scans_.fetch_add(1, std::memory_order_relaxed);
|
|
|
|
|
+ return true;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+std::optional<std::vector<std::string>>
|
|
|
|
|
+LmdbDocumentStore::lookupIndexAllTxn(ReadTxn& rtxn,
|
|
|
|
|
+ std::string_view collection,
|
|
|
|
|
+ const std::string& field,
|
|
|
|
|
+ uint64_t budget) {
|
|
|
|
|
+ auto dbi_opt = try_open_for_read(rtxn, index_subdb_name(collection, field));
|
|
|
|
|
+ if (!dbi_opt) return std::nullopt;
|
|
|
|
|
+
|
|
|
|
|
+ MDB_cursor* cur = nullptr;
|
|
|
|
|
+ mdb_check(mdb_cursor_open(rtxn.raw(), *dbi_opt, &cur), "cursor_open (index all)");
|
|
|
|
|
+ struct G { MDB_cursor* c; ~G() { if (c) mdb_cursor_close(c); } } g{cur};
|
|
|
|
|
+
|
|
|
|
|
+ std::vector<std::string> ids;
|
|
|
|
|
+ MDB_val k{0, nullptr};
|
|
|
|
|
+ MDB_val v{0, nullptr};
|
|
|
|
|
+ int rc = mdb_cursor_get(cur, &k, &v, MDB_FIRST);
|
|
|
|
|
+ while (rc == MDB_SUCCESS) {
|
|
|
|
|
+ if (!is_identity_key(to_sv(k))) {
|
|
|
|
|
+ ids.emplace_back(to_sv(v));
|
|
|
|
|
+ // Bail out rather than truncate, as everywhere else: a partial
|
|
|
|
|
+ // candidate list drops matching rows.
|
|
|
|
|
+ if (ids.size() > budget) return std::nullopt;
|
|
|
|
|
+ }
|
|
|
|
|
+ rc = mdb_cursor_get(cur, &k, &v, MDB_NEXT);
|
|
|
|
|
+ }
|
|
|
|
|
+ if (rc != MDB_SUCCESS && rc != MDB_NOTFOUND) throw_mdb(rc, "cursor next (index all)");
|
|
|
|
|
+ // An array-valued field yields several postings per row; dedupe so a row is
|
|
|
|
|
+ // named once.
|
|
|
|
|
+ std::sort(ids.begin(), ids.end());
|
|
|
|
|
+ ids.erase(std::unique(ids.begin(), ids.end()), ids.end());
|
|
|
|
|
+ return ids;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
std::optional<std::vector<std::string>>
|
|
std::optional<std::vector<std::string>>
|
|
|
LmdbDocumentStore::planIndexCandidates(ReadTxn& rtxn,
|
|
LmdbDocumentStore::planIndexCandidates(ReadTxn& rtxn,
|
|
|
std::string_view collection,
|
|
std::string_view collection,
|
|
@@ -991,9 +1127,7 @@ LmdbDocumentStore::planIndexCandidates(ReadTxn& rtxn,
|
|
|
const auto fields = indexed_fields(collection);
|
|
const auto fields = indexed_fields(collection);
|
|
|
if (fields.empty()) return std::nullopt;
|
|
if (fields.empty()) return std::nullopt;
|
|
|
|
|
|
|
|
- MDB_stat st{};
|
|
|
|
|
- if (mdb_stat(rtxn.raw(), coll_dbi, &st) != MDB_SUCCESS) return std::nullopt;
|
|
|
|
|
- const uint64_t total = st.ms_entries;
|
|
|
|
|
|
|
+ const uint64_t total = rowCountTxn(rtxn, coll_dbi);
|
|
|
if (total == 0) return std::nullopt;
|
|
if (total == 0) return std::nullopt;
|
|
|
const uint64_t budget = total / kIndexSelectivityDivisor;
|
|
const uint64_t budget = total / kIndexSelectivityDivisor;
|
|
|
|
|
|
|
@@ -1015,6 +1149,47 @@ LmdbDocumentStore::planIndexCandidates(ReadTxn& rtxn,
|
|
|
exacts.push_back({&f, *n});
|
|
exacts.push_back({&f, *n});
|
|
|
}
|
|
}
|
|
|
}
|
|
}
|
|
|
|
|
+
|
|
|
|
|
+ // v2.9.2 — IN is a UNION of posting lists, the mirror of the intersection
|
|
|
|
|
+ // below. Summing the per-value counts first means an IN over a broad set is
|
|
|
|
|
+ // rejected before any list is read.
|
|
|
|
|
+ for (const auto& f : query.filters) {
|
|
|
|
|
+ if (f.op != Op::IN || !f.value.is_array()) continue;
|
|
|
|
|
+ if (!is_indexed(f.field)) continue;
|
|
|
|
|
+ uint64_t sum = 0;
|
|
|
|
|
+ bool all_countable = true;
|
|
|
|
|
+ for (const auto& el : f.value) {
|
|
|
|
|
+ auto n = countIndexEqTxn(rtxn, collection, f.field, el);
|
|
|
|
|
+ if (!n) { all_countable = false; break; }
|
|
|
|
|
+ sum += *n;
|
|
|
|
|
+ }
|
|
|
|
|
+ if (!all_countable || sum > budget) continue;
|
|
|
|
|
+ std::vector<std::string> uni;
|
|
|
|
|
+ for (const auto& el : f.value) {
|
|
|
|
|
+ auto part = lookupIndexEqTxn(rtxn, collection, f.field, el);
|
|
|
|
|
+ if (!part) { all_countable = false; break; }
|
|
|
|
|
+ uni.insert(uni.end(), part->begin(), part->end());
|
|
|
|
|
+ }
|
|
|
|
|
+ if (!all_countable) continue;
|
|
|
|
|
+ std::sort(uni.begin(), uni.end());
|
|
|
|
|
+ uni.erase(std::unique(uni.begin(), uni.end()), uni.end());
|
|
|
|
|
+ union_scans_.fetch_add(1, std::memory_order_relaxed);
|
|
|
|
|
+ return uni;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // v2.9.2 — EXISTS=true is every posting the field has, which is selective
|
|
|
|
|
+ // exactly when the field is SPARSE. EXISTS=false cannot be served: rows
|
|
|
|
|
+ // without a posting are not enumerable from the index.
|
|
|
|
|
+ for (const auto& f : query.filters) {
|
|
|
|
|
+ if (f.op != Op::EXISTS) continue;
|
|
|
|
|
+ if (!is_indexed(f.field)) continue;
|
|
|
|
|
+ const bool wants = f.value.is_boolean() ? f.value.get<bool>() : true;
|
|
|
|
|
+ if (!wants) continue;
|
|
|
|
|
+ if (auto ids = lookupIndexAllTxn(rtxn, collection, f.field, budget)) {
|
|
|
|
|
+ exists_scans_.fetch_add(1, std::memory_order_relaxed);
|
|
|
|
|
+ return ids;
|
|
|
|
|
+ }
|
|
|
|
|
+ }
|
|
|
std::sort(exacts.begin(), exacts.end(),
|
|
std::sort(exacts.begin(), exacts.end(),
|
|
|
[](const Exact& a, const Exact& b) { return a.count < b.count; });
|
|
[](const Exact& a, const Exact& b) { return a.count < b.count; });
|
|
|
|
|
|
|
@@ -1226,14 +1401,38 @@ ScanResult LmdbDocumentStore::scan(std::string_view collection,
|
|
|
yyjson_doc_free(ydoc);
|
|
yyjson_doc_free(ydoc);
|
|
|
};
|
|
};
|
|
|
|
|
|
|
|
- auto candidates = planIndexCandidates(rtxn, collection, query, *dbi_opt);
|
|
|
|
|
- if (candidates) {
|
|
|
|
|
- indexed_scans_.fetch_add(1, std::memory_order_relaxed);
|
|
|
|
|
- } else {
|
|
|
|
|
- full_scans_.fetch_add(1, std::memory_order_relaxed);
|
|
|
|
|
|
|
+ // v2.9.2 — ORDERING from the index. Try this first: when it applies it
|
|
|
|
|
+ // reads the page and nothing else, where every other plan still visits
|
|
|
|
|
+ // every matching row.
|
|
|
|
|
+ bool ordered_by_index = false;
|
|
|
|
|
+ uint64_t ordered_total = 0;
|
|
|
|
|
+ if (query.filters.empty() && sorting) {
|
|
|
|
|
+ std::vector<std::string> ordered_ids;
|
|
|
|
|
+ if (orderedHitsFromIndex(rtxn, collection, query, *dbi_opt,
|
|
|
|
|
+ ordered_ids, ordered_total)) {
|
|
|
|
|
+ ordered_by_index = true;
|
|
|
|
|
+ hits.reserve(ordered_ids.size());
|
|
|
|
|
+ for (auto& id : ordered_ids) {
|
|
|
|
|
+ Hit h;
|
|
|
|
|
+ h.id = std::move(id);
|
|
|
|
|
+ hits.push_back(std::move(h));
|
|
|
|
|
+ }
|
|
|
|
|
+ }
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ std::optional<std::vector<std::string>> candidates;
|
|
|
|
|
+ if (!ordered_by_index) {
|
|
|
|
|
+ candidates = planIndexCandidates(rtxn, collection, query, *dbi_opt);
|
|
|
|
|
+ if (candidates) {
|
|
|
|
|
+ indexed_scans_.fetch_add(1, std::memory_order_relaxed);
|
|
|
|
|
+ } else {
|
|
|
|
|
+ full_scans_.fetch_add(1, std::memory_order_relaxed);
|
|
|
|
|
+ }
|
|
|
}
|
|
}
|
|
|
|
|
|
|
|
- if (candidates) {
|
|
|
|
|
|
|
+ if (ordered_by_index) {
|
|
|
|
|
+ // hits are already in final order, and only the page's worth exist.
|
|
|
|
|
+ } else if (candidates) {
|
|
|
// Indexed plan: visit only the rows the index named. Every filter is
|
|
// Indexed plan: visit only the rows the index named. Every filter is
|
|
|
// still applied to each one, so the index is allowed to be a
|
|
// still applied to each one, so the index is allowed to be a
|
|
|
// superset - it narrows work, it does not decide the answer.
|
|
// superset - it narrows work, it does not decide the answer.
|
|
@@ -1262,7 +1461,9 @@ ScanResult LmdbDocumentStore::scan(std::string_view collection,
|
|
|
}
|
|
}
|
|
|
}
|
|
}
|
|
|
|
|
|
|
|
- if (sorting) {
|
|
|
|
|
|
|
+ // Already ordered by the index walk - re-sorting would be wasted work, and
|
|
|
|
|
+ // the hits carry no sort key to sort by.
|
|
|
|
|
+ if (sorting && !ordered_by_index) {
|
|
|
const bool desc = query.sort->descending;
|
|
const bool desc = query.sort->descending;
|
|
|
std::stable_sort(hits.begin(), hits.end(),
|
|
std::stable_sort(hits.begin(), hits.end(),
|
|
|
[desc](const Hit& a, const Hit& b) {
|
|
[desc](const Hit& a, const Hit& b) {
|
|
@@ -1277,7 +1478,10 @@ ScanResult LmdbDocumentStore::scan(std::string_view collection,
|
|
|
});
|
|
});
|
|
|
}
|
|
}
|
|
|
|
|
|
|
|
- result.total_matched = hits.size();
|
|
|
|
|
|
|
+ // With an index-ordered walk `hits` holds only offset+limit rows, so its
|
|
|
|
|
+ // size is not the total. The total is exact and free there: no filters
|
|
|
|
|
+ // means every row matches, and mdb_stat already counted them.
|
|
|
|
|
+ result.total_matched = ordered_by_index ? ordered_total : hits.size();
|
|
|
const uint64_t start = std::min<uint64_t>(query.offset, hits.size());
|
|
const uint64_t start = std::min<uint64_t>(query.offset, hits.size());
|
|
|
const uint64_t end = std::min<uint64_t>(
|
|
const uint64_t end = std::min<uint64_t>(
|
|
|
static_cast<uint64_t>(query.offset) + query.limit, hits.size());
|
|
static_cast<uint64_t>(query.offset) + query.limit, hits.size());
|