|
|
@@ -16,6 +16,7 @@
|
|
|
#include <iostream>
|
|
|
#include <cstdio>
|
|
|
#include <algorithm>
|
|
|
+#include <functional>
|
|
|
#include <thread>
|
|
|
#include <set>
|
|
|
#include <string>
|
|
|
@@ -1177,6 +1178,240 @@ void test_range_contains_and_intersection() {
|
|
|
}
|
|
|
}
|
|
|
|
|
|
+
|
|
|
+// v2.9.2 — the index supplies the ORDER, not just filter candidates.
|
|
|
+//
|
|
|
+// "newest N, unfiltered" walked the whole collection to discover what to sort by:
|
|
|
+// 341ms to return one row from 592 MB of real data, with the index declared and
|
|
|
+// unused, because a Sort is not a filter. Walking the sort field's index reads the
|
|
|
+// page and nothing else.
|
|
|
+//
|
|
|
+// The dangerous part is not speed, it is that sort_documents places rows MISSING
|
|
|
+// the sort field FIRST when descending, while an index holds no posting for them.
|
|
|
+// A walk would silently omit them from the first page - wrong rows, not slow ones.
|
|
|
+// So every case here compares the ordered plan against the plain scan.
|
|
|
+void test_index_supplies_ordering() {
|
|
|
+ using Op = smartbotic::database::FilterOp;
|
|
|
+
|
|
|
+ auto make = [](LmdbDocumentStore& store, int n,
|
|
|
+ const std::function<nlohmann::json(int)>& body) {
|
|
|
+ for (int i = 0; i < n; ++i) {
|
|
|
+ Document d;
|
|
|
+ d.id = "r" + std::string(i < 10 ? "00" : (i < 100 ? "0" : "")) +
|
|
|
+ std::to_string(i);
|
|
|
+ d.collection = "c";
|
|
|
+ d.set_data(body(i));
|
|
|
+ store.put("c", d.id, d);
|
|
|
+ }
|
|
|
+ };
|
|
|
+ auto sig = [](LmdbDocumentStore& store, const smartbotic::database::Query& q) {
|
|
|
+ auto r = store.scan("c", q);
|
|
|
+ std::string out = "total=" + std::to_string(r.total_matched) +
|
|
|
+ " more=" + std::to_string(r.has_more ? 1 : 0) + " [";
|
|
|
+ for (const auto& d : r.documents) { out += d.id; out += ","; }
|
|
|
+ return out + "]";
|
|
|
+ };
|
|
|
+ auto Q = [](const char* field, bool desc, uint32_t limit, uint32_t offset) {
|
|
|
+ smartbotic::database::Query q;
|
|
|
+ q.sort = smartbotic::database::Sort{field, desc};
|
|
|
+ q.limit = limit;
|
|
|
+ q.offset = offset;
|
|
|
+ return q;
|
|
|
+ };
|
|
|
+
|
|
|
+ // ---- every row carries the sort field: the ordered plan applies ----
|
|
|
+ {
|
|
|
+ TmpEnv t("ord-total");
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
+ // Deliberate duplicate keys (i/3) so tie-breaking by id is exercised in
|
|
|
+ // both directions.
|
|
|
+ make(store, 300, [](int i) {
|
|
|
+ return nlohmann::json{{"seq", i}, {"dup", i / 3}};
|
|
|
+ });
|
|
|
+
|
|
|
+ const std::vector<std::pair<const char*, smartbotic::database::Query>> cases = {
|
|
|
+ {"desc, first page", Q("seq", true, 10, 0)},
|
|
|
+ {"asc, first page", Q("seq", false, 10, 0)},
|
|
|
+ {"desc, deep offset", Q("seq", true, 10, 250)},
|
|
|
+ {"asc, deep offset", Q("seq", false, 7, 33)},
|
|
|
+ {"desc, last partial page", Q("seq", true, 10, 295)},
|
|
|
+ {"offset past the end", Q("seq", true, 10, 999)},
|
|
|
+ {"limit=0", Q("seq", true, 0, 0)},
|
|
|
+ {"whole collection", Q("seq", false, 1000, 0)},
|
|
|
+ {"desc with tied keys", Q("dup", true, 12, 0)},
|
|
|
+ {"asc with tied keys", Q("dup", false, 12, 0)},
|
|
|
+ {"tied keys, deep offset", Q("dup", true, 5, 40)},
|
|
|
+ };
|
|
|
+ uint64_t ordered_used = 0;
|
|
|
+ for (const auto& [name, q] : cases) {
|
|
|
+ store.set_indexed_fields("c", {});
|
|
|
+ const std::string without = sig(store, q);
|
|
|
+ store.set_indexed_fields("c", {"seq", "dup"});
|
|
|
+ store.build_index("c", "seq");
|
|
|
+ store.build_index("c", "dup");
|
|
|
+ store.reset_index_plan_stats();
|
|
|
+ const std::string with = sig(store, q);
|
|
|
+ ordered_used += store.index_plan_stats().ordered_scans;
|
|
|
+ if (with != without) {
|
|
|
+ std::cerr << " scan: " << without.substr(0, 180) << "\n"
|
|
|
+ << " ordered: " << with.substr(0, 180) << "\n";
|
|
|
+ }
|
|
|
+ const std::string msg = std::string("ordered plan == scan: ") + name;
|
|
|
+ check(with == without, msg.c_str());
|
|
|
+ }
|
|
|
+
|
|
|
+ // EVERY case above must have used the ordered plan, not just one. The
|
|
|
+ // first version of this test asserted only a single query and so passed
|
|
|
+ // while the plan was silently never taken (the collection's row count
|
|
|
+ // included the identity sentinel, so entries never equalled rows).
|
|
|
+ check(ordered_used == cases.size(),
|
|
|
+ ("the index SUPPLIED THE ORDER in all " + std::to_string(cases.size()) +
|
|
|
+ " cases (got " + std::to_string(ordered_used) + ") - otherwise the "
|
|
|
+ "comparisons above are scan against scan").c_str());
|
|
|
+ }
|
|
|
+
|
|
|
+ // ---- THE trap: one row lacks the sort field ----
|
|
|
+ {
|
|
|
+ TmpEnv t("ord-missing");
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
+ make(store, 50, [](int i) {
|
|
|
+ nlohmann::json j{{"other", i}};
|
|
|
+ if (i != 7) j["seq"] = i; // r007 has no seq
|
|
|
+ return j;
|
|
|
+ });
|
|
|
+ store.set_indexed_fields("c", {});
|
|
|
+ const std::string without = sig(store, Q("seq", true, 5, 0));
|
|
|
+ store.set_indexed_fields("c", {"seq"});
|
|
|
+ store.build_index("c", "seq");
|
|
|
+ const std::string with = sig(store, Q("seq", true, 5, 0));
|
|
|
+
|
|
|
+ check(with == without,
|
|
|
+ "one row missing the sort field: DESCENDING puts it FIRST, and the "
|
|
|
+ "index has no posting for it - the ordered plan must decline");
|
|
|
+ store.reset_index_plan_stats();
|
|
|
+ (void)sig(store, Q("seq", true, 5, 0));
|
|
|
+ check(store.index_plan_stats().ordered_scans == 0,
|
|
|
+ "and it does decline - entries != rows is the guard");
|
|
|
+ }
|
|
|
+
|
|
|
+ // ---- an array-valued sort field must also decline ----
|
|
|
+ {
|
|
|
+ TmpEnv t("ord-array");
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
+ make(store, 40, [](int i) {
|
|
|
+ return nlohmann::json{{"seq", nlohmann::json::array({i})}};
|
|
|
+ });
|
|
|
+ store.set_indexed_fields("c", {});
|
|
|
+ const std::string without = sig(store, Q("seq", true, 5, 0));
|
|
|
+ store.set_indexed_fields("c", {"seq"});
|
|
|
+ store.build_index("c", "seq");
|
|
|
+ const std::string with = sig(store, Q("seq", true, 5, 0));
|
|
|
+ check(with == without,
|
|
|
+ "a one-element-array sort field agrees - it satisfies entries==rows, "
|
|
|
+ "so the fetched-page array check is what catches it");
|
|
|
+ }
|
|
|
+}
|
|
|
+
|
|
|
+// v2.9.2 — IN as a union of posting lists, EXISTS as all postings.
|
|
|
+void test_index_in_and_exists() {
|
|
|
+ using Op = smartbotic::database::FilterOp;
|
|
|
+ TmpEnv t("in-exists");
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
+
|
|
|
+ for (int i = 0; i < 400; ++i) {
|
|
|
+ Document d;
|
|
|
+ d.id = "r" + std::to_string(1000 + i);
|
|
|
+ d.collection = "c";
|
|
|
+ nlohmann::json j{{"grp", "g" + std::to_string(i % 40)}};
|
|
|
+ // `rare` exists on 8 of 400 rows, so EXISTS=true is highly selective.
|
|
|
+ if (i % 50 == 0) j["rare"] = i;
|
|
|
+ d.set_data(j);
|
|
|
+ store.put("c", d.id, d);
|
|
|
+ }
|
|
|
+
|
|
|
+ auto F = [](const char* f, Op op, const nlohmann::json& v) {
|
|
|
+ smartbotic::database::Filter x;
|
|
|
+ x.field = f; x.op = op; x.value = v;
|
|
|
+ return x;
|
|
|
+ };
|
|
|
+ auto sig = [&](const std::vector<smartbotic::database::Filter>& fs) {
|
|
|
+ smartbotic::database::Query q;
|
|
|
+ q.filters = fs;
|
|
|
+ q.limit = 1000;
|
|
|
+ auto r = store.scan("c", q);
|
|
|
+ std::vector<std::string> ids;
|
|
|
+ for (const auto& d : r.documents) ids.push_back(d.id);
|
|
|
+ std::sort(ids.begin(), ids.end());
|
|
|
+ std::string out = "total=" + std::to_string(r.total_matched) + " [";
|
|
|
+ for (const auto& i : ids) { out += i; out += ","; }
|
|
|
+ return out + "]";
|
|
|
+ };
|
|
|
+
|
|
|
+ const std::vector<std::pair<const char*, std::vector<smartbotic::database::Filter>>> cases = {
|
|
|
+ {"IN over three values", {F("grp", Op::IN, nlohmann::json::array({"g1","g2","g3"}))}},
|
|
|
+ {"IN with a missing value", {F("grp", Op::IN, nlohmann::json::array({"g1","nope"}))}},
|
|
|
+ {"IN over one value", {F("grp", Op::IN, nlohmann::json::array({"g5"}))}},
|
|
|
+ {"IN matching nothing", {F("grp", Op::IN, nlohmann::json::array({"x","y"}))}},
|
|
|
+ {"IN too broad to help", {F("grp", Op::IN, nlohmann::json::array(
|
|
|
+ {"g0","g1","g2","g3","g4","g5","g6","g7","g8","g9","g10","g11"}))}},
|
|
|
+ {"EXISTS true on a sparse field", {F("rare", Op::EXISTS, true)}},
|
|
|
+ {"EXISTS false on a sparse field", {F("rare", Op::EXISTS, false)}},
|
|
|
+ {"EXISTS true on a dense field", {F("grp", Op::EXISTS, true)}},
|
|
|
+ };
|
|
|
+
|
|
|
+ for (const auto& [name, filters] : cases) {
|
|
|
+ store.set_indexed_fields("c", {});
|
|
|
+ const std::string without = sig(filters);
|
|
|
+ store.set_indexed_fields("c", {"grp", "rare"});
|
|
|
+ store.build_index("c", "grp");
|
|
|
+ store.build_index("c", "rare");
|
|
|
+ const std::string with = sig(filters);
|
|
|
+ if (with != without) {
|
|
|
+ std::cerr << " scan: " << without.substr(0, 160) << "\n"
|
|
|
+ << " indexed: " << with.substr(0, 160) << "\n";
|
|
|
+ }
|
|
|
+ const std::string msg = std::string("indexed == scan: ") + name;
|
|
|
+ check(with == without, msg.c_str());
|
|
|
+ }
|
|
|
+
|
|
|
+ store.set_indexed_fields("c", {"grp", "rare"});
|
|
|
+ {
|
|
|
+ store.reset_index_plan_stats();
|
|
|
+ (void)sig({F("grp", Op::IN, nlohmann::json::array({"g1","g2","g3"}))});
|
|
|
+ check(store.index_plan_stats().union_scans == 1,
|
|
|
+ "IN over a narrow set UNIONS posting lists - 30 of 400 rows");
|
|
|
+ }
|
|
|
+ {
|
|
|
+ store.reset_index_plan_stats();
|
|
|
+ (void)sig({F("grp", Op::IN, nlohmann::json::array(
|
|
|
+ {"g0","g1","g2","g3","g4","g5","g6","g7","g8","g9","g10","g11"}))});
|
|
|
+ auto st = store.index_plan_stats();
|
|
|
+ check(st.union_scans == 0 && st.full_scans == 1,
|
|
|
+ "a broad IN is rejected on the summed counts, before any list is read");
|
|
|
+ }
|
|
|
+ {
|
|
|
+ store.reset_index_plan_stats();
|
|
|
+ (void)sig({F("rare", Op::EXISTS, true)});
|
|
|
+ check(store.index_plan_stats().exists_scans == 1,
|
|
|
+ "EXISTS=true on a sparse field is served from all its postings");
|
|
|
+ }
|
|
|
+ {
|
|
|
+ store.reset_index_plan_stats();
|
|
|
+ (void)sig({F("rare", Op::EXISTS, false)});
|
|
|
+ auto st = store.index_plan_stats();
|
|
|
+ check(st.exists_scans == 0 && st.full_scans == 1,
|
|
|
+ "EXISTS=false cannot be served - rows WITHOUT a posting are not "
|
|
|
+ "enumerable from the index, so it must scan");
|
|
|
+ }
|
|
|
+ {
|
|
|
+ store.reset_index_plan_stats();
|
|
|
+ (void)sig({F("grp", Op::EXISTS, true)});
|
|
|
+ auto st = store.index_plan_stats();
|
|
|
+ check(st.exists_scans == 0 && st.full_scans == 1,
|
|
|
+ "EXISTS=true on a field every row has is not selective, so it scans");
|
|
|
+ }
|
|
|
+}
|
|
|
+
|
|
|
} // namespace
|
|
|
|
|
|
int main() {
|
|
|
@@ -1199,6 +1434,8 @@ int main() {
|
|
|
test_indexed_and_unindexed_plans_agree();
|
|
|
test_empty_id_reads_as_absent();
|
|
|
test_range_contains_and_intersection();
|
|
|
+ test_index_supplies_ordering();
|
|
|
+ test_index_in_and_exists();
|
|
|
|
|
|
std::cout << "passed: " << g_pass << ", failed: " << g_fail << "\n";
|
|
|
return g_fail == 0 ? 0 : 1;
|