|
@@ -0,0 +1,124 @@
|
|
|
|
|
+// Regression: IN must behave the same on a SPARSE field as on a dense one.
|
|
|
|
|
+//
|
|
|
|
|
+// Reported by a consumer as "`parentExecutionId in (id, id, ...)` returns
|
|
|
|
|
+// nothing while each id queried alone matches, and the same `in` works on
|
|
|
|
|
+// `status` and `workflowId`" - with the difference identified as sparseness
|
|
|
|
|
+// (most executions have no `parentExecutionId` at all).
|
|
|
|
|
+//
|
|
|
|
|
+// It did not reproduce, here or against the live data the report came from. The
|
|
|
|
|
+// confound worth noting: on that instance `status` and `workflowId` are INDEXED
|
|
|
|
|
+// and `parentExecutionId` is not, so "works on those two, not on this one" also
|
|
|
|
|
+// separates the indexed union plan from the plain two-pass scan - a different
|
|
|
|
|
+// axis from sparseness. Both are covered below, along with the shapes a real
|
|
|
|
|
+// listing query adds (sort, limit, a second filter, values that match nothing).
|
|
|
|
|
+//
|
|
|
|
|
+// This exists so the claim is answerable by running something rather than by
|
|
|
|
|
+// arguing, and so a future change to the scan resolver or the IN union plan
|
|
|
|
|
+// cannot quietly make it true.
|
|
|
|
|
+
|
|
|
|
|
+#include <smartbotic/database/client.hpp>
|
|
|
|
|
+
|
|
|
|
|
+#include <iostream>
|
|
|
|
|
+#include <string>
|
|
|
|
|
+#include <vector>
|
|
|
|
|
+
|
|
|
|
|
+namespace {
|
|
|
|
|
+
|
|
|
|
|
+using C = smartbotic::database::Client;
|
|
|
|
|
+
|
|
|
|
|
+int failures = 0;
|
|
|
|
|
+int checks = 0;
|
|
|
|
|
+
|
|
|
|
|
+void check(bool ok, const std::string& what) {
|
|
|
|
|
+ ++checks;
|
|
|
|
|
+ std::cout << (ok ? " ok " : " FAIL ") << what << "\n";
|
|
|
|
|
+ if (!ok) ++failures;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+} // namespace
|
|
|
|
|
+
|
|
|
|
|
+int main(int argc, char** argv) {
|
|
|
|
|
+ if (argc < 3) {
|
|
|
|
|
+ std::cerr << "usage: " << argv[0] << " <address> <project>\n";
|
|
|
|
|
+ return 2;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ C::Config cfg;
|
|
|
|
|
+ cfg.address = argv[1];
|
|
|
|
|
+ cfg.project = argv[2];
|
|
|
|
|
+ C client(cfg);
|
|
|
|
|
+ if (!client.connect()) {
|
|
|
|
|
+ std::cerr << "could not connect to " << cfg.address << "\n";
|
|
|
|
|
+ return 2;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ const std::string coll = "executions";
|
|
|
|
|
+
|
|
|
|
|
+ // 200 rows; only 6 carry `parentExecutionId`, i.e. 3% - the sparsity the
|
|
|
|
|
+ // report describes. Two parents own 3 children each, so a per-value count is
|
|
|
|
|
+ // known exactly and an IN over both must return their sum.
|
|
|
|
|
+ const std::vector<std::string> parents = {"exec_parent_a", "exec_parent_b"};
|
|
|
|
|
+ for (int i = 0; i < 200; ++i) {
|
|
|
|
|
+ nlohmann::json d = {{"status", (i % 2) ? "completed" : "failed"},
|
|
|
|
|
+ {"workflowId", "wf1"},
|
|
|
|
|
+ {"triggerType", "manual"}};
|
|
|
|
|
+ if (i < 6) d["parentExecutionId"] = parents[i % 2];
|
|
|
|
|
+ d["startedAt"] = 1000000 + i;
|
|
|
|
|
+ client.upsert(coll, d, "e" + std::to_string(i));
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ auto count = [&](const C::QueryOptions& o) { return client.find(coll, o).size(); };
|
|
|
|
|
+
|
|
|
|
|
+ // ---- 1. the plain two-pass scan (no index declared on the field) ----
|
|
|
|
|
+ check(count({.filters = {C::Filter::eq("parentExecutionId", parents[0])}}) == 3,
|
|
|
|
|
+ "scan: EQ on the sparse field finds its 3 rows");
|
|
|
|
|
+ check(count({.filters = {C::Filter::in("parentExecutionId", {parents[0]})}}) == 3,
|
|
|
|
|
+ "scan: IN with one value equals EQ");
|
|
|
|
|
+ check(count({.filters = {C::Filter::in("parentExecutionId", {parents[0], parents[1]})}}) == 6,
|
|
|
|
|
+ "scan: IN over both parents returns the SUM, not nothing");
|
|
|
|
|
+ check(count({.filters = {C::Filter::in("parentExecutionId", {parents[0], "no_such_parent"})}}) == 3,
|
|
|
|
|
+ "scan: a non-matching value in the list does not suppress the matching one");
|
|
|
|
|
+
|
|
|
|
|
+ // The dense-field control. If this passed while the sparse one failed, the
|
|
|
|
|
+ // report's diagnosis would be right; it is here so the comparison is made.
|
|
|
|
|
+ check(count({.filters = {C::Filter::in("status", {"completed"})}, .limit = 1000}) == 100,
|
|
|
|
|
+ "scan: IN on a DENSE field is unaffected (the control)");
|
|
|
|
|
+
|
|
|
|
|
+ // ---- 2. the same queries with the field INDEXED (the union plan) ----
|
|
|
|
|
+ uint64_t indexed = 0;
|
|
|
|
|
+ check(client.createIndex(coll, "parentExecutionId", indexed),
|
|
|
|
|
+ "an index can be declared on the sparse field");
|
|
|
|
|
+ check(indexed == 6, "the backfill indexed only the rows that HAVE the field");
|
|
|
|
|
+
|
|
|
|
|
+ check(count({.filters = {C::Filter::in("parentExecutionId", {parents[0], parents[1]})}}) == 6,
|
|
|
|
|
+ "indexed: IN over both parents still returns 6");
|
|
|
|
|
+ check(count({.filters = {C::Filter::in("parentExecutionId", {parents[0], "no_such_parent"})}}) == 3,
|
|
|
|
|
+ "indexed: a value with an empty posting list does not empty the result");
|
|
|
|
|
+ check(count({.filters = {C::Filter::eq("parentExecutionId", parents[1])}}) == 3,
|
|
|
|
|
+ "indexed: EQ agrees with the scan");
|
|
|
|
|
+
|
|
|
|
|
+ // ---- 3. the shapes a real listing query adds ----
|
|
|
|
|
+ check(count({.filters = {C::Filter::in("parentExecutionId", {parents[0], parents[1]})},
|
|
|
|
|
+ .sortField = "startedAt", .sortDescending = true}) == 6,
|
|
|
|
|
+ "IN + sort returns the same rows");
|
|
|
|
|
+ check(count({.filters = {C::Filter::in("parentExecutionId", {parents[0], parents[1]})},
|
|
|
|
|
+ .sortField = "startedAt", .sortDescending = true, .limit = 4}) == 4,
|
|
|
|
|
+ "IN + sort + limit pages rather than emptying");
|
|
|
|
|
+ check(count({.filters = {C::Filter::in("parentExecutionId", {parents[0], parents[1]}),
|
|
|
|
|
+ C::Filter::eq("triggerType", "manual")}}) == 6,
|
|
|
|
|
+ "IN AND a second filter on a dense field");
|
|
|
|
|
+
|
|
|
|
|
+ // A long value list, most of which match nothing - the shape produced by
|
|
|
|
|
+ // "fetch the children of the executions on this page".
|
|
|
|
|
+ nlohmann::json many = nlohmann::json::array();
|
|
|
|
|
+ for (const auto& p : parents) many.push_back(p);
|
|
|
|
|
+ for (int i = 0; i < 60; ++i) many.push_back("exec_absent_" + std::to_string(i));
|
|
|
|
|
+ check(count({.filters = {C::Filter::in("parentExecutionId", many)}}) == 6,
|
|
|
|
|
+ "IN with 62 values, 60 of them absent, still returns exactly the 6");
|
|
|
|
|
+
|
|
|
|
|
+ for (int i = 0; i < 200; ++i) client.remove(coll, "e" + std::to_string(i));
|
|
|
|
|
+
|
|
|
|
|
+ std::cout << (failures == 0 ? "PASS" : "FAIL") << ": " << (checks - failures) << "/"
|
|
|
|
|
+ << checks << " checks\n";
|
|
|
|
|
+ return failures == 0 ? 0 : 1;
|
|
|
|
|
+}
|