|
@@ -0,0 +1,2936 @@
|
|
|
|
|
+// v2.11.0 T3 — maintain the relation reverse index on child writes.
|
|
|
|
|
+//
|
|
|
|
|
+// Storage-only: exercises LmdbDocumentStore::put()/del() maintaining the
|
|
|
|
|
+// reverse index declared via set_relations(), using the same set-difference
|
|
|
|
|
+// discipline as maintainIndexes(). No enforcement (Task 4) is involved here
|
|
|
|
|
+// - relation_index_child_count is used purely as the observation point.
|
|
|
|
|
+//
|
|
|
|
|
+// v2.11.0 T4 adds restrict/no_action delete enforcement below
|
|
|
|
|
+// (test_restrict_blocks_and_names_the_blockers,
|
|
|
|
|
+// test_relations_enforced_false_skips_the_check) - unit-level, against
|
|
|
|
|
+// RelationManager + RelationEnforcer + LmdbDocumentStore directly, no
|
|
|
|
|
+// grpc::ServerContext. The RPC-level check lands in Task 6/9.
|
|
|
|
|
+
|
|
|
|
|
+#include <algorithm>
|
|
|
|
|
+#include <atomic>
|
|
|
|
|
+#include <cstdio>
|
|
|
|
|
+#include <filesystem>
|
|
|
|
|
+#include <iostream>
|
|
|
|
|
+#include <string>
|
|
|
|
|
+#include <unistd.h>
|
|
|
|
|
+#include <vector>
|
|
|
|
|
+
|
|
|
|
|
+#include <nlohmann/json.hpp>
|
|
|
|
|
+
|
|
|
|
|
+#include "config/collection_config_manager.hpp"
|
|
|
|
|
+#include "document.hpp"
|
|
|
|
|
+#include "memory_store.hpp"
|
|
|
|
|
+#include "persistence/persistence_manager.hpp"
|
|
|
|
|
+#include "relations/relation_cascade.hpp"
|
|
|
|
|
+#include "relations/relation_enforcement.hpp"
|
|
|
|
|
+#include "relations/relation_manager.hpp"
|
|
|
|
|
+#include "storage/document_store_lmdb.hpp"
|
|
|
|
|
+#include "storage/lmdb_env.hpp"
|
|
|
|
|
+
|
|
|
|
|
+#include <lmdb.h>
|
|
|
|
|
+
|
|
|
|
|
+namespace fs = std::filesystem;
|
|
|
|
|
+
|
|
|
|
|
+using smartbotic::database::CascadeBlocked;
|
|
|
|
|
+using smartbotic::database::CascadeMutation;
|
|
|
|
|
+using smartbotic::database::CascadePlan;
|
|
|
|
|
+using smartbotic::database::CollectionConfigManager;
|
|
|
|
|
+using smartbotic::database::Document;
|
|
|
|
|
+using smartbotic::database::MemoryStore;
|
|
|
|
|
+using smartbotic::database::OnDelete;
|
|
|
|
|
+using smartbotic::database::PersistenceManager;
|
|
|
|
|
+using smartbotic::database::RelationBlock;
|
|
|
|
|
+using smartbotic::database::RelationEnforcer;
|
|
|
|
|
+using smartbotic::database::RelationInfo;
|
|
|
|
|
+using smartbotic::database::RelationManager;
|
|
|
|
|
+using smartbotic::database::RelationImpact;
|
|
|
|
|
+using smartbotic::database::WalEntry;
|
|
|
|
|
+using smartbotic::database::WalOpType;
|
|
|
|
|
+using smartbotic::database::WriteAheadLog;
|
|
|
|
|
+using smartbotic::database::applyCascadeToMemory;
|
|
|
|
|
+using smartbotic::database::commitCascadeLmdb;
|
|
|
|
|
+using smartbotic::database::describeDeleteImpacts;
|
|
|
|
|
+using smartbotic::database::executeCascade;
|
|
|
|
|
+using smartbotic::database::findRelationBlocks;
|
|
|
|
|
+using smartbotic::database::planCascade;
|
|
|
|
|
+using smartbotic::database::ttlExpiryDecision;
|
|
|
|
|
+using smartbotic::database::writeCascadeWal;
|
|
|
|
|
+using smartbotic::db::storage::LmdbDocumentStore;
|
|
|
|
|
+using smartbotic::db::storage::LmdbEnv;
|
|
|
|
|
+using smartbotic::db::storage::LmdbEnvOpts;
|
|
|
|
|
+using smartbotic::db::storage::MissingParentReference;
|
|
|
|
|
+using smartbotic::db::storage::RelationRef;
|
|
|
|
|
+
|
|
|
|
|
+namespace {
|
|
|
|
|
+
|
|
|
|
|
+int g_pass = 0;
|
|
|
|
|
+int g_fail = 0;
|
|
|
|
|
+
|
|
|
|
|
+void check(bool cond, const char* msg) {
|
|
|
|
|
+ if (cond) {
|
|
|
|
|
+ ++g_pass;
|
|
|
|
|
+ } else {
|
|
|
|
|
+ ++g_fail;
|
|
|
|
|
+ std::cerr << "FAIL: " << msg << "\n";
|
|
|
|
|
+ }
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+std::string make_tmpdir(const char* tag) {
|
|
|
|
|
+ static std::atomic<int> counter{0};
|
|
|
|
|
+ std::string path = "/tmp/relmaint-test-" + std::to_string(::getpid()) + "-" +
|
|
|
|
|
+ std::to_string(counter.fetch_add(1)) + "-" + tag;
|
|
|
|
|
+ std::error_code ec;
|
|
|
|
|
+ fs::remove_all(path, ec);
|
|
|
|
|
+ return path;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+struct TmpEnv {
|
|
|
|
|
+ std::string path;
|
|
|
|
|
+ LmdbEnv env;
|
|
|
|
|
+ explicit TmpEnv(const char* tag)
|
|
|
|
|
+ : path(make_tmpdir(tag)),
|
|
|
|
|
+ env(LmdbEnvOpts{path, 64ULL << 20, 256, 126, false}) {}
|
|
|
|
|
+ ~TmpEnv() {
|
|
|
|
|
+ std::error_code ec;
|
|
|
|
|
+ fs::remove_all(path, ec);
|
|
|
|
|
+ }
|
|
|
|
|
+ TmpEnv(const TmpEnv&) = delete;
|
|
|
|
|
+ TmpEnv& operator=(const TmpEnv&) = delete;
|
|
|
|
|
+};
|
|
|
|
|
+
|
|
|
|
|
+// v2.11.0 T12 — a PersistenceManager over its own tmpdir, for the WAL-first
|
|
|
|
|
+// cascade tests. Separate from TmpEnv (which wraps an LmdbEnv) because the
|
|
|
|
|
+// two are deliberately independent stores in this test file, exactly as
|
|
|
|
|
+// they are in the real service (executeCascade takes both, wired together
|
|
|
|
|
+// only by this test/the Delete handler, never by MemoryStore's own
|
|
|
|
|
+// persistCallback_ — see relation_cascade.hpp's file header).
|
|
|
|
|
+struct TmpPersistence {
|
|
|
|
|
+ std::string path;
|
|
|
|
|
+ PersistenceManager::Config cfg;
|
|
|
|
|
+ PersistenceManager pm;
|
|
|
|
|
+ explicit TmpPersistence(const char* tag)
|
|
|
|
|
+ : path(make_tmpdir(tag)), cfg(makeConfig(path)), pm(cfg) {}
|
|
|
|
|
+ ~TmpPersistence() {
|
|
|
|
|
+ std::error_code ec;
|
|
|
|
|
+ fs::remove_all(path, ec);
|
|
|
|
|
+ }
|
|
|
|
|
+ TmpPersistence(const TmpPersistence&) = delete;
|
|
|
|
|
+ TmpPersistence& operator=(const TmpPersistence&) = delete;
|
|
|
|
|
+
|
|
|
|
|
+private:
|
|
|
|
|
+ static PersistenceManager::Config makeConfig(const std::string& dir) {
|
|
|
|
|
+ PersistenceManager::Config c;
|
|
|
|
|
+ c.dataDir = dir;
|
|
|
|
|
+ return c;
|
|
|
|
|
+ }
|
|
|
|
|
+};
|
|
|
|
|
+
|
|
|
|
|
+// Replay every WAL entry under `dataDir`/wal directly, bypassing
|
|
|
|
|
+// PersistenceManager/MemoryStore entirely — the most direct way to answer
|
|
|
|
|
+// "does the WAL file actually contain this entry", independent of whatever
|
|
|
|
|
+// LMDB or MemoryStore ended up with.
|
|
|
|
|
+std::vector<WalEntry> replayRawWal(const std::string& dataDir) {
|
|
|
|
|
+ WriteAheadLog::Config wcfg;
|
|
|
|
|
+ wcfg.walDir = std::filesystem::path(dataDir) / "wal";
|
|
|
|
|
+ WriteAheadLog wal(wcfg);
|
|
|
|
|
+ std::vector<WalEntry> out;
|
|
|
|
|
+ wal.replay(0, [&](const WalEntry& e) { out.push_back(e); });
|
|
|
|
|
+ return out;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+bool walHasDelete(const std::vector<WalEntry>& entries,
|
|
|
|
|
+ const std::string& collection, const std::string& id) {
|
|
|
|
|
+ for (const auto& e : entries) {
|
|
|
|
|
+ if (e.opType == WalOpType::DELETE && e.collection == collection && e.documentId == id) {
|
|
|
|
|
+ return true;
|
|
|
|
|
+ }
|
|
|
|
|
+ }
|
|
|
|
|
+ return false;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+bool walHasUpdate(const std::vector<WalEntry>& entries,
|
|
|
|
|
+ const std::string& collection, const std::string& id) {
|
|
|
|
|
+ for (const auto& e : entries) {
|
|
|
|
|
+ if (e.opType == WalOpType::UPDATE && e.collection == collection && e.documentId == id) {
|
|
|
|
|
+ return true;
|
|
|
|
|
+ }
|
|
|
|
|
+ }
|
|
|
|
|
+ return false;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// -------------------------------------------------------------------------
|
|
|
|
|
+
|
|
|
|
|
+void test_index_follows_the_child_field() {
|
|
|
|
|
+ TmpEnv t("rel-maint");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions", {{"exec_wf", "workflowId"}});
|
|
|
|
|
+
|
|
|
|
|
+ auto put = [&](const std::string& id, const nlohmann::json& data) {
|
|
|
|
|
+ Document d; d.id = id; d.collection = "executions"; d.set_data(data);
|
|
|
|
|
+ store.put("executions", id, d);
|
|
|
|
|
+ };
|
|
|
|
|
+
|
|
|
|
|
+ put("e1", {{"workflowId", "wf-1"}});
|
|
|
|
|
+ put("e2", {{"workflowId", "wf-1"}});
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 2, "two children");
|
|
|
|
|
+
|
|
|
|
|
+ put("e2", {{"workflowId", "wf-2"}}); // re-point
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 1, "left the old parent");
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-2") == 1, "joined the new one");
|
|
|
|
|
+
|
|
|
|
|
+ store.del("executions", "e1");
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 0, "delete removes it");
|
|
|
|
|
+
|
|
|
|
|
+ // Absent and null are NOT references: they never block a delete and never
|
|
|
|
|
+ // count as dangling.
|
|
|
|
|
+ put("e3", {{"other", 1}});
|
|
|
|
|
+ put("e4", {{"workflowId", nullptr}});
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 0, "absent adds nothing");
|
|
|
|
|
+
|
|
|
|
|
+ // An ARRAY-valued reference contributes one posting per element.
|
|
|
|
|
+ store.set_relations("nodes", {{"node_creds", "config.credentialIds"}});
|
|
|
|
|
+ Document n; n.id = "n1"; n.collection = "nodes";
|
|
|
|
|
+ n.set_data({{"config", {{"credentialIds", {"c1", "c2"}}}}});
|
|
|
|
|
+ store.put("nodes", "n1", n);
|
|
|
|
|
+ check(store.relation_index_child_count("node_creds", "c1") == 1, "array element 1");
|
|
|
|
|
+ check(store.relation_index_child_count("node_creds", "c2") == 1, "array element 2");
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// A collection with no relations declared must pay nothing and never touch
|
|
|
|
|
+// any relation sub-db, mirroring how maintainIndexes short-circuits.
|
|
|
|
|
+void test_undeclared_collection_maintains_nothing() {
|
|
|
|
|
+ TmpEnv t("rel-none");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+
|
|
|
|
|
+ Document d; d.id = "x1"; d.collection = "misc"; d.set_data({{"workflowId", "wf-9"}});
|
|
|
|
|
+ store.put("misc", "x1", d);
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-9") == 0,
|
|
|
|
|
+ "no relation declared on this collection means no posting written");
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// Unrelated field updates must not touch the index - the set-difference is
|
|
|
|
|
+// exact, not "rewrite unconditionally".
|
|
|
|
|
+void test_unrelated_update_leaves_the_posting_alone() {
|
|
|
|
|
+ TmpEnv t("rel-stable");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions", {{"exec_wf", "workflowId"}});
|
|
|
|
|
+
|
|
|
|
|
+ Document d1; d1.id = "e1"; d1.collection = "executions";
|
|
|
|
|
+ d1.set_data({{"workflowId", "wf-1"}, {"status", "running"}});
|
|
|
|
|
+ store.put("executions", "e1", d1);
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 1, "initial posting");
|
|
|
|
|
+
|
|
|
|
|
+ Document d2; d2.id = "e1"; d2.collection = "executions";
|
|
|
|
|
+ d2.set_data({{"workflowId", "wf-1"}, {"status", "completed"}});
|
|
|
|
|
+ store.put("executions", "e1", d2);
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 1,
|
|
|
|
|
+ "still one posting after an unrelated field changed");
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// Survives across a fresh store instance over the same env - i.e. the reverse
|
|
|
|
|
+// index sub-db created at write time was registered via cacheCommittedDbi
|
|
|
|
|
+// after commit, not merely usable within the same process instance.
|
|
|
|
|
+void test_posting_visible_after_reopen() {
|
|
|
|
|
+ TmpEnv t("rel-reopen");
|
|
|
|
|
+ {
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions", {{"exec_wf", "workflowId"}});
|
|
|
|
|
+ Document d; d.id = "e1"; d.collection = "executions";
|
|
|
|
|
+ d.set_data({{"workflowId", "wf-1"}});
|
|
|
|
|
+ store.put("executions", "e1", d);
|
|
|
|
|
+ }
|
|
|
|
|
+ LmdbDocumentStore reopened(t.env);
|
|
|
|
|
+ check(reopened.relation_index_child_count("exec_wf", "wf-1") == 1,
|
|
|
|
|
+ "posting survives a fresh LmdbDocumentStore over the same env");
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// v2.11.0 T4 — restrict blocks a delete and names the blockers; no_action
|
|
|
|
|
+// permits it and leaves the reference dangling, deliberately.
|
|
|
|
|
+//
|
|
|
|
|
+// RelationManager's declaration registry lives in a MemoryStore's
|
|
|
|
|
+// `_relations` collection (see test_relation_manager.cpp's Fixture); the
|
|
|
|
|
+// reverse index it queries lives in a separate per-project LmdbDocumentStore
|
|
|
|
|
+// (T2/T3, exercised above). This test wires both together the way Task 6's
|
|
|
|
|
+// Delete handler eventually will.
|
|
|
|
|
+void test_restrict_blocks_and_names_the_blockers() {
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+
|
|
|
|
|
+ TmpEnv t("rel-enforce-restrict");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions", {{"exec_wf", "workflowId"}});
|
|
|
|
|
+
|
|
|
|
|
+ auto put = [&](const std::string& id, const nlohmann::json& data) {
|
|
|
|
|
+ Document d; d.id = id; d.collection = "executions"; d.set_data(data);
|
|
|
|
|
+ store.put("executions", id, d);
|
|
|
|
|
+ };
|
|
|
|
|
+ put("e1", {{"workflowId", "wf-1"}});
|
|
|
|
|
+ put("e2", {{"workflowId", "wf-1"}});
|
|
|
|
|
+
|
|
|
|
|
+ RelationInfo rel;
|
|
|
|
|
+ rel.name = "default:exec_wf";
|
|
|
|
|
+ rel.child = "default:executions";
|
|
|
|
|
+ rel.childField = "workflowId";
|
|
|
|
|
+ rel.parent = "default:workflows";
|
|
|
|
|
+ rel.onDelete = OnDelete::Restrict;
|
|
|
|
|
+
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "declared the restrict relation");
|
|
|
|
|
+
|
|
|
|
|
+ bool enforced = true;
|
|
|
|
|
+ RelationEnforcer enforcer(rm, store, enforced);
|
|
|
|
|
+
|
|
|
|
|
+ std::string err;
|
|
|
|
|
+ check(!enforcer.canDelete("default:workflows", "wf-1", err), "restrict blocks");
|
|
|
|
|
+ check(err.find("exec_wf") != std::string::npos, "names the relation");
|
|
|
|
|
+ check(err.find("2") != std::string::npos, "gives the child count");
|
|
|
|
|
+ check(err.find("e1") != std::string::npos, "samples a blocking id");
|
|
|
|
|
+
|
|
|
|
|
+ // no_action permits it and leaves the reference dangling, deliberately.
|
|
|
|
|
+ // RelationManager has no update-in-place; re-declare with the new
|
|
|
|
|
+ // policy the same way an operator would via dropRelation + createRelation.
|
|
|
|
|
+ check(rm.dropRelation(rel.name, mgrErr), "dropped to change on_delete");
|
|
|
|
|
+ rel.onDelete = OnDelete::NoAction;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "re-declared as no_action");
|
|
|
|
|
+
|
|
|
|
|
+ err.clear();
|
|
|
|
|
+ check(enforcer.canDelete("default:workflows", "wf-1", err), "no_action permits");
|
|
|
|
|
+
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// v2.11.0 T12 — cascade/set_null ARE now destructive (see the tests further
|
|
|
|
|
+// below: test_cascade_deletes_scalar_children_wal_first,
|
|
|
|
|
+// test_array_reference_pulls_id_and_keeps_document), but that destruction
|
|
|
|
|
+// happens entirely OUTSIDE findRelationBlocks()/canDelete() - those two only
|
|
|
|
|
+// ever implement `restrict`, deliberately (see relation_enforcement.hpp's
|
|
|
|
|
+// file header: "Only restrict blocks"). A relation declaring cascade or
|
|
|
|
|
+// set_null must therefore still never BLOCK a delete via this path, exactly
|
|
|
|
|
+// like no_action - the actual cascade/set_null execution is a separate step
|
|
|
|
|
+// (executeCascade, called by the Delete handler only after this check has
|
|
|
|
|
+// already permitted the delete).
|
|
|
|
|
+void test_cascade_and_set_null_never_block_via_restrict_check() {
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+
|
|
|
|
|
+ TmpEnv t("rel-enforce-cascade");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions", {{"exec_wf", "workflowId"}});
|
|
|
|
|
+
|
|
|
|
|
+ Document d; d.id = "e1"; d.collection = "executions";
|
|
|
|
|
+ d.set_data({{"workflowId", "wf-1"}});
|
|
|
|
|
+ store.put("executions", "e1", d);
|
|
|
|
|
+
|
|
|
|
|
+ RelationInfo rel;
|
|
|
|
|
+ rel.name = "default:exec_wf";
|
|
|
|
|
+ rel.child = "default:executions";
|
|
|
|
|
+ rel.childField = "workflowId";
|
|
|
|
|
+ rel.parent = "default:workflows";
|
|
|
|
|
+ rel.onDelete = OnDelete::Cascade;
|
|
|
|
|
+
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "declared a cascade relation");
|
|
|
|
|
+
|
|
|
|
|
+ auto blocks = findRelationBlocks(rm, store, "default:workflows", "wf-1");
|
|
|
|
|
+ check(blocks.empty(), "cascade does not block - not yet destructive (Task 12)");
|
|
|
|
|
+
|
|
|
|
|
+ check(rm.dropRelation(rel.name, mgrErr), "dropped to change on_delete");
|
|
|
|
|
+ rel.onDelete = OnDelete::SetNull;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "re-declared as set_null");
|
|
|
|
|
+
|
|
|
|
|
+ blocks = findRelationBlocks(rm, store, "default:workflows", "wf-1");
|
|
|
|
|
+ check(blocks.empty(), "set_null does not block - not yet destructive (Task 12)");
|
|
|
|
|
+
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// relationsEnforced=false (CollectionCfg, Task 8) skips the check entirely,
|
|
|
|
|
+// even though a restrict relation with live children exists. Deliberately
|
|
|
|
|
+// not retroactive - see RelationEnforcer::canDelete's doc comment.
|
|
|
|
|
+void test_relations_enforced_false_skips_the_check() {
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+
|
|
|
|
|
+ TmpEnv t("rel-enforce-disabled");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions", {{"exec_wf", "workflowId"}});
|
|
|
|
|
+
|
|
|
|
|
+ Document d; d.id = "e1"; d.collection = "executions";
|
|
|
|
|
+ d.set_data({{"workflowId", "wf-1"}});
|
|
|
|
|
+ store.put("executions", "e1", d);
|
|
|
|
|
+
|
|
|
|
|
+ RelationInfo rel;
|
|
|
|
|
+ rel.name = "default:exec_wf";
|
|
|
|
|
+ rel.child = "default:executions";
|
|
|
|
|
+ rel.childField = "workflowId";
|
|
|
|
|
+ rel.parent = "default:workflows";
|
|
|
|
|
+ rel.onDelete = OnDelete::Restrict;
|
|
|
|
|
+
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "declared the restrict relation");
|
|
|
|
|
+
|
|
|
|
|
+ bool enforced = false; // per-collection switch, Task 8
|
|
|
|
|
+ RelationEnforcer enforcer(rm, store, enforced);
|
|
|
|
|
+
|
|
|
|
|
+ std::string err;
|
|
|
|
|
+ check(enforcer.canDelete("default:workflows", "wf-1", err),
|
|
|
|
|
+ "enforcement disabled for this collection lets the delete through - and "
|
|
|
|
|
+ "the resulting dangling references will NOT be found by re-enabling it, "
|
|
|
|
|
+ "only by `relations check`");
|
|
|
|
|
+ check(err.empty(), "no error text is populated on a permitted delete");
|
|
|
|
|
+
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// v2.11.0 T5 — DescribeDelete's enumeration. The case that matters most:
|
|
|
|
|
+// a parent with children under a restrict relation AND children under a
|
|
|
|
|
+// no_action relation. Unlike findRelationBlocks (which only reports
|
|
|
|
|
+// blockers), describeDeleteImpacts must report BOTH relations, and mark
|
|
|
|
|
+// only the restrict one as blocking.
|
|
|
|
|
+void test_describe_delete_reports_restrict_and_no_action() {
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+
|
|
|
|
|
+ TmpEnv t("rel-describe-mixed");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions", {{"exec_wf", "workflowId"}});
|
|
|
|
|
+ store.set_relations("comments", {{"comment_wf", "workflowId"}});
|
|
|
|
|
+
|
|
|
|
|
+ auto put = [&](const std::string& coll, const std::string& id, const nlohmann::json& data) {
|
|
|
|
|
+ Document d; d.id = id; d.collection = coll; d.set_data(data);
|
|
|
|
|
+ store.put(coll, id, d);
|
|
|
|
|
+ };
|
|
|
|
|
+ put("executions", "e1", {{"workflowId", "wf-1"}});
|
|
|
|
|
+ put("executions", "e2", {{"workflowId", "wf-1"}});
|
|
|
|
|
+ put("comments", "c1", {{"workflowId", "wf-1"}});
|
|
|
|
|
+
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ RelationInfo restrictRel;
|
|
|
|
|
+ restrictRel.name = "default:exec_wf";
|
|
|
|
|
+ restrictRel.child = "default:executions";
|
|
|
|
|
+ restrictRel.childField = "workflowId";
|
|
|
|
|
+ restrictRel.parent = "default:workflows";
|
|
|
|
|
+ restrictRel.onDelete = OnDelete::Restrict;
|
|
|
|
|
+ check(rm.createRelation(restrictRel, mgrErr), "declared the restrict relation");
|
|
|
|
|
+
|
|
|
|
|
+ RelationInfo noActionRel;
|
|
|
|
|
+ noActionRel.name = "default:comment_wf";
|
|
|
|
|
+ noActionRel.child = "default:comments";
|
|
|
|
|
+ noActionRel.childField = "workflowId";
|
|
|
|
|
+ noActionRel.parent = "default:workflows";
|
|
|
|
|
+ noActionRel.onDelete = OnDelete::NoAction;
|
|
|
|
|
+ check(rm.createRelation(noActionRel, mgrErr), "declared the no_action relation");
|
|
|
|
|
+
|
|
|
|
|
+ auto impacts = describeDeleteImpacts(rm, store, "default:workflows", "wf-1");
|
|
|
|
|
+ check(impacts.size() == 2, "both relations are reported, not just the blocker");
|
|
|
|
|
+
|
|
|
|
|
+ const RelationImpact* restrictImpact = nullptr;
|
|
|
|
|
+ const RelationImpact* noActionImpact = nullptr;
|
|
|
|
|
+ for (const auto& imp : impacts) {
|
|
|
|
|
+ if (imp.relation == "default:exec_wf") restrictImpact = &imp;
|
|
|
|
|
+ if (imp.relation == "default:comment_wf") noActionImpact = &imp;
|
|
|
|
|
+ }
|
|
|
|
|
+ check(restrictImpact != nullptr, "restrict relation present");
|
|
|
|
|
+ check(noActionImpact != nullptr, "no_action relation present");
|
|
|
|
|
+ if (restrictImpact) {
|
|
|
|
|
+ check(restrictImpact->childCount == 2, "restrict relation counts both children");
|
|
|
|
|
+ check(restrictImpact->blocks, "restrict relation with live children blocks");
|
|
|
|
|
+ check(restrictImpact->sampleChildIds.size() == 2, "samples both blocking ids");
|
|
|
|
|
+ }
|
|
|
|
|
+ if (noActionImpact) {
|
|
|
|
|
+ check(noActionImpact->childCount == 1, "no_action relation counts its child");
|
|
|
|
|
+ check(!noActionImpact->blocks, "no_action never blocks, even with live children");
|
|
|
|
|
+ check(noActionImpact->sampleChildIds.size() == 1, "still samples the id for visibility");
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // Consistency with the actual enforcement decision: RelationEnforcer
|
|
|
|
|
+ // (what Delete() really calls) agrees that this delete is blocked, by
|
|
|
|
|
+ // exactly the relation describeDeleteImpacts flagged.
|
|
|
|
|
+ bool enforced = true;
|
|
|
|
|
+ RelationEnforcer enforcer(rm, store, enforced);
|
|
|
|
|
+ std::string err;
|
|
|
|
|
+ check(!enforcer.canDelete("default:workflows", "wf-1", err),
|
|
|
|
|
+ "a real delete right now would in fact be blocked, agreeing with the impact");
|
|
|
|
|
+
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// A relation with no live children is reported (so an operator can see the
|
|
|
|
|
+// declaration exists) but never blocks - zero postings is not a reference,
|
|
|
|
|
+// matching findRelationBlocks' "zero count is not a block" rule.
|
|
|
|
|
+void test_describe_delete_zero_children_does_not_block() {
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+
|
|
|
|
|
+ TmpEnv t("rel-describe-empty");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions", {{"exec_wf", "workflowId"}});
|
|
|
|
|
+
|
|
|
|
|
+ RelationInfo rel;
|
|
|
|
|
+ rel.name = "default:exec_wf";
|
|
|
|
|
+ rel.child = "default:executions";
|
|
|
|
|
+ rel.childField = "workflowId";
|
|
|
|
|
+ rel.parent = "default:workflows";
|
|
|
|
|
+ rel.onDelete = OnDelete::Restrict;
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "declared the relation");
|
|
|
|
|
+
|
|
|
|
|
+ auto impacts = describeDeleteImpacts(rm, store, "default:workflows", "wf-nonexistent");
|
|
|
|
|
+ check(impacts.size() == 1, "the declared relation is reported even with no children");
|
|
|
|
|
+ check(impacts[0].childCount == 0, "no children referencing this parent id");
|
|
|
|
|
+ check(!impacts[0].blocks, "zero children never blocks");
|
|
|
|
|
+ check(impacts[0].sampleChildIds.empty(), "no sample ids when there are no children");
|
|
|
|
|
+
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// A collection with no declared relations at all reports an empty impacts
|
|
|
|
|
+// list - describeDeleteImpacts, like findRelationBlocks, has nothing to say
|
|
|
|
|
+// about a parent nothing points at.
|
|
|
|
|
+void test_describe_delete_no_relations_declared() {
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+
|
|
|
|
|
+ TmpEnv t("rel-describe-none");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+
|
|
|
|
|
+ auto impacts = describeDeleteImpacts(rm, store, "default:orphan_collection", "anything");
|
|
|
|
|
+ check(impacts.empty(), "no relations declared means no impacts reported");
|
|
|
|
|
+
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// -------------------------------------------------------------------------
|
|
|
|
|
+// v2.11.0 T12 — cascade/set_null made destructive, WAL-first.
|
|
|
|
|
+// -------------------------------------------------------------------------
|
|
|
|
|
+
|
|
|
|
|
+// The brief's first required test: cascade delete of a parent with three
|
|
|
|
|
+// scalar-referencing children. Children gone, index entries gone, and -
|
|
|
|
|
+// the part most likely to be skipped - a WAL entry exists for every child
|
|
|
|
|
+// mutation (not just the parent). Would fail if regressed: if
|
|
|
|
|
+// commitCascadeLmdb() forgot to delete a child, that child's LMDB row or
|
|
|
|
|
+// index posting would survive; if writeCascadeWal() forgot a child, the raw
|
|
|
|
|
+// WAL replay would not contain its DELETE entry - this reads the WAL file
|
|
|
|
|
+// directly (replayRawWal), not through any code path that could paper over
|
|
|
|
|
+// a missing log call.
|
|
|
|
|
+void test_cascade_deletes_scalar_children_wal_first() {
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+ cfgManager.loadFromStore();
|
|
|
|
|
+
|
|
|
|
|
+ TmpEnv t("rel-cascade-scalar");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions", {{"exec_wf", "workflowId"}});
|
|
|
|
|
+
|
|
|
|
|
+ TmpPersistence p("rel-cascade-scalar-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ auto putBoth = [&](const std::string& id, const nlohmann::json& data) {
|
|
|
|
|
+ Document d; d.id = id; d.collection = "executions"; d.set_data(data);
|
|
|
|
|
+ store.put("executions", id, d);
|
|
|
|
|
+ mstore.loadDocument("default:executions", d);
|
|
|
|
|
+ };
|
|
|
|
|
+ putBoth("e1", {{"workflowId", "wf-1"}});
|
|
|
|
|
+ putBoth("e2", {{"workflowId", "wf-1"}});
|
|
|
|
|
+ putBoth("e3", {{"workflowId", "wf-1"}});
|
|
|
|
|
+
|
|
|
|
|
+ {
|
|
|
|
|
+ Document parent; parent.id = "wf-1"; parent.collection = "workflows";
|
|
|
|
|
+ parent.set_data({{"name", "example"}});
|
|
|
|
|
+ store.put("workflows", "wf-1", parent);
|
|
|
|
|
+ mstore.loadDocument("default:workflows", parent);
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ RelationInfo rel;
|
|
|
|
|
+ rel.name = "default:exec_wf";
|
|
|
|
|
+ rel.child = "default:executions";
|
|
|
|
|
+ rel.childField = "workflowId";
|
|
|
|
|
+ rel.parent = "default:workflows";
|
|
|
|
|
+ rel.onDelete = OnDelete::Cascade;
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "declared the cascade relation");
|
|
|
|
|
+
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 3, "three children indexed");
|
|
|
|
|
+
|
|
|
|
|
+ // v2.11.0 T12 review (C2) - a notify callback wired the way
|
|
|
|
|
+ // DatabaseGrpcImpl::Delete wires DatabaseService::notifyReplicationAndEvents,
|
|
|
|
|
+ // recorded here instead of actually queuing replication/publishing events
|
|
|
|
|
+ // (this test has no DatabaseService). Confirms executeCascade() actually
|
|
|
|
|
+ // drives it once per mutation plus once for the parent, with the right
|
|
|
|
|
+ // event type each time - the wiring C2 added, not just that it compiles.
|
|
|
|
|
+ struct Notification {
|
|
|
|
|
+ std::string collection, id;
|
|
|
|
|
+ bool hadDoc;
|
|
|
|
|
+ smartbotic::database::EventType eventType;
|
|
|
|
|
+ };
|
|
|
|
|
+ std::vector<Notification> notifications;
|
|
|
|
|
+ auto notify = [&](const std::string& coll, const std::string& id,
|
|
|
|
|
+ const std::optional<Document>& doc,
|
|
|
|
|
+ smartbotic::database::EventType et) {
|
|
|
|
|
+ notifications.push_back({coll, id, doc.has_value(), et});
|
|
|
|
|
+ };
|
|
|
|
|
+
|
|
|
|
|
+ const bool parentExisted = executeCascade(rm, store, p.pm, mstore, cfgManager,
|
|
|
|
|
+ "default:workflows", "wf-1", notify);
|
|
|
|
|
+ check(parentExisted, "the parent document was present and removed");
|
|
|
|
|
+
|
|
|
|
|
+ check(notifications.size() == 4, "notified for e1, e2, e3 and the parent - nothing missed, nothing extra");
|
|
|
|
|
+ int deleteNotifications = 0;
|
|
|
|
|
+ for (const auto& n : notifications) {
|
|
|
|
|
+ check(!n.hadDoc, "every notification here is a delete - no doc payload");
|
|
|
|
|
+ check(n.eventType == smartbotic::database::EventType::DELETE, "every notification is a DELETE");
|
|
|
|
|
+ if (n.eventType == smartbotic::database::EventType::DELETE) ++deleteNotifications;
|
|
|
|
|
+ }
|
|
|
|
|
+ check(deleteNotifications == 4, "all four notifications are DELETE (3 children + parent)");
|
|
|
|
|
+
|
|
|
|
|
+ // LMDB: children gone, index gone.
|
|
|
|
|
+ check(!store.get("executions", "e1").has_value(), "e1 gone from LMDB");
|
|
|
|
|
+ check(!store.get("executions", "e2").has_value(), "e2 gone from LMDB");
|
|
|
|
|
+ check(!store.get("executions", "e3").has_value(), "e3 gone from LMDB");
|
|
|
|
|
+ check(!store.get("workflows", "wf-1").has_value(), "parent gone from LMDB");
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 0,
|
|
|
|
|
+ "reverse index has no postings left for wf-1");
|
|
|
|
|
+
|
|
|
|
|
+ // MemoryStore: same, applied via applyCascadeToMemory / unloadDocument.
|
|
|
|
|
+ check(!mstore.get("default:executions", "e1").has_value(), "e1 gone from MemoryStore");
|
|
|
|
|
+ check(!mstore.get("default:executions", "e2").has_value(), "e2 gone from MemoryStore");
|
|
|
|
|
+ check(!mstore.get("default:executions", "e3").has_value(), "e3 gone from MemoryStore");
|
|
|
|
|
+ check(!mstore.get("default:workflows", "wf-1").has_value(), "parent gone from MemoryStore");
|
|
|
|
|
+
|
|
|
|
|
+ // WAL: every child mutation AND the parent got its own DELETE entry -
|
|
|
|
|
+ // read straight from the WAL file, not inferred from LMDB/MemoryStore
|
|
|
|
|
+ // state (which could be right for the wrong reason if WAL logging were
|
|
|
|
|
+ // silently skipped).
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+ auto entries = replayRawWal(p.path);
|
|
|
|
|
+ check(walHasDelete(entries, "default:executions", "e1"), "WAL has DELETE for e1");
|
|
|
|
|
+ check(walHasDelete(entries, "default:executions", "e2"), "WAL has DELETE for e2");
|
|
|
|
|
+ check(walHasDelete(entries, "default:executions", "e3"), "WAL has DELETE for e3");
|
|
|
|
|
+ check(walHasDelete(entries, "default:workflows", "wf-1"), "WAL has DELETE for the parent");
|
|
|
|
|
+
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// The brief's second required test, and the one the file header calls out
|
|
|
|
|
+// as the place a MySQL mental model actively misleads: an array-valued
|
|
|
|
|
+// reference under `cascade` pulls the id and KEEPS the document, exactly
|
|
|
|
|
+// like set_null would. Would fail if regressed: a naive port of "cascade
|
|
|
|
|
+// deletes the child" would delete the node here, which this test catches
|
|
|
|
|
+// directly (node must still exist afterward), not just check the array
|
|
|
|
|
+// shrank.
|
|
|
|
|
+void test_array_reference_pulls_id_and_keeps_document() {
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+ cfgManager.loadFromStore();
|
|
|
|
|
+
|
|
|
|
|
+ TmpEnv t("rel-cascade-array");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("nodes", {{"node_creds", "config.credentialIds"}});
|
|
|
|
|
+
|
|
|
|
|
+ TmpPersistence p("rel-cascade-array-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ {
|
|
|
|
|
+ Document n; n.id = "n1"; n.collection = "nodes";
|
|
|
|
|
+ n.set_data({{"config", {{"credentialIds", {"c1", "c2", "c3"}}}}});
|
|
|
|
|
+ store.put("nodes", "n1", n);
|
|
|
|
|
+ mstore.loadDocument("default:nodes", n);
|
|
|
|
|
+ }
|
|
|
|
|
+ {
|
|
|
|
|
+ Document parent; parent.id = "c1"; parent.collection = "credentials";
|
|
|
|
|
+ parent.set_data({{"name", "prod-key"}});
|
|
|
|
|
+ store.put("credentials", "c1", parent);
|
|
|
|
|
+ mstore.loadDocument("default:credentials", parent);
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // Cascade, deliberately - this is exactly the policy a MySQL-trained
|
|
|
|
|
+ // instinct expects to delete the node. It must not.
|
|
|
|
|
+ RelationInfo rel;
|
|
|
|
|
+ rel.name = "default:node_creds";
|
|
|
|
|
+ rel.child = "default:nodes";
|
|
|
|
|
+ rel.childField = "config.credentialIds";
|
|
|
|
|
+ rel.parent = "default:credentials";
|
|
|
|
|
+ rel.onDelete = OnDelete::Cascade;
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "declared the cascade relation on an array field");
|
|
|
|
|
+
|
|
|
|
|
+ // v2.11.0 T12 review (C2) - same notify-recording as the scalar test,
|
|
|
|
|
+ // to confirm the UPDATE (not DELETE) path also notifies correctly with
|
|
|
|
|
+ // the post-mutation document attached.
|
|
|
|
|
+ struct Notification {
|
|
|
|
|
+ std::string collection, id;
|
|
|
|
|
+ bool hadDoc;
|
|
|
|
|
+ smartbotic::database::EventType eventType;
|
|
|
|
|
+ };
|
|
|
|
|
+ std::vector<Notification> notifications;
|
|
|
|
|
+ auto notify = [&](const std::string& coll, const std::string& id,
|
|
|
|
|
+ const std::optional<Document>& doc,
|
|
|
|
|
+ smartbotic::database::EventType et) {
|
|
|
|
|
+ notifications.push_back({coll, id, doc.has_value(), et});
|
|
|
|
|
+ };
|
|
|
|
|
+
|
|
|
|
|
+ const bool parentExisted = executeCascade(rm, store, p.pm, mstore, cfgManager,
|
|
|
|
|
+ "default:credentials", "c1", notify);
|
|
|
|
|
+ check(parentExisted, "the credential document was present and removed");
|
|
|
|
|
+
|
|
|
|
|
+ check(notifications.size() == 2, "notified for the node update and the parent delete");
|
|
|
|
|
+ check(notifications[0].collection == "default:nodes" && notifications[0].id == "n1" &&
|
|
|
|
|
+ notifications[0].hadDoc &&
|
|
|
|
|
+ notifications[0].eventType == smartbotic::database::EventType::UPDATE,
|
|
|
|
|
+ "node notification is an UPDATE carrying the mutated doc");
|
|
|
|
|
+ check(notifications[1].collection == "default:credentials" && notifications[1].id == "c1" &&
|
|
|
|
|
+ !notifications[1].hadDoc &&
|
|
|
|
|
+ notifications[1].eventType == smartbotic::database::EventType::DELETE,
|
|
|
|
|
+ "parent notification is a DELETE with no doc payload");
|
|
|
|
|
+
|
|
|
|
|
+ // The node survives, in BOTH stores, with c1 pulled and the other two ids intact.
|
|
|
|
|
+ auto lmdbNode = store.get("nodes", "n1");
|
|
|
|
|
+ check(lmdbNode.has_value(), "node n1 still exists in LMDB - not deleted");
|
|
|
|
|
+ if (lmdbNode) {
|
|
|
|
|
+ auto ids = lmdbNode->data()["config"]["credentialIds"];
|
|
|
|
|
+ check(ids.is_array() && ids.size() == 2, "two ids remain");
|
|
|
|
|
+ check(std::find(ids.begin(), ids.end(), nlohmann::json("c1")) == ids.end(),
|
|
|
|
|
+ "c1 was pulled");
|
|
|
|
|
+ check(std::find(ids.begin(), ids.end(), nlohmann::json("c2")) != ids.end(),
|
|
|
|
|
+ "c2 survives");
|
|
|
|
|
+ check(std::find(ids.begin(), ids.end(), nlohmann::json("c3")) != ids.end(),
|
|
|
|
|
+ "c3 survives");
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ auto memNode = mstore.get("default:nodes", "n1");
|
|
|
|
|
+ check(memNode.has_value(), "node n1 still exists in MemoryStore - not deleted");
|
|
|
|
|
+ if (memNode) {
|
|
|
|
|
+ auto ids = memNode->data()["config"]["credentialIds"];
|
|
|
|
|
+ check(ids.is_array() && ids.size() == 2, "MemoryStore copy agrees: two ids remain");
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ check(store.relation_index_child_count("node_creds", "c1") == 0,
|
|
|
|
|
+ "the pulled posting is gone from the reverse index");
|
|
|
|
|
+ check(store.relation_index_child_count("node_creds", "c2") == 1,
|
|
|
|
|
+ "c2's posting is untouched");
|
|
|
|
|
+
|
|
|
|
|
+ check(!store.get("credentials", "c1").has_value(), "the credential itself IS deleted");
|
|
|
|
|
+
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+ auto entries = replayRawWal(p.path);
|
|
|
|
|
+ check(walHasUpdate(entries, "default:nodes", "n1"),
|
|
|
|
|
+ "WAL has an UPDATE for the node (not a DELETE) - the array rule");
|
|
|
|
|
+ check(walHasDelete(entries, "default:credentials", "c1"), "WAL has DELETE for the parent");
|
|
|
|
|
+
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// The brief's third required test: simulate a crash between the WAL write
|
|
|
|
|
+// (step 3) and the LMDB commit (step 4) by calling exactly those primitives
|
|
|
|
|
+// in that order and stopping there - never calling commitCascadeLmdb() or
|
|
|
|
|
+// applyCascadeToMemory(). This is what a real crash right after
|
|
|
|
|
+// PersistenceManager::flushWal() returns looks like from the next boot's
|
|
|
|
|
+// point of view: the WAL file has everything, LMDB has nothing.
|
|
|
|
|
+//
|
|
|
|
|
+// "Restart" = a FRESH MemoryStore, recovered via a second PersistenceManager
|
|
|
|
|
+// instance over the same dataDir (recover() replays from sequence 0 since
|
|
|
|
|
+// no snapshot exists - the fresh-install path in persistence_manager.cpp).
|
|
|
|
|
+// Recovery must converge on the correct post-cascade state even though the
|
|
|
|
|
+// LMDB transaction that would have made it "real" never ran - proving
|
|
|
|
|
+// MemoryStore's correctness after a crash depends on the WAL, not on how
|
|
|
|
|
+// far the LMDB commit got. Then recovering a SECOND time onto the SAME
|
|
|
|
|
+// (already-recovered) store checks that replaying an already-applied
|
|
|
|
|
+// delete is a no-op - it must not throw, and the state must not change.
|
|
|
|
|
+//
|
|
|
|
|
+// ⚠ Scope of what this test can and cannot prove: it exercises the WAL
|
|
|
|
|
+// primitives in isolation, so it proves the WAL entries are self-sufficient
|
|
|
|
|
+// for correct recovery. It does NOT, by itself, prove that executeCascade()
|
|
|
|
|
+// calls writeCascadeWal() before commitCascadeLmdb() - that ordering is
|
|
|
|
|
+// pinned by executeCascade() being three plain sequential statements next
|
|
|
|
|
+// to this comment in relation_cascade.cpp, not by a runtime fault
|
|
|
|
|
+// injection. See the Task 12 report for why a real fault-injection test
|
|
|
|
|
+// was judged not worth the complexity here.
|
|
|
|
|
+void test_crash_between_wal_and_commit_recovers() {
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+ cfgManager.loadFromStore();
|
|
|
|
|
+
|
|
|
|
|
+ TmpEnv t("rel-cascade-crash");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions", {{"exec_wf", "workflowId"}});
|
|
|
|
|
+
|
|
|
|
|
+ TmpPersistence p("rel-cascade-crash-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ // v2.11.0 T12 review (C1) - seed through the REAL WAL via
|
|
|
|
|
+ // p.pm.logInsert(), not just store.put()/mstore.loadDocument() (neither
|
|
|
|
|
+ // of which writes to the WAL). Without this the WAL below would contain
|
|
|
|
|
+ // ONLY the cascade's DELETE entries, freshStore would start from
|
|
|
|
|
+ // nothing, and "!freshStore.get(...).has_value()" would be true whether
|
|
|
|
|
+ // or not writeCascadeWal() wrote anything at all - a vacuous assertion
|
|
|
|
|
+ // (caught in review; confirmed by actually reverting this fix and
|
|
|
|
|
+ // re-running, see the Task 12 report). Seeding via logInsert makes the
|
|
|
|
|
+ // WAL read INSERT-then-DELETE per document, so recovery only ends up
|
|
|
|
|
+ // with nothing there if the DELETE entries genuinely got applied.
|
|
|
|
|
+ auto putBoth = [&](const std::string& id, const nlohmann::json& data) {
|
|
|
|
|
+ Document d; d.id = id; d.collection = "executions"; d.set_data(data);
|
|
|
|
|
+ store.put("executions", id, d);
|
|
|
|
|
+ p.pm.logInsert("default:executions", d);
|
|
|
|
|
+ };
|
|
|
|
|
+ putBoth("e1", {{"workflowId", "wf-1"}});
|
|
|
|
|
+ putBoth("e2", {{"workflowId", "wf-1"}});
|
|
|
|
|
+
|
|
|
|
|
+ {
|
|
|
|
|
+ Document parent; parent.id = "wf-1"; parent.collection = "workflows";
|
|
|
|
|
+ parent.set_data({{"name", "example"}});
|
|
|
|
|
+ store.put("workflows", "wf-1", parent);
|
|
|
|
|
+ p.pm.logInsert("default:workflows", parent);
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ RelationInfo rel;
|
|
|
|
|
+ rel.name = "default:exec_wf";
|
|
|
|
|
+ rel.child = "default:executions";
|
|
|
|
|
+ rel.childField = "workflowId";
|
|
|
|
|
+ rel.parent = "default:workflows";
|
|
|
|
|
+ rel.onDelete = OnDelete::Cascade;
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "declared the cascade relation");
|
|
|
|
|
+
|
|
|
|
|
+ // Steps 1+3 only - simulating a crash immediately after flushWal()
|
|
|
|
|
+ // returns, before commitCascadeLmdb()/applyCascadeToMemory() ever run.
|
|
|
|
|
+ CascadePlan plan = planCascade(rm, store, cfgManager, "default:workflows", "wf-1");
|
|
|
|
|
+ check(plan.mutations.size() == 2, "planned both children");
|
|
|
|
|
+ writeCascadeWal(p.pm, "default:workflows", "wf-1", plan);
|
|
|
|
|
+ p.pm.stop(); // closes the WAL file, like a process exiting
|
|
|
|
|
+
|
|
|
|
|
+ // Proof the "crash" really happened: LMDB was never touched by this
|
|
|
|
|
+ // cascade attempt - the children and parent are still exactly as they
|
|
|
|
|
+ // were.
|
|
|
|
|
+ check(store.get("executions", "e1").has_value(),
|
|
|
|
|
+ "LMDB was never committed - e1 is still there (this is the point)");
|
|
|
|
|
+ check(store.get("executions", "e2").has_value(),
|
|
|
|
|
+ "LMDB was never committed - e2 is still there");
|
|
|
|
|
+ check(store.get("workflows", "wf-1").has_value(),
|
|
|
|
|
+ "LMDB was never committed - the parent is still there");
|
|
|
|
|
+
|
|
|
|
|
+ // The WAL now genuinely contains 3 INSERTs (e1, e2, parent) followed by
|
|
|
|
|
+ // 3 DELETEs (e1, e2, parent) from writeCascadeWal() - read directly,
|
|
|
|
|
+ // the same way test_cascade_deletes_scalar_children_wal_first does,
|
|
|
|
|
+ // rather than only inferred from the recovered store's absence.
|
|
|
|
|
+ auto rawEntries = replayRawWal(p.path);
|
|
|
|
|
+ check(walHasDelete(rawEntries, "default:executions", "e1"), "WAL has DELETE for e1");
|
|
|
|
|
+ check(walHasDelete(rawEntries, "default:executions", "e2"), "WAL has DELETE for e2");
|
|
|
|
|
+ check(walHasDelete(rawEntries, "default:workflows", "wf-1"), "WAL has DELETE for the parent");
|
|
|
|
|
+
|
|
|
|
|
+ // "Restart": fresh MemoryStore, fresh PersistenceManager over the SAME
|
|
|
|
|
+ // dataDir, recover().
|
|
|
|
|
+ MemoryStore freshStore(MemoryStore::Config{});
|
|
|
|
|
+ freshStore.start();
|
|
|
|
|
+ PersistenceManager::Config cfg2;
|
|
|
|
|
+ cfg2.dataDir = p.path;
|
|
|
|
|
+ PersistenceManager pm2(cfg2);
|
|
|
|
|
+ auto outcome = pm2.recover(freshStore);
|
|
|
|
|
+ check(outcome.kind != smartbotic::database::RecoveryOutcome::Kind::Failed,
|
|
|
|
|
+ "recovery did not fail");
|
|
|
|
|
+ check(outcome.walEntriesReplayed >= 6,
|
|
|
|
|
+ "replayed the 3 inserts AND the 3 deletes (parent + two children)");
|
|
|
|
|
+
|
|
|
|
|
+ // Load-bearing: with the C1 fix, this is only possible because the WAL
|
|
|
|
|
+ // held both the INSERT and the DELETE for each id, and recovery applied
|
|
|
|
|
+ // both in order. Without the DELETE entries (the bug this test exists
|
|
|
|
|
+ // to catch), these would all be PRESENT after recovery instead.
|
|
|
|
|
+ check(!freshStore.get("default:executions", "e1").has_value(),
|
|
|
|
|
+ "recovery converges: e1 is gone from the recovered MemoryStore");
|
|
|
|
|
+ check(!freshStore.get("default:executions", "e2").has_value(),
|
|
|
|
|
+ "recovery converges: e2 is gone from the recovered MemoryStore");
|
|
|
|
|
+ check(!freshStore.get("default:workflows", "wf-1").has_value(),
|
|
|
|
|
+ "recovery converges: the parent is gone from the recovered MemoryStore");
|
|
|
|
|
+
|
|
|
|
|
+ // Replaying an already-applied delete is a no-op: recover() a SECOND
|
|
|
|
|
+ // time, onto the SAME already-recovered store. Must not throw and must
|
|
|
|
|
+ // not change the outcome.
|
|
|
|
|
+ auto outcome2 = pm2.recover(freshStore);
|
|
|
|
|
+ check(outcome2.kind != smartbotic::database::RecoveryOutcome::Kind::Failed,
|
|
|
|
|
+ "second recovery pass did not fail either");
|
|
|
|
|
+ check(!freshStore.get("default:executions", "e1").has_value(),
|
|
|
|
|
+ "still gone after replaying the same DELETE a second time - idempotent");
|
|
|
|
|
+
|
|
|
|
|
+ freshStore.stop();
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// v2.11.0 T12 review (I2) - a cascade must refuse rather than silently
|
|
|
|
|
+// destroy a grandchild protected by its own `restrict` relation.
|
|
|
|
|
+// workflows --cascade--> executions --restrict--> logs: deleting the
|
|
|
|
|
+// workflow would, without this check, delete e1 out from under a log entry
|
|
|
|
|
+// that names it, with canDelete() never having evaluated e1 at all (it only
|
|
|
|
|
+// ever evaluates the ORIGINAL parent, wf-1). Would fail if regressed: a
|
|
|
|
|
+// version of planCascade() without the grandchild check would throw
|
|
|
|
|
+// nothing here and instead silently proceed to delete e1.
|
|
|
|
|
+void test_cascade_refuses_when_grandchild_is_restrict_protected() {
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+ cfgManager.loadFromStore();
|
|
|
|
|
+
|
|
|
|
|
+ TmpEnv t("rel-cascade-grandchild");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions", {{"exec_wf", "workflowId"}});
|
|
|
|
|
+ store.set_relations("logs", {{"log_exec", "executionId"}});
|
|
|
|
|
+
|
|
|
|
|
+ TmpPersistence p("rel-cascade-grandchild-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ {
|
|
|
|
|
+ Document parent; parent.id = "wf-1"; parent.collection = "workflows";
|
|
|
|
|
+ parent.set_data({{"name", "example"}});
|
|
|
|
|
+ store.put("workflows", "wf-1", parent);
|
|
|
|
|
+ }
|
|
|
|
|
+ {
|
|
|
|
|
+ Document exec; exec.id = "e1"; exec.collection = "executions";
|
|
|
|
|
+ exec.set_data({{"workflowId", "wf-1"}});
|
|
|
|
|
+ store.put("executions", "e1", exec);
|
|
|
|
|
+ }
|
|
|
|
|
+ {
|
|
|
|
|
+ Document log; log.id = "log1"; log.collection = "logs";
|
|
|
|
|
+ log.set_data({{"executionId", "e1"}});
|
|
|
|
|
+ store.put("logs", "log1", log);
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ RelationInfo cascadeRel;
|
|
|
|
|
+ cascadeRel.name = "default:exec_wf";
|
|
|
|
|
+ cascadeRel.child = "default:executions";
|
|
|
|
|
+ cascadeRel.childField = "workflowId";
|
|
|
|
|
+ cascadeRel.parent = "default:workflows";
|
|
|
|
|
+ cascadeRel.onDelete = OnDelete::Cascade;
|
|
|
|
|
+ check(rm.createRelation(cascadeRel, mgrErr), "declared the cascade relation");
|
|
|
|
|
+
|
|
|
|
|
+ RelationInfo restrictRel;
|
|
|
|
|
+ restrictRel.name = "default:log_exec";
|
|
|
|
|
+ restrictRel.child = "default:logs";
|
|
|
|
|
+ restrictRel.childField = "executionId";
|
|
|
|
|
+ restrictRel.parent = "default:executions";
|
|
|
|
|
+ restrictRel.onDelete = OnDelete::Restrict;
|
|
|
|
|
+ check(rm.createRelation(restrictRel, mgrErr), "declared the grandchild restrict relation");
|
|
|
|
|
+
|
|
|
|
|
+ bool threw = false;
|
|
|
|
|
+ std::string thrownMessage;
|
|
|
|
|
+ try {
|
|
|
|
|
+ executeCascade(rm, store, p.pm, mstore, cfgManager, "default:workflows", "wf-1");
|
|
|
|
|
+ } catch (const CascadeBlocked& e) {
|
|
|
|
|
+ threw = true;
|
|
|
|
|
+ thrownMessage = e.what();
|
|
|
|
|
+ }
|
|
|
|
|
+ check(threw, "cascade refuses rather than silently deleting the restrict-protected grandchild");
|
|
|
|
|
+ check(thrownMessage.find("e1") != std::string::npos, "names the blocked child");
|
|
|
|
|
+ check(thrownMessage.find("log_exec") != std::string::npos || thrownMessage.find("logs") != std::string::npos,
|
|
|
|
|
+ "names the blocking relation or collection");
|
|
|
|
|
+
|
|
|
|
|
+ // Nothing was mutated anywhere - the throw happens inside planCascade(),
|
|
|
|
|
+ // before writeCascadeWal() ever runs.
|
|
|
|
|
+ check(store.get("workflows", "wf-1").has_value(), "parent untouched");
|
|
|
|
|
+ check(store.get("executions", "e1").has_value(), "the protected child untouched");
|
|
|
|
|
+ check(store.get("logs", "log1").has_value(), "the grandchild untouched");
|
|
|
|
|
+
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+ auto entries = replayRawWal(p.path);
|
|
|
|
|
+ check(entries.empty(), "nothing was ever written to the WAL - the refusal happened before step 3");
|
|
|
|
|
+
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// v2.11.0 T12 review round 3 (I1) - a crash between writeCascadeWal()'s
|
|
|
|
|
+// fsync and commitCascadeLmdb() leaves an UpdateChild mutation's WAL UPDATE
|
|
|
|
|
+// entry durable while its LMDB write never ran. Unlike DeleteChild (which
|
|
|
|
|
+// self-heals on replay because DELETE replay goes through
|
|
|
|
|
+// MemoryStore::remove(), which mirrors), UPDATE/UPSERT replay goes through
|
|
|
|
|
+// loadDocumentWithHistory(), which never mirrored - so before this fix,
|
|
|
|
|
+// LMDB would permanently keep serving the pre-cascade array/field. This
|
|
|
|
|
+// test drives the REAL cascade path (planCascade + writeCascadeWal against
|
|
|
|
|
+// an array reference, so the mutation really is an UpdateChild - not the
|
|
|
|
|
+// isolated primitive) through exactly that crash window, "restarts" with
|
|
|
|
|
+// the LMDB mirror wired the same way DatabaseService wires it in production
|
|
|
|
|
+// (setupComponents() wires the mirror BEFORE persistence_->recover() runs -
|
|
|
|
|
+// confirmed by reading database_service.cpp's initialize()), and asserts
|
|
|
|
|
+// LMDB converges, not just MemoryStore.
|
|
|
|
|
+void test_replayed_cascade_update_remirrors_to_lmdb_after_crash_window() {
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+ cfgManager.loadFromStore();
|
|
|
|
|
+
|
|
|
|
|
+ TmpEnv t("rel-remirror");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("nodes", {{"node_creds", "config.credentialIds"}});
|
|
|
|
|
+
|
|
|
|
|
+ TmpPersistence p("rel-remirror-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ // Seed through the REAL WAL (logInsert), same discipline as the C1 fix -
|
|
|
|
|
+ // these assertions must depend on genuine WAL replay, not just on
|
|
|
|
|
+ // store.put()/whatever MemoryStore starts with.
|
|
|
|
|
+ Document node; node.id = "n1"; node.collection = "nodes";
|
|
|
|
|
+ node.set_data({{"config", {{"credentialIds", {"c1", "c2", "c3"}}}}});
|
|
|
|
|
+ store.put("nodes", "n1", node);
|
|
|
|
|
+ p.pm.logInsert("default:nodes", node);
|
|
|
|
|
+
|
|
|
|
|
+ Document parent; parent.id = "c1"; parent.collection = "credentials";
|
|
|
|
|
+ parent.set_data({{"name", "prod-key"}});
|
|
|
|
|
+ store.put("credentials", "c1", parent);
|
|
|
|
|
+ p.pm.logInsert("default:credentials", parent);
|
|
|
|
|
+
|
|
|
|
|
+ RelationInfo rel;
|
|
|
|
|
+ rel.name = "default:node_creds";
|
|
|
|
|
+ rel.child = "default:nodes";
|
|
|
|
|
+ rel.childField = "config.credentialIds";
|
|
|
|
|
+ rel.parent = "default:credentials";
|
|
|
|
|
+ rel.onDelete = OnDelete::Cascade; // array reference -> UpdateChild (pull), not DeleteChild
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "declared the cascade relation on an array field");
|
|
|
|
|
+
|
|
|
|
|
+ // Steps 1+3 only - simulating a crash immediately after flushWal()
|
|
|
|
|
+ // returns, before commitCascadeLmdb()/applyCascadeToMemory() ever run.
|
|
|
|
|
+ CascadePlan plan = planCascade(rm, store, cfgManager, "default:credentials", "c1");
|
|
|
|
|
+ check(plan.mutations.size() == 1, "planned the one array-pull mutation");
|
|
|
|
|
+ check(plan.mutations[0].kind == CascadeMutation::Kind::UpdateChild,
|
|
|
|
|
+ "confirmed UpdateChild - this is exactly the case DeleteChild does NOT cover");
|
|
|
|
|
+ writeCascadeWal(p.pm, "default:credentials", "c1", plan);
|
|
|
|
|
+ p.pm.stop(); // closes the WAL file, like a process exiting
|
|
|
|
|
+
|
|
|
|
|
+ // Proof the "crash" really happened: LMDB was never touched by this
|
|
|
|
|
+ // cascade attempt - the node still has all three ids.
|
|
|
|
|
+ auto beforeRecovery = store.get("nodes", "n1");
|
|
|
|
|
+ check(beforeRecovery.has_value(), "n1 still in LMDB");
|
|
|
|
|
+ if (beforeRecovery) {
|
|
|
|
|
+ auto ids = beforeRecovery->data()["config"]["credentialIds"];
|
|
|
|
|
+ check(ids.is_array() && ids.size() == 3,
|
|
|
|
|
+ "LMDB was never committed - still the PRE-cascade array (this is the point)");
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // "Restart": fresh MemoryStore with the LMDB mirror wired to the SAME
|
|
|
|
|
+ // LmdbDocumentStore (matching production ordering - the mirror is wired
|
|
|
|
|
+ // in DatabaseService::setupComponents(), which runs BEFORE
|
|
|
|
|
+ // persistence_->recover() in initialize()), fresh PersistenceManager
|
|
|
|
|
+ // over the same dataDir, recover().
|
|
|
|
|
+ MemoryStore freshStore(MemoryStore::Config{});
|
|
|
|
|
+ freshStore.start();
|
|
|
|
|
+ std::atomic<bool> mirrorHealthy{true};
|
|
|
|
|
+ std::atomic<uint64_t> mirrorDrift{0};
|
|
|
|
|
+ freshStore.setDocumentStoreMirror(
|
|
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
|
|
+ &mirrorHealthy, &mirrorDrift);
|
|
|
|
|
+
|
|
|
|
|
+ PersistenceManager::Config cfg2;
|
|
|
|
|
+ cfg2.dataDir = p.path;
|
|
|
|
|
+ PersistenceManager pm2(cfg2);
|
|
|
|
|
+ auto outcome = pm2.recover(freshStore);
|
|
|
|
|
+ // v2.11.0 final review (finding 4) — recover() only COLLECTS the list now;
|
|
|
|
|
+ // DatabaseService::initialize() runs the pass after
|
|
|
|
|
+ // applyRelationDeclarations()/applyIndexDeclarations(). Mirrored here, in
|
|
|
|
|
+ // that order, so the test drives the production sequence.
|
|
|
|
|
+ pm2.runPendingRemirror(freshStore, outcome);
|
|
|
|
|
+ check(outcome.kind != smartbotic::database::RecoveryOutcome::Kind::Failed,
|
|
|
|
|
+ "recovery did not fail");
|
|
|
|
|
+ check(outcome.updatesRemirroredAfterReplay >= 1,
|
|
|
|
|
+ "the re-mirror pass handled at least the node's replayed UPDATE");
|
|
|
|
|
+
|
|
|
|
|
+ // MemoryStore converges (this part already worked before this fix).
|
|
|
|
|
+ auto memNode = freshStore.get("default:nodes", "n1");
|
|
|
|
|
+ check(memNode.has_value(), "n1 present in the recovered MemoryStore");
|
|
|
|
|
+ if (memNode) {
|
|
|
|
|
+ auto ids = memNode->data()["config"]["credentialIds"];
|
|
|
|
|
+ check(ids.is_array() && ids.size() == 2, "MemoryStore has the post-cascade array");
|
|
|
|
|
+ }
|
|
|
|
|
+ check(!freshStore.get("default:credentials", "c1").has_value(),
|
|
|
|
|
+ "parent gone from the recovered MemoryStore (DELETE replay already worked pre-fix)");
|
|
|
|
|
+
|
|
|
|
|
+ // LOAD-BEARING: LMDB now ALSO reflects the post-cascade array - only
|
|
|
|
|
+ // true because recover()'s remirror pass ran. Without the I1 fix this
|
|
|
|
|
+ // would still show 3 ids, exactly like `beforeRecovery` above.
|
|
|
|
|
+ auto lmdbAfter = store.get("nodes", "n1");
|
|
|
|
|
+ check(lmdbAfter.has_value(), "n1 still in LMDB after recovery");
|
|
|
|
|
+ if (lmdbAfter) {
|
|
|
|
|
+ auto ids = lmdbAfter->data()["config"]["credentialIds"];
|
|
|
|
|
+ check(ids.is_array() && ids.size() == 2,
|
|
|
|
|
+ "LMDB converged to the post-cascade array - the I1 fix");
|
|
|
|
|
+ check(std::find(ids.begin(), ids.end(), nlohmann::json("c1")) == ids.end(),
|
|
|
|
|
+ "c1 is gone from LMDB too, not just MemoryStore");
|
|
|
|
|
+ }
|
|
|
|
|
+ // The parent delete self-healed on its own even before this fix
|
|
|
|
|
+ // (DELETE replay mirrors via MemoryStore::remove()) - confirmed here so
|
|
|
|
|
+ // the test pins the WHOLE combined scenario, not just the new half.
|
|
|
|
|
+ check(!store.get("credentials", "c1").has_value(),
|
|
|
|
|
+ "parent also gone from LMDB after recovery");
|
|
|
|
|
+
|
|
|
|
|
+ freshStore.stop();
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// v2.11.0 T12 round-4 — the post-replay re-mirror pass must not be able to
|
|
|
|
|
+// stop the service from starting. Two throws are genuinely reachable from it
|
|
|
|
|
+// (see MemoryStore::remirrorDocuments): std::invalid_argument from
|
|
|
|
|
+// parseProjectCollection() on a malformed/legacy collection key, and a
|
|
|
|
|
+// storage fault from the put itself. Unguarded, either escaped recover(),
|
|
|
|
|
+// which DatabaseService::initialize() turns into "refuse to start" - a
|
|
|
|
|
+// deterministic boot loop on data that booted fine before.
|
|
|
|
|
+//
|
|
|
|
|
+// This also pins the batching contract: a row that throws inside a chunk
|
|
|
|
|
+// aborts that chunk's transaction, and the OTHER rows of that chunk must
|
|
|
|
|
+// still converge via the row-by-row retry.
|
|
|
|
|
+void test_a_failing_row_does_not_stop_recovery() {
|
|
|
|
|
+ TmpEnv t("rel-remirror-fail");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("rel-remirror-fail-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ // Row 1: an ordinary, well-formed document. Must converge.
|
|
|
|
|
+ Document good; good.id = "w1"; good.collection = "widgets";
|
|
|
|
|
+ good.set_data({{"v", 1}});
|
|
|
|
|
+ p.pm.logInsert("default:widgets", good);
|
|
|
|
|
+ good.set_data({{"v", 2}});
|
|
|
|
|
+ p.pm.logUpdate("default:widgets", good);
|
|
|
|
|
+
|
|
|
|
|
+ // Row 2: same project, so it lands in the SAME chunk as row 1 - but its
|
|
|
|
|
+ // id is past LMDB's 511-byte key limit, so the put throws
|
|
|
|
|
+ // (MDB_BAD_VALSIZE) from inside that chunk's transaction. This is the
|
|
|
|
|
+ // honest way to induce a mid-chunk throw, same technique as
|
|
|
|
|
+ // test_aborted_write_does_not_poison_the_collection (v2.8.0).
|
|
|
|
|
+ Document oversize; oversize.id = std::string(600, 'k'); oversize.collection = "widgets";
|
|
|
|
|
+ oversize.set_data({{"v", 1}});
|
|
|
|
|
+ p.pm.logInsert("default:widgets", oversize);
|
|
|
|
|
+ p.pm.logUpdate("default:widgets", oversize);
|
|
|
|
|
+
|
|
|
|
|
+ // Row 3: a malformed collection key. WAL replay itself never parses
|
|
|
|
|
+ // collection keys, so this replays fine and only the re-mirror pass
|
|
|
|
|
+ // trips on it - previously with an uncaught std::invalid_argument.
|
|
|
|
|
+ Document weird; weird.id = "x1"; weird.collection = "legacy";
|
|
|
|
|
+ weird.set_data({{"v", 1}});
|
|
|
|
|
+ p.pm.logInsert("weird:legacy:key", weird);
|
|
|
|
|
+ p.pm.logUpdate("weird:legacy:key", weird);
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore freshStore(MemoryStore::Config{});
|
|
|
|
|
+ freshStore.start();
|
|
|
|
|
+ std::atomic<bool> mirrorHealthy{true};
|
|
|
|
|
+ std::atomic<uint64_t> mirrorDrift{0};
|
|
|
|
|
+ freshStore.setDocumentStoreMirror(
|
|
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
|
|
+ &mirrorHealthy, &mirrorDrift);
|
|
|
|
|
+
|
|
|
|
|
+ PersistenceManager::Config cfg2;
|
|
|
|
|
+ cfg2.dataDir = p.path;
|
|
|
|
|
+ PersistenceManager pm2(cfg2);
|
|
|
|
|
+ auto outcome = pm2.recover(freshStore);
|
|
|
|
|
+ pm2.runPendingRemirror(freshStore, outcome); // finding 4 — see above
|
|
|
|
|
+
|
|
|
|
|
+ // THE finding: recovery completes.
|
|
|
|
|
+ check(outcome.kind != smartbotic::database::RecoveryOutcome::Kind::Failed,
|
|
|
|
|
+ "recovery completed despite two un-mirrorable rows - no boot loop");
|
|
|
|
|
+ check(outcome.walEntriesReplayed == 6,
|
|
|
|
|
+ "all six WAL entries replayed - the failures did not truncate replay");
|
|
|
|
|
+
|
|
|
|
|
+ // The failures are counted and reported, not swallowed silently.
|
|
|
|
|
+ check(outcome.updatesRemirrorFailed == 2,
|
|
|
|
|
+ "both un-mirrorable rows were counted as failed");
|
|
|
|
|
+ check(mirrorDrift.load() == 2,
|
|
|
|
|
+ "each failed row bumped mirror drift (the operator-visible signal)");
|
|
|
|
|
+ check(mirrorHealthy.load(),
|
|
|
|
|
+ "mirror health NOT flipped - one legacy row must not send every read "
|
|
|
|
|
+ "in the process to MemoryStore for the rest of its life");
|
|
|
|
|
+
|
|
|
|
|
+ // The other row in the same chunk still converged.
|
|
|
|
|
+ check(outcome.updatesRemirroredAfterReplay == 1, "the good row was re-mirrored");
|
|
|
|
|
+ auto lmdbGood = store.get("widgets", "w1");
|
|
|
|
|
+ check(lmdbGood.has_value(), "the good row reached LMDB despite sharing a chunk with a bad one");
|
|
|
|
|
+ if (lmdbGood) {
|
|
|
|
|
+ check(lmdbGood->data()["v"] == 2, "and it carries the REPLAYED update, not the insert");
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // MemoryStore replay itself was unaffected for every row, including the
|
|
|
|
|
+ // ones LMDB could not take.
|
|
|
|
|
+ check(freshStore.get("default:widgets", "w1").has_value(), "w1 in MemoryStore");
|
|
|
|
|
+ check(freshStore.get("weird:legacy:key", "x1").has_value(),
|
|
|
|
|
+ "the malformed-key row still recovered into MemoryStore - it is only "
|
|
|
|
|
+ "LMDB that cannot hold it");
|
|
|
|
|
+
|
|
|
|
|
+ freshStore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// v2.11.0 T12 round-4 (point 3) — INSERT must be tracked by the re-mirror
|
|
|
|
|
+// pass too. loadDocument() (INSERT replay) does not mirror, so a
|
|
|
|
|
+// delete-then-reinsert of the same id inside one replay window used to leave
|
|
|
|
|
+// LMDB with the row DELETED: the DELETE mirrored via remove(), the reinsert
|
|
|
|
|
+// did not.
|
|
|
|
|
+void test_reinsert_after_delete_converges_in_lmdb() {
|
|
|
|
|
+ TmpEnv t("rel-remirror-reinsert");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("rel-remirror-reinsert-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ Document d; d.id = "r1"; d.collection = "widgets";
|
|
|
|
|
+ d.set_data({{"v", 1}});
|
|
|
|
|
+ store.put("widgets", "r1", d); // LMDB starts in the pre-window state
|
|
|
|
|
+ p.pm.logInsert("default:widgets", d);
|
|
|
|
|
+ p.pm.logDelete("default:widgets", "r1");
|
|
|
|
|
+ d.set_data({{"v", 99}}); // reinserted with new content
|
|
|
|
|
+ p.pm.logInsert("default:widgets", d);
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore freshStore(MemoryStore::Config{});
|
|
|
|
|
+ freshStore.start();
|
|
|
|
|
+ std::atomic<bool> mirrorHealthy{true};
|
|
|
|
|
+ std::atomic<uint64_t> mirrorDrift{0};
|
|
|
|
|
+ freshStore.setDocumentStoreMirror(
|
|
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
|
|
+ &mirrorHealthy, &mirrorDrift);
|
|
|
|
|
+
|
|
|
|
|
+ PersistenceManager::Config cfg2;
|
|
|
|
|
+ cfg2.dataDir = p.path;
|
|
|
|
|
+ PersistenceManager pm2(cfg2);
|
|
|
|
|
+ auto outcome = pm2.recover(freshStore);
|
|
|
|
|
+ pm2.runPendingRemirror(freshStore, outcome); // finding 4 — see above
|
|
|
|
|
+ check(outcome.kind != smartbotic::database::RecoveryOutcome::Kind::Failed, "recovery completed");
|
|
|
|
|
+ check(outcome.updatesRemirroredAfterReplay == 1,
|
|
|
|
|
+ "the reinserted id was re-mirrored (INSERT is tracked, not just UPDATE/UPSERT)");
|
|
|
|
|
+
|
|
|
|
|
+ auto mem = freshStore.get("default:widgets", "r1");
|
|
|
|
|
+ check(mem.has_value(), "reinserted row present in MemoryStore");
|
|
|
|
|
+
|
|
|
|
|
+ // LOAD-BEARING: without INSERT tracking the DELETE's own mirror wins and
|
|
|
|
|
+ // LMDB has nothing here at all.
|
|
|
|
|
+ auto lmdb = store.get("widgets", "r1");
|
|
|
|
|
+ check(lmdb.has_value(), "reinserted row present in LMDB - the DELETE's mirror did not win");
|
|
|
|
|
+ if (lmdb) check(lmdb->data()["v"] == 99, "and it is the REINSERTED content, not the original");
|
|
|
|
|
+
|
|
|
|
|
+ freshStore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// -------------------------------------------------------------------------
|
|
|
|
|
+// v2.11.0 T13 — validate_on_write.
|
|
|
|
|
+//
|
|
|
|
|
+// The mitigation for the restrict race documented in relation_cascade.hpp
|
|
|
|
|
+// and the plan's self-review: a `restrict` check on delete runs in its own
|
|
|
|
|
+// read, so a child insert racing that check can still commit after the
|
|
|
|
|
+// parent is gone (LMDB serialises the two transactions, but the loser just
|
|
|
|
|
+// commits second). validate_on_write closes it by re-checking the parent's
|
|
|
|
|
+// existence with an mdb_get on the parent's own sub-db INSIDE the child's
|
|
|
|
|
+// write transaction - see LmdbDocumentStore::maintainRelations. What makes
|
|
|
|
|
+// it a genuine fix rather than a narrower window: the check and the write
|
|
|
|
|
+// commit as one unit, so there is no interval between them for the parent
|
|
|
|
|
+// to vanish in.
|
|
|
|
|
+// -------------------------------------------------------------------------
|
|
|
|
|
+
|
|
|
|
|
+void test_validate_on_write_rejects_missing_parent() {
|
|
|
|
|
+ TmpEnv t("validate-missing");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions",
|
|
|
|
|
+ {{"exec_wf", "workflowId", "workflows", true}});
|
|
|
|
|
+
|
|
|
|
|
+ Document d; d.id = "e1"; d.collection = "executions";
|
|
|
|
|
+ d.set_data({{"workflowId", "wf-ghost"}});
|
|
|
|
|
+
|
|
|
|
|
+ bool threw = false;
|
|
|
|
|
+ std::string what;
|
|
|
|
|
+ try {
|
|
|
|
|
+ store.put("executions", "e1", d);
|
|
|
|
|
+ } catch (const MissingParentReference& e) {
|
|
|
|
|
+ threw = true;
|
|
|
|
|
+ what = e.what();
|
|
|
|
|
+ }
|
|
|
|
|
+ check(threw, "validateOnWrite=true rejects a reference to a nonexistent parent");
|
|
|
|
|
+ check(what.find("exec_wf") != std::string::npos, "error names the relation");
|
|
|
|
|
+ check(what.find("workflowId") != std::string::npos, "error names the child field");
|
|
|
|
|
+ check(what.find("wf-ghost") != std::string::npos, "error names the missing parent id");
|
|
|
|
|
+
|
|
|
|
|
+ // A rejected write must leave no trace - not the document, not the
|
|
|
|
|
+ // reverse index posting.
|
|
|
|
|
+ check(!store.get("executions", "e1").has_value(),
|
|
|
|
|
+ "rejected insert left no document behind");
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-ghost") == 0,
|
|
|
|
|
+ "rejected insert left no reverse-index posting behind");
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+void test_validate_on_write_accepts_existing_parent() {
|
|
|
|
|
+ TmpEnv t("validate-present");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions",
|
|
|
|
|
+ {{"exec_wf", "workflowId", "workflows", true}});
|
|
|
|
|
+
|
|
|
|
|
+ Document w; w.id = "wf-1"; w.collection = "workflows";
|
|
|
|
|
+ w.set_data({{"name", "real workflow"}});
|
|
|
|
|
+ store.put("workflows", "wf-1", w);
|
|
|
|
|
+
|
|
|
|
|
+ Document d; d.id = "e1"; d.collection = "executions";
|
|
|
|
|
+ d.set_data({{"workflowId", "wf-1"}});
|
|
|
|
|
+ bool threw = false;
|
|
|
|
|
+ try {
|
|
|
|
|
+ store.put("executions", "e1", d);
|
|
|
|
|
+ } catch (const MissingParentReference&) {
|
|
|
|
|
+ threw = true;
|
|
|
|
|
+ }
|
|
|
|
|
+ check(!threw, "validateOnWrite=true accepts a reference to an existing parent");
|
|
|
|
|
+ check(store.get("executions", "e1").has_value(), "the child document was actually written");
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 1,
|
|
|
|
|
+ "and the reverse index posting was written");
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// With validateOnWrite=false (the default), the same missing-parent insert
|
|
|
|
|
+// succeeds and check_relation_dangling reports it - the brief's Step 1 case,
|
|
|
|
|
+// both halves.
|
|
|
|
|
+void test_validate_on_write_false_allows_dangling_and_check_reports_it() {
|
|
|
|
|
+ TmpEnv t("validate-off");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions",
|
|
|
|
|
+ {{"exec_wf", "workflowId", "workflows", false}});
|
|
|
|
|
+
|
|
|
|
|
+ Document d; d.id = "e1"; d.collection = "executions";
|
|
|
|
|
+ d.set_data({{"workflowId", "wf-ghost"}});
|
|
|
|
|
+ bool threw = false;
|
|
|
|
|
+ try {
|
|
|
|
|
+ store.put("executions", "e1", d);
|
|
|
|
|
+ } catch (const MissingParentReference&) {
|
|
|
|
|
+ threw = true;
|
|
|
|
|
+ }
|
|
|
|
|
+ check(!threw, "validateOnWrite=false lets the insert through");
|
|
|
|
|
+ check(store.get("executions", "e1").has_value(), "the dangling child document was written");
|
|
|
|
|
+
|
|
|
|
|
+ auto result = store.check_relation_dangling("exec_wf", "workflows", 100);
|
|
|
|
|
+ check(result.total == 1, "relations check reports one dangling parent id");
|
|
|
|
|
+ check(result.entries.size() == 1 && result.entries[0].parentId == "wf-ghost",
|
|
|
|
|
+ "and it is the ghost id");
|
|
|
|
|
+ check(result.entries[0].childCount == 1 &&
|
|
|
|
|
+ !result.entries[0].sampleChildIds.empty() &&
|
|
|
|
|
+ result.entries[0].sampleChildIds[0] == "e1",
|
|
|
|
|
+ "naming the dangling child");
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// Absent and null references are not references at all (same rule as T3's
|
|
|
|
|
+// index maintenance) - they must never be rejected, even under
|
|
|
|
|
+// validateOnWrite=true.
|
|
|
|
|
+void test_validate_on_write_never_rejects_absent_or_null() {
|
|
|
|
|
+ TmpEnv t("validate-absent-null");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions",
|
|
|
|
|
+ {{"exec_wf", "workflowId", "workflows", true}});
|
|
|
|
|
+
|
|
|
|
|
+ Document d1; d1.id = "e1"; d1.collection = "executions";
|
|
|
|
|
+ d1.set_data({{"other", 1}}); // workflowId absent entirely
|
|
|
|
|
+ bool threw1 = false;
|
|
|
|
|
+ try {
|
|
|
|
|
+ store.put("executions", "e1", d1);
|
|
|
|
|
+ } catch (const MissingParentReference&) {
|
|
|
|
|
+ threw1 = true;
|
|
|
|
|
+ }
|
|
|
|
|
+ check(!threw1, "an absent reference field is never rejected");
|
|
|
|
|
+
|
|
|
|
|
+ Document d2; d2.id = "e2"; d2.collection = "executions";
|
|
|
|
|
+ d2.set_data({{"workflowId", nullptr}});
|
|
|
|
|
+ bool threw2 = false;
|
|
|
|
|
+ try {
|
|
|
|
|
+ store.put("executions", "e2", d2);
|
|
|
|
|
+ } catch (const MissingParentReference&) {
|
|
|
|
|
+ threw2 = true;
|
|
|
|
|
+ }
|
|
|
|
|
+ check(!threw2, "an explicit null reference is never rejected");
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// Array-valued reference decision: ANY missing parent id rejects the WHOLE
|
|
|
|
|
+// write, not just that element - fail closed, and no partial state (some
|
|
|
|
|
+// elements resolvable, others not) is left behind.
|
|
|
|
|
+void test_validate_on_write_array_any_missing_rejects_whole_write() {
|
|
|
|
|
+ TmpEnv t("validate-array");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("nodes",
|
|
|
|
|
+ {{"node_creds", "config.credentialIds", "credentials", true}});
|
|
|
|
|
+
|
|
|
|
|
+ Document c1; c1.id = "c1"; c1.collection = "credentials";
|
|
|
|
|
+ c1.set_data({{"name", "real cred"}});
|
|
|
|
|
+ store.put("credentials", "c1", c1);
|
|
|
|
|
+ // c2 is deliberately never created.
|
|
|
|
|
+
|
|
|
|
|
+ Document n; n.id = "n1"; n.collection = "nodes";
|
|
|
|
|
+ n.set_data({{"config", {{"credentialIds", {"c1", "c2"}}}}});
|
|
|
|
|
+ bool threw = false;
|
|
|
|
|
+ std::string what;
|
|
|
|
|
+ try {
|
|
|
|
|
+ store.put("nodes", "n1", n);
|
|
|
|
|
+ } catch (const MissingParentReference& e) {
|
|
|
|
|
+ threw = true;
|
|
|
|
|
+ what = e.what();
|
|
|
|
|
+ }
|
|
|
|
|
+ check(threw, "one missing element in an array reference rejects the whole write");
|
|
|
|
|
+ check(what.find("c2") != std::string::npos, "error names the missing element, not the valid one");
|
|
|
|
|
+ check(!store.get("nodes", "n1").has_value(),
|
|
|
|
|
+ "rejected write left no document behind");
|
|
|
|
|
+ check(store.relation_index_child_count("node_creds", "c1") == 0,
|
|
|
|
|
+ "rejected write left no posting for the VALID element either - "
|
|
|
|
|
+ "no partial index state from a rejected write");
|
|
|
|
|
+ check(store.relation_index_child_count("node_creds", "c2") == 0,
|
|
|
|
|
+ "and none for the missing one");
|
|
|
|
|
+
|
|
|
|
|
+ // With every element resolvable, the write goes through and every
|
|
|
|
|
+ // element gets its posting.
|
|
|
|
|
+ Document c2; c2.id = "c2"; c2.collection = "credentials";
|
|
|
|
|
+ c2.set_data({{"name", "the missing one, now created"}});
|
|
|
|
|
+ store.put("credentials", "c2", c2);
|
|
|
|
|
+ threw = false;
|
|
|
|
|
+ try {
|
|
|
|
|
+ store.put("nodes", "n1", n);
|
|
|
|
|
+ } catch (const MissingParentReference&) {
|
|
|
|
|
+ threw = true;
|
|
|
|
|
+ }
|
|
|
|
|
+ check(!threw, "once every element resolves, the write succeeds");
|
|
|
|
|
+ check(store.relation_index_child_count("node_creds", "c1") == 1, "posting for c1");
|
|
|
|
|
+ check(store.relation_index_child_count("node_creds", "c2") == 1, "posting for c2");
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// An update that does not touch the reference field is not re-validated,
|
|
|
|
|
+// even if the previously-written reference has since gone dangling (e.g. the
|
|
|
|
|
+// parent was removed out from under a no_action/validateOnWrite=false-era
|
|
|
|
|
+// row, or validateOnWrite was turned on after the fact). Only NEWLY
|
|
|
|
|
+// introduced references (`to_add`) are checked - see maintainRelations'
|
|
|
|
|
+// comment for why: an unchanged reference already existed (or was already
|
|
|
|
|
+// dangling) before this write, and this task closes the race for writes
|
|
|
|
|
+// that introduce a reference, not for the mere fact that time has passed
|
|
|
|
|
+// since one was previously accepted.
|
|
|
|
|
+void test_validate_on_write_unrelated_update_not_rechecked() {
|
|
|
|
|
+ TmpEnv t("validate-unrelated-update");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions",
|
|
|
|
|
+ {{"exec_wf", "workflowId", "workflows", true}});
|
|
|
|
|
+
|
|
|
|
|
+ Document w; w.id = "wf-1"; w.collection = "workflows";
|
|
|
|
|
+ w.set_data({{"name", "real"}});
|
|
|
|
|
+ store.put("workflows", "wf-1", w);
|
|
|
|
|
+
|
|
|
|
|
+ Document d1; d1.id = "e1"; d1.collection = "executions";
|
|
|
|
|
+ d1.set_data({{"workflowId", "wf-1"}, {"status", "running"}});
|
|
|
|
|
+ store.put("executions", "e1", d1); // accepted: parent exists
|
|
|
|
|
+
|
|
|
|
|
+ store.del("workflows", "wf-1"); // parent now gone; reference dangles
|
|
|
|
|
+
|
|
|
|
|
+ // Rewrite e1 touching only `status` - workflowId is unchanged, so this
|
|
|
|
|
+ // must NOT re-validate it and must NOT throw.
|
|
|
|
|
+ Document d2; d2.id = "e1"; d2.collection = "executions";
|
|
|
|
|
+ d2.set_data({{"workflowId", "wf-1"}, {"status", "completed"}});
|
|
|
|
|
+ bool threw = false;
|
|
|
|
|
+ try {
|
|
|
|
|
+ store.put("executions", "e1", d2);
|
|
|
|
|
+ } catch (const MissingParentReference&) {
|
|
|
|
|
+ threw = true;
|
|
|
|
|
+ }
|
|
|
|
|
+ check(!threw, "an update that leaves the reference field unchanged is not re-validated");
|
|
|
|
|
+ check(store.get("executions", "e1")->data()["status"] == "completed",
|
|
|
|
|
+ "the update itself still applied");
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// v2.11.0 T13 round 3 (review finding) — rollback of an ALREADY-APPLIED
|
|
|
|
|
+// index mutation within the same write, not merely "the mutation never
|
|
|
|
|
+// happened because validation ran first."
|
|
|
|
|
+//
|
|
|
|
|
+// round 2's version of this test made `_relidx_wf_rel` itself only exist
|
|
|
|
|
+// inside the SAME aborted transaction (nothing had committed a wf_rel
|
|
|
|
|
+// posting beforehand), so relation_index_child_count("wf_rel", "wf-1")
|
|
|
|
|
+// still returned 0 through the "sub-db was never created"
|
|
|
|
|
+// (`if (!dbi_opt) return 0;`) shortcut regardless of whether the abort
|
|
|
|
|
+// rolled anything back - the exact case the comment claimed was
|
|
|
|
|
+// unavailable. Fixed by committing a SIBLING child (e0) first, in its own,
|
|
|
|
|
+// separate, successful write: `_relidx_wf_rel` is a real, already-existing
|
|
|
|
|
+// sub-db holding one committed posting (from e0) BEFORE the rejected write
|
|
|
|
|
+// (e1) runs. `executions` declares TWO relations: wf_rel
|
|
|
|
|
+// (validateOnWrite=false) and owner_rel (validateOnWrite=true). e1's write
|
|
|
|
|
+// applies wf_rel's mdb_put for real - into that already-existing sub-db -
|
|
|
|
|
+// before owner_rel's check runs and rejects. The assertion that matters is
|
|
|
|
|
+// that the count STAYS AT 1 (e0's), not 2: reading 2 would mean e1's
|
|
|
|
|
+// already-applied wf_rel mutation survived the abort. Because the sub-db
|
|
|
|
|
+// demonstrably existed beforehand (asserted directly, see the sanity check
|
|
|
|
|
+// below), the nullopt shortcut is not in play here, and the assertion
|
|
|
|
|
+// actually discriminates a rollback from a no-op.
|
|
|
|
|
+void test_validate_on_write_rolls_back_an_already_applied_sibling_relation() {
|
|
|
|
|
+ TmpEnv t("validate-rollback-sibling");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions", {
|
|
|
|
|
+ {"wf_rel", "workflowId", "workflows", false}, // validateOnWrite=false
|
|
|
|
|
+ {"owner_rel", "ownerId", "users", true}, // validateOnWrite=true
|
|
|
|
|
+ });
|
|
|
|
|
+
|
|
|
|
|
+ Document w; w.id = "wf-1"; w.collection = "workflows";
|
|
|
|
|
+ w.set_data({{"name", "real workflow"}});
|
|
|
|
|
+ store.put("workflows", "wf-1", w);
|
|
|
|
|
+
|
|
|
|
|
+ Document uReal; uReal.id = "u-real"; uReal.collection = "users";
|
|
|
|
|
+ uReal.set_data({{"name", "a real user"}});
|
|
|
|
|
+ store.put("users", "u-real", uReal);
|
|
|
|
|
+ // Deliberately no "users/u-ghost" - owner_rel's parent for e1 never exists.
|
|
|
|
|
+
|
|
|
|
|
+ // Commit a sibling child FIRST, in its own successful write, so
|
|
|
|
|
+ // `_relidx_wf_rel` is a real, already-existing sub-db with one committed
|
|
|
|
|
+ // posting before the rejected write below ever runs.
|
|
|
|
|
+ Document e0; e0.id = "e0"; e0.collection = "executions";
|
|
|
|
|
+ e0.set_data({{"workflowId", "wf-1"}, {"ownerId", "u-real"}});
|
|
|
|
|
+ store.put("executions", "e0", e0);
|
|
|
|
|
+ check(store.relation_index_child_count("wf_rel", "wf-1") == 1,
|
|
|
|
|
+ "sanity: wf_rel's sub-db already exists and holds e0's committed posting");
|
|
|
|
|
+
|
|
|
|
|
+ Document e; e.id = "e1"; e.collection = "executions";
|
|
|
|
|
+ e.set_data({{"workflowId", "wf-1"}, {"ownerId", "u-ghost"}});
|
|
|
|
|
+ bool threw = false;
|
|
|
|
|
+ try {
|
|
|
|
|
+ store.put("executions", "e1", e);
|
|
|
|
|
+ } catch (const MissingParentReference&) {
|
|
|
|
|
+ threw = true;
|
|
|
|
|
+ }
|
|
|
|
|
+ check(threw, "owner_rel's missing parent rejects the write");
|
|
|
|
|
+ check(!store.get("executions", "e1").has_value(),
|
|
|
|
|
+ "the document itself was rolled back");
|
|
|
|
|
+ check(store.relation_index_child_count("owner_rel", "u-ghost") == 0,
|
|
|
|
|
+ "owner_rel (the relation that rejected) has no posting for the ghost id");
|
|
|
|
|
+ check(store.relation_index_child_count("wf_rel", "wf-1") == 1,
|
|
|
|
|
+ "wf_rel's count STAYS AT 1 (e0's) rather than becoming 2 - e1's "
|
|
|
|
|
+ "already-applied mdb_put into this REAL, already-existing sub-db "
|
|
|
|
|
+ "was rolled back by the same transaction abort that rejected "
|
|
|
|
|
+ "owner_rel. A count of 2 here would mean the sibling relation's "
|
|
|
|
|
+ "mutation survived the abort.");
|
|
|
|
|
+
|
|
|
|
|
+ // Once owner_rel's parent exists, the identical write succeeds and BOTH
|
|
|
|
|
+ // relations end up with their postings - e0's plus e1's.
|
|
|
|
|
+ Document u; u.id = "u-ghost"; u.collection = "users";
|
|
|
|
|
+ u.set_data({{"name", "real user, now created"}});
|
|
|
|
|
+ store.put("users", "u-ghost", u);
|
|
|
|
|
+ threw = false;
|
|
|
|
|
+ try {
|
|
|
|
|
+ store.put("executions", "e1", e);
|
|
|
|
|
+ } catch (const MissingParentReference&) {
|
|
|
|
|
+ threw = true;
|
|
|
|
|
+ }
|
|
|
|
|
+ check(!threw, "once owner_rel's parent exists too, the write succeeds");
|
|
|
|
|
+ check(store.relation_index_child_count("wf_rel", "wf-1") == 2,
|
|
|
|
|
+ "wf_rel now has BOTH e0's (pre-existing) and e1's (just landed) postings");
|
|
|
|
|
+ check(store.relation_index_child_count("owner_rel", "u-ghost") == 1, "owner_rel posted for e1");
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// v2.11.0 T13 round 2 (review finding 1) — relationsEnforced is the
|
|
|
|
|
+// documented escape hatch (config/collection_config_manager.hpp) for "a
|
|
|
|
|
+// bulk import, or a collection under write pressure." Before this, the
|
|
|
|
|
+// only consumer was RelationEnforcer::canDelete (restrict/no_action on
|
|
|
|
|
+// delete); validate_on_write did not read it at all, so the only way to
|
|
|
|
|
+// stop a validate_on_write rejection was to drop and re-declare the
|
|
|
|
|
+// relation without validateOnWrite - not what the switch is for. Disabling
|
|
|
|
|
+// enforcement must also let a bulk-import-shaped write through even though
|
|
|
|
|
+// its parent is not loaded yet.
|
|
|
|
|
+void test_validate_on_write_relations_enforced_false_is_the_escape_hatch() {
|
|
|
|
|
+ TmpEnv t("validate-enforced-off");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ // relationsEnforced=false alongside validateOnWrite=true - the exact
|
|
|
|
|
+ // combination an operator reaches for mid-bulk-import.
|
|
|
|
|
+ store.set_relations("executions",
|
|
|
|
|
+ {{"exec_wf", "workflowId", "workflows", true, false}});
|
|
|
|
|
+
|
|
|
|
|
+ Document d; d.id = "e1"; d.collection = "executions";
|
|
|
|
|
+ d.set_data({{"workflowId", "wf-ghost"}});
|
|
|
|
|
+ bool threw = false;
|
|
|
|
|
+ try {
|
|
|
|
|
+ store.put("executions", "e1", d);
|
|
|
|
|
+ } catch (const MissingParentReference&) {
|
|
|
|
|
+ threw = true;
|
|
|
|
|
+ }
|
|
|
|
|
+ check(!threw, "relationsEnforced=false lets a validateOnWrite=true write through");
|
|
|
|
|
+ check(store.get("executions", "e1").has_value(),
|
|
|
|
|
+ "the escape-hatch write actually landed");
|
|
|
|
|
+
|
|
|
|
|
+ // Re-enabling enforcement does not retroactively touch what was already
|
|
|
|
|
+ // written (documented behaviour, mirrors canDelete's own log message) -
|
|
|
|
|
+ // but a NEW write with a missing parent is rejected again.
|
|
|
|
|
+ store.set_relations("executions",
|
|
|
|
|
+ {{"exec_wf", "workflowId", "workflows", true, true}});
|
|
|
|
|
+ Document d2; d2.id = "e2"; d2.collection = "executions";
|
|
|
|
|
+ d2.set_data({{"workflowId", "wf-ghost-2"}});
|
|
|
|
|
+ threw = false;
|
|
|
|
|
+ try {
|
|
|
|
|
+ store.put("executions", "e2", d2);
|
|
|
|
|
+ } catch (const MissingParentReference&) {
|
|
|
|
|
+ threw = true;
|
|
|
|
|
+ }
|
|
|
|
|
+ check(threw, "re-enabling enforcement rejects a new write with a missing parent");
|
|
|
|
|
+ check(store.get("executions", "e1").has_value(),
|
|
|
|
|
+ "the earlier escape-hatch write was not retroactively undone");
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// v2.11.0 T13 — demonstrates validation answers against write-time state,
|
|
|
|
|
+// not a stale earlier observation.
|
|
|
|
|
+//
|
|
|
|
|
+// Models one honest slice of the interleaving the plan describes: something
|
|
|
|
|
+// (an application-level pre-check, or the old restrict path's own read)
|
|
|
|
|
+// observes the parent present, and only AFTER that does the parent get
|
|
|
|
|
+// deleted - in its own committed transaction - before the child's write
|
|
|
|
|
+// happens. A check that trusted the earlier observation would let the
|
|
|
|
|
+// child insert through anyway.
|
|
|
|
|
+//
|
|
|
|
|
+// What this test does NOT establish: it does NOT distinguish "the mdb_get
|
|
|
|
|
+// runs inside the child's own write transaction" (the actual fix - see
|
|
|
|
|
+// maintainRelations, before the child's own mdb_put, same transaction the
|
|
|
|
|
+// caller commits) from "the mdb_get runs in a separate read transaction
|
|
|
|
|
+// opened immediately before the child's write transaction" (a narrowed
|
|
|
|
|
+// window, not a closed one). This test's sequence - get, then a
|
|
|
|
|
+// committed del, then put - would reject identically either way, because
|
|
|
|
|
+// the del is fully committed before either kind of check would run. That
|
|
|
|
|
+// placement is inside the transaction, not merely adjacent to it, is
|
|
|
|
|
+// established by reading the code (document_store_lmdb.cpp: the mdb_get is
|
|
|
|
|
+// at maintainRelations, called from put() before put()'s own mdb_put,
|
|
|
|
|
+// against the same wtxn the caller commits), not by this test.
|
|
|
|
|
+//
|
|
|
|
|
+// What this test DOES show: the validation's answer tracks the parent's
|
|
|
|
|
+// state as of when the check actually runs, not whatever an earlier,
|
|
|
|
|
+// separate read happened to observe - which is the necessary condition for
|
|
|
|
|
+// the fix to work at all, even though it is not sufficient to prove
|
|
|
|
|
+// placement by itself. It is deliberately single-threaded and
|
|
|
|
|
+// deterministic, not a real multi-thread stress test: LMDB is single-writer
|
|
|
|
|
+// (see try_open_for_read's file comment and document_store_lmdb.cpp's env
|
|
|
|
|
+// setup), so under real concurrency the only thing that can vary is
|
|
|
|
|
+// transaction ORDER, never interleaving within a transaction - serialising
|
|
|
|
|
+// "the delete's transaction commits, then the child write's transaction
|
|
|
|
|
+// begins" in program order is the deterministic equivalent of that
|
|
|
|
|
+// ordering, which is as much of the race as a single-process test can
|
|
|
|
|
+// exercise.
|
|
|
|
|
+void test_validate_on_write_closes_the_stale_check_race() {
|
|
|
|
|
+ TmpEnv t("validate-race");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("executions",
|
|
|
|
|
+ {{"exec_wf", "workflowId", "workflows", true}});
|
|
|
|
|
+
|
|
|
|
|
+ Document w; w.id = "wf-1"; w.collection = "workflows";
|
|
|
|
|
+ w.set_data({{"name", "about to be deleted"}});
|
|
|
|
|
+ store.put("workflows", "wf-1", w);
|
|
|
|
|
+
|
|
|
|
|
+ // The "stale check": some caller observes the parent present. This is
|
|
|
|
|
+ // exactly what a restrict check (or an application's own pre-flight
|
|
|
|
|
+ // lookup) does - a READ, complete and finished, before the write it is
|
|
|
|
|
+ // meant to gate.
|
|
|
|
|
+ check(store.get("workflows", "wf-1").has_value(),
|
|
|
|
|
+ "pre-check observes the parent present");
|
|
|
|
|
+
|
|
|
|
|
+ // The parent vanishes AFTER that check returned, in its own committed
|
|
|
|
|
+ // transaction - the race window a stale check cannot see across.
|
|
|
|
|
+ check(store.del("workflows", "wf-1"), "parent deleted after the check ran");
|
|
|
|
|
+
|
|
|
|
|
+ // The child write's OWN transaction begins only now, strictly after the
|
|
|
|
|
+ // delete's commit (LMDB's single-writer serialisation). If validation
|
|
|
|
|
+ // used the stale check's answer (or any state cached before this point)
|
|
|
|
|
+ // it would wrongly accept. It must instead re-read the parent's current
|
|
|
|
|
+ // state inside its own transaction and reject.
|
|
|
|
|
+ Document d; d.id = "e1"; d.collection = "executions";
|
|
|
|
|
+ d.set_data({{"workflowId", "wf-1"}});
|
|
|
|
|
+ bool threw = false;
|
|
|
|
|
+ try {
|
|
|
|
|
+ store.put("executions", "e1", d);
|
|
|
|
|
+ } catch (const MissingParentReference&) {
|
|
|
|
|
+ threw = true;
|
|
|
|
|
+ }
|
|
|
|
|
+ check(threw, "the child write is rejected against write-time state, "
|
|
|
|
|
+ "despite an earlier check having observed the parent present");
|
|
|
|
|
+ check(!store.get("executions", "e1").has_value(),
|
|
|
|
|
+ "no trace of the write the stale check would have allowed");
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+} // namespace
|
|
|
|
|
+
|
|
|
|
|
+
|
|
|
|
|
+// =========================================================================
|
|
|
|
|
+// v2.11.0 final review — the post-replay re-mirror pass runs AFTER arming
|
|
|
|
|
+// (finding 4) and its window is bounded by bytes as well as count (finding 3).
|
|
|
|
|
+// =========================================================================
|
|
|
|
|
+
|
|
|
|
|
+// FINDING 4, the consequential half: the pass writes documents through
|
|
|
|
|
+// LmdbDocumentStore::put(), which maintains every declared secondary index
|
|
|
|
|
+// INSIDE the document's own write transaction by reading
|
|
|
|
|
+// indexed_fields(collection) live. While the pass ran inside recover() - line
|
|
|
|
|
+// ~88 of DatabaseService::initialize(), where applyIndexDeclarations() is line
|
|
|
|
|
+// ~208 - that map was still EMPTY, so the pass moved the document forward and
|
|
|
|
|
+// left its postings behind. An index-served query then returns the row under
|
|
|
|
|
+// its OLD value and misses it under its new one: silently wrong rows, which is
|
|
|
|
|
+// exactly what putting maintainIndexes() inside the transaction exists to
|
|
|
|
|
+// prevent, and which bites ANY install with a v2.9 index declared, relations
|
|
|
|
|
+// or not.
|
|
|
|
|
+//
|
|
|
|
|
+// The simulated restart clears the in-process declaration (a real restart
|
|
|
|
|
+// rebuilds it from `_collection_meta`) and re-arms it only AFTER recover(),
|
|
|
|
|
+// which is the production ordering. Moving the runPendingRemirror() call above
|
|
|
|
|
+// the re-arm reproduces the bug.
|
|
|
|
|
+void test_remirror_maintains_the_secondary_index() {
|
|
|
|
|
+ TmpEnv t("rel-remirror-index");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("rel-remirror-index-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ store.set_indexed_fields("widgets", {"status"});
|
|
|
|
|
+
|
|
|
|
|
+ Document d; d.id = "w1"; d.collection = "widgets";
|
|
|
|
|
+ d.set_data({{"status", "queued"}});
|
|
|
|
|
+ store.put("widgets", "w1", d); // LMDB + posting under "queued"
|
|
|
|
|
+ p.pm.logInsert("default:widgets", d);
|
|
|
|
|
+
|
|
|
|
|
+ {
|
|
|
|
|
+ auto pre = store.index_lookup_eq("widgets", "status", nlohmann::json("queued"));
|
|
|
|
|
+ check(pre.has_value() && pre->size() == 1, "the pre-crash posting exists under 'queued'");
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // The crash window: a WAL UPDATE whose LMDB write never ran.
|
|
|
|
|
+ d.set_data({{"status", "done"}});
|
|
|
|
|
+ p.pm.logUpdate("default:widgets", d);
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+
|
|
|
|
|
+ // "Restart". The declaration is in-process state, so clear it: a real
|
|
|
|
|
+ // restart starts with an empty map and rebuilds it in
|
|
|
|
|
+ // applyIndexDeclarations().
|
|
|
|
|
+ store.set_indexed_fields("widgets", {});
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore freshStore(MemoryStore::Config{});
|
|
|
|
|
+ freshStore.start();
|
|
|
|
|
+ std::atomic<bool> mirrorHealthy{true};
|
|
|
|
|
+ std::atomic<uint64_t> mirrorDrift{0};
|
|
|
|
|
+ freshStore.setDocumentStoreMirror(
|
|
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
|
|
+ &mirrorHealthy, &mirrorDrift);
|
|
|
|
|
+
|
|
|
|
|
+ PersistenceManager::Config cfg2;
|
|
|
|
|
+ cfg2.dataDir = p.path;
|
|
|
|
|
+ PersistenceManager pm2(cfg2);
|
|
|
|
|
+ auto outcome = pm2.recover(freshStore);
|
|
|
|
|
+ check(outcome.kind != smartbotic::database::RecoveryOutcome::Kind::Failed, "recovery completed");
|
|
|
|
|
+ check(outcome.pendingRemirror.size() == 1,
|
|
|
|
|
+ "recover() COLLECTED the row instead of re-mirroring it itself");
|
|
|
|
|
+
|
|
|
|
|
+ // applyIndexDeclarations()' place in initialize(): before the pass.
|
|
|
|
|
+ store.set_indexed_fields("widgets", {"status"});
|
|
|
|
|
+ pm2.runPendingRemirror(freshStore, outcome);
|
|
|
|
|
+ check(outcome.updatesRemirroredAfterReplay == 1, "the row was re-mirrored");
|
|
|
|
|
+
|
|
|
|
|
+ // The document converged (this half worked before the fix too).
|
|
|
|
|
+ auto lmdb = store.get("widgets", "w1");
|
|
|
|
|
+ check(lmdb.has_value() && lmdb->data()["status"] == "done",
|
|
|
|
|
+ "LMDB holds the replayed value");
|
|
|
|
|
+
|
|
|
|
|
+ // LOAD-BEARING: the INDEX converged with it. Both directions matter - a
|
|
|
|
|
+ // stale posting under the old value is a wrong row returned, and a missing
|
|
|
|
|
+ // posting under the new value is a row silently omitted.
|
|
|
|
|
+ auto stale = store.index_lookup_eq("widgets", "status", nlohmann::json("queued"));
|
|
|
|
|
+ check(stale.has_value() && stale->empty(),
|
|
|
|
|
+ "no posting left under the OLD value - an index-served query cannot "
|
|
|
|
|
+ "return this row as if it were still queued");
|
|
|
|
|
+ auto fresh = store.index_lookup_eq("widgets", "status", nlohmann::json("done"));
|
|
|
|
|
+ check(fresh.has_value() && fresh->size() == 1 && (*fresh)[0] == "w1",
|
|
|
|
|
+ "the row IS found under its new value - the posting was maintained by "
|
|
|
|
|
+ "the pass, which is only possible because the index was armed first");
|
|
|
|
|
+
|
|
|
|
|
+ freshStore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// FINDING 4, the other half: with the pass running after arming, a
|
|
|
|
|
+// UniqueViolation from it is REACHABLE, which is what makes phase 3's single
|
|
|
|
|
+// retry earn its place. It could not fire at all while the pass ran inside
|
|
|
|
|
+// recover() (unique_fields_ was empty), so the retry, its "moved unique value"
|
|
|
|
|
+// rationale and persistence_manager.cpp's insertion-ORDERED comment were all
|
|
|
|
|
+// dead code justified by dead reasoning.
|
|
|
|
|
+//
|
|
|
|
|
+// The scenario, in WAL order: a NEW row b1 takes the email a1 currently holds,
|
|
|
|
|
+// and only then does a1 move off it. Re-mirroring in WAL order therefore tries
|
|
|
|
|
+// b1 first, against a1's not-yet-updated posting - a genuine conflict - and it
|
|
|
|
|
+// clears only once a1 has been re-mirrored. Replacing the retry with a bare
|
|
|
|
|
+// noteFailure() makes this test fail.
|
|
|
|
|
+void test_moved_unique_value_is_resolved_by_the_retry_pass() {
|
|
|
|
|
+ TmpEnv t("rel-remirror-unique");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("rel-remirror-unique-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ store.set_indexed_fields("people", {"email"});
|
|
|
|
|
+ store.set_unique_fields("people", {"email"});
|
|
|
|
|
+
|
|
|
|
|
+ // Pre-crash LMDB state: a1 holds the value, with its posting. Written
|
|
|
|
|
+ // straight to LMDB and deliberately NOT to the WAL - it is the state a
|
|
|
|
|
+ // snapshot would have carried, and keeping it out of the WAL is what puts
|
|
|
|
|
+ // b1 first in the re-mirror order.
|
|
|
|
|
+ Document a; a.id = "a1"; a.collection = "people";
|
|
|
|
|
+ a.set_data({{"email", "shared@example.com"}});
|
|
|
|
|
+ store.put("people", "a1", a);
|
|
|
|
|
+
|
|
|
|
|
+ // WAL order: b1 claims the value FIRST, a1 vacates it second. That order is
|
|
|
|
|
+ // what creates the transient conflict, and it is the order recover()
|
|
|
|
|
+ // preserves (insertion-ORDERED, deliberately - see persistence_manager.cpp).
|
|
|
|
|
+ Document b; b.id = "b1"; b.collection = "people";
|
|
|
|
|
+ b.set_data({{"email", "shared@example.com"}});
|
|
|
|
|
+ p.pm.logInsert("default:people", b);
|
|
|
|
|
+
|
|
|
|
|
+ a.set_data({{"email", "moved@example.com"}});
|
|
|
|
|
+ p.pm.logInsert("default:people", a);
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+
|
|
|
|
|
+ store.set_unique_fields("people", {});
|
|
|
|
|
+ store.set_indexed_fields("people", {});
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore freshStore(MemoryStore::Config{});
|
|
|
|
|
+ freshStore.start();
|
|
|
|
|
+ std::atomic<bool> mirrorHealthy{true};
|
|
|
|
|
+ std::atomic<uint64_t> mirrorDrift{0};
|
|
|
|
|
+ freshStore.setDocumentStoreMirror(
|
|
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
|
|
+ &mirrorHealthy, &mirrorDrift);
|
|
|
|
|
+
|
|
|
|
|
+ PersistenceManager::Config cfg2;
|
|
|
|
|
+ cfg2.dataDir = p.path;
|
|
|
|
|
+ PersistenceManager pm2(cfg2);
|
|
|
|
|
+ auto outcome = pm2.recover(freshStore);
|
|
|
|
|
+ check(outcome.pendingRemirror.size() == 2, "both ids collected, in WAL order");
|
|
|
|
|
+ check(outcome.pendingRemirror[0].second == "b1",
|
|
|
|
|
+ "b1 comes FIRST - the order is what creates the transient conflict");
|
|
|
|
|
+
|
|
|
|
|
+ store.set_indexed_fields("people", {"email"});
|
|
|
|
|
+ store.set_unique_fields("people", {"email"});
|
|
|
|
|
+ // chunkSize=1 so each row gets its own transaction: with a batch of two the
|
|
|
|
|
+ // whole chunk would abort and be retried row by row anyway, but forcing the
|
|
|
|
|
+ // per-row shape makes the deferral the pass's own, not put_batch's.
|
|
|
|
|
+ auto res = freshStore.remirrorDocuments(outcome.pendingRemirror, /*chunkSize=*/1);
|
|
|
|
|
+
|
|
|
|
|
+ check(res.remirrored == 2,
|
|
|
|
|
+ "BOTH rows converged - b1 was deferred on the conflict and committed by "
|
|
|
|
|
+ "the retry once a1 had vacated the value");
|
|
|
|
|
+ check(res.failed == 0, "nothing was written off as failed");
|
|
|
|
|
+ check(mirrorDrift.load() == 0, "no drift bumped - a resolved conflict is not drift");
|
|
|
|
|
+
|
|
|
|
|
+ auto la = store.get("people", "a1");
|
|
|
|
|
+ auto lb = store.get("people", "b1");
|
|
|
|
|
+ check(la.has_value() && la->data()["email"] == "moved@example.com", "a1 moved off the value");
|
|
|
|
|
+ check(lb.has_value() && lb->data()["email"] == "shared@example.com", "b1 holds the value now");
|
|
|
|
|
+ auto who = store.index_lookup_eq("people", "email", nlohmann::json("shared@example.com"));
|
|
|
|
|
+ check(who.has_value() && who->size() == 1 && (*who)[0] == "b1",
|
|
|
|
|
+ "exactly one row holds the unique value, and it is the right one");
|
|
|
|
|
+
|
|
|
|
|
+ freshStore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// FINDING 3: the window is bounded by BYTES as well as by count, and a window
|
|
|
|
|
+// boundary is crossed at all (no prior test did - every fixture fitted inside
|
|
|
|
|
+// one chunk of 256, so the streaming structure round 5 introduced was
|
|
|
|
|
+// unpinned).
|
|
|
|
|
+//
|
|
|
|
|
+// chunkBytes=1 forces one row per window, which is the extreme the byte budget
|
|
|
|
|
+// can reach; every row must still converge, and the loop must terminate. A
|
|
|
|
|
+// document larger than the whole budget forms a window of one rather than
|
|
|
|
|
+// stalling, which is what guarantees progress.
|
|
|
|
|
+void test_remirror_windows_are_bounded_by_bytes_and_by_count() {
|
|
|
|
|
+ TmpEnv t("rel-remirror-window");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("rel-remirror-window-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ constexpr int kRows = 7;
|
|
|
|
|
+ for (int i = 0; i < kRows; ++i) {
|
|
|
|
|
+ Document d; d.id = "w" + std::to_string(i); d.collection = "widgets";
|
|
|
|
|
+ d.set_data({{"n", i}, {"pad", std::string(4096, 'x')}});
|
|
|
|
|
+ p.pm.logInsert("default:widgets", d);
|
|
|
|
|
+ }
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore freshStore(MemoryStore::Config{});
|
|
|
|
|
+ freshStore.start();
|
|
|
|
|
+ std::atomic<bool> mirrorHealthy{true};
|
|
|
|
|
+ std::atomic<uint64_t> mirrorDrift{0};
|
|
|
|
|
+ freshStore.setDocumentStoreMirror(
|
|
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
|
|
+ &mirrorHealthy, &mirrorDrift);
|
|
|
|
|
+
|
|
|
|
|
+ PersistenceManager::Config cfg2;
|
|
|
|
|
+ cfg2.dataDir = p.path;
|
|
|
|
|
+ PersistenceManager pm2(cfg2);
|
|
|
|
|
+ auto outcome = pm2.recover(freshStore);
|
|
|
|
|
+ check(outcome.pendingRemirror.size() == kRows, "all rows collected");
|
|
|
|
|
+
|
|
|
|
|
+ // Byte budget of 1 with a generous count cap: the BYTES must be what closes
|
|
|
|
|
+ // each window. Before this fix chunkBytes did not exist and the count cap
|
|
|
|
|
+ // alone put all seven in one window - on this repo's ~2.9 MB documents, 256
|
|
|
|
|
+ // of them is ~750 MB resident plus as much again in LMDB dirty pages, in
|
|
|
|
|
+ // the happy path, on the boot path.
|
|
|
|
|
+ //
|
|
|
|
|
+ // The OBSERVABLE is the number of committed LMDB transactions, read from
|
|
|
|
|
+ // the env's own last-txnid. Row counts alone cannot discriminate here (all
|
|
|
|
|
+ // seven rows converge either way, via put_batch instead of put), so
|
|
|
|
|
+ // counting commits is what actually pins that the window closed per row.
|
|
|
|
|
+ auto lastTxnId = [&]() -> uint64_t {
|
|
|
|
|
+ MDB_envinfo info;
|
|
|
|
|
+ mdb_env_info(t.env.raw(), &info);
|
|
|
|
|
+ return static_cast<uint64_t>(info.me_last_txnid);
|
|
|
|
|
+ };
|
|
|
|
|
+ const uint64_t txnBefore = lastTxnId();
|
|
|
|
|
+ auto res = freshStore.remirrorDocuments(outcome.pendingRemirror,
|
|
|
|
|
+ /*chunkSize=*/1024, /*chunkBytes=*/1);
|
|
|
|
|
+ const uint64_t commits = lastTxnId() - txnBefore;
|
|
|
|
|
+ check(res.remirrored == kRows,
|
|
|
|
|
+ "every row converged with a byte budget smaller than one document - "
|
|
|
|
|
+ "an oversized row forms a window of one rather than stalling");
|
|
|
|
|
+ check(res.failed == 0, "no failures");
|
|
|
|
|
+ check(commits >= kRows,
|
|
|
|
|
+ "the BYTE budget closed each window: one commit per row, not one "
|
|
|
|
|
+ "commit for all seven (which is what the count cap alone produced, "
|
|
|
|
|
+ "and what makes 256 rows of 2.9 MB a ~750 MB resident window)");
|
|
|
|
|
+ for (int i = 0; i < kRows; ++i) {
|
|
|
|
|
+ auto got = store.get("widgets", "w" + std::to_string(i));
|
|
|
|
|
+ check(got.has_value() && got->data()["n"] == i, "row survived its own window");
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // And the count cap still closes a window on its own: two rows, chunkSize=1.
|
|
|
|
|
+ LmdbDocumentStore store2(t.env);
|
|
|
|
|
+ freshStore.setDocumentStoreMirror(
|
|
|
|
|
+ [&store2](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store2; },
|
|
|
|
|
+ &mirrorHealthy, &mirrorDrift);
|
|
|
|
|
+ auto res2 = freshStore.remirrorDocuments(outcome.pendingRemirror,
|
|
|
|
|
+ /*chunkSize=*/1,
|
|
|
|
|
+ /*chunkBytes=*/64ull * 1024 * 1024);
|
|
|
|
|
+ check(res2.remirrored == kRows, "count-bounded windows cover every row too");
|
|
|
|
|
+ check(res2.failed == 0, "no failures crossing count-bounded window boundaries");
|
|
|
|
|
+
|
|
|
|
|
+ freshStore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// FINDING 6: the cascade path builds every updatedDoc from the LMDB copy and
|
|
|
|
|
+// pushes it into MemoryStore. While the mirror is unhealthy or drifted LMDB may
|
|
|
|
|
+// be BEHIND - that is the entire reason reads fall back to MemoryStore - so the
|
|
|
|
|
+// cascade would overwrite MemoryStore's fresher child body with a stale one.
|
|
|
|
|
+// Real data loss, on a state this codebase treats as routine.
|
|
|
|
|
+void test_cascade_refuses_while_the_mirror_is_unhealthy_or_drifted() {
|
|
|
|
|
+ TmpEnv t("rel-cascade-gate");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("rel-cascade-gate-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+
|
|
|
|
|
+ std::atomic<bool> healthy{true};
|
|
|
|
|
+ std::atomic<uint64_t> drift{0};
|
|
|
|
|
+ mstore.setDocumentStoreMirror(
|
|
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
|
|
+ &healthy, &drift);
|
|
|
|
|
+
|
|
|
|
|
+ RelationInfo rel;
|
|
|
|
|
+ rel.name = "default:exec_wf";
|
|
|
|
|
+ rel.child = "default:executions";
|
|
|
|
|
+ rel.childField = "workflowId";
|
|
|
|
|
+ rel.parent = "default:workflows";
|
|
|
|
|
+ rel.onDelete = OnDelete::Cascade;
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "declared the cascade relation");
|
|
|
|
|
+ store.set_relations("executions",
|
|
|
|
|
+ {RelationRef{"exec_wf", "workflowId", "workflows", false, true}});
|
|
|
|
|
+
|
|
|
|
|
+ Document parent; parent.id = "wf-1"; parent.collection = "workflows";
|
|
|
|
|
+ parent.set_data({{"name", "wf-1"}});
|
|
|
|
|
+ store.put("workflows", "wf-1", parent);
|
|
|
|
|
+ mstore.loadDocument("default:workflows", parent);
|
|
|
|
|
+
|
|
|
|
|
+ Document child; child.id = "ex-1"; child.collection = "executions";
|
|
|
|
|
+ child.set_data({{"workflowId", "wf-1"}});
|
|
|
|
|
+ store.put("executions", "ex-1", child);
|
|
|
|
|
+ mstore.loadDocument("default:executions", child);
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 1, "one child indexed");
|
|
|
|
|
+
|
|
|
|
|
+ auto expectRefusal = [&](const char* what) {
|
|
|
|
|
+ bool refused = false;
|
|
|
|
|
+ std::string msg;
|
|
|
|
|
+ try {
|
|
|
|
|
+ executeCascade(rm, store, p.pm, mstore, cfgManager, "default:workflows", "wf-1", nullptr);
|
|
|
|
|
+ } catch (const CascadeBlocked& e) {
|
|
|
|
|
+ refused = true;
|
|
|
|
|
+ msg = e.what();
|
|
|
|
|
+ } catch (const std::exception& e) {
|
|
|
|
|
+ msg = std::string("wrong exception type: ") + e.what();
|
|
|
|
|
+ }
|
|
|
|
|
+ check(refused, what);
|
|
|
|
|
+ check(msg.find("LMDB mirror is") != std::string::npos,
|
|
|
|
|
+ "the refusal tells the operator WHY, not just that it failed");
|
|
|
|
|
+ // Nothing may have been touched anywhere: the gate runs before
|
|
|
|
|
+ // planCascade(), so before writeCascadeWal() and before any LMDB commit.
|
|
|
|
|
+ check(store.get("workflows", "wf-1").has_value(), "parent untouched in LMDB");
|
|
|
|
|
+ check(store.get("executions", "ex-1").has_value(), "child untouched in LMDB");
|
|
|
|
|
+ check(mstore.get("default:workflows", "wf-1").has_value(), "parent untouched in MemoryStore");
|
|
|
|
|
+ check(mstore.get("default:executions", "ex-1").has_value(), "child untouched in MemoryStore");
|
|
|
|
|
+ };
|
|
|
|
|
+
|
|
|
|
|
+ healthy.store(false);
|
|
|
|
|
+ expectRefusal("cascade refused while the mirror is UNHEALTHY");
|
|
|
|
|
+
|
|
|
|
|
+ healthy.store(true);
|
|
|
|
|
+ drift.store(1);
|
|
|
|
|
+ expectRefusal("cascade refused while the mirror has DRIFTED (health alone is not enough - "
|
|
|
|
|
+ "the re-mirror pass bumps drift without flipping health, by design)");
|
|
|
|
|
+
|
|
|
|
|
+ // And with a clean mirror it proceeds, so the gate is not a blanket refusal.
|
|
|
|
|
+ drift.store(0);
|
|
|
|
|
+ bool parentExisted = false;
|
|
|
|
|
+ try {
|
|
|
|
|
+ parentExisted = executeCascade(rm, store, p.pm, mstore, cfgManager,
|
|
|
|
|
+ "default:workflows", "wf-1", nullptr);
|
|
|
|
|
+ } catch (const std::exception& e) {
|
|
|
|
|
+ check(false, "cascade threw with a healthy mirror");
|
|
|
|
|
+ std::cerr << " (" << e.what() << ")\n";
|
|
|
|
|
+ }
|
|
|
|
|
+ check(parentExisted, "the cascade ran once the mirror was healthy and undrifted");
|
|
|
|
|
+ check(!store.get("executions", "ex-1").has_value(), "child cascaded away");
|
|
|
|
|
+ check(!store.get("workflows", "wf-1").has_value(), "parent deleted");
|
|
|
|
|
+
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+
|
|
|
|
|
+// FINDING 5: the gate that keeps T11's per-write Document deep copy off the
|
|
|
|
|
+// write path of every install that declares no rejecting constraint.
|
|
|
|
|
+//
|
|
|
|
|
+// Measured (this repo's shapes, same machine): the copy alone is 1.8 us at
|
|
|
|
|
+// 4 KB, 7.1 us at 64 KB and 364 us at 2.9 MB, all of it inside the
|
|
|
|
|
+// collection's unique_lock and all of it a fresh yyjson_mut_doc allocation.
|
|
|
|
|
+// The gate's own cost in the common case is one relaxed atomic load.
|
|
|
|
|
+//
|
|
|
|
|
+// The truth table is what matters, and one row is easy to get wrong: a
|
|
|
|
|
+// relation WITHOUT validateOnWrite maintains its reverse index on every put
|
|
|
|
|
+// but can never THROW, so it must not force a snapshot.
|
|
|
|
|
+void test_can_reject_writes_gates_the_undo_snapshot() {
|
|
|
|
|
+ TmpEnv t("rel-gate-undo");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+
|
|
|
|
|
+ check(!store.can_reject_writes("widgets"),
|
|
|
|
|
+ "nothing declared anywhere: no snapshot needed (the case that used to "
|
|
|
|
|
+ "pay for the copy on every install)");
|
|
|
|
|
+
|
|
|
|
|
+ store.set_indexed_fields("widgets", {"status"});
|
|
|
|
|
+ check(!store.can_reject_writes("widgets"),
|
|
|
|
|
+ "a plain secondary index cannot reject a write, so still no snapshot");
|
|
|
|
|
+
|
|
|
|
|
+ store.set_relations("widgets",
|
|
|
|
|
+ {RelationRef{"w_rel", "parentId", "parents", /*validateOnWrite=*/false,
|
|
|
|
|
+ /*relationsEnforced=*/true}});
|
|
|
|
|
+ check(!store.can_reject_writes("widgets"),
|
|
|
|
|
+ "a relation WITHOUT validate_on_write maintains the reverse index but "
|
|
|
|
|
+ "never throws - no snapshot");
|
|
|
|
|
+
|
|
|
|
|
+ store.set_relations("widgets",
|
|
|
|
|
+ {RelationRef{"w_rel", "parentId", "parents", /*validateOnWrite=*/true,
|
|
|
|
|
+ /*relationsEnforced=*/true}});
|
|
|
|
|
+ check(store.can_reject_writes("widgets"),
|
|
|
|
|
+ "validate_on_write CAN reject (MissingParentReference) - snapshot needed");
|
|
|
|
|
+ check(!store.can_reject_writes("others"),
|
|
|
|
|
+ "and it is per collection, not process-wide: an unrelated collection "
|
|
|
|
|
+ "still pays nothing");
|
|
|
|
|
+
|
|
|
|
|
+ store.set_relations("widgets", {});
|
|
|
|
|
+ check(!store.can_reject_writes("widgets"), "dropping the relation drops the need");
|
|
|
|
|
+
|
|
|
|
|
+ store.set_unique_fields("widgets", {"email"});
|
|
|
|
|
+ check(store.can_reject_writes("widgets"),
|
|
|
|
|
+ "a unique field CAN reject (UniqueViolation) - snapshot needed");
|
|
|
|
|
+ check(!store.can_reject_writes("others"), "still per collection");
|
|
|
|
|
+ store.set_unique_fields("widgets", {});
|
|
|
|
|
+ check(!store.can_reject_writes("widgets"), "and dropping it drops the need again");
|
|
|
|
|
+
|
|
|
|
|
+ // MemoryStore's view of the same question, which is what the write paths
|
|
|
|
|
+ // actually call. Fails SAFE when no mirror is wired.
|
|
|
|
|
+ MemoryStore ms(MemoryStore::Config{});
|
|
|
|
|
+ ms.start();
|
|
|
|
|
+ check(ms.needsUndoSnapshot("default:widgets"),
|
|
|
|
|
+ "with NO mirror wired the answer is 'take the snapshot' - fail safe");
|
|
|
|
|
+ std::atomic<bool> healthy{true};
|
|
|
|
|
+ std::atomic<uint64_t> drift{0};
|
|
|
|
|
+ ms.setDocumentStoreMirror(
|
|
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
|
|
+ &healthy, &drift);
|
|
|
|
|
+ check(!ms.needsUndoSnapshot("default:widgets"),
|
|
|
|
|
+ "wired, with nothing declared: no snapshot");
|
|
|
|
|
+ store.set_unique_fields("widgets", {"email"});
|
|
|
|
|
+ check(ms.needsUndoSnapshot("default:widgets"),
|
|
|
|
|
+ "wired, with a unique field declared: snapshot");
|
|
|
|
|
+ check(ms.needsUndoSnapshot("this is not a valid:collection:name"),
|
|
|
|
|
+ "an unparseable name fails safe too");
|
|
|
|
|
+ ms.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+
|
|
|
|
|
+// =========================================================================
|
|
|
|
|
+// v2.11.0 close-out — TTL EXPIRY MUST HANDLE CHILDREN LIKE A MANUAL DELETE.
|
|
|
|
|
+//
|
|
|
|
|
+// Operator's ruling. Before this, MemoryStore::expireDocuments() erased the
|
|
|
|
|
+// document and mirrored a DELETE with no relation enforcement whatsoever, so a
|
|
|
|
|
+// restrict-protected parent carrying a TTL silently vanished and orphaned every
|
|
|
|
|
+// child - strictly the worst of the three available behaviours.
|
|
|
|
|
+//
|
|
|
|
|
+// Every test below drives the REAL sweeper (MemoryStore::expireDocuments()) with
|
|
|
|
|
+// the REAL decision function (relations/relation_cascade.cpp's
|
|
|
|
|
+// ttlExpiryDecision(), which DatabaseService binds as the production hook), so
|
|
|
|
|
+// what is under test is the shipped path and not a re-statement of it.
|
|
|
|
|
+// =========================================================================
|
|
|
|
|
+
|
|
|
|
|
+// Residency probe. MemoryStore::get() deliberately hides an EXPIRED document
|
|
|
|
|
+// (Document::isExpired()), so "is it still there" cannot be asked with get()
|
|
|
|
|
+// when the whole point is that the document is past its TTL and still present.
|
|
|
|
|
+// getAllDocuments() does not filter.
|
|
|
|
|
+bool residentInMemory(MemoryStore& mstore, const std::string& collection,
|
|
|
|
|
+ const std::string& id) {
|
|
|
|
|
+ for (const auto& d : mstore.getAllDocuments(collection)) {
|
|
|
|
|
+ if (d.id == id) return true;
|
|
|
|
|
+ }
|
|
|
|
|
+ return false;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// Install the production decision function as the sweeper's hook. Exactly what
|
|
|
|
|
+// DatabaseService::ttlExpiryRelationDecision() does, minus the per-project store
|
|
|
|
|
+// resolution (this fixture has one store) and the replication notifier.
|
|
|
|
|
+void installTtlHook(MemoryStore& mstore, RelationManager& rm, LmdbDocumentStore& store,
|
|
|
|
|
+ PersistenceManager& pm, CollectionConfigManager& cfgManager,
|
|
|
|
|
+ const smartbotic::database::CascadeNotifyFn& notify = nullptr) {
|
|
|
|
|
+ mstore.setTtlExpiryRelationHook(
|
|
|
|
|
+ [&rm, &store, &pm, &mstore, &cfgManager, notify](const std::string& coll,
|
|
|
|
|
+ const std::string& id,
|
|
|
|
|
+ bool firstAttempt) {
|
|
|
|
|
+ return ttlExpiryDecision(rm, store, pm, mstore, cfgManager, coll, id, notify,
|
|
|
|
|
+ firstAttempt);
|
|
|
|
|
+ });
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// A parent carrying a TTL, plus one child, plus a declared relation. Returns
|
|
|
|
|
+// nothing; the caller owns every object so the fixtures stay explicit (this
|
|
|
|
|
+// file's established style).
|
|
|
|
|
+void seedTtlParentAndChild(MemoryStore& mstore, LmdbDocumentStore& store,
|
|
|
|
|
+ RelationManager& rm, OnDelete policy,
|
|
|
|
|
+ const std::string& childField = "workflowId",
|
|
|
|
|
+ const nlohmann::json& childData = {{"workflowId", "wf-1"}}) {
|
|
|
|
|
+ RelationInfo rel;
|
|
|
|
|
+ rel.name = "default:exec_wf";
|
|
|
|
|
+ rel.child = "default:executions";
|
|
|
|
|
+ rel.childField = childField;
|
|
|
|
|
+ rel.parent = "default:workflows";
|
|
|
|
|
+ rel.onDelete = policy;
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "declared the relation under test");
|
|
|
|
|
+ store.set_relations("executions",
|
|
|
|
|
+ {RelationRef{"exec_wf", childField, "workflows", false, true}});
|
|
|
|
|
+
|
|
|
|
|
+ // ⚠ expiresAt = 1 (one millisecond after the epoch) rather than "now minus
|
|
|
|
|
+ // something": expireDocuments() compares against currentTimeMs(), so 1 is
|
|
|
|
|
+ // unambiguously expired on any clock and the test cannot race the sweep.
|
|
|
|
|
+ Document parent;
|
|
|
|
|
+ parent.id = "wf-1";
|
|
|
|
|
+ parent.collection = "workflows";
|
|
|
|
|
+ parent.set_data({{"name", "wf-1"}});
|
|
|
|
|
+ parent.expiresAt = 1;
|
|
|
|
|
+ store.put("workflows", "wf-1", parent);
|
|
|
|
|
+ mstore.loadDocument("default:workflows", parent);
|
|
|
|
|
+
|
|
|
|
|
+ Document child;
|
|
|
|
|
+ child.id = "ex-1";
|
|
|
|
|
+ child.collection = "executions";
|
|
|
|
|
+ child.set_data(childData);
|
|
|
|
|
+ store.put("executions", "ex-1", child);
|
|
|
|
|
+ mstore.loadDocument("default:executions", child);
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// restrict: a MANUAL delete fails, so the expiry must not happen. The document
|
|
|
|
|
+// is left in place, its expiry stays armed, and the block is signalled.
|
|
|
|
|
+//
|
|
|
|
|
+// Would fail before the fix on its first assertion: the pre-v2.11.0 sweeper
|
|
|
|
|
+// returned 1 and erased the parent, leaving ex-1 pointing at nothing.
|
|
|
|
|
+void test_ttl_restrict_blocks_the_expiry() {
|
|
|
|
|
+ TmpEnv t("ttl-restrict");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("ttl-restrict-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ // Small retry cadence so the test does not have to run 60 sweeps.
|
|
|
|
|
+ MemoryStore::Config cfg;
|
|
|
|
|
+ cfg.ttlBlockedRetrySweeps = 3;
|
|
|
|
|
+ MemoryStore mstore(cfg);
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+
|
|
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::Restrict);
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 1, "one child indexed");
|
|
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager);
|
|
|
|
|
+
|
|
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
|
|
+ check(expired == 0, "the restrict-protected parent was NOT expired");
|
|
|
|
|
+ check(residentInMemory(mstore, "default:workflows", "wf-1"),
|
|
|
|
|
+ "the parent is still in MemoryStore - it OUTLIVES its TTL, deliberately");
|
|
|
|
|
+ check(store.get("workflows", "wf-1").has_value(), "the parent is still in LMDB");
|
|
|
|
|
+ check(mstore.get("default:executions", "ex-1").has_value(), "the child was not orphaned");
|
|
|
|
|
+ check(mstore.getStats().ttlExpiryBlockedByRelation == 1,
|
|
|
|
|
+ "the skip is counted, so an operator can see a document is stuck past its TTL");
|
|
|
|
|
+ check(mstore.getStats().expiredCount == 0, "and it is not counted as expired");
|
|
|
|
|
+
|
|
|
|
|
+ check(mstore.ttlBlockedDocumentCount() == 1,
|
|
|
|
|
+ "and it is tracked as ONE stuck document - the gauge an operator wants");
|
|
|
|
|
+
|
|
|
|
|
+ // ⚠ The next sweep must NOT re-examine it (close-out review, finding 2). A
|
|
|
|
|
+ // blocked document costs one LMDB read txn + cursor scan + one log line every
|
|
|
|
|
+ // time it is examined, and at the 1s default that is a permanent flood plus a
|
|
|
|
|
+ // permanent index-lookup load. It is skipped for free until its retry is due.
|
|
|
|
|
+ const uint64_t again = mstore.expireDocuments();
|
|
|
|
|
+ check(again == 0, "still not expired on the next sweep");
|
|
|
|
|
+ check(mstore.getStats().ttlExpiryBlockedByRelation == 1,
|
|
|
|
|
+ "and the relation was NOT re-consulted - no second refusal event");
|
|
|
|
|
+ check(residentInMemory(mstore, "default:workflows", "wf-1"), "the parent is still there");
|
|
|
|
|
+
|
|
|
|
|
+ // The expiry must stay ARMED, so once the retry comes due AND the child is
|
|
|
|
|
+ // gone, it expires - the block is the relation's, not a permanent quarantine.
|
|
|
|
|
+ store.del("executions", "ex-1");
|
|
|
|
|
+ check(mstore.remove("default:executions", "ex-1"), "child removed");
|
|
|
|
|
+ uint64_t total = 0;
|
|
|
|
|
+ for (uint32_t i = 0; i < cfg.ttlBlockedRetrySweeps + 1; ++i) {
|
|
|
|
|
+ total += mstore.expireDocuments();
|
|
|
|
|
+ }
|
|
|
|
|
+ check(total == 1, "when the retry comes due, the parent finally expires");
|
|
|
|
|
+ check(!residentInMemory(mstore, "default:workflows", "wf-1"), "the parent is gone now");
|
|
|
|
|
+ check(mstore.ttlBlockedDocumentCount() == 0, "and it is no longer counted as stuck");
|
|
|
|
|
+
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// cascade with a SCALAR reference: children are deleted, exactly as
|
|
|
|
|
+// executeCascade does for a manual delete, and the WAL carries every mutation.
|
|
|
|
|
+//
|
|
|
|
|
+// Would fail before the fix: the old sweeper wrote no WAL entry for ex-1 at all
|
|
|
|
|
+// (it only mirrored a DELETE for the parent), so walHasDelete for the child is
|
|
|
|
|
+// the assertion that a hand-rolled, LMDB-only sweeper cascade cannot satisfy.
|
|
|
|
|
+void test_ttl_cascade_deletes_children_like_a_manual_delete() {
|
|
|
|
|
+ TmpEnv t("ttl-cascade");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("ttl-cascade-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+
|
|
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::Cascade);
|
|
|
|
|
+
|
|
|
|
|
+ // The notify wiring: a TTL-driven cascade must drive replication and
|
|
|
|
|
+ // Subscribe events per mutation, or a follower never sees the child
|
|
|
|
|
+ // deletions and diverges permanently.
|
|
|
|
|
+ struct Notification { std::string collection, id; bool hadDoc; smartbotic::database::EventType et; };
|
|
|
|
|
+ std::vector<Notification> notifications;
|
|
|
|
|
+ auto notify = [&](const std::string& coll, const std::string& id,
|
|
|
|
|
+ const std::optional<Document>& doc, smartbotic::database::EventType et) {
|
|
|
|
|
+ notifications.push_back({coll, id, doc.has_value(), et});
|
|
|
|
|
+ };
|
|
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager, notify);
|
|
|
|
|
+
|
|
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
|
|
+ check(expired == 1, "the parent was expired");
|
|
|
|
|
+ check(!mstore.get("default:executions", "ex-1").has_value(),
|
|
|
|
|
+ "the child was CASCADED, not orphaned");
|
|
|
|
|
+ check(!store.get("executions", "ex-1").has_value(), "and it is gone from LMDB too");
|
|
|
|
|
+ check(!store.get("workflows", "wf-1").has_value(), "the parent is gone from LMDB");
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 0,
|
|
|
|
|
+ "the reverse-index posting went with it");
|
|
|
|
|
+ check(notifications.size() == 2,
|
|
|
|
|
+ "replication/events fired once per child plus once for the parent");
|
|
|
|
|
+
|
|
|
|
|
+ // Accounting (close-out review, minor 1): the parent is ONE expiry, and the
|
|
|
|
|
+ // one child is ONE delete. The cascade removes the parent through
|
|
|
|
|
+ // unloadDocument(), which counts a DELETE, so without compensating for that
|
|
|
|
|
+ // the parent was counted as both an expiry and a delete.
|
|
|
|
|
+ const auto st = mstore.getStats();
|
|
|
|
|
+ check(st.expiredCount == 1, "the parent counted as exactly one expiry");
|
|
|
|
|
+ check(st.deleteCount == 1,
|
|
|
|
|
+ "and the delete count covers the CHILD only - the parent is not counted twice");
|
|
|
|
|
+
|
|
|
|
|
+ // The property a hand-rolled sweeper cascade breaks: MemoryStore is rebuilt
|
|
|
|
|
+ // from snapshot + WAL, NEVER from LMDB, so a cascade whose child deletions
|
|
|
|
|
+ // exist only in LMDB has them RESURRECTED on the next boot.
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+ auto entries = replayRawWal(p.path);
|
|
|
|
|
+ check(walHasDelete(entries, "default:executions", "ex-1"),
|
|
|
|
|
+ "the WAL carries the CHILD's delete - without this the next boot resurrects it");
|
|
|
|
|
+ check(walHasDelete(entries, "default:workflows", "wf-1"),
|
|
|
|
|
+ "the WAL carries the parent's delete");
|
|
|
|
|
+
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// set_null with a SCALAR reference: the field is nulled and the child KEPT.
|
|
|
|
|
+void test_ttl_set_null_nulls_the_scalar_and_keeps_the_child() {
|
|
|
|
|
+ TmpEnv t("ttl-setnull");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("ttl-setnull-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+
|
|
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::SetNull);
|
|
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager);
|
|
|
|
|
+
|
|
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
|
|
+ check(expired == 1, "the parent was expired");
|
|
|
|
|
+ auto child = store.get("executions", "ex-1");
|
|
|
|
|
+ check(child.has_value(), "the child SURVIVED - set_null keeps the document");
|
|
|
|
|
+ if (child) {
|
|
|
|
|
+ check(child->data().contains("workflowId") && child->data()["workflowId"].is_null(),
|
|
|
|
|
+ "and its reference is null, not deleted and not left dangling");
|
|
|
|
|
+ }
|
|
|
|
|
+ auto memChild = mstore.get("default:executions", "ex-1");
|
|
|
|
|
+ check(memChild.has_value(), "the child survived in MemoryStore too");
|
|
|
|
|
+ if (memChild) {
|
|
|
|
|
+ check(memChild->data()["workflowId"].is_null(), "with the same nulled reference");
|
|
|
|
|
+ }
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 0,
|
|
|
|
|
+ "the posting is gone, since the reference no longer names wf-1");
|
|
|
|
|
+
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+ auto entries = replayRawWal(p.path);
|
|
|
|
|
+ check(walHasUpdate(entries, "default:executions", "ex-1"),
|
|
|
|
|
+ "the child's UPDATE is in the WAL - the mutation survives a restart");
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// An ARRAY-valued reference: the id is PULLED and the document KEPT, under
|
|
|
|
|
+// cascade as well as set_null. The place a MySQL mental model actively
|
|
|
|
|
+// misleads, and the same rule the request path follows - deleting a node
|
|
|
|
|
+// because one of its three credentials expired would be worse than useless.
|
|
|
|
|
+void test_ttl_array_reference_pulls_the_id_and_keeps_the_document() {
|
|
|
|
|
+ TmpEnv t("ttl-array");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("ttl-array-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+
|
|
|
|
|
+ // cascade, deliberately: the array rule must hold under the policy a
|
|
|
|
|
+ // MySQL-trained reader expects to delete the row.
|
|
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::Cascade, "workflowIds",
|
|
|
|
|
+ {{"workflowIds", {"wf-1", "wf-2"}}});
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 1,
|
|
|
|
|
+ "the array element is indexed as a posting");
|
|
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager);
|
|
|
|
|
+
|
|
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
|
|
+ check(expired == 1, "the parent was expired");
|
|
|
|
|
+ auto child = store.get("executions", "ex-1");
|
|
|
|
|
+ check(child.has_value(), "the child SURVIVED - an array match never deletes the document");
|
|
|
|
|
+ if (child) {
|
|
|
|
|
+ const auto& ids = child->data()["workflowIds"];
|
|
|
|
|
+ check(ids.is_array() && ids.size() == 1, "exactly one id was pulled");
|
|
|
|
|
+ check(ids.is_array() && !ids.empty() && ids[0] == "wf-2",
|
|
|
|
|
+ "and it was the expiring parent's id, not the surviving one");
|
|
|
|
|
+ }
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 0, "posting for wf-1 gone");
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-2") == 1, "posting for wf-2 intact");
|
|
|
|
|
+
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// no_action: the expiry proceeds and the reference is left dangling, which is
|
|
|
|
|
+// what it did before this change and must keep doing.
|
|
|
|
|
+void test_ttl_no_action_expires_and_leaves_the_reference_dangling() {
|
|
|
|
|
+ TmpEnv t("ttl-noaction");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("ttl-noaction-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+
|
|
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::NoAction);
|
|
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager);
|
|
|
|
|
+
|
|
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
|
|
+ check(expired == 1, "no_action does not block the expiry");
|
|
|
|
|
+ check(!mstore.get("default:workflows", "wf-1").has_value(), "the parent expired");
|
|
|
|
|
+ auto child = mstore.get("default:executions", "ex-1");
|
|
|
|
|
+ check(child.has_value(), "the child is untouched");
|
|
|
|
|
+ if (child) {
|
|
|
|
|
+ check(child->data()["workflowId"] == "wf-1",
|
|
|
|
|
+ "and its reference still names the gone parent - dangling, as declared");
|
|
|
|
|
+ }
|
|
|
|
|
+ check(mstore.getStats().ttlExpiryBlockedByRelation == 0, "nothing was blocked");
|
|
|
|
|
+
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// SCOPE: the common case - no relation names this collection as a parent - must
|
|
|
|
|
+// behave exactly as it did before, including the LMDB DELETE mirror and the
|
|
|
|
|
+// EXPIRE event. This is the regression guard for the two-phase rewrite of
|
|
|
|
|
+// expireDocuments(), which is what actually changed for every install.
|
|
|
|
|
+void test_ttl_with_no_relations_expires_exactly_as_before() {
|
|
|
|
|
+ TmpEnv t("ttl-plain");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("ttl-plain-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+
|
|
|
|
|
+ std::atomic<bool> healthy{true};
|
|
|
|
|
+ std::atomic<uint64_t> drift{0};
|
|
|
|
|
+ mstore.setDocumentStoreMirror(
|
|
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
|
|
+ &healthy, &drift);
|
|
|
|
|
+
|
|
|
|
|
+ int expireEvents = 0;
|
|
|
|
|
+ mstore.setEventCallback([&](const smartbotic::database::DatabaseEvent& ev) {
|
|
|
|
|
+ if (ev.type == smartbotic::database::EventType::EXPIRE) ++expireEvents;
|
|
|
|
|
+ });
|
|
|
|
|
+
|
|
|
|
|
+ Document doc;
|
|
|
|
|
+ doc.id = "s-1";
|
|
|
|
|
+ doc.collection = "sessions";
|
|
|
|
|
+ doc.set_data({{"token", "abc"}});
|
|
|
|
|
+ doc.expiresAt = 1;
|
|
|
|
|
+ store.put("sessions", "s-1", doc);
|
|
|
|
|
+ mstore.loadDocument("default:sessions", doc);
|
|
|
|
|
+
|
|
|
|
|
+ // Hook installed, with the relation cache EMPTY - the early-return path.
|
|
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager);
|
|
|
|
|
+
|
|
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
|
|
+ check(expired == 1, "an unrelated document still expires");
|
|
|
|
|
+ check(!mstore.get("default:sessions", "s-1").has_value(), "gone from MemoryStore");
|
|
|
|
|
+ check(!store.get("sessions", "s-1").has_value(),
|
|
|
|
|
+ "and the DELETE was mirrored to LMDB, as before");
|
|
|
|
|
+ check(expireEvents == 1, "exactly one EXPIRE event, as before");
|
|
|
|
|
+ check(mstore.getStats().expiredCount == 1, "counted as expired");
|
|
|
|
|
+ check(mstore.getStats().ttlExpiryBlockedByRelation == 0, "nothing blocked");
|
|
|
|
|
+
|
|
|
|
|
+ // A document whose TTL is in the future is not touched by any of this.
|
|
|
|
|
+ Document later;
|
|
|
|
|
+ later.id = "s-2";
|
|
|
|
|
+ later.collection = "sessions";
|
|
|
|
|
+ later.set_data({{"token", "def"}});
|
|
|
|
|
+ later.expiresAt = 1;
|
|
|
|
|
+ later.expiresAt = static_cast<uint64_t>(1) << 62; // far future
|
|
|
|
|
+ mstore.loadDocument("default:sessions", later);
|
|
|
|
|
+ check(mstore.expireDocuments() == 0, "an unexpired document is left alone");
|
|
|
|
|
+ check(mstore.get("default:sessions", "s-2").has_value(), "and is still there");
|
|
|
|
|
+
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// THE ORDERING PROPERTY, observed rather than asserted from the outside: the
|
|
|
|
|
+// cascade's WAL entries are durable BEFORE the LMDB transaction commits.
|
|
|
|
|
+//
|
|
|
|
|
+// Induced honestly, with the v2.4.4 identity sentinel: the child sub-db's
|
|
|
|
|
+// sentinel is overwritten with the wrong name using raw LMDB, so
|
|
|
|
|
+// commitCascadeLmdb()'s open_for_write() throws and its WriteTxn aborts
|
|
|
|
|
+// UNWRITTEN. Everything before it has already happened. If the sweeper wrote
|
|
|
|
|
+// LMDB first (or hand-rolled its own cascade), the WAL would be empty here and
|
|
|
|
|
+// the child's deletion would be lost on the next boot.
|
|
|
|
|
+void test_ttl_cascade_wal_is_durable_before_the_lmdb_commit() {
|
|
|
|
|
+ TmpEnv t("ttl-wal-first");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("ttl-wal-first-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+
|
|
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::Cascade);
|
|
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager);
|
|
|
|
|
+
|
|
|
|
|
+ // Overwrite the child sub-db's identity sentinel with a name that is not
|
|
|
|
|
+ // its own. open_for_write verifies it on every cached-handle reuse.
|
|
|
|
|
+ {
|
|
|
|
|
+ MDB_txn* txn = nullptr;
|
|
|
|
|
+ check(mdb_txn_begin(t.env.raw(), nullptr, 0, &txn) == 0, "tamper txn opened");
|
|
|
|
|
+ MDB_dbi dbi = 0;
|
|
|
|
|
+ check(mdb_dbi_open(txn, "executions", 0, &dbi) == 0, "child sub-db opened for tampering");
|
|
|
|
|
+ const std::string_view sentinel{"\0__subdb_identity__", 19};
|
|
|
|
|
+ MDB_val k{sentinel.size(), const_cast<char*>(sentinel.data())};
|
|
|
|
|
+ std::string wrong = "not_executions";
|
|
|
|
|
+ MDB_val v{wrong.size(), wrong.data()};
|
|
|
|
|
+ check(mdb_put(txn, dbi, &k, &v, 0) == 0, "sentinel overwritten");
|
|
|
|
|
+ check(mdb_txn_commit(txn) == 0, "tamper txn committed");
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
|
|
+ check(expired == 0, "the sweeper did NOT expire the parent when the cascade faulted");
|
|
|
|
|
+ check(residentInMemory(mstore, "default:workflows", "wf-1"),
|
|
|
|
|
+ "the parent is still in MemoryStore - never expired behind a failed cascade");
|
|
|
|
|
+ check(store.get("workflows", "wf-1").has_value(),
|
|
|
|
|
+ "and still in LMDB - commitCascadeLmdb's WriteTxn aborted unwritten");
|
|
|
|
|
+ check(store.get("executions", "ex-1").has_value(), "the child is still in LMDB too");
|
|
|
|
|
+ check(mstore.getStats().ttlExpiryBlockedByRelation == 1, "the skip is counted");
|
|
|
|
|
+
|
|
|
|
|
+ // ...and yet the WAL already describes the whole cascade. That is the
|
|
|
|
|
+ // ordering: WAL fsynced, THEN the LMDB commit attempted.
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+ auto entries = replayRawWal(p.path);
|
|
|
|
|
+ check(walHasDelete(entries, "default:executions", "ex-1"),
|
|
|
|
|
+ "the child's DELETE is already durable in the WAL, before any LMDB commit");
|
|
|
|
|
+ check(walHasDelete(entries, "default:workflows", "wf-1"),
|
|
|
|
|
+ "so is the parent's - the next boot's replay applies the cascade regardless");
|
|
|
|
|
+
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+
|
|
|
|
|
+// =========================================================================
|
|
|
|
|
+// v2.11.0 close-out review, FINDING 1 — a concurrent TTL change must not have
|
|
|
|
|
+// its document (and its children) deleted by an in-flight sweep.
|
|
|
|
|
+//
|
|
|
|
|
+// Phase 1 collects candidates and releases every lock; the hook may then DELETE
|
|
|
|
|
+// the document and cascade its children, and the hook has no view of
|
|
|
|
|
+// `expiresAt`, so `Handled` cannot re-check the way `Proceed` does. An
|
|
|
|
|
+// update()/patch() that extends or clears the TTL in that window would
|
|
|
|
|
+// otherwise destroy a document that is no longer expired - and its children with
|
|
|
|
|
+// it. The pre-v2.11.0 loop held the collection lock across the whole erase and
|
|
|
|
|
+// had no such window, so this is a cost of the two-phase split.
|
|
|
|
|
+//
|
|
|
|
|
+// ⚠ HOW THIS IS INDUCED, and why it is honest rather than staged: a
|
|
|
|
|
+// single-threaded test cannot interleave a real writer, so the write is made to
|
|
|
|
|
+// land at exactly the wrong moment from INSIDE the sweep. Two documents expire
|
|
|
|
|
+// in the same sweep, wf-early before wf-late (the expiration index is keyed by
|
|
|
|
|
+// expiry time and walked in ascending order, so the order is deterministic).
|
|
|
|
|
+// The hook, while handling wf-early, extends wf-late's TTL - which is precisely
|
|
|
|
|
+// "a write landed after phase 1 collected wf-late and before the sweeper got to
|
|
|
|
|
+// it". Everything about wf-late's path after that point is the production path.
|
|
|
|
|
+void test_ttl_concurrent_ttl_extension_is_not_expired() {
|
|
|
|
|
+ TmpEnv t("ttl-race");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("ttl-race-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+
|
|
|
|
|
+ RelationInfo rel;
|
|
|
|
|
+ rel.name = "default:exec_wf";
|
|
|
|
|
+ rel.child = "default:executions";
|
|
|
|
|
+ rel.childField = "workflowId";
|
|
|
|
|
+ rel.parent = "default:workflows";
|
|
|
|
|
+ rel.onDelete = OnDelete::Cascade;
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "declared the cascade relation");
|
|
|
|
|
+ store.set_relations("executions",
|
|
|
|
|
+ {RelationRef{"exec_wf", "workflowId", "workflows", false, true}});
|
|
|
|
|
+
|
|
|
|
|
+ const uint64_t farFuture = static_cast<uint64_t>(1) << 62;
|
|
|
|
|
+
|
|
|
|
|
+ auto seedParent = [&](const std::string& id, uint64_t expiresAt) {
|
|
|
|
|
+ Document d; d.id = id; d.collection = "workflows";
|
|
|
|
|
+ d.set_data({{"name", id}});
|
|
|
|
|
+ d.expiresAt = expiresAt;
|
|
|
|
|
+ store.put("workflows", id, d);
|
|
|
|
|
+ mstore.loadDocument("default:workflows", d);
|
|
|
|
|
+ };
|
|
|
|
|
+ auto seedChild = [&](const std::string& id, const std::string& parentId) {
|
|
|
|
|
+ Document d; d.id = id; d.collection = "executions";
|
|
|
|
|
+ d.set_data({{"workflowId", parentId}});
|
|
|
|
|
+ store.put("executions", id, d);
|
|
|
|
|
+ mstore.loadDocument("default:executions", d);
|
|
|
|
|
+ };
|
|
|
|
|
+
|
|
|
|
|
+ // expiresAt 1 sorts before 2, so wf-early is handled first in the sweep.
|
|
|
|
|
+ seedParent("wf-early", 1);
|
|
|
|
|
+ seedParent("wf-late", 2);
|
|
|
|
|
+ seedChild("ex-early", "wf-early");
|
|
|
|
|
+ seedChild("ex-late", "wf-late");
|
|
|
|
|
+
|
|
|
|
|
+ // The injected "concurrent" write: while the sweeper is busy expiring
|
|
|
|
|
+ // wf-early (a real cascade - WAL fsync plus an LMDB commit, which is what
|
|
|
|
|
+ // makes the window wide), wf-late's TTL is extended.
|
|
|
|
|
+ bool injected = false;
|
|
|
|
|
+ mstore.setTtlExpiryRelationHook(
|
|
|
|
|
+ [&](const std::string& coll, const std::string& id, bool firstAttempt) {
|
|
|
|
|
+ if (!injected && id == "wf-early") {
|
|
|
|
|
+ injected = true;
|
|
|
|
|
+ Document renewed;
|
|
|
|
|
+ renewed.id = "wf-late";
|
|
|
|
|
+ renewed.collection = "workflows";
|
|
|
|
|
+ renewed.set_data({{"name", "wf-late"}});
|
|
|
|
|
+ renewed.expiresAt = farFuture; // TTL extended
|
|
|
|
|
+ mstore.loadDocument("default:workflows", renewed);
|
|
|
|
|
+ }
|
|
|
|
|
+ return ttlExpiryDecision(rm, store, p.pm, mstore, cfgManager, coll, id,
|
|
|
|
|
+ nullptr, firstAttempt);
|
|
|
|
|
+ });
|
|
|
|
|
+
|
|
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
|
|
+
|
|
|
|
|
+ check(injected, "the injected concurrent TTL extension actually ran");
|
|
|
|
|
+ check(expired == 1, "exactly ONE document expired - wf-early only");
|
|
|
|
|
+
|
|
|
|
|
+ // wf-early: genuinely expired, cascaded as normal. Proves the sweep worked.
|
|
|
|
|
+ check(!residentInMemory(mstore, "default:workflows", "wf-early"), "wf-early expired");
|
|
|
|
|
+ check(!store.get("executions", "ex-early").has_value(), "its child cascaded away");
|
|
|
|
|
+
|
|
|
|
|
+ // wf-late: the whole point. Not expired, and - the data-loss half - its
|
|
|
|
|
+ // child was not cascaded either.
|
|
|
|
|
+ check(residentInMemory(mstore, "default:workflows", "wf-late"),
|
|
|
|
|
+ "wf-late was NOT expired - its TTL was extended after phase 1 collected it");
|
|
|
|
|
+ check(store.get("workflows", "wf-late").has_value(), "and it is intact in LMDB");
|
|
|
|
|
+ check(store.get("executions", "ex-late").has_value(),
|
|
|
|
|
+ "and ITS CHILD was not cascaded - this is the data-loss half of the race");
|
|
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-late") == 1,
|
|
|
|
|
+ "the reverse-index posting survived too");
|
|
|
|
|
+ check(mstore.getStats().ttlExpiryBlockedByRelation == 0,
|
|
|
|
|
+ "and it was not recorded as relation-blocked - it simply is not expired");
|
|
|
|
|
+
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// =========================================================================
|
|
|
|
|
+// v2.11.0 close-out review, FINDING 2 — a relation-blocked document must not
|
|
|
|
|
+// consume the sweep's candidate budget.
|
|
|
|
|
+//
|
|
|
|
|
+// Phase 1 never erases a blocked document's expiration index entry (deliberately
|
|
|
|
|
+// - it has to stay armed so the block can lift), so with one shared budget those
|
|
|
|
|
+// documents refill the cap on every sweep. `collections_` iteration order is
|
|
|
|
|
+// stable, so every collection after them SILENTLY STOPS BEING SWEPT - and TTL'd
|
|
|
|
|
+// parents under `restrict` is exactly the combination this feature creates, so
|
|
|
|
|
+// that is the expected steady state, not an edge case.
|
|
|
|
|
+//
|
|
|
|
|
+// Budgets come from Config so this can be shown at 3 documents instead of 10001.
|
|
|
|
|
+void test_ttl_blocked_documents_do_not_starve_the_sweep() {
|
|
|
|
|
+ TmpEnv t("ttl-starve");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("ttl-starve-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore::Config cfg;
|
|
|
|
|
+ cfg.ttlMaxCandidatesPerSweep = 3; // exactly filled by the blocked parents
|
|
|
|
|
+ cfg.ttlBlockedRetrySweeps = 100; // far enough away not to interfere
|
|
|
|
|
+ MemoryStore mstore(cfg);
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+
|
|
|
|
|
+ std::atomic<bool> healthy{true};
|
|
|
|
|
+ std::atomic<uint64_t> drift{0};
|
|
|
|
|
+ mstore.setDocumentStoreMirror(
|
|
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
|
|
+ &healthy, &drift);
|
|
|
|
|
+
|
|
|
|
|
+ RelationInfo rel;
|
|
|
|
|
+ rel.name = "default:exec_wf";
|
|
|
|
|
+ rel.child = "default:executions";
|
|
|
|
|
+ rel.childField = "workflowId";
|
|
|
|
|
+ rel.parent = "default:workflows";
|
|
|
|
|
+ rel.onDelete = OnDelete::Restrict;
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "declared the restrict relation");
|
|
|
|
|
+ store.set_relations("executions",
|
|
|
|
|
+ {RelationRef{"exec_wf", "workflowId", "workflows", false, true}});
|
|
|
|
|
+
|
|
|
|
|
+ // Three restrict-blocked TTL'd parents. Their expiry times are the earliest,
|
|
|
|
|
+ // so they are collected first and fill the budget on the first sweep.
|
|
|
|
|
+ for (int i = 0; i < 3; ++i) {
|
|
|
|
|
+ const std::string pid = "wf-" + std::to_string(i);
|
|
|
|
|
+ Document parent; parent.id = pid; parent.collection = "workflows";
|
|
|
|
|
+ parent.set_data({{"name", pid}});
|
|
|
|
|
+ parent.expiresAt = 1 + static_cast<uint64_t>(i);
|
|
|
|
|
+ store.put("workflows", pid, parent);
|
|
|
|
|
+ mstore.loadDocument("default:workflows", parent);
|
|
|
|
|
+
|
|
|
|
|
+ Document child; child.id = "ex-" + std::to_string(i);
|
|
|
|
|
+ child.collection = "executions";
|
|
|
|
|
+ child.set_data({{"workflowId", pid}});
|
|
|
|
|
+ store.put("executions", child.id, child);
|
|
|
|
|
+ mstore.loadDocument("default:executions", child);
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // The victim: a TTL'd document with NO children, expiring after the three
|
|
|
|
|
+ // blocked parents. Deliberately in the SAME collection, because that makes
|
|
|
|
|
+ // the ordering deterministic - `collections_` is an unordered_map, but a
|
|
|
|
|
+ // collection's expirationIndex is a std::map walked in ascending expiry
|
|
|
|
|
+ // order, so wf-0/1/2 (1, 2, 3) are always collected before wf-victim (100).
|
|
|
|
|
+ // Starvation inside one collection is the same defect as starvation across
|
|
|
|
|
+ // collections, and this way the test cannot pass or fail on hash order.
|
|
|
|
|
+ Document victim;
|
|
|
|
|
+ victim.id = "wf-victim";
|
|
|
|
|
+ victim.collection = "workflows";
|
|
|
|
|
+ victim.set_data({{"name", "wf-victim"}});
|
|
|
|
|
+ victim.expiresAt = 100;
|
|
|
|
|
+ store.put("workflows", "wf-victim", victim);
|
|
|
|
|
+ mstore.loadDocument("default:workflows", victim);
|
|
|
|
|
+
|
|
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager);
|
|
|
|
|
+
|
|
|
|
|
+ // Sweep 1: the three blocked parents are fresh, so they legitimately take
|
|
|
|
|
+ // the whole budget and the victim is not reached at all.
|
|
|
|
|
+ const uint64_t first = mstore.expireDocuments();
|
|
|
|
|
+ check(first == 0, "sweep 1 expired nothing - the budget went to the blocked parents");
|
|
|
|
|
+ check(mstore.getStats().ttlExpiryBlockedByRelation == 3, "all three parents were refused");
|
|
|
|
|
+ check(mstore.ttlBlockedDocumentCount() == 3, "and all three are tracked as stuck");
|
|
|
|
|
+ check(residentInMemory(mstore, "default:workflows", "wf-victim"),
|
|
|
|
|
+ "and the victim has not been reached yet");
|
|
|
|
|
+
|
|
|
|
|
+ // Sweep 2: the blocked parents are known and not due, so they must consume
|
|
|
|
|
+ // NOTHING and the victim finally gets in. With one shared budget it never
|
|
|
|
|
+ // would - the three blocked parents refill the cap on every sweep, forever.
|
|
|
|
|
+ uint64_t total = first;
|
|
|
|
|
+ total += mstore.expireDocuments();
|
|
|
|
|
+ check(total == 1,
|
|
|
|
|
+ "the childless document was expired despite three blocked parents filling "
|
|
|
|
|
+ "the budget - blocked documents no longer starve the sweep");
|
|
|
|
|
+ check(!residentInMemory(mstore, "default:workflows", "wf-victim"), "the victim is gone");
|
|
|
|
|
+ check(!store.get("workflows", "wf-victim").has_value(), "and its DELETE reached LMDB");
|
|
|
|
|
+ check(mstore.getStats().ttlExpiryBlockedByRelation == 3,
|
|
|
|
|
+ "and the blocked parents were not re-consulted - still three refusal events, "
|
|
|
|
|
+ "not three more per sweep");
|
|
|
|
|
+
|
|
|
|
|
+ // Ten more sweeps: still nothing re-examined, so the steady-state cost of a
|
|
|
|
|
+ // persistent block is zero rather than one LMDB scan + one WARN per document
|
|
|
|
|
+ // per second.
|
|
|
|
|
+ for (int i = 0; i < 10; ++i) mstore.expireDocuments();
|
|
|
|
|
+ check(mstore.getStats().ttlExpiryBlockedByRelation == 3,
|
|
|
|
|
+ "ten further sweeps consulted the relation zero times");
|
|
|
|
|
+ check(mstore.ttlBlockedDocumentCount() == 3, "and the three are still tracked as stuck");
|
|
|
|
|
+
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// Round-3 review, item 2 — dropping a blocked document's collection must not
|
|
|
|
|
+// leak its blocked-set entry. Phase 1 never revisits a key whose collection is
|
|
|
|
|
+// gone and dropCollection() knows nothing about the blocked set, so a leaked
|
|
|
|
|
+// entry would inflate ttlBlockedDocumentCount() - the gauge an operator reads -
|
|
|
|
|
+// for the rest of the process's life.
|
|
|
|
|
+void test_ttl_blocked_entry_is_released_when_the_collection_is_dropped() {
|
|
|
|
|
+ TmpEnv t("ttl-blocked-drop");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("ttl-blocked-drop-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore::Config cfg;
|
|
|
|
|
+ cfg.ttlBlockedRetrySweeps = 1; // due again on the very next sweep
|
|
|
|
|
+ MemoryStore mstore(cfg);
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+
|
|
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::Restrict);
|
|
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager);
|
|
|
|
|
+
|
|
|
|
|
+ check(mstore.expireDocuments() == 0, "the restrict-protected parent is blocked");
|
|
|
|
|
+ check(mstore.ttlBlockedDocumentCount() == 1, "and is counted as stuck");
|
|
|
|
|
+
|
|
|
|
|
+ // The collection goes away underneath it.
|
|
|
|
|
+ mstore.dropCollection("default:workflows");
|
|
|
|
|
+ check(mstore.expireDocuments() == 0, "nothing left to expire");
|
|
|
|
|
+ check(mstore.ttlBlockedDocumentCount() == 0,
|
|
|
|
|
+ "the blocked entry was released with the collection - the gauge does not "
|
|
|
|
|
+ "keep counting a document that no longer exists");
|
|
|
|
|
+
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// =========================================================================
|
|
|
|
|
+// v2.11.0 close-out — BOOT-PASS DRIFT MUST NOT DISABLE DESTRUCTIVE POLICIES.
|
|
|
|
|
+//
|
|
|
|
|
+// mirror_drift_count_ is never reset, and the post-replay re-mirror pass bumps
|
|
|
|
|
+// it for every row it cannot repair - a condition runPendingRemirror() treats as
|
|
|
|
|
+// advisory and which recurs on every boot, since the restart replays the same
|
|
|
|
|
+// failing row. Gated on the raw counter, ONE unrepairable row refused every
|
|
|
|
|
+// cascade/set_null delete and every CreateRelation for the process's life, with
|
|
|
|
|
+// an error message that told the operator to restart.
|
|
|
|
|
+// =========================================================================
|
|
|
|
|
+void test_boot_pass_drift_does_not_disable_the_cascade() {
|
|
|
|
|
+ TmpEnv t("rel-drift-baseline");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ TmpPersistence p("rel-drift-baseline-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+
|
|
|
|
|
+ std::atomic<bool> healthy{true};
|
|
|
|
|
+ std::atomic<uint64_t> drift{0};
|
|
|
|
|
+ mstore.setDocumentStoreMirror(
|
|
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
|
|
+ &healthy, &drift);
|
|
|
|
|
+
|
|
|
|
|
+ // The boot path: three rows the re-mirror pass could not write. Health is
|
|
|
|
|
+ // deliberately NOT flipped (the v2.8.1 lesson), so drift is the only signal.
|
|
|
|
|
+ drift.store(3);
|
|
|
|
|
+ check(mstore.mirrorDriftCount() == 3, "the raw counter still reports every stale row");
|
|
|
|
|
+ mstore.markMirrorDriftBaseline(); // what initialize() does last
|
|
|
|
|
+ check(mstore.mirrorDriftBaseline() == 3, "the boot path's drift is in the baseline");
|
|
|
|
|
+ check(mstore.mirrorDriftSinceBaseline() == 0,
|
|
|
|
|
+ "and nothing has drifted since READY");
|
|
|
|
|
+
|
|
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::Cascade);
|
|
|
|
|
+
|
|
|
|
|
+ bool parentExisted = false;
|
|
|
|
|
+ try {
|
|
|
|
|
+ parentExisted = executeCascade(rm, store, p.pm, mstore, cfgManager,
|
|
|
|
|
+ "default:workflows", "wf-1", nullptr);
|
|
|
|
|
+ } catch (const std::exception& e) {
|
|
|
|
|
+ check(false, "the cascade was refused because of drift the BOOT PASS accrued");
|
|
|
|
|
+ std::cerr << " (" << e.what() << ")\n";
|
|
|
|
|
+ }
|
|
|
|
|
+ check(parentExisted, "the cascade ran despite 3 rows the boot pass left stale");
|
|
|
|
|
+ check(!store.get("executions", "ex-1").has_value(), "and it actually cascaded");
|
|
|
|
|
+
|
|
|
|
|
+ // A LIVE write's drift still latches and still refuses - that half was
|
|
|
|
|
+ // right, and re-basing the gate must not have removed it.
|
|
|
|
|
+ drift.fetch_add(1);
|
|
|
|
|
+ check(mstore.mirrorDriftSinceBaseline() == 1, "post-READY drift is visible");
|
|
|
|
|
+ Document parent2;
|
|
|
|
|
+ parent2.id = "wf-2";
|
|
|
|
|
+ parent2.collection = "workflows";
|
|
|
|
|
+ parent2.set_data({{"name", "wf-2"}});
|
|
|
|
|
+ store.put("workflows", "wf-2", parent2);
|
|
|
|
|
+ mstore.loadDocument("default:workflows", parent2);
|
|
|
|
|
+ Document child2;
|
|
|
|
|
+ child2.id = "ex-2";
|
|
|
|
|
+ child2.collection = "executions";
|
|
|
|
|
+ child2.set_data({{"workflowId", "wf-2"}});
|
|
|
|
|
+ store.put("executions", "ex-2", child2);
|
|
|
|
|
+ mstore.loadDocument("default:executions", child2);
|
|
|
|
|
+
|
|
|
|
|
+ bool refused = false;
|
|
|
|
|
+ try {
|
|
|
|
|
+ executeCascade(rm, store, p.pm, mstore, cfgManager, "default:workflows", "wf-2", nullptr);
|
|
|
|
|
+ } catch (const CascadeBlocked&) {
|
|
|
|
|
+ refused = true;
|
|
|
|
|
+ }
|
|
|
|
|
+ check(refused, "drift accrued AFTER READY still refuses the cascade");
|
|
|
|
|
+ check(store.get("executions", "ex-2").has_value(), "and nothing was touched");
|
|
|
|
|
+
|
|
|
|
|
+ p.pm.stop();
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+// v2.11.0 close-out — runPendingRemirror() must RELEASE the id list.
|
|
|
|
|
+// RecoveryOutcome lives as DatabaseService::recovery_outcome_ for the process
|
|
|
|
|
+// lifetime; the pass runs exactly once and nothing reads the list afterwards, so
|
|
|
|
|
+// keeping it resident pinned ~100 bytes per replayed row (an install's whole
|
|
|
|
|
+// history under --recovery-mode=wal_only) for nothing.
|
|
|
|
|
+void test_pending_remirror_list_is_released_after_the_pass() {
|
|
|
|
|
+ TmpPersistence p("remirror-release");
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+
|
|
|
|
|
+ smartbotic::database::RecoveryOutcome outcome;
|
|
|
|
|
+ for (int i = 0; i < 500; ++i) {
|
|
|
|
|
+ outcome.pendingRemirror.emplace_back("default:widgets", "w-" + std::to_string(i));
|
|
|
|
|
+ }
|
|
|
|
|
+ check(outcome.pendingRemirror.size() == 500, "the list starts populated");
|
|
|
|
|
+
|
|
|
|
|
+ p.pm.runPendingRemirror(mstore, outcome);
|
|
|
|
|
+
|
|
|
|
|
+ check(outcome.pendingRemirror.empty(),
|
|
|
|
|
+ "the id list is cleared once the pass has run");
|
|
|
|
|
+ check(outcome.pendingRemirror.capacity() == 0,
|
|
|
|
|
+ "and its CAPACITY is released - clear() alone keeps the whole allocation");
|
|
|
|
|
+
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+int main() {
|
|
|
|
|
+ std::cout << "=== test_relation_enforcement ===\n";
|
|
|
|
|
+ test_index_follows_the_child_field();
|
|
|
|
|
+ test_undeclared_collection_maintains_nothing();
|
|
|
|
|
+ test_unrelated_update_leaves_the_posting_alone();
|
|
|
|
|
+ test_posting_visible_after_reopen();
|
|
|
|
|
+ test_restrict_blocks_and_names_the_blockers();
|
|
|
|
|
+ test_cascade_and_set_null_never_block_via_restrict_check();
|
|
|
|
|
+ test_relations_enforced_false_skips_the_check();
|
|
|
|
|
+ test_describe_delete_reports_restrict_and_no_action();
|
|
|
|
|
+ test_describe_delete_zero_children_does_not_block();
|
|
|
|
|
+ test_describe_delete_no_relations_declared();
|
|
|
|
|
+ test_cascade_deletes_scalar_children_wal_first();
|
|
|
|
|
+ test_array_reference_pulls_id_and_keeps_document();
|
|
|
|
|
+ test_crash_between_wal_and_commit_recovers();
|
|
|
|
|
+ test_cascade_refuses_when_grandchild_is_restrict_protected();
|
|
|
|
|
+ test_replayed_cascade_update_remirrors_to_lmdb_after_crash_window();
|
|
|
|
|
+ test_a_failing_row_does_not_stop_recovery();
|
|
|
|
|
+ test_reinsert_after_delete_converges_in_lmdb();
|
|
|
|
|
+ test_validate_on_write_rejects_missing_parent();
|
|
|
|
|
+ test_validate_on_write_accepts_existing_parent();
|
|
|
|
|
+ test_validate_on_write_false_allows_dangling_and_check_reports_it();
|
|
|
|
|
+ test_validate_on_write_never_rejects_absent_or_null();
|
|
|
|
|
+ test_validate_on_write_array_any_missing_rejects_whole_write();
|
|
|
|
|
+ test_validate_on_write_unrelated_update_not_rechecked();
|
|
|
|
|
+ test_validate_on_write_rolls_back_an_already_applied_sibling_relation();
|
|
|
|
|
+ test_validate_on_write_relations_enforced_false_is_the_escape_hatch();
|
|
|
|
|
+ test_validate_on_write_closes_the_stale_check_race();
|
|
|
|
|
+ test_remirror_maintains_the_secondary_index();
|
|
|
|
|
+ test_moved_unique_value_is_resolved_by_the_retry_pass();
|
|
|
|
|
+ test_remirror_windows_are_bounded_by_bytes_and_by_count();
|
|
|
|
|
+ test_cascade_refuses_while_the_mirror_is_unhealthy_or_drifted();
|
|
|
|
|
+ test_can_reject_writes_gates_the_undo_snapshot();
|
|
|
|
|
+ test_ttl_restrict_blocks_the_expiry();
|
|
|
|
|
+ test_ttl_cascade_deletes_children_like_a_manual_delete();
|
|
|
|
|
+ test_ttl_set_null_nulls_the_scalar_and_keeps_the_child();
|
|
|
|
|
+ test_ttl_array_reference_pulls_the_id_and_keeps_the_document();
|
|
|
|
|
+ test_ttl_no_action_expires_and_leaves_the_reference_dangling();
|
|
|
|
|
+ test_ttl_with_no_relations_expires_exactly_as_before();
|
|
|
|
|
+ test_ttl_cascade_wal_is_durable_before_the_lmdb_commit();
|
|
|
|
|
+ test_ttl_concurrent_ttl_extension_is_not_expired();
|
|
|
|
|
+ test_ttl_blocked_documents_do_not_starve_the_sweep();
|
|
|
|
|
+ test_ttl_blocked_entry_is_released_when_the_collection_is_dropped();
|
|
|
|
|
+ test_boot_pass_drift_does_not_disable_the_cascade();
|
|
|
|
|
+ test_pending_remirror_list_is_released_after_the_pass();
|
|
|
|
|
+
|
|
|
|
|
+ std::cout << "passed: " << g_pass << ", failed: " << g_fail << "\n";
|
|
|
|
|
+ return g_fail == 0 ? 0 : 1;
|
|
|
|
|
+}
|