|
|
@@ -58,6 +58,7 @@ using smartbotic::database::describeDeleteImpacts;
|
|
|
using smartbotic::database::executeCascade;
|
|
|
using smartbotic::database::findRelationBlocks;
|
|
|
using smartbotic::database::planCascade;
|
|
|
+using smartbotic::database::ttlExpiryDecision;
|
|
|
using smartbotic::database::writeCascadeWal;
|
|
|
using smartbotic::db::storage::LmdbDocumentStore;
|
|
|
using smartbotic::db::storage::LmdbEnv;
|
|
|
@@ -2079,6 +2080,529 @@ void test_can_reject_writes_gates_the_undo_snapshot() {
|
|
|
ms.stop();
|
|
|
}
|
|
|
|
|
|
+
|
|
|
+// =========================================================================
|
|
|
+// v2.11.0 close-out — TTL EXPIRY MUST HANDLE CHILDREN LIKE A MANUAL DELETE.
|
|
|
+//
|
|
|
+// Operator's ruling. Before this, MemoryStore::expireDocuments() erased the
|
|
|
+// document and mirrored a DELETE with no relation enforcement whatsoever, so a
|
|
|
+// restrict-protected parent carrying a TTL silently vanished and orphaned every
|
|
|
+// child - strictly the worst of the three available behaviours.
|
|
|
+//
|
|
|
+// Every test below drives the REAL sweeper (MemoryStore::expireDocuments()) with
|
|
|
+// the REAL decision function (relations/relation_cascade.cpp's
|
|
|
+// ttlExpiryDecision(), which DatabaseService binds as the production hook), so
|
|
|
+// what is under test is the shipped path and not a re-statement of it.
|
|
|
+// =========================================================================
|
|
|
+
|
|
|
+// Residency probe. MemoryStore::get() deliberately hides an EXPIRED document
|
|
|
+// (Document::isExpired()), so "is it still there" cannot be asked with get()
|
|
|
+// when the whole point is that the document is past its TTL and still present.
|
|
|
+// getAllDocuments() does not filter.
|
|
|
+bool residentInMemory(MemoryStore& mstore, const std::string& collection,
|
|
|
+ const std::string& id) {
|
|
|
+ for (const auto& d : mstore.getAllDocuments(collection)) {
|
|
|
+ if (d.id == id) return true;
|
|
|
+ }
|
|
|
+ return false;
|
|
|
+}
|
|
|
+
|
|
|
+// Install the production decision function as the sweeper's hook. Exactly what
|
|
|
+// DatabaseService::ttlExpiryRelationDecision() does, minus the per-project store
|
|
|
+// resolution (this fixture has one store) and the replication notifier.
|
|
|
+void installTtlHook(MemoryStore& mstore, RelationManager& rm, LmdbDocumentStore& store,
|
|
|
+ PersistenceManager& pm, CollectionConfigManager& cfgManager,
|
|
|
+ const smartbotic::database::CascadeNotifyFn& notify = nullptr) {
|
|
|
+ mstore.setTtlExpiryRelationHook(
|
|
|
+ [&rm, &store, &pm, &mstore, &cfgManager, notify](const std::string& coll,
|
|
|
+ const std::string& id) {
|
|
|
+ return ttlExpiryDecision(rm, store, pm, mstore, cfgManager, coll, id, notify);
|
|
|
+ });
|
|
|
+}
|
|
|
+
|
|
|
+// A parent carrying a TTL, plus one child, plus a declared relation. Returns
|
|
|
+// nothing; the caller owns every object so the fixtures stay explicit (this
|
|
|
+// file's established style).
|
|
|
+void seedTtlParentAndChild(MemoryStore& mstore, LmdbDocumentStore& store,
|
|
|
+ RelationManager& rm, OnDelete policy,
|
|
|
+ const std::string& childField = "workflowId",
|
|
|
+ const nlohmann::json& childData = {{"workflowId", "wf-1"}}) {
|
|
|
+ RelationInfo rel;
|
|
|
+ rel.name = "default:exec_wf";
|
|
|
+ rel.child = "default:executions";
|
|
|
+ rel.childField = childField;
|
|
|
+ rel.parent = "default:workflows";
|
|
|
+ rel.onDelete = policy;
|
|
|
+ std::string mgrErr;
|
|
|
+ check(rm.createRelation(rel, mgrErr), "declared the relation under test");
|
|
|
+ store.set_relations("executions",
|
|
|
+ {RelationRef{"exec_wf", childField, "workflows", false, true}});
|
|
|
+
|
|
|
+ // ⚠ expiresAt = 1 (one millisecond after the epoch) rather than "now minus
|
|
|
+ // something": expireDocuments() compares against currentTimeMs(), so 1 is
|
|
|
+ // unambiguously expired on any clock and the test cannot race the sweep.
|
|
|
+ Document parent;
|
|
|
+ parent.id = "wf-1";
|
|
|
+ parent.collection = "workflows";
|
|
|
+ parent.set_data({{"name", "wf-1"}});
|
|
|
+ parent.expiresAt = 1;
|
|
|
+ store.put("workflows", "wf-1", parent);
|
|
|
+ mstore.loadDocument("default:workflows", parent);
|
|
|
+
|
|
|
+ Document child;
|
|
|
+ child.id = "ex-1";
|
|
|
+ child.collection = "executions";
|
|
|
+ child.set_data(childData);
|
|
|
+ store.put("executions", "ex-1", child);
|
|
|
+ mstore.loadDocument("default:executions", child);
|
|
|
+}
|
|
|
+
|
|
|
+// restrict: a MANUAL delete fails, so the expiry must not happen. The document
|
|
|
+// is left in place, its expiry stays armed, and the block is signalled.
|
|
|
+//
|
|
|
+// Would fail before the fix on its first assertion: the pre-v2.11.0 sweeper
|
|
|
+// returned 1 and erased the parent, leaving ex-1 pointing at nothing.
|
|
|
+void test_ttl_restrict_blocks_the_expiry() {
|
|
|
+ TmpEnv t("ttl-restrict");
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
+ TmpPersistence p("ttl-restrict-wal");
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
+
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
+ mstore.start();
|
|
|
+ RelationManager rm(mstore);
|
|
|
+ rm.loadFromStore();
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
+
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::Restrict);
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 1, "one child indexed");
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager);
|
|
|
+
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
+ check(expired == 0, "the restrict-protected parent was NOT expired");
|
|
|
+ check(residentInMemory(mstore, "default:workflows", "wf-1"),
|
|
|
+ "the parent is still in MemoryStore - it OUTLIVES its TTL, deliberately");
|
|
|
+ check(store.get("workflows", "wf-1").has_value(), "the parent is still in LMDB");
|
|
|
+ check(mstore.get("default:executions", "ex-1").has_value(), "the child was not orphaned");
|
|
|
+ check(mstore.getStats().ttlExpiryBlockedByRelation == 1,
|
|
|
+ "the skip is counted, so an operator can see a document is stuck past its TTL");
|
|
|
+ check(mstore.getStats().expiredCount == 0, "and it is not counted as expired");
|
|
|
+
|
|
|
+ // The expiry must stay ARMED so the next sweep retries - a skip that also
|
|
|
+ // dropped the expiration index entry would leave the document permanently
|
|
|
+ // unexpirable even after the last child went away.
|
|
|
+ const uint64_t again = mstore.expireDocuments();
|
|
|
+ check(again == 0, "still blocked on the second sweep");
|
|
|
+ check(mstore.getStats().ttlExpiryBlockedByRelation == 2,
|
|
|
+ "the second sweep re-examined it, so the expiry is still armed");
|
|
|
+
|
|
|
+ // And once the child is gone, the same sweep expires it - the block is the
|
|
|
+ // relation's, not a permanent quarantine.
|
|
|
+ store.del("executions", "ex-1");
|
|
|
+ check(mstore.remove("default:executions", "ex-1"), "child removed");
|
|
|
+ const uint64_t third = mstore.expireDocuments();
|
|
|
+ check(third == 1, "with no children left, the retry finally expires the parent");
|
|
|
+ check(!mstore.get("default:workflows", "wf-1").has_value(), "the parent is gone now");
|
|
|
+
|
|
|
+ p.pm.stop();
|
|
|
+ mstore.stop();
|
|
|
+}
|
|
|
+
|
|
|
+// cascade with a SCALAR reference: children are deleted, exactly as
|
|
|
+// executeCascade does for a manual delete, and the WAL carries every mutation.
|
|
|
+//
|
|
|
+// Would fail before the fix: the old sweeper wrote no WAL entry for ex-1 at all
|
|
|
+// (it only mirrored a DELETE for the parent), so walHasDelete for the child is
|
|
|
+// the assertion that a hand-rolled, LMDB-only sweeper cascade cannot satisfy.
|
|
|
+void test_ttl_cascade_deletes_children_like_a_manual_delete() {
|
|
|
+ TmpEnv t("ttl-cascade");
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
+ TmpPersistence p("ttl-cascade-wal");
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
+
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
+ mstore.start();
|
|
|
+ RelationManager rm(mstore);
|
|
|
+ rm.loadFromStore();
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
+
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::Cascade);
|
|
|
+
|
|
|
+ // The notify wiring: a TTL-driven cascade must drive replication and
|
|
|
+ // Subscribe events per mutation, or a follower never sees the child
|
|
|
+ // deletions and diverges permanently.
|
|
|
+ struct Notification { std::string collection, id; bool hadDoc; smartbotic::database::EventType et; };
|
|
|
+ std::vector<Notification> notifications;
|
|
|
+ auto notify = [&](const std::string& coll, const std::string& id,
|
|
|
+ const std::optional<Document>& doc, smartbotic::database::EventType et) {
|
|
|
+ notifications.push_back({coll, id, doc.has_value(), et});
|
|
|
+ };
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager, notify);
|
|
|
+
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
+ check(expired == 1, "the parent was expired");
|
|
|
+ check(!mstore.get("default:executions", "ex-1").has_value(),
|
|
|
+ "the child was CASCADED, not orphaned");
|
|
|
+ check(!store.get("executions", "ex-1").has_value(), "and it is gone from LMDB too");
|
|
|
+ check(!store.get("workflows", "wf-1").has_value(), "the parent is gone from LMDB");
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 0,
|
|
|
+ "the reverse-index posting went with it");
|
|
|
+ check(notifications.size() == 2,
|
|
|
+ "replication/events fired once per child plus once for the parent");
|
|
|
+
|
|
|
+ // The property a hand-rolled sweeper cascade breaks: MemoryStore is rebuilt
|
|
|
+ // from snapshot + WAL, NEVER from LMDB, so a cascade whose child deletions
|
|
|
+ // exist only in LMDB has them RESURRECTED on the next boot.
|
|
|
+ p.pm.stop();
|
|
|
+ auto entries = replayRawWal(p.path);
|
|
|
+ check(walHasDelete(entries, "default:executions", "ex-1"),
|
|
|
+ "the WAL carries the CHILD's delete - without this the next boot resurrects it");
|
|
|
+ check(walHasDelete(entries, "default:workflows", "wf-1"),
|
|
|
+ "the WAL carries the parent's delete");
|
|
|
+
|
|
|
+ mstore.stop();
|
|
|
+}
|
|
|
+
|
|
|
+// set_null with a SCALAR reference: the field is nulled and the child KEPT.
|
|
|
+void test_ttl_set_null_nulls_the_scalar_and_keeps_the_child() {
|
|
|
+ TmpEnv t("ttl-setnull");
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
+ TmpPersistence p("ttl-setnull-wal");
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
+
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
+ mstore.start();
|
|
|
+ RelationManager rm(mstore);
|
|
|
+ rm.loadFromStore();
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
+
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::SetNull);
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager);
|
|
|
+
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
+ check(expired == 1, "the parent was expired");
|
|
|
+ auto child = store.get("executions", "ex-1");
|
|
|
+ check(child.has_value(), "the child SURVIVED - set_null keeps the document");
|
|
|
+ if (child) {
|
|
|
+ check(child->data().contains("workflowId") && child->data()["workflowId"].is_null(),
|
|
|
+ "and its reference is null, not deleted and not left dangling");
|
|
|
+ }
|
|
|
+ auto memChild = mstore.get("default:executions", "ex-1");
|
|
|
+ check(memChild.has_value(), "the child survived in MemoryStore too");
|
|
|
+ if (memChild) {
|
|
|
+ check(memChild->data()["workflowId"].is_null(), "with the same nulled reference");
|
|
|
+ }
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 0,
|
|
|
+ "the posting is gone, since the reference no longer names wf-1");
|
|
|
+
|
|
|
+ p.pm.stop();
|
|
|
+ auto entries = replayRawWal(p.path);
|
|
|
+ check(walHasUpdate(entries, "default:executions", "ex-1"),
|
|
|
+ "the child's UPDATE is in the WAL - the mutation survives a restart");
|
|
|
+ mstore.stop();
|
|
|
+}
|
|
|
+
|
|
|
+// An ARRAY-valued reference: the id is PULLED and the document KEPT, under
|
|
|
+// cascade as well as set_null. The place a MySQL mental model actively
|
|
|
+// misleads, and the same rule the request path follows - deleting a node
|
|
|
+// because one of its three credentials expired would be worse than useless.
|
|
|
+void test_ttl_array_reference_pulls_the_id_and_keeps_the_document() {
|
|
|
+ TmpEnv t("ttl-array");
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
+ TmpPersistence p("ttl-array-wal");
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
+
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
+ mstore.start();
|
|
|
+ RelationManager rm(mstore);
|
|
|
+ rm.loadFromStore();
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
+
|
|
|
+ // cascade, deliberately: the array rule must hold under the policy a
|
|
|
+ // MySQL-trained reader expects to delete the row.
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::Cascade, "workflowIds",
|
|
|
+ {{"workflowIds", {"wf-1", "wf-2"}}});
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 1,
|
|
|
+ "the array element is indexed as a posting");
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager);
|
|
|
+
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
+ check(expired == 1, "the parent was expired");
|
|
|
+ auto child = store.get("executions", "ex-1");
|
|
|
+ check(child.has_value(), "the child SURVIVED - an array match never deletes the document");
|
|
|
+ if (child) {
|
|
|
+ const auto& ids = child->data()["workflowIds"];
|
|
|
+ check(ids.is_array() && ids.size() == 1, "exactly one id was pulled");
|
|
|
+ check(ids.is_array() && !ids.empty() && ids[0] == "wf-2",
|
|
|
+ "and it was the expiring parent's id, not the surviving one");
|
|
|
+ }
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-1") == 0, "posting for wf-1 gone");
|
|
|
+ check(store.relation_index_child_count("exec_wf", "wf-2") == 1, "posting for wf-2 intact");
|
|
|
+
|
|
|
+ p.pm.stop();
|
|
|
+ mstore.stop();
|
|
|
+}
|
|
|
+
|
|
|
+// no_action: the expiry proceeds and the reference is left dangling, which is
|
|
|
+// what it did before this change and must keep doing.
|
|
|
+void test_ttl_no_action_expires_and_leaves_the_reference_dangling() {
|
|
|
+ TmpEnv t("ttl-noaction");
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
+ TmpPersistence p("ttl-noaction-wal");
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
+
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
+ mstore.start();
|
|
|
+ RelationManager rm(mstore);
|
|
|
+ rm.loadFromStore();
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
+
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::NoAction);
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager);
|
|
|
+
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
+ check(expired == 1, "no_action does not block the expiry");
|
|
|
+ check(!mstore.get("default:workflows", "wf-1").has_value(), "the parent expired");
|
|
|
+ auto child = mstore.get("default:executions", "ex-1");
|
|
|
+ check(child.has_value(), "the child is untouched");
|
|
|
+ if (child) {
|
|
|
+ check(child->data()["workflowId"] == "wf-1",
|
|
|
+ "and its reference still names the gone parent - dangling, as declared");
|
|
|
+ }
|
|
|
+ check(mstore.getStats().ttlExpiryBlockedByRelation == 0, "nothing was blocked");
|
|
|
+
|
|
|
+ p.pm.stop();
|
|
|
+ mstore.stop();
|
|
|
+}
|
|
|
+
|
|
|
+// SCOPE: the common case - no relation names this collection as a parent - must
|
|
|
+// behave exactly as it did before, including the LMDB DELETE mirror and the
|
|
|
+// EXPIRE event. This is the regression guard for the two-phase rewrite of
|
|
|
+// expireDocuments(), which is what actually changed for every install.
|
|
|
+void test_ttl_with_no_relations_expires_exactly_as_before() {
|
|
|
+ TmpEnv t("ttl-plain");
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
+ TmpPersistence p("ttl-plain-wal");
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
+
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
+ mstore.start();
|
|
|
+ RelationManager rm(mstore);
|
|
|
+ rm.loadFromStore();
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
+
|
|
|
+ std::atomic<bool> healthy{true};
|
|
|
+ std::atomic<uint64_t> drift{0};
|
|
|
+ mstore.setDocumentStoreMirror(
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
+ &healthy, &drift);
|
|
|
+
|
|
|
+ int expireEvents = 0;
|
|
|
+ mstore.setEventCallback([&](const smartbotic::database::DatabaseEvent& ev) {
|
|
|
+ if (ev.type == smartbotic::database::EventType::EXPIRE) ++expireEvents;
|
|
|
+ });
|
|
|
+
|
|
|
+ Document doc;
|
|
|
+ doc.id = "s-1";
|
|
|
+ doc.collection = "sessions";
|
|
|
+ doc.set_data({{"token", "abc"}});
|
|
|
+ doc.expiresAt = 1;
|
|
|
+ store.put("sessions", "s-1", doc);
|
|
|
+ mstore.loadDocument("default:sessions", doc);
|
|
|
+
|
|
|
+ // Hook installed, with the relation cache EMPTY - the early-return path.
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager);
|
|
|
+
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
+ check(expired == 1, "an unrelated document still expires");
|
|
|
+ check(!mstore.get("default:sessions", "s-1").has_value(), "gone from MemoryStore");
|
|
|
+ check(!store.get("sessions", "s-1").has_value(),
|
|
|
+ "and the DELETE was mirrored to LMDB, as before");
|
|
|
+ check(expireEvents == 1, "exactly one EXPIRE event, as before");
|
|
|
+ check(mstore.getStats().expiredCount == 1, "counted as expired");
|
|
|
+ check(mstore.getStats().ttlExpiryBlockedByRelation == 0, "nothing blocked");
|
|
|
+
|
|
|
+ // A document whose TTL is in the future is not touched by any of this.
|
|
|
+ Document later;
|
|
|
+ later.id = "s-2";
|
|
|
+ later.collection = "sessions";
|
|
|
+ later.set_data({{"token", "def"}});
|
|
|
+ later.expiresAt = 1;
|
|
|
+ later.expiresAt = static_cast<uint64_t>(1) << 62; // far future
|
|
|
+ mstore.loadDocument("default:sessions", later);
|
|
|
+ check(mstore.expireDocuments() == 0, "an unexpired document is left alone");
|
|
|
+ check(mstore.get("default:sessions", "s-2").has_value(), "and is still there");
|
|
|
+
|
|
|
+ p.pm.stop();
|
|
|
+ mstore.stop();
|
|
|
+}
|
|
|
+
|
|
|
+// THE ORDERING PROPERTY, observed rather than asserted from the outside: the
|
|
|
+// cascade's WAL entries are durable BEFORE the LMDB transaction commits.
|
|
|
+//
|
|
|
+// Induced honestly, with the v2.4.4 identity sentinel: the child sub-db's
|
|
|
+// sentinel is overwritten with the wrong name using raw LMDB, so
|
|
|
+// commitCascadeLmdb()'s open_for_write() throws and its WriteTxn aborts
|
|
|
+// UNWRITTEN. Everything before it has already happened. If the sweeper wrote
|
|
|
+// LMDB first (or hand-rolled its own cascade), the WAL would be empty here and
|
|
|
+// the child's deletion would be lost on the next boot.
|
|
|
+void test_ttl_cascade_wal_is_durable_before_the_lmdb_commit() {
|
|
|
+ TmpEnv t("ttl-wal-first");
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
+ TmpPersistence p("ttl-wal-first-wal");
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
+
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
+ mstore.start();
|
|
|
+ RelationManager rm(mstore);
|
|
|
+ rm.loadFromStore();
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
+
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::Cascade);
|
|
|
+ installTtlHook(mstore, rm, store, p.pm, cfgManager);
|
|
|
+
|
|
|
+ // Overwrite the child sub-db's identity sentinel with a name that is not
|
|
|
+ // its own. open_for_write verifies it on every cached-handle reuse.
|
|
|
+ {
|
|
|
+ MDB_txn* txn = nullptr;
|
|
|
+ check(mdb_txn_begin(t.env.raw(), nullptr, 0, &txn) == 0, "tamper txn opened");
|
|
|
+ MDB_dbi dbi = 0;
|
|
|
+ check(mdb_dbi_open(txn, "executions", 0, &dbi) == 0, "child sub-db opened for tampering");
|
|
|
+ const std::string_view sentinel{"\0__subdb_identity__", 19};
|
|
|
+ MDB_val k{sentinel.size(), const_cast<char*>(sentinel.data())};
|
|
|
+ std::string wrong = "not_executions";
|
|
|
+ MDB_val v{wrong.size(), wrong.data()};
|
|
|
+ check(mdb_put(txn, dbi, &k, &v, 0) == 0, "sentinel overwritten");
|
|
|
+ check(mdb_txn_commit(txn) == 0, "tamper txn committed");
|
|
|
+ }
|
|
|
+
|
|
|
+ const uint64_t expired = mstore.expireDocuments();
|
|
|
+ check(expired == 0, "the sweeper did NOT expire the parent when the cascade faulted");
|
|
|
+ check(residentInMemory(mstore, "default:workflows", "wf-1"),
|
|
|
+ "the parent is still in MemoryStore - never expired behind a failed cascade");
|
|
|
+ check(store.get("workflows", "wf-1").has_value(),
|
|
|
+ "and still in LMDB - commitCascadeLmdb's WriteTxn aborted unwritten");
|
|
|
+ check(store.get("executions", "ex-1").has_value(), "the child is still in LMDB too");
|
|
|
+ check(mstore.getStats().ttlExpiryBlockedByRelation == 1, "the skip is counted");
|
|
|
+
|
|
|
+ // ...and yet the WAL already describes the whole cascade. That is the
|
|
|
+ // ordering: WAL fsynced, THEN the LMDB commit attempted.
|
|
|
+ p.pm.stop();
|
|
|
+ auto entries = replayRawWal(p.path);
|
|
|
+ check(walHasDelete(entries, "default:executions", "ex-1"),
|
|
|
+ "the child's DELETE is already durable in the WAL, before any LMDB commit");
|
|
|
+ check(walHasDelete(entries, "default:workflows", "wf-1"),
|
|
|
+ "so is the parent's - the next boot's replay applies the cascade regardless");
|
|
|
+
|
|
|
+ mstore.stop();
|
|
|
+}
|
|
|
+
|
|
|
+// =========================================================================
|
|
|
+// v2.11.0 close-out — BOOT-PASS DRIFT MUST NOT DISABLE DESTRUCTIVE POLICIES.
|
|
|
+//
|
|
|
+// mirror_drift_count_ is never reset, and the post-replay re-mirror pass bumps
|
|
|
+// it for every row it cannot repair - a condition runPendingRemirror() treats as
|
|
|
+// advisory and which recurs on every boot, since the restart replays the same
|
|
|
+// failing row. Gated on the raw counter, ONE unrepairable row refused every
|
|
|
+// cascade/set_null delete and every CreateRelation for the process's life, with
|
|
|
+// an error message that told the operator to restart.
|
|
|
+// =========================================================================
|
|
|
+void test_boot_pass_drift_does_not_disable_the_cascade() {
|
|
|
+ TmpEnv t("rel-drift-baseline");
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
+ TmpPersistence p("rel-drift-baseline-wal");
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
+
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
+ mstore.start();
|
|
|
+ RelationManager rm(mstore);
|
|
|
+ rm.loadFromStore();
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
+
|
|
|
+ std::atomic<bool> healthy{true};
|
|
|
+ std::atomic<uint64_t> drift{0};
|
|
|
+ mstore.setDocumentStoreMirror(
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
+ &healthy, &drift);
|
|
|
+
|
|
|
+ // The boot path: three rows the re-mirror pass could not write. Health is
|
|
|
+ // deliberately NOT flipped (the v2.8.1 lesson), so drift is the only signal.
|
|
|
+ drift.store(3);
|
|
|
+ check(mstore.mirrorDriftCount() == 3, "the raw counter still reports every stale row");
|
|
|
+ mstore.markMirrorDriftBaseline(); // what initialize() does last
|
|
|
+ check(mstore.mirrorDriftBaseline() == 3, "the boot path's drift is in the baseline");
|
|
|
+ check(mstore.mirrorDriftSinceBaseline() == 0,
|
|
|
+ "and nothing has drifted since READY");
|
|
|
+
|
|
|
+ seedTtlParentAndChild(mstore, store, rm, OnDelete::Cascade);
|
|
|
+
|
|
|
+ bool parentExisted = false;
|
|
|
+ try {
|
|
|
+ parentExisted = executeCascade(rm, store, p.pm, mstore, cfgManager,
|
|
|
+ "default:workflows", "wf-1", nullptr);
|
|
|
+ } catch (const std::exception& e) {
|
|
|
+ check(false, "the cascade was refused because of drift the BOOT PASS accrued");
|
|
|
+ std::cerr << " (" << e.what() << ")\n";
|
|
|
+ }
|
|
|
+ check(parentExisted, "the cascade ran despite 3 rows the boot pass left stale");
|
|
|
+ check(!store.get("executions", "ex-1").has_value(), "and it actually cascaded");
|
|
|
+
|
|
|
+ // A LIVE write's drift still latches and still refuses - that half was
|
|
|
+ // right, and re-basing the gate must not have removed it.
|
|
|
+ drift.fetch_add(1);
|
|
|
+ check(mstore.mirrorDriftSinceBaseline() == 1, "post-READY drift is visible");
|
|
|
+ Document parent2;
|
|
|
+ parent2.id = "wf-2";
|
|
|
+ parent2.collection = "workflows";
|
|
|
+ parent2.set_data({{"name", "wf-2"}});
|
|
|
+ store.put("workflows", "wf-2", parent2);
|
|
|
+ mstore.loadDocument("default:workflows", parent2);
|
|
|
+ Document child2;
|
|
|
+ child2.id = "ex-2";
|
|
|
+ child2.collection = "executions";
|
|
|
+ child2.set_data({{"workflowId", "wf-2"}});
|
|
|
+ store.put("executions", "ex-2", child2);
|
|
|
+ mstore.loadDocument("default:executions", child2);
|
|
|
+
|
|
|
+ bool refused = false;
|
|
|
+ try {
|
|
|
+ executeCascade(rm, store, p.pm, mstore, cfgManager, "default:workflows", "wf-2", nullptr);
|
|
|
+ } catch (const CascadeBlocked&) {
|
|
|
+ refused = true;
|
|
|
+ }
|
|
|
+ check(refused, "drift accrued AFTER READY still refuses the cascade");
|
|
|
+ check(store.get("executions", "ex-2").has_value(), "and nothing was touched");
|
|
|
+
|
|
|
+ p.pm.stop();
|
|
|
+ mstore.stop();
|
|
|
+}
|
|
|
+
|
|
|
+// v2.11.0 close-out — runPendingRemirror() must RELEASE the id list.
|
|
|
+// RecoveryOutcome lives as DatabaseService::recovery_outcome_ for the process
|
|
|
+// lifetime; the pass runs exactly once and nothing reads the list afterwards, so
|
|
|
+// keeping it resident pinned ~100 bytes per replayed row (an install's whole
|
|
|
+// history under --recovery-mode=wal_only) for nothing.
|
|
|
+void test_pending_remirror_list_is_released_after_the_pass() {
|
|
|
+ TmpPersistence p("remirror-release");
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
+ mstore.start();
|
|
|
+
|
|
|
+ smartbotic::database::RecoveryOutcome outcome;
|
|
|
+ for (int i = 0; i < 500; ++i) {
|
|
|
+ outcome.pendingRemirror.emplace_back("default:widgets", "w-" + std::to_string(i));
|
|
|
+ }
|
|
|
+ check(outcome.pendingRemirror.size() == 500, "the list starts populated");
|
|
|
+
|
|
|
+ p.pm.runPendingRemirror(mstore, outcome);
|
|
|
+
|
|
|
+ check(outcome.pendingRemirror.empty(),
|
|
|
+ "the id list is cleared once the pass has run");
|
|
|
+ check(outcome.pendingRemirror.capacity() == 0,
|
|
|
+ "and its CAPACITY is released - clear() alone keeps the whole allocation");
|
|
|
+
|
|
|
+ mstore.stop();
|
|
|
+}
|
|
|
+
|
|
|
int main() {
|
|
|
std::cout << "=== test_relation_enforcement ===\n";
|
|
|
test_index_follows_the_child_field();
|
|
|
@@ -2112,6 +2636,15 @@ int main() {
|
|
|
test_remirror_windows_are_bounded_by_bytes_and_by_count();
|
|
|
test_cascade_refuses_while_the_mirror_is_unhealthy_or_drifted();
|
|
|
test_can_reject_writes_gates_the_undo_snapshot();
|
|
|
+ test_ttl_restrict_blocks_the_expiry();
|
|
|
+ test_ttl_cascade_deletes_children_like_a_manual_delete();
|
|
|
+ test_ttl_set_null_nulls_the_scalar_and_keeps_the_child();
|
|
|
+ test_ttl_array_reference_pulls_the_id_and_keeps_the_document();
|
|
|
+ test_ttl_no_action_expires_and_leaves_the_reference_dangling();
|
|
|
+ test_ttl_with_no_relations_expires_exactly_as_before();
|
|
|
+ test_ttl_cascade_wal_is_durable_before_the_lmdb_commit();
|
|
|
+ test_boot_pass_drift_does_not_disable_the_cascade();
|
|
|
+ test_pending_remirror_list_is_released_after_the_pass();
|
|
|
|
|
|
std::cout << "passed: " << g_pass << ", failed: " << g_fail << "\n";
|
|
|
return g_fail == 0 ? 0 : 1;
|