|
@@ -959,6 +959,130 @@ void test_cascade_refuses_when_grandchild_is_restrict_protected() {
|
|
|
mstore.stop();
|
|
mstore.stop();
|
|
|
}
|
|
}
|
|
|
|
|
|
|
|
|
|
+// v2.11.0 T12 review round 3 (I1) - a crash between writeCascadeWal()'s
|
|
|
|
|
+// fsync and commitCascadeLmdb() leaves an UpdateChild mutation's WAL UPDATE
|
|
|
|
|
+// entry durable while its LMDB write never ran. Unlike DeleteChild (which
|
|
|
|
|
+// self-heals on replay because DELETE replay goes through
|
|
|
|
|
+// MemoryStore::remove(), which mirrors), UPDATE/UPSERT replay goes through
|
|
|
|
|
+// loadDocumentWithHistory(), which never mirrored - so before this fix,
|
|
|
|
|
+// LMDB would permanently keep serving the pre-cascade array/field. This
|
|
|
|
|
+// test drives the REAL cascade path (planCascade + writeCascadeWal against
|
|
|
|
|
+// an array reference, so the mutation really is an UpdateChild - not the
|
|
|
|
|
+// isolated primitive) through exactly that crash window, "restarts" with
|
|
|
|
|
+// the LMDB mirror wired the same way DatabaseService wires it in production
|
|
|
|
|
+// (setupComponents() wires the mirror BEFORE persistence_->recover() runs -
|
|
|
|
|
+// confirmed by reading database_service.cpp's initialize()), and asserts
|
|
|
|
|
+// LMDB converges, not just MemoryStore.
|
|
|
|
|
+void test_replayed_cascade_update_remirrors_to_lmdb_after_crash_window() {
|
|
|
|
|
+ MemoryStore mstore(MemoryStore::Config{});
|
|
|
|
|
+ mstore.start();
|
|
|
|
|
+ RelationManager rm(mstore);
|
|
|
|
|
+ rm.loadFromStore();
|
|
|
|
|
+ CollectionConfigManager cfgManager(mstore);
|
|
|
|
|
+ cfgManager.loadFromStore();
|
|
|
|
|
+
|
|
|
|
|
+ TmpEnv t("rel-remirror");
|
|
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
|
|
+ store.set_relations("nodes", {{"node_creds", "config.credentialIds"}});
|
|
|
|
|
+
|
|
|
|
|
+ TmpPersistence p("rel-remirror-wal");
|
|
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
|
|
+
|
|
|
|
|
+ // Seed through the REAL WAL (logInsert), same discipline as the C1 fix -
|
|
|
|
|
+ // these assertions must depend on genuine WAL replay, not just on
|
|
|
|
|
+ // store.put()/whatever MemoryStore starts with.
|
|
|
|
|
+ Document node; node.id = "n1"; node.collection = "nodes";
|
|
|
|
|
+ node.set_data({{"config", {{"credentialIds", {"c1", "c2", "c3"}}}}});
|
|
|
|
|
+ store.put("nodes", "n1", node);
|
|
|
|
|
+ p.pm.logInsert("default:nodes", node);
|
|
|
|
|
+
|
|
|
|
|
+ Document parent; parent.id = "c1"; parent.collection = "credentials";
|
|
|
|
|
+ parent.set_data({{"name", "prod-key"}});
|
|
|
|
|
+ store.put("credentials", "c1", parent);
|
|
|
|
|
+ p.pm.logInsert("default:credentials", parent);
|
|
|
|
|
+
|
|
|
|
|
+ RelationInfo rel;
|
|
|
|
|
+ rel.name = "default:node_creds";
|
|
|
|
|
+ rel.child = "default:nodes";
|
|
|
|
|
+ rel.childField = "config.credentialIds";
|
|
|
|
|
+ rel.parent = "default:credentials";
|
|
|
|
|
+ rel.onDelete = OnDelete::Cascade; // array reference -> UpdateChild (pull), not DeleteChild
|
|
|
|
|
+ std::string mgrErr;
|
|
|
|
|
+ check(rm.createRelation(rel, mgrErr), "declared the cascade relation on an array field");
|
|
|
|
|
+
|
|
|
|
|
+ // Steps 1+3 only - simulating a crash immediately after flushWal()
|
|
|
|
|
+ // returns, before commitCascadeLmdb()/applyCascadeToMemory() ever run.
|
|
|
|
|
+ CascadePlan plan = planCascade(rm, store, cfgManager, "default:credentials", "c1");
|
|
|
|
|
+ check(plan.mutations.size() == 1, "planned the one array-pull mutation");
|
|
|
|
|
+ check(plan.mutations[0].kind == CascadeMutation::Kind::UpdateChild,
|
|
|
|
|
+ "confirmed UpdateChild - this is exactly the case DeleteChild does NOT cover");
|
|
|
|
|
+ writeCascadeWal(p.pm, "default:credentials", "c1", plan);
|
|
|
|
|
+ p.pm.stop(); // closes the WAL file, like a process exiting
|
|
|
|
|
+
|
|
|
|
|
+ // Proof the "crash" really happened: LMDB was never touched by this
|
|
|
|
|
+ // cascade attempt - the node still has all three ids.
|
|
|
|
|
+ auto beforeRecovery = store.get("nodes", "n1");
|
|
|
|
|
+ check(beforeRecovery.has_value(), "n1 still in LMDB");
|
|
|
|
|
+ if (beforeRecovery) {
|
|
|
|
|
+ auto ids = beforeRecovery->data()["config"]["credentialIds"];
|
|
|
|
|
+ check(ids.is_array() && ids.size() == 3,
|
|
|
|
|
+ "LMDB was never committed - still the PRE-cascade array (this is the point)");
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // "Restart": fresh MemoryStore with the LMDB mirror wired to the SAME
|
|
|
|
|
+ // LmdbDocumentStore (matching production ordering - the mirror is wired
|
|
|
|
|
+ // in DatabaseService::setupComponents(), which runs BEFORE
|
|
|
|
|
+ // persistence_->recover() in initialize()), fresh PersistenceManager
|
|
|
|
|
+ // over the same dataDir, recover().
|
|
|
|
|
+ MemoryStore freshStore(MemoryStore::Config{});
|
|
|
|
|
+ freshStore.start();
|
|
|
|
|
+ std::atomic<bool> mirrorHealthy{true};
|
|
|
|
|
+ std::atomic<uint64_t> mirrorDrift{0};
|
|
|
|
|
+ freshStore.setDocumentStoreMirror(
|
|
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
|
|
+ &mirrorHealthy, &mirrorDrift);
|
|
|
|
|
+
|
|
|
|
|
+ PersistenceManager::Config cfg2;
|
|
|
|
|
+ cfg2.dataDir = p.path;
|
|
|
|
|
+ PersistenceManager pm2(cfg2);
|
|
|
|
|
+ auto outcome = pm2.recover(freshStore);
|
|
|
|
|
+ check(outcome.kind != smartbotic::database::RecoveryOutcome::Kind::Failed,
|
|
|
|
|
+ "recovery did not fail");
|
|
|
|
|
+ check(outcome.updatesRemirroredAfterReplay >= 1,
|
|
|
|
|
+ "recover() re-mirrored at least the node's replayed UPDATE");
|
|
|
|
|
+
|
|
|
|
|
+ // MemoryStore converges (this part already worked before this fix).
|
|
|
|
|
+ auto memNode = freshStore.get("default:nodes", "n1");
|
|
|
|
|
+ check(memNode.has_value(), "n1 present in the recovered MemoryStore");
|
|
|
|
|
+ if (memNode) {
|
|
|
|
|
+ auto ids = memNode->data()["config"]["credentialIds"];
|
|
|
|
|
+ check(ids.is_array() && ids.size() == 2, "MemoryStore has the post-cascade array");
|
|
|
|
|
+ }
|
|
|
|
|
+ check(!freshStore.get("default:credentials", "c1").has_value(),
|
|
|
|
|
+ "parent gone from the recovered MemoryStore (DELETE replay already worked pre-fix)");
|
|
|
|
|
+
|
|
|
|
|
+ // LOAD-BEARING: LMDB now ALSO reflects the post-cascade array - only
|
|
|
|
|
+ // true because recover()'s remirror pass ran. Without the I1 fix this
|
|
|
|
|
+ // would still show 3 ids, exactly like `beforeRecovery` above.
|
|
|
|
|
+ auto lmdbAfter = store.get("nodes", "n1");
|
|
|
|
|
+ check(lmdbAfter.has_value(), "n1 still in LMDB after recovery");
|
|
|
|
|
+ if (lmdbAfter) {
|
|
|
|
|
+ auto ids = lmdbAfter->data()["config"]["credentialIds"];
|
|
|
|
|
+ check(ids.is_array() && ids.size() == 2,
|
|
|
|
|
+ "LMDB converged to the post-cascade array - the I1 fix");
|
|
|
|
|
+ check(std::find(ids.begin(), ids.end(), nlohmann::json("c1")) == ids.end(),
|
|
|
|
|
+ "c1 is gone from LMDB too, not just MemoryStore");
|
|
|
|
|
+ }
|
|
|
|
|
+ // The parent delete self-healed on its own even before this fix
|
|
|
|
|
+ // (DELETE replay mirrors via MemoryStore::remove()) - confirmed here so
|
|
|
|
|
+ // the test pins the WHOLE combined scenario, not just the new half.
|
|
|
|
|
+ check(!store.get("credentials", "c1").has_value(),
|
|
|
|
|
+ "parent also gone from LMDB after recovery");
|
|
|
|
|
+
|
|
|
|
|
+ freshStore.stop();
|
|
|
|
|
+ mstore.stop();
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
} // namespace
|
|
} // namespace
|
|
|
|
|
|
|
|
int main() {
|
|
int main() {
|
|
@@ -977,6 +1101,7 @@ int main() {
|
|
|
test_array_reference_pulls_id_and_keeps_document();
|
|
test_array_reference_pulls_id_and_keeps_document();
|
|
|
test_crash_between_wal_and_commit_recovers();
|
|
test_crash_between_wal_and_commit_recovers();
|
|
|
test_cascade_refuses_when_grandchild_is_restrict_protected();
|
|
test_cascade_refuses_when_grandchild_is_restrict_protected();
|
|
|
|
|
+ test_replayed_cascade_update_remirrors_to_lmdb_after_crash_window();
|
|
|
|
|
|
|
|
std::cout << "passed: " << g_pass << ", failed: " << g_fail << "\n";
|
|
std::cout << "passed: " << g_pass << ", failed: " << g_fail << "\n";
|
|
|
return g_fail == 0 ? 0 : 1;
|
|
return g_fail == 0 ? 0 : 1;
|