|
|
@@ -1083,6 +1083,143 @@ void test_replayed_cascade_update_remirrors_to_lmdb_after_crash_window() {
|
|
|
mstore.stop();
|
|
|
}
|
|
|
|
|
|
+// v2.11.0 T12 round-4 — the post-replay re-mirror pass must not be able to
|
|
|
+// stop the service from starting. Two throws are genuinely reachable from it
|
|
|
+// (see MemoryStore::remirrorDocuments): std::invalid_argument from
|
|
|
+// parseProjectCollection() on a malformed/legacy collection key, and a
|
|
|
+// storage fault from the put itself. Unguarded, either escaped recover(),
|
|
|
+// which DatabaseService::initialize() turns into "refuse to start" - a
|
|
|
+// deterministic boot loop on data that booted fine before.
|
|
|
+//
|
|
|
+// This also pins the batching contract: a row that throws inside a chunk
|
|
|
+// aborts that chunk's transaction, and the OTHER rows of that chunk must
|
|
|
+// still converge via the row-by-row retry.
|
|
|
+void test_a_failing_row_does_not_stop_recovery() {
|
|
|
+ TmpEnv t("rel-remirror-fail");
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
+ TmpPersistence p("rel-remirror-fail-wal");
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
+
|
|
|
+ // Row 1: an ordinary, well-formed document. Must converge.
|
|
|
+ Document good; good.id = "w1"; good.collection = "widgets";
|
|
|
+ good.set_data({{"v", 1}});
|
|
|
+ p.pm.logInsert("default:widgets", good);
|
|
|
+ good.set_data({{"v", 2}});
|
|
|
+ p.pm.logUpdate("default:widgets", good);
|
|
|
+
|
|
|
+ // Row 2: same project, so it lands in the SAME chunk as row 1 - but its
|
|
|
+ // id is past LMDB's 511-byte key limit, so the put throws
|
|
|
+ // (MDB_BAD_VALSIZE) from inside that chunk's transaction. This is the
|
|
|
+ // honest way to induce a mid-chunk throw, same technique as
|
|
|
+ // test_aborted_write_does_not_poison_the_collection (v2.8.0).
|
|
|
+ Document oversize; oversize.id = std::string(600, 'k'); oversize.collection = "widgets";
|
|
|
+ oversize.set_data({{"v", 1}});
|
|
|
+ p.pm.logInsert("default:widgets", oversize);
|
|
|
+ p.pm.logUpdate("default:widgets", oversize);
|
|
|
+
|
|
|
+ // Row 3: a malformed collection key. WAL replay itself never parses
|
|
|
+ // collection keys, so this replays fine and only the re-mirror pass
|
|
|
+ // trips on it - previously with an uncaught std::invalid_argument.
|
|
|
+ Document weird; weird.id = "x1"; weird.collection = "legacy";
|
|
|
+ weird.set_data({{"v", 1}});
|
|
|
+ p.pm.logInsert("weird:legacy:key", weird);
|
|
|
+ p.pm.logUpdate("weird:legacy:key", weird);
|
|
|
+ p.pm.stop();
|
|
|
+
|
|
|
+ MemoryStore freshStore(MemoryStore::Config{});
|
|
|
+ freshStore.start();
|
|
|
+ std::atomic<bool> mirrorHealthy{true};
|
|
|
+ std::atomic<uint64_t> mirrorDrift{0};
|
|
|
+ freshStore.setDocumentStoreMirror(
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
+ &mirrorHealthy, &mirrorDrift);
|
|
|
+
|
|
|
+ PersistenceManager::Config cfg2;
|
|
|
+ cfg2.dataDir = p.path;
|
|
|
+ PersistenceManager pm2(cfg2);
|
|
|
+ auto outcome = pm2.recover(freshStore);
|
|
|
+
|
|
|
+ // THE finding: recovery completes.
|
|
|
+ check(outcome.kind != smartbotic::database::RecoveryOutcome::Kind::Failed,
|
|
|
+ "recovery completed despite two un-mirrorable rows - no boot loop");
|
|
|
+ check(outcome.walEntriesReplayed == 6,
|
|
|
+ "all six WAL entries replayed - the failures did not truncate replay");
|
|
|
+
|
|
|
+ // The failures are counted and reported, not swallowed silently.
|
|
|
+ check(outcome.updatesRemirrorFailed == 2,
|
|
|
+ "both un-mirrorable rows were counted as failed");
|
|
|
+ check(mirrorDrift.load() == 2,
|
|
|
+ "each failed row bumped mirror drift (the operator-visible signal)");
|
|
|
+ check(mirrorHealthy.load(),
|
|
|
+ "mirror health NOT flipped - one legacy row must not send every read "
|
|
|
+ "in the process to MemoryStore for the rest of its life");
|
|
|
+
|
|
|
+ // The other row in the same chunk still converged.
|
|
|
+ check(outcome.updatesRemirroredAfterReplay == 1, "the good row was re-mirrored");
|
|
|
+ auto lmdbGood = store.get("widgets", "w1");
|
|
|
+ check(lmdbGood.has_value(), "the good row reached LMDB despite sharing a chunk with a bad one");
|
|
|
+ if (lmdbGood) {
|
|
|
+ check(lmdbGood->data()["v"] == 2, "and it carries the REPLAYED update, not the insert");
|
|
|
+ }
|
|
|
+
|
|
|
+ // MemoryStore replay itself was unaffected for every row, including the
|
|
|
+ // ones LMDB could not take.
|
|
|
+ check(freshStore.get("default:widgets", "w1").has_value(), "w1 in MemoryStore");
|
|
|
+ check(freshStore.get("weird:legacy:key", "x1").has_value(),
|
|
|
+ "the malformed-key row still recovered into MemoryStore - it is only "
|
|
|
+ "LMDB that cannot hold it");
|
|
|
+
|
|
|
+ freshStore.stop();
|
|
|
+}
|
|
|
+
|
|
|
+// v2.11.0 T12 round-4 (point 3) — INSERT must be tracked by the re-mirror
|
|
|
+// pass too. loadDocument() (INSERT replay) does not mirror, so a
|
|
|
+// delete-then-reinsert of the same id inside one replay window used to leave
|
|
|
+// LMDB with the row DELETED: the DELETE mirrored via remove(), the reinsert
|
|
|
+// did not.
|
|
|
+void test_reinsert_after_delete_converges_in_lmdb() {
|
|
|
+ TmpEnv t("rel-remirror-reinsert");
|
|
|
+ LmdbDocumentStore store(t.env);
|
|
|
+ TmpPersistence p("rel-remirror-reinsert-wal");
|
|
|
+ check(p.pm.start(), "persistence manager started");
|
|
|
+
|
|
|
+ Document d; d.id = "r1"; d.collection = "widgets";
|
|
|
+ d.set_data({{"v", 1}});
|
|
|
+ store.put("widgets", "r1", d); // LMDB starts in the pre-window state
|
|
|
+ p.pm.logInsert("default:widgets", d);
|
|
|
+ p.pm.logDelete("default:widgets", "r1");
|
|
|
+ d.set_data({{"v", 99}}); // reinserted with new content
|
|
|
+ p.pm.logInsert("default:widgets", d);
|
|
|
+ p.pm.stop();
|
|
|
+
|
|
|
+ MemoryStore freshStore(MemoryStore::Config{});
|
|
|
+ freshStore.start();
|
|
|
+ std::atomic<bool> mirrorHealthy{true};
|
|
|
+ std::atomic<uint64_t> mirrorDrift{0};
|
|
|
+ freshStore.setDocumentStoreMirror(
|
|
|
+ [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
|
|
|
+ &mirrorHealthy, &mirrorDrift);
|
|
|
+
|
|
|
+ PersistenceManager::Config cfg2;
|
|
|
+ cfg2.dataDir = p.path;
|
|
|
+ PersistenceManager pm2(cfg2);
|
|
|
+ auto outcome = pm2.recover(freshStore);
|
|
|
+ check(outcome.kind != smartbotic::database::RecoveryOutcome::Kind::Failed, "recovery completed");
|
|
|
+ check(outcome.updatesRemirroredAfterReplay == 1,
|
|
|
+ "the reinserted id was re-mirrored (INSERT is tracked, not just UPDATE/UPSERT)");
|
|
|
+
|
|
|
+ auto mem = freshStore.get("default:widgets", "r1");
|
|
|
+ check(mem.has_value(), "reinserted row present in MemoryStore");
|
|
|
+
|
|
|
+ // LOAD-BEARING: without INSERT tracking the DELETE's own mirror wins and
|
|
|
+ // LMDB has nothing here at all.
|
|
|
+ auto lmdb = store.get("widgets", "r1");
|
|
|
+ check(lmdb.has_value(), "reinserted row present in LMDB - the DELETE's mirror did not win");
|
|
|
+ if (lmdb) check(lmdb->data()["v"] == 99, "and it is the REINSERTED content, not the original");
|
|
|
+
|
|
|
+ freshStore.stop();
|
|
|
+}
|
|
|
+
|
|
|
} // namespace
|
|
|
|
|
|
int main() {
|
|
|
@@ -1102,6 +1239,8 @@ int main() {
|
|
|
test_crash_between_wal_and_commit_recovers();
|
|
|
test_cascade_refuses_when_grandchild_is_restrict_protected();
|
|
|
test_replayed_cascade_update_remirrors_to_lmdb_after_crash_window();
|
|
|
+ test_a_failing_row_does_not_stop_recovery();
|
|
|
+ test_reinsert_after_delete_converges_in_lmdb();
|
|
|
|
|
|
std::cout << "passed: " << g_pass << ", failed: " << g_fail << "\n";
|
|
|
return g_fail == 0 ? 0 : 1;
|