소스 검색

Merge relations + unique constraints (v2.11.0)

Referential integrity: relations declared in _relations, a DUPSORT reverse index
maintained inside the document's own transaction, restrict / cascade / set_null /
no_action with WAL-before-LMDB sequencing, DescribeDelete, a bootstrap scan so
declaring a relation on populated data actually protects it, relations check,
per-collection enforcement switches, boot re-arm with self-heal, and
validate_on_write closing the restrict race inside the child's transaction.

TTL expiry runs the same on_delete policy as a manual delete (MySQL-style):
restrict blocks the expiry so the document outlives its TTL, visibly; cascade and
set_null go through the real cascade path.

Unique constraints are enforced. They existed since v2.10.0 but were unreachable,
because a violation thrown from put() was swallowed by applyDualWriteMirror and
also flipped mirror health, sending every read in the process to MemoryStore.

Fixed on the way, and not about relations: the v2.10.0 post-replay re-mirror pass
ran before index arming, so re-mirrored rows got no index maintenance - the
document moved forward in LMDB while its postings did not, producing silently
wrong rows on any install with a v2.9 index.

31 commits, each reviewed per task, then a whole-branch review, a fix wave and a
close-out. ctest 23/23.
fszontagh 1 개월 전
부모
커밋
54a813218a
52개의 변경된 파일과 12891개의 추가작업 그리고 285개의 파일을 삭제
  1. 1 0
      .gitignore
  2. 7 1
      CLAUDE.md
  3. 1 1
      VERSION
  4. 256 8
      cli/main.cpp
  5. 278 0
      client/include/smartbotic/database/client.hpp
  6. 341 3
      client/src/client.cpp
  7. 28 31
      docs/ROADMAP.md
  8. 215 0
      proto/database.proto
  9. 4 0
      service/CMakeLists.txt
  10. 28 7
      service/src/config/collection_config_manager.cpp
  11. 55 0
      service/src/config/collection_config_manager.hpp
  12. 888 29
      service/src/database_grpc_impl.cpp
  13. 93 1
      service/src/database_grpc_impl.hpp
  14. 491 65
      service/src/database_service.cpp
  15. 78 0
      service/src/database_service.hpp
  16. 870 47
      service/src/memory_store.cpp
  17. 490 1
      service/src/memory_store.hpp
  18. 57 1
      service/src/migrations/migration_runner.cpp
  19. 34 1
      service/src/migrations/migration_runner.hpp
  20. 114 1
      service/src/persistence/persistence_manager.cpp
  21. 90 0
      service/src/persistence/persistence_manager.hpp
  22. 21 4
      service/src/persistence/wal.cpp
  23. 489 0
      service/src/relations/relation_cascade.cpp
  24. 435 0
      service/src/relations/relation_cascade.hpp
  25. 163 0
      service/src/relations/relation_enforcement.cpp
  26. 172 0
      service/src/relations/relation_enforcement.hpp
  27. 20 0
      service/src/relations/relation_index.cpp
  28. 50 0
      service/src/relations/relation_index.hpp
  29. 357 0
      service/src/relations/relation_manager.cpp
  30. 160 0
      service/src/relations/relation_manager.hpp
  31. 48 0
      service/src/storage/document_store.hpp
  32. 677 40
      service/src/storage/document_store_lmdb.cpp
  33. 304 16
      service/src/storage/document_store_lmdb.hpp
  34. 22 0
      service/src/storage/dual_write_mirror.hpp
  35. 147 0
      tests/CMakeLists.txt
  36. 93 0
      tests/load_test/relidx_drop_tool.cpp
  37. 12 2
      tests/load_test/test_client_namespacing.sh
  38. 75 0
      tests/load_test/test_policy_enforcement.cpp
  39. 12 2
      tests/load_test/test_policy_enforcement.sh
  40. 268 0
      tests/load_test/test_relations.cpp
  41. 333 0
      tests/load_test/test_relations.sh
  42. 140 0
      tests/load_test/test_relations_client_e2e.cpp
  43. 76 0
      tests/load_test/test_relations_client_e2e.sh
  44. 12 2
      tests/load_test/test_v24_tls_auth.sh
  45. 12 2
      tests/load_test/test_views_multiproject.sh
  46. 101 0
      tests/test_dual_write_mirror.cpp
  47. 111 19
      tests/test_eviction.cpp
  48. 2936 0
      tests/test_relation_enforcement.cpp
  49. 341 0
      tests/test_relation_index.cpp
  50. 545 0
      tests/test_relation_manager.cpp
  51. 267 0
      tests/test_subdb_identity.cpp
  52. 73 1
      tests/test_timestamp_precision.cpp

+ 1 - 0
.gitignore

@@ -28,3 +28,4 @@ core.*
 
 .claude/worktrees/
 .playwright-mcp/
+.superpowers/

파일 크기가 너무 크기때문에 변경 상태를 표시하지 않습니다.
+ 7 - 1
CLAUDE.md


+ 1 - 1
VERSION

@@ -1 +1 @@
-2.10.0
+2.11.0

+ 256 - 8
cli/main.cpp

@@ -18,6 +18,7 @@
 #include <sys/stat.h>
 #include <unistd.h>
 
+#include <algorithm>
 #include <cerrno>
 #include <cstdio>
 #include <cstring>
@@ -33,6 +34,10 @@ namespace {
 
 struct Args {
     std::string address = "localhost:9004";
+    // v2.11.0 T6b — needed to exercise relation management against a
+    // non-default project from the CLI. Empty means the client's own
+    // "default" (unchanged behaviour for every existing command).
+    std::string project;
     std::string command;
     std::vector<std::string> params;
 };
@@ -80,6 +85,20 @@ void printUsage() {
               << "          remove an index\n"
               << "  " << C_CYAN << "index-values" << C_RESET << " <coll> <field> [n] [asc|desc]"
               << "  distinct values + counts\n"
+              << C_BOLD << "  Relations / referential integrity (v2.11.0+)" << C_RESET << "\n"
+              << "  " << C_CYAN << "relations" << C_RESET
+              << "                             List declared relations\n"
+              << "  " << C_CYAN << "relation" << C_RESET << " <name>"
+              << "                    Show one relation's declaration\n"
+              << "  " << C_CYAN << "relation-create" << C_RESET
+              << " <name> <child> <child_field> <parent> [on_delete] [validate_on_write]\n"
+              << "  " << C_CYAN << "relation-drop" << C_RESET << " <name>\n"
+              << "  " << C_CYAN << "relation-check" << C_RESET << " <name>"
+              << "                    Report dangling references (read-only)\n"
+              << "  " << C_CYAN << "configure-relations" << C_RESET
+              << " <collection> <on|off>   Enable/disable enforcement for a collection\n"
+              << "  " << C_CYAN << "describe-delete" << C_RESET << " <collection> <id>"
+              << "        What would happen if this document were deleted\n"
               << "  " << C_CYAN << "security-set" << C_RESET << " <project> <on|off> [enforce|audit]\n"
               << "  " << C_CYAN << "policies" << C_RESET << " [project]                   List principals with a policy\n"
               << "  " << C_CYAN << "policy" << C_RESET << " <project> <principal>       Show one policy\n"
@@ -95,7 +114,8 @@ void printUsage() {
               << "  " << C_CYAN << "reconcile-subdbs" << C_RESET << " --env <path>       Repair misfiled documents (dry run by default)\n"
               << "    " << C_DIM << "[--project NAME] [--apply]   STOP THE SERVICE and back up before --apply" << C_RESET << "\n\n"
               << C_BOLD << "Options:" << C_RESET << "\n"
-              << "  --address HOST:PORT    Database address (default: localhost:9004)\n";
+              << "  --address HOST:PORT    Database address (default: localhost:9004)\n"
+              << "  --project NAME         Operate as this project namespace (default: default)\n";
 }
 
 // ---------------------------------------------------------------------------
@@ -545,7 +565,25 @@ bool execCommand(smartbotic::database::Client& client,
 
         if (cmd == "remove" || cmd == "delete") {
             if (params.size() < 2) { printError("usage: remove <collection> <id>"); return false; }
-            client.remove(params[0], params[1]);
+            // v2.11.0 T6b — the return value used to be discarded and "ok
+            // removed" printed unconditionally. That was merely sloppy
+            // before referential integrity: now a delete can be REFUSED (a
+            // relation with on_delete=restrict/no_action still has children
+            // referencing it), and the server reports that as a real error,
+            // not deleted=false. Report the actual outcome, and surface the
+            // server's message on refusal so an operator learns what
+            // blocked them rather than being told it worked.
+            std::string err;
+            bool deleted = client.remove(params[0], params[1], err);
+            if (!err.empty()) {
+                printError("remove " + params[0] + "/" + params[1] + ": " + err);
+                return false;
+            }
+            if (!deleted) {
+                std::cout << C_YELLOW << "not found" << C_RESET << " "
+                          << params[0] << "/" << params[1] << "\n";
+                return true;
+            }
             std::cout << C_GREEN << "ok" << C_RESET << " removed " << params[0] << "/" << params[1] << "\n";
             return true;
         }
@@ -568,21 +606,28 @@ bool execCommand(smartbotic::database::Client& client,
         // an operator can see that coming.
         if (cmd == "indexes") {
             if (params.empty()) { printError("usage: indexes <collection>"); return false; }
-            auto list = client.listIndexes(params[0]);
+            std::vector<bool> uniqueFlags;
+            auto list = client.listIndexes(params[0], uniqueFlags);
             if (list.empty()) {
                 std::cout << "no indexes on " << params[0] << "\n";
                 return true;
             }
-            std::cout << C_BOLD << "field                          distinct      entries"
+            std::cout << C_BOLD << "field                          distinct      entries  unique"
                       << C_RESET << "\n";
-            for (const auto& i : list) {
+            for (size_t idx = 0; idx < list.size(); ++idx) {
+                const auto& i = list[idx];
+                const bool unique = idx < uniqueFlags.size() && uniqueFlags[idx];
                 std::cout << "  " << i.field
                           << std::string(i.field.size() < 29 ? 29 - i.field.size() : 1, ' ')
                           << i.distinctValues
                           << std::string(std::to_string(i.distinctValues).size() < 13
                                              ? 13 - std::to_string(i.distinctValues).size()
                                              : 1, ' ')
-                          << i.entries << "\n";
+                          << i.entries
+                          << std::string(std::to_string(i.entries).size() < 9
+                                             ? 9 - std::to_string(i.entries).size()
+                                             : 1, ' ')
+                          << (unique ? "yes" : "no") << "\n";
             }
             return true;
         }
@@ -611,9 +656,31 @@ bool execCommand(smartbotic::database::Client& client,
 
         if (cmd == "index-create") {
             if (params.size() < 2) {
-                printError("usage: index-create <collection> <field>");
+                printError("usage: index-create <collection> <field> [--unique]");
                 return false;
             }
+            const bool unique = std::find(params.begin(), params.end(), "--unique") != params.end();
+            if (unique) {
+                // v2.11.0 T11 — a duplicate-refusal is not a generic failure:
+                // surface the examples so the operator can go fix the data
+                // rather than guess what "could not create the index" meant.
+                auto result = client.createUniqueIndex(params[0], params[1]);
+                if (!result.success) {
+                    printError(result.error);
+                    if (!result.duplicateExamples.empty()) {
+                        std::cout << "  colliding document ids: ";
+                        for (size_t i = 0; i < result.duplicateExamples.size(); ++i) {
+                            if (i) std::cout << ", ";
+                            std::cout << result.duplicateExamples[i];
+                        }
+                        std::cout << "\n";
+                    }
+                    return false;
+                }
+                std::cout << "indexed " << result.rowsIndexed << " existing row(s) on "
+                          << params[0] << "#" << params[1] << " (unique)\n";
+                return true;
+            }
             uint64_t rows = 0;
             if (!client.createIndex(params[0], params[1], rows)) {
                 printError("could not create the index (see the service log)");
@@ -637,6 +704,183 @@ bool execCommand(smartbotic::database::Client& client,
             return true;
         }
 
+        // ===== v2.11.0 T6b — relations (referential integrity) =====
+        //
+        // Declaration is admin-only: `_relations` is a system collection and a
+        // relation names another collection's schema, which is not ordinary
+        // per-collection write access. Follows the `indexes` output shape.
+        if (cmd == "relations") {
+            auto list = client.listRelations();
+            if (list.empty()) {
+                std::cout << "no relations declared\n";
+                return true;
+            }
+            std::cout << C_BOLD << "name                 child.field -> parent                    on_delete    enforced-at-write"
+                      << C_RESET << "\n";
+            for (const auto& r : list) {
+                std::cout << "  " << C_CYAN << r.name << C_RESET
+                          << std::string(r.name.size() < 19 ? 19 - r.name.size() : 1, ' ')
+                          << r.child << "." << r.childField << " -> " << r.parent
+                          << "  " << r.onDelete
+                          << (r.validateOnWrite ? "  (validate_on_write)" : "") << "\n";
+            }
+            return true;
+        }
+
+        if (cmd == "relation") {
+            if (params.empty()) { printError("usage: relation <name>"); return false; }
+            auto r = client.getRelationInfo(params[0]);
+            if (!r) { printError("relation not found: " + params[0]); return false; }
+            std::cout << C_BOLD << r->name << C_RESET << ":\n"
+                      << "  child:              " << r->child << "\n"
+                      << "  child_field:        " << r->childField << "\n"
+                      << "  parent:             " << r->parent << "\n"
+                      << "  on_delete:          " << r->onDelete << "\n"
+                      << "  validate_on_write:  " << (r->validateOnWrite ? "yes" : "no") << "\n";
+            return true;
+        }
+
+        if (cmd == "relation-create") {
+            if (params.size() < 4) {
+                printError("usage: relation-create <name> <child> <child_field> <parent> "
+                           "[on_delete] [validate_on_write]\n"
+                           "  on_delete: restrict (default) | cascade | set_null | no_action\n"
+                           "  note (v2.11.0+): cascade deletes the referencing document "
+                           "(scalar reference) or pulls the id from the array and keeps the "
+                           "document (array reference - cascade and set_null are the same "
+                           "for arrays); set_null nulls the scalar field. no_action permits "
+                           "the delete and leaves the reference dangling.");
+                return false;
+            }
+            const std::string onDelete = params.size() > 4 ? params[4] : "restrict";
+            const bool validateOnWrite = params.size() > 5
+                && (params[5] == "true" || params[5] == "1" || params[5] == "yes");
+            // v2.11.0 T7 - the server backfills the reverse index over rows
+            // already in the child collection as part of this call, so
+            // rowsIndexed reports coverage the same way index-create does.
+            uint64_t rowsIndexed = 0;
+            if (!client.createRelation(params[0], params[1], params[2], params[3],
+                                       onDelete, validateOnWrite, rowsIndexed)) {
+                printError("could not create the relation (see the service log)");
+                return false;
+            }
+            std::cout << "declared relation " << params[0] << " (" << params[1] << "."
+                      << params[2] << " -> " << params[3] << ", on_delete=" << onDelete << ")\n"
+                      << "indexed " << rowsIndexed << " existing row(s) in " << params[1] << "\n";
+            return true;
+        }
+
+        if (cmd == "relation-drop") {
+            if (params.empty()) { printError("usage: relation-drop <name>"); return false; }
+            if (!client.dropRelation(params[0])) {
+                printError("could not drop the relation (see the service log)");
+                return false;
+            }
+            std::cout << "dropped relation " << params[0] << "\n";
+            return true;
+        }
+
+        // v2.11.0 T7 - "does this relation's reverse index actually match
+        // live data?" Read-only, changes nothing. Walks the whole reverse
+        // index (cost is proportional to distinct parents referenced, not to
+        // the child collection's size) so this is a migration/operator tool,
+        // not something to run in a loop.
+        if (cmd == "relation-check") {
+            if (params.empty()) { printError("usage: relation-check <name>"); return false; }
+            auto result = client.checkRelation(params[0]);
+            if (!result.success) {
+                printError("could not check the relation: " + result.error);
+                return false;
+            }
+            if (result.totalDangling == 0) {
+                std::cout << C_GREEN << "clean" << C_RESET << " - no dangling references for "
+                          << params[0] << "\n";
+                return true;
+            }
+            std::cout << C_RED << result.totalDangling << " dangling reference(s)" << C_RESET
+                      << " for " << params[0] << ":\n";
+            for (const auto& d : result.dangling) {
+                std::cout << "  parent " << d.parentId << " does not exist, referenced by "
+                          << d.childCount << " child document(s)";
+                if (!d.sampleChildIds.empty()) {
+                    std::cout << " (e.g. ";
+                    for (size_t i = 0; i < d.sampleChildIds.size(); ++i) {
+                        if (i) std::cout << ", ";
+                        std::cout << d.sampleChildIds[i];
+                    }
+                    std::cout << ")";
+                }
+                std::cout << "\n";
+            }
+            if (result.dangling.size() < result.totalDangling) {
+                std::cout << "  ... " << (result.totalDangling - result.dangling.size())
+                          << " more not shown\n";
+            }
+            return true;
+        }
+
+        // v2.11.0 T8 — the only reachable path to relations_enforced besides
+        // grpcurl. A partial update: touches only this one knob.
+        if (cmd == "configure-relations") {
+            if (params.size() < 2) {
+                printError("usage: configure-relations <collection> <on|off>");
+                return false;
+            }
+            if (params[1] != "on" && params[1] != "off") {
+                printError("expected 'on' or 'off', got: " + params[1]);
+                return false;
+            }
+            const bool enforced = params[1] == "on";
+            if (!client.setRelationsEnforced(params[0], enforced)) {
+                printError("could not update relations_enforced (see the service log)");
+                return false;
+            }
+            std::cout << C_GREEN << "ok" << C_RESET << " " << params[0]
+                      << " relations_enforced=" << (enforced ? "on" : "off") << "\n";
+            return true;
+        }
+
+        // v2.11.0 T5 — DescribeDelete: "what would happen if I deleted this?"
+        // without deleting anything. A per-collection READ, not admin - any
+        // caller who can read the collection can ask this. Cheap enough to
+        // run before every delete: counts come from the reverse index.
+        if (cmd == "describe-delete") {
+            if (params.size() < 2) {
+                printError("usage: describe-delete <collection> <id>");
+                return false;
+            }
+            auto d = client.describeDelete(params[0], params[1]);
+            if (!d.success) {
+                printError("could not describe the delete: " + d.error);
+                return false;
+            }
+            if (d.impacts.empty()) {
+                std::cout << "no relations reference " << params[0] << "/" << params[1]
+                          << " - safe to delete\n";
+                return true;
+            }
+            std::cout << (d.wouldBeBlocked
+                              ? (C_RED + std::string("would be BLOCKED") + C_RESET)
+                              : (C_GREEN + std::string("would proceed") + C_RESET))
+                      << " deleting " << params[0] << "/" << params[1] << ":\n";
+            for (const auto& imp : d.impacts) {
+                std::cout << "  " << (imp.blocks ? C_RED : C_DIM) << "[" << imp.onDelete << "]"
+                          << C_RESET << " " << imp.relation << ": " << imp.childCount
+                          << " child document(s) in " << imp.childCollection
+                          << " via " << imp.childField;
+                if (!imp.sampleChildIds.empty()) {
+                    std::cout << " (e.g. ";
+                    for (size_t i = 0; i < imp.sampleChildIds.size(); ++i) {
+                        if (i) std::cout << ", ";
+                        std::cout << imp.sampleChildIds[i];
+                    }
+                    std::cout << ")";
+                }
+                std::cout << (imp.blocks ? "  BLOCKS" : "") << "\n";
+            }
+            return true;
+        }
+
         // ===== v2.7.0 access policy =====
         //
         // Policy lives in the `_policies` collection and is managed through the
@@ -957,6 +1201,8 @@ int main(int argc, char* argv[]) {
         std::string arg = argv[i];
         if (arg == "--address" && i + 1 < argc) {
             args.address = argv[++i];
+        } else if (arg == "--project" && i + 1 < argc) {
+            args.project = argv[++i];
         } else if (arg == "--help" || arg == "-h") {
             printUsage();
             return 0;
@@ -983,7 +1229,9 @@ int main(int argc, char* argv[]) {
     }
 
     // Connect
-    smartbotic::database::Client client({.address = args.address});
+    smartbotic::database::Client::Config clientCfg{.address = args.address};
+    if (!args.project.empty()) clientCfg.project = args.project;
+    smartbotic::database::Client client(clientCfg);
     client.connect();
 
     // Scriptable mode: single command

+ 278 - 0
client/include/smartbotic/database/client.hpp

@@ -211,6 +211,24 @@ public:
      */
     bool remove(const std::string& collection, const std::string& id);
 
+    /**
+     * Delete a document, reporting WHY on refusal.
+     *
+     * A relation with on_delete="restrict" (or the enforced default,
+     * "no_action") can refuse a delete that still has children referencing
+     * it - the server returns a real gRPC FAILED_PRECONDITION, not
+     * `deleted=false`. The single-return-value remove() above discarded
+     * that message (logged it at spdlog::error and nothing else), which
+     * left an operator unable to tell "already gone" apart from "blocked".
+     * This is an OVERLOAD, not a new parameter on the existing signature,
+     * so the exported symbol callers already link against is untouched.
+     *
+     * @param errorOut populated with the server's message when the call
+     *   fails for any reason (transport error, or a refusal). Empty when
+     *   the document simply didn't exist (deleted=false, no error).
+     */
+    bool remove(const std::string& collection, const std::string& id, std::string& errorOut);
+
     /**
      * Check if a document exists.
      */
@@ -655,6 +673,42 @@ public:
                      uint64_t& rowsIndexed);
     bool dropIndex(const std::string& collection, const std::string& field);
 
+    /**
+     * Result of createUniqueIndex() below. A brand-new struct, not a member
+     * added to IndexDefinition or CreateIndexResponse's shape - same ABI
+     * reasoning as the note above createIndex.
+     */
+    struct CreateIndexResult {
+        bool success = false;
+        std::string error;
+        uint64_t rowsIndexed = 0;
+        bool alreadyExisted = false;
+        /// Populated only when a fresh unique declaration was refused because
+        /// the field already holds duplicate values: up to five document ids
+        /// that collide, one per colliding value.
+        std::vector<std::string> duplicateExamples;
+    };
+
+    /**
+     * Declare `field` a secondary index AND a UNIQUE constraint: two
+     * documents in `collection` may never hold the same value for it. v2.11.0.
+     *
+     * Same backfill-before-declare semantics as createIndex(); idempotent
+     * for a field that is already unique. Declaring uniqueness over a field
+     * that already holds duplicate values is REFUSED - result.success is
+     * false, result.error explains it, and result.duplicateExamples carries
+     * sample colliding document ids so the caller can go fix the data rather
+     * than guess. A refused declaration over a field that had no index
+     * before this call leaves no index behind either.
+     *
+     * A separate method rather than a `bool unique` overload of
+     * createIndex(): the richer result (duplicate examples on refusal) is
+     * worth its own name rather than an overload set that only sometimes
+     * needs it.
+     */
+    [[nodiscard]] CreateIndexResult createUniqueIndex(const std::string& collection,
+                                                       const std::string& field);
+
     /** One declared index, as reported by listIndexes(). */
     struct IndexDefinition {
         std::string field;
@@ -666,6 +720,16 @@ public:
     };
     [[nodiscard]] std::vector<IndexDefinition> listIndexes(const std::string& collection);
 
+    /**
+     * Same as listIndexes() above, but also reports which fields carry a
+     * UNIQUE constraint via a parallel out-vector (uniqueFlags[i]
+     * corresponds to the returned IndexDefinition at the same position) -
+     * an out-parameter rather than a member added to IndexDefinition, same
+     * ABI reasoning as everywhere else on this class.
+     */
+    [[nodiscard]] std::vector<IndexDefinition> listIndexes(const std::string& collection,
+                                                            std::vector<bool>& uniqueFlags);
+
     /**
      * Distinct values an indexed field holds, with how many rows hold each. v2.10.0.
      *
@@ -701,6 +765,220 @@ public:
      */
     [[nodiscard]] std::optional<ViewDefinition> getViewInfo(const std::string& name);
 
+    // ===== Relation (Referential Integrity) Management — v2.11.0 T6b =====
+
+    /**
+     * A declared relation, as reported by listRelations() / getRelationInfo().
+     * Mirrors the server-side RelationDefinition proto message. This is a
+     * brand-new struct, not a member added to an existing one - safe under
+     * the unchanged soname (see the ABI note on createIndex above).
+     */
+    struct RelationDefinition {
+        std::string name;
+        std::string child;
+        std::string childField;
+        std::string parent;
+        std::string onDelete = "restrict";  // restrict | cascade | set_null | no_action
+        bool validateOnWrite = false;
+        uint64_t createdAt = 0;
+        uint64_t updatedAt = 0;
+    };
+
+    /**
+     * Declare a relation: documents in `childCollection` reference documents
+     * in `parentCollection` through `childField` (a dot-path that may
+     * resolve to a single id or an array of ids).
+     *
+     * Admin-only: `_relations` is a system collection and a relation names
+     * another collection's schema, so this is not ordinary per-collection
+     * write access.
+     *
+     * `onDelete` is one of "restrict" (default), "cascade", "set_null",
+     * "no_action". ALL FOUR ARE ENFORCED as of v2.11.0. An unrecognised value
+     * is REFUSED, not coerced to "restrict".
+     *
+     *   - "restrict"  — the parent delete fails with FAILED_PRECONDITION while
+     *                   any child still references it; the message names the
+     *                   relation, the child count and up to five child ids.
+     *   - "cascade"   — DESTRUCTIVE. Deleting the parent deletes every child
+     *                   that references it, atomically with the parent. If the
+     *                   reference is an ARRAY of ids, the parent's id is pulled
+     *                   from the array and the child is KEPT (cascade and
+     *                   set_null collapse to the same behaviour there) - an
+     *                   array reference is a many-to-many edge, not the child's
+     *                   reason to exist. Refused if a child is itself protected
+     *                   by its own `restrict` relation; NOT recursive, so a
+     *                   grandchild under a cascade relation on the child is left
+     *                   dangling for `relations check` to find.
+     *   - "set_null"  — DESTRUCTIVE (a write, not a delete). The child's
+     *                   reference field is set to null and the child is kept.
+     *                   Array references behave as under "cascade".
+     *   - "no_action" — the delete proceeds and the reference is left dangling.
+     *                   The documented escape hatch.
+     *
+     * `validateOnWrite` is also enforced as of v2.11.0: a write to the child
+     * whose reference names a parent that does not exist is rejected with
+     * FAILED_PRECONDITION. An empty-string reference is a hard rejection.
+     *
+     * Per-collection `relations_enforced` (configureCollection) turns EVERY
+     * on_delete policy off for a collection, not just restrict - so a bulk
+     * import can be loaded without enforcement and checked afterwards with
+     * `relations check`. Turning it back on does NOT retroactively find
+     * references created while it was off.
+     *
+     * A cross-project relation (name/child/parent resolving to different
+     * projects) is refused server-side: no LMDB transaction spans two
+     * project envs, so it could never be enforced atomically.
+     *
+     * @return true on success.
+     */
+    bool createRelation(const std::string& name, const std::string& childCollection,
+                        const std::string& childField, const std::string& parentCollection,
+                        const std::string& onDelete = "restrict",
+                        bool validateOnWrite = false);
+
+    /**
+     * Same as createRelation() above, but also reports how many existing
+     * rows in `childCollection` the server's bootstrap scan indexed (v2.11.0
+     * T7). Declaring a relation over a collection that already has data
+     * backfills the reverse index immediately - this is how a caller learns
+     * that happened and how much it covered, the same shape as
+     * createIndex(...,rowsIndexed).
+     *
+     * An OVERLOAD, not a default out-parameter added to the call above and
+     * not a struct member - same ABI reasoning as createIndex.
+     */
+    bool createRelation(const std::string& name, const std::string& childCollection,
+                        const std::string& childField, const std::string& parentCollection,
+                        const std::string& onDelete, bool validateOnWrite,
+                        uint64_t& rowsIndexed);
+
+    /**
+     * Drop a relation by name. Admin-only.
+     */
+    bool dropRelation(const std::string& name);
+
+    /**
+     * List all relations declared in this client's project. Admin-only.
+     */
+    [[nodiscard]] std::vector<RelationDefinition> listRelations();
+
+    /**
+     * Look up a single relation by name. Admin-only.
+     */
+    [[nodiscard]] std::optional<RelationDefinition> getRelationInfo(const std::string& name);
+
+    /**
+     * Set whether declared relations are enforced for `collection`.
+     * Defaults to true (enforced) - declaring a relation names its child
+     * and parent explicitly, so the declaration IS the opt-in.
+     *
+     * A METHOD rather than a member on CollectionConfig: adding a field to
+     * that public struct would change its size under an unchanged soname,
+     * which is exactly what crashed an installed consumer in v2.7.1 (see
+     * the ABI note above createIndex). This sends a ConfigureCollection
+     * partial update touching ONLY relations_enforced - timestampPrecision
+     * and versioningEnabled are left as they are.
+     */
+    bool setRelationsEnforced(const std::string& collection, bool enforced);
+
+    /**
+     * Read whether relations are currently enforced for `collection`.
+     * Defaults to true when the collection has no explicit config.
+     */
+    [[nodiscard]] bool getRelationsEnforced(const std::string& collection);
+
+    /**
+     * One relation's impact on deleting a specific document, as reported by
+     * describeDelete(). A brand-new struct, not a member added to an
+     * existing one - safe under the unchanged soname.
+     */
+    struct RelationImpact {
+        std::string relation;
+        std::string childCollection;
+        std::string childField;
+        std::string onDelete;              // restrict | cascade | set_null | no_action
+        uint64_t childCount = 0;
+        std::vector<std::string> sampleChildIds;   // at most five
+        bool blocks = false;
+    };
+
+    /**
+     * The result of describeDelete(): whether deleting collection/id would
+     * be blocked right now, and why. Read-only - describeDelete() never
+     * deletes anything.
+     */
+    struct DescribeDeleteResult {
+        bool success = false;
+        std::string error;
+        bool wouldBeBlocked = false;
+        std::vector<RelationImpact> impacts;
+    };
+
+    /**
+     * "What would happen if I deleted this?" - answered without deleting
+     * anything. Reports EVERY relation whose parent is `collection`,
+     * including cascade/no_action/set_null (with blocks=false), not just
+     * the ones that would block - the whole point is showing an operator
+     * what each declared policy will do. `impacts[i].blocks` is ANDed
+     * with the collection's live relations_enforced flag, so it and
+     * wouldBeBlocked agree with what an actual delete() would do right now.
+     *
+     * Gated as an ordinary per-collection READ, not admin - it changes
+     * nothing, but child counts and sample ids are facts about data.
+     * Cheap: counts come from the reverse index (mdb_cursor_count), not a
+     * collection scan, so this is meant to be called before every delete.
+     */
+    [[nodiscard]] DescribeDeleteResult describeDelete(const std::string& collection,
+                                                       const std::string& id);
+
+    /**
+     * One dangling reference found by checkRelation(): a child row points at
+     * a parent id that does not exist. A brand-new struct, not a member
+     * added to an existing one - safe under the unchanged soname.
+     */
+    struct DanglingReference {
+        std::string parentId;
+        uint64_t childCount = 0;                   // children of this parent, from the index
+        std::vector<std::string> sampleChildIds;    // at most five
+    };
+
+    /**
+     * Result of checkRelation(): every dangling parent id it found, capped
+     * (see the server implementation) with `totalDangling` giving the exact
+     * count so the cap never understates the problem silently.
+     */
+    struct CheckRelationResult {
+        bool success = false;
+        std::string error;
+        uint64_t totalDangling = 0;
+        std::vector<DanglingReference> dangling;    // capped
+    };
+
+    /**
+     * "Does this relation's reverse index actually match live data?" v2.11.0
+     * T7 - the bootstrap-scan gap this task closes. A relation declared over
+     * a collection that already had rows is backfilled by createRelation()
+     * itself now, but a relation restored from a snapshot predating that fix,
+     * or one whose index sub-db was otherwise lost, can still carry stale
+     * postings for parent ids that no longer exist. This walks the reverse
+     * index and reports them. Read-only - changes nothing.
+     *
+     * ⚠ Cost is proportional to the number of DISTINCT parents referenced,
+     * not `childCollection`'s row count, but that is still a full index
+     * walk. Treat this as an operator/migration tool, not something to call
+     * on a hot path or in a request-serving loop.
+     *
+     * Admin-only, like the other relation-management calls (createRelation,
+     * dropRelation, listRelations, getRelationInfo) - NOT a per-collection
+     * read on the child, even though it changes nothing. Answering requires
+     * looking the relation up first to learn its child collection, and
+     * gating after that lookup would make "relation does not exist" and
+     * "relation exists but you can't read it" distinguishable responses -
+     * a probe-able existence oracle.
+     */
+    [[nodiscard]] CheckRelationResult checkRelation(const std::string& name);
+
     // ===== Event Subscription =====
 
     using EventCallback = std::function<void(const std::string& collection,

+ 341 - 3
client/src/client.cpp

@@ -480,7 +480,14 @@ public:
         return {response.id(), response.inserted()};
     }
 
-    bool remove(const std::string& collection, const std::string& id) {
+    // errorOut is optional: nullptr keeps the old single-value remove()'s
+    // behaviour (log at spdlog::error and swallow the message). Non-null is
+    // what lets a caller distinguish "already gone" from "blocked" - a
+    // relation with on_delete=restrict/no_action refuses via a real gRPC
+    // FAILED_PRECONDITION, not deleted=false, and that message is the only
+    // way an operator learns WHAT blocked the delete.
+    bool removeImpl(const std::string& collection, const std::string& id,
+                    std::string* errorOut) {
         smartbotic::databasepb::DeleteRequest request;
         request.set_collection(qualify(collection));
         request.set_id(id);
@@ -507,12 +514,22 @@ public:
         }
         if (!status.ok()) {
             spdlog::error("Client::remove failed after retries: {}", status.error_message());
+            if (errorOut != nullptr) *errorOut = status.error_message();
             return false;
         }
 
         return response.deleted();
     }
 
+    bool remove(const std::string& collection, const std::string& id) {
+        return removeImpl(collection, id, nullptr);
+    }
+
+    bool remove(const std::string& collection, const std::string& id, std::string& errorOut) {
+        errorOut.clear();
+        return removeImpl(collection, id, &errorOut);
+    }
+
     bool exists(const std::string& collection, const std::string& id) {
         smartbotic::databasepb::ExistsRequest request;
         request.set_collection(qualify(collection));
@@ -1091,6 +1108,50 @@ public:
         return response.found();
     }
 
+    // v2.11.0 T6b — a METHOD, not a CollectionConfig member (see the header
+    // doc comment for why). Sends a partial update touching ONLY
+    // relations_enforced: timestamp_precision is left unset (empty string,
+    // the server's "leave unchanged" sentinel) and versioning_enabled is
+    // left absent (proto3 field presence), so neither is disturbed.
+    bool setRelationsEnforced(const std::string& collection, bool enforced) {
+        smartbotic::databasepb::ConfigureCollectionRequest request;
+        request.set_collection(qualify(collection));
+        request.mutable_config()->set_relations_enforced(enforced);
+
+        smartbotic::databasepb::ConfigureCollectionResponse response;
+        grpc::ClientContext context;
+        setDeadline(context);
+
+        auto status = stub_->ConfigureCollection(&context, request, &response);
+        if (!status.ok()) {
+            spdlog::error("Client::setRelationsEnforced failed: {}", status.error_message());
+            return false;
+        }
+        if (!response.success()) {
+            spdlog::error("Client::setRelationsEnforced rejected: {}", response.error());
+            return false;
+        }
+        return true;
+    }
+
+    bool getRelationsEnforced(const std::string& collection) {
+        smartbotic::databasepb::GetCollectionConfigRequest request;
+        request.set_collection(qualify(collection));
+        smartbotic::databasepb::GetCollectionConfigResponse response;
+        grpc::ClientContext context;
+        setDeadline(context);
+
+        auto status = stub_->GetCollectionConfig(&context, request, &response);
+        if (!status.ok()) {
+            spdlog::error("Client::getRelationsEnforced failed: {}", status.error_message());
+            return true;  // fail toward the safe (enforced) default
+        }
+        // Defaults to true (enforced) when never explicitly configured -
+        // mirrors the server's CollectionCfg default.
+        return !response.config().has_relations_enforced()
+                   || response.config().relations_enforced();
+    }
+
     Client::TimestampMigrationResult migrateCollectionTimestamps(
         const std::string& collection,
         const std::string& fromPrecision,
@@ -1208,7 +1269,38 @@ public:
         return response.success();
     }
 
-    std::vector<Client::IndexDefinition> listIndexes(const std::string& collection) {
+    // v2.11.0 T11 — unique constraints.
+    Client::CreateIndexResult createUniqueIndex(const std::string& collection,
+                                                const std::string& field) {
+        smartbotic::databasepb::CreateIndexRequest request;
+        request.set_collection(qualify(collection));
+        request.set_field(field);
+        request.set_unique(true);
+        smartbotic::databasepb::CreateIndexResponse response;
+        grpc::ClientContext context;
+        setDeadline(context);
+
+        Client::CreateIndexResult out;
+        auto status = stub_->CreateIndex(&context, request, &response);
+        if (!status.ok()) {
+            out.error = status.error_message();
+            spdlog::error("Client::createUniqueIndex failed: {}", out.error);
+            return out;
+        }
+        out.success = response.success();
+        out.error = response.error();
+        out.rowsIndexed = response.rows_indexed();
+        out.alreadyExisted = response.already_existed();
+        out.duplicateExamples.assign(response.duplicate_examples().begin(),
+                                     response.duplicate_examples().end());
+        if (!out.success) {
+            spdlog::error("Client::createUniqueIndex rejected: {}", out.error);
+        }
+        return out;
+    }
+
+    std::vector<Client::IndexDefinition> listIndexes(const std::string& collection,
+                                                      std::vector<bool>* uniqueFlags) {
         smartbotic::databasepb::ListIndexesRequest request;
         request.set_collection(qualify(collection));
         smartbotic::databasepb::ListIndexesResponse response;
@@ -1222,12 +1314,17 @@ public:
             return out;
         }
         out.reserve(response.indexes_size());
+        if (uniqueFlags != nullptr) {
+            uniqueFlags->clear();
+            uniqueFlags->reserve(response.indexes_size());
+        }
         for (const auto& i : response.indexes()) {
             Client::IndexDefinition d;
             d.field = i.field();
             d.distinctValues = i.distinct_values();
             d.entries = i.entries();
             out.push_back(std::move(d));
+            if (uniqueFlags != nullptr) uniqueFlags->push_back(i.unique());
         }
         return out;
     }
@@ -1356,6 +1453,189 @@ public:
         return v;
     }
 
+    // ===== v2.11.0 T6b — relation (referential integrity) management =====
+
+    Client::RelationDefinition relationFromProto(
+        const smartbotic::databasepb::RelationDefinition& pb) const {
+        Client::RelationDefinition r;
+        // Round-trip like listViews(): the caller declared "posts_by_user",
+        // it should list back as "posts_by_user", not "myproject:posts_by_user".
+        r.name = unqualify(pb.name());
+        r.child = unqualify(pb.child());
+        r.childField = pb.child_field();
+        r.parent = unqualify(pb.parent());
+        r.onDelete = pb.on_delete().empty() ? "restrict" : pb.on_delete();
+        r.validateOnWrite = pb.validate_on_write();
+        r.createdAt = pb.created_at();
+        r.updatedAt = pb.updated_at();
+        return r;
+    }
+
+    bool createRelation(const std::string& name, const std::string& childCollection,
+                        const std::string& childField, const std::string& parentCollection,
+                        const std::string& onDelete, bool validateOnWrite,
+                        uint64_t* rowsIndexed) {
+        smartbotic::databasepb::CreateRelationRequest request;
+        // name, child AND parent all need qualifying - qualifying only
+        // `child`/`parent` but not `name` is exactly the v2.4.2 createView
+        // bug (registry keyed on the bare name, every lookup sent the
+        // qualified one, the keys never met).
+        request.set_name(qualify(name));
+        request.set_child(qualify(childCollection));
+        request.set_child_field(childField);
+        request.set_parent(qualify(parentCollection));
+        request.set_on_delete(onDelete);
+        request.set_validate_on_write(validateOnWrite);
+
+        smartbotic::databasepb::CreateRelationResponse response;
+        grpc::ClientContext context;
+        setDeadline(context);
+
+        auto status = stub_->CreateRelation(&context, request, &response);
+        if (!status.ok()) {
+            spdlog::error("Client::createRelation failed: {}", status.error_message());
+            return false;
+        }
+        if (!response.success()) {
+            spdlog::error("Client::createRelation rejected: {}", response.error());
+            return false;
+        }
+        // v2.11.0 T7 - the server's bootstrap scan already backfilled the
+        // reverse index over any rows present in childCollection.
+        if (rowsIndexed != nullptr) *rowsIndexed = response.rows_indexed();
+        return true;
+    }
+
+    bool dropRelation(const std::string& name) {
+        smartbotic::databasepb::DropRelationRequest request;
+        request.set_name(qualify(name));
+        smartbotic::databasepb::DropRelationResponse response;
+        grpc::ClientContext context;
+        setDeadline(context);
+
+        auto status = stub_->DropRelation(&context, request, &response);
+        if (!status.ok()) {
+            spdlog::error("Client::dropRelation failed: {}", status.error_message());
+            return false;
+        }
+        if (!response.success()) {
+            spdlog::error("Client::dropRelation rejected: {}", response.error());
+            return false;
+        }
+        return true;
+    }
+
+    std::vector<Client::RelationDefinition> listRelations() {
+        smartbotic::databasepb::ListRelationsRequest request;
+        // Scope the listing to this client's workspace, like listViews().
+        request.set_project(config_.project);
+        smartbotic::databasepb::ListRelationsResponse response;
+        grpc::ClientContext context;
+        setDeadline(context);
+
+        std::vector<Client::RelationDefinition> out;
+        auto status = stub_->ListRelations(&context, request, &response);
+        if (!status.ok()) {
+            spdlog::error("Client::listRelations failed: {}", status.error_message());
+            return out;
+        }
+        out.reserve(response.relations_size());
+        for (const auto& pb : response.relations()) {
+            out.push_back(relationFromProto(pb));
+        }
+        return out;
+    }
+
+    std::optional<Client::RelationDefinition> getRelationInfo(const std::string& name) {
+        smartbotic::databasepb::GetRelationInfoRequest request;
+        request.set_name(qualify(name));
+        smartbotic::databasepb::GetRelationInfoResponse response;
+        grpc::ClientContext context;
+        setDeadline(context);
+
+        auto status = stub_->GetRelationInfo(&context, request, &response);
+        if (!status.ok() || !response.found()) {
+            return std::nullopt;
+        }
+        return relationFromProto(response.relation());
+    }
+
+    // v2.11.0 T5 — "what would happen if I deleted this?" Gated as an
+    // ordinary read server-side, so `collection` is qualified the same way
+    // get()/find() qualify theirs - not the admin-only qualify() used by
+    // the relation-management calls above (which is the same call, just a
+    // reminder this one follows the read family, not the admin family).
+    Client::DescribeDeleteResult describeDelete(const std::string& collection,
+                                                const std::string& id) {
+        smartbotic::databasepb::DescribeDeleteRequest request;
+        request.set_collection(qualify(collection));
+        request.set_id(id);
+
+        smartbotic::databasepb::DescribeDeleteResponse response;
+        grpc::ClientContext context;
+        setDeadline(context);
+
+        Client::DescribeDeleteResult out;
+        auto status = stub_->DescribeDelete(&context, request, &response);
+        if (!status.ok()) {
+            spdlog::error("Client::describeDelete failed: {}", status.error_message());
+            out.success = false;
+            out.error = status.error_message();
+            return out;
+        }
+        out.success = response.success();
+        out.error = response.error();
+        out.wouldBeBlocked = response.would_be_blocked();
+        out.impacts.reserve(response.impacts_size());
+        for (const auto& pbImpact : response.impacts()) {
+            Client::RelationImpact impact;
+            impact.relation = unqualify(pbImpact.relation());
+            impact.childCollection = unqualify(pbImpact.child_collection());
+            impact.childField = pbImpact.child_field();
+            impact.onDelete = pbImpact.on_delete();
+            impact.childCount = pbImpact.child_count();
+            impact.sampleChildIds.assign(pbImpact.sample_child_ids().begin(),
+                                         pbImpact.sample_child_ids().end());
+            impact.blocks = pbImpact.blocks();
+            out.impacts.push_back(std::move(impact));
+        }
+        return out;
+    }
+
+    // v2.11.0 T7 - "does this relation's reverse index actually match live
+    // data?" Read-only; qualified like the other relation-management calls
+    // (getRelationInfo etc.), since a relation name is admin-scoped the same
+    // way regardless of which call reads it.
+    Client::CheckRelationResult checkRelation(const std::string& name) {
+        smartbotic::databasepb::CheckRelationRequest request;
+        request.set_name(qualify(name));
+
+        smartbotic::databasepb::CheckRelationResponse response;
+        grpc::ClientContext context;
+        setDeadline(context);
+
+        Client::CheckRelationResult out;
+        auto status = stub_->CheckRelation(&context, request, &response);
+        if (!status.ok()) {
+            spdlog::error("Client::checkRelation failed: {}", status.error_message());
+            out.success = false;
+            out.error = status.error_message();
+            return out;
+        }
+        out.success = response.success();
+        out.error = response.error();
+        out.totalDangling = response.total_dangling();
+        out.dangling.reserve(response.dangling_size());
+        for (const auto& pbd : response.dangling()) {
+            Client::DanglingReference d;
+            d.parentId = pbd.parent_id();
+            d.childCount = pbd.child_count();
+            d.sampleChildIds.assign(pbd.sample_child_ids().begin(), pbd.sample_child_ids().end());
+            out.dangling.push_back(std::move(d));
+        }
+        return out;
+    }
+
     // ===== Event Subscription =====
 
     class SubscriptionHandle {
@@ -2039,6 +2319,10 @@ bool Client::remove(const std::string& collection, const std::string& id) {
     return impl_->remove(collection, id);
 }
 
+bool Client::remove(const std::string& collection, const std::string& id, std::string& errorOut) {
+    return impl_->remove(collection, id, errorOut);
+}
+
 bool Client::exists(const std::string& collection, const std::string& id) {
     return impl_->exists(collection, id);
 }
@@ -2187,8 +2471,18 @@ bool Client::dropIndex(const std::string& collection, const std::string& field)
     return impl_->dropIndex(collection, field);
 }
 
+Client::CreateIndexResult Client::createUniqueIndex(const std::string& collection,
+                                                    const std::string& field) {
+    return impl_->createUniqueIndex(collection, field);
+}
+
 std::vector<Client::IndexDefinition> Client::listIndexes(const std::string& collection) {
-    return impl_->listIndexes(collection);
+    return impl_->listIndexes(collection, nullptr);
+}
+
+std::vector<Client::IndexDefinition> Client::listIndexes(const std::string& collection,
+                                                          std::vector<bool>& uniqueFlags) {
+    return impl_->listIndexes(collection, &uniqueFlags);
 }
 
 std::vector<Client::IndexValue> Client::indexValues(const std::string& collection,
@@ -2205,6 +2499,50 @@ std::optional<Client::ViewDefinition> Client::getViewInfo(const std::string& nam
     return impl_->getViewInfo(name);
 }
 
+bool Client::createRelation(const std::string& name, const std::string& childCollection,
+                            const std::string& childField, const std::string& parentCollection,
+                            const std::string& onDelete, bool validateOnWrite) {
+    return impl_->createRelation(name, childCollection, childField, parentCollection,
+                                 onDelete, validateOnWrite, nullptr);
+}
+
+bool Client::createRelation(const std::string& name, const std::string& childCollection,
+                            const std::string& childField, const std::string& parentCollection,
+                            const std::string& onDelete, bool validateOnWrite,
+                            uint64_t& rowsIndexed) {
+    return impl_->createRelation(name, childCollection, childField, parentCollection,
+                                 onDelete, validateOnWrite, &rowsIndexed);
+}
+
+bool Client::dropRelation(const std::string& name) {
+    return impl_->dropRelation(name);
+}
+
+std::vector<Client::RelationDefinition> Client::listRelations() {
+    return impl_->listRelations();
+}
+
+std::optional<Client::RelationDefinition> Client::getRelationInfo(const std::string& name) {
+    return impl_->getRelationInfo(name);
+}
+
+bool Client::setRelationsEnforced(const std::string& collection, bool enforced) {
+    return impl_->setRelationsEnforced(collection, enforced);
+}
+
+bool Client::getRelationsEnforced(const std::string& collection) {
+    return impl_->getRelationsEnforced(collection);
+}
+
+Client::DescribeDeleteResult Client::describeDelete(const std::string& collection,
+                                                     const std::string& id) {
+    return impl_->describeDelete(collection, id);
+}
+
+Client::CheckRelationResult Client::checkRelation(const std::string& name) {
+    return impl_->checkRelation(name);
+}
+
 std::shared_ptr<void> Client::subscribe(const std::vector<std::string>& collections, EventCallback callback) {
     return impl_->subscribe(collections, std::move(callback));
 }

+ 28 - 31
docs/ROADMAP.md

@@ -1,6 +1,6 @@
 # Smartbotic Database - Status and Roadmap
 
-**Current version: 2.10.0** (see `VERSION`). Last reviewed: 2026-08-09.
+**Current version: 2.11.0** (see `VERSION`). Last reviewed: 2026-08-10.
 
 This is the single authoritative statement of what exists and what does not.
 If any other document in this repository disagrees with this one, this one is
@@ -53,7 +53,7 @@ Consequences of that substitution, which trip up readers:
 
 ## Shipped
 
-Every item below is in the installed product as of 2.10.0. `CLAUDE.md` has the
+Every item below is in the installed product as of 2.11.0. `CLAUDE.md` has the
 detail and the failure modes.
 
 - JSON document store: collections, version history, field-level encryption, TTL
@@ -66,6 +66,10 @@ detail and the failure modes.
 - Per-listener TLS and bearer-token auth
 - Durable snapshots with tiered recovery and read-only lockout
 - Replication, events/subscribe, migrations, set operations
+- Referential integrity: relations with restrict / cascade / set_null / no_action,
+  a DUPSORT reverse index, `DescribeDelete`, `relations check`, per-collection
+  enforcement switches, and TTL expiry running the same policy as a manual delete
+- Unique constraints, enforced inside the document's own transaction
 - Paging fast path (no filter, no sort) and the two-pass filtered/sorted scan
 - Secondary indexes on declared fields: equality, IN, CONTAINS, EXISTS,
   ranges, intersection, result ordering, filtered totals, and distinct
@@ -136,35 +140,28 @@ This is the v2.x arc's architectural endpoint. It touches the write path that
 produced the v2.4.3, v2.4.4 and v2.8.0 LMDB handle incidents, so it wants to land
 in small reviewable pieces with the sub-db identity sentinel kept intact.
 
-### 5. Relations, and unique constraints - one piece of work
-
-Referential integrity does not exist: deleting a parent leaves children pointing
-at nothing, silently, with no way to detect it. `smartbotic-automation` has
-`workflows` referenced by `executions`, `users` by `sessions`, and `credentials`
-from node configuration.
-
-**The design is re-validated and current** as of 2026-08-09:
-`docs/superpowers/specs/2026-08-03-relations-design.md`. Read its status section
-first - it lists what survived re-validation, what was wrong, and the constraints
-that post-date it. **The plan to execute is
-`docs/superpowers/plans/2026-08-09-relations-v2.11.0.md`** (13 tasks, Phase A
-additive and shippable alone, Phase B the write-path work). The 3278-line
-`2026-08-04-relations-v2.5.0.md` beside it is superseded and must not be executed.
-
-**Relations and unique constraints share one blocker**, so schedule them
-together. Both need `LmdbDocumentStore` operations to accept a caller's
-`WriteTxn` - relations for atomic cascade across parent, children and index
-sub-dbs; uniqueness so a rejection can propagate instead of being swallowed by
-`applyDualWriteMirror`, which currently catches every exception, bumps mirror
-drift and flips `mirror_healthy_` (the v2.8.1 fault). MemoryStore also mutates
-before the mirror runs, so a clean rejection needs the in-memory write rolled
-back. The uniqueness check itself is already built and tested, sitting unreachable
-behind that.
-
-Also requested and specified: **per-collection enable/disable** for relation
-enforcement and uniqueness, persisted in `CollectionCfg` alongside
-`versioningEnabled` and `indexedFields`, and re-armed at boot the way
-`applyIndexDeclarations()` already does.
+### 5. Relations - what is left after v2.11.0
+
+Shipped in v2.11.0; see the `CLAUDE.md` entry for the full surface. Remaining,
+all deliberate and none blocking:
+
+- **`restrict` is checked one level deep.** A cascade that would destroy a
+  `restrict`-protected grandchild is refused rather than recursing. Recursion
+  needs an unbounded transaction and its own WAL-first story.
+- **A cascade that fails after its WAL fsync still applies on the next restart.**
+  Inherent to WAL-before-LMDB; documented in `relation_cascade.hpp` and the error.
+- **One narrow race remains:** a write extending a TTL *during* a cascade's
+  fsync-plus-commit. Only reachable for a `cascade`/`set_null` relation whose
+  parent collection carries a TTL and is renewed after expiry. The fix is a
+  per-document claim flag that makes the concurrent write lose visibly - **not** a
+  versioned delete, which does not close it (no version-checked delete primitive
+  exists, and the cascade never goes through `remove()`).
+- **Cascade is unbounded in memory and transaction size** for a very wide parent;
+  a pre-flight child-count cap would bound both.
+- `_relations` has no direct-document-write interception, unlike `_policies`, so
+  an admin editing the declaration by hand has no effect until restart. A
+  replication follower likewise does not re-arm until restart.
+- No reserved `_`-prefix check on relation names.
 
 ### 6. Smaller known gaps
 

+ 215 - 0
proto/database.proto

@@ -70,6 +70,37 @@ service DatabaseService {
     rpc ListViews(ListViewsRequest) returns (ListViewsResponse);
     rpc GetViewInfo(GetViewInfoRequest) returns (GetViewInfoResponse);
 
+    // v2.11.0 T6b — relation (referential integrity) management. Admin-only:
+    // `_relations` is a system collection and a relation names another
+    // collection's schema, so declaring one is not ordinary per-collection
+    // write access.
+    rpc CreateRelation(CreateRelationRequest) returns (CreateRelationResponse);
+    rpc DropRelation(DropRelationRequest) returns (DropRelationResponse);
+    rpc ListRelations(ListRelationsRequest) returns (ListRelationsResponse);
+    rpc GetRelationInfo(GetRelationInfoRequest) returns (GetRelationInfoResponse);
+
+    // v2.11.0 T5 — "what would happen if I deleted this?" without deleting
+    // it. Read-only and gated as an ordinary per-collection read, not admin.
+    rpc DescribeDelete(DescribeDeleteRequest) returns (DescribeDeleteResponse);
+
+    // v2.11.0 T7 — "does this relation's reverse index actually match live
+    // data?" A relation declared over a collection that already had rows
+    // before v2.11.0 T7 shipped (or one whose sub-db was lost/restored from
+    // an older snapshot) can carry postings for parent ids that no longer
+    // exist. This walks the reverse index and probes each parent id;
+    // mutates nothing. ⚠ Cost is proportional to the number of DISTINCT
+    // parents referenced, not collection size, but on a very large child
+    // collection that is still a full index walk - treat this as an
+    // operator/migration tool, not something to call on a hot path.
+    // Admin-only, like the other relation-management RPCs above (NOT gated
+    // as a per-collection read on `child`, despite being read-only and
+    // reporting facts about that collection's data): answering requires
+    // resolving the relation first to learn its `child`, and gating after
+    // that lookup would make a nonexistent-relation response distinguishable
+    // from an existing-but-unauthorized one - a probe-able existence oracle,
+    // which is exactly what gate() exists to prevent elsewhere.
+    rpc CheckRelation(CheckRelationRequest) returns (CheckRelationResponse);
+
     // Collection configuration
     rpc ConfigureCollection(ConfigureCollectionRequest) returns (ConfigureCollectionResponse);
 
@@ -885,6 +916,147 @@ message GetViewInfoResponse {
     bool found = 2;
 }
 
+// ===== Relation (Referential Integrity) Operations — v2.11.0 T6b =====
+//
+// A relation says: documents in `child` reference documents in `parent`
+// through `child_field` (a dot-path that may resolve to a single id or an
+// array of ids). Declarations are persisted in the global `_relations`
+// system collection (mirrors `_views`) and are project-qualified like any
+// other collection-carrying name. Management is admin-only.
+
+message RelationDefinition {
+    string name = 1;               // relation name, project-qualified
+    string child = 2;               // collection holding the reference, project-qualified
+    string child_field = 3;         // dot-path on the child; may resolve to an array of ids
+    string parent = 4;              // collection being referenced, project-qualified
+    // "restrict" (default) | "cascade" | "set_null" | "no_action".
+    //
+    // ALL FOUR ARE ENFORCED as of v2.11.0, and an unrecognised value is
+    // REFUSED by CreateRelation rather than coerced to "restrict".
+    //   restrict  - the parent delete fails FAILED_PRECONDITION while any
+    //               child still references it.
+    //   cascade   - DESTRUCTIVE: children referencing the parent are deleted
+    //               atomically with it. If child_field resolves to an ARRAY,
+    //               the parent id is pulled from the array and the child is
+    //               KEPT (cascade and set_null collapse there). Refused if a
+    //               child is itself protected by its own restrict relation.
+    //               Not recursive.
+    //   set_null  - DESTRUCTIVE: the child's reference field is set to null
+    //               and the child is kept. Arrays behave as under cascade.
+    //   no_action - the delete proceeds, the reference is left dangling.
+    string on_delete = 5;
+    // ENFORCED as of v2.11.0: a write to the child naming a parent that does
+    // not exist is rejected with FAILED_PRECONDITION. An empty-string
+    // reference is a hard rejection.
+    bool validate_on_write = 6;
+    uint64 created_at = 7;
+    uint64 updated_at = 8;
+}
+
+message CreateRelationRequest {
+    string name = 1;
+    string child = 2;
+    string child_field = 3;
+    string parent = 4;
+    string on_delete = 5;           // empty defaults to "restrict"
+    bool validate_on_write = 6;
+}
+
+message CreateRelationResponse {
+    bool success = 1;
+    string error = 2;
+    // v2.11.0 T7 — rows indexed by the bootstrap scan. Declaring a relation
+    // over a collection that already has rows backfills the reverse index
+    // immediately (mirrors CreateIndexResponse.rows_indexed), so enforcement
+    // covers pre-existing children from the moment the relation is declared,
+    // not only writes made afterwards.
+    uint64 rows_indexed = 3;
+}
+
+message DropRelationRequest {
+    string name = 1;
+}
+
+message DropRelationResponse {
+    bool success = 1;
+    string error = 2;
+}
+
+message ListRelationsRequest {
+    // Restrict the listing to one project namespace. Empty lists every
+    // project (operator/CLI use). Clients always send their own project so
+    // one workspace never enumerates another's relations.
+    string project = 1;
+}
+
+message ListRelationsResponse {
+    repeated RelationDefinition relations = 1;
+}
+
+message GetRelationInfoRequest {
+    string name = 1;
+}
+
+message GetRelationInfoResponse {
+    RelationDefinition relation = 1;
+    bool found = 2;
+}
+
+// v2.11.0 T5 — "what would happen if I deleted this?", answered without
+// deleting anything. A relational database exposes its constraints through
+// DDL you can read; a document store has none, so this has to be a query.
+// Gated as an ordinary READ on `collection` (it changes nothing), unlike
+// the relation-management RPCs above which are admin-only.
+message DescribeDeleteRequest {
+    string collection = 1;
+    string id = 2;
+}
+
+message RelationImpact {
+    string relation = 1;
+    string child_collection = 2;
+    string child_field = 3;
+    string on_delete = 4;        // restrict | cascade | set_null | no_action
+    uint64 child_count = 5;
+    repeated string sample_child_ids = 6;   // at most five
+    bool blocks = 7;
+}
+
+message DescribeDeleteResponse {
+    bool success = 1;
+    string error = 2;
+    bool would_be_blocked = 3;
+    repeated RelationImpact impacts = 4;
+}
+
+// v2.11.0 T7 — the bootstrap-scan gap. Declaring a relation with T3 alone
+// only maintains the reverse index going FORWARD from writes made after
+// declaration; rows already in the child collection were never indexed, so
+// enforcement silently missed them. This RPC answers "is that still true
+// right now, for this relation" without changing anything.
+message CheckRelationRequest {
+    string name = 1;              // relation name, project-qualified like the others
+}
+
+// One parent id referenced by the child collection's data, whose parent
+// document does not exist.
+message DanglingReference {
+    string parent_id = 1;
+    uint64 child_count = 2;             // children of this parent, from the index
+    repeated string sample_child_ids = 3;   // at most five
+}
+
+message CheckRelationResponse {
+    bool success = 1;
+    string error = 2;
+    // Every dangling parent id found, not just the reported sample - so a
+    // capped `dangling` list below does not silently understate the problem.
+    uint64 total_dangling = 3;
+    // Capped (see the server implementation) so a badly out-of-sync relation
+    // cannot blow up the response size; total_dangling is the true count.
+    repeated DanglingReference dangling = 4;
+}
+
 // ===== Collection Configuration =====
 
 // Per-collection configuration for timestamp precision and other runtime knobs.
@@ -908,6 +1080,18 @@ message CollectionConfig {
     // which means "unlimited", not "off".
     optional bool versioning_enabled = 2;
 
+    // v2.11.0 T8 — whether declared relations are enforced for this collection.
+    //
+    // `optional` for the same reason versioning_enabled is: a plain bool
+    // defaults to false, so a ConfigureCollection call that meant to set
+    // only timestamp_precision would silently switch relation enforcement
+    // OFF. With presence, an absent field means "leave unchanged".
+    //
+    // Defaults to true (enforced) when never explicitly configured -
+    // declaring a relation names its child and parent explicitly, so the
+    // declaration IS the opt-in.
+    optional bool relations_enforced = 3;
+
     // Room for future per-collection knobs
 }
 
@@ -930,6 +1114,12 @@ message CreateIndexRequest {
     // document metadata fields (_id, _created_at, _updated_at, _version) are
     // addressable too.
     string field = 2;
+    // v2.11.0 T11 — when true, the field carries a UNIQUE constraint: two
+    // documents may never hold the same value for it. Declaring this over a
+    // collection that already contains duplicate values is REFUSED (see
+    // CreateIndexResponse.duplicate_examples) rather than accepted and left
+    // silently unenforced against the data that already violates it.
+    bool unique = 3;
 }
 
 message CreateIndexResponse {
@@ -941,6 +1131,11 @@ message CreateIndexResponse {
     uint64 rows_indexed = 3;
     // True when the index already existed - creation is idempotent.
     bool already_existed = 4;
+    // v2.11.0 T11 — set when a `unique=true` request was refused because the
+    // field already holds duplicate values. `error` explains the refusal;
+    // these are up to five example document ids that collide, one entry per
+    // colliding value, so the caller can go fix the data rather than guess.
+    repeated string duplicate_examples = 5;
 }
 
 message DropIndexRequest {
@@ -964,6 +1159,8 @@ message IndexInfo {
     // postings are concentrated in few values will not be used by the planner.
     uint64 distinct_values = 2;
     uint64 entries = 3;
+    // v2.11.0 T11 — true when this index also carries a UNIQUE constraint.
+    bool unique = 4;
 }
 
 message ListIndexesResponse {
@@ -1065,4 +1262,22 @@ message GetMemoryStatsResponse {
     uint64 last_eviction_timestamp = 6;  // ms since epoch, 0 if none
     uint64 last_eviction_docs = 7;
     uint64 last_eviction_bytes_freed = 8;
+
+    // v2.11.0 — TTL expiries a relation is refusing.
+    //
+    // ttl_blocked_documents is a GAUGE: how many documents are past their TTL
+    // right now and NOT being expired because deleting them would break a
+    // `restrict` relation (or because the cascade path is refusing while the
+    // LMDB mirror is unhealthy/drifted). It falls when a block lifts. Nonzero
+    // means those documents are outliving their retention policy - the
+    // deliberate alternative to silently orphaning their children.
+    //
+    // ttl_expiry_blocked_events is a monotonic COUNTER of refusal events, so it
+    // keeps rising while a block persists (each blocked document is retried on a
+    // slow cadence). Use the gauge for "how bad is it now", the counter for
+    // "is this still happening".
+    //
+    // Both are zero on any install that declares no relations.
+    uint64 ttl_blocked_documents = 9;
+    uint64 ttl_expiry_blocked_events = 10;
 }

+ 4 - 0
service/CMakeLists.txt

@@ -57,6 +57,10 @@ set(DATABASE_SERVICE_SOURCES
     src/migrations/migration_runner.cpp
     src/views/projection.cpp
     src/views/view_manager.cpp
+    src/relations/relation_manager.cpp
+    src/relations/relation_index.cpp
+    src/relations/relation_enforcement.cpp
+    src/relations/relation_cascade.cpp
     src/config/collection_config_manager.cpp
     src/config/config_loader.cpp
     src/storage/lmdb_env.cpp

+ 28 - 7
service/src/config/collection_config_manager.cpp

@@ -1,6 +1,7 @@
 #include "collection_config_manager.hpp"
 
 #include "../memory_store.hpp"
+#include "../project_addressing.hpp"
 
 #include <nlohmann/json.hpp>
 #include <spdlog/spdlog.h>
@@ -13,7 +14,9 @@ nlohmann::json toJson(const CollectionCfg& cfg) {
     return {
         {"timestamp_precision", cfg.timestampPrecision},
         {"versioning_enabled", cfg.versioningEnabled},
-        {"indexed_fields", cfg.indexedFields}
+        {"indexed_fields", cfg.indexedFields},
+        {"relations_enforced", cfg.relationsEnforced},
+        {"unique_fields", cfg.uniqueFields}
     };
 }
 
@@ -26,11 +29,25 @@ CollectionCfg fromJson(const nlohmann::json& j) {
     c.versioningEnabled = j.value("versioning_enabled", true);
     // Absent means no indexes, which is what every pre-v2.9.0 config record has.
     c.indexedFields = j.value("indexed_fields", std::vector<std::string>{});
+    // v2.11.0 T8 — absent on every pre-2.11 record, which must keep enforcing
+    // (declaring a relation is the opt-in) and must keep having no unique
+    // fields (nothing enforces those yet).
+    c.relationsEnforced = j.value("relations_enforced", true);
+    c.uniqueFields = j.value("unique_fields", std::vector<std::string>{});
     return c;
 }
 
 } // anonymous namespace
 
+std::string CollectionConfigManager::canonicalKey(const std::string& collection) {
+    if (collection.empty() || collection[0] == '_') return collection;
+    try {
+        return resolveCollection(collection).qualified;
+    } catch (const std::exception&) {
+        return collection;   // see the header: never throws on a hot path
+    }
+}
+
 CollectionConfigManager::CollectionConfigManager(MemoryStore& store) : store_(store) {}
 
 void CollectionConfigManager::loadFromStore() {
@@ -64,7 +81,7 @@ void CollectionConfigManager::loadFromStore() {
         if (result.documents.empty()) break;
         for (const auto& doc : result.documents) {
             if (doc.id.empty()) continue;
-            cache_[doc.id] = fromJson(doc.data());
+            cache_[canonicalKey(doc.id)] = fromJson(doc.data());
         }
         if (result.documents.size() < kPage) break;
         offset += kPage;
@@ -91,9 +108,13 @@ bool CollectionConfigManager::setConfig(const std::string& collection,
         return false;
     }
 
-    // Persist via upsert (document ID = collection name)
+    // Persist via upsert (document ID = the CANONICAL collection name, so a
+    // bare-named caller and a qualified-named caller write the same record
+    // rather than two half-configs - see canonicalKey()).
+    const std::string key = canonicalKey(collection);
+
     Document doc;
-    doc.id = collection;
+    doc.id = key;
     doc.set_data(toJson(cfg));
     try {
         store_.upsert(SYSTEM_COLLECTION, doc);
@@ -104,7 +125,7 @@ bool CollectionConfigManager::setConfig(const std::string& collection,
 
     {
         std::unique_lock<std::shared_mutex> wlock(cacheMutex_);
-        cache_[collection] = cfg;
+        cache_[key] = cfg;
     }
     spdlog::info("CollectionConfigManager: set '{}' timestamp_precision={}",
                  collection, cfg.timestampPrecision);
@@ -113,7 +134,7 @@ bool CollectionConfigManager::setConfig(const std::string& collection,
 
 CollectionCfg CollectionConfigManager::configFor(const std::string& collection) const {
     std::shared_lock<std::shared_mutex> lock(cacheMutex_);
-    auto it = cache_.find(collection);
+    auto it = cache_.find(canonicalKey(collection));
     if (it == cache_.end()) {
         return CollectionCfg{};  // default (ms)
     }
@@ -122,7 +143,7 @@ CollectionCfg CollectionConfigManager::configFor(const std::string& collection)
 
 bool CollectionConfigManager::hasExplicitConfig(const std::string& collection) const {
     std::shared_lock<std::shared_mutex> lock(cacheMutex_);
-    return cache_.contains(collection);
+    return cache_.contains(canonicalKey(collection));
 }
 
 } // namespace smartbotic::database

+ 55 - 0
service/src/config/collection_config_manager.hpp

@@ -49,6 +49,40 @@ struct CollectionCfg {
     // set_indexed_fields(). If that did not happen, writes would stop maintaining
     // an index that queries still consult - stale rows returned as if current.
     std::vector<std::string> indexedFields;
+
+    // v2.11.0 T8 — whether declared relations are enforced for this collection.
+    //
+    // Defaults to true, unlike versioningEnabled/indexedFields which default
+    // to "off"/empty: declaring a relation names its child and parent
+    // explicitly, so the declaration IS the opt-in. A declared constraint
+    // that silently did nothing would be worse than no constraint at all.
+    //
+    // Lives here for the same reason versioningEnabled does: `_collection_meta`
+    // is an ordinary collection, so it is WAL'd and snapshotted for free.
+    //
+    // Consumed by Task 4's delete enforcement (restrict/no_action on delete):
+    // read via configFor(collection).relationsEnforced before refusing a
+    // delete that would orphan a child row - that consumer IS just a
+    // per-call read of this flag, no arming needed.
+    //
+    // v2.11.0 T13 round 2 — validate_on_write is different: it runs inside
+    // LmdbDocumentStore, which cannot read CollectionConfigManager itself
+    // (keyed by qualified name; the per-project storage layer deliberately
+    // doesn't carry that - see RelationRef's comment). So this flag IS
+    // pushed into another component for that consumer, the same way
+    // indexedFields is pushed via LmdbDocumentStore::set_indexed_fields():
+    // ConfigureCollection re-arms the collection's relations
+    // (armRelationsForChild) whenever it changes this flag, so
+    // RelationRef::relationsEnforced stays a fresh snapshot rather than
+    // going stale.
+    bool relationsEnforced = true;
+
+    // v2.11.0 T8 — persisted only. Nothing enforces uniqueness yet (Task 11
+    // activates it). A constraint that is advertised but unenforced would be
+    // worse than an absent one, so this deliberately has no RPC surface and
+    // no proto field yet - it exists here purely so the storage shape is
+    // settled before enforcement lands.
+    std::vector<std::string> uniqueFields;
 };
 
 /**
@@ -95,6 +129,27 @@ public:
     std::unordered_map<std::string, CollectionCfg> allConfigs() const;
 
 private:
+    /**
+     * v2.11.0 final review (finding 8) — the cache key for a USER collection
+     * is its canonical "<project>:<collection>" form, applied on every read
+     * and every write so the two cannot disagree.
+     *
+     * Without this, ConfigureCollection stored under whatever string the
+     * caller sent (`request->collection()`, raw) while the relation-arming
+     * path reads the canonical form - so `configureCollection("executions",
+     * relations_enforced=false)` wrote a key nothing ever reads, and the
+     * collection kept enforcing. Third occurrence of the bare-vs-qualified
+     * class in this repository (v2.4.2 views, v2.4.5 collections), so it is
+     * fixed at the keying layer rather than at each call site.
+     *
+     * `_`-prefixed system collections are left EXACTLY as given: they are
+     * global, not project-scoped, and `default:_views` does not exist - the
+     * same rule Client::qualify() follows since v2.7.0. Unparseable names are
+     * also returned unchanged, so a bad name reads back the default config
+     * instead of throwing on a hot path.
+     */
+    static std::string canonicalKey(const std::string& collection);
+
     MemoryStore& store_;
     mutable std::shared_mutex cacheMutex_;
     std::unordered_map<std::string, CollectionCfg> cache_;

파일 크기가 너무 크기때문에 변경 상태를 표시하지 않습니다.
+ 888 - 29
service/src/database_grpc_impl.cpp


+ 93 - 1
service/src/database_grpc_impl.hpp

@@ -9,6 +9,7 @@
 #include "views/view_manager.hpp"
 #include "config/collection_config_manager.hpp"
 #include "security/policy_manager.hpp"
+#include "relations/relation_manager.hpp"
 #include "auth/principal.hpp"
 
 #include <database.grpc.pb.h>
@@ -42,7 +43,8 @@ public:
         EncryptionManager& encryption,
         ViewManager& view_manager,
         CollectionConfigManager& config_manager,
-        PolicyManager& policy_manager
+        PolicyManager& policy_manager,
+        RelationManager& relation_manager
     );
 
     ~DatabaseGrpcImpl() override = default;
@@ -262,6 +264,50 @@ public:
         pb::GetViewInfoResponse* response
     ) override;
 
+    // ===== Relation Operations (v2.11.0 T6b) =====
+
+    grpc::Status CreateRelation(
+        grpc::ServerContext* context,
+        const pb::CreateRelationRequest* request,
+        pb::CreateRelationResponse* response
+    ) override;
+
+    grpc::Status DropRelation(
+        grpc::ServerContext* context,
+        const pb::DropRelationRequest* request,
+        pb::DropRelationResponse* response
+    ) override;
+
+    grpc::Status ListRelations(
+        grpc::ServerContext* context,
+        const pb::ListRelationsRequest* request,
+        pb::ListRelationsResponse* response
+    ) override;
+
+    grpc::Status GetRelationInfo(
+        grpc::ServerContext* context,
+        const pb::GetRelationInfoRequest* request,
+        pb::GetRelationInfoResponse* response
+    ) override;
+
+    // v2.11.0 T5 — "what would happen if I deleted this?" Read-only, gated
+    // as an ordinary per-collection read (not admin) - unlike the relation
+    // management RPCs above.
+    grpc::Status DescribeDelete(
+        grpc::ServerContext* context,
+        const pb::DescribeDeleteRequest* request,
+        pb::DescribeDeleteResponse* response
+    ) override;
+
+    // v2.11.0 T7 — dangling-reference report for one relation. Read-only,
+    // mutates nothing; gated as an ordinary per-collection read on the
+    // relation's CHILD collection, not admin.
+    grpc::Status CheckRelation(
+        grpc::ServerContext* context,
+        const pb::CheckRelationRequest* request,
+        pb::CheckRelationResponse* response
+    ) override;
+
     // ===== Collection Config Operations =====
 
     // v2.9.0 — secondary index management.
@@ -403,6 +449,51 @@ private:
     static std::vector<Filter> fromProtoFilters(const google::protobuf::RepeatedPtrField<pb::Filter>& filters);
     static Query fromProtoQuery(const pb::FindRequest& request);
 
+    // v2.11.0 T6b — re-arm a child collection's reverse index with its
+    // CURRENT set of relations (from relation_manager_), right after a
+    // runtime create/drop. Without this, a relation declared at runtime has
+    // an empty reverse index until the next restart's
+    // DatabaseService::applyRelationDeclarations(), and enforcement silently
+    // permits every delete in the meantime. No-op on a project with no LMDB
+    // substrate for `childQualified`. Advisory: logs and returns on failure,
+    // since the declaration itself already succeeded and persisted.
+    void armRelationsForChild(const std::string& childQualified);
+
+    // -------------------------------------------------------------------
+    // v2.11.0 final review (finding 1) — referential integrity for deletes,
+    // extracted out of Delete() so BATCHDELETE SHARES IT.
+    //
+    // BatchDelete used to call store_.bulkDelete() directly: no canDelete, no
+    // cascade, no set_null. Any caller with ordinary write access could delete
+    // a restrict-protected parent simply by putting its id in a batch, which
+    // defeats the entire feature through a sibling RPC. Exactly the miss class
+    // the v2.7.0 policy audit caught for Upsert/Batch*/Subscribe - found then
+    // by auditing every handler rather than by reading one.
+    //
+    // Both helpers are no-ops for `_`-prefixed system collections and for
+    // projects with no LMDB substrate (the reverse index lives in LMDB), and
+    // both map an LMDB fault to a non-OK status rather than letting it escape
+    // - see finding 2.
+    // -------------------------------------------------------------------
+
+    // Restrict/no_action only: non-OK means this delete must be refused, and
+    // the message names every blocking relation, its child count and sample
+    // child ids. Reads nothing but the reverse index's cursor counts, so it is
+    // cheap enough to run over a whole batch before touching anything.
+    grpc::Status checkRelationDeleteAllowed(const std::string& collection,
+                                            const std::string& id);
+
+    // Cascade/set_null. `handled` false means this collection declares no
+    // destructive relation and the caller must perform the ordinary delete
+    // itself; true means the delete (parent included) has ALREADY been
+    // performed atomically and `parentExisted` says whether there was a
+    // parent to remove. Assumes checkRelationDeleteAllowed() has already
+    // permitted the delete, exactly as Delete's original ordering did.
+    grpc::Status performRelationCascade(const std::string& collection,
+                                        const std::string& id,
+                                        bool& handled,
+                                        bool& parentExisted);
+
     DatabaseService& service_;
     MemoryStore& store_;
     PersistenceManager& persistence_;
@@ -412,6 +503,7 @@ private:
     ViewManager& view_manager_;
     CollectionConfigManager& config_manager_;
     PolicyManager& policy_manager_;
+    RelationManager& relation_manager_;
 
     // v2.7.0 — token -> principal, resolved per call. Populated by
     // DatabaseService from the union of all listeners' keys.

+ 491 - 65
service/src/database_service.cpp

@@ -4,6 +4,8 @@
 #include "storage/document_store.hpp"
 #include "storage/document_store_lmdb.hpp"
 #include "storage/dual_write_mirror.hpp"
+#include "relations/relation_cascade.hpp"
+#include "relations/relation_enforcement.hpp"
 #include "auth/auth_interceptor.hpp"
 #include "auth/principal.hpp"
 #include "tls/cert_generator.hpp"
@@ -195,6 +197,21 @@ bool DatabaseService::initialize() {
         // migration-created documents are stamped with the correct precision.
         config_manager_->loadFromStore();
         policy_manager_->loadFromStore();
+        relation_manager_->loadFromStore();
+        applyRelationDeclarations();
+
+        // v2.11.0 close-out — teach the TTL sweeper about relations. Installed
+        // HERE, after loadFromStore() and the arming pass, deliberately: the
+        // hook consults the declaration cache and the reverse index, and an
+        // expiry that fired before either was ready would have decided from an
+        // empty cache. Until this line the sweeper behaves exactly as it did
+        // pre-v2.11.0 (which is also what every unit fixture that never
+        // installs the hook keeps doing).
+        store_->setTtlExpiryRelationHook(
+            [this](const std::string& qualifiedCollection, const std::string& id,
+                   bool firstAttempt) {
+                return ttlExpiryRelationDecision(qualifiedCollection, id, firstAttempt);
+            });
 
         // v2.9.0 — re-apply persisted index declarations to each project's LMDB
         // store. This is load-bearing, not bookkeeping: the declaration is what
@@ -205,6 +222,30 @@ bool DatabaseService::initialize() {
         // with nothing logged.
         applyIndexDeclarations();
 
+        // v2.11.0 final review (finding 4) — THE POST-REPLAY RE-MIRROR PASS
+        // RUNS HERE, and the position is load-bearing.
+        //
+        // PersistenceManager::recover() (above, before any of the arming
+        // steps) only collects the id list. The pass writes documents through
+        // LmdbDocumentStore::put(), which reads indexed_fields(),
+        // unique_fields() and relations() LIVE and maintains every index
+        // inside the document's own write transaction. Run from inside
+        // recover(), all three maps were still empty, so it repaired the
+        // document and left every posting derived from it stale: an
+        // index-served query then returns the row under its OLD value and
+        // misses it under its new one - silently wrong rows, which is exactly
+        // what putting maintainIndexes() inside the transaction exists to
+        // prevent, and which bit any install with a v2.9 index declared
+        // whether or not it uses relations. It also made the reverse-index
+        // half of relation_cascade.hpp's convergence claim false, and made
+        // UniqueViolation (and therefore the pass's own retry phase)
+        // unreachable.
+        //
+        // Must be after applyRelationDeclarations() AND
+        // applyIndexDeclarations(); backfillIntoDocStore() below already sits
+        // at this point for the same reason. Never throws.
+        persistence_->runPendingRemirror(*store_, recovery_outcome_);
+
         // Run migrations if enabled
         if (config_.migrations.enabled && !config_.migrations.directory.empty()) {
             if (!runMigrations()) {
@@ -247,6 +288,30 @@ bool DatabaseService::initialize() {
             }
         }
 
+        // v2.11.0 close-out — LAST ACT OF THE BOOT PATH, and the position is
+        // load-bearing. Everything above that can bump the drift counter (the
+        // post-replay re-mirror pass, backfillIntoDocStore(), a malformed
+        // legacy collection key) is now behind the baseline, so
+        // mirrorDriftSinceReady() reports only drift a LIVE write caused.
+        // Destructive relation policies gate on that, not on the raw counter:
+        // see MemoryStore::markMirrorDriftBaseline() for the trap this closes
+        // (one unrepairable row disabling every cascade/set_null delete and
+        // every CreateRelation for the process's life, with an error message
+        // telling the operator to restart, which re-incurs it). Reads keep
+        // gating on the RAW counter - a stale row genuinely means MemoryStore
+        // is ahead, and that fallback must stay.
+        store_->markMirrorDriftBaseline();
+        if (store_->mirrorDriftBaseline() > 0) {
+            spdlog::warn("mirror drift at READY: {} - accrued on the boot path (see the "
+                         "ERROR lines above for the exact rows). Reads for this process "
+                         "will be served from MemoryStore rather than LMDB. Destructive "
+                         "relation policies are NOT disabled by this, but the rows named "
+                         "above are still stale in LMDB until each is rewritten through "
+                         "an ordinary write; a restart re-attempts and, if the cause "
+                         "persists, re-incurs the same count.",
+                         store_->mirrorDriftBaseline());
+        }
+
         spdlog::info("Database service initialized successfully");
         return true;
 
@@ -319,7 +384,83 @@ bool DatabaseService::dropProject(const std::string& name, std::string& error) {
         error = "project registry not initialized";
         return false;
     }
-    return projects_->drop(name, error);
+
+    // v2.11.0 close-out — capture the relation declarations belonging to this
+    // project BEFORE the drop. Relations are validated same-project at
+    // creation (no LMDB transaction spans two envs), so `child` and `parent`
+    // are always in the same project as `name` itself; listRelations(name)
+    // therefore names every declaration the drop invalidates. Captured first
+    // because listRelations needs the project to still be nameable and, more
+    // importantly, because un-arming needs the store to still exist.
+    std::vector<RelationInfo> doomed;
+    if (relation_manager_) {
+        try {
+            doomed = relation_manager_->listRelations(name);
+        } catch (const std::exception& e) {
+            spdlog::warn("v2.11 relations: could not enumerate relations of project '{}' "
+                         "before dropping it ({}); its declarations may be left behind",
+                         name, e.what());
+        }
+    }
+
+    if (!projects_->drop(name, error)) return false;
+
+    // ---- The drop succeeded; the declarations now point at a namespace that
+    // no longer exists. Sweep them out of the global `_relations` collection
+    // and un-arm each affected child collection.
+    //
+    // Order (drop first, sweep second) is deliberate: the registry is the one
+    // that refuses "default" and rejects unknown names, so sweeping first would
+    // destroy declarations for a project whose drop then failed. The cost of
+    // this order is that the per-project store is usually already gone by the
+    // time we try to un-arm, which makes the un-arm a no-op - correct, since a
+    // store that no longer exists cannot be maintaining a reverse index. It
+    // still runs, because ProjectStoreRegistry may hold a live
+    // LmdbDocumentStore for a re-created project of the same name.
+    //
+    // ⚠ Advisory, never fatal: the project IS dropped at this point, and a
+    // leftover declaration is litter, not corruption (its child and parent
+    // collections are both gone, so nothing can be enforced against them). A
+    // failure here must not report the drop as failed.
+    for (const auto& r : doomed) {
+        try {
+            std::string err;
+            if (!relation_manager_->dropRelation(r.name, err)) {
+                spdlog::warn("v2.11 relations: could not drop declaration '{}' left by "
+                             "dropped project '{}': {}", r.name, name, err);
+                continue;
+            }
+            const auto rc = resolveCollection(r.child);
+            auto* lmdb = dynamic_cast<smartbotic::db::storage::LmdbDocumentStore*>(
+                docStore(rc.project));
+            if (lmdb != nullptr) {
+                // Empty list = "this collection has no relations", which is what
+                // set_relations replaces the previous list with. Every relation
+                // of a dropped project goes, so clearing the child wholesale is
+                // exactly right here (unlike DropRelation, which must re-arm the
+                // survivors).
+                lmdb->set_relations(rc.collection, {});
+            }
+            spdlog::info("v2.11 relations: dropped declaration '{}' with the project '{}'",
+                         r.name, name);
+        } catch (const std::exception& e) {
+            spdlog::warn("v2.11 relations: could not clean up declaration '{}' after "
+                         "dropping project '{}': {}", r.name, name, e.what());
+        }
+    }
+
+    // ⚠ KNOWN, NOT FIXED HERE (v2.11.0 close-out): _views, _policies and
+    // _collection_meta have the SAME gap - ProjectStoreRegistry::drop() removes
+    // the project's env and nothing else, so a dropped project leaves its view
+    // definitions, access policies and per-collection configs behind in those
+    // global system collections too. Not a shared fix: each manager keys and
+    // caches differently (ViewManager keys `<project>:<name>`, PolicyManager
+    // `<project>:<principal>`, CollectionConfigManager `<project>:<collection>`
+    // and canonicalises the cache key only), so each needs its own sweep, and
+    // PolicyManager's has a security dimension this one does not (a re-created
+    // project inheriting a stale policy). Scoped out deliberately; recorded so
+    // it is not rediscovered as new.
+    return true;
 }
 
 bool DatabaseService::runMigrations() {
@@ -332,8 +473,23 @@ bool DatabaseService::runMigrations() {
     migrationConfig.autoApply = config_.migrations.autoApply;
     migrationConfig.failOnError = config_.migrations.failOnError;
 
-    migrationRunner_ = std::make_unique<MigrationRunner>(*store_, *view_manager_, migrationConfig);
-    return migrationRunner_->runMigrations();
+    migrationRunner_ = std::make_unique<MigrationRunner>(
+        *store_, *view_manager_, *relation_manager_, migrationConfig);
+    const bool ok = migrationRunner_->runMigrations();
+
+    // v2.11.0 close-out — a `create_relation` migration op only PERSISTS the
+    // declaration. Arming the write path and building the reverse index over
+    // rows that already exist is what applyRelationDeclarations() does, and it
+    // has already run for this boot (initialize() calls it before migrations),
+    // so without this a migration-declared relation would maintain no reverse
+    // index and enforce nothing until the NEXT restart - the "declared but not
+    // enforcing" state this feature's self-heal exists to make unreachable.
+    // Idempotent and cheap: re-arming an already-armed child replaces an
+    // identical list, and the bootstrap loop skips any relation whose index
+    // sub-db already exists. Runs even when the migration run reported failure,
+    // because a partially-applied run may still have created a relation.
+    applyRelationDeclarations(/*afterMigrations=*/true);
+    return ok;
 }
 
 void DatabaseService::applyIndexDeclarations() {
@@ -347,9 +503,17 @@ void DatabaseService::applyIndexDeclarations() {
                 dynamic_cast<smartbotic::db::storage::LmdbDocumentStore*>(ds);
             if (lmdb == nullptr) continue;
             lmdb->set_indexed_fields(rc.collection, cfg.indexedFields);
+            // v2.11.0 T11 — arm uniqueFields the same way. This is the fix
+            // for the exact failure mode the type was built to avoid:
+            // uniqueFields has been persisted since T8 with no RPC or proto
+            // surface, which was safe (nothing consulted it); now that
+            // set_unique_fields is reachable, a restart that skipped this
+            // call would silently stop enforcing a constraint every write
+            // handler still advertises as active.
+            lmdb->set_unique_fields(rc.collection, cfg.uniqueFields);
             ++applied;
-            spdlog::info("v2.9 index: {} field(s) active on {}",
-                         cfg.indexedFields.size(), qualified);
+            spdlog::info("v2.9 index: {} field(s) active on {} ({} unique)",
+                         cfg.indexedFields.size(), qualified, cfg.uniqueFields.size());
 
             // Self-heal a declared index whose sub-db is absent. That happens
             // when the KEY FORMAT VERSION in the sub-db prefix changes (v2.9.1
@@ -379,6 +543,162 @@ void DatabaseService::applyIndexDeclarations() {
     }
 }
 
+MemoryStore::TtlExpiryAction DatabaseService::ttlExpiryRelationDecision(
+    const std::string& qualifiedCollection, const std::string& id, bool firstAttempt) {
+    using Action = MemoryStore::TtlExpiryAction;
+
+    // Thin binder over relations/relation_cascade.cpp's ttlExpiryDecision() -
+    // the policy itself lives there so it is unit-testable against the same
+    // fixtures the cascade tests use, without a DatabaseService. All this does
+    // is resolve the per-project LMDB store and supply the replication/events
+    // notifier.
+    if (!relation_manager_ || !config_manager_ || !store_ || !persistence_) {
+        return Action::Proceed;
+    }
+    if (qualifiedCollection.empty() || qualifiedCollection[0] == '_') return Action::Proceed;
+
+    try {
+        const auto rc = resolveCollection(qualifiedCollection);
+        auto* lmdb = dynamic_cast<smartbotic::db::storage::LmdbDocumentStore*>(
+            docStore(rc.project));
+        // No LMDB substrate means no reverse index to consult, so the
+        // pre-v2.11.0 behaviour is all that is available.
+        if (lmdb == nullptr) return Action::Proceed;
+
+        // Replication + Subscribe events, driven per mutation - the same wiring
+        // DatabaseGrpcImpl::performRelationCascade() uses. Without it a
+        // TTL-driven cascade would be invisible to followers and subscribers,
+        // and a follower that never saw the child deletions diverges
+        // permanently (the v2.3.1 class of bug).
+        auto notify = [this](const std::string& coll, const std::string& docId,
+                             const std::optional<Document>& doc, EventType eventType) {
+            notifyReplicationAndEvents(coll, docId, doc, eventType);
+        };
+
+        // firstAttempt -> the refusal (if any) is logged at WARN; a retry of an
+        // already-reported document logs at DEBUG. The sweeper owns that
+        // decision because it is the only thing that knows the document has
+        // been refused before.
+        return smartbotic::database::ttlExpiryDecision(
+            *relation_manager_, *lmdb, *persistence_, *store_, *config_manager_,
+            qualifiedCollection, id, notify, firstAttempt);
+    } catch (const std::exception& e) {
+        // Never expire a parent whose children could not be evaluated.
+        spdlog::error("TTL expiry of '{}/{}': could not resolve its storage ({}) - the "
+                      "document is left in place and will be retried",
+                      qualifiedCollection, id, e.what());
+        return Action::Skip;
+    }
+}
+
+void DatabaseService::applyRelationDeclarations(bool afterMigrations) {
+    // Group by (project, bare child collection) — set_relations() is a
+    // per-collection call on that project's LmdbDocumentStore and replaces
+    // whatever was declared before, so every relation sharing a child
+    // collection must land in one call.
+    std::unordered_map<std::string,
+                        std::vector<smartbotic::db::storage::RelationRef>> byChild;
+    // Track which project each grouping key belongs to alongside the bare
+    // collection name, since the map key alone doesn't carry it.
+    std::unordered_map<std::string, std::pair<std::string, std::string>> keyToProjectCollection;
+
+    for (const auto& r : relation_manager_->listRelations()) {
+        try {
+            const auto rn = resolveCollection(r.name);
+            const auto rc = resolveCollection(r.child);
+            // v2.11.0 T13 — parent resolved to its bare name, same reasoning
+            // as armRelationsForChild's mirror of this construction.
+            const auto pc = resolveCollection(r.parent);
+            const std::string key = rc.project + ":" + rc.collection;
+            // v2.11.0 T13 round 2 — snapshot relationsEnforced at boot-time
+            // arming, same reasoning as armRelationsForChild's mirror of
+            // this construction. `key` is already the qualified child name
+            // configFor expects.
+            const bool enforced = config_manager_->configFor(key).relationsEnforced;
+            byChild[key].push_back(smartbotic::db::storage::RelationRef{
+                rn.collection, r.childField, pc.collection, r.validateOnWrite, enforced});
+            keyToProjectCollection[key] = {rc.project, rc.collection};
+        } catch (const std::exception& e) {
+            // Advisory per relation: one unparseable declaration must not
+            // stop the rest from being armed. Loud, because a skipped
+            // relation silently maintains no reverse index for its child.
+            spdlog::error("v2.11 relations: could not apply declaration '{}': {}",
+                          r.name, e.what());
+        }
+    }
+
+    size_t applied = 0;
+    for (const auto& [key, refs] : byChild) {
+        const auto& [project, collection] = keyToProjectCollection[key];
+        try {
+            // ⚠ v2.11.0 close-out — getOrCreate, not get(), on the POST-MIGRATION
+            // pass. docStore() is projects_->get(), which does NOT create an env,
+            // and a `create_relation` migration in a project that has never been
+            // written to has no env yet: the declaration persisted, this pass
+            // skipped it, and the relation maintained no reverse index and
+            // enforced NOTHING until the next restart - the exact
+            // "declared but not enforcing" state the op was added to avoid, just
+            // narrowed to new projects. A migration declaring a relation IS a
+            // statement that the project exists, and the mirror resolver would
+            // getOrCreate the same env on the first write anyway.
+            //
+            // The BOOT pass deliberately keeps get(): creating envs there would
+            // resurrect the env of a project whose declarations are merely stale.
+            auto* ds = (afterMigrations && projects_)
+                           ? projects_->getOrCreate(project)
+                           : docStore(project);
+            auto* lmdb = dynamic_cast<smartbotic::db::storage::LmdbDocumentStore*>(ds);
+            if (lmdb == nullptr) continue;
+            lmdb->set_relations(collection, refs);
+            ++applied;
+            spdlog::info("v2.11 relations: {} relation(s) active with child '{}:{}'",
+                         refs.size(), project, collection);
+
+            // v2.11.0 T7 self-heal — mirrors applyIndexDeclarations()'s
+            // rebuild of a declared-but-absent index sub-db, for the same
+            // reason: a relation whose reverse index sub-db is missing is
+            // exactly the "declared but not enforcing" bug this task exists
+            // to close. Reachable paths that leave a relation in this state:
+            // a snapshot/backup restored from before the sub-db existed, or
+            // an operator manually dropping the `_relidx1_*` sub-db (there
+            // is no dropRelation-only-the-index tool). CreateRelation's own
+            // bootstrap scan (see database_grpc_impl.cpp) covers the normal
+            // declare-time path; this covers everything else.
+            //
+            // ⚠ v2.11.0 close-out (review minor 2) — the SAME code path is the
+            // NORMAL, expected one for a relation a migration just declared:
+            // migrations run after this pass, so runMigrations() calls it again
+            // and the brand-new relation legitimately has no sub-db yet. Logging
+            // "declared but its sub-db was absent (restored snapshot or a manual
+            // removal)" at WARN there describes a fault that did not happen.
+            // `afterMigrations` distinguishes the two so the text and the level
+            // match the event.
+            for (const auto& ref : refs) {
+                if (lmdb->relation_index_exists(ref.name)) continue;
+                const uint64_t rows =
+                    lmdb->build_relation_index(ref.name, collection, ref.childField);
+                if (afterMigrations) {
+                    spdlog::info("v2.11 relations: built the reverse index for '{}' over {} "
+                                 "existing row(s) in '{}:{}' - a migration declared this "
+                                 "relation on this boot",
+                                 ref.name, rows, project, collection);
+                } else {
+                    spdlog::warn("v2.11 relations: rebuilt '{}' over {} row(s) in '{}:{}' - the "
+                                 "relation was declared but its reverse index sub-db was absent "
+                                 "(restored snapshot predating it, or a manual removal)",
+                                 ref.name, rows, project, collection);
+                }
+            }
+        } catch (const std::exception& e) {
+            spdlog::error("v2.11 relations: could not apply declarations for '{}:{}': {}",
+                          project, collection, e.what());
+        }
+    }
+    if (applied > 0) {
+        spdlog::info("v2.11 relations: applied declarations for {} child collection(s)", applied);
+    }
+}
+
 void DatabaseService::auditSubdbPlacement() {
     if (!projects_) return;
 
@@ -470,9 +790,26 @@ bool DatabaseService::backfillIntoDocStore() {
         for (const auto& doc : docs) {
             std::optional<Document> opt_doc(doc);
             const uint64_t drift_before = mirror_drift_count_.load(std::memory_order_relaxed);
-            smartbotic::db::storage::applyDualWriteMirror(
-                ds, mirror_healthy_, mirror_drift_count_,
-                pc.collection, doc.id, opt_doc, EventType::INSERT);
+            // v2.11.0 T13 round 2 — UniqueViolation/MissingParentReference are
+            // rethrown UNCAUGHT by applyDualWriteMirror (deliberately - see
+            // both exceptions' header comments), unlike a genuine mirror
+            // fault which is swallowed and only shows up as a drift bump
+            // below. Uncaught here would escape this loop, backfillIntoDocStore
+            // and initialize() itself, refusing startup over one rejected row.
+            // Practically unreachable today - backfill only runs migrating a
+            // v1.x dataset, which predates both `_relations` and unique-index
+            // declarations - but catching removes that reasoning burden for
+            // the next reader, same as any other row here that fails.
+            try {
+                smartbotic::db::storage::applyDualWriteMirror(
+                    ds, mirror_healthy_, mirror_drift_count_,
+                    pc.collection, doc.id, opt_doc, EventType::INSERT);
+            } catch (const std::exception& e) {
+                spdlog::error("v2.3 backfill: mirror rejected {}/{}: {} - row stays "
+                              "in MemoryStore only, LMDB does not have it",
+                              qualified, doc.id, e.what());
+                ++total_failures;
+            }
             if (mirror_drift_count_.load(std::memory_order_relaxed) > drift_before) {
                 ++total_failures;
             }
@@ -804,6 +1141,20 @@ DatabaseService::Config DatabaseService::parseConfig(const nlohmann::json& json)
             config.evictionMaxEpisodePercent = memory.value("eviction_max_episode_percent", config.evictionMaxEpisodePercent);
         }
 
+        // v2.11.0 close-out — TTL sweep budgets. Its own block rather than
+        // `memory`, because these bound expiry work per sweep, not memory.
+        if (db.contains("ttl")) {
+            auto& ttl = db["ttl"];
+            config.ttlMaxCandidatesPerSweep =
+                ttl.value("max_candidates_per_sweep", config.ttlMaxCandidatesPerSweep);
+            config.ttlMaxBlockedRetriesPerSweep =
+                ttl.value("max_blocked_retries_per_sweep", config.ttlMaxBlockedRetriesPerSweep);
+            config.ttlBlockedRetrySweeps =
+                ttl.value("blocked_retry_sweeps", config.ttlBlockedRetrySweeps);
+            config.ttlBlockedSummarySweeps =
+                ttl.value("blocked_summary_sweeps", config.ttlBlockedSummarySweeps);
+        }
+
         // Persistence settings
         if (db.contains("persistence")) {
             auto& persistence = db["persistence"];
@@ -988,6 +1339,11 @@ void DatabaseService::setupComponents() {
     storeConfig.memoryEmergencyPercent = config_.memoryEmergencyPercent;
     storeConfig.evictionBurstThreshold = config_.evictionBurstThreshold;
     storeConfig.evictionMaxEpisodePercent = config_.evictionMaxEpisodePercent;
+    // v2.11.0 close-out — TTL sweep budgets (storage.ttl).
+    storeConfig.ttlMaxCandidatesPerSweep = config_.ttlMaxCandidatesPerSweep;
+    storeConfig.ttlMaxBlockedRetriesPerSweep = config_.ttlMaxBlockedRetriesPerSweep;
+    storeConfig.ttlBlockedRetrySweeps = config_.ttlBlockedRetrySweeps;
+    storeConfig.ttlBlockedSummarySweeps = config_.ttlBlockedSummarySweeps;
     store_ = std::make_unique<MemoryStore>(storeConfig);
 
     // Create view manager (cache loaded in initialize() after persistence recovery)
@@ -999,6 +1355,9 @@ void DatabaseService::setupComponents() {
     // v2.8.0 — access policy. Constructed here; its cache is loaded after
     // recovery alongside the view and collection-config caches.
     policy_manager_ = std::make_unique<PolicyManager>(*store_);
+    // v2.11.0 T6a — relation declarations. Constructed here; its cache is
+    // loaded after recovery alongside the view/config/policy caches.
+    relation_manager_ = std::make_unique<RelationManager>(*store_);
     store_->setConfigManager(config_manager_.get());
 
     // v1.9.0 — disk-resident version history. Replaces the in-heap
@@ -1070,59 +1429,10 @@ void DatabaseService::setupComponents() {
         // race window where readers could see a doc in MemoryStore but not
         // yet in LMDB. The callback now does only WAL + replication + events.
 
-        // Queue for replication broadcast
-        if (replication_ && config_.replicationEnabled) {
-            databasepb::ReplicationEntry entry;
-            entry.set_collection(collection);
-            entry.set_document_id(id);
-            // v1.8.0 — stamp our nodeId so peers can attribute the write to
-            // its actual origin. Pre-v1.8 broadcasts left this empty, which
-            // is now treated by receivers as "legacy / unknown origin".
-            entry.set_node_id(config_.nodeId);
-            entry.set_global_timestamp(
-                std::chrono::duration_cast<std::chrono::milliseconds>(
-                    std::chrono::system_clock::now().time_since_epoch()
-                ).count());
-
-            switch (eventType) {
-                case EventType::INSERT:
-                    entry.set_op(databasepb::OP_INSERT);
-                    // Send the full Document envelope (id, collection, data,
-                    // version, timestamps, encryption state). The follower's
-                    // applyReplicatedEntry uses Document::fromJson() which
-                    // expects this envelope — sending only doc->data silently
-                    // dropped every user field on the other side.
-                    if (doc) entry.set_data(doc->toJson().dump());
-                    break;
-                case EventType::UPDATE:
-                    entry.set_op(databasepb::OP_UPDATE);
-                    if (doc) entry.set_data(doc->toJson().dump());
-                    break;
-                case EventType::DELETE:
-                    entry.set_op(databasepb::OP_DELETE);
-                    break;
-                default:
-                    break;
-            }
-
-            replication_->queueForReplication(entry);
-        }
-
-        // Publish event
-        if (events_) {
-            DatabaseEvent event;
-            event.type = eventType;
-            event.collection = collection;
-            event.documentId = id;
-            event.timestamp = std::chrono::duration_cast<std::chrono::milliseconds>(
-                std::chrono::system_clock::now().time_since_epoch()
-            ).count();
-            event.nodeId = config_.nodeId;
-            if (doc) {
-                event.data = doc->data();
-            }
-            events_->publish(event);
-        }
+        // v2.11.0 T12 review (C2) — replication + events extracted into
+        // notifyReplicationAndEvents() so relations/relation_cascade.cpp can
+        // drive them explicitly too. See that method's doc comment.
+        notifyReplicationAndEvents(collection, id, doc, eventType);
     });
 
     // Set up document load callback for LRU eviction recovery.
@@ -1177,7 +1487,7 @@ void DatabaseService::setupComponents() {
     // Create gRPC implementations
     storageImpl_ = std::make_unique<DatabaseGrpcImpl>(
         *this, *store_, *persistence_, *events_, *files_, *encryption_, *view_manager_, *config_manager_,
-        *policy_manager_
+        *policy_manager_, *relation_manager_
     );
     // v2.7.0 — hand the impl the union of every listener's keys so it can
     // resolve a principal per call. A token is only accepted if some listener's
@@ -1200,6 +1510,65 @@ void DatabaseService::setupComponents() {
     );
 }
 
+void DatabaseService::notifyReplicationAndEvents(const std::string& collection,
+                                                  const std::string& id,
+                                                  const std::optional<Document>& doc,
+                                                  EventType eventType) {
+    // Queue for replication broadcast
+    if (replication_ && config_.replicationEnabled) {
+        databasepb::ReplicationEntry entry;
+        entry.set_collection(collection);
+        entry.set_document_id(id);
+        // v1.8.0 — stamp our nodeId so peers can attribute the write to
+        // its actual origin. Pre-v1.8 broadcasts left this empty, which
+        // is now treated by receivers as "legacy / unknown origin".
+        entry.set_node_id(config_.nodeId);
+        entry.set_global_timestamp(
+            std::chrono::duration_cast<std::chrono::milliseconds>(
+                std::chrono::system_clock::now().time_since_epoch()
+            ).count());
+
+        switch (eventType) {
+            case EventType::INSERT:
+                entry.set_op(databasepb::OP_INSERT);
+                // Send the full Document envelope (id, collection, data,
+                // version, timestamps, encryption state). The follower's
+                // applyReplicatedEntry uses Document::fromJson() which
+                // expects this envelope — sending only doc->data silently
+                // dropped every user field on the other side.
+                if (doc) entry.set_data(doc->toJson().dump());
+                break;
+            case EventType::UPDATE:
+                entry.set_op(databasepb::OP_UPDATE);
+                if (doc) entry.set_data(doc->toJson().dump());
+                break;
+            case EventType::DELETE:
+                entry.set_op(databasepb::OP_DELETE);
+                break;
+            default:
+                break;
+        }
+
+        replication_->queueForReplication(entry);
+    }
+
+    // Publish event
+    if (events_) {
+        DatabaseEvent event;
+        event.type = eventType;
+        event.collection = collection;
+        event.documentId = id;
+        event.timestamp = std::chrono::duration_cast<std::chrono::milliseconds>(
+            std::chrono::system_clock::now().time_since_epoch()
+        ).count();
+        event.nodeId = config_.nodeId;
+        if (doc) {
+            event.data = doc->data();
+        }
+        events_->publish(event);
+    }
+}
+
 void DatabaseService::applyReplicatedEntry(const databasepb::ReplicationEntry& entry) {
     try {
         // v1.8.0 — also append the replicated entry to the local WAL, tagged
@@ -1246,9 +1615,66 @@ void DatabaseService::applyReplicatedEntry(const databasepb::ReplicationEntry& e
                                 entry.collection());
                     if (auto* ds = projects_->getOrCreate(pc.project)) {
                         std::optional<Document> opt_doc(doc);
-                        smartbotic::db::storage::applyDualWriteMirror(
-                            ds, mirror_healthy_, mirror_drift_count_,
-                            pc.collection, doc.id, opt_doc, EventType::INSERT);
+                        try {
+                            smartbotic::db::storage::applyDualWriteMirror(
+                                ds, mirror_healthy_, mirror_drift_count_,
+                                pc.collection, doc.id, opt_doc, EventType::INSERT);
+                        } catch (const smartbotic::db::storage::UniqueViolation& e) {
+                            // v2.11.0 T11 finding 1 — deliberate choice, not an
+                            // accident of the generic catch below. By this
+                            // point store_->loadDocument() above has ALREADY
+                            // put the row into MemoryStore (loadDocument
+                            // bypasses the persist callback specifically to
+                            // avoid re-replicating), so unlike every
+                            // client-facing write path there is no "reject the
+                            // whole operation" available: the origin node
+                            // already committed this write and every other
+                            // follower is expected to converge to it too.
+                            // Undoing loadDocument here would silently diverge
+                            // this follower's data from the rest of the
+                            // cluster over a constraint that may not even
+                            // exist on the origin (this follower could have
+                            // declared the unique field locally, after the
+                            // fact — replication carries no guarantee the
+                            // origin shares this node's config). So
+                            // MemoryStore keeps the row — it is the true
+                            // record of what replication decided — and the
+                            // mirror is marked unhealthy ON PURPOSE: this is
+                            // exactly the pre-T11 swallow-and-flip outcome,
+                            // chosen deliberately for this one caller so an
+                            // operator sees the drift (and LMDB-first reads
+                            // fall back to MemoryStore, which has the answer)
+                            // rather than nothing signalling a permanent gap
+                            // between the two stores.
+                            spdlog::error(
+                                "v2.11 replication: unique constraint violated applying "
+                                "{}/{}: {} - MemoryStore holds the row, LMDB does not; "
+                                "marking the mirror unhealthy so reads fall back instead "
+                                "of silently disagreeing with MemoryStore",
+                                entry.collection(), doc.id, e.what());
+                            mirror_healthy_.store(false, std::memory_order_release);
+                            mirror_drift_count_.fetch_add(1, std::memory_order_relaxed);
+                        } catch (const smartbotic::db::storage::MissingParentReference& e) {
+                            // v2.11.0 T13 — same deliberate swallow-and-flip as
+                            // the UniqueViolation catch just above, for the same
+                            // reason: loadDocument() has already committed this
+                            // row into MemoryStore, the origin already accepted
+                            // the write, and this follower may not even have the
+                            // same validate_on_write declaration (or the same
+                            // parent data) as the origin did at the time it
+                            // wrote. MemoryStore keeps the row; the mirror is
+                            // marked unhealthy on purpose so an operator sees
+                            // the drift and reads fall back to MemoryStore
+                            // rather than the two stores silently disagreeing.
+                            spdlog::error(
+                                "v2.11 replication: validate_on_write rejected applying "
+                                "{}/{}: {} - MemoryStore holds the row, LMDB does not; "
+                                "marking the mirror unhealthy so reads fall back instead "
+                                "of silently disagreeing with MemoryStore",
+                                entry.collection(), doc.id, e.what());
+                            mirror_healthy_.store(false, std::memory_order_release);
+                            mirror_drift_count_.fetch_add(1, std::memory_order_relaxed);
+                        }
                     }
                 }
 

+ 78 - 0
service/src/database_service.hpp

@@ -13,6 +13,7 @@
 #include "views/view_manager.hpp"
 #include "config/collection_config_manager.hpp"
 #include "security/policy_manager.hpp"
+#include "relations/relation_manager.hpp"
 
 // LMDB storage substrate (v2.0+). DocumentStore + LmdbEnv live alongside the
 // v1.x MemoryStore, which is still the write entry point and a bounded read
@@ -150,6 +151,19 @@ public:
         // may evict. 0 disables. See MemoryStore::Config for the rationale.
         uint32_t evictionMaxEpisodePercent = 50;
 
+        // v2.11.0 close-out (round-3 review, item 5) — TTL sweep budgets, under
+        // `storage.ttl` in config.json. Mirrors of MemoryStore::Config's fields
+        // of the same name; see there for what each one bounds and why the
+        // blocked-document budget is separate from the fresh one. Present here
+        // because a header calling them tunable while nothing read them from
+        // config meant production always ran the defaults and an operator facing
+        // a blocked-document flood could not change the cadence without a
+        // rebuild.
+        uint32_t ttlMaxCandidatesPerSweep = 10000;
+        uint32_t ttlMaxBlockedRetriesPerSweep = 100;
+        uint32_t ttlBlockedRetrySweeps = 60;
+        uint32_t ttlBlockedSummarySweeps = 300;
+
         // Persistence settings
         uint32_t walSyncIntervalMs = 100;
         uint32_t snapshotIntervalSec = 3600;
@@ -289,6 +303,36 @@ public:
     uint64_t mirrorDriftCount() const noexcept {
         return mirror_drift_count_.load(std::memory_order_relaxed);
     }
+    // v2.11.0 close-out — drift accrued AFTER the boot path finished. What
+    // destructive relation policies gate on; every READ gate keeps using
+    // mirrorDriftCount() above. See MemoryStore::markMirrorDriftBaseline().
+    uint64_t mirrorDriftSinceReady() const noexcept {
+        return store_ == nullptr ? 0 : store_->mirrorDriftSinceBaseline();
+    }
+    uint64_t mirrorDriftAtReady() const noexcept {
+        return store_ == nullptr ? 0 : store_->mirrorDriftBaseline();
+    }
+
+    // v2.11.0 T12 review (C2) — replication queueing + Subscribe event
+    // publication, extracted out of the MemoryStore persistCallback_
+    // lambda (setupComponents()) so a caller that deliberately bypasses
+    // that callback can still drive both explicitly, in the same shape
+    // the callback itself uses. This is what a cascade's `executeCascade`
+    // (relations/relation_cascade.cpp) calls once per child mutation and
+    // once for the parent delete, AFTER MemoryStore has been updated —
+    // mirroring how v2.3.1 fixed replicated-entry apply by explicitly
+    // driving `applyDualWriteMirror` rather than relying on a callback
+    // that had already been bypassed for the same "don't double up on
+    // WAL/mirror" reason. `collection` is the QUALIFIED name (what
+    // `ReplicationEntry`/`DatabaseEvent` both expect); `doc` is null for a
+    // delete. Does NOT touch WAL or the LMDB mirror — those are the
+    // caller's job, done separately, and calling this a second time for
+    // the same mutation would double-queue replication and double-publish
+    // the event, so callers must call it exactly once per mutation.
+    void notifyReplicationAndEvents(const std::string& collection,
+                                    const std::string& id,
+                                    const std::optional<Document>& doc,
+                                    EventType eventType);
 
     // v2.3 Stage F — project CRUD entry points (delegate to registry).
     // The registry is the source of truth for listing / creating / dropping.
@@ -322,6 +366,37 @@ private:
     // persisted in _collection_meta. Must run at boot, after config load.
     void applyIndexDeclarations();
 
+    // v2.11.0 close-out — MemoryStore::TtlExpiryRelationHook. Decides what the
+    // TTL sweeper must do about an expiring parent's children, and PERFORMS the
+    // cascade itself when one is required, by calling the ordinary
+    // relations/relation_cascade.cpp machinery (WAL-before-LMDB) rather than a
+    // second, sweeper-local implementation - a hand-rolled cascade that wrote
+    // only LMDB would have its child deletions resurrected by the next boot's
+    // WAL replay, since MemoryStore is rebuilt from snapshot + WAL and NOT from
+    // LMDB.
+    //
+    // Runs on the cleanup thread with NO MemoryStore lock held (that is
+    // MemoryStore::expireDocuments()' phase-2 contract, and it is what makes
+    // calling back into MemoryStore here safe). Installed in setupComponents();
+    // must never throw (the sweeper catches, but a throw per document would
+    // stop expiry making progress).
+    MemoryStore::TtlExpiryAction ttlExpiryRelationDecision(
+        const std::string& qualifiedCollection, const std::string& id, bool firstAttempt);
+
+    // v2.11.0 T6a — arm each project's LmdbDocumentStore with the relation
+    // declarations loaded by relation_manager_->loadFromStore(), grouped by
+    // CHILD collection. LmdbDocumentStore has no knowledge of RelationManager
+    // and maintains the reverse index for nothing until told to via
+    // set_relations() (see Task 3's report) — this is what tells it, once at
+    // boot. Task 6b's createRelation/dropRelation RPCs must call
+    // set_relations() again on live mutation; this only covers what was
+    // already persisted at startup.
+    // `afterMigrations` only affects LOGGING (see the self-heal block): a
+    // relation whose reverse index sub-db is absent is a fault at boot and the
+    // expected state for a relation a migration just declared, and the same code
+    // handles both.
+    void applyRelationDeclarations(bool afterMigrations = false);
+
     // v2.3 Stage C — atomic rename of <dataDir>/env/ into
     // <dataDir>/projects/default/env/ when the v2.2 layout is detected
     // and the new layout doesn't yet exist. Idempotent. Refuses to start
@@ -364,6 +439,9 @@ private:
     // v2.8.0 — per-project access policy. Owned here so its cache is loaded
     // once, after recovery, alongside ViewManager and CollectionConfigManager.
     std::unique_ptr<PolicyManager> policy_manager_;
+    // v2.11.0 T6a — relation declarations. Owned here so its cache is loaded
+    // once, after recovery, alongside ViewManager/CollectionConfigManager/PolicyManager.
+    std::unique_ptr<RelationManager> relation_manager_;
 
     // gRPC
     std::unique_ptr<DatabaseGrpcImpl> storageImpl_;

파일 크기가 너무 크기때문에 변경 상태를 표시하지 않습니다.
+ 870 - 47
service/src/memory_store.cpp


+ 490 - 1
service/src/memory_store.hpp

@@ -112,6 +112,29 @@ public:
         // the estimator is wrong, so it is an ERROR, not a warning.
         uint32_t evictionMaxEpisodePercent = 50;
 
+        // v2.11.0 close-out (review finding 2) — TTL sweep budgets.
+        //
+        // `ttlMaxCandidatesPerSweep` bounds how many freshly-expired documents
+        // one sweep collects, because the two-phase sweeper materialises the
+        // candidate list before releasing its locks and an unbounded list is a
+        // retention hazard on a large backlog.
+        //
+        // The other two exist because a document a relation refuses to let
+        // expire is NEVER erased from the expiration index - it has to stay
+        // armed so the block can lift - so without separate accounting those
+        // documents refill the fresh budget on every sweep and silently starve
+        // every collection after them in iteration order. Blocked ids are
+        // remembered, skipped for free until `ttlBlockedRetrySweeps` sweeps have
+        // passed, and then retried against their own small budget.
+        //
+        // Wired into the config loader under `storage.ttl` (see parseConfig), so
+        // an operator facing a blocked-document flood can change the cadence
+        // without a rebuild - which is what makes calling them tunable true.
+        uint32_t ttlMaxCandidatesPerSweep = 10000;
+        uint32_t ttlMaxBlockedRetriesPerSweep = 100;
+        uint32_t ttlBlockedRetrySweeps = 60;      // ~1 min at the 1s default
+        uint32_t ttlBlockedSummarySweeps = 300;   // ~5 min at the 1s default
+
         Config()
             : maxMemoryBytes(800ULL * 1024 * 1024)   // 800 MB default
             , expirationCheckIntervalMs(1000)         // Check TTL every second
@@ -194,6 +217,107 @@ public:
                                 std::atomic<bool>* healthy,
                                 std::atomic<uint64_t>* drift);
 
+    /**
+     * v2.11.0 final review (finding 6) — read-only view of the mirror's
+     * health, for code that holds a MemoryStore& but not a DatabaseService&
+     * (relations/relation_cascade.cpp). DatabaseService exposes the same two
+     * values to the gRPC layer; these read the very same atomics.
+     *
+     * ⚠ "No mirror wired" reports HEALTHY with zero drift, deliberately: with
+     * no document store there is no second substrate that could be behind,
+     * so there is nothing to be unhealthy about. This matches
+     * mirrorWriteToDocStore(), which silently no-ops in the same state rather
+     * than treating it as a fault, and it keeps in-process unit fixtures that
+     * never wire a mirror from being refused every destructive operation.
+     */
+    [[nodiscard]] bool mirrorHealthy() const noexcept {
+        return mirrorHealthy_ == nullptr || mirrorHealthy_->load(std::memory_order_relaxed);
+    }
+    [[nodiscard]] uint64_t mirrorDriftCount() const noexcept {
+        return mirrorDriftCount_ == nullptr
+            ? 0
+            : mirrorDriftCount_->load(std::memory_order_relaxed);
+    }
+
+    /**
+     * v2.11.0 close-out — DRIFT ACCRUED AFTER THE BOOT PATH FINISHED, which is
+     * what a destructive relation policy must gate on. Read the whole of this
+     * before using mirrorDriftCount() for a new gate.
+     *
+     * mirrorDriftCount_ is never reset, and the post-replay re-mirror pass
+     * (remirrorDocuments()) deliberately bumps it - and deliberately does NOT
+     * flip health - for every row it could not write. That is correct for
+     * READS: a stale LMDB row means MemoryStore is ahead, and a nonzero drift
+     * count is exactly what routes reads to MemoryStore.
+     *
+     * It was NOT correct as the gate for cascade/set_null deletes and
+     * CreateRelation. Those refuse while drift is nonzero, drift is never
+     * reset, and a row the boot pass cannot repair is re-attempted and
+     * re-failed on EVERY boot - so one unrepairable row permanently disabled
+     * every destructive relation policy service-wide, on a condition
+     * runPendingRemirror() explicitly treats as advisory ("recovery is NOT
+     * failed by this"), and the error text told the operator to restart, which
+     * cannot help.
+     *
+     * DatabaseService::initialize() calls markMirrorDriftBaseline() as its last
+     * act, so everything the boot path accrued - the re-mirror pass, the
+     * backfill, a malformed legacy collection key - lands in the baseline, and
+     * this function reports only drift a LIVE write caused. Live drift still
+     * latches for the process lifetime and still refuses, which is the part
+     * that was right: a mirror that broke under traffic is exactly when a
+     * cascade must not push a stale LMDB body back into MemoryStore.
+     *
+     * ⚠ HONEST LIMIT, accepted deliberately (operator's ruling): the specific
+     * rows the boot pass left stale are still stale, and a cascade that touches
+     * one of them can still overwrite a fresher MemoryStore body with the stale
+     * LMDB copy. Excluding them from the gate trades that narrow, per-row risk
+     * for not disabling every destructive policy on every collection forever.
+     * A per-row remedy would need the failed id set retained, which is the
+     * unbounded-retention problem finding 3 removed. Repair the row (rewrite
+     * it through the ordinary write path) to clear the staleness itself.
+     */
+    void markMirrorDriftBaseline() noexcept {
+        mirrorDriftBaseline_.store(mirrorDriftCount(), std::memory_order_relaxed);
+    }
+    [[nodiscard]] uint64_t mirrorDriftBaseline() const noexcept {
+        return mirrorDriftBaseline_.load(std::memory_order_relaxed);
+    }
+    [[nodiscard]] uint64_t mirrorDriftSinceBaseline() const noexcept {
+        const uint64_t total = mirrorDriftCount();
+        const uint64_t base = mirrorDriftBaseline_.load(std::memory_order_relaxed);
+        // Saturating: the baseline can only ever be <= total (it is a snapshot
+        // of the same monotonic counter), but subtracting unsigned without the
+        // guard would wrap into "billions of drift" if that ever stopped being
+        // true, which fails in the loudest possible wrong direction.
+        return total > base ? total - base : 0;
+    }
+
+    /**
+     * v2.11.0 final review (finding 5) — does a write to `collection` need
+     * T11's pre-write undo snapshot?
+     *
+     * True only when the mirror can REJECT the write (a declared unique field,
+     * or a relation with validateOnWrite) - which is the only situation where
+     * mirrorDocOrUndo()'s undo can ever run. T11 added
+     * `const Document original = it->second;` to update, updateIfVersion,
+     * upsert, patchDocument, setAdd, setRemove and restoreToVersion
+     * unconditionally. Document's copy constructor is
+     * yyjson_mut_val_mut_copy: O(size), a fresh yyjson_mut_doc, taken inside
+     * the collection's unique_lock. On this repo's measured document shapes
+     * that is millions of bytes and milliseconds of allocation per write, on
+     * every install, added to the path v2.8.0 spent a release making cheaper -
+     * and where nothing is declared, the copy is never read.
+     *
+     * Fails SAFE: no mirror wired, an unresolvable collection name, or a
+     * backend that does not implement the query all answer true, so the
+     * snapshot is taken. A wrong `true` costs a copy; a wrong `false` would
+     * leave a rejected write's mutation sitting in MemoryStore unenforced,
+     * which is why the guard inside each undo lambda refuses to restore from a
+     * snapshot that was never taken rather than clobbering the row with an
+     * empty Document.
+     */
+    [[nodiscard]] bool needsUndoSnapshot(const std::string& collection) const;
+
     // ===== Collection Management =====
 
     /**
@@ -265,6 +389,188 @@ public:
      */
     bool remove(const std::string& collection, const std::string& id);
 
+    /**
+     * v2.11.0 T12 — memory-only counterpart to loadDocument()/
+     * loadDocumentWithHistory() for a delete. Cascade (relations/
+     * relation_cascade.cpp) WAL-logs and commits the LMDB mutation itself,
+     * in that order, BEFORE this runs — see the Delete handler's cascade
+     * path in database_grpc_impl.cpp. Calling the ordinary remove() here
+     * would re-log the delete to WAL (a redundant second entry) and
+     * re-attempt the LMDB mirror (a redundant WriteTxn against a row
+     * already gone) for a mutation that is already durable — harmless but
+     * wasteful, and it blurs the "WAL and the LMDB commit each happen
+     * exactly once, in that order" property the crash test depends on.
+     *
+     * Mirrors what remove() does structurally (history, expiration index,
+     * vector, stats) MINUS the two things the caller already handled: the
+     * LMDB mirror and the WAL log. No persist/event callback fires either,
+     * for the same reason — this step is memory bookkeeping only.
+     *
+     * Returns false if the id was not present, which is expected and not
+     * an error: MemoryStore is a bounded cache and may simply not hold a
+     * child the cascade is deleting from LMDB (see the v2.4.4 notes on
+     * eviction). Nothing to undo there.
+     */
+    bool unloadDocument(const std::string& collection, const std::string& id);
+
+    /**
+     * v2.11.0 T12 review (I1) — explicitly re-mirror an id's CURRENT
+     * in-memory state into the LMDB DocumentStore, without touching
+     * MemoryStore itself or the WAL.
+     *
+     * WHY THIS EXISTS: loadDocument()/loadDocumentWithHistory() — what WAL
+     * replay uses to reconstruct MemoryStore for INSERT/UPDATE/UPSERT
+     * entries at boot — do NOT mirror to LMDB (unlike every ordinary write
+     * path, where mirrorDocOrUndo()/mirrorWriteToDocStore() always run
+     * BEFORE the WAL entry is even logged — see update()/insertWithVector()
+     * above). Round 4 correction — that asymmetry is NOT harmless for the
+     * ordinary write path either, and the round-3 claim that "a WAL entry
+     * implies the LMDB write already succeeded" is false:
+     * applyDualWriteMirror() SWALLOWS every LMDB fault except
+     * UniqueViolation (log ERROR, bump drift, flip mirror_healthy_, no
+     * rethrow — storage/dual_write_mirror.hpp), so mirrorDocOrUndo()
+     * returns normally and emitPersist() logs the WAL entry anyway. A WAL
+     * entry therefore only implies "the mirror did not throw
+     * UniqueViolation". So the divergence this pass repairs is NOT
+     * cascade-only: it also occurs on any ordinary write whose mirror fault
+     * was swallowed, and re-mirroring at boot repairs those too. The
+     * divergence merely becomes RELIABLY reachable the moment something
+     * logs a WAL entry BEFORE its LMDB write — which
+     * relations/relation_cascade.cpp's cascade delete deliberately does
+     * (WAL-first, for its own crash-safety reasons: see that file's
+     * header). A crash between that WAL write and the LMDB commit leaves a
+     * WAL UPDATE entry (a set_null or array-pull mutation) whose LMDB
+     * write never actually happened; DELETE entries self-heal because
+     * remove() DOES mirror, but UPDATE/UPSERT entries replayed via
+     * loadDocumentWithHistory() do not, and LMDB then permanently serves
+     * the pre-cascade row until something else happens to rewrite it.
+     *
+     * ⚠ WHEN THIS RUNS, and why it matters (v2.11.0 final review, finding 4).
+     * PersistenceManager::recover() only COLLECTS the id list, into
+     * RecoveryOutcome::pendingRemirror. The pass itself runs from
+     * DatabaseService::initialize(), AFTER applyRelationDeclarations() and
+     * applyIndexDeclarations(), via PersistenceManager::runPendingRemirror().
+     *
+     * It used to run inside recover(), which is line 88 of initialize() while
+     * the declarations are armed at lines ~199 and ~208 - so put() read
+     * indexed_fields(collection), unique_fields(collection) and
+     * relations(collection) while ALL THREE MAPS WERE STILL EMPTY. The pass
+     * therefore repaired the document and corrupted everything derived from
+     * it:
+     *   - no secondary-index maintenance, so the document moved forward in
+     *     LMDB while its postings did not. An index-served query then returns
+     *     the row under its OLD value and misses it under its new one -
+     *     silently wrong rows, precisely what putting maintainIndexes() inside
+     *     the write transaction exists to prevent. This bit ANY install with a
+     *     v2.9 index declared, relations or not.
+     *   - no reverse-index maintenance, so relation_cascade.hpp's claim that
+     *     this pass closes "a live reverse-index posting pointing at a
+     *     now-deleted parent" was false.
+     *   - UniqueViolation was unreachable, which made phase 3's retry and the
+     *     insertion-ORDERED tracking in persistence_manager.cpp dead code
+     *     justified by dead reasoning.
+     * backfillIntoDocStore() already sits at that point in initialize() and is
+     * already guarded for these exceptions, so the ordering is precedented.
+     *
+     * One entry per DISTINCT id that WAL replay applied as
+     * INSERT/UPDATE/UPSERT (deduplicated, and skipped if a later DELETE for
+     * that id was also replayed — remove() already mirrored that) — a
+     * targeted post-replay pass, not a mirror call on
+     * every replayed entry, which would multiply LMDB writes by however
+     * many times a hot document was rewritten since the last snapshot for
+     * no benefit (the ordinary case's LMDB write already happened at
+     * original write time and needs no repeating).
+     *
+     * Safe to call whether or not a given id actually needed it:
+     * re-mirroring an already-correct row is an idempotent overwrite with
+     * the same content.
+     *
+     * ⚠ DOCUMENTS ONLY — never vectors. There is no put_vector() equivalent
+     * here, so a replayed write on a collection with vector_dimension > 0
+     * converges its document and leaves the `_vectors_<collection>` sidecar
+     * stale (SimilaritySearch would score the pre-crash vector). Present
+     * since this pass was introduced; not closed, recorded so "replayed
+     * writes converge" is not read as covering vectors.
+     *
+     * v2.11.0 T12 round-5 — the single-id remirrorDocument() that used to
+     * sit here is GONE: recover() moved to the batched form and it had zero
+     * remaining call sites. Its return semantics had also silently changed
+     * (false, not true, for a system collection or a failed put), so keeping
+     * an untested wrapper with a stale contract was worse than deleting it.
+     */
+    struct RemirrorBatchResult {
+        // Rows whose LMDB write committed as part of this call.
+        uint64_t remirrored = 0;
+        // Rows this call could not re-mirror. Each one is logged at ERROR
+        // naming its collection and id, and bumps the mirror drift counter.
+        // A nonzero value means "those rows stay stale in LMDB", never
+        // "recovery failed" — see below.
+        uint64_t failed = 0;
+    };
+
+    /**
+     * v2.11.0 T12 round-4 — the batched re-mirror pass, and the only entry
+     * point (see the block above for WHY the pass exists at all).
+     *
+     * WHY BATCHED: a per-document call routes to LmdbDocumentStore::put(),
+     * which opens and commits its OWN WriteTxn. The env is opened without
+     * MDB_NOSYNC, so that is one fsync per document, on the boot path,
+     * before sd_notify(READY=1). WAL size is bounded only by
+     * snapshotIntervalSec (3600) and maxWalSizeMb (100), and
+     * --recovery-mode=wal_only can replay the entire history — tens of
+     * thousands of distinct ids on a busy install, against a 10-minute
+     * systemd start watchdog. This commits in chunks of `chunkSize`
+     * instead, via DocumentStore::put_batch() (whose LMDB override holds the
+     * Task 10 beginWrite/put(WriteTxn&)/commit-then-cache discipline), so N
+     * documents in one project cost ceil(N / chunkSize) fsyncs.
+     *
+     * ⚠ STREAMED, not resolved-then-committed (round 5). Documents are
+     * COPIED out of MemoryStore to be written, and `docs` can name an
+     * install's entire history under --recovery-mode=wal_only. Resolving all
+     * of them before the first commit made peak memory O(total bytes
+     * replayed) — tens of GB on this repo's measured multi-MB document
+     * shapes — which traded a slow boot for an OOM-killed one. Each
+     * `chunkSize` window is resolved, committed and RELEASED before the next
+     * is resolved, so peak memory is O(chunkSize documents). Do not hoist
+     * the resolve step back out of the loop.
+     *
+     * ⚠ NEVER THROWS, and that is the whole point of its error handling. A
+     * re-mirror is a REPAIR pass: failing one row must degrade to "that row
+     * stays stale in LMDB", which is exactly the state the pass exists to
+     * improve on and is strictly better than refusing to boot. An escaping
+     * exception here would propagate out of recover() and turn a working
+     * recovery into a deterministic boot loop on data that booted fine
+     * before. Two throws are genuinely reachable: UniqueViolation (the pass
+     * runs against a STALE index, so a unique value that MOVED between two
+     * rows conflicts with the other row's not-yet-re-mirrored posting) and
+     * std::invalid_argument from parseProjectCollection() on a malformed or
+     * legacy collection key (WAL replay itself never parses collection
+     * keys, so such a key replays fine and only this pass would trip on
+     * it). Both are caught per row, and the whole body additionally sits
+     * under a catch-all so an allocation failure while copying a document
+     * cannot escape either.
+     *
+     * ONE BAD ROW MUST NOT POISON ITS CHUNK: a throw from row N aborts that
+     * chunk's transaction (nothing in it committed, and NOTHING is cached —
+     * the Task 10 invariant: to_cache is applied only after a successful
+     * commit), so the whole chunk is then retried ROW BY ROW, each in its
+     * own transaction. The rows that can commit do; only the genuinely bad
+     * one is counted as failed. Rows that fail individually get ONE further
+     * retry pass after every other row has been re-mirrored, which is what
+     * actually resolves the moved-unique-value case: once the row that used
+     * to hold the value has been re-mirrored, its stale posting is gone and
+     * the row that now holds it commits.
+     *
+     * Rows whose collection is missing from MemoryStore, whose id is absent,
+     * or whose collection is `_`-prefixed (system collections are not
+     * mirrored — same rule as applyDualWriteMirror) are silently skipped:
+     * not remirrored, not failed.
+     */
+    RemirrorBatchResult remirrorDocuments(
+        const std::vector<std::pair<std::string, std::string>>& docs,
+        size_t chunkSize = 256,
+        uint64_t chunkBytes = 64ull * 1024 * 1024);
+
     /**
      * Check if a document exists.
      */
@@ -398,13 +704,108 @@ public:
 
     // ===== TTL Management =====
 
+    /**
+     * v2.11.0 close-out — WHAT A TTL EXPIRY MUST DO ABOUT THE EXPIRING
+     * DOCUMENT'S CHILDREN. Returned by the relation hook below, one call per
+     * candidate document, with NO MemoryStore lock held.
+     *
+     * Operator's ruling: a TTL-deleted parent handles its children exactly the
+     * way a manually deleted parent does - like MySQL. Before this, the sweeper
+     * erased the document and mirrored a DELETE with no relation enforcement at
+     * all, so a `restrict`-protected parent silently vanished and orphaned
+     * every child.
+     */
+    enum class TtlExpiryAction {
+        // No relation has anything to say about this document (the common case,
+        // and what a build with no relations declared always returns): expire
+        // it exactly as pre-v2.11.0 did - erase, mirror a DELETE, emit EXPIRE.
+        Proceed,
+        // The expiry must NOT happen: a `restrict` relation blocks it (a manual
+        // delete would have failed with FAILED_PRECONDITION), or the cascade
+        // machinery refused because the LMDB mirror is unhealthy/drifted.
+        // ⚠ The document is left in place WITH ITS EXPIRY STILL ARMED, so a later
+        // sweep retries - which means it OUTLIVES ITS TTL for as long as the
+        // block lasts. That is a retention-policy surprise, so the first refusal
+        // per document logs a WARN naming the relation and the blocking child
+        // count, a periodic summary line reports how many are still stuck, and
+        // Stats::ttlExpiryBlockedByRelation counts refusal events.
+        // The id is remembered (expireDocuments()' ttlBlocked_) so subsequent
+        // sweeps skip it for free - no LMDB read, no log - until its retry is
+        // due, and so it cannot consume the sweep's fresh-candidate budget.
+        Skip,
+        // The hook already performed the whole delete through the ordinary
+        // cascade machinery (relations/relation_cascade.cpp: WAL for the parent
+        // and every child mutation, fsync, one atomic LMDB commit, then the
+        // MemoryStore apply, with replication + Subscribe events driven per
+        // mutation). The sweeper must NOT touch the document again - doing so
+        // would double-log the delete to the WAL and re-mirror it.
+        Handled
+    };
+    // `firstAttempt` is false when this document has already been refused by a
+    // previous sweep and is being retried (see expireDocuments()' blocked set).
+    // It exists so the hook can log its refusal ONCE per document instead of
+    // once per document per sweep: at the 1s default sweep interval, a
+    // persistent `restrict` block on N parents was N WARN lines per second,
+    // indefinitely, and a flood is not a signal.
+    using TtlExpiryRelationHook =
+        std::function<TtlExpiryAction(const std::string& qualifiedCollection,
+                                      const std::string& id,
+                                      bool firstAttempt)>;
+
+    /**
+     * Install the hook above. DatabaseService does this once, at boot, after
+     * the RelationManager is loaded - MemoryStore has no access to
+     * RelationManager, LmdbDocumentStore, PersistenceManager or
+     * CollectionConfigManager, all four of which a cascade needs.
+     *
+     * Unset (the default, and every unit fixture that does not need it) is
+     * exactly the pre-v2.11.0 behaviour.
+     */
+    void setTtlExpiryRelationHook(TtlExpiryRelationHook hook);
+
     /**
      * Expire documents that have passed their TTL.
      * Called automatically by background thread.
-     * @return Number of documents expired
+     *
+     * ⚠ LOCKING (v2.11.0 close-out): this runs in TWO PHASES and the split is
+     * load-bearing. Phase 1 collects candidates under the global read lock plus
+     * each collection's write lock, exactly as before. Phase 2 releases both and
+     * then processes one document at a time, taking that collection's lock
+     * afresh per document. The relation hook is only ever called in phase 2,
+     * with NO MemoryStore lock held, because a cascade re-enters MemoryStore
+     * (applyCascadeToMemory takes each affected collection's lock in turn, and
+     * getOrCreateCollection takes globalMutex_ EXCLUSIVELY): calling it from
+     * inside phase 1 would self-deadlock on globalMutex_ - a std::shared_mutex
+     * is not upgradable - and would deadlock outright against a request thread
+     * whenever the child collection and the parent collection differed.
+     *
+     * ⚠ THE RESIDUAL RACE, stated because it is not fully closed (close-out
+     * review, finding 1). Phase 2 re-checks residency and `expiresAt` under the
+     * collection lock IMMEDIATELY before invoking the hook, and the `Proceed`
+     * path re-checks again in the same critical section as its erase (that one
+     * is airtight). A `Handled` cascade cannot: it spans a WAL fsync and an LMDB
+     * commit and the hook has no view of `expiresAt`, so a write that extends or
+     * clears the TTL DURING the cascade still loses. Closing that needs the
+     * expiry decision and the cascade to share one atomic unit, which nothing in
+     * the current design provides - see the close-out report.
+     *
+     * @return Number of documents actually expired (Skip does not count)
      */
     uint64_t expireDocuments();
 
+    /**
+     * v2.11.0 close-out — how many documents are currently past their TTL and
+     * NOT being expired because a relation blocks the delete. A GAUGE (it falls
+     * when a block lifts), unlike Stats::ttlExpiryBlockedByRelation, which
+     * counts refusal events. Zero on any install that declares no relations.
+     *
+     * Surfaced to operators as GetMemoryStats.ttl_blocked_documents (with the
+     * event counter as ttl_expiry_blocked_events). It is the only visibility
+     * there is for a document outliving its retention policy, so if a caller is
+     * added here it must reach an RPC too.
+     */
+    [[nodiscard]] size_t ttlBlockedDocumentCount() const;
+
     // ===== Statistics =====
 
     struct Stats {
@@ -412,6 +813,19 @@ public:
         uint64_t totalCollections = 0;
         uint64_t estimatedMemoryBytes = 0;
         uint64_t expiredCount = 0;
+        // v2.11.0 close-out — TTL expiry REFUSAL EVENTS, because a relation
+        // blocked the delete (a `restrict` relation with live children, or the
+        // cascade path refusing while the LMDB mirror is unhealthy/drifted).
+        //
+        // ⚠ EVENTS, NOT DISTINCT DOCUMENTS - it is a rate, and it keeps growing
+        // while a block persists, because a blocked document is retried
+        // (every ~60 sweeps; see expireDocuments()). For "how many documents are
+        // stuck past their TTL right now", which is the gauge an operator
+        // actually wants, use MemoryStore::ttlBlockedDocumentCount().
+        //
+        // Nonzero is an operator signal, not an error: the alternative was
+        // silently orphaning the children.
+        uint64_t ttlExpiryBlockedByRelation = 0;
         uint64_t insertCount = 0;
         uint64_t updateCount = 0;
         uint64_t deleteCount = 0;
@@ -764,6 +1178,27 @@ private:
     void mirrorWriteToDocStore(const std::string& collection, const std::string& id,
                                 const std::optional<Document>& doc, EventType eventType);
 
+    // v2.11.0 T11 — mirrors an INSERT/UPDATE doc write, and on a rejected
+    // UniqueViolation runs `undo` (which must restore CollectionData to its
+    // exact pre-mutation state — the document map entry, the expiration
+    // index, and coll->vectors, whichever this call site touched before
+    // mirroring) and then rethrows. `doc` is always present for the
+    // INSERT/UPDATE case this exists for, unlike the DELETE-capable
+    // mirrorWriteToDocStore above.
+    //
+    // Every write path that mutates CollectionData BEFORE calling the mirror
+    // MUST route through this rather than mirrorWriteToDocStore directly:
+    // applyDualWriteMirror now rethrows UniqueViolation instead of
+    // swallowing it (storage/dual_write_mirror.hpp), and MemoryStore mutates
+    // its map before the mirror runs — so without an undo, a rejected write
+    // still leaves the row in memory even though the caller was told it
+    // failed. DELETE call sites don't need this: maintainIndexes only checks
+    // uniqueness on postings being ADDED, never on ones being removed, so a
+    // delete cannot raise UniqueViolation.
+    void mirrorDocOrUndo(const std::string& collection, const std::string& id,
+                         const Document& doc, EventType eventType,
+                         const std::function<void()>& undo);
+
     // v2.0 Stage 5 — mirror a vector write/delete into the LMDB sub-db
     // `_vectors_<collection>`. Called from within the per-collection
     // write lock, after the doc mirror. For INSERT/UPDATE: pass the
@@ -890,8 +1325,62 @@ private:
     // unset, mirrorWriteToDocStore / mirrorVectorToDocStore are no-ops
     // (preserves test/bootstrap paths that don't wire LMDB).
     DocumentStoreResolver docStoreResolver_;
+    // v2.11.0 close-out — see setTtlExpiryRelationHook(). Written once at boot
+    // (before the expiration thread does anything that consults it) and read
+    // from the cleanup thread only.
+    TtlExpiryRelationHook ttlExpiryRelationHook_;
+
+    // v2.11.0 close-out (review finding 2) — documents a relation is currently
+    // refusing to let expire, mapped to the sweep number at which to retry.
+    // Exists so a persistent `restrict` block costs nothing per sweep (no LMDB
+    // read txn, no cursor scan, no log line) and, crucially, cannot refill the
+    // per-sweep candidate cap - which would silently stop every collection after
+    // it in `collections_` iteration order from being swept at all.
+    //
+    // Guarded by its own mutex rather than a collection lock: it is keyed across
+    // collections, so no single collection lock covers it.
+    //
+    // ⚠ LOCK ORDER, and it was WRONG in the first cut (round-3 review, item 1):
+    // ttlBlockedMutex_ is ordered **BEFORE** globalMutex_ and coll->mutex, not
+    // after. Phase 1 of expireDocuments() holds it across the acquisition of
+    // both, so the ONLY legal edge is
+    //
+    //     ttlBlockedMutex_  ->  globalMutex_  ->  coll->mutex
+    //
+    // and it must NEVER be acquired while either of the others is held. The
+    // first cut called clearTtlBlocked() from inside phase 2's
+    // globalMutex_+coll->mutex scope, which closed the cycle: sweeper A holding
+    // ttlBlockedMutex_ and waiting for coll->mutex, sweeper B holding
+    // coll->mutex and waiting for ttlBlockedMutex_ - and because both then pin
+    // globalMutex_ shared forever, the next getOrCreateCollection() for a new
+    // collection blocks on the exclusive acquire and the WHOLE STORE hangs
+    // (restart only). Every other call site takes it with no other lock held,
+    // which is also legal. Nothing taken under it takes any other lock.
+    //
+    // Not reachable through the shipped service today - expirationLoop() is the
+    // only caller of expireDocuments() and no RPC triggers a sweep - but a
+    // second caller, or the obvious next improvement (clearing a block from
+    // update()/dropRelation() so a lifted block retries at once), makes it live.
+    mutable std::mutex ttlBlockedMutex_;
+    std::unordered_map<std::string, uint64_t> ttlBlocked_;
+    // Atomic because they are read and written OUTSIDE ttlBlockedMutex_ (the
+    // sweep number is taken before phase 1 acquires anything), so with two
+    // concurrent sweepers a plain member is a data race on the retry schedule.
+    std::atomic<uint64_t> ttlSweepCounter_{0};
+    std::atomic<uint64_t> ttlBlockedSummarySweep_{0};
+
+    static std::string ttlBlockedKey(const std::string& collection, const std::string& id);
+    void clearTtlBlocked(const std::string& collection, const std::string& id);
+    // Called by dropCollection(), from OUTSIDE globalMutex_ - see the lock-order
+    // note above and the call site.
+    void clearTtlBlockedForCollection(const std::string& collection);
+
     std::atomic<bool>* mirrorHealthy_ = nullptr;
     std::atomic<uint64_t>* mirrorDriftCount_ = nullptr;
+    // v2.11.0 close-out — drift already accrued when the boot path finished.
+    // See markMirrorDriftBaseline(). Owned here rather than in DatabaseService
+    // because relation_cascade.cpp only ever gets a MemoryStore&.
+    std::atomic<uint64_t> mirrorDriftBaseline_{0};
 
     // Per-document last-write WAL sequence map. Populated by the persist
     // callback after each WAL append and by recovery's WAL replay. Used

+ 57 - 1
service/src/migrations/migration_runner.cpp

@@ -11,9 +11,11 @@
 
 namespace smartbotic::database {
 
-MigrationRunner::MigrationRunner(MemoryStore& store, ViewManager& view_manager, Config config)
+MigrationRunner::MigrationRunner(MemoryStore& store, ViewManager& view_manager,
+                                 RelationManager& relation_manager, Config config)
     : store_(store)
     , view_manager_(view_manager)
+    , relation_manager_(relation_manager)
     , config_(std::move(config))
 {
     // Ensure migrations collection exists
@@ -341,6 +343,60 @@ bool MigrationRunner::applyOperation(const nlohmann::json& operation) {
         return true;
     }
 
+    // v2.11.0 close-out — create_relation. Modelled on create_view above: same
+    // file shape, same idempotency contract ("already exists" is success on
+    // replay), and the same decision to validate through the manager rather
+    // than writing the declaration record directly.
+    if (type == "create_relation") {
+        RelationInfo r;
+        r.name = operation.value("name", "");
+        r.child = operation.value("child", "");
+        r.childField = operation.value("child_field", "");
+        r.parent = operation.value("parent", "");
+        if (r.name.empty() || r.child.empty() || r.childField.empty() || r.parent.empty()) {
+            spdlog::error("create_relation: name, child, child_field and parent are all "
+                          "required (got name='{}' child='{}' child_field='{}' parent='{}')",
+                          r.name, r.child, r.childField, r.parent);
+            return false;
+        }
+
+        // VALIDATE, do not coerce — the same call the CreateRelation RPC makes,
+        // for the reason recorded in the header: an unrecognised value used to
+        // become Restrict silently, which stopped being a safe default the
+        // moment cascade/set_null became genuinely destructive. Absent means
+        // the documented default.
+        const std::string onDelete = operation.value("on_delete", "");
+        if (onDelete.empty()) {
+            r.onDelete = OnDelete::Restrict;
+        } else if (auto parsed = parseOnDelete(onDelete); parsed.has_value()) {
+            r.onDelete = *parsed;
+        } else {
+            spdlog::error("create_relation '{}': on_delete must be one of {} (got '{}')",
+                          r.name, kOnDeleteValues, onDelete);
+            return false;
+        }
+        r.validateOnWrite = operation.value("validate_on_write", false);
+
+        std::string err;
+        if (!relation_manager_.createRelation(r, err)) {
+            // Idempotent on replay, exactly as create_view is. Everything else
+            // (cross-project, malformed name, unknown collection) is a real
+            // failure and must fail the migration - a relation the author
+            // believes is armed and is not is the whole class of bug this
+            // feature exists to prevent.
+            if (err.find("already exists") != std::string::npos) {
+                spdlog::debug("create_relation '{}': already exists (idempotent)", r.name);
+                return true;
+            }
+            spdlog::error("create_relation '{}': {}", r.name, err);
+            return false;
+        }
+        spdlog::info("create_relation '{}': {}.{} -> {} (on_delete={})",
+                     r.name, r.child, r.childField, r.parent,
+                     onDelete.empty() ? "restrict" : onDelete);
+        return true;
+    }
+
     if (type == "insert") {
         std::string collection = operation.value("collection", "");
         std::string id = operation.value("id", "");

+ 34 - 1
service/src/migrations/migration_runner.hpp

@@ -1,6 +1,7 @@
 #pragma once
 
 #include "../memory_store.hpp"
+#include "../relations/relation_manager.hpp"
 #include "../views/view_manager.hpp"
 
 #include <filesystem>
@@ -50,6 +51,32 @@ namespace smartbotic::database {
  *   GTE/LT/LTE/IN/CONTAINS/EXISTS/REGEX/SEARCH — via string or integer
  *   op field), and `default_sort` (field + descending). Idempotent:
  *   "already exists" is treated as success on migration replay.
+ * - create_relation: Declare a referential-integrity relation (v2.11.0).
+ *   Required: `name`, `child`, `child_field`, `parent`. Optional:
+ *   `on_delete` (restrict|cascade|set_null|no_action, default restrict) and
+ *   `validate_on_write` (bool, default false). Idempotent: "already
+ *   exists" is treated as success on migration replay.
+ *
+ *   ⚠ Goes through RelationManager::createRelation, deliberately and not as
+ *   a shortcut, so the same-project rule and the on_delete validation apply
+ *   exactly as they do to the CreateRelation RPC. An unrecognised on_delete
+ *   FAILS the migration rather than being coerced to restrict - the same
+ *   decision the RPC made in the v2.11.0 final review, and it matters more
+ *   here: a typo in a file that ships in a deb would otherwise arm restrict
+ *   on every install while the author believed cascade was armed.
+ *
+ *   ⚠ Names are project-qualified the same way every other collection name
+ *   in a migration file is: a bare `users` resolves to `default:users`. A
+ *   consumer declaring relations in a non-default project must write the
+ *   qualified `<project>:<name>` form, and all four of name/child/parent
+ *   must name the SAME project (no LMDB transaction spans two project envs).
+ *
+ *   Arming and the bootstrap scan over existing rows are NOT done here:
+ *   DatabaseService::runMigrations() calls applyRelationDeclarations() after
+ *   the run, which arms every declaration and builds any reverse index whose
+ *   sub-db is absent. Doing it per-op would duplicate that logic; doing it
+ *   nowhere would leave a migration-declared relation unenforced until the
+ *   next restart.
  */
 class MigrationRunner {
 public:
@@ -59,7 +86,12 @@ public:
         bool failOnError = true;
     };
 
-    MigrationRunner(MemoryStore& store, ViewManager& view_manager, Config config);
+    // v2.11.0 close-out — `relation_manager` is what the create_relation op
+    // declares through. A reference, not a pointer: every construction site
+    // has one, and an optional RelationManager would make "the op silently did
+    // nothing" a reachable state.
+    MigrationRunner(MemoryStore& store, ViewManager& view_manager,
+                    RelationManager& relation_manager, Config config);
 
     /**
      * Run all pending migrations.
@@ -110,6 +142,7 @@ private:
 
     MemoryStore& store_;
     ViewManager& view_manager_;
+    RelationManager& relation_manager_;
     Config config_;
 
     /**

+ 114 - 1
service/src/persistence/persistence_manager.cpp

@@ -4,6 +4,8 @@
 
 #include <optional>
 #include <stdexcept>
+#include <unordered_map>
+#include <utility>
 
 namespace smartbotic::database {
 
@@ -203,11 +205,82 @@ RecoveryOutcome PersistenceManager::recover(MemoryStore& store) {
             outcome.failureReason = "Failed to open WAL for replay";
             return false;
         }
-        uint64_t replayed = wal_->replay(fromSequence, [&store](const WalEntry& entry) {
+        // v2.11.0 T12 review (I1) — track every (collection, id) applied as
+        // INSERT/UPDATE/UPSERT during this replay, deduplicated, so it can be
+        // explicitly re-mirrored to LMDB once replay finishes. Erased again
+        // on a later DELETE for the same id: remove() (DELETE's own replay
+        // path) already mirrors, so there is nothing left to re-mirror, and
+        // re-mirroring a doc that replay just deleted from MemoryStore would
+        // be wrong (remirrorDocuments() would correctly skip a missing id,
+        // but dropping it here avoids the pointless lookup). See
+        // MemoryStore::remirrorDocuments()'s doc comment for why this step
+        // exists at all: WAL replay's INSERT/UPDATE path never mirrors on
+        // its own, unlike every ordinary write, and
+        // relations/relation_cascade.cpp's WAL-first cascade can leave a
+        // replayed UPDATE whose LMDB write never actually happened.
+        //
+        // INSERT is tracked too (round 4): loadDocument() does not mirror
+        // either, so a DELETE-then-reinsert of the same id inside one replay
+        // window used to leave LMDB with the row deleted — the DELETE
+        // mirrored, the reinsert did not. Tracking it costs nothing, and it
+        // makes "replayed writes converge" true generally rather than for
+        // two of the three op types.
+        //
+        // Insertion-ORDERED (a vector plus a dedup index, not just a map):
+        // the order rows are re-mirrored in decides which of two rows wins a
+        // transient unique-index conflict, and an unordered_map made that
+        // non-deterministic between boots. WAL order is the order the writes
+        // originally happened in, which is the least surprising choice.
+        //
+        // v2.11.0 final review (finding 4) — this reasoning is LIVE again. It
+        // was dead while the pass ran inside recover(): unique_fields_ was
+        // still empty at that point, so no transient conflict could occur and
+        // the ordering decided nothing. The pass now runs from
+        // DatabaseService::initialize() after applyIndexDeclarations(), so a
+        // unique index really is armed and the order really does pick a
+        // winner.
+        std::vector<std::pair<std::string, std::string>> touched;
+        std::unordered_map<std::string, size_t> touchedIndex;
+        std::vector<bool> touchedAlive;
+        uint64_t replayed = wal_->replay(fromSequence, [&](const WalEntry& entry) {
             applyWalEntry(store, entry);
+            const std::string key = entry.collection + "\x1f" + entry.documentId;
+            if (entry.opType == WalOpType::INSERT || entry.opType == WalOpType::UPDATE ||
+                entry.opType == WalOpType::UPSERT) {
+                auto it = touchedIndex.find(key);
+                if (it == touchedIndex.end()) {
+                    touchedIndex.emplace(key, touched.size());
+                    touched.emplace_back(entry.collection, entry.documentId);
+                    touchedAlive.push_back(true);
+                } else {
+                    touchedAlive[it->second] = true;
+                }
+            } else if (entry.opType == WalOpType::DELETE) {
+                auto it = touchedIndex.find(key);
+                if (it != touchedIndex.end()) touchedAlive[it->second] = false;
+            }
         });
         outcome.walEntriesReplayed = replayed;
         spdlog::info("Replayed {} WAL entries from sequence {}", replayed, fromSequence);
+
+        std::vector<std::pair<std::string, std::string>> toRemirror;
+        toRemirror.reserve(touched.size());
+        for (size_t i = 0; i < touched.size(); ++i) {
+            if (touchedAlive[i]) toRemirror.push_back(touched[i]);
+        }
+        // v2.11.0 final review (finding 4) — HANDED OFF, not run here.
+        // recover() runs at DatabaseService::initialize() line ~88, while
+        // relation and index declarations are armed at ~199/~208. Running the
+        // pass here meant LmdbDocumentStore::put() consulted
+        // indexed_fields()/unique_fields()/relations() while all three were
+        // still empty: the document was repaired and every index posting
+        // derived from it was left stale, which an index-served query then
+        // returns as a silently wrong row. initialize() calls
+        // runPendingRemirror() after arming instead.
+        outcome.pendingRemirror = std::move(toRemirror);
+        spdlog::info("Collected {} document(s) for the post-replay re-mirror pass "
+                     "(runs after index/relation declarations are armed)",
+                     outcome.pendingRemirror.size());
         // Anchor the WAL's sequence_ counter to the snapshot's walSequence.
         // Without this, the steady-state outcome of `truncateBefore(walSeq)`
         // after every snapshot — which deletes every WAL file because all
@@ -370,6 +443,41 @@ RecoveryOutcome PersistenceManager::recover(MemoryStore& store) {
     return outcome;
 }
 
+void PersistenceManager::runPendingRemirror(MemoryStore& store, RecoveryOutcome& outcome) {
+    if (outcome.pendingRemirror.empty()) return;
+
+    // BATCHED, and it never throws — both deliberate, see
+    // MemoryStore::remirrorDocuments(). Per-document transactions meant one
+    // fsync per replayed id on the boot path before sd_notify(READY=1); an
+    // escaping exception meant the service refused to start, permanently, on
+    // data that booted fine before.
+    auto remirror = store.remirrorDocuments(outcome.pendingRemirror);
+    outcome.updatesRemirroredAfterReplay = remirror.remirrored;
+    outcome.updatesRemirrorFailed = remirror.failed;
+    if (remirror.remirrored > 0) {
+        spdlog::info("Re-mirrored {} document(s) to LMDB after WAL replay "
+                    "(closes the window where a WAL-first writer, e.g. a "
+                    "cascade delete, logged an UPDATE whose LMDB write "
+                    "never ran before a crash)", remirror.remirrored);
+    }
+    if (remirror.failed > 0) {
+        spdlog::error("{} document(s) could not be re-mirrored to LMDB after WAL "
+                      "replay (each logged above with its collection and id). "
+                      "Those rows stay stale in LMDB; recovery is NOT failed by "
+                      "this - a repair pass that cannot repair one row must not "
+                      "stop the service from starting.", remirror.failed);
+    }
+
+    // v2.11.0 close-out — release the id list. RecoveryOutcome is held by
+    // DatabaseService for the process lifetime; the pass runs exactly once and
+    // nothing reads the list after it, so keeping it resident pinned ~100
+    // bytes per replayed row for nothing (millions of rows under
+    // --recovery-mode=wal_only). shrink_to_fit() as well as clear(), because
+    // clear() alone keeps the capacity - which is the entire allocation.
+    outcome.pendingRemirror.clear();
+    outcome.pendingRemirror.shrink_to_fit();
+}
+
 uint64_t PersistenceManager::logInsert(const std::string& collection, const Document& doc,
                                         const std::string& originNodeId) {
     if (!running_.load()) return 0;
@@ -397,6 +505,11 @@ uint64_t PersistenceManager::logDelete(const std::string& collection, const std:
     return seq;
 }
 
+void PersistenceManager::flushWal() {
+    if (!wal_) return;
+    wal_->sync();
+}
+
 uint64_t PersistenceManager::logUpsert(const std::string& collection, const Document& doc,
                                         const std::string& originNodeId) {
     if (!running_.load()) return 0;

+ 90 - 0
service/src/persistence/persistence_manager.hpp

@@ -12,6 +12,8 @@
 #include <mutex>
 #include <thread>
 #include <unordered_map>
+#include <utility>
+#include <vector>
 
 namespace smartbotic::database {
 
@@ -48,6 +50,56 @@ struct RecoveryOutcome {
     size_t snapshotsAttempted = 0;
     size_t snapshotsAvailable = 0;
 
+    // v2.11.0 T12 review (I1) — how many distinct (collection, id) pairs
+    // WAL replay applied as INSERT/UPDATE/UPSERT and then explicitly
+    // re-mirrored to LMDB afterward (one batched
+    // MemoryStore::remirrorDocuments() pass from
+    // PersistenceManager::recover() — see that method and
+    // remirrorDocuments()'s doc comment for why this step exists: WAL
+    // replay's INSERT/UPDATE path does not mirror on its own, unlike every
+    // ordinary write, and relations/relation_cascade.cpp's WAL-first
+    // cascade can leave a replayed UPDATE whose LMDB write never actually
+    // ran). Exposed for tests and operator visibility; not itself an
+    // error signal — a nonzero count on a box that has never run a
+    // cascade just means ordinary UPDATEs got a redundant (harmless,
+    // idempotent) re-mirror.
+    uint64_t updatesRemirroredAfterReplay = 0;
+
+    // v2.11.0 T12 round-4 — rows the post-replay re-mirror pass could NOT
+    // write to LMDB (a malformed/legacy collection key, or a unique-index
+    // conflict that survived the retry pass). Each is logged at ERROR with
+    // its collection and id and bumps the mirror drift counter. Nonzero
+    // means "those rows stay stale in LMDB", NOT "recovery failed": a
+    // repair pass that cannot repair one row must never stop the service
+    // from starting, which is what an unguarded throw here used to do.
+    uint64_t updatesRemirrorFailed = 0;
+
+    // v2.11.0 final review (finding 4) — the (collection, id) pairs WAL
+    // replay applied as INSERT/UPDATE/UPSERT and that still need mirroring
+    // to LMDB. recover() only COLLECTS this list; the pass itself runs from
+    // DatabaseService::initialize() via runPendingRemirror(), AFTER
+    // applyRelationDeclarations()/applyIndexDeclarations().
+    //
+    // ⚠ THE ORDERING IS THE POINT, not a refactor. Run from inside
+    // recover(), the pass wrote documents while indexed_fields_,
+    // unique_fields_ and relations_ were all still empty, so it repaired the
+    // document and left every index and reverse-index posting derived from
+    // it stale - silently wrong rows on any install with a v2.9 index
+    // declared. See MemoryStore::remirrorDocuments()'s doc comment.
+    //
+    // Empty after a recovery that replayed nothing, and after ForceEmpty.
+    //
+    // ⚠ RELEASED BY runPendingRemirror() ONCE THE PASS HAS RUN (v2.11.0
+    // close-out). RecoveryOutcome lives as DatabaseService::recovery_outcome_
+    // for the whole process lifetime, so leaving this populated pinned the
+    // entire replayed id set - ~100 bytes per entry, and the entry count is
+    // an install's whole history under --recovery-mode=wal_only - resident
+    // for nothing, since the pass runs exactly once and nothing reads the
+    // list afterwards. The COUNTS (updatesRemirroredAfterReplay /
+    // updatesRemirrorFailed) are what operator surfaces report and they are
+    // deliberately kept.
+    std::vector<std::pair<std::string, std::string>> pendingRemirror;
+
     bool isNonTrivial() const {
         return kind != Kind::TrivialSuccess && kind != Kind::FreshInstall;
     }
@@ -118,6 +170,24 @@ public:
      */
     RecoveryOutcome recover(MemoryStore& store);
 
+    /**
+     * v2.11.0 final review (finding 4) — run the post-replay re-mirror pass
+     * over outcome.pendingRemirror and record its counts back onto `outcome`.
+     *
+     * MUST be called AFTER the caller has armed relation and index
+     * declarations (DatabaseService::initialize() calls it right after
+     * applyIndexDeclarations()), because LmdbDocumentStore::put() reads
+     * indexed_fields()/unique_fields()/relations() live and maintains every
+     * index inside the document's own write transaction. Running it earlier -
+     * which is what recover() itself used to do - repairs the document and
+     * leaves every posting derived from it stale.
+     *
+     * Idempotent, and never throws (MemoryStore::remirrorDocuments guarantees
+     * that; a repair pass must not be able to stop the service from
+     * starting). Safe to call with an empty list.
+     */
+    void runPendingRemirror(MemoryStore& store, RecoveryOutcome& outcome);
+
     /**
      * Log an insert operation.
      * `originNodeId` (v1.8.0) is the node that originally produced this
@@ -175,6 +245,26 @@ public:
     uint64_t logVecDelete(const std::string& collection, const std::string& docId,
                           const std::string& originNodeId = "");
 
+    /**
+     * v2.11.0 T12 — force the WAL to fsync NOW, synchronously.
+     *
+     * The background walSyncLoop() fsyncs every walSyncIntervalMs (100ms by
+     * default), which is fine for the ordinary per-document write path
+     * (that path mirrors to LMDB first, under MemoryStore's collection
+     * lock, and only logs to WAL — durably or not — afterward; see
+     * memory_store.cpp's remove()/update()). A cascade delete is the one
+     * place that ordering is deliberately reversed: WAL entries for the
+     * parent and every child mutation must be durable (written AND
+     * fsynced) BEFORE the LMDB transaction that mutates them commits,
+     * because MemoryStore is rebuilt at boot from snapshot + WAL replay,
+     * NOT from LMDB. Relying on the 100ms timer would leave a window where
+     * a crash lands entries in the WAL FILE but not yet fsynced to disk —
+     * indistinguishable, after a real crash, from "never written" — so the
+     * cascade path calls this explicitly instead of waiting on the timer.
+     * See relations/relation_cascade.cpp's writeCascadeWal().
+     */
+    void flushWal();
+
     /**
      * Force an immediate snapshot.
      * Blocks until snapshot is complete.

+ 21 - 4
service/src/persistence/wal.cpp

@@ -8,6 +8,9 @@
 #include <iomanip>
 #include <sstream>
 
+#include <fcntl.h>
+#include <unistd.h>
+
 namespace smartbotic::database {
 
 // ===== CRC32 Implementation =====
@@ -443,10 +446,24 @@ uint64_t WriteAheadLog::append(WalEntry entry) {
 }
 
 void WriteAheadLog::sync() {
-    if (currentFile_.is_open()) {
-        currentFile_.flush();
-        // Note: For true durability, we'd use fsync() here
-        // std::filesystem doesn't provide this, would need OS-specific code
+    if (!currentFile_.is_open()) return;
+    currentFile_.flush();
+    // v2.11.0 T12 — real fsync, not just flush(). flush() only pushes the
+    // C++ stream buffer into the OS page cache; on a crash before the OS
+    // itself writes that page back to disk, a "flushed" WAL entry is
+    // indistinguishable from one never written at all. The cascade
+    // WAL-first design (PersistenceManager::flushWal(), called before the
+    // LMDB commit — see relations/relation_cascade.cpp) depends on
+    // durability being REAL here, not merely buffered: a crash between an
+    // LMDB-only cascade write and an un-fsynced "durable" WAL entry is
+    // exactly the resurrection bug this task exists to close. std::ofstream
+    // has no portable fsync, so this reopens the same path by fd — cheap
+    // (one open/fsync/close), and correct on the POSIX targets this
+    // project ships for.
+    const int fd = ::open(currentFilePath().c_str(), O_WRONLY);
+    if (fd >= 0) {
+        ::fsync(fd);
+        ::close(fd);
     }
 }
 

+ 489 - 0
service/src/relations/relation_cascade.cpp

@@ -0,0 +1,489 @@
+#include "relation_cascade.hpp"
+
+#include <chrono>
+#include <limits>
+#include <unordered_map>
+
+#include <spdlog/spdlog.h>
+#include <spdlog/fmt/fmt.h>
+
+#include "../config/collection_config_manager.hpp"
+#include "../memory_store.hpp"
+#include "../persistence/persistence_manager.hpp"
+#include "../project_addressing.hpp"
+#include "../storage/document_store_lmdb.hpp"
+#include "../storage/filter_eval.hpp"
+#include "../storage/lmdb_txn.hpp"
+#include "relation_enforcement.hpp"
+
+namespace smartbotic::database {
+
+namespace {
+
+// v2.11.0 T12 review (I3) — mirrors MemoryStore::currentTimeFor() exactly
+// (memory_store.cpp): "ms" (the fallback) unless the collection's
+// CollectionCfg declares "ns". planCascade() only has an
+// LmdbDocumentStore&, not a MemoryStore&, so it cannot call
+// currentTimeFor() directly (private, and the wrong layer to reach across
+// to for one timestamp) — hence the duplication here rather than a shared
+// call. Getting this wrong is not a rounding error: resolveFilterValue()
+// returns Document::updatedAt raw, with no unit normalisation on any read/
+// sort/filter path (the 10^15-magnitude heuristic exists in exactly two
+// places — eviction's hot-write floor and migrateCollectionTimestamps'
+// idempotency guard — neither of which is a general-purpose read path). A
+// millisecond stamp on an "ns"-precision collection (the default since
+// v2.2.0) sorts as if written in 1970 and never matches an
+// `_updated_at > <ns-value>` filter.
+uint64_t nowForCollection(CollectionConfigManager& configManager,
+                          const std::string& qualifiedCollection) {
+    const auto cfg = configManager.configFor(qualifiedCollection);
+    const auto now = std::chrono::system_clock::now().time_since_epoch();
+    if (cfg.timestampPrecision == "ns") {
+        return static_cast<uint64_t>(
+            std::chrono::duration_cast<std::chrono::nanoseconds>(now).count());
+    }
+    return static_cast<uint64_t>(
+        std::chrono::duration_cast<std::chrono::milliseconds>(now).count());
+}
+
+std::vector<std::string> splitDotPath(const std::string& path) {
+    std::vector<std::string> parts;
+    size_t start = 0;
+    while (start <= path.size()) {
+        const size_t dot = path.find('.', start);
+        parts.push_back(path.substr(start, dot == std::string::npos ? std::string::npos : dot - start));
+        if (dot == std::string::npos) break;
+        start = dot + 1;
+    }
+    return parts;
+}
+
+// Set `root`'s value at `path` to null. Mirrors filter_eval::getJsonPath's
+// read semantics: dotted paths descend OBJECTS only, arrays are never
+// indexed into. No-ops (rather than throws) if the path doesn't resolve —
+// the caller only reaches here after confirming the field held a live
+// reference, but a defensive no-op is cheaper than a second round of
+// validation and strictly safer than crashing mid-cascade over a race.
+void setDotPathNull(nlohmann::json& root, const std::string& path) {
+    const auto parts = splitDotPath(path);
+    if (parts.empty()) return;
+    nlohmann::json* cur = &root;
+    for (size_t i = 0; i + 1 < parts.size(); ++i) {
+        if (!cur->is_object()) return;
+        auto it = cur->find(parts[i]);
+        if (it == cur->end()) return;
+        cur = &(*it);
+    }
+    if (!cur->is_object()) return;
+    (*cur)[parts.back()] = nullptr;
+}
+
+// Remove every string array element equal to `id` at `path`, in place.
+// No-op if the path doesn't resolve to an array.
+void pullDotPathArrayId(nlohmann::json& root, const std::string& path, const std::string& id) {
+    const auto parts = splitDotPath(path);
+    if (parts.empty()) return;
+    nlohmann::json* cur = &root;
+    for (size_t i = 0; i + 1 < parts.size(); ++i) {
+        if (!cur->is_object()) return;
+        auto it = cur->find(parts[i]);
+        if (it == cur->end()) return;
+        cur = &(*it);
+    }
+    if (!cur->is_object()) return;
+    auto it = cur->find(parts.back());
+    if (it == cur->end() || !it->is_array()) return;
+    nlohmann::json pruned = nlohmann::json::array();
+    for (const auto& el : *it) {
+        if (!(el.is_string() && el.get<std::string>() == id)) pruned.push_back(el);
+    }
+    *it = std::move(pruned);
+}
+
+// Document-level wrappers: materialise the full tree, mutate the copy,
+// re-encode. Document::data()/set_data() are the sanctioned way to touch
+// the binary-backed representation from outside doc_binary.{hpp,cpp} — see
+// document.hpp's class comment.
+void setFieldNullInPlace(Document& doc, const std::string& path) {
+    nlohmann::json data = doc.data();
+    setDotPathNull(data, path);
+    doc.set_data(data);
+}
+
+void pullArrayIdInPlace(Document& doc, const std::string& path, const std::string& id) {
+    nlohmann::json data = doc.data();
+    pullDotPathArrayId(data, path, id);
+    doc.set_data(data);
+}
+
+// Running state for one child document while planCascade() folds every
+// relation that touches it. Keyed by (bare child collection, child id) so
+// two different relations from the same parent landing on the same child
+// accumulate onto ONE working copy instead of one clobbering the other's
+// edit — see the header comment on the merge rule.
+struct WorkingChild {
+    std::string collectionQualified;
+    std::string collectionBare;
+    std::string childId;
+    bool deleted = false;
+    Document doc;
+};
+
+} // namespace
+
+CascadePlan planCascade(RelationManager& relations,
+                        smartbotic::db::storage::LmdbDocumentStore& store,
+                        CollectionConfigManager& configManager,
+                        const std::string& qualifiedParentCollection,
+                        const std::string& parentId) {
+    CascadePlan plan;
+
+    try {
+        const auto parentRc = resolveCollection(qualifiedParentCollection);
+        plan.parentHadVector = store.get_vector(parentRc.collection, parentId).has_value();
+    } catch (const std::exception&) {
+        // Malformed parent name would already have failed the delete
+        // upstream (canDelete/the handler's own resolveCollection); treat
+        // as "no vector" rather than let a cosmetic lookup abort planning.
+    }
+
+    std::vector<WorkingChild> working;
+    std::unordered_map<std::string, size_t> indexOf;   // "<bare-coll>\x1f<id>" -> working[]
+
+    for (const auto& r : relations.relationsWithParent(qualifiedParentCollection)) {
+        if (r.onDelete != OnDelete::Cascade && r.onDelete != OnDelete::SetNull) continue;
+
+        std::string bareRelation;
+        smartbotic::database::ResolvedCollection childRc;
+        try {
+            bareRelation = resolveCollection(r.name).collection;
+            childRc = resolveCollection(r.child);
+        } catch (const std::exception& e) {
+            spdlog::error("relations: cascade planning skipped malformed relation '{}': {}",
+                         r.name, e.what());
+            continue;
+        }
+
+        // Unbounded — relation_index_children truncates at `limit`, and a
+        // cascade that silently dropped children past some cap would be
+        // worse than no cascade at all (a dangling reference an operator
+        // can find with `relations check`, versus data quietly left
+        // referencing a deleted parent forever). SIZE_MAX asks for
+        // everything; see the .hpp file header.
+        const auto childIds = store.relation_index_children(
+            bareRelation, parentId, std::numeric_limits<size_t>::max());
+
+        for (const auto& childId : childIds) {
+            const std::string key = childRc.collection + "\x1f" + childId;
+            size_t idx;
+            auto found = indexOf.find(key);
+            if (found == indexOf.end()) {
+                auto childDoc = store.get(childRc.collection, childId);
+                if (!childDoc) continue;   // raced away between index read and here
+                WorkingChild w;
+                w.collectionQualified = r.child;
+                w.collectionBare = childRc.collection;
+                w.childId = childId;
+                w.doc = std::move(*childDoc);
+                working.push_back(std::move(w));
+                idx = working.size() - 1;
+                indexOf.emplace(key, idx);
+            } else {
+                idx = found->second;
+            }
+
+            WorkingChild& w = working[idx];
+            if (w.deleted) continue;   // already slated for deletion — nothing further to do
+
+            auto val = smartbotic::db::storage::filter_eval::resolveFilterValue(w.doc, r.childField);
+            if (!val) continue;   // field no longer present — race, skip this relation's effect
+
+            if (val->is_array()) {
+                // ⚠ THE ARRAY RULE — cascade and set_null collapse. See the
+                // .hpp file header: pull the id, keep the document, no
+                // matter which policy is declared.
+                pullArrayIdInPlace(w.doc, r.childField, parentId);
+            } else if (val->is_string() && val->get<std::string>() == parentId) {
+                if (r.onDelete == OnDelete::Cascade) {
+                    // v2.11.0 T12 review (I2) — this child is about to be
+                    // DELETED. Before committing to that, check whether IT
+                    // is itself protected by a restrict relation (a
+                    // grandchild of the delete being planned) — otherwise
+                    // this cascade would silently destroy a row that
+                    // relation exists to protect, with canDelete() never
+                    // having seen it (canDelete() only ever evaluates the
+                    // ORIGINAL parent/id). See the .hpp file header's "NOT
+                    // RECURSIVE" section for why this refuses rather than
+                    // cascading further, and why it is unconditional
+                    // (not gated by the child's own relationsEnforced).
+                    auto blocks = findRelationBlocks(relations, store, w.collectionQualified, w.childId);
+                    if (!blocks.empty()) {
+                        throw CascadeBlocked(
+                            "cascade delete of '" + qualifiedParentCollection + "/" + parentId +
+                            "' would delete '" + w.collectionQualified + "/" + w.childId +
+                            "', which is itself protected: " +
+                            formatRelationBlockError(w.collectionQualified, w.childId, blocks));
+                    }
+                    w.deleted = true;
+                    continue;   // don't touch w.doc further — it's being deleted
+                }
+                setFieldNullInPlace(w.doc, r.childField);
+            } else {
+                continue;   // no longer actually references parentId — race, skip
+            }
+            w.doc.version += 1;
+            w.doc.updatedAt = nowForCollection(configManager, w.collectionQualified);
+        }
+    }
+
+    plan.mutations.reserve(working.size());
+    for (auto& w : working) {
+        CascadeMutation mut;
+        mut.childCollectionQualified = w.collectionQualified;
+        mut.childCollectionBare = w.collectionBare;
+        mut.childId = w.childId;
+        if (w.deleted) {
+            mut.kind = CascadeMutation::Kind::DeleteChild;
+            mut.hadVector = store.get_vector(w.collectionBare, w.childId).has_value();
+        } else {
+            mut.kind = CascadeMutation::Kind::UpdateChild;
+            mut.updatedDoc = std::move(w.doc);
+        }
+        plan.mutations.push_back(std::move(mut));
+    }
+
+    return plan;
+}
+
+void writeCascadeWal(PersistenceManager& persistence,
+                     const std::string& qualifiedParentCollection,
+                     const std::string& parentId,
+                     const CascadePlan& plan) {
+    for (const auto& m : plan.mutations) {
+        if (m.kind == CascadeMutation::Kind::DeleteChild) {
+            persistence.logDelete(m.childCollectionQualified, m.childId);
+            if (m.hadVector) persistence.logVecDelete(m.childCollectionQualified, m.childId);
+        } else {
+            persistence.logUpdate(m.childCollectionQualified, *m.updatedDoc);
+        }
+    }
+    persistence.logDelete(qualifiedParentCollection, parentId);
+    if (plan.parentHadVector) persistence.logVecDelete(qualifiedParentCollection, parentId);
+
+    // MANDATORY — see the .hpp file header's WAL-before-LMDB explanation.
+    // Every entry above must be durable before commitCascadeLmdb() runs.
+    persistence.flushWal();
+}
+
+bool commitCascadeLmdb(smartbotic::db::storage::LmdbDocumentStore& store,
+                       const std::string& qualifiedParentCollection,
+                       const std::string& parentId,
+                       const CascadePlan& plan) {
+    const auto parentRc = resolveCollection(qualifiedParentCollection);
+
+    smartbotic::db::storage::WriteTxn wtxn = store.beginWrite();
+    std::vector<std::pair<std::string, unsigned int>> to_cache;
+
+    for (const auto& m : plan.mutations) {
+        if (m.kind == CascadeMutation::Kind::DeleteChild) {
+            store.del(wtxn, m.childCollectionBare, m.childId, to_cache);
+            if (m.hadVector) store.del_vector(wtxn, m.childCollectionBare, m.childId, to_cache);
+        } else {
+            store.put(wtxn, m.childCollectionBare, m.childId, *m.updatedDoc, to_cache);
+        }
+    }
+
+    const bool parentExisted = store.del(wtxn, parentRc.collection, parentId, to_cache);
+    if (plan.parentHadVector) store.del_vector(wtxn, parentRc.collection, parentId, to_cache);
+
+    // ONE commit for the whole cascade — all-or-nothing. commitAndCache()
+    // caches every handle in to_cache only AFTER this commit succeeds.
+    store.commitAndCache(wtxn, to_cache);
+    return parentExisted;
+}
+
+void applyCascadeToMemory(MemoryStore& memStore,
+                          const std::string& qualifiedParentCollection,
+                          const std::string& parentId,
+                          const CascadePlan& plan,
+                          bool parentExisted,
+                          const CascadeNotifyFn& notify) {
+    for (const auto& m : plan.mutations) {
+        if (m.kind == CascadeMutation::Kind::DeleteChild) {
+            memStore.unloadDocument(m.childCollectionQualified, m.childId);
+            if (notify) notify(m.childCollectionQualified, m.childId, std::nullopt, EventType::DELETE);
+        } else {
+            memStore.loadDocumentWithHistory(m.childCollectionQualified, *m.updatedDoc);
+            if (notify) notify(m.childCollectionQualified, m.childId, m.updatedDoc, EventType::UPDATE);
+        }
+    }
+    memStore.unloadDocument(qualifiedParentCollection, parentId);
+    // Gated on the LMDB-confirmed parentExisted, not MemoryStore's own
+    // return — see the .hpp file header. MemoryStore may not have had the
+    // parent cached (evicted) even though LMDB genuinely deleted it, or
+    // vice versa; LMDB is the source of truth for "did this really happen".
+    if (notify && parentExisted) {
+        notify(qualifiedParentCollection, parentId, std::nullopt, EventType::DELETE);
+    }
+}
+
+bool executeCascade(RelationManager& relations,
+                    smartbotic::db::storage::LmdbDocumentStore& store,
+                    PersistenceManager& persistence,
+                    MemoryStore& memStore,
+                    CollectionConfigManager& configManager,
+                    const std::string& qualifiedParentCollection,
+                    const std::string& parentId,
+                    const CascadeNotifyFn& notify) {
+    // v2.11.0 final review (finding 6) — REFUSE THE DESTRUCTIVE PATH while
+    // the mirror is unhealthy or has drifted.
+    //
+    // planCascade() builds every `updatedDoc` from the LMDB copy
+    // (store.get(...)), and applyCascadeToMemory() then pushes that copy into
+    // MemoryStore via loadDocumentWithHistory(). In the unhealthy/drifted
+    // state LMDB may be BEHIND - that is precisely why reads fall back to
+    // MemoryStore and why six read gates in database_grpc_impl.cpp test this
+    // pair - so the cascade would overwrite MemoryStore's fresher child body
+    // with a stale one. That is real data loss, produced by a repair-shaped
+    // operation, on a state this codebase treats as routine.
+    //
+    // Thrown as CascadeBlocked so the Delete handler maps it to
+    // FAILED_PRECONDITION with the message intact: nothing has been written
+    // anywhere at this point (this is before writeCascadeWal()), which is
+    // exactly the contract CascadeBlocked already carries. Same refusal
+    // shape, and the same reasoning, as the CreateIndex(unique) and
+    // CreateRelation gates.
+    // ⚠ v2.11.0 close-out — the drift half of this gate is
+    // mirrorDriftSinceBaseline(), NOT mirrorDriftCount(). Drift is never
+    // reset and the boot path's post-replay re-mirror pass bumps it for every
+    // row it cannot repair - a condition runPendingRemirror() deliberately
+    // treats as advisory and which recurs on every boot, so the raw counter
+    // made ONE unrepairable row refuse every cascade/set_null delete
+    // service-wide, permanently, while the message above told the operator to
+    // restart. See MemoryStore::markMirrorDriftBaseline() for the full
+    // reasoning and for the residual per-row risk this accepts.
+    if (!memStore.mirrorHealthy() || memStore.mirrorDriftSinceBaseline() != 0) {
+        throw CascadeBlocked(
+            "refusing the cascade/set_null delete of '" + qualifiedParentCollection +
+            "/" + parentId + "': the LMDB mirror is " +
+            (memStore.mirrorHealthy()
+                 ? "drifted since startup (" +
+                       std::to_string(memStore.mirrorDriftSinceBaseline()) +
+                       " row(s) a live write left stale)"
+                 : "unhealthy") +
+            ", so the child documents this cascade would rewrite are read from a substrate "
+            "that may be behind MemoryStore - applying them would overwrite fresher data "
+            "with older data. Restrict/no_action deletes are unaffected. Fix the mirror "
+            "(see the ERROR lines naming the failed collection and id): drift accrued "
+            "under live traffic latches for this process, so a restart clears it ONLY if "
+            "the underlying cause is gone. Or use on_delete=no_action if the dangling "
+            "reference is acceptable.");
+    }
+
+    const CascadePlan plan = planCascade(relations, store, configManager,
+                                         qualifiedParentCollection, parentId);
+    writeCascadeWal(persistence, qualifiedParentCollection, parentId, plan);
+    const bool parentExisted = commitCascadeLmdb(store, qualifiedParentCollection, parentId, plan);
+    applyCascadeToMemory(memStore, qualifiedParentCollection, parentId, plan, parentExisted, notify);
+    return parentExisted;
+}
+
+MemoryStore::TtlExpiryAction ttlExpiryDecision(
+    RelationManager& relations,
+    smartbotic::db::storage::LmdbDocumentStore& store,
+    PersistenceManager& persistence,
+    MemoryStore& memStore,
+    CollectionConfigManager& configManager,
+    const std::string& qualifiedParentCollection,
+    const std::string& parentId,
+    const CascadeNotifyFn& notify,
+    bool logBlockAtWarn) {
+    using Action = MemoryStore::TtlExpiryAction;
+
+    // One helper for both refusal sites, so the two cannot drift in either text
+    // or level. See the header: the level is a volume decision only.
+    const auto logBlock = [logBlockAtWarn](const std::string& msg) {
+        if (logBlockAtWarn) spdlog::warn("{}", msg);
+        else spdlog::debug("{}", msg);
+    };
+
+    // ---- Scope: the no-relations case must cost exactly one map lookup. ----
+    // Same early-out shape the write path uses. Nothing below this line runs
+    // for a collection that is nobody's parent, which is every collection on
+    // every install that declares no relations.
+    if (qualifiedParentCollection.empty() || qualifiedParentCollection[0] == '_') {
+        return Action::Proceed;
+    }
+    const auto parentRelations = relations.relationsWithParent(qualifiedParentCollection);
+    if (parentRelations.empty()) return Action::Proceed;
+
+    try {
+        // relationsEnforced gates the sweeper exactly as it gates the Delete
+        // handler: off means every on_delete policy is skipped, so a TTL expiry
+        // behaves as it did pre-v2.11.0 rather than half-enforcing. Named local
+        // - RelationEnforcer binds the flag by reference.
+        const CollectionCfg cfg = configManager.configFor(qualifiedParentCollection);
+        if (!cfg.relationsEnforced) return Action::Proceed;
+
+        // ---- restrict: a manual delete FAILS, so the expiry must not happen.
+        RelationEnforcer enforcer(relations, store, cfg.relationsEnforced);
+        std::string err;
+        if (!enforcer.canDelete(qualifiedParentCollection, parentId, err)) {
+            // ⚠ Stated where an operator will see it: this document now
+            // OUTLIVES ITS TTL, indefinitely, for as long as a child keeps
+            // referencing it. That is the accepted price of not orphaning the
+            // children, and the alternative is what this change removes.
+            logBlock(fmt::format(
+                "TTL expiry of '{}/{}' is BLOCKED by a restrict relation, so the document "
+                "remains past its TTL and will be retried on a slower cadence. How many are "
+                "stuck right now: GetMemoryStats.ttl_blocked_documents (a periodic summary "
+                "line also reports it). Reason: {}",
+                qualifiedParentCollection, parentId, err));
+            return Action::Skip;
+        }
+
+        // ---- cascade / set_null: only reach for the cascade machinery when a
+        // destructive policy actually has work to do. A restrict-only or
+        // no_action-only parent falls through to the ordinary expiry, which
+        // keeps the EXPIRE event and the expiredCount stat exactly as they were.
+        bool hasDestructive = false;
+        for (const auto& r : parentRelations) {
+            if (r.onDelete == OnDelete::Cascade || r.onDelete == OnDelete::SetNull) {
+                hasDestructive = true;
+                break;
+            }
+        }
+        if (!hasDestructive) return Action::Proceed;
+
+        try {
+            (void)executeCascade(relations, store, persistence, memStore, configManager,
+                                 qualifiedParentCollection, parentId, notify);
+            return Action::Handled;
+        } catch (const CascadeBlocked& e) {
+            // Either a restrict-protected grandchild, or the mirror
+            // unhealthy/drifted refusal. Nothing was written in either case
+            // (both are thrown before writeCascadeWal()), so skipping is a
+            // clean no-op and the next sweep retries. Expiring the parent
+            // anyway is exactly the orphaning this function exists to remove.
+            logBlock(fmt::format(
+                "TTL expiry of '{}/{}' skipped: {} - the document remains past its TTL and "
+                "will be retried on a slower cadence",
+                qualifiedParentCollection, parentId, e.what()));
+            return Action::Skip;
+        }
+    } catch (const std::exception& e) {
+        // A malformed name, MDB_READERS_FULL, a mid-cascade fault. Never expire
+        // a parent whose children could not be handled.
+        // ⚠ A cascade that threw AFTER writeCascadeWal() fsynced is already
+        // durable and will apply on the next restart - the same caveat the
+        // Delete handler documents. Skipping is still right: the sweeper must
+        // not also delete the parent behind that cascade's back.
+        spdlog::error("TTL expiry of '{}/{}' failed while handling its children ({}) - the "
+                      "document is left in place and will be retried; if the failure came "
+                      "after the cascade's WAL fsync, that cascade will still apply on the "
+                      "next restart", qualifiedParentCollection, parentId, e.what());
+        return Action::Skip;
+    }
+}
+
+} // namespace smartbotic::database

+ 435 - 0
service/src/relations/relation_cascade.hpp

@@ -0,0 +1,435 @@
+// v2.11.0 T12 — cascade and set_null, made destructive and WAL-first.
+//
+// Today (through T9/Phase A) a relation declaring `cascade` or `set_null`
+// behaves exactly like `no_action`: the delete goes through and the
+// reference is left dangling. This module makes them actually act, while
+// leaving `restrict` (relation_enforcement.hpp) untouched — that check
+// still runs first, upstream of everything here, and still aborts the
+// whole delete with FAILED_PRECONDITION when it blocks.
+//
+// ⚠ THE PLACE A RELATIONAL MENTAL MODEL ACTIVELY MISLEADS: for an
+// ARRAY-VALUED reference, `cascade` and `set_null` COLLAPSE TO THE SAME
+// BEHAVIOUR — pull the parent id out of the array and keep the document.
+// A MySQL-trained instinct says `cascade` means "delete the child row";
+// that is only true for a SCALAR reference. Deleting a node because one of
+// its three credentials went away would be worse than useless. Only a
+// scalar reference under `cascade` deletes the child document. Every array
+// match — under either policy — and every scalar match under `set_null`,
+// mutates the child in place instead.
+//
+// ⚠ WAL-BEFORE-LMDB, AND WHY: MemoryStore is rebuilt at boot from snapshot
+// + WAL replay, NOT from LMDB (LMDB is a read-serving mirror, kept in sync
+// under MemoryStore's per-collection lock on the ordinary write path — see
+// memory_store.cpp's mirrorWriteToDocStore). A cascade that committed its
+// LMDB transaction before the WAL entries for the parent and every child
+// mutation were written and fsynced would, on a crash in that window, come
+// back on restart with the children RESURRECTED — reconstructed from WAL
+// exactly as they were before the cascade, now pointing at a parent that
+// LMDB (correctly) no longer has. Ordering it the other way — WAL first,
+// fsynced, THEN one atomic LMDB commit — means a crash in that window
+// instead leaves LMDB one step behind a WAL that already describes the
+// full cascade; the next boot's WAL replay reconstructs the correct
+// (post-cascade) MemoryStore regardless of whether LMDB got that far, and
+// replaying an already-applied delete/update against MemoryStore is a
+// no-op (MemoryStore::remove()/unloadDocument() on an absent id just
+// returns false). See writeCascadeWal()/commitCascadeLmdb() below for the
+// exact sequence, and tests/test_relation_enforcement.cpp for the crash
+// simulation that pins this.
+//
+// ⚠ WHAT "WAL-BEFORE-LMDB" DOES NOT FIX (review finding I1): MemoryStore
+// recovers correctly from the WAL alone, as above. LMDB itself does NOT
+// converge for every mutation kind after a crash in the WAL→commit window.
+// DELETE replay goes through MemoryStore::remove(), which still mirrors to
+// LMDB (memory_store.cpp), so a crashed-and-recovered DeleteChild/parent
+// delete eventually reaches LMDB too, on the next boot's replay. UPDATE
+// replay (a SetNull/array-pull mutation) goes through
+// loadDocumentWithHistory(), which mirrors NOTHING — and the one-time
+// backfillIntoDocStore() sync is short-circuited by the schema_version=2
+// marker on any already-migrated install, so it will not run again. A
+// crash in that window therefore leaves LMDB permanently serving the
+// PRE-cascade child (parent id still in the array, or the field still
+// non-null) — reads are LMDB-first, so this is user-visible, not just an
+// internal inconsistency — plus a live reverse-index posting pointing at a
+// now-deleted parent, invisible until that child is next rewritten through
+// the ordinary write path (which re-mirrors it).
+//
+// CLOSED (round 3, refined in round 4) by a post-replay re-mirror pass:
+// every DISTINCT (collection, id) that replay applied as
+// INSERT/UPDATE/UPSERT, minus any a later DELETE removed, is collected by
+// PersistenceManager::recover() into RecoveryOutcome::pendingRemirror and
+// pushed to LMDB via MemoryStore::remirrorDocuments().
+// It is BATCHED (chunked transactions — one fsync per document on the boot
+// path was minutes of startup on a large WAL) and it NEVER THROWS (an
+// escaping exception from a repair pass turned recover() into a refusal to
+// start).
+//
+// ⚠ THE REVERSE-INDEX HALF OF THIS ONLY BECAME TRUE IN v2.11.0's FINAL
+// REVIEW (finding 4). While the pass ran inside recover() it wrote documents
+// before DatabaseService::initialize() had called
+// applyRelationDeclarations(), so LmdbDocumentStore::relations(collection)
+// was empty and maintainRelations() did nothing: the child document
+// converged and the stale posting named above did NOT. Same for secondary
+// indexes and for unique constraints. The pass now runs from initialize()
+// after both arming steps (PersistenceManager::runPendingRemirror()), which
+// is what makes "plus a live reverse-index posting pointing at a now-deleted
+// parent" actually closed rather than merely claimed. Do not move it back.
+//
+// It is deliberately not cascade-specific: WAL entries carry no
+// "came from a cascade" marker, and the same divergence occurs whenever
+// applyDualWriteMirror() SWALLOWED an ordinary write's LMDB fault (it
+// swallows everything but UniqueViolation), so a WAL entry never did imply
+// its LMDB write succeeded. Do not assume LMDB is exempt from the crash
+// window just because MemoryStore is — it converges at the NEXT boot, not
+// at the moment of the crash.
+//
+// ⚠ NOT RECURSIVE, AND WHAT THAT MEANS FOR `restrict` (review finding I2):
+// planCascade() only ever looks at relations whose parent is the document
+// actually being deleted. A child this cascade deletes is never itself
+// re-planned as a parent — grandchildren under a Cascade/SetNull relation
+// declared on that child are left dangling (the same class of leftover
+// `relations check`, Task 7, exists to find — not new, and not silently
+// worse than what no_action already leaves behind everywhere else). The
+// dangerous case is different: if a grandchild relation declares
+// `restrict`, deleting the top-level parent would otherwise destroy a
+// row that relation exists to protect, and `RelationEnforcer::canDelete()`
+// never sees it — it only ever evaluates the ORIGINAL parent/id. planCascade()
+// closes that specific hole (not the general dangling-grandchild one): for
+// every child this plan would actually DELETE (never for one it only
+// updates — that child survives), it calls findRelationBlocks() against
+// that child exactly as if IT were being deleted, and throws CascadeBlocked
+// (caught by the Delete handler, mapped to the same FAILED_PRECONDITION
+// restrict itself uses) if anything blocks. I chose refuse-the-whole-
+// cascade over recursive cascading because recursion changes the shape of
+// this entire module — WAL-first and one-atomic-txn both need to extend to
+// an unbounded graph walk rather than one level, which is a much larger
+// change than this finding's fix budget covers — and because refusing is
+// the conservative, safety-preserving choice: a cascade that would bypass
+// a restrict guard is now loudly rejected rather than silently destroying
+// the row that guard exists to protect. This check runs unconditionally,
+// NOT gated by the grandchild collection's own relationsEnforced — being
+// more cautious than strictly required here is deliberate.
+//
+// ⚠ MINOR, DOCUMENTED NOT FIXED: a schema shape this makes permanently
+// undeletable. If the restrict-blocking grandchild is ITSELF one of the
+// children this same plan would delete — e.g. workflows --cascade-->
+// logs, and logs --restrict--> executions, where executions ALSO cascades
+// from workflows and references a log row — findRelationBlocks() still
+// sees that log row's posting (the check runs against the CURRENT reverse
+// index, not against "what this same plan is also about to remove"), so
+// the whole cascade refuses even though, if the plan were applied as one
+// atomic unit, both the log row and its blocker would vanish together and
+// nothing would actually be left dangling. This is conservative and safe
+// (a false refusal, never a false permit), but an operator who designs a
+// schema shaped like this will find the parent refuses to delete no
+// matter what, with no cascade/set_null policy able to route around it —
+// only dropping or loosening the grandchild relation, or deleting the
+// grandchild row out of band first, resolves it. Fixing this properly
+// means checking each blocked-by posting against the REST OF THE PLAN
+// (not just the live index) before refusing, which needs the plan to be
+// built as a fixed point over multiple passes rather than the current
+// single pass over relationsWithParent() — judged not worth doing for
+// what is likely a rare schema shape, but worth an operator knowing why
+// their delete refuses.
+//
+// ⚠ REFUSED WHILE THE MIRROR IS UNHEALTHY OR DRIFTED (v2.11.0 final review,
+// finding 6). executeCascade() checks MemoryStore::mirrorHealthy() and
+// mirrorDriftCount() as its FIRST action and throws CascadeBlocked if either
+// says the two substrates have diverged. planCascade() builds every
+// `updatedDoc` from the LMDB copy and applyCascadeToMemory() then pushes that
+// copy into MemoryStore, so in a state where LMDB is legitimately BEHIND -
+// which is the entire reason reads fall back to MemoryStore, and why six read
+// gates in database_grpc_impl.cpp test this same pair - the cascade would
+// overwrite MemoryStore's fresher child body with a stale one. That is real
+// data loss caused by a repair-shaped operation. Nothing is written before
+// the check, so the refusal is a clean no-op, and restrict/no_action deletes
+// are unaffected. Same reasoning and same shape as the CreateIndex(unique)
+// and CreateRelation gates.
+//
+// Sequence, in full (mirrors the design doc):
+//   0. Refuse if the LMDB mirror is unhealthy or drifted (see above).
+//   1. Resolve children through the reverse index — planCascade(). Also
+//      where the restrict/grandchild check above runs, and where
+//      CascadeBlocked can be thrown.
+//   2. restrict still blocks upstream (relation_enforcement.hpp) — this
+//      module is never reached if that already refused the delete.
+//   3. WAL-log the parent delete + every child mutation, fsync —
+//      writeCascadeWal().
+//   4. One atomic LMDB WriteTxn: mutate children, update the reverse
+//      index (automatic — see maintainRelations in document_store_lmdb.cpp),
+//      delete the parent, commit — commitCascadeLmdb().
+//   5. Apply the same mutations to MemoryStore, taking each affected
+//      collection's lock in turn — applyCascadeToMemory(). Also where the
+//      optional replication/Subscribe-event notification (see below) fires,
+//      once per mutation, after that mutation's MemoryStore apply.
+// executeCascade() runs 3 → 4 → 5 in order; it is what DatabaseGrpcImpl::
+// Delete calls once RelationEnforcer::canDelete() has already permitted
+// the delete.
+//
+// ⚠ FAILURE AFTER THE WAL COMMIT (review finding I4, corrected in round 2):
+// steps 3-5 are not themselves transactional as a WHOLE — only step 4 (the
+// LMDB txn) is all-or-nothing. If commitCascadeLmdb() or
+// applyCascadeToMemory() throws AFTER writeCascadeWal() has already
+// fsynced, the mutation is already DURABLE (the next boot's WAL replay
+// will apply it) even though the caller receives an error. What "neither
+// store reflects it yet" would get wrong: that is only true if
+// commitCascadeLmdb() is what threw (LMDB's WriteTxn aborts unwritten, so
+// LMDB is genuinely still pre-cascade, and step 5 never ran either). If
+// applyCascadeToMemory() is what threw instead, LMDB has ALREADY committed
+// (step 4 succeeded) — and reads are LMDB-first, so the cascade IS ALREADY
+// VISIBLE through ordinary reads at that point, even though the caller is
+// being told the call failed. MemoryStore in that case may be only
+// PARTIALLY applied (some mutations done, some not — applyCascadeToMemory()
+// applies one document at a time, not atomically), and `notify` may have
+// already fired for whichever mutations did apply before the throw.
+// DatabaseGrpcImpl::Delete's error text distinguishes these two cases by
+// stating both possibilities rather than asserting the wrong one. This is a
+// direct, unavoidable consequence of WAL-first: making the WAL entry
+// contingent on the LMDB commit succeeding would reopen exactly the
+// resurrection window this module exists to close. There is no
+// compensating "un-write" of the WAL entry, because replaying it again is
+// safe (idempotent) but erasing it is not (a second failure between
+// erasure and re-fsync would then silently lose the cascade for real).
+//
+// Replication + Subscribe events (review finding C2): the ordinary
+// per-document write path gets both for free because
+// MemoryStore::remove()/update() call emitPersist() -> persistCallback_ ->
+// DatabaseService's lambda, which does WAL + replication + events in one
+// block. executeCascade() cannot reuse that callback directly — it already
+// does its own WAL logging (step 3) and its own LMDB mirroring (step 4),
+// so going back through the ordinary MemoryStore write path in step 5
+// would re-log every mutation to WAL a second time and re-attempt the LMDB
+// mirror a second time, undermining the "WAL and the LMDB commit each
+// happen exactly once, in that order" property this whole module exists to
+// establish. Instead, DatabaseService::notifyReplicationAndEvents() (the
+// replication+events TAIL of that same lambda, extracted so it can be
+// called on its own) is passed in as `notify` and invoked once per
+// mutation, right after that mutation's MemoryStore apply — mirroring how
+// v2.3.1 fixed replicated-entry apply by explicitly driving
+// applyDualWriteMirror rather than relying on a callback that had already
+// been bypassed for the same reason. `notify` is optional (nullptr skips
+// it) so the unit tests in tests/test_relation_enforcement.cpp, which have
+// no DatabaseService, keep working unchanged.
+#pragma once
+
+#include <functional>
+#include <optional>
+#include <stdexcept>
+#include <string>
+#include <vector>
+
+#include "document.hpp"
+// v2.11.0 close-out — for MemoryStore::TtlExpiryAction, the return type of
+// ttlExpiryDecision() below. MemoryStore was previously only forward-declared
+// here; the enum has to be a complete type at this point, and it belongs on
+// MemoryStore because MemoryStore's sweeper is what acts on it.
+#include "memory_store.hpp"
+#include "relation_manager.hpp"
+
+namespace smartbotic::db::storage {
+class LmdbDocumentStore;
+}
+
+namespace smartbotic::database {
+
+class PersistenceManager;
+class MemoryStore;
+class CollectionConfigManager;
+
+// Thrown by planCascade() when proceeding would silently destroy a row
+// protected by its own `restrict` relation (a grandchild of the delete
+// being planned) — see the file header's "NOT RECURSIVE" section.
+// DatabaseGrpcImpl::Delete catches this and maps it to FAILED_PRECONDITION,
+// the same status code a direct restrict refusal uses.
+class CascadeBlocked : public std::runtime_error {
+public:
+    explicit CascadeBlocked(std::string msg) : std::runtime_error(std::move(msg)) {}
+};
+
+// One resolved child-side mutation a cascade/set_null relation requires.
+struct CascadeMutation {
+    enum class Kind { DeleteChild, UpdateChild };
+    Kind kind = Kind::DeleteChild;
+    std::string childCollectionQualified;   // for MemoryStore/WAL (e.g. "default:executions")
+    std::string childCollectionBare;        // for LmdbDocumentStore (e.g. "executions")
+    std::string childId;
+    bool hadVector = false;                 // DeleteChild only — also drop the vector
+    std::optional<Document> updatedDoc;     // UpdateChild only — the full post-mutation doc
+};
+
+struct CascadePlan {
+    std::vector<CascadeMutation> mutations;
+    bool parentHadVector = false;
+};
+
+// Called once per mutation (child), and once more for the parent, from
+// applyCascadeToMemory() — AFTER that specific document's MemoryStore
+// apply — to drive replication + Subscribe events explicitly. See the file
+// header's "Replication + Subscribe events" section. `doc` is nullopt for
+// a delete. Typically bound to
+// DatabaseService::notifyReplicationAndEvents(); nullptr/empty skips
+// notification entirely (what every unit test in
+// tests/test_relation_enforcement.cpp does, having no DatabaseService).
+using CascadeNotifyFn = std::function<void(const std::string& qualifiedCollection,
+                                           const std::string& id,
+                                           const std::optional<Document>& doc,
+                                           EventType eventType)>;
+
+// Step 1 — read-only except for the throw path. Resolves every
+// Cascade/SetNull relation whose parent is qualifiedParentCollection,
+// walks the reverse index for parentId under each (unbounded — a cascade
+// must see every child, never a truncated sample), reads the current child
+// documents, and decides Delete vs Update per the array rule in the file
+// header. For every child this plan would DELETE, also checks whether that
+// child is itself protected by a `restrict` relation (see the file
+// header's "NOT RECURSIVE" section) and throws CascadeBlocked if so —
+// the only way this function mutates anything is by NOT returning.
+//
+// `configManager` supplies the per-collection timestamp_precision each
+// mutated child is stamped with (see the file header — mirrors
+// MemoryStore::currentTimeFor() exactly, since planCascade() only has an
+// LmdbDocumentStore&, not a MemoryStore&, to ask directly).
+//
+// If the SAME child is targeted by two different relations from this
+// parent (e.g. two fields on one child collection both referencing it),
+// and the two would disagree (one wants Delete, the other Update), Delete
+// wins — a child slated for deletion is not resurrected by a later Update
+// in the same plan. This is a narrow, deliberately simple merge rule; see
+// the Task 12 report for the reasoning.
+CascadePlan planCascade(RelationManager& relations,
+                        smartbotic::db::storage::LmdbDocumentStore& store,
+                        CollectionConfigManager& configManager,
+                        const std::string& qualifiedParentCollection,
+                        const std::string& parentId);
+
+// Step 3 — WAL-log the parent delete and every mutation in `plan`, then
+// force an fsync (PersistenceManager::flushWal()). MUST run, and MUST
+// complete, before commitCascadeLmdb() — see the file header.
+void writeCascadeWal(PersistenceManager& persistence,
+                     const std::string& qualifiedParentCollection,
+                     const std::string& parentId,
+                     const CascadePlan& plan);
+
+// Step 4 — one atomic LMDB WriteTxn: apply every mutation in `plan`
+// (deleting or updating each child, which also maintains the reverse index
+// and any secondary indexes automatically — see maintainRelations/
+// maintainIndexes), then delete the parent (and its vector, if any).
+// All-or-nothing: any exception leaves neither a child nor the parent
+// mutated — the WriteTxn's destructor aborts unwritten. Returns whether the
+// PARENT document itself was present and removed (independent of how many
+// children were mutated — a cascade run against an already-gone parent id
+// still cleans up any dangling children the reverse index still names, the
+// same repair `relations check` (Task 7) exists to find).
+//
+// ⚠ See the file header's "LMDB does not converge" note (I1): this step's
+// own commit is atomic, but if it or applyCascadeToMemory() never runs at
+// all (the process crashed between writeCascadeWal()'s fsync and here),
+// LMDB stays pre-cascade for UpdateChild mutations specifically until that
+// row is next rewritten through the ordinary write path.
+bool commitCascadeLmdb(smartbotic::db::storage::LmdbDocumentStore& store,
+                       const std::string& qualifiedParentCollection,
+                       const std::string& parentId,
+                       const CascadePlan& plan);
+
+// Step 5 — apply the same mutations to MemoryStore, taking each affected
+// collection's lock IN TURN (one call per document, not one lock spanning
+// the whole cascade). Memory-only, no WAL/mirror side effects — see
+// MemoryStore::unloadDocument()'s header comment for why. If `notify` is
+// set, it is called once per mutation (and once for the parent, gated on
+// `parentExisted` — commitCascadeLmdb()'s return, the LMDB-confirmed
+// answer, NOT whatever MemoryStore's own cache happened to hold, which may
+// have evicted the parent independently) immediately after that document's
+// MemoryStore apply — see the file header's "Replication + Subscribe
+// events" section. Every CHILD mutation notifies unconditionally: each one
+// was already confirmed real by planCascade()'s LMDB read and committed by
+// commitCascadeLmdb(), independent of whether THIS node's MemoryStore
+// cache happened to hold that child.
+void applyCascadeToMemory(MemoryStore& store,
+                          const std::string& qualifiedParentCollection,
+                          const std::string& parentId,
+                          const CascadePlan& plan,
+                          bool parentExisted,
+                          const CascadeNotifyFn& notify = nullptr);
+
+// Orchestrates steps 3 → 4 → 5, in order, with nothing in between. What
+// DatabaseGrpcImpl::Delete calls once RelationEnforcer::canDelete() (step
+// 2 — restrict) has already permitted the delete; this function does not
+// re-check restrict and must not be reached if that refused. May throw
+// CascadeBlocked from the planning step (see above) before anything is
+// written anywhere. Returns commitCascadeLmdb()'s result — whether the
+// parent document itself was present and removed — for the handler's
+// DeleteResponse.deleted field.
+// v2.11.0 close-out — WHAT A TTL EXPIRY MUST DO ABOUT THE EXPIRING PARENT'S
+// CHILDREN, and the cascade itself when one is required.
+//
+// Operator's ruling: a TTL-deleted parent handles its children exactly the way
+// a manually deleted parent does, like MySQL. Before this,
+// MemoryStore::expireDocuments() erased the document and mirrored a DELETE with
+// no relation enforcement at all, so a `restrict`-protected parent silently
+// vanished and orphaned every child - strictly the worst of the available
+// behaviours.
+//
+// This function is the sweeper's whole decision, and it lives HERE rather than
+// in DatabaseService so it is directly unit-testable against the same fixtures
+// the cascade tests use (DatabaseService::ttlExpiryRelationDecision() is a thin
+// binder over it, and MemoryStore calls it through
+// MemoryStore::TtlExpiryRelationHook). By policy:
+//   restrict  -> Skip. A manual delete FAILS, so the expiry must not happen.
+//                The document stays, its expiry stays armed, the next sweep
+//                retries - which means it OUTLIVES ITS TTL for as long as a
+//                child references it. Deliberate, and signalled: a WARN per
+//                skip plus MemoryStore::Stats::ttlExpiryBlockedByRelation.
+//   cascade   -> Handled, via executeCascade() below. NOT a second cascade
+//                implementation: WAL-before-LMDB is the whole reason this path
+//                exists (MemoryStore is rebuilt from snapshot + WAL, never from
+//                LMDB, so a sweeper that wrote only LMDB would have its child
+//                deletions resurrected by the next boot's replay).
+//   set_null  -> Handled, same call. Scalar reference nulled; an array-valued
+//                reference has the id pulled and the document kept, exactly as
+//                the request path does.
+//   no_action -> Proceed. Reference left dangling, as today.
+// and Proceed also when no relation names this collection as a parent (the
+// common case, early-returned before any LMDB or config work), or when the
+// collection's relationsEnforced is off (the documented escape hatch, gating
+// every on_delete policy and not just restrict, exactly as the Delete handler
+// does).
+//
+// Skips - never orphans - if executeCascade() refuses because the LMDB mirror is
+// unhealthy or has drifted since READY, and if anything throws. `notify` is
+// forwarded to executeCascade() so a TTL-driven cascade drives replication and
+// Subscribe events per mutation like any other; without it a follower would
+// never see the child deletions and would diverge permanently.
+//
+// ⚠ MUST be called with NO MemoryStore lock held - it re-enters MemoryStore
+// (applyCascadeToMemory takes each affected collection's lock, and
+// getOrCreateCollection takes globalMutex_ exclusively). That is exactly what
+// expireDocuments()' two-phase structure guarantees.
+//
+// `logBlockAtWarn` controls the VOLUME of the refusal log, never the decision:
+// true logs the block at WARN with the blocking relation and child count (what
+// the first refusal for a document does), false logs the same text at DEBUG
+// (what a retry of an already-reported document does). At the sweeper's 1s
+// default, a persistent `restrict` block on N parents was N WARN lines per
+// second forever, which buries every other operator signal. The periodic
+// "still stuck" summary comes from expireDocuments() itself, which is the only
+// place that knows the total.
+MemoryStore::TtlExpiryAction ttlExpiryDecision(
+    RelationManager& relations,
+    smartbotic::db::storage::LmdbDocumentStore& store,
+    PersistenceManager& persistence,
+    MemoryStore& memStore,
+    CollectionConfigManager& configManager,
+    const std::string& qualifiedParentCollection,
+    const std::string& parentId,
+    const CascadeNotifyFn& notify = nullptr,
+    bool logBlockAtWarn = true);
+
+bool executeCascade(RelationManager& relations,
+                    smartbotic::db::storage::LmdbDocumentStore& store,
+                    PersistenceManager& persistence,
+                    MemoryStore& memStore,
+                    CollectionConfigManager& configManager,
+                    const std::string& qualifiedParentCollection,
+                    const std::string& parentId,
+                    const CascadeNotifyFn& notify = nullptr);
+
+} // namespace smartbotic::database

+ 163 - 0
service/src/relations/relation_enforcement.cpp

@@ -0,0 +1,163 @@
+#include "relation_enforcement.hpp"
+
+#include "../project_addressing.hpp"
+#include "../storage/document_store_lmdb.hpp"
+
+#include <sstream>
+
+#include <spdlog/spdlog.h>
+
+namespace {
+
+// Shared per-relation lookup step used by both findRelationBlocks() and
+// describeDeleteImpacts() - factored out after a review flagged the two as
+// independently duplicating bare-name resolution + child count + sampling.
+// Two copies of a count query is exactly the failure pattern behind the
+// v2.4.4 misfiled-rows incident and the v2.7.1 divergent-count bug: they
+// only need to disagree once for DescribeDelete to say "safe" while an
+// actual delete blocks, or vice versa. One implementation, two call sites.
+struct RelationLookup {
+    std::string bareRelation;              // resolveCollection(r.name).collection
+    uint64_t childCount = 0;
+    std::vector<std::string> sampleChildIds;   // populated only if childCount > 0
+    bool resolvable = false;               // false => r.name could not be resolved; caller must skip
+};
+
+RelationLookup lookupRelationCounts(smartbotic::db::storage::LmdbDocumentStore& store,
+                                     const smartbotic::database::RelationInfo& r,
+                                     const std::string& parentId) {
+    RelationLookup out;
+
+    // relation_index_* take the BARE relation name; RelationInfo::name is
+    // project-qualified. Malformed names should never happen (createRelation
+    // validates), but neither a delete nor a describe query is the place to
+    // throw over a data problem in an unrelated declaration - skip it.
+    try {
+        out.bareRelation = smartbotic::database::resolveCollection(r.name).collection;
+    } catch (const std::exception&) {
+        return out; // resolvable stays false
+    }
+    out.resolvable = true;
+
+    out.childCount = store.relation_index_child_count(out.bareRelation, parentId);
+    if (out.childCount > 0) {
+        out.sampleChildIds = store.relation_index_children(out.bareRelation, parentId, 5);
+    }
+    return out;
+}
+
+} // namespace
+
+namespace smartbotic::database {
+
+std::vector<RelationBlock> findRelationBlocks(
+    RelationManager& relations,
+    smartbotic::db::storage::LmdbDocumentStore& store,
+    const std::string& qualifiedParentCollection,
+    const std::string& parentId) {
+    std::vector<RelationBlock> out;
+
+    for (const auto& r : relations.relationsWithParent(qualifiedParentCollection)) {
+        // Only restrict blocks here. no_action is the documented escape
+        // hatch. cascade/set_null (v2.11.0 T12+) ARE destructive, but that
+        // destruction is handled entirely by relations/relation_cascade.cpp,
+        // AFTER this function has permitted the delete - so they still fall
+        // through this filter unblocked, on purpose, not because they are
+        // still permissive placeholders. See relation_cascade.hpp's file
+        // header for what actually happens to them.
+        if (r.onDelete != OnDelete::Restrict) continue;
+
+        const auto lookup = lookupRelationCounts(store, r, parentId);
+        if (!lookup.resolvable) continue;
+        if (lookup.childCount == 0) continue; // no reference -> nothing to block on
+
+        RelationBlock block;
+        block.relation = r.name;
+        block.childCollection = r.child;
+        block.childCount = lookup.childCount;
+        block.sampleChildIds = lookup.sampleChildIds;
+        out.push_back(std::move(block));
+    }
+
+    return out;
+}
+
+std::vector<RelationImpact> describeDeleteImpacts(
+    RelationManager& relations,
+    smartbotic::db::storage::LmdbDocumentStore& store,
+    const std::string& qualifiedParentCollection,
+    const std::string& parentId) {
+    std::vector<RelationImpact> out;
+
+    for (const auto& r : relations.relationsWithParent(qualifiedParentCollection)) {
+        const auto lookup = lookupRelationCounts(store, r, parentId);
+        if (!lookup.resolvable) continue;
+
+        RelationImpact impact;
+        impact.relation = r.name;
+        impact.childCollection = r.child;
+        impact.childField = r.childField;
+        impact.onDelete = r.onDelete;
+        impact.childCount = lookup.childCount;
+        impact.sampleChildIds = lookup.sampleChildIds;
+        impact.blocks = (r.onDelete == OnDelete::Restrict) && (lookup.childCount > 0);
+
+        out.push_back(std::move(impact));
+    }
+
+    return out;
+}
+
+std::string formatRelationBlockError(
+    const std::string& qualifiedParentCollection,
+    const std::string& parentId,
+    const std::vector<RelationBlock>& blocks) {
+    if (blocks.empty()) return "";
+
+    std::ostringstream os;
+    os << "cannot delete '" << qualifiedParentCollection << "/" << parentId
+       << "': " << blocks.size() << " relation(s) still reference it -";
+    for (const auto& b : blocks) {
+        os << " relation '" << b.relation << "' has " << b.childCount
+           << " child document(s) in '" << b.childCollection << "' referencing it";
+        if (!b.sampleChildIds.empty()) {
+            os << " (e.g. ";
+            for (size_t i = 0; i < b.sampleChildIds.size(); ++i) {
+                if (i) os << ", ";
+                os << b.sampleChildIds[i];
+            }
+            os << ")";
+        }
+        os << ";";
+    }
+    os << " delete or re-point the referencing document(s) first, or change "
+          "the relation's on_delete to no_action to permit the dangling "
+          "reference.";
+    return os.str();
+}
+
+RelationEnforcer::RelationEnforcer(RelationManager& relations,
+                                    smartbotic::db::storage::LmdbDocumentStore& store,
+                                    const bool& relationsEnforced)
+    : relations_(relations), store_(store), relationsEnforced_(relationsEnforced) {}
+
+bool RelationEnforcer::canDelete(const std::string& qualifiedParentCollection,
+                                  const std::string& parentId,
+                                  std::string& err) {
+    if (!relationsEnforced_) {
+        spdlog::info(
+            "relations: enforcement disabled for '{}' - permitting delete of '{}' "
+            "unchecked. This is NOT retroactive: dangling references this creates "
+            "will not be found by re-enabling enforcement, only by `relations check`.",
+            qualifiedParentCollection, parentId);
+        return true;
+    }
+
+    auto blocks = findRelationBlocks(relations_, store_, qualifiedParentCollection, parentId);
+    if (blocks.empty()) return true;
+
+    err = formatRelationBlockError(qualifiedParentCollection, parentId, blocks);
+    return false;
+}
+
+} // namespace smartbotic::database

+ 172 - 0
service/src/relations/relation_enforcement.hpp

@@ -0,0 +1,172 @@
+// v2.11.0 T4 — restrict/no_action enforcement on parent delete.
+//
+// The pure decision logic for "may this delete proceed", kept free of
+// grpc::ServerContext and the pb:: types entirely so it is unit-testable
+// against just a RelationManager + LmdbDocumentStore (see
+// tests/test_relation_enforcement.cpp). Task 6 wires a live RelationManager
+// into DatabaseGrpcImpl and calls into this from Delete(), after the access
+// gate and before store_.remove(); Task 9 exercises it over the real gRPC
+// boundary.
+//
+// Scope: only `restrict` blocks HERE. `no_action` permits the delete and
+// leaves the reference dangling - the documented escape hatch. A relation
+// declaring `cascade` or `set_null` also falls through this filter
+// unblocked, but as of v2.11.0 T12 that is NOT permissive placeholder
+// behaviour - those policies are genuinely destructive, just handled by a
+// separate module (relations/relation_cascade.cpp), called by the Delete
+// handler AFTER this check has permitted the delete. Do not read "not
+// blocked here" as "does nothing" - see relation_cascade.hpp's file header
+// for what a cascade/set_null delete actually does, including the one
+// place it can still refuse: a grandchild protected by its own `restrict`
+// relation.
+
+#pragma once
+
+#include <cstdint>
+#include <string>
+#include <vector>
+
+#include "relation_manager.hpp"
+
+namespace smartbotic::db::storage {
+class LmdbDocumentStore;
+}
+
+namespace smartbotic::database {
+
+// One relation that is currently blocking a delete.
+struct RelationBlock {
+    std::string relation;        // qualified name, e.g. "default:exec_wf"
+    std::string childCollection; // qualified, e.g. "default:executions"
+    uint64_t childCount = 0;
+    std::vector<std::string> sampleChildIds;   // at most five
+};
+
+// Every relation that blocks deleting qualifiedParentCollection/parentId
+// right now. Empty means the delete may proceed as far as relations are
+// concerned.
+//
+// Deliberately does NOT consult relationsEnforced - that is the caller's
+// decision about whether to invoke this at all. A caller that always needs
+// the true answer regardless of the flag exists too: `relations check`
+// (Task 7) has to find dangling references produced while enforcement was
+// off, so the query itself must stay honest about the flag being someone
+// else's business.
+//
+// Only Restrict relations are considered; see the file header for why
+// Cascade/SetNull are treated as permissive here. Absent and null child
+// references are not represented in the index at all (Task 3), so a
+// relation with zero matching postings is silently skipped - there is no
+// second code path that reasons about "no reference" separately from "zero
+// count".
+//
+// Costs no document reads: relation_index_child_count is backed by
+// mdb_cursor_count. `relation` is looked up via relationsWithParent(), and
+// its bare name (stripped of the project qualifier) is what the index
+// calls expect.
+std::vector<RelationBlock> findRelationBlocks(
+    RelationManager& relations,
+    smartbotic::db::storage::LmdbDocumentStore& store,
+    const std::string& qualifiedParentCollection,
+    const std::string& parentId);
+
+// Render findRelationBlocks() output into the operator-facing
+// FAILED_PRECONDITION message text: names every blocking relation, its
+// child count, and up to five sample blocking ids, so an operator learns
+// what to do next from the message alone. Returns "" for an empty list
+// (defined total for clarity; callers should not invoke this when
+// findRelationBlocks() returned nothing).
+std::string formatRelationBlockError(
+    const std::string& qualifiedParentCollection,
+    const std::string& parentId,
+    const std::vector<RelationBlock>& blocks);
+
+// One relation's impact on deleting qualifiedParentCollection/parentId, for
+// DescribeDelete (Task 5).
+//
+// Unlike RelationBlock/findRelationBlocks, this enumerates EVERY relation
+// whose parent is this collection - restrict, cascade, set_null AND
+// no_action - because DescribeDelete exists to show an operator what each
+// declared policy will do, not just which ones currently block. `blocks`
+// is the structural answer ("this relation's policy is restrict and it has
+// live children right now"): true only for Restrict with childCount > 0,
+// mirroring findRelationBlocks' own filter. It deliberately does NOT fold
+// in CollectionCfg::relationsEnforced - same reasoning as
+// findRelationBlocks: whether the flag is consulted is the caller's
+// decision, not this query's. DescribeDelete's handler ANDs `blocks` with
+// the live relationsEnforced value to produce would_be_blocked, so a
+// disabled collection still shows an operator which relations WOULD block
+// if it were turned back on.
+struct RelationImpact {
+    std::string relation;          // qualified name, e.g. "default:exec_wf"
+    std::string childCollection;   // qualified, e.g. "default:executions"
+    std::string childField;        // dot-path on the child
+    OnDelete onDelete = OnDelete::Restrict;
+    uint64_t childCount = 0;
+    std::vector<std::string> sampleChildIds;   // at most five
+    bool blocks = false;
+};
+
+// Every relation whose parent is qualifiedParentCollection, each annotated
+// with its live child count and blocking status against parentId. Empty
+// means this collection has no declared relations at all (not the same as
+// "the delete is safe" - see `blocks` on each entry, and would_be_blocked
+// in the DescribeDelete response).
+//
+// Same cost profile as findRelationBlocks: relation_index_child_count is
+// backed by mdb_cursor_count, so this reads no documents and is cheap
+// enough for a UI to call before every delete.
+std::vector<RelationImpact> describeDeleteImpacts(
+    RelationManager& relations,
+    smartbotic::db::storage::LmdbDocumentStore& store,
+    const std::string& qualifiedParentCollection,
+    const std::string& parentId);
+
+// Thin, stateful wrapper around findRelationBlocks()/formatRelationBlockError()
+// that also applies the relationsEnforced switch (Task 8). Exists so the
+// decision is directly unit-testable (see tests/test_relation_enforcement.cpp)
+// and directly reusable, unmodified, from Task 6's Delete handler.
+//
+// Holds references only, no ownership. `relationsEnforced` is bound by
+// reference (normally CollectionCfg::relationsEnforced, read per call via
+// config_manager_.configFor(qualifiedCollection)) so it is re-read live,
+// not snapshotted at construction - matching how every other per-call
+// config read in this codebase behaves.
+//
+// ⚠ Lifetime: relationsEnforced must outlive the RelationEnforcer. Bind it
+// to a named local, e.g.:
+//     auto cfg = config_manager_.configFor(qualified);
+//     RelationEnforcer enforcer(relations_, *store, cfg.relationsEnforced);
+// NOT to a temporary's subobject (config_manager_.configFor(qualified).relationsEnforced
+// used directly as the constructor argument) - configFor() returns
+// CollectionCfg BY VALUE, so that temporary is destroyed at the end of the
+// full expression and the reference would dangle for every call after
+// construction.
+class RelationEnforcer {
+public:
+    RelationEnforcer(RelationManager& relations,
+                      smartbotic::db::storage::LmdbDocumentStore& store,
+                      const bool& relationsEnforced);
+
+    // Returns true if the delete of qualifiedParentCollection/parentId may
+    // proceed. False means at least one restrict relation still has
+    // children; err is populated with a message naming the relation(s),
+    // child count(s) and sample blocking ids - see formatRelationBlockError().
+    //
+    // relationsEnforced == false skips the check entirely and returns true
+    // - deliberately, and NOT retroactively: dangling references created
+    // while it was off are not found by turning it back on, only by
+    // `relations check` (Task 7). Logged at INFO on every skip, because
+    // DeleteResponse has no message field to carry this to the caller - the
+    // server log is the only place an operator can learn it happened.
+    bool canDelete(const std::string& qualifiedParentCollection,
+                   const std::string& parentId,
+                   std::string& err);
+
+private:
+    RelationManager& relations_;
+    smartbotic::db::storage::LmdbDocumentStore& store_;
+    const bool& relationsEnforced_;
+};
+
+} // namespace smartbotic::database

+ 20 - 0
service/src/relations/relation_index.cpp

@@ -0,0 +1,20 @@
+// v2.11.0 T2 — reverse index sub-db naming for relations.
+
+#include "relations/relation_index.hpp"
+
+namespace smartbotic::db::storage {
+
+std::string relation_index_subdb(std::string_view relationBareName) {
+    std::string out;
+    out.reserve(kRelIndexPrefix.size() + relationBareName.size());
+    out.append(kRelIndexPrefix);
+    out.append(relationBareName);
+    return out;
+}
+
+bool is_relation_index_subdb(std::string_view subdb) {
+    return subdb.size() > kRelIndexPrefix.size() &&
+           subdb.compare(0, kRelIndexPrefix.size(), kRelIndexPrefix) == 0;
+}
+
+}  // namespace smartbotic::db::storage

+ 50 - 0
service/src/relations/relation_index.hpp

@@ -0,0 +1,50 @@
+// v2.11.0 T2 — reverse index sub-db naming for relations.
+//
+// A relation says "documents in `child` reference documents in `parent`
+// through `childField`". Enforcement (Restrict/Cascade/etc. on delete, and
+// DescribeDelete) needs the inverse question answered fast: "which child
+// documents reference this parent id, and how many?" Scanning `child` for
+// every delete would be a full collection scan per parent delete, which is
+// exactly the cost this repo already paid down once for secondary indexes
+// (see storage/secondary_index.hpp) - so the reverse index reuses that shape.
+//
+// Layout: one LMDB sub-db per relation, opened MDB_DUPSORT.
+//   key  = parent id
+//   data = child id
+// DUPSORT stores a parent's children as a sorted, deduplicated set - exactly
+// a posting list, and mdb_cursor_count answers "how many children" without
+// reading any of them (the operation restrict/DescribeDelete need).
+//
+// This header is intentionally free of relations/relation_manager.hpp - the
+// storage layer does not know about RelationInfo, project qualification, or
+// enforcement policy, the same way secondary_index.hpp does not know about
+// Query or FilterOp. Callers (Task 3+) pass the BARE relation name; a relation
+// sub-db lives inside a per-project LMDB env, so the project is already
+// implied by which env this is - qualifying here would double-encode it.
+
+#pragma once
+
+#include <string>
+#include <string_view>
+
+namespace smartbotic::db::storage {
+
+// Prefix marking a sub-db as a relation reverse index. Leading '_' means
+// is_system_subdb() already treats these as internal, so they stay out of
+// list_collections(). The trailing digit is a KEY FORMAT VERSION, the same
+// convention as kIndexSubdbPrefix: v2.9.1 had to bump _idx_ to _idx2_ when its
+// encoding changed, so a stale index is never read under new rules. Paying
+// that forward here costs nothing and avoids a repeat of that migration pain
+// if the relation index's shape ever needs to change.
+inline constexpr std::string_view kRelIndexPrefix = "_relidx1_";
+
+// Sub-db name for one relation's reverse index. `relationBareName` is the
+// UNQUALIFIED relation name (e.g. "exec_wf", not "myproj:exec_wf") - see the
+// file comment above for why.
+std::string relation_index_subdb(std::string_view relationBareName);
+
+// True if a sub-db name is a relation reverse index. Used to skip these when
+// walking sub-dbs and to open them with MDB_DUPSORT (see prime_dbi_cache).
+bool is_relation_index_subdb(std::string_view subdb);
+
+}  // namespace smartbotic::db::storage

+ 357 - 0
service/src/relations/relation_manager.cpp

@@ -0,0 +1,357 @@
+#include "relation_manager.hpp"
+
+#include "../memory_store.hpp"
+#include "../project_addressing.hpp"
+
+#include <chrono>
+#include <nlohmann/json.hpp>
+#include <spdlog/spdlog.h>
+
+namespace smartbotic::database {
+
+namespace {
+
+uint64_t nowMs() {
+    return std::chrono::duration_cast<std::chrono::milliseconds>(
+        std::chrono::system_clock::now().time_since_epoch()).count();
+}
+
+nlohmann::json toJson(const RelationInfo& r) {
+    return nlohmann::json{
+        {"name", r.name},
+        {"child", r.child},
+        {"child_field", r.childField},
+        {"parent", r.parent},
+        {"on_delete", onDeleteToString(r.onDelete)},
+        {"validate_on_write", r.validateOnWrite},
+        {"created_at", r.createdAt},
+        {"updated_at", r.updatedAt}
+    };
+}
+
+RelationInfo fromJson(const nlohmann::json& j) {
+    RelationInfo r;
+    r.name = j.value("name", "");
+    r.child = j.value("child", "");
+    r.childField = j.value("child_field", "");
+    r.parent = j.value("parent", "");
+    // v2.11.0 final review (finding 9) — the persisted value goes through the
+    // same single parser as the RPC. A record whose on_delete is not one of
+    // the four falls back to Restrict, which is the safe direction (it
+    // refuses deletes rather than performing an unintended destructive one),
+    // but it is logged at ERROR: silently coercing is what made a typo
+    // indistinguishable from a deliberate choice. Deliberately NOT skipped
+    // the way a cross-project record is: dropping the relation entirely would
+    // remove protection, which is strictly worse than over-restricting.
+    {
+        const std::string raw = j.value("on_delete", std::string("restrict"));
+        if (auto parsed = parseOnDelete(raw)) {
+            r.onDelete = *parsed;
+        } else {
+            r.onDelete = OnDelete::Restrict;
+            spdlog::error("RelationManager: relation '{}' has an unrecognised "
+                          "on_delete '{}' (expected {}); treating it as 'restrict' "
+                          "- fix the declaration, this is not what was asked for",
+                          r.name, raw, kOnDeleteValues);
+        }
+    }
+    r.validateOnWrite = j.value("validate_on_write", false);
+    r.createdAt = j.value("created_at", uint64_t{0});
+    r.updatedAt = j.value("updated_at", uint64_t{0});
+    return r;
+}
+
+} // anonymous namespace
+
+std::string onDeleteToString(OnDelete v) {
+    switch (v) {
+        case OnDelete::Restrict: return "restrict";
+        case OnDelete::Cascade: return "cascade";
+        case OnDelete::SetNull: return "set_null";
+        case OnDelete::NoAction: return "no_action";
+    }
+    return "restrict";
+}
+
+std::optional<OnDelete> parseOnDelete(const std::string& s) {
+    if (s == "restrict") return OnDelete::Restrict;
+    if (s == "cascade") return OnDelete::Cascade;
+    if (s == "set_null") return OnDelete::SetNull;
+    if (s == "no_action") return OnDelete::NoAction;
+    return std::nullopt;
+}
+
+std::string RelationManager::canonical(const std::string& name) {
+    if (name.empty()) return name;
+    try {
+        return resolveCollection(name).qualified;
+    } catch (const std::exception&) {
+        return name;   // see the header: unparseable matches nothing, never throws
+    }
+}
+
+RelationManager::RelationManager(MemoryStore& store) : store_(store) {}
+
+void RelationManager::loadFromStore() {
+    std::unique_lock<std::shared_mutex> lock(mutex_);
+    cache_.clear();
+
+    // Ensure _relations collection exists.
+    CollectionOptions opts;
+    store_.createCollection(SYSTEM_COLLECTION, opts);
+
+    // Page explicitly. Query::limit defaults to 100, and limit=0 returns
+    // nothing (not everything) - the same trap that has already shipped
+    // as a bug in ViewManager, PolicyManager and CollectionConfigManager.
+    constexpr uint32_t kPage = 500;
+    uint32_t offset = 0;
+    // Collected during the walk and applied after it: mutating _relations
+    // while paging over it would shift the offsets underneath us.
+    std::vector<std::pair<std::string, RelationInfo>> rekeys;
+    while (true) {
+        Query q;
+        q.limit = kPage;
+        q.offset = offset;
+        auto res = store_.find(SYSTEM_COLLECTION, q);
+        if (res.documents.empty()) break;
+        for (const auto& d : res.documents) {
+            if (d.id.empty()) continue;
+            RelationInfo r = fromJson(d.data());
+            if (r.name.empty()) continue;
+
+            // v2.11.0 T13 round 2 — createRelation refuses a cross-project
+            // declaration at write time (see below), but this load loop is
+            // the OTHER entry point into the cache and did not re-check it.
+            // A legacy or hand-written `_relations` document naming a
+            // cross-project parent would arm a bare parent collection name
+            // that then gets resolved inside the CHILD's own project env
+            // (armRelationsForChild/applyRelationDeclarations only ever
+            // resolve `r.parent` against the child's project) - so
+            // validate_on_write would check an unrelated, wrong collection
+            // for existence, silently accepting or rejecting for the wrong
+            // reason. Skip and log rather than fail the whole load: one bad
+            // record must not stop every other relation from arming, same
+            // reasoning as applyRelationDeclarations' per-relation try/catch.
+            try {
+                const auto rn = resolveCollection(r.name);
+                const auto rc = resolveCollection(r.child);
+                const auto rp = resolveCollection(r.parent);
+                // v2.11.0 final review (finding 8) — canonicalise on load, so
+                // the cache is keyed the same way regardless of what form the
+                // stored record used. Legacy/hand-written records naming
+                // "exec_wf" and "default:exec_wf" now resolve to the same
+                // entry instead of coexisting as two relations that arm the
+                // same collection and overwrite each other.
+                r.name = rn.qualified;
+                r.child = rc.qualified;
+                r.parent = rp.qualified;
+                if (rc.project != rp.project || rc.project != rn.project) {
+                    spdlog::error(
+                        "RelationManager: skipping relation '{}' - name/child/parent "
+                        "span more than one project ({}/{}/{}), which createRelation() "
+                        "refuses today; this record predates that check or was written "
+                        "by hand",
+                        r.name, rn.project, rc.project, rp.project);
+                    continue;
+                }
+            } catch (const std::exception& e) {
+                spdlog::error("RelationManager: skipping unparseable relation '{}': {}",
+                              r.name, e.what());
+                continue;
+            }
+
+            // v2.11.0 final review (finding 8) — a record stored under a
+            // non-canonical document id is re-keyed in the store as well,
+            // not just in the cache: dropRelation() removes by the canonical
+            // name, so leaving the row under its old id would make the
+            // relation undroppable. Same migration shape ViewManager uses for
+            // its legacy bare-named views. Advisory - a failed re-key leaves
+            // the cache correct and logs, it does not fail the load.
+            if (d.id != r.name) {
+                rekeys.push_back({d.id, r});
+            }
+
+            cache_[r.name] = std::move(r);
+        }
+        if (res.documents.size() < kPage) break;
+        offset += kPage;
+    }
+
+    for (const auto& [oldId, rel] : rekeys) {
+        try {
+            Document rekeyed;
+            rekeyed.id = rel.name;
+            rekeyed.set_data(toJson(rel));
+            store_.upsert(SYSTEM_COLLECTION, rekeyed);
+            store_.remove(SYSTEM_COLLECTION, oldId);
+            spdlog::warn("RelationManager: re-keyed legacy relation '{}' -> '{}'",
+                         oldId, rel.name);
+        } catch (const std::exception& e) {
+            spdlog::warn("RelationManager: could not re-key legacy relation '{}': {}",
+                         oldId, e.what());
+        }
+    }
+
+    spdlog::info("RelationManager: loaded {} relation(s) from {}", cache_.size(), SYSTEM_COLLECTION);
+}
+
+bool RelationManager::createRelation(const RelationInfo& r, std::string& errorOut) {
+    if (r.name.empty()) {
+        errorOut = "relation name is required";
+        return false;
+    }
+    if (r.child.empty()) {
+        errorOut = "child collection is required";
+        return false;
+    }
+    if (r.parent.empty()) {
+        errorOut = "parent collection is required";
+        return false;
+    }
+    if (r.childField.empty()) {
+        errorOut = "child_field is required";
+        return false;
+    }
+
+    // Each project owns its own LMDB env and no transaction spans two, so a
+    // cross-project relation could never be enforced atomically.
+    ResolvedCollection rn, rc, rp;
+    try {
+        rn = resolveCollection(r.name);
+        rc = resolveCollection(r.child);
+        rp = resolveCollection(r.parent);
+    } catch (const std::exception& e) {
+        errorOut = std::string("invalid relation/child/parent name: ") + e.what();
+        return false;
+    }
+    if (rc.project != rp.project || rc.project != rn.project) {
+        errorOut = "relation, child and parent must be in one project (no "
+                   "transaction spans two project envs)";
+        return false;
+    }
+
+    RelationInfo out = r;
+    // v2.11.0 final review (finding 8) — CANONICALISE BEFORE ANYTHING ELSE.
+    // The declaration is persisted and cached under the canonical form only,
+    // so a caller sending "exec_wf" and one sending "default:exec_wf" now
+    // create, find and collide with the SAME relation. Before this, the raw
+    // string was stored verbatim, which is how a raw-gRPC caller could arm a
+    // second, near-identical relation over the same child collection and
+    // clobber the first one's armed state.
+    out.name = rn.qualified;
+    out.child = rc.qualified;
+    out.parent = rp.qualified;
+
+    {
+        std::shared_lock<std::shared_mutex> rlock(mutex_);
+        if (cache_.contains(out.name)) {
+            errorOut = "relation '" + out.name + "' already exists";
+            return false;
+        }
+    }
+
+    out.createdAt = nowMs();
+    out.updatedAt = out.createdAt;
+
+    Document doc;
+    doc.id = out.name;
+    doc.set_data(toJson(out));
+    try {
+        std::string id = store_.insert(SYSTEM_COLLECTION, doc);
+        if (id.empty()) {
+            errorOut = "failed to persist relation declaration";
+            return false;
+        }
+    } catch (const std::exception& e) {
+        errorOut = std::string("failed to persist relation declaration: ") + e.what();
+        return false;
+    }
+
+    {
+        std::unique_lock<std::shared_mutex> wlock(mutex_);
+        cache_[out.name] = out;
+    }
+    spdlog::info("RelationManager: created relation '{}' ({} -> {} via {})",
+                 out.name, out.child, out.parent, out.childField);
+    return true;
+}
+
+bool RelationManager::dropRelation(const std::string& qualifiedName, std::string& errorOut) {
+    // finding 8 — canonicalise, so dropping "exec_wf" drops the relation a
+    // caller created as "default:exec_wf". Reported back under the name the
+    // CALLER used, since that is the string they can act on.
+    const std::string key = canonical(qualifiedName);
+    {
+        std::shared_lock<std::shared_mutex> rlock(mutex_);
+        if (!cache_.contains(key)) {
+            errorOut = "relation '" + qualifiedName + "' does not exist";
+            return false;
+        }
+    }
+
+    bool removed = store_.remove(SYSTEM_COLLECTION, key);
+    if (!removed) {
+        errorOut = "failed to remove relation from store";
+        return false;
+    }
+
+    {
+        std::unique_lock<std::shared_mutex> wlock(mutex_);
+        cache_.erase(key);
+    }
+    spdlog::info("RelationManager: dropped relation '{}'", key);
+    return true;
+}
+
+std::optional<RelationInfo> RelationManager::getRelation(const std::string& qualifiedName) const {
+    const std::string key = canonical(qualifiedName);
+    std::shared_lock<std::shared_mutex> lock(mutex_);
+    auto it = cache_.find(key);
+    if (it == cache_.end()) return std::nullopt;
+    return it->second;
+}
+
+std::vector<RelationInfo> RelationManager::listRelations(const std::string& project) const {
+    std::shared_lock<std::shared_mutex> lock(mutex_);
+    std::vector<RelationInfo> out;
+    out.reserve(cache_.size());
+    for (const auto& [_, r] : cache_) {
+        if (!project.empty()) {
+            ResolvedCollection rn;
+            try {
+                rn = resolveCollection(r.name);
+            } catch (const std::exception&) {
+                continue;
+            }
+            if (rn.project != project) continue;
+        }
+        out.push_back(r);
+    }
+    return out;
+}
+
+std::vector<RelationInfo> RelationManager::relationsWithParent(const std::string& collection) const {
+    // finding 8 — canonicalise the needle. Cached `parent` values are always
+    // canonical (createRelation and loadFromStore both make them so), so a
+    // bare argument would otherwise match nothing and silently report "this
+    // collection is nobody's parent", permitting every delete.
+    const std::string key = canonical(collection);
+    std::shared_lock<std::shared_mutex> lock(mutex_);
+    std::vector<RelationInfo> out;
+    for (const auto& [_, r] : cache_) {
+        if (r.parent == key) out.push_back(r);
+    }
+    return out;
+}
+
+std::vector<RelationInfo> RelationManager::relationsWithChild(const std::string& collection) const {
+    const std::string key = canonical(collection);
+    std::shared_lock<std::shared_mutex> lock(mutex_);
+    std::vector<RelationInfo> out;
+    for (const auto& [_, r] : cache_) {
+        if (r.child == key) out.push_back(r);
+    }
+    return out;
+}
+
+} // namespace smartbotic::database

+ 160 - 0
service/src/relations/relation_manager.hpp

@@ -0,0 +1,160 @@
+#pragma once
+
+#include <optional>
+#include <shared_mutex>
+#include <string>
+#include <unordered_map>
+#include <vector>
+
+namespace smartbotic::database {
+
+class MemoryStore;
+
+/**
+ * Behavior when the PARENT side of a relation is deleted. Not enforced by
+ * this task - RelationManager only stores the declaration. Enforcement
+ * (Restrict/Cascade/SetNull/NoAction semantics on delete) is a later task.
+ */
+enum class OnDelete { Restrict, Cascade, SetNull, NoAction };
+
+/**
+ * v2.11.0 final review (finding 9) — the SINGLE string->OnDelete parser.
+ *
+ * Returns nullopt for anything that is not exactly one of "restrict",
+ * "cascade", "set_null", "no_action". There used to be two copies of this
+ * (relationOnDeleteFromString in database_grpc_impl.cpp and
+ * onDeleteFromString in relation_manager.cpp) and BOTH silently coerced an
+ * unrecognised string to Restrict. That failed safe while cascade/set_null
+ * were inert, but T12 made those strings destructive in the other
+ * direction: an operator who types "Cascade" was told the relation was
+ * created and believed cascade was armed while `restrict` actually was, so
+ * the deletes they expected to cascade started failing with
+ * FAILED_PRECONDITION instead - and nothing anywhere named the typo.
+ *
+ * Rendering back to a string lives in onDeleteToString(), also here, so the
+ * two directions cannot drift.
+ */
+std::optional<OnDelete> parseOnDelete(const std::string& s);
+std::string onDeleteToString(OnDelete v);
+
+// Every string an on_delete value may take, for error messages.
+inline constexpr const char* kOnDeleteValues = "restrict|cascade|set_null|no_action";
+
+/**
+ * In-memory representation of a relation declaration.
+ * Mirrors the (future) RelationDefinition proto message.
+ *
+ * A relation says: documents in `child` reference documents in `parent`
+ * through `childField` (a dot-path that may resolve to a single id or an
+ * array of ids). `onDelete` and `validateOnWrite` describe how a later
+ * enforcement layer should behave; this task does not enforce them.
+ */
+struct RelationInfo {
+    std::string name;            // qualified "<project>:<name>"
+    std::string child;           // qualified "<project>:<collection>"
+    std::string childField;      // dot-path, may resolve to an array of ids
+    std::string parent;          // qualified "<project>:<collection>"
+    OnDelete onDelete = OnDelete::Restrict;
+    bool validateOnWrite = false;
+    uint64_t createdAt = 0;
+    uint64_t updatedAt = 0;
+};
+
+/**
+ * RelationManager — loads, caches, and manages relation declarations.
+ *
+ * Relations are persisted as documents in the `_relations` system
+ * collection, shared globally across projects (like `_views`), so each
+ * relation's `name` carries its own project qualifier and is used as the
+ * document id. On startup, loadFromStore() populates an in-memory cache
+ * for O(1) lookup. Mutations (createRelation/dropRelation) update both
+ * the store and the cache.
+ *
+ * This task builds the declaration registry only: no enforcement (no
+ * on-write validation, no on-delete cascade/restrict), no LMDB reverse
+ * index. Those are later tasks in the relations-v2.11.0 plan.
+ */
+class RelationManager {
+public:
+    static constexpr const char* SYSTEM_COLLECTION = "_relations";
+
+    explicit RelationManager(MemoryStore& store);
+
+    /**
+     * Load all relation declarations from the _relations collection into
+     * the cache. Call once at startup, AFTER MemoryStore has loaded
+     * persisted state. Pages explicitly - Query::limit defaults to 100.
+     */
+    void loadFromStore();
+
+    /**
+     * Declare a relation. Fails if the name already exists, or if name,
+     * child and parent do not all resolve to the same project (no LMDB
+     * transaction spans two project envs, so cross-project relations
+     * could never be enforced atomically).
+     * @return true on success. On failure errorOut is populated.
+     */
+    bool createRelation(const RelationInfo& r, std::string& errorOut);
+
+    /**
+     * Drop a relation by its qualified name.
+     */
+    bool dropRelation(const std::string& qualifiedName, std::string& errorOut);
+
+    /**
+     * Lookup a relation by its qualified name. Returns nullopt if absent.
+     */
+    std::optional<RelationInfo> getRelation(const std::string& qualifiedName) const;
+
+    /**
+     * List relations. If `project` is empty, lists every relation across
+     * every project; otherwise filters to relations whose name is
+     * qualified with that project.
+     */
+    std::vector<RelationInfo> listRelations(const std::string& project = "") const;
+
+    // Relations whose PARENT is this collection - what a delete must
+    // consult. See the canonicalisation note below: the argument may be
+    // bare or qualified.
+    std::vector<RelationInfo> relationsWithParent(const std::string& collection) const;
+
+    // Relations whose CHILD is this collection - what a write must
+    // maintain. Same canonicalisation as relationsWithParent.
+    std::vector<RelationInfo> relationsWithChild(const std::string& collection) const;
+
+private:
+    /**
+     * v2.11.0 final review (finding 8) — EVERY name crossing this class's
+     * boundary is canonicalised to "<project>:<name>" here, on the way in
+     * and on the way out, and the cache is keyed by the canonical form only.
+     *
+     * This is the STRUCTURAL fix for the bare-vs-qualified bug class, which
+     * this repository has now shipped three times: v2.4.2 lost every view
+     * for two releases (createView qualified `collection` but not `name`),
+     * v2.4.5 gave every collection a phantom twin (createCollection sent a
+     * bare name while insert sent a qualified one), and on this branch
+     * armRelationsForChild looked relations up under the caller's raw
+     * string while boot arming used the canonical one - so a raw-gRPC caller
+     * naming "executions" instead of "default:executions" could REPLACE or
+     * ERASE the real relation's armed state for the life of the process,
+     * logged only as "re-armed N relation(s)", and a restart silently
+     * repaired it.
+     *
+     * Spot-fixing each call site is what produced three occurrences. Doing
+     * it at the one funnel every entry point already goes through (the RPC
+     * handlers, the boot re-arm, the CLI and loadFromStore all reach
+     * relations exclusively through this class) makes the mismatch
+     * unrepresentable instead of merely absent today.
+     *
+     * Unparseable input is returned UNCHANGED rather than throwing: a bad
+     * name then simply matches nothing, which is what the callers already
+     * handle, and no lookup should be able to abort a delete or a boot.
+     */
+    static std::string canonical(const std::string& name);
+
+    MemoryStore& store_;
+    mutable std::shared_mutex mutex_;
+    std::unordered_map<std::string, RelationInfo> cache_;
+};
+
+} // namespace smartbotic::database

+ 48 - 0
service/src/storage/document_store.hpp

@@ -46,6 +46,35 @@ public:
                      std::string_view id,
                      const smartbotic::database::Document& doc) = 0;
 
+    // v2.11.0 T12 round-4 — put several documents (possibly across several
+    // collections) as ONE atomic unit.
+    //
+    // Contract: all-or-nothing. Either every item is written, or the call
+    // throws and NOTHING it touched was written. Callers that need
+    // per-row failure isolation retry the batch through single-item put()
+    // calls after catching — which is what MemoryStore::remirrorDocuments()
+    // does, so one bad row cannot leave a whole chunk stale.
+    //
+    // Why it exists: the post-WAL-replay re-mirror pass runs on the boot
+    // path before sd_notify(READY=1), and put() commits its own transaction
+    // (one fsync) per document. On a crash boot with a large WAL, or under
+    // --recovery-mode=wal_only, per-document commits are minutes of startup.
+    //
+    // `items` holds NON-OWNING Document pointers - they must outlive the
+    // call. The default implementation is a non-atomic loop over put(), for
+    // backends (and test doubles) with no transaction concept of their own;
+    // LmdbDocumentStore overrides it with a single WriteTxn.
+    struct BatchPutItem {
+        std::string_view collection;
+        std::string_view id;
+        const smartbotic::database::Document* doc;
+    };
+    virtual void put_batch(const std::vector<BatchPutItem>& items) {
+        for (const auto& it : items) {
+            if (it.doc) put(it.collection, it.id, *it.doc);
+        }
+    }
+
     // Read a document. Returns nullopt if absent or collection doesn't exist.
     virtual std::optional<smartbotic::database::Document>
     get(std::string_view collection, std::string_view id) = 0;
@@ -79,6 +108,25 @@ public:
     // Drop a collection sub-db entirely. Returns true if it existed.
     virtual bool drop_collection(std::string_view collection) = 0;
 
+    // v2.11.0 final review (finding 5) — can a put() on this collection
+    // REJECT the write (UniqueViolation / MissingParentReference)?
+    //
+    // MemoryStore's update-shaped write paths must snapshot the pre-write
+    // Document so they can undo their in-memory mutation when the mirror
+    // rejects it (see MemoryStore::mirrorDocOrUndo). That copy is
+    // yyjson_mut_val_mut_copy - O(document size), a fresh allocation, taken
+    // inside the collection's unique_lock - and T11 added it unconditionally,
+    // so every install paid it on every update whether or not any unique
+    // field or relation existed. Where none exists the undo can never fire, so
+    // the copy is pure cost, on the exact path v2.8.0 spent a release making
+    // cheaper.
+    //
+    // The DEFAULT IS true - fail safe. A backend that does not know must be
+    // treated as if it can reject, because the alternative (skipping the
+    // snapshot when a rejection is in fact possible) means a rejected write
+    // leaves its mutation in MemoryStore unenforced.
+    virtual bool can_reject_writes(std::string_view /*collection*/) { return true; }
+
     // ==========================================================================
     // v2.0 Stage 5 — vector sub-db operations.
     //

+ 677 - 40
service/src/storage/document_store_lmdb.cpp

@@ -50,6 +50,7 @@
 #include "doc_binary.hpp"
 #include "document.hpp"
 #include "json_parse.hpp"
+#include "relations/relation_index.hpp"
 #include "storage/filter_eval.hpp"
 #include "storage/lmdb_dbi.hpp"
 #include "storage/lmdb_env.hpp"
@@ -194,6 +195,31 @@ std::optional<nlohmann::json> resolve_from_yyjson(yyjson_val* root,
     return to_json(cur);
 }
 
+// v2.11.0 T3 — the set of parent ids a reference field value names.
+//
+// Relation ids are raw strings (unlike secondary index keys, they need no
+// type-tagged encoding - see relation_index.hpp), so this is simpler than
+// encode_index_keys: a string value names one id, an array value names one
+// id per STRING element (non-string elements are not valid ids and are
+// skipped rather than stringified, since an id must round-trip exactly for
+// relation_index_remove to find it again). Absent/null must never reach
+// here as a reference - callers pass std::nullopt for those, which this
+// treats the same as "no ids".
+std::vector<std::string> extract_relation_ids(const std::optional<nlohmann::json>& v) {
+    std::vector<std::string> out;
+    if (!v || v->is_null()) return out;
+    if (v->is_string()) {
+        out.push_back(v->get<std::string>());
+    } else if (v->is_array()) {
+        for (const auto& el : *v) {
+            if (el.is_string()) out.push_back(el.get<std::string>());
+        }
+    }
+    std::sort(out.begin(), out.end());
+    out.erase(std::unique(out.begin(), out.end()), out.end());
+    return out;
+}
+
 void sort_documents(std::vector<smartbotic::database::Document>& docs,
                     const smartbotic::database::Sort& sort) {
     std::sort(docs.begin(), docs.end(),
@@ -364,6 +390,27 @@ void LmdbDocumentStore::cacheCommittedDbi(std::string_view collection,
     dbi_cache_[std::string(collection)] = dbi;
 }
 
+std::optional<unsigned int>
+LmdbDocumentStore::cachedDbi(std::string_view collection) {
+    std::lock_guard<std::mutex> lock(cache_mutex_);
+    auto it = dbi_cache_.find(std::string(collection));
+    if (it == dbi_cache_.end()) return std::nullopt;
+    return it->second;
+}
+
+WriteTxn LmdbDocumentStore::beginWrite() {
+    return WriteTxn(env_);
+}
+
+void LmdbDocumentStore::commitAndCache(
+    WriteTxn& wtxn,
+    const std::vector<std::pair<std::string, unsigned int>>& to_cache) {
+    // Commit BEFORE any handle is cached - the v2.8.0 lesson, same as every
+    // no-txn wrapper in this file.
+    wtxn.commit();
+    for (const auto& [sub, d] : to_cache) cacheCommittedDbi(sub, d);
+}
+
 std::optional<unsigned int>
 LmdbDocumentStore::try_open_for_read(ReadTxn& rtxn,
                                       std::string_view collection) {
@@ -449,8 +496,11 @@ size_t LmdbDocumentStore::prime_dbi_cache() {
     for (const auto& n : names) {
         MDB_dbi dbi = 0;
         // Index sub-dbs are MDB_DUPSORT; pass the flag on reopen so the handle
-        // agrees with how the sub-db was created.
-        const unsigned int flags = is_index_subdb(n) ? MDB_DUPSORT : 0u;
+        // agrees with how the sub-db was created. Relation reverse indexes
+        // (v2.11.0 T2) are DUPSORT for the same reason - one parent id key
+        // holding a sorted set of child ids.
+        const unsigned int flags =
+            (is_index_subdb(n) || is_relation_index_subdb(n)) ? MDB_DUPSORT : 0u;
         int rc = mdb_dbi_open(wtxn.raw(), n.c_str(), flags, &dbi);
         if (rc == MDB_NOTFOUND) continue;          // vanished under us; ignore
         if (rc != MDB_SUCCESS) throw_mdb(rc, "dbi_open (prime)");
@@ -471,30 +521,76 @@ size_t LmdbDocumentStore::prime_dbi_cache() {
 void LmdbDocumentStore::put(std::string_view collection,
                              std::string_view id,
                              const smartbotic::database::Document& doc) {
-    std::string payload = encode_document(doc);
     WriteTxn wtxn(env_);
+    std::vector<std::pair<std::string, unsigned int>> to_cache;
+    put(wtxn, collection, id, doc, to_cache);
+    // Commit BEFORE any handle is cached - the v2.8.0 lesson. Every handle in
+    // to_cache is private to wtxn until this succeeds.
+    wtxn.commit();
+    for (const auto& [sub, d] : to_cache) cacheCommittedDbi(sub, d);
+}
+
+void LmdbDocumentStore::put_batch(const std::vector<BatchPutItem>& items) {
+    if (items.empty()) return;
+    WriteTxn wtxn(env_);
+    std::vector<std::pair<std::string, unsigned int>> to_cache;
+    for (const auto& item : items) {
+        if (!item.doc) continue;
+        put(wtxn, item.collection, item.id, *item.doc, to_cache);
+    }
+    // Commit BEFORE any handle is cached - the v2.8.0 lesson. If any put()
+    // above threw, wtxn's destructor aborts everything and to_cache is
+    // discarded unapplied, which is exactly right: LMDB closes handles whose
+    // opening transaction aborted, so caching one would poison the sub-db for
+    // the life of the process.
+    wtxn.commit();
+    for (const auto& [sub, d] : to_cache) cacheCommittedDbi(sub, d);
+}
+
+void LmdbDocumentStore::put(WriteTxn& wtxn,
+                             std::string_view collection,
+                             std::string_view id,
+                             const smartbotic::database::Document& doc,
+                             std::vector<std::pair<std::string, unsigned int>>& to_cache) {
+    std::string payload = encode_document(doc);
     unsigned int dbi = open_for_write(wtxn, collection);
+    // Recorded immediately, not after mdb_put succeeds: del() and the
+    // maintainIndexes/maintainRelations helpers all record their handle right
+    // after opening it, and a handle from a transaction that later aborts is
+    // harmless to have recorded (the caller only ever applies to_cache after
+    // ITS commit succeeds). Recording it late, only after mdb_put, meant a
+    // throw from mdb_put on a brand-new collection left the collection's own
+    // handle out of to_cache while any index/relation handles it. A caller
+    // that somehow committed after catching would then have committed a
+    // sub-db whose handle nothing had cached - a repeat of the v2.8.1 class
+    // of bug ("callers must remember" is not an enforcement mechanism).
+    to_cache.emplace_back(std::string(collection), dbi);
     MDB_val k = to_val(id);
 
     // v2.9.0 — index maintenance runs in THIS transaction, so a write that
     // throws after this point rolls the index back with the document. An
     // asynchronously-maintained index would let a query read entries for a row
     // that was never stored, and return silently wrong rows rather than an error.
-    std::vector<std::pair<std::string, unsigned int>> index_dbis;
-    if (!indexed_fields(collection).empty()) {
+    const bool needs_index = !indexed_fields(collection).empty();
+    const bool needs_relations = !relations(collection).empty();
+    if (needs_index || needs_relations) {
         MDB_val old{0, nullptr};
         const int rc = mdb_get(wtxn.raw(), dbi, &k, &old);
         if (rc != MDB_SUCCESS && rc != MDB_NOTFOUND) throw_mdb(rc, "get (pre-index)");
         const std::string_view old_payload =
             rc == MDB_SUCCESS ? to_sv(old) : std::string_view{};
-        maintainIndexes(wtxn, collection, id, old_payload, &doc, index_dbis);
+        if (needs_index) {
+            maintainIndexes(wtxn, collection, id, old_payload, &doc, to_cache);
+        }
+        if (needs_relations) {
+            maintainRelations(wtxn, collection, id, old_payload, &doc, to_cache);
+        }
     }
 
     MDB_val v = to_val(payload);
     mdb_check(mdb_put(wtxn.raw(), dbi, &k, &v, 0), "put");
-    wtxn.commit();
-    cacheCommittedDbi(collection, dbi);
-    for (const auto& [sub, d] : index_dbis) cacheCommittedDbi(sub, d);
+    // NOT cached here - see the header contract. The caller caches every
+    // entry in to_cache (recorded above) only after ITS commit succeeds.
 }
 
 
@@ -638,6 +734,198 @@ void LmdbDocumentStore::maintainIndexes(
     }
 }
 
+// -------------------------------------------------------------------------
+// v2.11.0 T3 — relation reverse-index maintenance on child writes.
+// -------------------------------------------------------------------------
+
+void LmdbDocumentStore::set_relations(std::string_view collection,
+                                       std::vector<RelationRef> rels) {
+    std::lock_guard<std::mutex> lock(relations_mutex_);
+    if (rels.empty()) {
+        relations_.erase(std::string(collection));
+    } else {
+        relations_[std::string(collection)] = std::move(rels);
+    }
+    // finding 5 — count only collections that can actually REJECT a write. A
+    // relation without validateOnWrite still maintains its reverse index on
+    // every put, but it never throws, so it cannot make an undo snapshot
+    // necessary.
+    size_t rejecting = 0;
+    for (const auto& [_, refs] : relations_) {
+        for (const auto& ref : refs) {
+            if (ref.validateOnWrite) { ++rejecting; break; }
+        }
+    }
+    rejecting_relation_collections_.store(rejecting, std::memory_order_relaxed);
+}
+
+bool LmdbDocumentStore::can_reject_writes(std::string_view collection) {
+    // The overwhelmingly common case: nothing anywhere in this process has a
+    // rejecting constraint declared, so no mutex is taken and no string is
+    // constructed. This is the check that keeps T11's undo snapshot off the
+    // write path of every install that does not use unique fields or
+    // validate_on_write.
+    if (rejecting_unique_collections_.load(std::memory_order_relaxed) == 0 &&
+        rejecting_relation_collections_.load(std::memory_order_relaxed) == 0) {
+        return false;
+    }
+    const std::string key(collection);
+    {
+        std::lock_guard<std::mutex> lock(index_mutex_);
+        if (unique_fields_.contains(key)) return true;
+    }
+    {
+        std::lock_guard<std::mutex> lock(relations_mutex_);
+        auto it = relations_.find(key);
+        if (it != relations_.end()) {
+            for (const auto& ref : it->second) {
+                if (ref.validateOnWrite) return true;
+            }
+        }
+    }
+    return false;
+}
+
+std::vector<RelationRef> LmdbDocumentStore::relations(std::string_view collection) {
+    std::lock_guard<std::mutex> lock(relations_mutex_);
+    auto it = relations_.find(std::string(collection));
+    if (it == relations_.end()) return {};
+    return it->second;
+}
+
+void LmdbDocumentStore::maintainRelations(
+    WriteTxn& wtxn,
+    std::string_view collection,
+    std::string_view id,
+    std::string_view old_payload,
+    const smartbotic::database::Document* new_doc,
+    std::vector<std::pair<std::string, unsigned int>>& to_cache) {
+
+    const auto rels = relations(collection);
+    if (rels.empty()) return;   // collections with no relations pay nothing
+
+    // Old references come from the STORED bytes, parsed with yyjson and
+    // resolved one field at a time - never materialised into a Document,
+    // for the same reason maintainIndexes avoids it (decode_document was
+    // measured at 88% of write cost on a large row).
+    std::unordered_map<std::string, std::vector<std::string>> old_ids_by_field;
+    if (!old_payload.empty()) {
+        yyjson_doc* d = yyjson_read(old_payload.data(), old_payload.size(), 0);
+        if (d) {
+            yyjson_val* root = yyjson_doc_get_root(d);
+            for (const auto& r : rels) {
+                if (old_ids_by_field.count(r.childField)) continue;
+                old_ids_by_field[r.childField] =
+                    extract_relation_ids(resolve_from_yyjson(root, r.childField));
+            }
+            yyjson_doc_free(d);
+        }
+    }
+
+    for (const auto& r : rels) {
+        std::vector<std::string> new_ids;
+        if (new_doc != nullptr) {
+            // resolveFilterValue is the SAME resolver queries/filters use, so a
+            // relation and a query can never disagree about which field a
+            // dot-path names.
+            new_ids = extract_relation_ids(
+                filter_eval::resolveFilterValue(*new_doc, r.childField));
+        }
+        std::vector<std::string> old_ids;
+        if (auto it = old_ids_by_field.find(r.childField); it != old_ids_by_field.end()) {
+            old_ids = it->second;
+        }
+
+        // Unchanged reference costs no index write - the common case for an
+        // update that touches other fields. Both sides are sorted/deduped by
+        // extract_relation_ids, so this comparison is exact.
+        if (old_ids == new_ids) continue;
+
+        std::vector<std::string> to_remove;
+        std::vector<std::string> to_add;
+        std::set_difference(old_ids.begin(), old_ids.end(), new_ids.begin(), new_ids.end(),
+                            std::back_inserter(to_remove));
+        std::set_difference(new_ids.begin(), new_ids.end(), old_ids.begin(), old_ids.end(),
+                            std::back_inserter(to_add));
+        if (to_remove.empty() && to_add.empty()) continue;
+
+        // v2.11.0 T13 — validate_on_write: every id newly appearing in
+        // `to_add` must name a parent that already exists, checked with an
+        // mdb_get on the parent's own sub-db INSIDE this same write
+        // transaction. Only `to_add` is checked, not the full `new_ids` set -
+        // a reference that did not change this write already existed (or was
+        // already dangling) before this write and is not what this task
+        // closes the race for; re-validating it on every unrelated field
+        // update would also defeat the "unchanged reference costs no write"
+        // short-circuit above.
+        //
+        // Array-valued reference decision: ANY missing parent id rejects the
+        // WHOLE write, not just that element. Silently keeping the
+        // resolvable elements while dropping the unresolvable ones would
+        // discard data the caller explicitly supplied without telling them,
+        // and would make the same array partially "valid" depending on write
+        // order - the same fail-closed, all-or-nothing reasoning
+        // UniqueViolation already applies to the rest of the document.
+        //
+        // Absent/null references never reach here: extract_relation_ids /
+        // resolveFilterValue never put them in `new_ids`, so they can never
+        // land in `to_add`.
+        //
+        // r.relationsEnforced gates this exactly like it gates restrict/
+        // no_action in RelationEnforcer::canDelete: the documented escape
+        // hatch for a bulk import or a collection under write pressure is
+        // "enforcement off means every relation policy is skipped, not just
+        // restrict" - a validate_on_write rejection during a bulk import
+        // whose parents are not loaded yet is precisely the case the switch
+        // exists for. Before this, the only way to stop the rejection was to
+        // drop and re-declare the relation without validateOnWrite, which is
+        // not what the switch is for.
+        if (r.validateOnWrite && r.relationsEnforced && !to_add.empty()) {
+            // open_for_write, not try_open_for_read: this call happens
+            // inside our own write transaction (the only kind that may call
+            // mdb_dbi_open - see try_open_for_read's file comment), and
+            // passing MDB_CREATE is harmless here - if the parent collection
+            // truly has never been written, mdb_get below finds nothing for
+            // every id, this throws, and the whole transaction (including
+            // that dbi's creation) aborts with it. The parent's own handle
+            // is queued into `to_cache` like every other handle this
+            // function opens, so a later child write reuses the cached one
+            // once THIS write commits.
+            const unsigned int parent_dbi = open_for_write(wtxn, r.parent);
+            to_cache.emplace_back(r.parent, parent_dbi);
+            for (const auto& parentId : to_add) {
+                MDB_val pk = to_val(parentId);
+                MDB_val pv{0, nullptr};
+                const int prc = mdb_get(wtxn.raw(), parent_dbi, &pk, &pv);
+                if (prc == MDB_NOTFOUND) {
+                    throw MissingParentReference(r.name, r.childField, parentId);
+                }
+                if (prc != MDB_SUCCESS) throw_mdb(prc, "get (validate_on_write)");
+            }
+        }
+
+        const std::string sub = relation_index_subdb(r.name);
+        const unsigned int dbi = open_for_write(wtxn, sub, MDB_DUPSORT);
+        to_cache.emplace_back(sub, dbi);
+
+        for (const auto& parentId : to_remove) {
+            MDB_val pk = to_val(parentId);
+            MDB_val cv = to_val(id);
+            // DUPSORT: passing the data removes just this (parent, child) pair -
+            // siblings of the same parent are untouched.
+            const int rc = mdb_del(wtxn.raw(), dbi, &pk, &cv);
+            if (rc != MDB_SUCCESS && rc != MDB_NOTFOUND) throw_mdb(rc, "relation index del");
+        }
+        for (const auto& parentId : to_add) {
+            MDB_val pk = to_val(parentId);
+            MDB_val cv = to_val(id);
+            // MDB_NODUPDATA makes a repeat put a no-op rather than an error.
+            const int rc = mdb_put(wtxn.raw(), dbi, &pk, &cv, MDB_NODUPDATA);
+            if (rc != MDB_SUCCESS && rc != MDB_KEYEXIST) throw_mdb(rc, "relation index put");
+        }
+    }
+}
+
 std::optional<uint64_t>
 LmdbDocumentStore::index_count_eq(std::string_view collection,
                                    const std::string& field,
@@ -996,6 +1284,278 @@ bool LmdbDocumentStore::drop_index(std::string_view collection,
     return true;
 }
 
+// -------------------------------------------------------------------------
+// v2.11.0 T2 — relation reverse index.
+//
+// Same DUPSORT-posting-list shape as secondary indexes above: key = parent
+// id, data = child id. mdb_cursor_count answers "how many children" without
+// reading any of them - the operation restrict/DescribeDelete (Task 4/5)
+// need. No document-write hook calls relation_index_add yet (Task 3); these
+// are pure storage primitives.
+// -------------------------------------------------------------------------
+
+bool LmdbDocumentStore::relation_index_add(std::string_view relation,
+                                            std::string_view parentId,
+                                            std::string_view childId) {
+    const std::string sub = relation_index_subdb(relation);
+    WriteTxn wtxn(env_);
+    const unsigned int dbi = open_for_write(wtxn, sub, MDB_DUPSORT);
+    MDB_val k = to_val(parentId);
+    MDB_val v = to_val(childId);
+    const int rc = mdb_put(wtxn.raw(), dbi, &k, &v, MDB_NODUPDATA);
+    // MDB_KEYEXIST under MDB_NODUPDATA means this exact (key, data) pair is
+    // already present - idempotent re-add, not an error.
+    if (rc != MDB_SUCCESS && rc != MDB_KEYEXIST) throw_mdb(rc, "relation index add");
+    wtxn.commit();
+    // MANDATORY - see the global constraint. A sub-db created at runtime is
+    // invisible to try_open_for_read until its opening transaction commits
+    // AND the handle is cached; skipping this leaves the read path reporting
+    // "no such sub-db" for the life of the process.
+    cacheCommittedDbi(sub, dbi);
+    return true;
+}
+
+bool LmdbDocumentStore::relation_index_remove(std::string_view relation,
+                                               std::string_view parentId,
+                                               std::string_view childId) {
+    const std::string sub = relation_index_subdb(relation);
+    {
+        ReadTxn rtxn(env_);
+        if (!try_open_for_read(rtxn, sub)) return false;
+    }
+    WriteTxn wtxn(env_);
+    const unsigned int dbi = open_for_write(wtxn, sub, MDB_DUPSORT);
+    MDB_val k = to_val(parentId);
+    MDB_val v = to_val(childId);
+    // Passing data narrows the delete to exactly this (key, data) pair -
+    // siblings under the same parent key are untouched.
+    const int rc = mdb_del(wtxn.raw(), dbi, &k, &v);
+    if (rc != MDB_SUCCESS && rc != MDB_NOTFOUND) throw_mdb(rc, "relation index remove");
+    wtxn.commit();
+    cacheCommittedDbi(sub, dbi);
+    return rc == MDB_SUCCESS;
+}
+
+uint64_t LmdbDocumentStore::relation_index_child_count(std::string_view relation,
+                                                         std::string_view parentId) {
+    const std::string sub = relation_index_subdb(relation);
+    ReadTxn rtxn(env_);
+    auto dbi_opt = try_open_for_read(rtxn, sub);
+    if (!dbi_opt) return 0;
+    MDB_cursor* cur = nullptr;
+    mdb_check(mdb_cursor_open(rtxn.raw(), *dbi_opt, &cur), "cursor_open (relidx)");
+    struct CursorGuard {
+        MDB_cursor* c;
+        ~CursorGuard() { if (c) mdb_cursor_close(c); }
+    } guard{cur};
+    MDB_val k = to_val(parentId);
+    MDB_val v{0, nullptr};
+    if (mdb_cursor_get(cur, &k, &v, MDB_SET) != MDB_SUCCESS) return 0;
+    size_t n = 0;
+    mdb_check(mdb_cursor_count(cur, &n), "cursor_count (relidx)");
+    return static_cast<uint64_t>(n);
+}
+
+std::vector<std::string>
+LmdbDocumentStore::relation_index_children(std::string_view relation,
+                                            std::string_view parentId,
+                                            size_t limit) {
+    std::vector<std::string> out;
+    if (limit == 0) return out;
+    const std::string sub = relation_index_subdb(relation);
+    ReadTxn rtxn(env_);
+    auto dbi_opt = try_open_for_read(rtxn, sub);
+    if (!dbi_opt) return out;
+    MDB_cursor* cur = nullptr;
+    mdb_check(mdb_cursor_open(rtxn.raw(), *dbi_opt, &cur), "cursor_open (relidx children)");
+    struct CursorGuard {
+        MDB_cursor* c;
+        ~CursorGuard() { if (c) mdb_cursor_close(c); }
+    } guard{cur};
+    MDB_val k = to_val(parentId);
+    MDB_val v{0, nullptr};
+    // MDB_SET positions the cursor on the key but leaves which dup is current
+    // unspecified across LMDB versions; MDB_FIRST_DUP makes it explicit.
+    int rc = mdb_cursor_get(cur, &k, &v, MDB_SET);
+    if (rc != MDB_SUCCESS) {
+        if (rc == MDB_NOTFOUND) return out;
+        throw_mdb(rc, "cursor_get (relidx children)");
+    }
+    rc = mdb_cursor_get(cur, &k, &v, MDB_FIRST_DUP);
+    if (rc != MDB_SUCCESS) {
+        if (rc == MDB_NOTFOUND) return out;
+        throw_mdb(rc, "cursor first_dup (relidx children)");
+    }
+    while (out.size() < limit) {
+        const std::string_view child = to_sv(v);
+        if (!is_index_meta_key(child)) out.emplace_back(child);
+        rc = mdb_cursor_get(cur, &k, &v, MDB_NEXT_DUP);
+        if (rc == MDB_NOTFOUND) break;
+        if (rc != MDB_SUCCESS) throw_mdb(rc, "cursor next_dup (relidx children)");
+    }
+    return out;
+}
+
+// -------------------------------------------------------------------------
+// v2.11.0 T7 — the bootstrap-scan gap.
+//
+// T3 wired relation_index_add/remove into put()/del(), but that only
+// maintains the index going FORWARD from the moment set_relations() arms a
+// collection. Declaring a relation over a collection that already had rows
+// (the common case - a relation is usually declared once the schema is
+// known, not on day one of an empty collection) never indexed those rows:
+// the reverse index started empty and stayed empty for every pre-existing
+// child, so a parent delete against it silently found zero children and
+// went through as if nothing referenced it. Worse than no feature, because
+// declaring the relation looked like protection.
+//
+// build_relation_index closes that gap the same way build_index closes it
+// for secondary indexes: walk the child collection once, resolve
+// childField with yyjson only (never materialise a Document - the same
+// reasoning as maintainRelations), and post every reference found. Idempotent
+// via MDB_NODUPDATA, so calling this twice over an already-indexed relation
+// (e.g. CreateRelation re-run, or an operator re-running a migration) posts
+// no duplicates.
+// -------------------------------------------------------------------------
+
+uint64_t LmdbDocumentStore::build_relation_index(std::string_view relation,
+                                                  std::string_view childCollection,
+                                                  const std::string& childField) {
+    std::vector<std::pair<std::string, std::string>> rows;   // id -> payload
+    {
+        ReadTxn rtxn(env_);
+        auto dbi_opt = try_open_for_read(rtxn, childCollection);
+        if (!dbi_opt) return 0;
+        MDB_cursor* cur = nullptr;
+        mdb_check(mdb_cursor_open(rtxn.raw(), *dbi_opt, &cur), "cursor_open (relbuild)");
+        struct G { MDB_cursor* c; ~G() { if (c) mdb_cursor_close(c); } } g{cur};
+        MDB_val k{0, nullptr};
+        MDB_val v{0, nullptr};
+        int rc = mdb_cursor_get(cur, &k, &v, MDB_FIRST);
+        while (rc == MDB_SUCCESS) {
+            if (!is_identity_key(to_sv(k))) {
+                rows.emplace_back(std::string(to_sv(k)), std::string(to_sv(v)));
+            }
+            rc = mdb_cursor_get(cur, &k, &v, MDB_NEXT);
+        }
+    }
+
+    const std::string sub = relation_index_subdb(relation);
+    uint64_t indexed = 0;
+    {
+        WriteTxn wtxn(env_);
+        const unsigned int dbi = open_for_write(wtxn, sub, MDB_DUPSORT);
+        for (const auto& [id, payload] : rows) {
+            yyjson_doc* d = yyjson_read(payload.data(), payload.size(), 0);
+            if (!d) continue;
+            auto val = resolve_from_yyjson(yyjson_doc_get_root(d), childField);
+            yyjson_doc_free(d);
+            const auto parentIds = extract_relation_ids(val);
+            if (parentIds.empty()) continue;   // absent/null/unresolvable - not a reference
+            for (const auto& parentId : parentIds) {
+                MDB_val pk = to_val(parentId);
+                MDB_val cv = to_val(id);
+                const int rc = mdb_put(wtxn.raw(), dbi, &pk, &cv, MDB_NODUPDATA);
+                if (rc != MDB_SUCCESS && rc != MDB_KEYEXIST) {
+                    throw_mdb(rc, "relation index build put");
+                }
+            }
+            ++indexed;
+        }
+        wtxn.commit();
+        // MANDATORY, same as relation_index_add - the sub-db may be brand
+        // new (a relation declared on a collection with no prior postings
+        // still gets an empty sub-db created here), and until this runs it
+        // is invisible to try_open_for_read for the rest of the process.
+        cacheCommittedDbi(sub, dbi);
+    }
+    return indexed;
+}
+
+bool LmdbDocumentStore::relation_index_exists(std::string_view relation) {
+    const std::string sub = relation_index_subdb(relation);
+    ReadTxn rtxn(env_);
+    return try_open_for_read(rtxn, sub).has_value();
+}
+
+LmdbDocumentStore::DanglingCheckResult
+LmdbDocumentStore::check_relation_dangling(std::string_view relation,
+                                            std::string_view parentCollection,
+                                            size_t maxResults) {
+    DanglingCheckResult out;
+
+    const std::string sub = relation_index_subdb(relation);
+    ReadTxn rtxn(env_);
+    auto idx_dbi = try_open_for_read(rtxn, sub);
+    if (!idx_dbi) return out;   // never built - nothing to report (see relation_index_exists)
+
+    // The parent collection may legitimately not exist yet (nothing has been
+    // written to it) - every referenced parent id is then dangling by
+    // definition, which the mdb_get loop below already produces correctly
+    // when parent_dbi is nullopt (mdb_get is only called when it is set).
+    auto parent_dbi = try_open_for_read(rtxn, parentCollection);
+
+    MDB_cursor* cur = nullptr;
+    mdb_check(mdb_cursor_open(rtxn.raw(), *idx_dbi, &cur), "cursor_open (relation check)");
+    struct G { MDB_cursor* c; ~G() { if (c) mdb_cursor_close(c); } } g{cur};
+
+    MDB_val k{0, nullptr};
+    MDB_val v{0, nullptr};
+    int rc = mdb_cursor_get(cur, &k, &v, MDB_FIRST);
+    while (rc == MDB_SUCCESS) {
+        const auto key = to_sv(k);
+        // ⚠ Top-level walk: is_index_meta_key(), NOT is_identity_key(). This
+        // sub-db carries the v2.4.4 identity sentinel like any other, and a
+        // key here IS a real posting-list key (a parent id) - exactly the
+        // position where a real sentinel collision must be skipped, unlike
+        // the dead-code check inside one parent's dup set that Task 2 left
+        // as a Minor finding.
+        if (is_index_meta_key(key)) {
+            rc = mdb_cursor_get(cur, &k, &v, MDB_NEXT_NODUP);
+            continue;
+        }
+
+        bool parentExists = false;
+        if (parent_dbi) {
+            MDB_val pk = to_val(key);
+            MDB_val pv{0, nullptr};
+            const int grc = mdb_get(rtxn.raw(), *parent_dbi, &pk, &pv);
+            if (grc == MDB_SUCCESS) parentExists = true;
+            else if (grc != MDB_NOTFOUND) throw_mdb(grc, "get (relation check parent probe)");
+        }
+
+        if (!parentExists) {
+            ++out.total;
+            if (out.entries.size() < maxResults) {
+                DanglingReference ref;
+                ref.parentId = std::string(key);
+                // Cursor is still positioned on this key's first dup entry;
+                // cursor_count and a bounded walk of MDB_NEXT_DUP cost
+                // nothing extra beyond what a normal walk already touches.
+                size_t n = 0;
+                mdb_check(mdb_cursor_count(cur, &n), "cursor_count (relation check)");
+                ref.childCount = static_cast<uint64_t>(n);
+                MDB_val sk = k;
+                MDB_val sv{0, nullptr};
+                int src = mdb_cursor_get(cur, &sk, &sv, MDB_FIRST_DUP);
+                while (src == MDB_SUCCESS && ref.sampleChildIds.size() < 5) {
+                    const std::string_view child = to_sv(sv);
+                    if (!is_index_meta_key(child)) ref.sampleChildIds.emplace_back(child);
+                    src = mdb_cursor_get(cur, &sk, &sv, MDB_NEXT_DUP);
+                }
+                if (src != MDB_SUCCESS && src != MDB_NOTFOUND) {
+                    throw_mdb(src, "cursor next_dup (relation check sample)");
+                }
+                out.entries.push_back(std::move(ref));
+            }
+        }
+
+        rc = mdb_cursor_get(cur, &k, &v, MDB_NEXT_NODUP);
+    }
+    if (rc != MDB_SUCCESS && rc != MDB_NOTFOUND) throw_mdb(rc, "cursor next (relation check)");
+    return out;
+}
+
 std::optional<smartbotic::database::Document>
 LmdbDocumentStore::get(std::string_view collection, std::string_view id) {
     // An empty id is a zero-length LMDB key, which mdb_get rejects with
@@ -1021,43 +1581,77 @@ bool LmdbDocumentStore::del(std::string_view collection, std::string_view id) {
     // Same reasoning as get(): a zero-length key cannot exist, so there is
     // nothing to delete rather than an error to raise.
     if (id.empty()) return false;
-    // Open as a write txn unconditionally so we have MDB_CREATE available
-    // if the collection doesn't exist yet — but in that case there's
-    // nothing to delete; just probe with a read txn first to avoid
-    // accidentally creating an empty sub-db on a no-op delete.
-    {
-        ReadTxn rtxn(env_);
-        auto dbi_opt = try_open_for_read(rtxn, collection);
-        if (!dbi_opt) return false;
-    }
+    // Probe the cache first to avoid accidentally creating an empty sub-db on
+    // a no-op delete - no transaction needed for this, see cachedDbi().
+    if (!cachedDbi(collection)) return false;
 
     WriteTxn wtxn(env_);
+    std::vector<std::pair<std::string, unsigned int>> to_cache;
+    const bool existed = del(wtxn, collection, id, to_cache);
+    // Commit BEFORE any handle is cached - the v2.8.0 lesson.
+    wtxn.commit();
+    for (const auto& [sub, d] : to_cache) cacheCommittedDbi(sub, d);
+    return existed;
+}
+
+bool LmdbDocumentStore::del(WriteTxn& wtxn,
+                             std::string_view collection,
+                             std::string_view id,
+                             std::vector<std::pair<std::string, unsigned int>>& to_cache) {
+    if (id.empty()) return false;
+    // Same no-op-avoidance as the no-txn form above: don't spring an empty
+    // sub-db into existence for a collection that has never been written.
+    // cachedDbi() alone only sees ALREADY-COMMITTED collections, though - a
+    // collection this same caller-owned transaction created earlier (via an
+    // uncommitted put(wtxn, ...) call sharing this to_cache) would not be in
+    // the process-wide cache yet, and treating it as "doesn't exist" would
+    // silently drop the delete instead of routing it through open_for_write.
+    // By construction, anything opened earlier in THIS transaction is
+    // already in to_cache (every txn-accepting overload records its handle
+    // there), so checking both closes that gap rather than merely
+    // documenting it.
+    const bool known_this_txn = std::any_of(
+        to_cache.begin(), to_cache.end(),
+        [&](const auto& p) { return p.first == collection; });
+    if (!cachedDbi(collection) && !known_this_txn) return false;
+
     unsigned int dbi = open_for_write(wtxn, collection);
+    // Recorded immediately, not after mdb_del - see put(WriteTxn&, ...)'s
+    // comment. mdb_get (below) and maintainIndexes/maintainRelations can all
+    // throw (including UniqueViolation), and a throw between here and the
+    // old late emplace_back left the collection's own handle out of
+    // to_cache while any index/relation handles opened along the way were
+    // in it. Harmless for the no-txn wrapper above (its WriteTxn just aborts
+    // unwritten either way) but exactly Finding 2's shape one function over,
+    // and it becomes live the moment a cascade shares to_cache across calls,
+    // catches per-op exceptions, and commits the rest of the transaction.
+    to_cache.emplace_back(std::string(collection), dbi);
     MDB_val k = to_val(id);
 
     // Remove index entries before the row goes, while its stored bytes are
     // still readable - they are the only record of which index keys it owns.
-    std::vector<std::pair<std::string, unsigned int>> index_dbis;
-    if (!indexed_fields(collection).empty()) {
+    const bool needs_index = !indexed_fields(collection).empty();
+    const bool needs_relations = !relations(collection).empty();
+    if (needs_index || needs_relations) {
         MDB_val old{0, nullptr};
         const int grc = mdb_get(wtxn.raw(), dbi, &k, &old);
         if (grc == MDB_SUCCESS) {
-            maintainIndexes(wtxn, collection, id, to_sv(old), nullptr, index_dbis);
+            if (needs_index) {
+                maintainIndexes(wtxn, collection, id, to_sv(old), nullptr, to_cache);
+            }
+            if (needs_relations) {
+                maintainRelations(wtxn, collection, id, to_sv(old), nullptr, to_cache);
+            }
         } else if (grc != MDB_NOTFOUND) {
             throw_mdb(grc, "get (pre-index del)");
         }
     }
 
     int rc = mdb_del(wtxn.raw(), dbi, &k, nullptr);
-    if (rc == MDB_NOTFOUND) {
-        wtxn.commit();
-        cacheCommittedDbi(collection, dbi);
-        return false;
-    }
+    // NOT cached here - the caller caches every entry in to_cache (recorded
+    // above) only after ITS commit succeeds.
+    if (rc == MDB_NOTFOUND) return false;
     if (rc != MDB_SUCCESS) throw_mdb(rc, "del");
-    wtxn.commit();
-    cacheCommittedDbi(collection, dbi);
-    for (const auto& [sub, d] : index_dbis) cacheCommittedDbi(sub, d);
     return true;
 }
 
@@ -1122,6 +1716,12 @@ void LmdbDocumentStore::set_unique_fields(std::string_view collection,
     } else {
         unique_fields_[std::string(collection)] = std::move(fields);
     }
+    // finding 5 — recomputed under the same mutex that owns the map, so the
+    // counter can never disagree with it. Recomputed rather than incremented:
+    // this setter is idempotent-by-replacement, so a blind ++ would double
+    // count a re-declaration.
+    rejecting_unique_collections_.store(unique_fields_.size(),
+                                       std::memory_order_relaxed);
 }
 
 std::vector<std::string>
@@ -1971,34 +2571,71 @@ void LmdbDocumentStore::put_vector(std::string_view collection,
                                     std::string_view id,
                                     const std::vector<float>& vec) {
     if (vec.empty()) return;  // mirror the migration tool's no-op semantics
-    const std::string subdb = vector_subdb_name(collection);
     WriteTxn wtxn(env_);
+    std::vector<std::pair<std::string, unsigned int>> to_cache;
+    put_vector(wtxn, collection, id, vec, to_cache);
+    wtxn.commit();
+    for (const auto& [sub, d] : to_cache) cacheCommittedDbi(sub, d);
+}
+
+void LmdbDocumentStore::put_vector(WriteTxn& wtxn,
+                                    std::string_view collection,
+                                    std::string_view id,
+                                    const std::vector<float>& vec,
+                                    std::vector<std::pair<std::string, unsigned int>>& to_cache) {
+    if (vec.empty()) return;  // mirror the migration tool's no-op semantics
+    const std::string subdb = vector_subdb_name(collection);
     unsigned int dbi = open_for_write(wtxn, subdb);
+    // Recorded immediately, not after mdb_put succeeds - see put(WriteTxn&,
+    // ...)'s comment. A throw from mdb_put on a brand-new vector sub-db must
+    // not leave its handle out of to_cache.
+    to_cache.emplace_back(subdb, dbi);
     MDB_val k = to_val(id);
     MDB_val v{vec.size() * sizeof(float),
               const_cast<void*>(static_cast<const void*>(vec.data()))};
     mdb_check(mdb_put(wtxn.raw(), dbi, &k, &v, 0), "put_vector");
-    wtxn.commit();
-    cacheCommittedDbi(subdb, dbi);
 }
 
 bool LmdbDocumentStore::del_vector(std::string_view collection, std::string_view id) {
     const std::string subdb = vector_subdb_name(collection);
     // Probe first so we don't create an empty vectors sub-db just to
-    // discover the vector isn't there.
-    {
-        ReadTxn rtxn(env_);
-        auto dbi_opt = try_open_for_read(rtxn, subdb);
-        if (!dbi_opt) return false;
-    }
+    // discover the vector isn't there. Cache-only, see cachedDbi().
+    if (!cachedDbi(subdb)) return false;
     WriteTxn wtxn(env_);
+    std::vector<std::pair<std::string, unsigned int>> to_cache;
+    const bool existed = del_vector(wtxn, collection, id, to_cache);
+    wtxn.commit();
+    for (const auto& [sub, d] : to_cache) cacheCommittedDbi(sub, d);
+    return existed;
+}
+
+bool LmdbDocumentStore::del_vector(WriteTxn& wtxn,
+                                    std::string_view collection,
+                                    std::string_view id,
+                                    std::vector<std::pair<std::string, unsigned int>>& to_cache) {
+    const std::string subdb = vector_subdb_name(collection);
+    // Same same-transaction gap as del(WriteTxn&, ...): cachedDbi() only
+    // sees already-committed sub-dbs, so a vector sub-db opened by an
+    // earlier put_vector(wtxn, ...) call sharing this to_cache would not be
+    // visible there yet. Check to_cache too - by construction it already
+    // holds any handle opened earlier in this transaction (see the
+    // emplace_back in put_vector(WriteTxn&, ...) above). Compared against
+    // the VECTOR sub-db's own name, not the collection's, so a collection or
+    // index handle recorded earlier in the same to_cache cannot be mistaken
+    // for it.
+    const bool known_this_txn = std::any_of(
+        to_cache.begin(), to_cache.end(),
+        [&](const auto& p) { return p.first == subdb; });
+    if (!cachedDbi(subdb) && !known_this_txn) return false;
     unsigned int dbi = open_for_write(wtxn, subdb);
+    // Recorded immediately, matching put_vector(WriteTxn&, ...) and
+    // del(WriteTxn&, ...) - a handle from a transaction that later aborts is
+    // harmless, since the caller only applies to_cache after its own commit.
+    to_cache.emplace_back(subdb, dbi);
     MDB_val k = to_val(id);
     int rc = mdb_del(wtxn.raw(), dbi, &k, nullptr);
     if (rc == MDB_NOTFOUND) return false;
     if (rc != MDB_SUCCESS) throw_mdb(rc, "del_vector");
-    wtxn.commit();
-    cacheCommittedDbi(subdb, dbi);
     return true;
 }
 

+ 304 - 16
service/src/storage/document_store_lmdb.hpp

@@ -46,6 +46,67 @@ public:
     std::string existing_id;
 };
 
+// v2.11.0 T13 — thrown when a write to a child collection introduces a
+// reference (via a relation declared with validateOnWrite=true) to a
+// parent id that does not exist. Checked with an mdb_get on the parent's own
+// sub-db INSIDE the child's write transaction (see maintainRelations) - not
+// before it, not after - which is what closes the restrict race: LMDB
+// serialises writers, so this transaction cannot begin until any transaction
+// that deleted the parent has already committed, and there is no interval
+// between the check and this write's own commit for the parent to vanish in.
+//
+// A distinct type, same reasoning as UniqueViolation: the gRPC layer answers
+// FAILED_PRECONDITION instead of INTERNAL, and applyDualWriteMirror /
+// mirrorDocOrUndo must rethrow it untouched rather than treat it as a mirror
+// fault - see UniqueViolation's comment on set_unique_fields for why that
+// distinction matters.
+class MissingParentReference : public std::runtime_error {
+public:
+    MissingParentReference(std::string relation, std::string childField,
+                            std::string parentId)
+        : std::runtime_error("relation '" + relation + "': field '" + childField +
+                             "' references parent id '" + parentId +
+                             "' which does not exist"),
+          relation(std::move(relation)),
+          childField(std::move(childField)),
+          parentId(std::move(parentId)) {}
+    std::string relation;
+    std::string childField;
+    std::string parentId;
+};
+
+// v2.11.0 T3 — a child-side relation reference, as the storage layer needs
+// it to maintain the reverse index on a write. Deliberately NOT
+// RelationInfo (relations/relation_manager.hpp): this layer stays free of
+// project qualification, OnDelete policy, etc. - see relation_index.hpp's
+// file comment. `name` is the BARE relation name (the caller strips the
+// project); `childField` is the dot-path on the CHILD document (the
+// collection this ref is declared against) that holds the parent id, or an
+// array of them.
+//
+// v2.11.0 T13 — `parent` is the BARE parent collection name (same project as
+// the child - see relation_manager.hpp's cross-project refusal), and
+// `validateOnWrite` mirrors RelationInfo::validateOnWrite. Both stay
+// optional-by-default (empty / false) so every pre-T13 aggregate-init call
+// site (`RelationRef{name, childField}`) keeps compiling unchanged.
+// v2.11.0 T13 round 2 — `relationsEnforced` mirrors CollectionCfg::
+// relationsEnforced (config/collection_config_manager.hpp), snapshotted at
+// arming time (armRelationsForChild / applyRelationDeclarations) rather than
+// read live: the storage layer stays free of CollectionConfigManager the
+// same way it stays free of RelationManager (see the file comment above) -
+// that dependency is keyed by the QUALIFIED collection name, which this
+// per-project layer deliberately does not carry. Staying current after a
+// configureCollection flip is the caller's job: ConfigureCollection
+// re-arms the collection's relations after changing the flag, exactly like
+// any other change to a relation's declared shape.
+struct RelationRef {
+    std::string name;
+    std::string childField;
+    std::string parent;
+    bool validateOnWrite = false;
+    bool relationsEnforced = true;
+};
+
 class LmdbDocumentStore : public DocumentStore {
 public:
     explicit LmdbDocumentStore(LmdbEnv& env);
@@ -76,24 +137,33 @@ public:
                             std::vector<std::string> fields);
     std::vector<std::string> indexed_fields(std::string_view collection);
 
-    // ⚠ v2.10.0 — UNREACHABLE BY DESIGN, pending write-path work. Nothing calls
-    // set_unique_fields, so the enforcement below never runs, and no RPC or config
-    // key exposes it. Kept because the mechanism is correct in isolation and is
-    // the base for finishing the feature.
+    // v2.11.0 T11 — ACTIVE. Called from three places: DatabaseGrpcImpl::CreateIndex
+    // and ::DropIndex (per-call, gated on a duplicate-value refusal — see
+    // find_duplicate_values below), and DatabaseService::applyIndexDeclarations
+    // at boot (re-arming from CollectionCfg::uniqueFields, persisted in
+    // `_collection_meta`). Exposed on the wire via CreateIndexRequest.unique and
+    // reported by ListIndexes.
     //
-    // Why it cannot be turned on yet: the check throws from put(), which runs
-    // inside applyDualWriteMirror - and that catches every exception, logs it,
-    // bumps mirror drift and flips mirror_healthy_. So a rejection would be
-    // swallowed (the row still lands in MemoryStore, unenforced) AND would send
-    // every read in the process to MemoryStore, which is the v2.8.1 fault. Worse,
-    // MemoryStore mutates BEFORE the mirror runs, so a clean rejection needs the
-    // in-memory write rolled back.
+    // Why this took its own task to turn on: the check throws UniqueViolation
+    // from put(), which runs inside applyDualWriteMirror - and that used to catch
+    // every exception, log it, bump mirror drift and flip mirror_healthy_. A
+    // rejection would have been swallowed (the row still lands in MemoryStore,
+    // unenforced) AND would have sent every read in the process to MemoryStore,
+    // the v2.8.1 fault. Worse, MemoryStore mutates BEFORE the mirror runs, so a
+    // clean rejection needed the in-memory write rolled back.
     //
-    // Finishing it means either propagating UniqueViolation through the mirror
-    // without touching health, plus rollback, or moving enforcement ahead of the
-    // MemoryStore mutation under the same collection lock. Both touch the write
-    // path that produced the v2.4.3, v2.4.4 and v2.8.0 incidents, so it wants its
-    // own change. See docs/ROADMAP.md.
+    // The fix (T11): applyDualWriteMirror catches UniqueViolation separately and
+    // rethrows it untouched - health/drift stay exactly as they were - and every
+    // MemoryStore write path that can raise it (insert/update/upsert/
+    // updateIfVersion/patchDocument/setAdd/restoreToVersion) undoes its own
+    // mutation via mirrorDocOrUndo() before letting it propagate. The gRPC
+    // handlers map it to ALREADY_EXISTS. See dual_write_mirror.hpp and
+    // memory_store.cpp's mirrorDocOrUndo for the mechanics, and
+    // database_service.cpp's replicated-entry apply path for the one caller of
+    // applyDualWriteMirror that deliberately keeps the old swallow-and-flip
+    // behaviour (loadDocument there already mutated MemoryStore before the
+    // mirror runs, and there is no "reject the whole operation" available on a
+    // replication-apply path).
     void set_unique_fields(std::string_view collection,
                           std::vector<std::string> fields);
     std::vector<std::string> unique_fields(std::string_view collection);
@@ -182,6 +252,106 @@ public:
     // Remove an index entirely.
     bool drop_index(std::string_view collection, const std::string& field);
 
+    // v2.11.0 T2 — relation reverse index: parent id -> child ids, DUPSORT.
+    // `relation` is the BARE relation name (see relations/relation_index.hpp);
+    // callers strip the project qualifier before reaching here. No document-
+    // write hooks call these yet (Task 3); no enforcement reads them yet
+    // (Task 4). This is storage only.
+
+    // Record that `childId` references `parentId` through `relation`.
+    // Idempotent: re-adding an existing pair is a no-op (MDB_NODUPDATA), not a
+    // duplicate posting.
+    bool relation_index_add(std::string_view relation, std::string_view parentId,
+                            std::string_view childId);
+
+    // Remove exactly the (parentId, childId) pair. Other children of the same
+    // parent are untouched. Returns false if the pair was not present.
+    bool relation_index_remove(std::string_view relation, std::string_view parentId,
+                               std::string_view childId);
+
+    // Number of children `parentId` has under `relation`, without reading them
+    // (mdb_cursor_count) — the O(1)-ish count restrict/DescribeDelete need. 0
+    // for an unreferenced parent or a relation with no index yet; neither is
+    // an error.
+    uint64_t relation_index_child_count(std::string_view relation,
+                                        std::string_view parentId);
+
+    // Up to `limit` child ids of `parentId` under `relation`. Empty if none.
+    std::vector<std::string> relation_index_children(std::string_view relation,
+                                                      std::string_view parentId,
+                                                      size_t limit);
+
+    // v2.11.0 T7 — populate a relation's reverse index over rows already
+    // present in `childCollection`, mirroring build_index(). Idempotent via
+    // MDB_NODUPDATA: re-running over an already-indexed relation posts no
+    // duplicates. Returns rows indexed (rows that had a resolvable, non-null
+    // reference under `childField`; absent/null contribute nothing, same as
+    // the write-path maintainRelations()).
+    //
+    // ⚠ Cost is proportional to `childCollection`'s size - a full collection
+    // walk, same as build_index. On a large collection this call blocks for
+    // the duration; it belongs in a migration/maintenance window, not a
+    // request path a caller is waiting on synchronously in a tight loop.
+    uint64_t build_relation_index(std::string_view relation,
+                                  std::string_view childCollection,
+                                  const std::string& childField);
+
+    // True if this relation's reverse index sub-db has been created at least
+    // once (via relation_index_add or build_relation_index) - independent of
+    // whether it currently holds any postings, since build_relation_index
+    // creates the (possibly empty) sub-db even when nothing matched. Used to
+    // tell "declared but never populated" (missing sub-db - the T7 bug) apart
+    // from "populated, currently empty" (sub-db exists with zero postings).
+    bool relation_index_exists(std::string_view relation);
+
+    // v2.11.0 T7 — dangling references: child rows pointing at a parent id
+    // that does not exist in `parentCollection`. Read-only; mutates nothing.
+    //
+    // Walks the TOP LEVEL of the reverse index (every distinct parent id),
+    // skipping reserved meta keys via is_index_meta_key() - not
+    // is_identity_key(), since a top-level walk is exactly where a real
+    // sentinel collision is possible (see the T2 review note this task
+    // closes). For each parent id absent from `parentCollection`, records
+    // its child count (mdb_cursor_count - free, already positioned) and up
+    // to 5 sample child ids.
+    //
+    // `maxResults` caps the ENTRIES returned, not the work done: every
+    // parent id in the index is still probed, so `total` is always exact
+    // even when `entries` is capped. Cost is proportional to the number of
+    // DISTINCT parents referenced, not to `childCollection`'s row count -
+    // still a full index walk, so treat this as a migration/operator tool on
+    // a large relation, not a hot-path call.
+    struct DanglingReference {
+        std::string parentId;
+        uint64_t childCount = 0;
+        std::vector<std::string> sampleChildIds;   // at most 5
+    };
+    struct DanglingCheckResult {
+        uint64_t total = 0;
+        std::vector<DanglingReference> entries;    // capped at maxResults
+    };
+    DanglingCheckResult check_relation_dangling(std::string_view relation,
+                                                std::string_view parentCollection,
+                                                size_t maxResults);
+
+    // v2.11.0 T3 — declare which relations have `collection` as their CHILD
+    // side, so put()/del() know to maintain the reverse index. Empty removes
+    // the declaration. Mirrors set_indexed_fields: maintenance runs inside
+    // the same write transaction as the document, and a collection with no
+    // relations declared pays nothing (the write path checks this map and
+    // returns immediately).
+    void set_relations(std::string_view collection, std::vector<RelationRef> rels);
+    std::vector<RelationRef> relations(std::string_view collection);
+
+    // v2.11.0 final review (finding 5) — see DocumentStore::can_reject_writes.
+    //
+    // True iff `collection` has at least one declared unique field, or at
+    // least one declared relation with validateOnWrite (the only two things
+    // put() rejects a write for). The common case - nothing declared anywhere
+    // in the process - is answered by ONE relaxed atomic load with no mutex
+    // taken at all, which is what makes it usable on the hot write path.
+    bool can_reject_writes(std::string_view collection) override;
+
     // Outcome of the constructor's priming pass, for the owner to report.
     // prime_error() is empty on success.
     size_t primed_count() const noexcept { return primed_count_; }
@@ -196,11 +366,49 @@ public:
              std::string_view id,
              const smartbotic::database::Document& doc) override;
 
+    // v2.11.0 T12 round-4 — see DocumentStore::put_batch. One WriteTxn for
+    // the whole vector, one commit, then (and only then) the accumulated
+    // handles are cached: the same commit-before-cache discipline as every
+    // other wrapper in this file. Throws without having written anything if
+    // any item fails.
+    void put_batch(const std::vector<BatchPutItem>& items) override;
+
     std::optional<smartbotic::database::Document>
     get(std::string_view collection, std::string_view id) override;
 
     bool del(std::string_view collection, std::string_view id) override;
 
+    // v2.11.0 T10 — txn-accepting overloads. Task 12's atomic cascade needs
+    // ONE transaction spanning the parent document, each affected child, and
+    // every index/relation sub-db a write touches - the no-txn forms above
+    // each open and commit their own WriteTxn, so they cannot compose into
+    // one atomic unit. Internal helpers already take a WriteTxn& (
+    // open_for_write, maintainIndexes, maintainRelations); this extends that
+    // pattern to the public surface rather than inventing a new one.
+    //
+    // ⚠ These do NOT commit and do NOT cache - both stay the caller's job.
+    // Caching before commit is the v2.8.0 bug: a later abort of the SAME
+    // caller-owned transaction would leave a closed handle in the cache,
+    // poisoning the collection with EINVAL for the life of the process. Every
+    // MDB_dbi this call opens (the collection's own handle, plus any
+    // index/relation sub-db maintenance touched) is appended to `to_cache`;
+    // call cacheCommittedDbi() for each entry ONLY after wtxn.commit()
+    // succeeds. Never skip that step either - see try_open_for_read: reads
+    // are cache-only, so a committed-but-uncached sub-db reads as absent for
+    // the rest of the process's life (this is how relations would silently
+    // stop enforcing). The no-txn forms above are thin wrappers that do
+    // exactly this: open, call, commit, then cache every returned handle.
+    void put(class WriteTxn& wtxn,
+             std::string_view collection,
+             std::string_view id,
+             const smartbotic::database::Document& doc,
+             std::vector<std::pair<std::string, unsigned int>>& to_cache);
+
+    bool del(class WriteTxn& wtxn,
+             std::string_view collection,
+             std::string_view id,
+             std::vector<std::pair<std::string, unsigned int>>& to_cache);
+
     uint64_t count(std::string_view collection) override;
 
     ScanResult scan(std::string_view collection,
@@ -215,6 +423,19 @@ public:
                     std::string_view id,
                     const std::vector<float>& vec) override;
     bool del_vector(std::string_view collection, std::string_view id) override;
+
+    // v2.11.0 T10 — txn-accepting vector equivalents. Same contract as
+    // put(WriteTxn&, ...)/del(WriteTxn&, ...) above: no commit, no cache,
+    // handles opened are appended to `to_cache` for the caller.
+    void put_vector(class WriteTxn& wtxn,
+                    std::string_view collection,
+                    std::string_view id,
+                    const std::vector<float>& vec,
+                    std::vector<std::pair<std::string, unsigned int>>& to_cache);
+    bool del_vector(class WriteTxn& wtxn,
+                    std::string_view collection,
+                    std::string_view id,
+                    std::vector<std::pair<std::string, unsigned int>>& to_cache);
     std::optional<std::vector<float>>
     get_vector(std::string_view collection, std::string_view id) override;
     void scan_vectors(
@@ -223,6 +444,34 @@ public:
                            const float* data,
                            size_t count)> callback) override;
 
+    // v2.11.0 T12 — open a caller-owned WriteTxn spanning several
+    // txn-accepting put()/del()/put_vector()/del_vector() calls across
+    // MULTIPLE collections (parent + every cascade-affected child + the
+    // relation/index sub-dbs those touch) — exactly what an atomic cascade
+    // needs and what the txn-accepting overloads above were built for (see
+    // their header comment). Never opens a second MDB_env: this reuses
+    // env_, the same instance every internal caller already shares.
+    //
+    // The caller owns wtxn's lifetime: call every put()/del() it needs
+    // against ONE shared `to_cache`, then commitAndCache() exactly once.
+    // Letting wtxn go out of scope uncommitted (or calling .abort())
+    // aborts everything done through it — nothing partial can survive.
+    class WriteTxn beginWrite();
+
+    // v2.11.0 T12 — commit `wtxn`, then cache every handle accumulated in
+    // `to_cache`, in that order. Companion to beginWrite(): the same
+    // commit-then-cache discipline every no-txn wrapper in this file
+    // already applies internally (put(), del(), put_vector(), del_vector()),
+    // exposed for a caller that opened its own WriteTxn via beginWrite() and
+    // shared one to_cache across several calls. Never cache before commit -
+    // a handle whose opening transaction later aborts is closed by LMDB, and
+    // caching it anyway is the v2.8.0 bug (EINVAL for the life of the
+    // process). This is the ONLY sanctioned way for outside code to reach
+    // the private per-handle cache — deliberately, so "commit before cache"
+    // cannot be gotten wrong by a caller that forgets the ordering.
+    void commitAndCache(class WriteTxn& wtxn,
+                        const std::vector<std::pair<std::string, unsigned int>>& to_cache);
+
 private:
     // Sub-db handle cache. MDB_dbi is typedef'd to unsigned int — held as
     // that bare type to avoid pulling <lmdb.h> into this header.
@@ -239,6 +488,21 @@ private:
     std::mutex index_mutex_;
     std::unordered_map<std::string, std::vector<std::string>> indexed_fields_;
     std::unordered_map<std::string, std::vector<std::string>> unique_fields_;
+
+    // v2.11.0 T3 — collection (as CHILD) -> declared relation refs. Its own
+    // mutex for the same reason indexed_fields_ has one: the write path
+    // consults it on every put/del and must not contend with the dbi cache.
+    std::mutex relations_mutex_;
+    std::unordered_map<std::string, std::vector<RelationRef>> relations_;
+
+    // finding 5 — how many collections currently have a rejecting constraint
+    // declared (a unique field, or a relation with validateOnWrite), across
+    // BOTH maps. Maintained by set_unique_fields()/set_relations() so
+    // can_reject_writes() can answer "no, and nothing anywhere does" without
+    // taking either mutex. Never negative: each setter recomputes its own
+    // contribution rather than incrementing blindly.
+    std::atomic<size_t> rejecting_unique_collections_{0};
+    std::atomic<size_t> rejecting_relation_collections_{0};
     mutable std::atomic<uint64_t> indexed_scans_{0};
     mutable std::atomic<uint64_t> full_scans_{0};
     mutable std::atomic<uint64_t> declined_unselective_{0};
@@ -388,6 +652,20 @@ private:
                          const smartbotic::database::Document* new_doc,
                          std::vector<std::pair<std::string, unsigned int>>& to_cache);
 
+    // v2.11.0 T3 — bring every relation declared with `collection` as CHILD
+    // into line with a write, inside `wtxn`. Same shape as maintainIndexes:
+    // old references come from the stored bytes (yyjson only), new
+    // references from filter_eval::resolveFilterValue() on `new_doc` (null
+    // on delete), diffed with std::set_difference so an unchanged reference
+    // costs no write. Handles opened here are appended to `to_cache` for the
+    // caller to cache AFTER its commit.
+    void maintainRelations(class WriteTxn& wtxn,
+                           std::string_view collection,
+                           std::string_view id,
+                           std::string_view old_payload,
+                           const smartbotic::database::Document* new_doc,
+                           std::vector<std::pair<std::string, unsigned int>>& to_cache);
+
     // v2.8.0 — record a handle in the cache, to be called ONLY after the
     // transaction that opened it has committed. LMDB closes a handle whose
     // opening transaction aborts, so caching any earlier leaves a closed handle
@@ -396,6 +674,16 @@ private:
     void cacheCommittedDbi(std::string_view collection, unsigned int dbi);
     std::optional<unsigned int> try_open_for_read(class ReadTxn& rtxn,
                                                    std::string_view collection);
+
+    // v2.11.0 T10 — cache-only existence probe, no transaction required.
+    // Mirrors try_open_for_read's cache-hit path exactly (which never
+    // touches its ReadTxn& argument either — the cache is complete by
+    // construction once primed, see that function's long comment) and its
+    // cache-miss path (nullopt). Lets a txn-accepting overload ask "does
+    // this collection already exist" without opening a second transaction
+    // inside the caller's already-open WriteTxn — nesting a ReadTxn there
+    // would be needless and this answers exactly the same question.
+    std::optional<unsigned int> cachedDbi(std::string_view collection);
 };
 
 }  // namespace smartbotic::db::storage

+ 22 - 0
service/src/storage/dual_write_mirror.hpp

@@ -27,6 +27,7 @@
 #include "document.hpp"
 #include "memory_store.hpp"
 #include "storage/document_store.hpp"
+#include "storage/document_store_lmdb.hpp"
 
 namespace smartbotic::db::storage {
 
@@ -54,6 +55,27 @@ inline void applyDualWriteMirror(
             default:
                 break;
         }
+    } catch (const UniqueViolation&) {
+        // v2.11.0 T11 — a rejected write is the CALLER's business, not a
+        // storage fault. Genuine LMDB faults still fall into the catch
+        // below and degrade the mirror; a UniqueViolation must not, on
+        // either count:
+        //   - bumping mirror_drift_count would report a data-integrity
+        //     problem that doesn't exist - the write correctly did not land.
+        //   - flipping mirror_healthy_ would send every read in the
+        //     process to MemoryStore for the rest of its life over a
+        //     single duplicate-key rejection. That's the v2.8.1 fault.
+        // Rethrown as-is so the caller (MemoryStore::insert/update/etc.)
+        // can undo its own in-memory mutation, and so it eventually reaches
+        // the gRPC handler as ALREADY_EXISTS rather than INTERNAL.
+        throw;
+    } catch (const MissingParentReference&) {
+        // v2.11.0 T13 — same reasoning as the UniqueViolation catch above,
+        // for the same reason: a rejected write because its parent does not
+        // exist is the caller's business, not a mirror fault. Rethrown as-is
+        // so MemoryStore undoes its own mutation and the gRPC handler
+        // answers FAILED_PRECONDITION rather than INTERNAL.
+        throw;
     } catch (const std::exception& e) {
         spdlog::error("v2.0 mirror failed coll={} id={} op={}: {}",
                       collection, id, static_cast<int>(eventType), e.what());

+ 147 - 0
tests/CMakeLists.txt

@@ -353,6 +353,7 @@ add_executable(test_document_store
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/lmdb_txn.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/lmdb_dbi.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/document_store_lmdb.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/relations/relation_index.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/secondary_index.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/subdb_identity.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/json_parse.cpp
@@ -392,6 +393,7 @@ add_executable(test_migrate_v1_to_v2
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/lmdb_txn.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/lmdb_dbi.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/document_store_lmdb.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/relations/relation_index.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/secondary_index.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/subdb_identity.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/migrate_v1_to_v2.cpp
@@ -448,6 +450,7 @@ add_executable(test_dual_write_mirror
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/lmdb_txn.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/lmdb_dbi.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/document_store_lmdb.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/relations/relation_index.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/secondary_index.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/subdb_identity.cpp
 )
@@ -533,6 +536,7 @@ add_executable(test_subdb_identity
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/lmdb_dbi.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/subdb_identity.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/document_store_lmdb.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/relations/relation_index.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/secondary_index.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/json_parse.cpp
     ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/doc_binary.cpp
@@ -555,6 +559,109 @@ endif()
 
 add_test(NAME test_subdb_identity COMMAND test_subdb_identity)
 
+# v2.11.0 T2 — relation reverse index (parent id -> child ids, DUPSORT).
+# Only the sources relation_index_{add,remove,child_count,children} actually
+# need: the LMDB store itself plus its direct dependencies. No relations/
+# relation_manager.cpp here - the storage layer is deliberately free of that
+# dependency (see relations/relation_index.hpp).
+add_executable(test_relation_index
+    test_relation_index.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/lmdb_env.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/lmdb_txn.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/lmdb_dbi.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/subdb_identity.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/document_store_lmdb.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/secondary_index.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/relations/relation_index.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/json_parse.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/doc_binary.cpp
+)
+
+target_include_directories(test_relation_index PRIVATE
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src
+    ${LMDB_INCLUDE_DIR}
+    ${yyjson_INCLUDE_DIRS}
+)
+
+target_link_libraries(test_relation_index PRIVATE ${LMDB_LIBRARY})
+target_link_libraries(test_relation_index PRIVATE ${yyjson_LIBRARIES})
+
+if(TARGET nlohmann_json::nlohmann_json)
+    target_link_libraries(test_relation_index PRIVATE nlohmann_json::nlohmann_json)
+else()
+    target_include_directories(test_relation_index PRIVATE ${NLOHMANN_JSON_INCLUDE_DIRS})
+endif()
+
+add_test(NAME test_relation_index COMMAND test_relation_index)
+
+# v2.11.0 T3/T4 — relation reverse-index maintenance on child writes (T3),
+# plus restrict/no_action delete enforcement (T4). T4 needs the declaration
+# registry (RelationManager, backed by a MemoryStore) alongside the LMDB
+# reverse index, so unlike test_relation_index this target does pull in
+# relations/relation_manager.cpp and memory_store.cpp and their dependencies.
+add_executable(test_relation_enforcement
+    test_relation_enforcement.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/lmdb_env.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/lmdb_txn.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/lmdb_dbi.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/subdb_identity.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/document_store_lmdb.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/storage/secondary_index.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/relations/relation_index.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/relations/relation_manager.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/relations/relation_enforcement.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/relations/relation_cascade.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/memory_store.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/views/view_manager.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/views/projection.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/config/collection_config_manager.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/persistence/history_store.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/persistence/wal.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/persistence/snapshot.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/persistence/persistence_manager.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/json_parse.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/doc_binary.cpp
+)
+
+target_include_directories(test_relation_enforcement PRIVATE
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src
+    ${LMDB_INCLUDE_DIR}
+    ${yyjson_INCLUDE_DIRS}
+)
+
+target_link_libraries(test_relation_enforcement PRIVATE ${LMDB_LIBRARY})
+target_link_libraries(test_relation_enforcement PRIVATE ${yyjson_LIBRARIES})
+
+if(TARGET nlohmann_json::nlohmann_json)
+    target_link_libraries(test_relation_enforcement PRIVATE nlohmann_json::nlohmann_json)
+else()
+    target_include_directories(test_relation_enforcement PRIVATE ${NLOHMANN_JSON_INCLUDE_DIRS})
+endif()
+
+if(TARGET spdlog::spdlog)
+    target_link_libraries(test_relation_enforcement PRIVATE spdlog::spdlog)
+else()
+    target_link_libraries(test_relation_enforcement PRIVATE ${SPDLOG_LIBRARIES})
+    target_include_directories(test_relation_enforcement PRIVATE ${SPDLOG_INCLUDE_DIRS})
+endif()
+find_package(Threads REQUIRED)
+target_link_libraries(test_relation_enforcement PRIVATE Threads::Threads)
+
+# v2.11.0 T12 — persistence_manager.cpp now linked in (WAL-first cascade
+# tests need a real PersistenceManager); it pulls in snapshot.cpp, which
+# uses LZ4 compression, same as test_migrate_v1_to_v2 above.
+if(PKG_CONFIG_FOUND)
+    pkg_check_modules(RELENF_LZ4 QUIET liblz4)
+endif()
+if(RELENF_LZ4_FOUND)
+    target_link_libraries(test_relation_enforcement PRIVATE ${RELENF_LZ4_LIBRARIES})
+    target_include_directories(test_relation_enforcement PRIVATE ${RELENF_LZ4_INCLUDE_DIRS})
+else()
+    target_link_libraries(test_relation_enforcement PRIVATE lz4)
+endif()
+
+add_test(NAME test_relation_enforcement COMMAND test_relation_enforcement)
+
 # v2.9.0 — secondary index key encoding. Pins the one property that matters:
 # an index key comparison must mean the same thing as the scan's comparison,
 # because two paths answering one question that disagree return wrong data
@@ -718,3 +825,43 @@ endif()
 find_package(Threads REQUIRED)
 target_link_libraries(test_view_manager_paging PRIVATE Threads::Threads)
 add_test(NAME test_view_manager_paging COMMAND test_view_manager_paging)
+
+# v2.11.0 T1 — RelationManager: the declaration registry for referential
+# integrity relations. Same paging trap as ViewManager/PolicyManager/
+# CollectionConfigManager — loadFromStore() must page explicitly.
+add_executable(test_relation_manager
+    test_relation_manager.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/relations/relation_manager.cpp
+    # v2.11.0 close-out — the create_relation migration op is tested here rather
+    # than in a binary of its own: MigrationRunner needs exactly the MemoryStore
+    # + ViewManager + RelationManager trio this target already links.
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/migrations/migration_runner.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/views/view_manager.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/views/projection.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/memory_store.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/config/collection_config_manager.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/persistence/history_store.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/persistence/wal.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/json_parse.cpp
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src/doc_binary.cpp
+)
+target_include_directories(test_relation_manager PRIVATE
+    ${CMAKE_CURRENT_SOURCE_DIR}/../service/src
+    ${yyjson_INCLUDE_DIRS}
+)
+target_link_libraries(test_relation_manager PRIVATE ${yyjson_LIBRARIES})
+if(TARGET nlohmann_json::nlohmann_json)
+    target_link_libraries(test_relation_manager PRIVATE nlohmann_json::nlohmann_json)
+else()
+    target_include_directories(test_relation_manager PRIVATE ${NLOHMANN_JSON_INCLUDE_DIRS})
+endif()
+if(TARGET spdlog::spdlog)
+    target_link_libraries(test_relation_manager PRIVATE spdlog::spdlog)
+else()
+    target_link_libraries(test_relation_manager PRIVATE ${SPDLOG_LIBRARIES})
+    target_include_directories(test_relation_manager PRIVATE ${SPDLOG_INCLUDE_DIRS})
+endif()
+target_link_libraries(test_relation_manager PRIVATE Threads::Threads)
+# migration_runner.cpp uses RAND_bytes for the generate_secret op.
+target_link_libraries(test_relation_manager PRIVATE OpenSSL::Crypto)
+add_test(NAME test_relation_manager COMMAND test_relation_manager)

+ 93 - 0
tests/load_test/relidx_drop_tool.cpp

@@ -0,0 +1,93 @@
+// v2.11.0 T9 — standalone LMDB sub-db removal, used ONLY to manufacture the
+// "relation declared but its reverse index sub-db is missing" state that
+// DatabaseService::applyRelationDeclarations()'s boot-time self-heal (T7)
+// exists to recover from. That self-heal path had no test anywhere in the
+// suite (see task-9-brief.md) because producing the precondition needs a
+// service-level fixture: the sub-db has to actually be gone, not simulated.
+//
+// This is the exact "audit(path, ...) is CLI-only, separate process" pattern
+// already established by storage/subdb_placement.cpp — it opens its own
+// MDB_env directly on the data directory and must ONLY run while the real
+// service process has that path closed. Running it against a live service's
+// env would violate the documented POSIX-lock hazard (closing any fd on a
+// file drops every lock the process holds on it) and corrupt the running
+// service's LMDB state. test_relations.sh enforces this by only invoking the
+// tool between stop_server and the next start_server.
+//
+// Usage: relidx_drop_tool <env_dir> <subdb_name>
+//   Drops the named sub-db (mdb_drop with del=1, which removes both its
+//   contents and its entry in the env's name table) and exits 0. Exits 1 if
+//   the sub-db does not exist or on any LMDB error.
+
+#include <lmdb.h>
+
+#include <cstdint>
+#include <iostream>
+#include <string>
+
+int main(int argc, char** argv) {
+    if (argc != 3) {
+        std::cerr << "usage: relidx_drop_tool <env_dir> <subdb_name>\n";
+        return 2;
+    }
+    const std::string envDir = argv[1];
+    const std::string subdb = argv[2];
+
+    MDB_env* env = nullptr;
+    if (mdb_env_create(&env) != MDB_SUCCESS) {
+        std::cerr << "mdb_env_create failed\n";
+        return 1;
+    }
+    // Same offline-tool ceiling as subdb_placement.cpp's EnvHandle.
+    mdb_env_set_maxdbs(env, 512);
+    mdb_env_set_mapsize(env, 2ULL << 30);
+
+    int rc = mdb_env_open(env, envDir.c_str(), 0, 0664);
+    if (rc != MDB_SUCCESS) {
+        std::cerr << "mdb_env_open('" << envDir << "') failed: " << mdb_strerror(rc)
+                  << " — this tool must run only while the service holding "
+                     "this env is stopped\n";
+        mdb_env_close(env);
+        return 1;
+    }
+
+    MDB_txn* txn = nullptr;
+    rc = mdb_txn_begin(env, nullptr, 0, &txn);
+    if (rc != MDB_SUCCESS) {
+        std::cerr << "mdb_txn_begin failed: " << mdb_strerror(rc) << "\n";
+        mdb_env_close(env);
+        return 1;
+    }
+
+    MDB_dbi dbi = 0;
+    rc = mdb_dbi_open(txn, subdb.c_str(), 0, &dbi);
+    if (rc != MDB_SUCCESS) {
+        std::cerr << "mdb_dbi_open('" << subdb << "') failed: " << mdb_strerror(rc)
+                  << " (sub-db already absent?)\n";
+        mdb_txn_abort(txn);
+        mdb_env_close(env);
+        return 1;
+    }
+
+    // del=1 removes the sub-database's contents AND its handle/name-table
+    // entry — the exact state applyRelationDeclarations()'s self-heal must
+    // recover from (relation_index_exists() returns false).
+    rc = mdb_drop(txn, dbi, 1);
+    if (rc != MDB_SUCCESS) {
+        std::cerr << "mdb_drop('" << subdb << "') failed: " << mdb_strerror(rc) << "\n";
+        mdb_txn_abort(txn);
+        mdb_env_close(env);
+        return 1;
+    }
+
+    rc = mdb_txn_commit(txn);
+    if (rc != MDB_SUCCESS) {
+        std::cerr << "mdb_txn_commit failed: " << mdb_strerror(rc) << "\n";
+        mdb_env_close(env);
+        return 1;
+    }
+
+    mdb_env_close(env);
+    std::cout << "dropped sub-db '" << subdb << "' in " << envDir << "\n";
+    return 0;
+}

+ 12 - 2
tests/load_test/test_client_namespacing.sh

@@ -5,9 +5,19 @@
 
 set -euo pipefail
 
-cd "$(dirname "$0")"
+# v2.11.0 final review (finding 10) — SCRIPT-RELATIVE, resolved ONCE and
+# BEFORE any cd.
+#
+# Three of these scripts hardcoded `ROOT=/data/smartbotic-database`, so running
+# them from a git worktree built and tested the MAIN checkout instead of the
+# branch under test: a green result was evidence about main, not about the
+# change. The two that were already script-relative computed ROOT from a
+# RELATIVE "$0" AFTER cd-ing, so `bash tests/load_test/<script>.sh` from the
+# repo root failed outright - they only worked when invoked by absolute path.
+SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
+ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
+cd "$SCRIPT_DIR"
 
-ROOT=/data/smartbotic-database
 DIR=/tmp/sbdb-client-namespacing-e2e
 PORT=9011
 

+ 75 - 0
tests/load_test/test_policy_enforcement.cpp

@@ -125,6 +125,81 @@ int main(int argc, char** argv) {
     ck(denied([&]{ (void)reader.get("_policies", "secured:reader"); }),
        "a non-admin cannot read _policies - it describes the access model");
 
+    // ---- Phase 10: v2.11.0 T7 review fix — CheckRelation must not become
+    // an existence oracle. The first cut resolved the relation (to find its
+    // child collection to gate on) BEFORE checking access, so a nonexistent
+    // relation and an existing-but-unauthorized one returned distinguishable
+    // responses (OK/success=false/"does not exist" vs. PERMISSION_DENIED).
+    // A non-admin caller could then map which relation names exist by
+    // probing. Fixed by gating admin-only, before any lookup — this proves
+    // it: a stranger (no policy at all) must see byte-identical denials for
+    // a relation that exists and one that does not.
+    ck(ops.createRelation("docs_rel", "docs", "refId", "docs"),
+       "admin can declare a relation to have something real to probe against");
+
+    auto existing = stranger.checkRelation("docs_rel");
+    auto missing = stranger.checkRelation("does_not_exist_rel");
+    ck(!existing.success, "stranger denied on an EXISTING relation");
+    ck(!missing.success, "stranger denied on a NONEXISTENT relation");
+    ck(!existing.error.empty() && existing.error == missing.error,
+       "identical denial text for both - a stranger cannot tell 'exists but "
+       "denied' apart from 'does not exist' by probing");
+    ck(existing.error.find("PERMISSION_DENIED") != std::string::npos ||
+       existing.error.find("access denied") != std::string::npos,
+       "the denial is in fact the admin gate, not some other failure "
+       "(so this test isn't accidentally passing by both sides erroring "
+       "for unrelated reasons)");
+
+    // The admin path still works and still distinguishes the two cases -
+    // the fix must not have made CheckRelation useless for its actual
+    // audience, only opaque to callers who cannot use it at all.
+    auto adminExisting = ops.checkRelation("docs_rel");
+    auto adminMissing = ops.checkRelation("does_not_exist_rel");
+    ck(adminExisting.success, "admin succeeds on the existing relation");
+    ck(!adminMissing.success && adminMissing.error.find("does not exist") != std::string::npos,
+       "admin still gets a real 'does not exist' for a bogus name");
+
+    // ---- Phase 11: v2.11.0 T9 — admin gating on the other four
+    // relation-management RPCs. T6b (CreateRelation/DropRelation/
+    // ListRelations/GetRelationInfo) shipped `requireAnyAdmin()` gates but
+    // left no e2e proving a non-admin is actually refused under a secured
+    // project - the exact miss class the v2.7.0 audit caught for
+    // Upsert/Batch*/Subscribe (coverage established by walking every
+    // handler, not from a list). `docs_rel` (created by ops in Phase 10)
+    // is reused here as something real to probe: each check proves BOTH
+    // that the stranger is refused AND that the refusal is really the
+    // admin gate (ops can still do it), the same shape as Phase 10.
+    ck(!stranger.createRelation("stranger_rel", "docs", "x", "docs"),
+       "stranger is denied on CreateRelation");
+    ck(!ops.getRelationInfo("stranger_rel").has_value(),
+       "and nothing was actually created by the denied attempt");
+
+    ck(!stranger.dropRelation("docs_rel"),
+       "stranger is denied on DropRelation");
+    ck(ops.getRelationInfo("docs_rel").has_value(),
+       "and 'docs_rel' still exists - the denied attempt did not drop it");
+
+    {
+        auto strangerList = stranger.listRelations();
+        bool sawDocsRel = false;
+        for (const auto& r : strangerList) if (r.name == "docs_rel") sawDocsRel = true;
+        ck(!sawDocsRel, "stranger's ListRelations does not include 'docs_rel'");
+
+        auto opsList = ops.listRelations();
+        bool adminSawDocsRel = false;
+        for (const auto& r : opsList) if (r.name == "docs_rel") adminSawDocsRel = true;
+        ck(adminSawDocsRel, "admin's ListRelations DOES include it - proving the "
+                            "stranger's empty-of-it result is the gate, not the "
+                            "relation being gone");
+    }
+
+    ck(!stranger.getRelationInfo("docs_rel").has_value(),
+       "stranger is denied on GetRelationInfo for a relation that DOES "
+       "exist (proven by the admin listing above) - so this is a real "
+       "denial, not 'not found'");
+    ck(ops.getRelationInfo("docs_rel").has_value(),
+       "admin's GetRelationInfo still works for the same name");
+
     std::cout << "\npassed=" << pass << " failed=" << fail << "\n";
     return fail == 0 ? 0 : 1;
 }

+ 12 - 2
tests/load_test/test_policy_enforcement.sh

@@ -8,9 +8,19 @@
 # real operator workflow.
 
 set -euo pipefail
-cd "$(dirname "$0")"
+# v2.11.0 final review (finding 10) — SCRIPT-RELATIVE, resolved ONCE and
+# BEFORE any cd.
+#
+# Three of these scripts hardcoded `ROOT=/data/smartbotic-database`, so running
+# them from a git worktree built and tested the MAIN checkout instead of the
+# branch under test: a green result was evidence about main, not about the
+# change. The two that were already script-relative computed ROOT from a
+# RELATIVE "$0" AFTER cd-ing, so `bash tests/load_test/<script>.sh` from the
+# repo root failed outright - they only worked when invoked by absolute path.
+SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
+ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
+cd "$SCRIPT_DIR"
 
-ROOT=/data/smartbotic-database
 DIR=/tmp/sbdb-policy-e2e
 PORT=9012
 

+ 268 - 0
tests/load_test/test_relations.cpp

@@ -0,0 +1,268 @@
+// v2.11.0 T9 — relations end to end, over gRPC, through the real client,
+// including a real service restart and a manufactured self-heal fixture.
+//
+// tests/load_test/test_relations_client_e2e.cpp already covers the basic
+// client/server-boundary shape (non-default project, createRelation/
+// listRelations/getRelationInfo qualify-by-name correctly, restrict blocks
+// a delete, cross-project relation names don't collide). This driver adds
+// exactly what that suite structurally cannot, because it never restarts
+// the service:
+//
+//   1. A relation declaration SURVIVES a restart, and — the load-bearing
+//      half — a child document written AFTER the restart is indexed. That
+//      proves DatabaseService::applyRelationDeclarations() re-armed the
+//      write-path maintenance hook, not merely that RelationManager
+//      remembered the row. (Mirrors test_indexes.cpp's identical proof for
+//      secondary indexes.)
+//   2. Boot-time self-heal: with the relation's `_relidx1_*` reverse-index
+//      sub-db manufactured missing (via relidx_drop_tool, run by
+//      test_relations.sh while the service is stopped), a restart must
+//      rebuild the index from the CHILD ROWS THAT ALREADY EXISTED before
+//      the sub-db was dropped — not just index rows written afterward —
+//      and enforcement on the original relationship must work again.
+//   3. Per-project isolation: two projects each declare a relation under
+//      the identical bare name; a project must not see the other
+//      project's children when evaluating restrict/DescribeDelete. Same
+//      failure shape as the v2.4.2 createView bug.
+//
+// Usage: test_relations <address> <projectA> <projectB> <phase>
+//   phase=setup           - declare relations in both projects, insert data,
+//                            assert restrict/no_action/DescribeDelete/isolation
+//   phase=verify           - AFTER restart #1 (plain restart, no sub-db
+//                            tampering): declaration survived, restrict still
+//                            blocks the pre-restart relationship, and a NEW
+//                            child written post-restart is indexed
+//   phase=verify_selfheal - AFTER restart #2 (with projectA's relidx sub-db
+//                            dropped between stop and start): restrict still
+//                            blocks the ORIGINAL pre-drop relationship
+//                            (proving the rebuild recovered existing rows,
+//                            not just future writes), and post-restart
+//                            writes are still indexed too
+
+#include <smartbotic/database/client.hpp>
+
+#include <iostream>
+#include <string>
+
+using namespace smartbotic::database;
+
+namespace {
+
+int g_pass = 0;
+int g_fail = 0;
+
+void check(bool cond, const std::string& msg) {
+    if (cond) {
+        ++g_pass;
+        std::cout << "  ok   " << msg << "\n";
+    } else {
+        ++g_fail;
+        std::cout << "  FAIL " << msg << "\n";
+    }
+}
+
+constexpr const char* kRelation = "wf_exec";
+constexpr const char* kParent = "workflows";
+constexpr const char* kChild = "executions";
+constexpr const char* kChildField = "workflowId";
+
+// Refusal text is pinned from service/src/relations/relation_enforcement.cpp
+// (formatRelationBlockError) — NOT transcribed from any report. A prior task
+// report mis-transcribed "child document(s)" as "document(s)"; assert the
+// real string so that mistake can't repeat silently here.
+bool looksLikeRestrictRefusal(const std::string& err, const std::string& qualifiedRelation,
+                              const std::string& qualifiedChildColl, uint64_t count) {
+    if (err.find("cannot delete '") == std::string::npos) return false;
+    if (err.find("relation '" + qualifiedRelation + "' has " + std::to_string(count) +
+                  " child document(s) in '" + qualifiedChildColl + "' referencing it") ==
+        std::string::npos) {
+        return false;
+    }
+    if (err.find("or change the relation's on_delete to no_action to permit the "
+                 "dangling reference.") == std::string::npos) {
+        return false;
+    }
+    return true;
+}
+
+void setup(Client& a, Client& b, const std::string& projA, const std::string& projB) {
+    std::cout << "-- setup --\n";
+
+    // Identically-named relation declared independently in two projects.
+    // RelationManager keys by the project-qualified name, so this must NOT
+    // collide — same shape the v2.4.2 createView bug got wrong.
+    check(a.createRelation(kRelation, kChild, kChildField, kParent),
+          "projectA declares 'wf_exec'");
+    check(b.createRelation(kRelation, kChild, kChildField, kParent),
+          "projectB declares the SAME bare name 'wf_exec' without collision");
+
+    // A no_action relation in project A, to prove restrict isn't the only
+    // policy path this suite exercises.
+    check(a.createRelation("wf_logs_na", "logs", kChildField, kParent, "no_action"),
+          "projectA declares a no_action relation");
+
+    // Parent + child in A only.
+    a.upsert(kParent, nlohmann::json{{"name", "wf-1"}}, "wf-1");
+    a.upsert(kChild, nlohmann::json{{kChildField, "wf-1"}}, "ex-a1");
+
+    // Same parent id exists in B, but with NO referencing child — this is
+    // the isolation probe: B's identically-named relation must report zero
+    // children for a parent id that DOES have a child in A.
+    b.upsert(kParent, nlohmann::json{{"name", "wf-1"}}, "wf-1");
+
+    // ---- DescribeDelete, non-destructive, before any delete attempt ------
+    auto descA = a.describeDelete(kParent, "wf-1");
+    check(descA.success, "projectA describeDelete succeeds");
+    bool foundA = false;
+    for (const auto& imp : descA.impacts) {
+        if (imp.relation != kRelation) continue;
+        foundA = true;
+        check(imp.childCollection == kChild, "impact reports the bare child collection");
+        check(imp.childField == kChildField, "impact reports the child field");
+        check(imp.onDelete == "restrict", "impact reports the on_delete policy");
+        check(imp.childCount == 1, "projectA sees exactly its own 1 child");
+        check(imp.blocks, "impact.blocks is true (relations_enforced defaults on)");
+    }
+    check(foundA, "projectA describeDelete includes the wf_exec impact");
+
+    auto descB = b.describeDelete(kParent, "wf-1");
+    check(descB.success, "projectB describeDelete succeeds");
+    bool foundB = false;
+    for (const auto& imp : descB.impacts) {
+        if (imp.relation != kRelation) continue;
+        foundB = true;
+        check(imp.childCount == 0,
+              "projectB sees ZERO children for the SAME parent id 'wf-1' - "
+              "A's child does not leak across the project boundary");
+        check(!imp.blocks, "projectB's describeDelete is not blocked");
+    }
+    check(foundB, "projectB describeDelete includes its own wf_exec impact");
+
+    // ---- restrict actually blocks in A, and the message names the relation
+    std::string err;
+    bool deleted = a.remove(kParent, "wf-1", err);
+    check(!deleted, "projectA: restrict blocks deleting a referenced parent");
+    check(looksLikeRestrictRefusal(err, projA + ":" + kRelation, projA + ":" + kChild, 1),
+          "the refusal message matches formatRelationBlockError's real text "
+          "('child document(s)', not 'document(s)')");
+
+    // ---- restrict does NOT block in B: no children reference wf-1 there --
+    err.clear();
+    deleted = b.remove(kParent, "wf-1", err);
+    check(deleted, "projectB: the identically-named relation does not block - "
+                   "isolation holds for the actual delete, not just DescribeDelete");
+    check(err.empty(), "no error on projectB's successful delete");
+
+    // ---- no_action permits a dangling reference -----------------------
+    a.upsert(kParent, nlohmann::json{{"name", "wf-2"}}, "wf-2");
+    a.upsert("logs", nlohmann::json{{kChildField, "wf-2"}}, "log-a1");
+    err.clear();
+    deleted = a.remove(kParent, "wf-2", err);
+    check(deleted, "no_action permits deleting a parent with a dangling child");
+    check(err.empty(), "no error on the no_action delete");
+}
+
+void verify(Client& a, const std::string& projA) {
+    std::cout << "-- verify after restart #1 (plain restart) --\n";
+
+    auto listed = a.listRelations();
+    bool found = false;
+    for (const auto& r : listed) if (r.name == kRelation) found = true;
+    check(found, "the declaration SURVIVED the restart - listRelations still "
+                 "reports 'wf_exec'");
+
+    const std::string qualRelation = projA + ":" + kRelation;
+    const std::string qualChild = projA + ":" + kChild;
+
+    // The pre-restart relationship must still block.
+    std::string err;
+    bool deleted = a.remove(kParent, "wf-1", err);
+    check(!deleted, "restrict still blocks the pre-restart relationship "
+                    "(wf-1/ex-a1) after the restart");
+    check(looksLikeRestrictRefusal(err, qualRelation, qualChild, 1),
+          "the refusal still matches the pinned restrict-block text");
+
+    // THE load-bearing check: a child written AFTER the restart must be
+    // indexed. If applyRelationDeclarations() had merely restored the
+    // in-memory RelationManager cache without calling set_relations() on
+    // the LMDB store, writes after this point would go unmaintained and
+    // this would silently NOT block.
+    a.upsert(kParent, nlohmann::json{{"name", "wf-3"}}, "wf-3");
+    a.upsert(kChild, nlohmann::json{{kChildField, "wf-3"}}, "ex-a3");
+    err.clear();
+    deleted = a.remove(kParent, "wf-3", err);
+    check(!deleted, "a child written AFTER the restart IS indexed - the "
+                    "write path is armed, not merely remembered");
+    check(looksLikeRestrictRefusal(err, qualRelation, qualChild, 1),
+          "the refusal for the post-restart relationship matches the "
+          "pinned text too");
+}
+
+void verifySelfHeal(Client& a, const std::string& projA) {
+    std::cout << "-- verify after restart #2 (relidx sub-db manufactured "
+                 "missing, then restarted) --\n";
+
+    auto listed = a.listRelations();
+    bool found = false;
+    for (const auto& r : listed) if (r.name == kRelation) found = true;
+    check(found, "the declaration still exists after the self-heal restart "
+                 "(dropping the INDEX sub-db must not touch the declaration "
+                 "in RelationManager's own store)");
+
+    const std::string qualRelation = projA + ":" + kRelation;
+    const std::string qualChild = projA + ":" + kChild;
+
+    // THE point of this phase: wf-1/ex-a1 was written and indexed BEFORE
+    // the sub-db was dropped. If the self-heal rebuild only covered new
+    // writes (or didn't run at all), this relationship would no longer
+    // block, and a real delete of wf-1 would silently succeed, leaving
+    // ex-a1 dangling with nothing to say so.
+    std::string err;
+    bool deleted = a.remove(kParent, "wf-1", err);
+    check(!deleted, "restrict STILL blocks the ORIGINAL relationship after "
+                    "the sub-db was dropped and the service restarted - "
+                    "the boot-time self-heal rebuilt it from the existing "
+                    "child row, not merely from future writes");
+    check(looksLikeRestrictRefusal(err, qualRelation, qualChild, 1),
+          "the refusal matches the pinned restrict-block text");
+
+    // And maintenance is still armed post-self-heal for brand new writes.
+    a.upsert(kParent, nlohmann::json{{"name", "wf-4"}}, "wf-4");
+    a.upsert(kChild, nlohmann::json{{kChildField, "wf-4"}}, "ex-a4");
+    err.clear();
+    deleted = a.remove(kParent, "wf-4", err);
+    check(!deleted, "a child written after the self-heal restart is indexed too");
+}
+
+}  // namespace
+
+int main(int argc, char** argv) {
+    if (argc < 5) {
+        std::cerr << "usage: test_relations <address> <projectA> <projectB> "
+                     "<setup|verify|verify_selfheal>\n";
+        return 2;
+    }
+    const std::string address = argv[1];
+    const std::string projA = argv[2];
+    const std::string projB = argv[3];
+    const std::string phase = argv[4];
+
+    Client a({.address = address, .project = projA});
+    a.connect();
+
+    if (phase == "setup") {
+        Client b({.address = address, .project = projB});
+        b.connect();
+        setup(a, b, projA, projB);
+    } else if (phase == "verify") {
+        verify(a, projA);
+    } else if (phase == "verify_selfheal") {
+        verifySelfHeal(a, projA);
+    } else {
+        std::cerr << "unknown phase '" << phase << "'\n";
+        return 2;
+    }
+
+    std::cout << "\npassed=" << g_pass << " failed=" << g_fail << "\n";
+    return g_fail == 0 ? 0 : 1;
+}

+ 333 - 0
tests/load_test/test_relations.sh

@@ -0,0 +1,333 @@
+#!/usr/bin/env bash
+# v2.11.0 T9 — relations end to end, including a real restart and a
+# manufactured boot-time self-heal fixture. See test_relations.cpp for what
+# each phase asserts and why. Companion to test_relations_client_e2e.sh
+# (basic client/server-boundary shape, no restart) — this script exists
+# because the restart + sub-db-tampering machinery needed here has no
+# equivalent in that single-boot harness. Modeled on test_indexes.sh, which
+# established the setup/verify restart pattern for secondary indexes.
+
+set -uo pipefail
+
+BUILD="${1:-build}"
+REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)"
+PORT=19413
+PROJECT_A=relproj_a
+PROJECT_B=relproj_b
+WORK="$(mktemp -d)"
+DRIVER="$WORK/test_relations"
+DROP_TOOL="$WORK/relidx_drop_tool"
+SERVER_PID=""
+
+cleanup() {
+    [[ -n "$SERVER_PID" ]] && kill "$SERVER_PID" 2>/dev/null
+    wait "$SERVER_PID" 2>/dev/null
+    rm -rf "$WORK"
+}
+trap cleanup EXIT
+
+fail() { echo "FAIL: $*" >&2; exit 1; }
+
+echo "=== building the driver and the offline relidx-drop tool ==="
+g++ -std=c++20 -O1 -o "$DRIVER" "$REPO/tests/load_test/test_relations.cpp" \
+    -I"$REPO/client/include" -I"$REPO/$BUILD/client" \
+    -L"$REPO/$BUILD/client" -lsmartbotic-db-client -lspdlog -lfmt \
+    || fail "driver did not compile"
+g++ -std=c++20 -O1 -o "$DROP_TOOL" "$REPO/tests/load_test/relidx_drop_tool.cpp" \
+    -llmdb \
+    || fail "relidx_drop_tool did not compile"
+
+mkdir -p "$WORK/data"
+
+# v2.11.0 close-out — the create_relation MIGRATION OP, over a real boot.
+# Declaring schema in migration files is how consumers ship views, so this is the
+# surface that matters for relations too. Written before the first boot so the op
+# runs on boot 1, which is also the only boot on which the reverse index for it
+# does not yet exist - the case whose log text minor 2 was about.
+MIGPROJ=relproj_m
+mkdir -p "$WORK/migrations"
+cat > "$WORK/migrations/001_relation.json" <<EOF
+{
+  "version": "001",
+  "name": "declare_mig_rel",
+  "operations": [
+    {"type": "create_collection", "collection": "$MIGPROJ:workflows"},
+    {"type": "create_collection", "collection": "$MIGPROJ:executions"},
+    {"type": "create_relation",
+     "name": "$MIGPROJ:mig_rel",
+     "child": "$MIGPROJ:executions",
+     "child_field": "workflowId",
+     "parent": "$MIGPROJ:workflows",
+     "on_delete": "restrict"}
+  ]
+}
+EOF
+
+cat > "$WORK/config.json" <<EOF
+{
+  "storage": {
+    "data_directory": "$WORK/data",
+    "bind_address": "127.0.0.1",
+    "rpc_port": $PORT,
+    "encryption": { "enabled": false, "key_file": "$WORK/data/storage.key" },
+    "migrations": { "enabled": true, "directory": "$WORK/migrations" },
+    "ttl": {
+      "max_candidates_per_sweep": 5000,
+      "max_blocked_retries_per_sweep": 50,
+      "blocked_retry_sweeps": 30,
+      "blocked_summary_sweeps": 120
+    },
+    "replication": { "enabled": false }
+  }
+}
+EOF
+
+start_server() {
+    local log="$1"
+    "$REPO/$BUILD/service/smartbotic-database" --config "$WORK/config.json" > "$log" 2>&1 &
+    SERVER_PID=$!
+    for _ in $(seq 1 60); do
+        grep -q "Notified systemd: READY" "$log" 2>/dev/null && return 0
+        kill -0 "$SERVER_PID" 2>/dev/null || { cat "$log"; fail "server exited during startup"; }
+        sleep 0.5
+    done
+    cat "$log"
+    fail "server never became ready"
+}
+
+stop_server() {
+    [[ -n "$SERVER_PID" ]] || return 0
+    kill "$SERVER_PID" 2>/dev/null
+    wait "$SERVER_PID" 2>/dev/null
+    SERVER_PID=""
+}
+
+export LD_LIBRARY_PATH="$REPO/$BUILD/client:${LD_LIBRARY_PATH:-}"
+
+echo
+echo "=== boot 1: declare relations, insert data, assert restrict/no_action/isolation ==="
+start_server "$WORK/boot1.log"
+"$DRIVER" "127.0.0.1:$PORT" "$PROJECT_A" "$PROJECT_B" setup || fail "setup phase"
+
+echo
+echo "=== restarting the service (plain restart, no tampering) ==="
+stop_server
+start_server "$WORK/boot2.log"
+
+grep -q "v2.11 relations" "$WORK/boot2.log" \
+    || { grep -iE "relation|error" "$WORK/boot2.log" | tail -20
+         fail "the service did not re-apply relation declarations at boot"; }
+echo "  boot log: $(grep 'v2.11 relations' "$WORK/boot2.log" | tail -1 | sed 's/.*\] //')"
+
+echo
+echo "=== phase: verify declaration + write-path survived the restart ==="
+"$DRIVER" "127.0.0.1:$PORT" "$PROJECT_A" "$PROJECT_B" verify || fail "verify phase"
+
+# -------------------------------------------------------------------------
+# v2.11.0 final review (findings 1 and 9) — the surfaces the C++ client does
+# not expose. Client has no batchDelete()/dropCollection-with-relations and no
+# way to send a malformed on_delete, so these go over the raw wire.
+#
+# State at this point (established by the setup phase and carried across the
+# restart): relproj_a:workflows/wf-1 is a restrict-protected parent with
+# exactly one child, relproj_a:executions/ex-a1.
+# -------------------------------------------------------------------------
+GRPCURL="$(command -v grpcurl || true)"
+if [[ -z "$GRPCURL" ]]; then
+    fail "grpcurl is required for the BatchDelete/DropCollection/on_delete phase"
+fi
+rpc() {
+    local method="$1" body="$2"
+    "$GRPCURL" -plaintext -import-path "$REPO/proto" -proto database.proto \
+        -d "$body" "127.0.0.1:$PORT" "smartbotic.databasepb.DatabaseService/$method" 2>&1
+}
+
+echo
+echo "=== phase: BatchDelete must NOT bypass relation enforcement (finding 1) ==="
+OUT="$(rpc BatchDelete "{\"collection\":\"$PROJECT_A:workflows\",\"ids\":[\"wf-1\"]}" || true)"
+grep -q "FailedPrecondition" <<<"$OUT" \
+    || { echo "$OUT"; fail "BatchDelete of a restrict-protected parent was NOT refused"; }
+grep -q "wf_exec" <<<"$OUT" \
+    || { echo "$OUT"; fail "the BatchDelete refusal does not name the blocking relation"; }
+grep -q "nothing was deleted" <<<"$OUT" \
+    || { echo "$OUT"; fail "the BatchDelete refusal does not state that nothing was deleted"; }
+echo "  refused: $(head -2 <<<"$OUT" | tr '\n' ' ')"
+
+OUT="$(rpc Exists "{\"collection\":\"$PROJECT_A:workflows\",\"id\":\"wf-1\"}" || true)"
+grep -q '"exists": true' <<<"$OUT" \
+    || { echo "$OUT"; fail "the protected parent did not survive the refused BatchDelete"; }
+echo "  the parent is still there - the refusal mutated nothing"
+
+echo
+echo "=== phase: DropCollection must refuse a collection in a relation (finding 1 audit) ==="
+OUT="$(rpc DropCollection "{\"name\":\"$PROJECT_A:workflows\"}" || true)"
+grep -q "FailedPrecondition" <<<"$OUT" \
+    || { echo "$OUT"; fail "DropCollection on a relation parent was NOT refused"; }
+grep -q "participates in" <<<"$OUT" \
+    || { echo "$OUT"; fail "the DropCollection refusal does not explain why"; }
+OUT="$(rpc Exists "{\"collection\":\"$PROJECT_A:workflows\",\"id\":\"wf-1\"}" || true)"
+grep -q '"exists": true' <<<"$OUT" \
+    || { echo "$OUT"; fail "DropCollection destroyed the collection despite refusing"; }
+echo "  refused, and the collection is intact"
+
+echo
+echo "=== phase: an unrecognised on_delete is REFUSED, not coerced (finding 9) ==="
+OUT="$(rpc CreateRelation "{\"name\":\"$PROJECT_A:typo_rel\",\"child\":\"$PROJECT_A:executions\",\"child_field\":\"workflowId\",\"parent\":\"$PROJECT_A:workflows\",\"on_delete\":\"Cascade\"}" || true)"
+grep -q "on_delete must be one of" <<<"$OUT" \
+    || { echo "$OUT"; fail "a typo'd on_delete was silently coerced instead of refused"; }
+grep -q "Cascade" <<<"$OUT" \
+    || { echo "$OUT"; fail "the refusal does not echo the value that was rejected"; }
+OUT="$(rpc GetRelationInfo "{\"name\":\"$PROJECT_A:typo_rel\"}" || true)"
+grep -q '"found": true' <<<"$OUT" \
+    && { echo "$OUT"; fail "the refused relation was persisted anyway"; }
+echo "  refused, and nothing was declared"
+
+echo
+echo "=== phase: BatchDelete still deletes what it is allowed to (finding 1) ==="
+OUT="$(rpc BatchDelete "{\"collection\":\"$PROJECT_A:executions\",\"ids\":[\"ex-a1\"]}" || true)"
+grep -q '"deletedCount": "1"' <<<"$OUT" \
+    || { echo "$OUT"; fail "BatchDelete of an unprotected child did not delete it"; }
+OUT="$(rpc BatchDelete "{\"collection\":\"$PROJECT_A:workflows\",\"ids\":[\"wf-1\"]}" || true)"
+grep -q '"deletedCount": "1"' <<<"$OUT" \
+    || { echo "$OUT"; fail "BatchDelete still refused the parent after its last child was removed"; }
+echo "  the child went first, then the parent - enforcement, not a blanket refusal"
+
+# Put the fixture back for the self-heal phase below, which expects one child.
+# ⚠ UpsertRequest.data is `bytes`, so grpcurl needs it BASE64-ENCODED. Passing
+# the raw JSON string there is accepted and silently stores nothing, which is
+# exactly what the first attempt did.
+b64() { printf '%s' "$1" | base64 -w0; }
+PARENT_B64="$(b64 '{"name":"wf-1"}')"
+CHILD_B64="$(b64 '{"workflowId":"wf-1"}')"
+rpc Upsert "{\"collection\":\"$PROJECT_A:workflows\",\"id\":\"wf-1\",\"data\":\"$PARENT_B64\"}" >/dev/null
+rpc Upsert "{\"collection\":\"$PROJECT_A:executions\",\"id\":\"ex-a1\",\"data\":\"$CHILD_B64\"}" >/dev/null
+OUT="$(rpc Exists "{\"collection\":\"$PROJECT_A:executions\",\"id\":\"ex-a1\"}" || true)"
+grep -q '"exists": true' <<<"$OUT" \
+    || { echo "$OUT"; fail "could not restore the e2e fixture for the self-heal phase"; }
+# And the reverse-index posting must be back, or the self-heal phase below is
+# asserting against an empty index and would pass for the wrong reason.
+OUT="$(rpc Delete "{\"collection\":\"$PROJECT_A:workflows\",\"id\":\"wf-1\"}" || true)"
+grep -q "FailedPrecondition" <<<"$OUT" \
+    || { echo "$OUT"; fail "the restored child is not indexed - the fixture is not equivalent"; }
+echo "  fixture restored, and the restored child is indexed"
+
+echo
+echo "=== manufacturing a missing reverse-index sub-db (service stopped) ==="
+stop_server
+ENV_A="$WORK/data/projects/$PROJECT_A/env"
+[[ -d "$ENV_A" ]] || fail "expected project A env dir at $ENV_A"
+"$DROP_TOOL" "$ENV_A" "_relidx1_wf_exec" || fail "relidx_drop_tool could not drop the sub-db"
+
+echo
+echo "=== restarting the service (self-heal must rebuild the dropped sub-db) ==="
+start_server "$WORK/boot3.log"
+
+grep -q "rebuilt 'wf_exec'" "$WORK/boot3.log" \
+    || { grep -iE "relation|error" "$WORK/boot3.log" | tail -20
+         fail "boot did not self-heal the missing relidx sub-db - the T7 self-heal path did not run"; }
+echo "  boot log: $(grep "rebuilt 'wf_exec'" "$WORK/boot3.log" | tail -1 | sed 's/.*\] //')"
+
+echo
+echo "=== phase: verify the self-heal rebuilt the index and enforcement works again ==="
+"$DRIVER" "127.0.0.1:$PORT" "$PROJECT_A" "$PROJECT_B" verify_selfheal || fail "verify_selfheal phase"
+
+echo
+echo "=== phase: the create_relation MIGRATION OP declared and ARMED a relation ==="
+# The op only persists the declaration; runMigrations() re-arms afterwards,
+# because migrations run AFTER the boot-time arming pass. Without that re-arm the
+# relation would maintain no reverse index and enforce nothing until the NEXT
+# restart - so this phase is the test of the ordering claim, not just the op.
+OUT="$(rpc GetRelationInfo "{\"name\":\"$MIGPROJ:mig_rel\"}" || true)"
+grep -q '"found": true' <<<"$OUT" \
+    || { echo "$OUT"; fail "the create_relation migration op did not declare the relation"; }
+echo "  declared by migration: $MIGPROJ:mig_rel"
+
+# Armed? Insert a parent and a child, then try to delete the parent. A relation
+# that was never armed maintains no posting, so the delete would SUCCEED - which
+# is exactly the silent failure mode.
+#
+# ⚠ This check alone cannot prove it was armed on the boot that DECLARED it: two
+# restarts have happened since, and the ordinary boot-time pass would have armed
+# it on either of them. The boot1.log assertion below is what pins that, and it
+# is the one that fails if the post-migration re-arm is removed.
+M_PARENT_B64="$(b64 '{"name":"wf-m1"}')"
+M_CHILD_B64="$(b64 '{"workflowId":"wf-m1"}')"
+rpc Upsert "{\"collection\":\"$MIGPROJ:workflows\",\"id\":\"wf-m1\",\"data\":\"$M_PARENT_B64\"}" >/dev/null
+rpc Upsert "{\"collection\":\"$MIGPROJ:executions\",\"id\":\"ex-m1\",\"data\":\"$M_CHILD_B64\"}" >/dev/null
+OUT="$(rpc Delete "{\"collection\":\"$MIGPROJ:workflows\",\"id\":\"wf-m1\"}" || true)"
+grep -q "FailedPrecondition" <<<"$OUT" \
+    || { echo "$OUT"; fail "a migration-declared relation is NOT enforcing - the post-migration re-arm did not happen"; }
+grep -q "mig_rel" <<<"$OUT" \
+    || { echo "$OUT"; fail "the refusal does not name the migration-declared relation"; }
+echo "  and it is enforcing: the parent delete was refused by $MIGPROJ:mig_rel"
+
+# minor 2: on the boot that first applies the migration, the reverse index
+# legitimately does not exist yet. That must NOT be logged as the
+# "declared but its sub-db was absent (restored snapshot or manual removal)"
+# fault - the text describes a fault and the event is the normal path.
+grep -q "a migration declared this relation on this boot" "$WORK/boot1.log" \
+    || { grep -iE "mig_rel" "$WORK/boot1.log" | tail -5
+         fail "the migration-declared relation's index build was not logged as the expected path"; }
+grep -E "rebuilt 'mig_rel'" "$WORK/boot1.log" \
+    && fail "the expected path was logged as the 'sub-db was absent' FAULT"
+echo "  boot log: $(grep 'a migration declared this relation' "$WORK/boot1.log" | tail -1 | sed 's/.*\] //')"
+
+echo
+echo "=== phase: the TTL blocked-document gauge is reachable over the wire ==="
+# The gauge is the ONLY operator visibility for a document outliving its TTL
+# because a relation blocks the delete. Before the close-out review it existed
+# only as a C++ method with no RPC, which is not a signal. This asserts the field
+# is part of GetMemoryStats and the RPC still answers after the proto change; the
+# nonzero path is unit-tested (test_relation_enforcement's restrict/starvation
+# tests assert the gauge directly).
+OUT="$(rpc GetMemoryStats '{}' || true)"
+grep -q "pressureLevel" <<<"$OUT" \
+    || { echo "$OUT"; fail "GetMemoryStats did not answer after the proto change"; }
+if ! "$GRPCURL" -plaintext -import-path "$REPO/proto" -proto database.proto \
+        -msg-template describe smartbotic.databasepb.GetMemoryStatsResponse 2>&1 \
+        | grep -q "ttlBlockedDocuments"; then
+    fail "ttl_blocked_documents is not on GetMemoryStatsResponse - the gauge is unreachable"
+fi
+echo "  GetMemoryStatsResponse carries ttlBlockedDocuments / ttlExpiryBlockedEvents"
+
+echo
+echo "=== phase: DropProject must clean up its relation declarations (close-out) ==="
+# Declarations live in the GLOBAL _relations collection, so dropping a project
+# used to leave them pointing at a namespace that no longer exists. There is no
+# client surface for CreateProject/DropProject relation cleanup, and no unit
+# fixture builds a DatabaseService, so this is the only level it can be tested at.
+PROJECT_C=relproj_c
+rpc CreateProject "{\"name\":\"$PROJECT_C\"}" >/dev/null
+C_PARENT_B64="$(b64 '{"name":"wf-c1"}')"
+C_CHILD_B64="$(b64 '{"workflowId":"wf-c1"}')"
+rpc Upsert "{\"collection\":\"$PROJECT_C:workflows\",\"id\":\"wf-c1\",\"data\":\"$C_PARENT_B64\"}" >/dev/null
+rpc Upsert "{\"collection\":\"$PROJECT_C:executions\",\"id\":\"ex-c1\",\"data\":\"$C_CHILD_B64\"}" >/dev/null
+OUT="$(rpc CreateRelation "{\"name\":\"$PROJECT_C:wf_exec_c\",\"child\":\"$PROJECT_C:executions\",\"child_field\":\"workflowId\",\"parent\":\"$PROJECT_C:workflows\",\"on_delete\":\"restrict\"}" || true)"
+grep -q '"success": true' <<<"$OUT" \
+    || { echo "$OUT"; fail "could not declare a relation in the throwaway project"; }
+OUT="$(rpc ListRelations "{\"project\":\"$PROJECT_C\"}" || true)"
+grep -q "wf_exec_c" <<<"$OUT" \
+    || { echo "$OUT"; fail "the declaration is not there to begin with - fixture is wrong"; }
+echo "  declared $PROJECT_C:wf_exec_c"
+
+# ⚠ DropProjectResponse's field is `dropped`, not `success` (CreateRelation's
+# is `success`) - checking for the wrong one made this phase fail on a drop that
+# had actually worked.
+OUT="$(rpc DropProject "{\"name\":\"$PROJECT_C\"}" || true)"
+grep -q '"dropped": true' <<<"$OUT" \
+    || { echo "$OUT"; fail "DropProject failed"; }
+
+OUT="$(rpc ListRelations "{\"project\":\"$PROJECT_C\"}" || true)"
+grep -q "wf_exec_c" <<<"$OUT" \
+    && { echo "$OUT"; fail "DropProject left a relation declaration pointing at a dropped namespace"; }
+OUT="$(rpc GetRelationInfo "{\"name\":\"$PROJECT_C:wf_exec_c\"}" || true)"
+grep -q '"found": true' <<<"$OUT" \
+    && { echo "$OUT"; fail "the declaration is still individually resolvable after the drop"; }
+# And the drop must not have taken anyone else's declarations with it.
+OUT="$(rpc ListRelations "{\"project\":\"$PROJECT_A\"}" || true)"
+grep -q "wf_exec" <<<"$OUT" \
+    || { echo "$OUT"; fail "DropProject removed relations belonging to ANOTHER project"; }
+echo "  dropped with the project, and project A's relations are untouched"
+
+echo
+echo "ALL RELATIONS E2E CHECKS PASSED"

+ 140 - 0
tests/load_test/test_relations_client_e2e.cpp

@@ -0,0 +1,140 @@
+// v2.11.0 T6b client/server boundary test.
+//
+// Covers the exact gap that let the v2.4.2 createView bug ship for two
+// releases: qualifying `collection`/`child`/`parent` but not `name` leaves
+// the manager's registry keyed on the bare name while every lookup sends
+// the qualified one, so the keys never meet and the declaration silently
+// behaves as if it does not exist. Relations have the same registry shape
+// (name -> RelationInfo, keyed by the project-qualified name) and the same
+// exposure, so this is the first thing to prove, with a NON-DEFAULT project.
+//
+// tests/test_relation_manager.cpp and tests/test_relation_enforcement.cpp
+// exercise RelationManager and RelationEnforcer in-process; neither drives a
+// real Client against a real server, which is the only place a
+// qualify/unqualify mismatch is observable.
+
+#include <smartbotic/database/client.hpp>
+#include <iostream>
+#include <string>
+
+using namespace smartbotic::database;
+static int pass = 0, fail = 0;
+static void ck(bool c, const char* m) {
+    if (c) { ++pass; } else { ++fail; std::cerr << "FAIL: " << m << "\n"; }
+}
+
+int main() {
+    // ---- Part 1: create/list/get under a NON-DEFAULT project, over gRPC.
+    Client c({.address = "127.0.0.1:9012", .project = "acme"});
+    c.connect();
+
+    ck(c.createRelation("exec_wf", "executions", "workflowId", "workflows"),
+       "createRelation succeeds with a non-default project");
+
+    auto listed = c.listRelations();
+    bool found = false;
+    for (const auto& r : listed) {
+        if (r.name == "exec_wf") {
+            found = true;
+            ck(r.child == "executions", "listRelations reports the BARE child name (round-trip)");
+            ck(r.parent == "workflows", "listRelations reports the BARE parent name (round-trip)");
+            ck(r.childField == "workflowId", "listRelations reports child_field verbatim");
+            ck(r.onDelete == "restrict", "listRelations reports the default on_delete");
+        }
+    }
+    ck(found, "listRelations sees the relation created under its BARE name - "
+              "this is the exact key-mismatch v2.4.2 shipped for createView");
+
+    auto info = c.getRelationInfo("exec_wf");
+    ck(info.has_value(), "getRelationInfo finds the relation by its bare name");
+    ck(info && info->child == "executions", "getRelationInfo round-trips child");
+    ck(info && info->parent == "workflows", "getRelationInfo round-trips parent");
+
+    // Another project must not see this one - same isolation contract as
+    // listViews()/listCollections().
+    Client other({.address = "127.0.0.1:9012", .project = "other_proj"});
+    other.connect();
+    auto theirRelations = other.listRelations();
+    bool leaked = false;
+    for (const auto& r : theirRelations) if (r.name == "exec_wf") leaked = true;
+    ck(!leaked, "listRelations does not leak another project's relation");
+
+    // ---- Part 2: cross-project relations are refused (RelationManager
+    // validates this; the RPC must surface the reason, not swallow it).
+    ck(!c.createRelation("bad_cross", "executions", "x", "other_proj:workflows"),
+       "cross-project relation is refused");
+
+    // ---- Part 3: enforcement blocks a delete with children, and drop
+    // requires the relation to exist first.
+    c.upsert("workflows", nlohmann::json{{"name", "wf-1"}}, "wf-1");
+    c.upsert("executions", nlohmann::json{{"workflowId", "wf-1"}}, "ex-1");
+
+    std::string err;
+    bool deleted = c.remove("workflows", "wf-1", err);
+    ck(!deleted, "delete of a referenced parent is refused");
+    ck(!err.empty(), "remove(..., errorOut) surfaces the refusal message");
+
+    ck(c.getRelationsEnforced("workflows"), "relations_enforced defaults to true");
+    ck(c.setRelationsEnforced("workflows", false), "setRelationsEnforced(false) accepted");
+    ck(!c.getRelationsEnforced("workflows"), "setRelationsEnforced round-trips");
+
+    err.clear();
+    deleted = c.remove("workflows", "wf-1", err);
+    ck(deleted, "delete succeeds once enforcement is disabled for the collection");
+    ck(err.empty(), "no error on a successful delete");
+
+    ck(c.setRelationsEnforced("workflows", true), "re-enable relations_enforced");
+
+    // ---- Part 3b: DescribeDelete over the boundary, with project
+    // qualification. Only exercised in the `default` project by tests/T5's
+    // own unit coverage; this is the first time it goes through a real
+    // Client with a non-default project, so it is the first place a
+    // qualify/unqualify mismatch in the DescribeDelete RPC specifically
+    // would be observable (see client.cpp's describeDelete: `collection` is
+    // qualified like get()/find(), `relation`/`childCollection` in the
+    // response are unqualified back to bare names the same way
+    // listRelations() unqualifies).
+    c.upsert("workflows", nlohmann::json{{"name", "wf-2"}}, "wf-2");
+    c.upsert("executions", nlohmann::json{{"workflowId", "wf-2"}}, "ex-2");
+
+    auto desc = c.describeDelete("workflows", "wf-2");
+    ck(desc.success, "describeDelete succeeds for a non-default project");
+    ck(desc.wouldBeBlocked, "describeDelete reports the delete would be blocked");
+    bool sawImpact = false;
+    for (const auto& imp : desc.impacts) {
+        if (imp.relation != "exec_wf") continue;
+        sawImpact = true;
+        ck(imp.childCollection == "executions",
+           "describeDelete reports the BARE child collection, unqualified "
+           "back from 'acme:executions'");
+        ck(imp.childField == "workflowId", "describeDelete reports child_field verbatim");
+        ck(imp.onDelete == "restrict", "describeDelete reports the on_delete policy");
+        ck(imp.childCount == 1, "describeDelete counts exactly the one referencing child");
+        ck(imp.blocks, "describeDelete's per-impact blocks agrees with wouldBeBlocked");
+        ck(!imp.sampleChildIds.empty() && imp.sampleChildIds[0] == "ex-2",
+           "describeDelete samples the actual referencing child id");
+    }
+    ck(sawImpact, "describeDelete's impacts include the 'exec_wf' relation by "
+                  "its BARE name - the qualify/unqualify round trip works for "
+                  "this RPC specifically, not just listRelations/getRelationInfo");
+
+    // DescribeDelete must never actually delete anything.
+    ck(c.get("workflows", "wf-2").has_value(),
+       "describeDelete is read-only - the parent document still exists");
+
+    // And a real delete of wf-2 is in fact still blocked, matching what
+    // describeDelete predicted.
+    err.clear();
+    ck(!c.remove("workflows", "wf-2", err), "the real delete agrees with describeDelete");
+    (void)c.remove("executions", "ex-2");  // leave wf-2 behind; harmless
+
+    ck(c.dropRelation("exec_wf"), "dropRelation succeeds");
+    ck(!c.getRelationInfo("exec_wf").has_value(), "relation is gone after drop");
+    ck(!c.dropRelation("exec_wf"), "dropping a gone relation fails (not idempotent-success)");
+
+    // c.remove leaves ex-1 behind; harmless, the process exits.
+    (void)c.remove("executions", "ex-1");
+
+    std::cout << "\npassed=" << pass << " failed=" << fail << "\n";
+    return fail == 0 ? 0 : 1;
+}

+ 76 - 0
tests/load_test/test_relations_client_e2e.sh

@@ -0,0 +1,76 @@
+#!/usr/bin/env bash
+# v2.11.0 T6b client/server boundary e2e — relation management over gRPC
+# with a non-default project. See test_relations_client_e2e.cpp for what
+# each assertion covers and why the unit suite could not catch it.
+
+set -euo pipefail
+
+# v2.11.0 final review (finding 10) — SCRIPT-RELATIVE, resolved ONCE and
+# BEFORE any cd.
+#
+# Three of these scripts hardcoded `ROOT=/data/smartbotic-database`, so running
+# them from a git worktree built and tested the MAIN checkout instead of the
+# branch under test: a green result was evidence about main, not about the
+# change. The two that were already script-relative computed ROOT from a
+# RELATIVE "$0" AFTER cd-ing, so `bash tests/load_test/<script>.sh` from the
+# repo root failed outright - they only worked when invoked by absolute path.
+SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
+ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
+cd "$SCRIPT_DIR"
+
+DIR=/tmp/sbdb-relations-client-e2e
+PORT=9012
+
+rm -rf "$DIR"
+mkdir -p "$DIR/data"
+
+cat > "$DIR/config.json" <<EOF
+{
+  "storage": {
+    "data_directory": "$DIR/data",
+    "bind_address": "127.0.0.1",
+    "rpc_port": $PORT,
+    "encryption": { "enabled": false },
+    "migrations": { "enabled": false },
+    "replication": { "enabled": false }
+  }
+}
+EOF
+
+echo "=== building test binary ==="
+g++ -std=c++20 -O1 -o "$DIR/test_relations_client_e2e" test_relations_client_e2e.cpp \
+    -I"$ROOT/client/include" -I"$ROOT/build/client" \
+    -L"$ROOT/build/client" -lsmartbotic-db-client -lspdlog -lfmt
+
+echo "=== booting server on $PORT ==="
+"$ROOT/build/service/smartbotic-database" --config "$DIR/config.json" \
+    > "$DIR/server.log" 2>&1 &
+SERVER_PID=$!
+cleanup() { kill "$SERVER_PID" 2>/dev/null || true; wait "$SERVER_PID" 2>/dev/null || true; }
+trap cleanup EXIT
+
+for _ in $(seq 1 60); do
+    grep -q "Notified systemd: READY" "$DIR/server.log" && break
+    sleep 0.5
+done
+if ! grep -q "Notified systemd: READY" "$DIR/server.log"; then
+    echo "server failed to become ready; log:"
+    tail -30 "$DIR/server.log"
+    exit 1
+fi
+
+echo "=== running assertions ==="
+set +e
+LD_LIBRARY_PATH="$ROOT/build/client" "$DIR/test_relations_client_e2e"
+RC=$?
+set -e
+
+if [[ $RC -ne 0 ]]; then
+    echo ""
+    echo "FAILED — server log tail:"
+    tail -30 "$DIR/server.log"
+    exit $RC
+fi
+
+echo ""
+echo "OK — relations client/server e2e passed"

+ 12 - 2
tests/load_test/test_v24_tls_auth.sh

@@ -14,9 +14,19 @@
 
 set -euo pipefail
 
-cd "$(dirname "$0")"
+# v2.11.0 final review (finding 10) — SCRIPT-RELATIVE, resolved ONCE and
+# BEFORE any cd.
+#
+# Three of these scripts hardcoded `ROOT=/data/smartbotic-database`, so running
+# them from a git worktree built and tested the MAIN checkout instead of the
+# branch under test: a green result was evidence about main, not about the
+# change. The two that were already script-relative computed ROOT from a
+# RELATIVE "$0" AFTER cd-ing, so `bash tests/load_test/<script>.sh` from the
+# repo root failed outright - they only worked when invoked by absolute path.
+SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
+ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
+cd "$SCRIPT_DIR"
 
-ROOT=/data/smartbotic-database
 DIR=/tmp/sbdbv24-tls-auth-e2e
 
 rm -rf "$DIR"

+ 12 - 2
tests/load_test/test_views_multiproject.sh

@@ -20,9 +20,19 @@
 
 set -euo pipefail
 
-cd "$(dirname "$0")"
+# v2.11.0 final review (finding 10) — SCRIPT-RELATIVE, resolved ONCE and
+# BEFORE any cd.
+#
+# Three of these scripts hardcoded `ROOT=/data/smartbotic-database`, so running
+# them from a git worktree built and tested the MAIN checkout instead of the
+# branch under test: a green result was evidence about main, not about the
+# change. The two that were already script-relative computed ROOT from a
+# RELATIVE "$0" AFTER cd-ing, so `bash tests/load_test/<script>.sh` from the
+# repo root failed outright - they only worked when invoked by absolute path.
+SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
+ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
+cd "$SCRIPT_DIR"
 
-ROOT=/data/smartbotic-database
 DIR=/tmp/sbdb-views-multiproject
 PORT=9078
 

+ 101 - 0
tests/test_dual_write_mirror.cpp

@@ -190,6 +190,105 @@ void test_unit_insert_without_doc_is_noop() {
     check(healthy.load() && drift.load() == 0, "INSERT with nullopt doc: clean");
 }
 
+// ---------- v2.11.0 T11 — unique constraint activation ----------
+//
+// The check itself (LmdbDocumentStore::set_unique_fields +
+// maintainIndexes's UniqueViolation throw) is tested in isolation elsewhere.
+// What's under test here is the wiring: a UniqueViolation raised inside
+// put() during the mirror must (1) actually reject the write instead of
+// being swallowed like every other mirror exception, (2) leave no trace of
+// the rejected write in MemoryStore, and (3) NOT be treated as a mirror
+// fault — mirror_healthy_ and mirror_drift_count_ must be untouched. #3 is
+// the one that is easy to get wrong: it is the exact v2.8.1 failure mode
+// (a single bad write taking every read in the process off LMDB) triggered
+// by a constraint doing its job correctly.
+
+void test_unique_violation_on_insert_rejects_and_leaves_no_trace() {
+    TmpEnv tmp("unique-insert");
+    LmdbDocumentStore doc_store(tmp.env);
+    doc_store.set_indexed_fields("users", {"email"});
+    doc_store.set_unique_fields("users", {"email"});
+
+    std::atomic<bool> healthy{true};
+    std::atomic<uint64_t> drift{0};
+
+    MemoryStore::Config cfg;
+    cfg.nodeId = "test-node";
+    cfg.maxMemoryBytes = 64ULL * 1024 * 1024;
+    MemoryStore mem(cfg);
+    mem.setDocumentStoreMirror(
+        [&](std::string_view) -> DocumentStore* { return &doc_store; },
+        &healthy, &drift);
+
+    Document d1 = make_doc("u1", "users", {{"email", "alice@example.com"}});
+    std::string id1 = mem.insert("users", d1);
+    check(id1 == "u1", "unique/insert: first row with the value succeeds");
+
+    Document d2 = make_doc("u2", "users", {{"email", "alice@example.com"}});
+    bool threw = false;
+    try {
+        mem.insert("users", d2);
+    } catch (const smartbotic::db::storage::UniqueViolation&) {
+        threw = true;
+    }
+    check(threw, "unique/insert: duplicate value is rejected");
+    check(!mem.get("users", "u2").has_value(),
+          "unique/insert: no row left behind in MemoryStore");
+    check(healthy.load(),
+          "unique/insert: mirror health untouched by a rejected write");
+    check(drift.load() == 0,
+          "unique/insert: mirror drift untouched by a rejected write");
+
+    // The first row must still be intact and still readable from LMDB —
+    // a rejected second write must not have disturbed it.
+    check(mem.get("users", "u1").has_value(), "unique/insert: first row intact");
+    check(doc_store.get("users", "u1").has_value(),
+          "unique/insert: first row still in LMDB");
+}
+
+void test_unique_violation_on_update_reverts_to_prior_state() {
+    TmpEnv tmp("unique-update");
+    LmdbDocumentStore doc_store(tmp.env);
+    doc_store.set_indexed_fields("users", {"email"});
+    doc_store.set_unique_fields("users", {"email"});
+
+    std::atomic<bool> healthy{true};
+    std::atomic<uint64_t> drift{0};
+
+    MemoryStore::Config cfg;
+    cfg.nodeId = "test-node";
+    cfg.maxMemoryBytes = 64ULL * 1024 * 1024;
+    MemoryStore mem(cfg);
+    mem.setDocumentStoreMirror(
+        [&](std::string_view) -> DocumentStore* { return &doc_store; },
+        &healthy, &drift);
+
+    Document d1 = make_doc("u1", "users", {{"email", "taken@example.com"}});
+    mem.insert("users", d1);
+    Document d2 = make_doc("u2", "users", {{"email", "mine@example.com"}});
+    mem.insert("users", d2);
+
+    Document update2 = make_doc("u2", "users", {{"email", "taken@example.com"}});
+    bool threw = false;
+    try {
+        mem.update("users", "u2", update2);
+    } catch (const smartbotic::db::storage::UniqueViolation&) {
+        threw = true;
+    }
+    check(threw, "unique/update: colliding update is rejected");
+
+    auto got = mem.get("users", "u2");
+    check(got.has_value(), "unique/update: row u2 still exists");
+    check(got.has_value() && got->data().value("email", "") == "mine@example.com",
+          "unique/update: u2's email reverted to its pre-update value");
+    check(got.has_value() && got->version == 1,
+          "unique/update: version did not advance on a rejected write");
+    check(healthy.load(),
+          "unique/update: mirror health untouched by a rejected write");
+    check(drift.load() == 0,
+          "unique/update: mirror drift untouched by a rejected write");
+}
+
 // ---------- Integration via MemoryStore callback ----------
 //
 // Mirrors the DatabaseService wiring: install a callback on MemoryStore
@@ -441,6 +540,8 @@ int main() {
     test_unit_insert_update_delete_round_trip();
     test_unit_failure_marks_degraded();
     test_unit_insert_without_doc_is_noop();
+    test_unique_violation_on_insert_rejects_and_leaves_no_trace();
+    test_unique_violation_on_update_reverts_to_prior_state();
     test_integration_via_memory_store_callback();
     test_vector_mirror_round_trip();
     test_vector_mirror_empty_is_noop_on_insert();

+ 111 - 19
tests/test_eviction.cpp

@@ -25,6 +25,19 @@ using namespace smartbotic::database;
 
 namespace {
 
+// Shared pass/fail counters. New assertions use check() rather than assert()
+// so a failure names what it wanted and the binary keeps running to report
+// everything, instead of aborting on the first one. (The older tests in this
+// file still use assert(); tests/CMakeLists.txt applies -UNDEBUG so those are
+// live in every build type - see the v2.4.3 note there.)
+int g_pass = 0;
+int g_fail = 0;
+
+void check(bool cond, const char* msg) {
+    if (cond) { ++g_pass; }
+    else { ++g_fail; std::cerr << "FAIL: " << msg << "\n"; }
+}
+
 MemoryStore::Config evictionTestConfig(uint64_t maxMb = 1, uint32_t chunkSize = 100) {
     MemoryStore::Config cfg;
     cfg.nodeId = "test";
@@ -110,22 +123,48 @@ void test_hot_write_floor_protects_recent_writes() {
               << liveDocs << "/500 preserved)\n";
 }
 
+// Priority bias: with two equally-sized collections, eviction must take from
+// the Low-priority one before the High-priority one.
+//
+// This test USED to call store.start() before filling, then sleep 500ms and
+// compare survivor counts. That raced its own fill loop against the 20ms
+// eviction tick, and the outcome depended on how fast inserts happened to be:
+//
+//   - Optimised build: all 800 inserts land inside the first 20ms tick, so
+//     eviction sees both collections at 400 and the bias assertion holds.
+//   - Unoptimised build (a plain `cmake -B build -G Ninja`, i.e. no
+//     CMAKE_BUILD_TYPE, which is what CLAUDE.md documents): only ~290 inserts
+//     land in 20ms, so the first tick fires while "important" is still filling
+//     and "archive" is empty or absent. Eviction then evicts only from
+//     "important", latches the v2.4.3 50%-per-episode drain cap against that
+//     partial resident set, and stays PAUSED for the rest of the episode.
+//     "archive" is filled afterwards and never evicted at all, so
+//     archiveLive (400) > importantLive — a guaranteed failure that says
+//     nothing about the priority logic.
+//
+// Measured 10 failures in 12 runs unoptimised, 0 in 12 optimised, at the
+// relations merge-base as well as at branch HEAD. The product was never wrong;
+// the test's precondition was unstated.
+//
+// So the precondition is now established BEFORE the eviction thread exists:
+// fill both collections with the store stopped, assert the resident set is
+// exactly what the comparison assumes, and only then start eviction. The
+// settle wait polls until the live count stops moving instead of trusting a
+// fixed 500ms, and a timeout fails loudly rather than asserting on a
+// half-evicted store.
 void test_priority_low_evicted_first() {
-    // Use a 4 MB cap with target=50% so eviction frees down to ~2 MB,
-    // leaving survivors we can compare. With docs ~2.5 KB each and 400 docs
-    // total (~1 MB data + overhead), we'll be just over hard threshold and
-    // eviction will pick victims with Low priority first.
+    // 4 MB cap, target 25%: the fill lands well above the hard threshold so
+    // eviction has real work, and there are survivors left to compare.
     auto cfg = evictionTestConfig(4);
     cfg.hotWriteFloorMs = 0;           // disable hot-write protection
     cfg.evictionCheckIntervalMs = 20;
     cfg.memorySoftPercent = 30;        // early trickle
     cfg.memoryHardPercent = 50;        // hit hard with our fill
     cfg.memoryEmergencyPercent = 90;
-    cfg.evictionTargetPercent = 25;    // clear down to 25% — leaves headroom
+    cfg.evictionTargetPercent = 25;    // clear down to 25% - leaves headroom
     cfg.evictionChunkSize = 50;        // small chunks, biased selection has effect
     cfg.maxEvictionPassesPerTrigger = 10;
     MemoryStore store(cfg);
-    store.start();
 
     // Two collections: one HIGH priority, one LOW
     CollectionOptions highOpts;
@@ -136,25 +175,74 @@ void test_priority_low_evicted_first() {
     lowOpts.memoryPriority = MemoryPriority::Low;
     store.createCollection("archive", lowOpts);
 
-    fillStore(store, "important", 400, 2048);
-    fillStore(store, "archive", 400, 2048);
+    // NOTE: store.start() has deliberately NOT been called yet. Nothing in
+    // insert() needs the eviction or expiration thread, so the fill below runs
+    // with no concurrent evictor and the resident set is fully determined.
+    const uint64_t kPerCollection = 400;
+    fillStore(store, "important", static_cast<int>(kPerCollection), 2048);
+    fillStore(store, "archive", static_cast<int>(kPerCollection), 2048);
+
+    auto liveCounts = [&]() {
+        uint64_t important = 0, archive = 0;
+        for (const auto& c : store.getMemoryStatsSnapshot().collections) {
+            if (c.collection == "important") important = c.documentCount;
+            if (c.collection == "archive")   archive = c.documentCount;
+        }
+        return std::pair<uint64_t, uint64_t>{important, archive};
+    };
+
+    // Precondition, asserted rather than assumed: both collections are whole,
+    // and the store is over the hard threshold so eviction is guaranteed to
+    // run. If a future change makes the fill cheap enough not to trip
+    // pressure, this fails loudly instead of the test passing vacuously.
+    {
+        auto [important0, archive0] = liveCounts();
+        check(important0 == kPerCollection,
+              "precondition: 'important' fully resident before eviction starts");
+        check(archive0 == kPerCollection,
+              "precondition: 'archive' fully resident before eviction starts");
+        const MemoryPressure p0 = store.pressure();
+        check(p0 == MemoryPressure::Hard || p0 == MemoryPressure::Emergency,
+              "precondition: fill must put the store under hard/emergency pressure");
+    }
 
-    // Let eviction run several passes and settle
-    std::this_thread::sleep_for(std::chrono::milliseconds(500));
+    store.start();
 
-    auto stats = store.getMemoryStatsSnapshot();
-    uint64_t importantLive = 0, archiveLive = 0;
-    for (const auto& c : stats.collections) {
-        if (c.collection == "important") importantLive = c.documentCount;
-        if (c.collection == "archive")   archiveLive = c.documentCount;
+    // Settle: poll until the live count has stopped moving. Eviction pauses
+    // itself once it hits the drain cap or reaches the target, so this
+    // converges quickly; the deadline exists only so a hang reports rather
+    // than comparing a half-evicted store.
+    const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(10);
+    uint64_t lastTotal = kPerCollection * 2;
+    int stableTicks = 0;
+    bool evictionRan = false;
+    bool settled = false;
+    while (std::chrono::steady_clock::now() < deadline) {
+        std::this_thread::sleep_for(std::chrono::milliseconds(25));
+        auto [important, archive] = liveCounts();
+        const uint64_t total = important + archive;
+        if (total < kPerCollection * 2) evictionRan = true;
+        if (evictionRan && total == lastTotal) {
+            if (++stableTicks >= 8) { settled = true; break; }  // ~200ms of no change
+        } else {
+            stableTicks = 0;
+        }
+        lastTotal = total;
     }
+    check(evictionRan, "eviction must run at all given the fill exceeds the cap");
+    check(settled, "eviction must settle within the deadline");
+
+    auto [importantLive, archiveLive] = liveCounts();
 
     // LOW priority should be evicted at least as much as HIGH.
-    assert(archiveLive <= importantLive);
+    check(archiveLive <= importantLive,
+          "low-priority collection must not outlive the high-priority one");
     // Some eviction MUST have happened given our fill vs cap.
-    assert(importantLive < 400 || archiveLive < 400);
+    check(importantLive < kPerCollection || archiveLive < kPerCollection,
+          "some eviction must have happened");
     // Expect strict bias once any eviction has run.
-    assert(archiveLive < importantLive);
+    check(archiveLive < importantLive,
+          "low-priority collection must be evicted strictly more than high");
 
     store.stop();
     std::cout << "PASS: priority=Low collection evicted more than priority=High "
@@ -259,6 +347,10 @@ int main() {
     test_priority_low_evicted_first();
     test_eviction_chunk_size_honored();
     test_eviction_never_drains_the_store();
-    std::cout << "\nAll eviction tests PASSED!\n";
+    if (g_fail != 0) {
+        std::cerr << "\n" << g_fail << " check(s) FAILED (" << g_pass << " passed)\n";
+        return 1;
+    }
+    std::cout << "\nAll eviction tests PASSED! (" << g_pass << " checks)\n";
     return 0;
 }

+ 2936 - 0
tests/test_relation_enforcement.cpp

@@ -0,0 +1,2936 @@
+// v2.11.0 T3 — maintain the relation reverse index on child writes.
+//
+// Storage-only: exercises LmdbDocumentStore::put()/del() maintaining the
+// reverse index declared via set_relations(), using the same set-difference
+// discipline as maintainIndexes(). No enforcement (Task 4) is involved here
+// - relation_index_child_count is used purely as the observation point.
+//
+// v2.11.0 T4 adds restrict/no_action delete enforcement below
+// (test_restrict_blocks_and_names_the_blockers,
+// test_relations_enforced_false_skips_the_check) - unit-level, against
+// RelationManager + RelationEnforcer + LmdbDocumentStore directly, no
+// grpc::ServerContext. The RPC-level check lands in Task 6/9.
+
+#include <algorithm>
+#include <atomic>
+#include <cstdio>
+#include <filesystem>
+#include <iostream>
+#include <string>
+#include <unistd.h>
+#include <vector>
+
+#include <nlohmann/json.hpp>
+
+#include "config/collection_config_manager.hpp"
+#include "document.hpp"
+#include "memory_store.hpp"
+#include "persistence/persistence_manager.hpp"
+#include "relations/relation_cascade.hpp"
+#include "relations/relation_enforcement.hpp"
+#include "relations/relation_manager.hpp"
+#include "storage/document_store_lmdb.hpp"
+#include "storage/lmdb_env.hpp"
+
+#include <lmdb.h>
+
+namespace fs = std::filesystem;
+
+using smartbotic::database::CascadeBlocked;
+using smartbotic::database::CascadeMutation;
+using smartbotic::database::CascadePlan;
+using smartbotic::database::CollectionConfigManager;
+using smartbotic::database::Document;
+using smartbotic::database::MemoryStore;
+using smartbotic::database::OnDelete;
+using smartbotic::database::PersistenceManager;
+using smartbotic::database::RelationBlock;
+using smartbotic::database::RelationEnforcer;
+using smartbotic::database::RelationInfo;
+using smartbotic::database::RelationManager;
+using smartbotic::database::RelationImpact;
+using smartbotic::database::WalEntry;
+using smartbotic::database::WalOpType;
+using smartbotic::database::WriteAheadLog;
+using smartbotic::database::applyCascadeToMemory;
+using smartbotic::database::commitCascadeLmdb;
+using smartbotic::database::describeDeleteImpacts;
+using smartbotic::database::executeCascade;
+using smartbotic::database::findRelationBlocks;
+using smartbotic::database::planCascade;
+using smartbotic::database::ttlExpiryDecision;
+using smartbotic::database::writeCascadeWal;
+using smartbotic::db::storage::LmdbDocumentStore;
+using smartbotic::db::storage::LmdbEnv;
+using smartbotic::db::storage::LmdbEnvOpts;
+using smartbotic::db::storage::MissingParentReference;
+using smartbotic::db::storage::RelationRef;
+
+namespace {
+
+int g_pass = 0;
+int g_fail = 0;
+
+void check(bool cond, const char* msg) {
+    if (cond) {
+        ++g_pass;
+    } else {
+        ++g_fail;
+        std::cerr << "FAIL: " << msg << "\n";
+    }
+}
+
+std::string make_tmpdir(const char* tag) {
+    static std::atomic<int> counter{0};
+    std::string path = "/tmp/relmaint-test-" + std::to_string(::getpid()) + "-" +
+                       std::to_string(counter.fetch_add(1)) + "-" + tag;
+    std::error_code ec;
+    fs::remove_all(path, ec);
+    return path;
+}
+
+struct TmpEnv {
+    std::string path;
+    LmdbEnv env;
+    explicit TmpEnv(const char* tag)
+        : path(make_tmpdir(tag)),
+          env(LmdbEnvOpts{path, 64ULL << 20, 256, 126, false}) {}
+    ~TmpEnv() {
+        std::error_code ec;
+        fs::remove_all(path, ec);
+    }
+    TmpEnv(const TmpEnv&) = delete;
+    TmpEnv& operator=(const TmpEnv&) = delete;
+};
+
+// v2.11.0 T12 — a PersistenceManager over its own tmpdir, for the WAL-first
+// cascade tests. Separate from TmpEnv (which wraps an LmdbEnv) because the
+// two are deliberately independent stores in this test file, exactly as
+// they are in the real service (executeCascade takes both, wired together
+// only by this test/the Delete handler, never by MemoryStore's own
+// persistCallback_ — see relation_cascade.hpp's file header).
+struct TmpPersistence {
+    std::string path;
+    PersistenceManager::Config cfg;
+    PersistenceManager pm;
+    explicit TmpPersistence(const char* tag)
+        : path(make_tmpdir(tag)), cfg(makeConfig(path)), pm(cfg) {}
+    ~TmpPersistence() {
+        std::error_code ec;
+        fs::remove_all(path, ec);
+    }
+    TmpPersistence(const TmpPersistence&) = delete;
+    TmpPersistence& operator=(const TmpPersistence&) = delete;
+
+private:
+    static PersistenceManager::Config makeConfig(const std::string& dir) {
+        PersistenceManager::Config c;
+        c.dataDir = dir;
+        return c;
+    }
+};
+
+// Replay every WAL entry under `dataDir`/wal directly, bypassing
+// PersistenceManager/MemoryStore entirely — the most direct way to answer
+// "does the WAL file actually contain this entry", independent of whatever
+// LMDB or MemoryStore ended up with.
+std::vector<WalEntry> replayRawWal(const std::string& dataDir) {
+    WriteAheadLog::Config wcfg;
+    wcfg.walDir = std::filesystem::path(dataDir) / "wal";
+    WriteAheadLog wal(wcfg);
+    std::vector<WalEntry> out;
+    wal.replay(0, [&](const WalEntry& e) { out.push_back(e); });
+    return out;
+}
+
+bool walHasDelete(const std::vector<WalEntry>& entries,
+                  const std::string& collection, const std::string& id) {
+    for (const auto& e : entries) {
+        if (e.opType == WalOpType::DELETE && e.collection == collection && e.documentId == id) {
+            return true;
+        }
+    }
+    return false;
+}
+
+bool walHasUpdate(const std::vector<WalEntry>& entries,
+                  const std::string& collection, const std::string& id) {
+    for (const auto& e : entries) {
+        if (e.opType == WalOpType::UPDATE && e.collection == collection && e.documentId == id) {
+            return true;
+        }
+    }
+    return false;
+}
+
+// -------------------------------------------------------------------------
+
+void test_index_follows_the_child_field() {
+    TmpEnv t("rel-maint");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions", {{"exec_wf", "workflowId"}});
+
+    auto put = [&](const std::string& id, const nlohmann::json& data) {
+        Document d; d.id = id; d.collection = "executions"; d.set_data(data);
+        store.put("executions", id, d);
+    };
+
+    put("e1", {{"workflowId", "wf-1"}});
+    put("e2", {{"workflowId", "wf-1"}});
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 2, "two children");
+
+    put("e2", {{"workflowId", "wf-2"}});          // re-point
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 1, "left the old parent");
+    check(store.relation_index_child_count("exec_wf", "wf-2") == 1, "joined the new one");
+
+    store.del("executions", "e1");
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 0, "delete removes it");
+
+    // Absent and null are NOT references: they never block a delete and never
+    // count as dangling.
+    put("e3", {{"other", 1}});
+    put("e4", {{"workflowId", nullptr}});
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 0, "absent adds nothing");
+
+    // An ARRAY-valued reference contributes one posting per element.
+    store.set_relations("nodes", {{"node_creds", "config.credentialIds"}});
+    Document n; n.id = "n1"; n.collection = "nodes";
+    n.set_data({{"config", {{"credentialIds", {"c1", "c2"}}}}});
+    store.put("nodes", "n1", n);
+    check(store.relation_index_child_count("node_creds", "c1") == 1, "array element 1");
+    check(store.relation_index_child_count("node_creds", "c2") == 1, "array element 2");
+}
+
+// A collection with no relations declared must pay nothing and never touch
+// any relation sub-db, mirroring how maintainIndexes short-circuits.
+void test_undeclared_collection_maintains_nothing() {
+    TmpEnv t("rel-none");
+    LmdbDocumentStore store(t.env);
+
+    Document d; d.id = "x1"; d.collection = "misc"; d.set_data({{"workflowId", "wf-9"}});
+    store.put("misc", "x1", d);
+    check(store.relation_index_child_count("exec_wf", "wf-9") == 0,
+          "no relation declared on this collection means no posting written");
+}
+
+// Unrelated field updates must not touch the index - the set-difference is
+// exact, not "rewrite unconditionally".
+void test_unrelated_update_leaves_the_posting_alone() {
+    TmpEnv t("rel-stable");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions", {{"exec_wf", "workflowId"}});
+
+    Document d1; d1.id = "e1"; d1.collection = "executions";
+    d1.set_data({{"workflowId", "wf-1"}, {"status", "running"}});
+    store.put("executions", "e1", d1);
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 1, "initial posting");
+
+    Document d2; d2.id = "e1"; d2.collection = "executions";
+    d2.set_data({{"workflowId", "wf-1"}, {"status", "completed"}});
+    store.put("executions", "e1", d2);
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 1,
+          "still one posting after an unrelated field changed");
+}
+
+// Survives across a fresh store instance over the same env - i.e. the reverse
+// index sub-db created at write time was registered via cacheCommittedDbi
+// after commit, not merely usable within the same process instance.
+void test_posting_visible_after_reopen() {
+    TmpEnv t("rel-reopen");
+    {
+        LmdbDocumentStore store(t.env);
+        store.set_relations("executions", {{"exec_wf", "workflowId"}});
+        Document d; d.id = "e1"; d.collection = "executions";
+        d.set_data({{"workflowId", "wf-1"}});
+        store.put("executions", "e1", d);
+    }
+    LmdbDocumentStore reopened(t.env);
+    check(reopened.relation_index_child_count("exec_wf", "wf-1") == 1,
+          "posting survives a fresh LmdbDocumentStore over the same env");
+}
+
+// v2.11.0 T4 — restrict blocks a delete and names the blockers; no_action
+// permits it and leaves the reference dangling, deliberately.
+//
+// RelationManager's declaration registry lives in a MemoryStore's
+// `_relations` collection (see test_relation_manager.cpp's Fixture); the
+// reverse index it queries lives in a separate per-project LmdbDocumentStore
+// (T2/T3, exercised above). This test wires both together the way Task 6's
+// Delete handler eventually will.
+void test_restrict_blocks_and_names_the_blockers() {
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+
+    TmpEnv t("rel-enforce-restrict");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions", {{"exec_wf", "workflowId"}});
+
+    auto put = [&](const std::string& id, const nlohmann::json& data) {
+        Document d; d.id = id; d.collection = "executions"; d.set_data(data);
+        store.put("executions", id, d);
+    };
+    put("e1", {{"workflowId", "wf-1"}});
+    put("e2", {{"workflowId", "wf-1"}});
+
+    RelationInfo rel;
+    rel.name = "default:exec_wf";
+    rel.child = "default:executions";
+    rel.childField = "workflowId";
+    rel.parent = "default:workflows";
+    rel.onDelete = OnDelete::Restrict;
+
+    std::string mgrErr;
+    check(rm.createRelation(rel, mgrErr), "declared the restrict relation");
+
+    bool enforced = true;
+    RelationEnforcer enforcer(rm, store, enforced);
+
+    std::string err;
+    check(!enforcer.canDelete("default:workflows", "wf-1", err), "restrict blocks");
+    check(err.find("exec_wf") != std::string::npos, "names the relation");
+    check(err.find("2") != std::string::npos, "gives the child count");
+    check(err.find("e1") != std::string::npos, "samples a blocking id");
+
+    // no_action permits it and leaves the reference dangling, deliberately.
+    // RelationManager has no update-in-place; re-declare with the new
+    // policy the same way an operator would via dropRelation + createRelation.
+    check(rm.dropRelation(rel.name, mgrErr), "dropped to change on_delete");
+    rel.onDelete = OnDelete::NoAction;
+    check(rm.createRelation(rel, mgrErr), "re-declared as no_action");
+
+    err.clear();
+    check(enforcer.canDelete("default:workflows", "wf-1", err), "no_action permits");
+
+    mstore.stop();
+}
+
+// v2.11.0 T12 — cascade/set_null ARE now destructive (see the tests further
+// below: test_cascade_deletes_scalar_children_wal_first,
+// test_array_reference_pulls_id_and_keeps_document), but that destruction
+// happens entirely OUTSIDE findRelationBlocks()/canDelete() - those two only
+// ever implement `restrict`, deliberately (see relation_enforcement.hpp's
+// file header: "Only restrict blocks"). A relation declaring cascade or
+// set_null must therefore still never BLOCK a delete via this path, exactly
+// like no_action - the actual cascade/set_null execution is a separate step
+// (executeCascade, called by the Delete handler only after this check has
+// already permitted the delete).
+void test_cascade_and_set_null_never_block_via_restrict_check() {
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+
+    TmpEnv t("rel-enforce-cascade");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions", {{"exec_wf", "workflowId"}});
+
+    Document d; d.id = "e1"; d.collection = "executions";
+    d.set_data({{"workflowId", "wf-1"}});
+    store.put("executions", "e1", d);
+
+    RelationInfo rel;
+    rel.name = "default:exec_wf";
+    rel.child = "default:executions";
+    rel.childField = "workflowId";
+    rel.parent = "default:workflows";
+    rel.onDelete = OnDelete::Cascade;
+
+    std::string mgrErr;
+    check(rm.createRelation(rel, mgrErr), "declared a cascade relation");
+
+    auto blocks = findRelationBlocks(rm, store, "default:workflows", "wf-1");
+    check(blocks.empty(), "cascade does not block - not yet destructive (Task 12)");
+
+    check(rm.dropRelation(rel.name, mgrErr), "dropped to change on_delete");
+    rel.onDelete = OnDelete::SetNull;
+    check(rm.createRelation(rel, mgrErr), "re-declared as set_null");
+
+    blocks = findRelationBlocks(rm, store, "default:workflows", "wf-1");
+    check(blocks.empty(), "set_null does not block - not yet destructive (Task 12)");
+
+    mstore.stop();
+}
+
+// relationsEnforced=false (CollectionCfg, Task 8) skips the check entirely,
+// even though a restrict relation with live children exists. Deliberately
+// not retroactive - see RelationEnforcer::canDelete's doc comment.
+void test_relations_enforced_false_skips_the_check() {
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+
+    TmpEnv t("rel-enforce-disabled");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions", {{"exec_wf", "workflowId"}});
+
+    Document d; d.id = "e1"; d.collection = "executions";
+    d.set_data({{"workflowId", "wf-1"}});
+    store.put("executions", "e1", d);
+
+    RelationInfo rel;
+    rel.name = "default:exec_wf";
+    rel.child = "default:executions";
+    rel.childField = "workflowId";
+    rel.parent = "default:workflows";
+    rel.onDelete = OnDelete::Restrict;
+
+    std::string mgrErr;
+    check(rm.createRelation(rel, mgrErr), "declared the restrict relation");
+
+    bool enforced = false;   // per-collection switch, Task 8
+    RelationEnforcer enforcer(rm, store, enforced);
+
+    std::string err;
+    check(enforcer.canDelete("default:workflows", "wf-1", err),
+          "enforcement disabled for this collection lets the delete through - and "
+          "the resulting dangling references will NOT be found by re-enabling it, "
+          "only by `relations check`");
+    check(err.empty(), "no error text is populated on a permitted delete");
+
+    mstore.stop();
+}
+
+// v2.11.0 T5 — DescribeDelete's enumeration. The case that matters most:
+// a parent with children under a restrict relation AND children under a
+// no_action relation. Unlike findRelationBlocks (which only reports
+// blockers), describeDeleteImpacts must report BOTH relations, and mark
+// only the restrict one as blocking.
+void test_describe_delete_reports_restrict_and_no_action() {
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+
+    TmpEnv t("rel-describe-mixed");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions", {{"exec_wf", "workflowId"}});
+    store.set_relations("comments", {{"comment_wf", "workflowId"}});
+
+    auto put = [&](const std::string& coll, const std::string& id, const nlohmann::json& data) {
+        Document d; d.id = id; d.collection = coll; d.set_data(data);
+        store.put(coll, id, d);
+    };
+    put("executions", "e1", {{"workflowId", "wf-1"}});
+    put("executions", "e2", {{"workflowId", "wf-1"}});
+    put("comments", "c1", {{"workflowId", "wf-1"}});
+
+    std::string mgrErr;
+    RelationInfo restrictRel;
+    restrictRel.name = "default:exec_wf";
+    restrictRel.child = "default:executions";
+    restrictRel.childField = "workflowId";
+    restrictRel.parent = "default:workflows";
+    restrictRel.onDelete = OnDelete::Restrict;
+    check(rm.createRelation(restrictRel, mgrErr), "declared the restrict relation");
+
+    RelationInfo noActionRel;
+    noActionRel.name = "default:comment_wf";
+    noActionRel.child = "default:comments";
+    noActionRel.childField = "workflowId";
+    noActionRel.parent = "default:workflows";
+    noActionRel.onDelete = OnDelete::NoAction;
+    check(rm.createRelation(noActionRel, mgrErr), "declared the no_action relation");
+
+    auto impacts = describeDeleteImpacts(rm, store, "default:workflows", "wf-1");
+    check(impacts.size() == 2, "both relations are reported, not just the blocker");
+
+    const RelationImpact* restrictImpact = nullptr;
+    const RelationImpact* noActionImpact = nullptr;
+    for (const auto& imp : impacts) {
+        if (imp.relation == "default:exec_wf") restrictImpact = &imp;
+        if (imp.relation == "default:comment_wf") noActionImpact = &imp;
+    }
+    check(restrictImpact != nullptr, "restrict relation present");
+    check(noActionImpact != nullptr, "no_action relation present");
+    if (restrictImpact) {
+        check(restrictImpact->childCount == 2, "restrict relation counts both children");
+        check(restrictImpact->blocks, "restrict relation with live children blocks");
+        check(restrictImpact->sampleChildIds.size() == 2, "samples both blocking ids");
+    }
+    if (noActionImpact) {
+        check(noActionImpact->childCount == 1, "no_action relation counts its child");
+        check(!noActionImpact->blocks, "no_action never blocks, even with live children");
+        check(noActionImpact->sampleChildIds.size() == 1, "still samples the id for visibility");
+    }
+
+    // Consistency with the actual enforcement decision: RelationEnforcer
+    // (what Delete() really calls) agrees that this delete is blocked, by
+    // exactly the relation describeDeleteImpacts flagged.
+    bool enforced = true;
+    RelationEnforcer enforcer(rm, store, enforced);
+    std::string err;
+    check(!enforcer.canDelete("default:workflows", "wf-1", err),
+          "a real delete right now would in fact be blocked, agreeing with the impact");
+
+    mstore.stop();
+}
+
+// A relation with no live children is reported (so an operator can see the
+// declaration exists) but never blocks - zero postings is not a reference,
+// matching findRelationBlocks' "zero count is not a block" rule.
+void test_describe_delete_zero_children_does_not_block() {
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+
+    TmpEnv t("rel-describe-empty");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions", {{"exec_wf", "workflowId"}});
+
+    RelationInfo rel;
+    rel.name = "default:exec_wf";
+    rel.child = "default:executions";
+    rel.childField = "workflowId";
+    rel.parent = "default:workflows";
+    rel.onDelete = OnDelete::Restrict;
+    std::string mgrErr;
+    check(rm.createRelation(rel, mgrErr), "declared the relation");
+
+    auto impacts = describeDeleteImpacts(rm, store, "default:workflows", "wf-nonexistent");
+    check(impacts.size() == 1, "the declared relation is reported even with no children");
+    check(impacts[0].childCount == 0, "no children referencing this parent id");
+    check(!impacts[0].blocks, "zero children never blocks");
+    check(impacts[0].sampleChildIds.empty(), "no sample ids when there are no children");
+
+    mstore.stop();
+}
+
+// A collection with no declared relations at all reports an empty impacts
+// list - describeDeleteImpacts, like findRelationBlocks, has nothing to say
+// about a parent nothing points at.
+void test_describe_delete_no_relations_declared() {
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+
+    TmpEnv t("rel-describe-none");
+    LmdbDocumentStore store(t.env);
+
+    auto impacts = describeDeleteImpacts(rm, store, "default:orphan_collection", "anything");
+    check(impacts.empty(), "no relations declared means no impacts reported");
+
+    mstore.stop();
+}
+
+// -------------------------------------------------------------------------
+// v2.11.0 T12 — cascade/set_null made destructive, WAL-first.
+// -------------------------------------------------------------------------
+
+// The brief's first required test: cascade delete of a parent with three
+// scalar-referencing children. Children gone, index entries gone, and -
+// the part most likely to be skipped - a WAL entry exists for every child
+// mutation (not just the parent). Would fail if regressed: if
+// commitCascadeLmdb() forgot to delete a child, that child's LMDB row or
+// index posting would survive; if writeCascadeWal() forgot a child, the raw
+// WAL replay would not contain its DELETE entry - this reads the WAL file
+// directly (replayRawWal), not through any code path that could paper over
+// a missing log call.
+void test_cascade_deletes_scalar_children_wal_first() {
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+    cfgManager.loadFromStore();
+
+    TmpEnv t("rel-cascade-scalar");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions", {{"exec_wf", "workflowId"}});
+
+    TmpPersistence p("rel-cascade-scalar-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    auto putBoth = [&](const std::string& id, const nlohmann::json& data) {
+        Document d; d.id = id; d.collection = "executions"; d.set_data(data);
+        store.put("executions", id, d);
+        mstore.loadDocument("default:executions", d);
+    };
+    putBoth("e1", {{"workflowId", "wf-1"}});
+    putBoth("e2", {{"workflowId", "wf-1"}});
+    putBoth("e3", {{"workflowId", "wf-1"}});
+
+    {
+        Document parent; parent.id = "wf-1"; parent.collection = "workflows";
+        parent.set_data({{"name", "example"}});
+        store.put("workflows", "wf-1", parent);
+        mstore.loadDocument("default:workflows", parent);
+    }
+
+    RelationInfo rel;
+    rel.name = "default:exec_wf";
+    rel.child = "default:executions";
+    rel.childField = "workflowId";
+    rel.parent = "default:workflows";
+    rel.onDelete = OnDelete::Cascade;
+    std::string mgrErr;
+    check(rm.createRelation(rel, mgrErr), "declared the cascade relation");
+
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 3, "three children indexed");
+
+    // v2.11.0 T12 review (C2) - a notify callback wired the way
+    // DatabaseGrpcImpl::Delete wires DatabaseService::notifyReplicationAndEvents,
+    // recorded here instead of actually queuing replication/publishing events
+    // (this test has no DatabaseService). Confirms executeCascade() actually
+    // drives it once per mutation plus once for the parent, with the right
+    // event type each time - the wiring C2 added, not just that it compiles.
+    struct Notification {
+        std::string collection, id;
+        bool hadDoc;
+        smartbotic::database::EventType eventType;
+    };
+    std::vector<Notification> notifications;
+    auto notify = [&](const std::string& coll, const std::string& id,
+                      const std::optional<Document>& doc,
+                      smartbotic::database::EventType et) {
+        notifications.push_back({coll, id, doc.has_value(), et});
+    };
+
+    const bool parentExisted = executeCascade(rm, store, p.pm, mstore, cfgManager,
+                                              "default:workflows", "wf-1", notify);
+    check(parentExisted, "the parent document was present and removed");
+
+    check(notifications.size() == 4, "notified for e1, e2, e3 and the parent - nothing missed, nothing extra");
+    int deleteNotifications = 0;
+    for (const auto& n : notifications) {
+        check(!n.hadDoc, "every notification here is a delete - no doc payload");
+        check(n.eventType == smartbotic::database::EventType::DELETE, "every notification is a DELETE");
+        if (n.eventType == smartbotic::database::EventType::DELETE) ++deleteNotifications;
+    }
+    check(deleteNotifications == 4, "all four notifications are DELETE (3 children + parent)");
+
+    // LMDB: children gone, index gone.
+    check(!store.get("executions", "e1").has_value(), "e1 gone from LMDB");
+    check(!store.get("executions", "e2").has_value(), "e2 gone from LMDB");
+    check(!store.get("executions", "e3").has_value(), "e3 gone from LMDB");
+    check(!store.get("workflows", "wf-1").has_value(), "parent gone from LMDB");
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 0,
+          "reverse index has no postings left for wf-1");
+
+    // MemoryStore: same, applied via applyCascadeToMemory / unloadDocument.
+    check(!mstore.get("default:executions", "e1").has_value(), "e1 gone from MemoryStore");
+    check(!mstore.get("default:executions", "e2").has_value(), "e2 gone from MemoryStore");
+    check(!mstore.get("default:executions", "e3").has_value(), "e3 gone from MemoryStore");
+    check(!mstore.get("default:workflows", "wf-1").has_value(), "parent gone from MemoryStore");
+
+    // WAL: every child mutation AND the parent got its own DELETE entry -
+    // read straight from the WAL file, not inferred from LMDB/MemoryStore
+    // state (which could be right for the wrong reason if WAL logging were
+    // silently skipped).
+    p.pm.stop();
+    auto entries = replayRawWal(p.path);
+    check(walHasDelete(entries, "default:executions", "e1"), "WAL has DELETE for e1");
+    check(walHasDelete(entries, "default:executions", "e2"), "WAL has DELETE for e2");
+    check(walHasDelete(entries, "default:executions", "e3"), "WAL has DELETE for e3");
+    check(walHasDelete(entries, "default:workflows", "wf-1"), "WAL has DELETE for the parent");
+
+    mstore.stop();
+}
+
+// The brief's second required test, and the one the file header calls out
+// as the place a MySQL mental model actively misleads: an array-valued
+// reference under `cascade` pulls the id and KEEPS the document, exactly
+// like set_null would. Would fail if regressed: a naive port of "cascade
+// deletes the child" would delete the node here, which this test catches
+// directly (node must still exist afterward), not just check the array
+// shrank.
+void test_array_reference_pulls_id_and_keeps_document() {
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+    cfgManager.loadFromStore();
+
+    TmpEnv t("rel-cascade-array");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("nodes", {{"node_creds", "config.credentialIds"}});
+
+    TmpPersistence p("rel-cascade-array-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    {
+        Document n; n.id = "n1"; n.collection = "nodes";
+        n.set_data({{"config", {{"credentialIds", {"c1", "c2", "c3"}}}}});
+        store.put("nodes", "n1", n);
+        mstore.loadDocument("default:nodes", n);
+    }
+    {
+        Document parent; parent.id = "c1"; parent.collection = "credentials";
+        parent.set_data({{"name", "prod-key"}});
+        store.put("credentials", "c1", parent);
+        mstore.loadDocument("default:credentials", parent);
+    }
+
+    // Cascade, deliberately - this is exactly the policy a MySQL-trained
+    // instinct expects to delete the node. It must not.
+    RelationInfo rel;
+    rel.name = "default:node_creds";
+    rel.child = "default:nodes";
+    rel.childField = "config.credentialIds";
+    rel.parent = "default:credentials";
+    rel.onDelete = OnDelete::Cascade;
+    std::string mgrErr;
+    check(rm.createRelation(rel, mgrErr), "declared the cascade relation on an array field");
+
+    // v2.11.0 T12 review (C2) - same notify-recording as the scalar test,
+    // to confirm the UPDATE (not DELETE) path also notifies correctly with
+    // the post-mutation document attached.
+    struct Notification {
+        std::string collection, id;
+        bool hadDoc;
+        smartbotic::database::EventType eventType;
+    };
+    std::vector<Notification> notifications;
+    auto notify = [&](const std::string& coll, const std::string& id,
+                      const std::optional<Document>& doc,
+                      smartbotic::database::EventType et) {
+        notifications.push_back({coll, id, doc.has_value(), et});
+    };
+
+    const bool parentExisted = executeCascade(rm, store, p.pm, mstore, cfgManager,
+                                              "default:credentials", "c1", notify);
+    check(parentExisted, "the credential document was present and removed");
+
+    check(notifications.size() == 2, "notified for the node update and the parent delete");
+    check(notifications[0].collection == "default:nodes" && notifications[0].id == "n1" &&
+          notifications[0].hadDoc &&
+          notifications[0].eventType == smartbotic::database::EventType::UPDATE,
+          "node notification is an UPDATE carrying the mutated doc");
+    check(notifications[1].collection == "default:credentials" && notifications[1].id == "c1" &&
+          !notifications[1].hadDoc &&
+          notifications[1].eventType == smartbotic::database::EventType::DELETE,
+          "parent notification is a DELETE with no doc payload");
+
+    // The node survives, in BOTH stores, with c1 pulled and the other two ids intact.
+    auto lmdbNode = store.get("nodes", "n1");
+    check(lmdbNode.has_value(), "node n1 still exists in LMDB - not deleted");
+    if (lmdbNode) {
+        auto ids = lmdbNode->data()["config"]["credentialIds"];
+        check(ids.is_array() && ids.size() == 2, "two ids remain");
+        check(std::find(ids.begin(), ids.end(), nlohmann::json("c1")) == ids.end(),
+              "c1 was pulled");
+        check(std::find(ids.begin(), ids.end(), nlohmann::json("c2")) != ids.end(),
+              "c2 survives");
+        check(std::find(ids.begin(), ids.end(), nlohmann::json("c3")) != ids.end(),
+              "c3 survives");
+    }
+
+    auto memNode = mstore.get("default:nodes", "n1");
+    check(memNode.has_value(), "node n1 still exists in MemoryStore - not deleted");
+    if (memNode) {
+        auto ids = memNode->data()["config"]["credentialIds"];
+        check(ids.is_array() && ids.size() == 2, "MemoryStore copy agrees: two ids remain");
+    }
+
+    check(store.relation_index_child_count("node_creds", "c1") == 0,
+          "the pulled posting is gone from the reverse index");
+    check(store.relation_index_child_count("node_creds", "c2") == 1,
+          "c2's posting is untouched");
+
+    check(!store.get("credentials", "c1").has_value(), "the credential itself IS deleted");
+
+    p.pm.stop();
+    auto entries = replayRawWal(p.path);
+    check(walHasUpdate(entries, "default:nodes", "n1"),
+          "WAL has an UPDATE for the node (not a DELETE) - the array rule");
+    check(walHasDelete(entries, "default:credentials", "c1"), "WAL has DELETE for the parent");
+
+    mstore.stop();
+}
+
+// The brief's third required test: simulate a crash between the WAL write
+// (step 3) and the LMDB commit (step 4) by calling exactly those primitives
+// in that order and stopping there - never calling commitCascadeLmdb() or
+// applyCascadeToMemory(). This is what a real crash right after
+// PersistenceManager::flushWal() returns looks like from the next boot's
+// point of view: the WAL file has everything, LMDB has nothing.
+//
+// "Restart" = a FRESH MemoryStore, recovered via a second PersistenceManager
+// instance over the same dataDir (recover() replays from sequence 0 since
+// no snapshot exists - the fresh-install path in persistence_manager.cpp).
+// Recovery must converge on the correct post-cascade state even though the
+// LMDB transaction that would have made it "real" never ran - proving
+// MemoryStore's correctness after a crash depends on the WAL, not on how
+// far the LMDB commit got. Then recovering a SECOND time onto the SAME
+// (already-recovered) store checks that replaying an already-applied
+// delete is a no-op - it must not throw, and the state must not change.
+//
+// ⚠ Scope of what this test can and cannot prove: it exercises the WAL
+// primitives in isolation, so it proves the WAL entries are self-sufficient
+// for correct recovery. It does NOT, by itself, prove that executeCascade()
+// calls writeCascadeWal() before commitCascadeLmdb() - that ordering is
+// pinned by executeCascade() being three plain sequential statements next
+// to this comment in relation_cascade.cpp, not by a runtime fault
+// injection. See the Task 12 report for why a real fault-injection test
+// was judged not worth the complexity here.
+void test_crash_between_wal_and_commit_recovers() {
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+    cfgManager.loadFromStore();
+
+    TmpEnv t("rel-cascade-crash");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions", {{"exec_wf", "workflowId"}});
+
+    TmpPersistence p("rel-cascade-crash-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    // v2.11.0 T12 review (C1) - seed through the REAL WAL via
+    // p.pm.logInsert(), not just store.put()/mstore.loadDocument() (neither
+    // of which writes to the WAL). Without this the WAL below would contain
+    // ONLY the cascade's DELETE entries, freshStore would start from
+    // nothing, and "!freshStore.get(...).has_value()" would be true whether
+    // or not writeCascadeWal() wrote anything at all - a vacuous assertion
+    // (caught in review; confirmed by actually reverting this fix and
+    // re-running, see the Task 12 report). Seeding via logInsert makes the
+    // WAL read INSERT-then-DELETE per document, so recovery only ends up
+    // with nothing there if the DELETE entries genuinely got applied.
+    auto putBoth = [&](const std::string& id, const nlohmann::json& data) {
+        Document d; d.id = id; d.collection = "executions"; d.set_data(data);
+        store.put("executions", id, d);
+        p.pm.logInsert("default:executions", d);
+    };
+    putBoth("e1", {{"workflowId", "wf-1"}});
+    putBoth("e2", {{"workflowId", "wf-1"}});
+
+    {
+        Document parent; parent.id = "wf-1"; parent.collection = "workflows";
+        parent.set_data({{"name", "example"}});
+        store.put("workflows", "wf-1", parent);
+        p.pm.logInsert("default:workflows", parent);
+    }
+
+    RelationInfo rel;
+    rel.name = "default:exec_wf";
+    rel.child = "default:executions";
+    rel.childField = "workflowId";
+    rel.parent = "default:workflows";
+    rel.onDelete = OnDelete::Cascade;
+    std::string mgrErr;
+    check(rm.createRelation(rel, mgrErr), "declared the cascade relation");
+
+    // Steps 1+3 only - simulating a crash immediately after flushWal()
+    // returns, before commitCascadeLmdb()/applyCascadeToMemory() ever run.
+    CascadePlan plan = planCascade(rm, store, cfgManager, "default:workflows", "wf-1");
+    check(plan.mutations.size() == 2, "planned both children");
+    writeCascadeWal(p.pm, "default:workflows", "wf-1", plan);
+    p.pm.stop();   // closes the WAL file, like a process exiting
+
+    // Proof the "crash" really happened: LMDB was never touched by this
+    // cascade attempt - the children and parent are still exactly as they
+    // were.
+    check(store.get("executions", "e1").has_value(),
+          "LMDB was never committed - e1 is still there (this is the point)");
+    check(store.get("executions", "e2").has_value(),
+          "LMDB was never committed - e2 is still there");
+    check(store.get("workflows", "wf-1").has_value(),
+          "LMDB was never committed - the parent is still there");
+
+    // The WAL now genuinely contains 3 INSERTs (e1, e2, parent) followed by
+    // 3 DELETEs (e1, e2, parent) from writeCascadeWal() - read directly,
+    // the same way test_cascade_deletes_scalar_children_wal_first does,
+    // rather than only inferred from the recovered store's absence.
+    auto rawEntries = replayRawWal(p.path);
+    check(walHasDelete(rawEntries, "default:executions", "e1"), "WAL has DELETE for e1");
+    check(walHasDelete(rawEntries, "default:executions", "e2"), "WAL has DELETE for e2");
+    check(walHasDelete(rawEntries, "default:workflows", "wf-1"), "WAL has DELETE for the parent");
+
+    // "Restart": fresh MemoryStore, fresh PersistenceManager over the SAME
+    // dataDir, recover().
+    MemoryStore freshStore(MemoryStore::Config{});
+    freshStore.start();
+    PersistenceManager::Config cfg2;
+    cfg2.dataDir = p.path;
+    PersistenceManager pm2(cfg2);
+    auto outcome = pm2.recover(freshStore);
+    check(outcome.kind != smartbotic::database::RecoveryOutcome::Kind::Failed,
+          "recovery did not fail");
+    check(outcome.walEntriesReplayed >= 6,
+          "replayed the 3 inserts AND the 3 deletes (parent + two children)");
+
+    // Load-bearing: with the C1 fix, this is only possible because the WAL
+    // held both the INSERT and the DELETE for each id, and recovery applied
+    // both in order. Without the DELETE entries (the bug this test exists
+    // to catch), these would all be PRESENT after recovery instead.
+    check(!freshStore.get("default:executions", "e1").has_value(),
+          "recovery converges: e1 is gone from the recovered MemoryStore");
+    check(!freshStore.get("default:executions", "e2").has_value(),
+          "recovery converges: e2 is gone from the recovered MemoryStore");
+    check(!freshStore.get("default:workflows", "wf-1").has_value(),
+          "recovery converges: the parent is gone from the recovered MemoryStore");
+
+    // Replaying an already-applied delete is a no-op: recover() a SECOND
+    // time, onto the SAME already-recovered store. Must not throw and must
+    // not change the outcome.
+    auto outcome2 = pm2.recover(freshStore);
+    check(outcome2.kind != smartbotic::database::RecoveryOutcome::Kind::Failed,
+          "second recovery pass did not fail either");
+    check(!freshStore.get("default:executions", "e1").has_value(),
+          "still gone after replaying the same DELETE a second time - idempotent");
+
+    freshStore.stop();
+    mstore.stop();
+}
+
+// v2.11.0 T12 review (I2) - a cascade must refuse rather than silently
+// destroy a grandchild protected by its own `restrict` relation.
+// workflows --cascade--> executions --restrict--> logs: deleting the
+// workflow would, without this check, delete e1 out from under a log entry
+// that names it, with canDelete() never having evaluated e1 at all (it only
+// ever evaluates the ORIGINAL parent, wf-1). Would fail if regressed: a
+// version of planCascade() without the grandchild check would throw
+// nothing here and instead silently proceed to delete e1.
+void test_cascade_refuses_when_grandchild_is_restrict_protected() {
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+    cfgManager.loadFromStore();
+
+    TmpEnv t("rel-cascade-grandchild");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions", {{"exec_wf", "workflowId"}});
+    store.set_relations("logs", {{"log_exec", "executionId"}});
+
+    TmpPersistence p("rel-cascade-grandchild-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    {
+        Document parent; parent.id = "wf-1"; parent.collection = "workflows";
+        parent.set_data({{"name", "example"}});
+        store.put("workflows", "wf-1", parent);
+    }
+    {
+        Document exec; exec.id = "e1"; exec.collection = "executions";
+        exec.set_data({{"workflowId", "wf-1"}});
+        store.put("executions", "e1", exec);
+    }
+    {
+        Document log; log.id = "log1"; log.collection = "logs";
+        log.set_data({{"executionId", "e1"}});
+        store.put("logs", "log1", log);
+    }
+
+    std::string mgrErr;
+    RelationInfo cascadeRel;
+    cascadeRel.name = "default:exec_wf";
+    cascadeRel.child = "default:executions";
+    cascadeRel.childField = "workflowId";
+    cascadeRel.parent = "default:workflows";
+    cascadeRel.onDelete = OnDelete::Cascade;
+    check(rm.createRelation(cascadeRel, mgrErr), "declared the cascade relation");
+
+    RelationInfo restrictRel;
+    restrictRel.name = "default:log_exec";
+    restrictRel.child = "default:logs";
+    restrictRel.childField = "executionId";
+    restrictRel.parent = "default:executions";
+    restrictRel.onDelete = OnDelete::Restrict;
+    check(rm.createRelation(restrictRel, mgrErr), "declared the grandchild restrict relation");
+
+    bool threw = false;
+    std::string thrownMessage;
+    try {
+        executeCascade(rm, store, p.pm, mstore, cfgManager, "default:workflows", "wf-1");
+    } catch (const CascadeBlocked& e) {
+        threw = true;
+        thrownMessage = e.what();
+    }
+    check(threw, "cascade refuses rather than silently deleting the restrict-protected grandchild");
+    check(thrownMessage.find("e1") != std::string::npos, "names the blocked child");
+    check(thrownMessage.find("log_exec") != std::string::npos || thrownMessage.find("logs") != std::string::npos,
+          "names the blocking relation or collection");
+
+    // Nothing was mutated anywhere - the throw happens inside planCascade(),
+    // before writeCascadeWal() ever runs.
+    check(store.get("workflows", "wf-1").has_value(), "parent untouched");
+    check(store.get("executions", "e1").has_value(), "the protected child untouched");
+    check(store.get("logs", "log1").has_value(), "the grandchild untouched");
+
+    p.pm.stop();
+    auto entries = replayRawWal(p.path);
+    check(entries.empty(), "nothing was ever written to the WAL - the refusal happened before step 3");
+
+    mstore.stop();
+}
+
+// v2.11.0 T12 review round 3 (I1) - a crash between writeCascadeWal()'s
+// fsync and commitCascadeLmdb() leaves an UpdateChild mutation's WAL UPDATE
+// entry durable while its LMDB write never ran. Unlike DeleteChild (which
+// self-heals on replay because DELETE replay goes through
+// MemoryStore::remove(), which mirrors), UPDATE/UPSERT replay goes through
+// loadDocumentWithHistory(), which never mirrored - so before this fix,
+// LMDB would permanently keep serving the pre-cascade array/field. This
+// test drives the REAL cascade path (planCascade + writeCascadeWal against
+// an array reference, so the mutation really is an UpdateChild - not the
+// isolated primitive) through exactly that crash window, "restarts" with
+// the LMDB mirror wired the same way DatabaseService wires it in production
+// (setupComponents() wires the mirror BEFORE persistence_->recover() runs -
+// confirmed by reading database_service.cpp's initialize()), and asserts
+// LMDB converges, not just MemoryStore.
+void test_replayed_cascade_update_remirrors_to_lmdb_after_crash_window() {
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+    cfgManager.loadFromStore();
+
+    TmpEnv t("rel-remirror");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("nodes", {{"node_creds", "config.credentialIds"}});
+
+    TmpPersistence p("rel-remirror-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    // Seed through the REAL WAL (logInsert), same discipline as the C1 fix -
+    // these assertions must depend on genuine WAL replay, not just on
+    // store.put()/whatever MemoryStore starts with.
+    Document node; node.id = "n1"; node.collection = "nodes";
+    node.set_data({{"config", {{"credentialIds", {"c1", "c2", "c3"}}}}});
+    store.put("nodes", "n1", node);
+    p.pm.logInsert("default:nodes", node);
+
+    Document parent; parent.id = "c1"; parent.collection = "credentials";
+    parent.set_data({{"name", "prod-key"}});
+    store.put("credentials", "c1", parent);
+    p.pm.logInsert("default:credentials", parent);
+
+    RelationInfo rel;
+    rel.name = "default:node_creds";
+    rel.child = "default:nodes";
+    rel.childField = "config.credentialIds";
+    rel.parent = "default:credentials";
+    rel.onDelete = OnDelete::Cascade;   // array reference -> UpdateChild (pull), not DeleteChild
+    std::string mgrErr;
+    check(rm.createRelation(rel, mgrErr), "declared the cascade relation on an array field");
+
+    // Steps 1+3 only - simulating a crash immediately after flushWal()
+    // returns, before commitCascadeLmdb()/applyCascadeToMemory() ever run.
+    CascadePlan plan = planCascade(rm, store, cfgManager, "default:credentials", "c1");
+    check(plan.mutations.size() == 1, "planned the one array-pull mutation");
+    check(plan.mutations[0].kind == CascadeMutation::Kind::UpdateChild,
+          "confirmed UpdateChild - this is exactly the case DeleteChild does NOT cover");
+    writeCascadeWal(p.pm, "default:credentials", "c1", plan);
+    p.pm.stop();   // closes the WAL file, like a process exiting
+
+    // Proof the "crash" really happened: LMDB was never touched by this
+    // cascade attempt - the node still has all three ids.
+    auto beforeRecovery = store.get("nodes", "n1");
+    check(beforeRecovery.has_value(), "n1 still in LMDB");
+    if (beforeRecovery) {
+        auto ids = beforeRecovery->data()["config"]["credentialIds"];
+        check(ids.is_array() && ids.size() == 3,
+              "LMDB was never committed - still the PRE-cascade array (this is the point)");
+    }
+
+    // "Restart": fresh MemoryStore with the LMDB mirror wired to the SAME
+    // LmdbDocumentStore (matching production ordering - the mirror is wired
+    // in DatabaseService::setupComponents(), which runs BEFORE
+    // persistence_->recover() in initialize()), fresh PersistenceManager
+    // over the same dataDir, recover().
+    MemoryStore freshStore(MemoryStore::Config{});
+    freshStore.start();
+    std::atomic<bool> mirrorHealthy{true};
+    std::atomic<uint64_t> mirrorDrift{0};
+    freshStore.setDocumentStoreMirror(
+        [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
+        &mirrorHealthy, &mirrorDrift);
+
+    PersistenceManager::Config cfg2;
+    cfg2.dataDir = p.path;
+    PersistenceManager pm2(cfg2);
+    auto outcome = pm2.recover(freshStore);
+    // v2.11.0 final review (finding 4) — recover() only COLLECTS the list now;
+    // DatabaseService::initialize() runs the pass after
+    // applyRelationDeclarations()/applyIndexDeclarations(). Mirrored here, in
+    // that order, so the test drives the production sequence.
+    pm2.runPendingRemirror(freshStore, outcome);
+    check(outcome.kind != smartbotic::database::RecoveryOutcome::Kind::Failed,
+          "recovery did not fail");
+    check(outcome.updatesRemirroredAfterReplay >= 1,
+          "the re-mirror pass handled at least the node's replayed UPDATE");
+
+    // MemoryStore converges (this part already worked before this fix).
+    auto memNode = freshStore.get("default:nodes", "n1");
+    check(memNode.has_value(), "n1 present in the recovered MemoryStore");
+    if (memNode) {
+        auto ids = memNode->data()["config"]["credentialIds"];
+        check(ids.is_array() && ids.size() == 2, "MemoryStore has the post-cascade array");
+    }
+    check(!freshStore.get("default:credentials", "c1").has_value(),
+          "parent gone from the recovered MemoryStore (DELETE replay already worked pre-fix)");
+
+    // LOAD-BEARING: LMDB now ALSO reflects the post-cascade array - only
+    // true because recover()'s remirror pass ran. Without the I1 fix this
+    // would still show 3 ids, exactly like `beforeRecovery` above.
+    auto lmdbAfter = store.get("nodes", "n1");
+    check(lmdbAfter.has_value(), "n1 still in LMDB after recovery");
+    if (lmdbAfter) {
+        auto ids = lmdbAfter->data()["config"]["credentialIds"];
+        check(ids.is_array() && ids.size() == 2,
+              "LMDB converged to the post-cascade array - the I1 fix");
+        check(std::find(ids.begin(), ids.end(), nlohmann::json("c1")) == ids.end(),
+              "c1 is gone from LMDB too, not just MemoryStore");
+    }
+    // The parent delete self-healed on its own even before this fix
+    // (DELETE replay mirrors via MemoryStore::remove()) - confirmed here so
+    // the test pins the WHOLE combined scenario, not just the new half.
+    check(!store.get("credentials", "c1").has_value(),
+          "parent also gone from LMDB after recovery");
+
+    freshStore.stop();
+    mstore.stop();
+}
+
+// v2.11.0 T12 round-4 — the post-replay re-mirror pass must not be able to
+// stop the service from starting. Two throws are genuinely reachable from it
+// (see MemoryStore::remirrorDocuments): std::invalid_argument from
+// parseProjectCollection() on a malformed/legacy collection key, and a
+// storage fault from the put itself. Unguarded, either escaped recover(),
+// which DatabaseService::initialize() turns into "refuse to start" - a
+// deterministic boot loop on data that booted fine before.
+//
+// This also pins the batching contract: a row that throws inside a chunk
+// aborts that chunk's transaction, and the OTHER rows of that chunk must
+// still converge via the row-by-row retry.
+void test_a_failing_row_does_not_stop_recovery() {
+    TmpEnv t("rel-remirror-fail");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("rel-remirror-fail-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    // Row 1: an ordinary, well-formed document. Must converge.
+    Document good; good.id = "w1"; good.collection = "widgets";
+    good.set_data({{"v", 1}});
+    p.pm.logInsert("default:widgets", good);
+    good.set_data({{"v", 2}});
+    p.pm.logUpdate("default:widgets", good);
+
+    // Row 2: same project, so it lands in the SAME chunk as row 1 - but its
+    // id is past LMDB's 511-byte key limit, so the put throws
+    // (MDB_BAD_VALSIZE) from inside that chunk's transaction. This is the
+    // honest way to induce a mid-chunk throw, same technique as
+    // test_aborted_write_does_not_poison_the_collection (v2.8.0).
+    Document oversize; oversize.id = std::string(600, 'k'); oversize.collection = "widgets";
+    oversize.set_data({{"v", 1}});
+    p.pm.logInsert("default:widgets", oversize);
+    p.pm.logUpdate("default:widgets", oversize);
+
+    // Row 3: a malformed collection key. WAL replay itself never parses
+    // collection keys, so this replays fine and only the re-mirror pass
+    // trips on it - previously with an uncaught std::invalid_argument.
+    Document weird; weird.id = "x1"; weird.collection = "legacy";
+    weird.set_data({{"v", 1}});
+    p.pm.logInsert("weird:legacy:key", weird);
+    p.pm.logUpdate("weird:legacy:key", weird);
+    p.pm.stop();
+
+    MemoryStore freshStore(MemoryStore::Config{});
+    freshStore.start();
+    std::atomic<bool> mirrorHealthy{true};
+    std::atomic<uint64_t> mirrorDrift{0};
+    freshStore.setDocumentStoreMirror(
+        [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
+        &mirrorHealthy, &mirrorDrift);
+
+    PersistenceManager::Config cfg2;
+    cfg2.dataDir = p.path;
+    PersistenceManager pm2(cfg2);
+    auto outcome = pm2.recover(freshStore);
+    pm2.runPendingRemirror(freshStore, outcome);   // finding 4 — see above
+
+    // THE finding: recovery completes.
+    check(outcome.kind != smartbotic::database::RecoveryOutcome::Kind::Failed,
+          "recovery completed despite two un-mirrorable rows - no boot loop");
+    check(outcome.walEntriesReplayed == 6,
+          "all six WAL entries replayed - the failures did not truncate replay");
+
+    // The failures are counted and reported, not swallowed silently.
+    check(outcome.updatesRemirrorFailed == 2,
+          "both un-mirrorable rows were counted as failed");
+    check(mirrorDrift.load() == 2,
+          "each failed row bumped mirror drift (the operator-visible signal)");
+    check(mirrorHealthy.load(),
+          "mirror health NOT flipped - one legacy row must not send every read "
+          "in the process to MemoryStore for the rest of its life");
+
+    // The other row in the same chunk still converged.
+    check(outcome.updatesRemirroredAfterReplay == 1, "the good row was re-mirrored");
+    auto lmdbGood = store.get("widgets", "w1");
+    check(lmdbGood.has_value(), "the good row reached LMDB despite sharing a chunk with a bad one");
+    if (lmdbGood) {
+        check(lmdbGood->data()["v"] == 2, "and it carries the REPLAYED update, not the insert");
+    }
+
+    // MemoryStore replay itself was unaffected for every row, including the
+    // ones LMDB could not take.
+    check(freshStore.get("default:widgets", "w1").has_value(), "w1 in MemoryStore");
+    check(freshStore.get("weird:legacy:key", "x1").has_value(),
+          "the malformed-key row still recovered into MemoryStore - it is only "
+          "LMDB that cannot hold it");
+
+    freshStore.stop();
+}
+
+// v2.11.0 T12 round-4 (point 3) — INSERT must be tracked by the re-mirror
+// pass too. loadDocument() (INSERT replay) does not mirror, so a
+// delete-then-reinsert of the same id inside one replay window used to leave
+// LMDB with the row DELETED: the DELETE mirrored via remove(), the reinsert
+// did not.
+void test_reinsert_after_delete_converges_in_lmdb() {
+    TmpEnv t("rel-remirror-reinsert");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("rel-remirror-reinsert-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    Document d; d.id = "r1"; d.collection = "widgets";
+    d.set_data({{"v", 1}});
+    store.put("widgets", "r1", d);           // LMDB starts in the pre-window state
+    p.pm.logInsert("default:widgets", d);
+    p.pm.logDelete("default:widgets", "r1");
+    d.set_data({{"v", 99}});                 // reinserted with new content
+    p.pm.logInsert("default:widgets", d);
+    p.pm.stop();
+
+    MemoryStore freshStore(MemoryStore::Config{});
+    freshStore.start();
+    std::atomic<bool> mirrorHealthy{true};
+    std::atomic<uint64_t> mirrorDrift{0};
+    freshStore.setDocumentStoreMirror(
+        [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
+        &mirrorHealthy, &mirrorDrift);
+
+    PersistenceManager::Config cfg2;
+    cfg2.dataDir = p.path;
+    PersistenceManager pm2(cfg2);
+    auto outcome = pm2.recover(freshStore);
+    pm2.runPendingRemirror(freshStore, outcome);   // finding 4 — see above
+    check(outcome.kind != smartbotic::database::RecoveryOutcome::Kind::Failed, "recovery completed");
+    check(outcome.updatesRemirroredAfterReplay == 1,
+          "the reinserted id was re-mirrored (INSERT is tracked, not just UPDATE/UPSERT)");
+
+    auto mem = freshStore.get("default:widgets", "r1");
+    check(mem.has_value(), "reinserted row present in MemoryStore");
+
+    // LOAD-BEARING: without INSERT tracking the DELETE's own mirror wins and
+    // LMDB has nothing here at all.
+    auto lmdb = store.get("widgets", "r1");
+    check(lmdb.has_value(), "reinserted row present in LMDB - the DELETE's mirror did not win");
+    if (lmdb) check(lmdb->data()["v"] == 99, "and it is the REINSERTED content, not the original");
+
+    freshStore.stop();
+}
+
+// -------------------------------------------------------------------------
+// v2.11.0 T13 — validate_on_write.
+//
+// The mitigation for the restrict race documented in relation_cascade.hpp
+// and the plan's self-review: a `restrict` check on delete runs in its own
+// read, so a child insert racing that check can still commit after the
+// parent is gone (LMDB serialises the two transactions, but the loser just
+// commits second). validate_on_write closes it by re-checking the parent's
+// existence with an mdb_get on the parent's own sub-db INSIDE the child's
+// write transaction - see LmdbDocumentStore::maintainRelations. What makes
+// it a genuine fix rather than a narrower window: the check and the write
+// commit as one unit, so there is no interval between them for the parent
+// to vanish in.
+// -------------------------------------------------------------------------
+
+void test_validate_on_write_rejects_missing_parent() {
+    TmpEnv t("validate-missing");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions",
+                        {{"exec_wf", "workflowId", "workflows", true}});
+
+    Document d; d.id = "e1"; d.collection = "executions";
+    d.set_data({{"workflowId", "wf-ghost"}});
+
+    bool threw = false;
+    std::string what;
+    try {
+        store.put("executions", "e1", d);
+    } catch (const MissingParentReference& e) {
+        threw = true;
+        what = e.what();
+    }
+    check(threw, "validateOnWrite=true rejects a reference to a nonexistent parent");
+    check(what.find("exec_wf") != std::string::npos, "error names the relation");
+    check(what.find("workflowId") != std::string::npos, "error names the child field");
+    check(what.find("wf-ghost") != std::string::npos, "error names the missing parent id");
+
+    // A rejected write must leave no trace - not the document, not the
+    // reverse index posting.
+    check(!store.get("executions", "e1").has_value(),
+          "rejected insert left no document behind");
+    check(store.relation_index_child_count("exec_wf", "wf-ghost") == 0,
+          "rejected insert left no reverse-index posting behind");
+}
+
+void test_validate_on_write_accepts_existing_parent() {
+    TmpEnv t("validate-present");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions",
+                        {{"exec_wf", "workflowId", "workflows", true}});
+
+    Document w; w.id = "wf-1"; w.collection = "workflows";
+    w.set_data({{"name", "real workflow"}});
+    store.put("workflows", "wf-1", w);
+
+    Document d; d.id = "e1"; d.collection = "executions";
+    d.set_data({{"workflowId", "wf-1"}});
+    bool threw = false;
+    try {
+        store.put("executions", "e1", d);
+    } catch (const MissingParentReference&) {
+        threw = true;
+    }
+    check(!threw, "validateOnWrite=true accepts a reference to an existing parent");
+    check(store.get("executions", "e1").has_value(), "the child document was actually written");
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 1,
+          "and the reverse index posting was written");
+}
+
+// With validateOnWrite=false (the default), the same missing-parent insert
+// succeeds and check_relation_dangling reports it - the brief's Step 1 case,
+// both halves.
+void test_validate_on_write_false_allows_dangling_and_check_reports_it() {
+    TmpEnv t("validate-off");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions",
+                        {{"exec_wf", "workflowId", "workflows", false}});
+
+    Document d; d.id = "e1"; d.collection = "executions";
+    d.set_data({{"workflowId", "wf-ghost"}});
+    bool threw = false;
+    try {
+        store.put("executions", "e1", d);
+    } catch (const MissingParentReference&) {
+        threw = true;
+    }
+    check(!threw, "validateOnWrite=false lets the insert through");
+    check(store.get("executions", "e1").has_value(), "the dangling child document was written");
+
+    auto result = store.check_relation_dangling("exec_wf", "workflows", 100);
+    check(result.total == 1, "relations check reports one dangling parent id");
+    check(result.entries.size() == 1 && result.entries[0].parentId == "wf-ghost",
+          "and it is the ghost id");
+    check(result.entries[0].childCount == 1 &&
+          !result.entries[0].sampleChildIds.empty() &&
+          result.entries[0].sampleChildIds[0] == "e1",
+          "naming the dangling child");
+}
+
+// Absent and null references are not references at all (same rule as T3's
+// index maintenance) - they must never be rejected, even under
+// validateOnWrite=true.
+void test_validate_on_write_never_rejects_absent_or_null() {
+    TmpEnv t("validate-absent-null");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions",
+                        {{"exec_wf", "workflowId", "workflows", true}});
+
+    Document d1; d1.id = "e1"; d1.collection = "executions";
+    d1.set_data({{"other", 1}});   // workflowId absent entirely
+    bool threw1 = false;
+    try {
+        store.put("executions", "e1", d1);
+    } catch (const MissingParentReference&) {
+        threw1 = true;
+    }
+    check(!threw1, "an absent reference field is never rejected");
+
+    Document d2; d2.id = "e2"; d2.collection = "executions";
+    d2.set_data({{"workflowId", nullptr}});
+    bool threw2 = false;
+    try {
+        store.put("executions", "e2", d2);
+    } catch (const MissingParentReference&) {
+        threw2 = true;
+    }
+    check(!threw2, "an explicit null reference is never rejected");
+}
+
+// Array-valued reference decision: ANY missing parent id rejects the WHOLE
+// write, not just that element - fail closed, and no partial state (some
+// elements resolvable, others not) is left behind.
+void test_validate_on_write_array_any_missing_rejects_whole_write() {
+    TmpEnv t("validate-array");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("nodes",
+                        {{"node_creds", "config.credentialIds", "credentials", true}});
+
+    Document c1; c1.id = "c1"; c1.collection = "credentials";
+    c1.set_data({{"name", "real cred"}});
+    store.put("credentials", "c1", c1);
+    // c2 is deliberately never created.
+
+    Document n; n.id = "n1"; n.collection = "nodes";
+    n.set_data({{"config", {{"credentialIds", {"c1", "c2"}}}}});
+    bool threw = false;
+    std::string what;
+    try {
+        store.put("nodes", "n1", n);
+    } catch (const MissingParentReference& e) {
+        threw = true;
+        what = e.what();
+    }
+    check(threw, "one missing element in an array reference rejects the whole write");
+    check(what.find("c2") != std::string::npos, "error names the missing element, not the valid one");
+    check(!store.get("nodes", "n1").has_value(),
+          "rejected write left no document behind");
+    check(store.relation_index_child_count("node_creds", "c1") == 0,
+          "rejected write left no posting for the VALID element either - "
+          "no partial index state from a rejected write");
+    check(store.relation_index_child_count("node_creds", "c2") == 0,
+          "and none for the missing one");
+
+    // With every element resolvable, the write goes through and every
+    // element gets its posting.
+    Document c2; c2.id = "c2"; c2.collection = "credentials";
+    c2.set_data({{"name", "the missing one, now created"}});
+    store.put("credentials", "c2", c2);
+    threw = false;
+    try {
+        store.put("nodes", "n1", n);
+    } catch (const MissingParentReference&) {
+        threw = true;
+    }
+    check(!threw, "once every element resolves, the write succeeds");
+    check(store.relation_index_child_count("node_creds", "c1") == 1, "posting for c1");
+    check(store.relation_index_child_count("node_creds", "c2") == 1, "posting for c2");
+}
+
+// An update that does not touch the reference field is not re-validated,
+// even if the previously-written reference has since gone dangling (e.g. the
+// parent was removed out from under a no_action/validateOnWrite=false-era
+// row, or validateOnWrite was turned on after the fact). Only NEWLY
+// introduced references (`to_add`) are checked - see maintainRelations'
+// comment for why: an unchanged reference already existed (or was already
+// dangling) before this write, and this task closes the race for writes
+// that introduce a reference, not for the mere fact that time has passed
+// since one was previously accepted.
+void test_validate_on_write_unrelated_update_not_rechecked() {
+    TmpEnv t("validate-unrelated-update");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions",
+                        {{"exec_wf", "workflowId", "workflows", true}});
+
+    Document w; w.id = "wf-1"; w.collection = "workflows";
+    w.set_data({{"name", "real"}});
+    store.put("workflows", "wf-1", w);
+
+    Document d1; d1.id = "e1"; d1.collection = "executions";
+    d1.set_data({{"workflowId", "wf-1"}, {"status", "running"}});
+    store.put("executions", "e1", d1);   // accepted: parent exists
+
+    store.del("workflows", "wf-1");      // parent now gone; reference dangles
+
+    // Rewrite e1 touching only `status` - workflowId is unchanged, so this
+    // must NOT re-validate it and must NOT throw.
+    Document d2; d2.id = "e1"; d2.collection = "executions";
+    d2.set_data({{"workflowId", "wf-1"}, {"status", "completed"}});
+    bool threw = false;
+    try {
+        store.put("executions", "e1", d2);
+    } catch (const MissingParentReference&) {
+        threw = true;
+    }
+    check(!threw, "an update that leaves the reference field unchanged is not re-validated");
+    check(store.get("executions", "e1")->data()["status"] == "completed",
+          "the update itself still applied");
+}
+
+// v2.11.0 T13 round 3 (review finding) — rollback of an ALREADY-APPLIED
+// index mutation within the same write, not merely "the mutation never
+// happened because validation ran first."
+//
+// round 2's version of this test made `_relidx_wf_rel` itself only exist
+// inside the SAME aborted transaction (nothing had committed a wf_rel
+// posting beforehand), so relation_index_child_count("wf_rel", "wf-1")
+// still returned 0 through the "sub-db was never created"
+// (`if (!dbi_opt) return 0;`) shortcut regardless of whether the abort
+// rolled anything back - the exact case the comment claimed was
+// unavailable. Fixed by committing a SIBLING child (e0) first, in its own,
+// separate, successful write: `_relidx_wf_rel` is a real, already-existing
+// sub-db holding one committed posting (from e0) BEFORE the rejected write
+// (e1) runs. `executions` declares TWO relations: wf_rel
+// (validateOnWrite=false) and owner_rel (validateOnWrite=true). e1's write
+// applies wf_rel's mdb_put for real - into that already-existing sub-db -
+// before owner_rel's check runs and rejects. The assertion that matters is
+// that the count STAYS AT 1 (e0's), not 2: reading 2 would mean e1's
+// already-applied wf_rel mutation survived the abort. Because the sub-db
+// demonstrably existed beforehand (asserted directly, see the sanity check
+// below), the nullopt shortcut is not in play here, and the assertion
+// actually discriminates a rollback from a no-op.
+void test_validate_on_write_rolls_back_an_already_applied_sibling_relation() {
+    TmpEnv t("validate-rollback-sibling");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions", {
+        {"wf_rel", "workflowId", "workflows", false},   // validateOnWrite=false
+        {"owner_rel", "ownerId", "users", true},         // validateOnWrite=true
+    });
+
+    Document w; w.id = "wf-1"; w.collection = "workflows";
+    w.set_data({{"name", "real workflow"}});
+    store.put("workflows", "wf-1", w);
+
+    Document uReal; uReal.id = "u-real"; uReal.collection = "users";
+    uReal.set_data({{"name", "a real user"}});
+    store.put("users", "u-real", uReal);
+    // Deliberately no "users/u-ghost" - owner_rel's parent for e1 never exists.
+
+    // Commit a sibling child FIRST, in its own successful write, so
+    // `_relidx_wf_rel` is a real, already-existing sub-db with one committed
+    // posting before the rejected write below ever runs.
+    Document e0; e0.id = "e0"; e0.collection = "executions";
+    e0.set_data({{"workflowId", "wf-1"}, {"ownerId", "u-real"}});
+    store.put("executions", "e0", e0);
+    check(store.relation_index_child_count("wf_rel", "wf-1") == 1,
+          "sanity: wf_rel's sub-db already exists and holds e0's committed posting");
+
+    Document e; e.id = "e1"; e.collection = "executions";
+    e.set_data({{"workflowId", "wf-1"}, {"ownerId", "u-ghost"}});
+    bool threw = false;
+    try {
+        store.put("executions", "e1", e);
+    } catch (const MissingParentReference&) {
+        threw = true;
+    }
+    check(threw, "owner_rel's missing parent rejects the write");
+    check(!store.get("executions", "e1").has_value(),
+          "the document itself was rolled back");
+    check(store.relation_index_child_count("owner_rel", "u-ghost") == 0,
+          "owner_rel (the relation that rejected) has no posting for the ghost id");
+    check(store.relation_index_child_count("wf_rel", "wf-1") == 1,
+          "wf_rel's count STAYS AT 1 (e0's) rather than becoming 2 - e1's "
+          "already-applied mdb_put into this REAL, already-existing sub-db "
+          "was rolled back by the same transaction abort that rejected "
+          "owner_rel. A count of 2 here would mean the sibling relation's "
+          "mutation survived the abort.");
+
+    // Once owner_rel's parent exists, the identical write succeeds and BOTH
+    // relations end up with their postings - e0's plus e1's.
+    Document u; u.id = "u-ghost"; u.collection = "users";
+    u.set_data({{"name", "real user, now created"}});
+    store.put("users", "u-ghost", u);
+    threw = false;
+    try {
+        store.put("executions", "e1", e);
+    } catch (const MissingParentReference&) {
+        threw = true;
+    }
+    check(!threw, "once owner_rel's parent exists too, the write succeeds");
+    check(store.relation_index_child_count("wf_rel", "wf-1") == 2,
+          "wf_rel now has BOTH e0's (pre-existing) and e1's (just landed) postings");
+    check(store.relation_index_child_count("owner_rel", "u-ghost") == 1, "owner_rel posted for e1");
+}
+
+// v2.11.0 T13 round 2 (review finding 1) — relationsEnforced is the
+// documented escape hatch (config/collection_config_manager.hpp) for "a
+// bulk import, or a collection under write pressure." Before this, the
+// only consumer was RelationEnforcer::canDelete (restrict/no_action on
+// delete); validate_on_write did not read it at all, so the only way to
+// stop a validate_on_write rejection was to drop and re-declare the
+// relation without validateOnWrite - not what the switch is for. Disabling
+// enforcement must also let a bulk-import-shaped write through even though
+// its parent is not loaded yet.
+void test_validate_on_write_relations_enforced_false_is_the_escape_hatch() {
+    TmpEnv t("validate-enforced-off");
+    LmdbDocumentStore store(t.env);
+    // relationsEnforced=false alongside validateOnWrite=true - the exact
+    // combination an operator reaches for mid-bulk-import.
+    store.set_relations("executions",
+                        {{"exec_wf", "workflowId", "workflows", true, false}});
+
+    Document d; d.id = "e1"; d.collection = "executions";
+    d.set_data({{"workflowId", "wf-ghost"}});
+    bool threw = false;
+    try {
+        store.put("executions", "e1", d);
+    } catch (const MissingParentReference&) {
+        threw = true;
+    }
+    check(!threw, "relationsEnforced=false lets a validateOnWrite=true write through");
+    check(store.get("executions", "e1").has_value(),
+          "the escape-hatch write actually landed");
+
+    // Re-enabling enforcement does not retroactively touch what was already
+    // written (documented behaviour, mirrors canDelete's own log message) -
+    // but a NEW write with a missing parent is rejected again.
+    store.set_relations("executions",
+                        {{"exec_wf", "workflowId", "workflows", true, true}});
+    Document d2; d2.id = "e2"; d2.collection = "executions";
+    d2.set_data({{"workflowId", "wf-ghost-2"}});
+    threw = false;
+    try {
+        store.put("executions", "e2", d2);
+    } catch (const MissingParentReference&) {
+        threw = true;
+    }
+    check(threw, "re-enabling enforcement rejects a new write with a missing parent");
+    check(store.get("executions", "e1").has_value(),
+          "the earlier escape-hatch write was not retroactively undone");
+}
+
+// v2.11.0 T13 — demonstrates validation answers against write-time state,
+// not a stale earlier observation.
+//
+// Models one honest slice of the interleaving the plan describes: something
+// (an application-level pre-check, or the old restrict path's own read)
+// observes the parent present, and only AFTER that does the parent get
+// deleted - in its own committed transaction - before the child's write
+// happens. A check that trusted the earlier observation would let the
+// child insert through anyway.
+//
+// What this test does NOT establish: it does NOT distinguish "the mdb_get
+// runs inside the child's own write transaction" (the actual fix - see
+// maintainRelations, before the child's own mdb_put, same transaction the
+// caller commits) from "the mdb_get runs in a separate read transaction
+// opened immediately before the child's write transaction" (a narrowed
+// window, not a closed one). This test's sequence - get, then a
+// committed del, then put - would reject identically either way, because
+// the del is fully committed before either kind of check would run. That
+// placement is inside the transaction, not merely adjacent to it, is
+// established by reading the code (document_store_lmdb.cpp: the mdb_get is
+// at maintainRelations, called from put() before put()'s own mdb_put,
+// against the same wtxn the caller commits), not by this test.
+//
+// What this test DOES show: the validation's answer tracks the parent's
+// state as of when the check actually runs, not whatever an earlier,
+// separate read happened to observe - which is the necessary condition for
+// the fix to work at all, even though it is not sufficient to prove
+// placement by itself. It is deliberately single-threaded and
+// deterministic, not a real multi-thread stress test: LMDB is single-writer
+// (see try_open_for_read's file comment and document_store_lmdb.cpp's env
+// setup), so under real concurrency the only thing that can vary is
+// transaction ORDER, never interleaving within a transaction - serialising
+// "the delete's transaction commits, then the child write's transaction
+// begins" in program order is the deterministic equivalent of that
+// ordering, which is as much of the race as a single-process test can
+// exercise.
+void test_validate_on_write_closes_the_stale_check_race() {
+    TmpEnv t("validate-race");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions",
+                        {{"exec_wf", "workflowId", "workflows", true}});
+
+    Document w; w.id = "wf-1"; w.collection = "workflows";
+    w.set_data({{"name", "about to be deleted"}});
+    store.put("workflows", "wf-1", w);
+
+    // The "stale check": some caller observes the parent present. This is
+    // exactly what a restrict check (or an application's own pre-flight
+    // lookup) does - a READ, complete and finished, before the write it is
+    // meant to gate.
+    check(store.get("workflows", "wf-1").has_value(),
+          "pre-check observes the parent present");
+
+    // The parent vanishes AFTER that check returned, in its own committed
+    // transaction - the race window a stale check cannot see across.
+    check(store.del("workflows", "wf-1"), "parent deleted after the check ran");
+
+    // The child write's OWN transaction begins only now, strictly after the
+    // delete's commit (LMDB's single-writer serialisation). If validation
+    // used the stale check's answer (or any state cached before this point)
+    // it would wrongly accept. It must instead re-read the parent's current
+    // state inside its own transaction and reject.
+    Document d; d.id = "e1"; d.collection = "executions";
+    d.set_data({{"workflowId", "wf-1"}});
+    bool threw = false;
+    try {
+        store.put("executions", "e1", d);
+    } catch (const MissingParentReference&) {
+        threw = true;
+    }
+    check(threw, "the child write is rejected against write-time state, "
+                "despite an earlier check having observed the parent present");
+    check(!store.get("executions", "e1").has_value(),
+          "no trace of the write the stale check would have allowed");
+}
+
+}  // namespace
+
+
+// =========================================================================
+// v2.11.0 final review — the post-replay re-mirror pass runs AFTER arming
+// (finding 4) and its window is bounded by bytes as well as count (finding 3).
+// =========================================================================
+
+// FINDING 4, the consequential half: the pass writes documents through
+// LmdbDocumentStore::put(), which maintains every declared secondary index
+// INSIDE the document's own write transaction by reading
+// indexed_fields(collection) live. While the pass ran inside recover() - line
+// ~88 of DatabaseService::initialize(), where applyIndexDeclarations() is line
+// ~208 - that map was still EMPTY, so the pass moved the document forward and
+// left its postings behind. An index-served query then returns the row under
+// its OLD value and misses it under its new one: silently wrong rows, which is
+// exactly what putting maintainIndexes() inside the transaction exists to
+// prevent, and which bites ANY install with a v2.9 index declared, relations
+// or not.
+//
+// The simulated restart clears the in-process declaration (a real restart
+// rebuilds it from `_collection_meta`) and re-arms it only AFTER recover(),
+// which is the production ordering. Moving the runPendingRemirror() call above
+// the re-arm reproduces the bug.
+void test_remirror_maintains_the_secondary_index() {
+    TmpEnv t("rel-remirror-index");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("rel-remirror-index-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    store.set_indexed_fields("widgets", {"status"});
+
+    Document d; d.id = "w1"; d.collection = "widgets";
+    d.set_data({{"status", "queued"}});
+    store.put("widgets", "w1", d);            // LMDB + posting under "queued"
+    p.pm.logInsert("default:widgets", d);
+
+    {
+        auto pre = store.index_lookup_eq("widgets", "status", nlohmann::json("queued"));
+        check(pre.has_value() && pre->size() == 1, "the pre-crash posting exists under 'queued'");
+    }
+
+    // The crash window: a WAL UPDATE whose LMDB write never ran.
+    d.set_data({{"status", "done"}});
+    p.pm.logUpdate("default:widgets", d);
+    p.pm.stop();
+
+    // "Restart". The declaration is in-process state, so clear it: a real
+    // restart starts with an empty map and rebuilds it in
+    // applyIndexDeclarations().
+    store.set_indexed_fields("widgets", {});
+
+    MemoryStore freshStore(MemoryStore::Config{});
+    freshStore.start();
+    std::atomic<bool> mirrorHealthy{true};
+    std::atomic<uint64_t> mirrorDrift{0};
+    freshStore.setDocumentStoreMirror(
+        [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
+        &mirrorHealthy, &mirrorDrift);
+
+    PersistenceManager::Config cfg2;
+    cfg2.dataDir = p.path;
+    PersistenceManager pm2(cfg2);
+    auto outcome = pm2.recover(freshStore);
+    check(outcome.kind != smartbotic::database::RecoveryOutcome::Kind::Failed, "recovery completed");
+    check(outcome.pendingRemirror.size() == 1,
+          "recover() COLLECTED the row instead of re-mirroring it itself");
+
+    // applyIndexDeclarations()' place in initialize(): before the pass.
+    store.set_indexed_fields("widgets", {"status"});
+    pm2.runPendingRemirror(freshStore, outcome);
+    check(outcome.updatesRemirroredAfterReplay == 1, "the row was re-mirrored");
+
+    // The document converged (this half worked before the fix too).
+    auto lmdb = store.get("widgets", "w1");
+    check(lmdb.has_value() && lmdb->data()["status"] == "done",
+          "LMDB holds the replayed value");
+
+    // LOAD-BEARING: the INDEX converged with it. Both directions matter - a
+    // stale posting under the old value is a wrong row returned, and a missing
+    // posting under the new value is a row silently omitted.
+    auto stale = store.index_lookup_eq("widgets", "status", nlohmann::json("queued"));
+    check(stale.has_value() && stale->empty(),
+          "no posting left under the OLD value - an index-served query cannot "
+          "return this row as if it were still queued");
+    auto fresh = store.index_lookup_eq("widgets", "status", nlohmann::json("done"));
+    check(fresh.has_value() && fresh->size() == 1 && (*fresh)[0] == "w1",
+          "the row IS found under its new value - the posting was maintained by "
+          "the pass, which is only possible because the index was armed first");
+
+    freshStore.stop();
+}
+
+// FINDING 4, the other half: with the pass running after arming, a
+// UniqueViolation from it is REACHABLE, which is what makes phase 3's single
+// retry earn its place. It could not fire at all while the pass ran inside
+// recover() (unique_fields_ was empty), so the retry, its "moved unique value"
+// rationale and persistence_manager.cpp's insertion-ORDERED comment were all
+// dead code justified by dead reasoning.
+//
+// The scenario, in WAL order: a NEW row b1 takes the email a1 currently holds,
+// and only then does a1 move off it. Re-mirroring in WAL order therefore tries
+// b1 first, against a1's not-yet-updated posting - a genuine conflict - and it
+// clears only once a1 has been re-mirrored. Replacing the retry with a bare
+// noteFailure() makes this test fail.
+void test_moved_unique_value_is_resolved_by_the_retry_pass() {
+    TmpEnv t("rel-remirror-unique");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("rel-remirror-unique-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    store.set_indexed_fields("people", {"email"});
+    store.set_unique_fields("people", {"email"});
+
+    // Pre-crash LMDB state: a1 holds the value, with its posting. Written
+    // straight to LMDB and deliberately NOT to the WAL - it is the state a
+    // snapshot would have carried, and keeping it out of the WAL is what puts
+    // b1 first in the re-mirror order.
+    Document a; a.id = "a1"; a.collection = "people";
+    a.set_data({{"email", "shared@example.com"}});
+    store.put("people", "a1", a);
+
+    // WAL order: b1 claims the value FIRST, a1 vacates it second. That order is
+    // what creates the transient conflict, and it is the order recover()
+    // preserves (insertion-ORDERED, deliberately - see persistence_manager.cpp).
+    Document b; b.id = "b1"; b.collection = "people";
+    b.set_data({{"email", "shared@example.com"}});
+    p.pm.logInsert("default:people", b);
+
+    a.set_data({{"email", "moved@example.com"}});
+    p.pm.logInsert("default:people", a);
+    p.pm.stop();
+
+    store.set_unique_fields("people", {});
+    store.set_indexed_fields("people", {});
+
+    MemoryStore freshStore(MemoryStore::Config{});
+    freshStore.start();
+    std::atomic<bool> mirrorHealthy{true};
+    std::atomic<uint64_t> mirrorDrift{0};
+    freshStore.setDocumentStoreMirror(
+        [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
+        &mirrorHealthy, &mirrorDrift);
+
+    PersistenceManager::Config cfg2;
+    cfg2.dataDir = p.path;
+    PersistenceManager pm2(cfg2);
+    auto outcome = pm2.recover(freshStore);
+    check(outcome.pendingRemirror.size() == 2, "both ids collected, in WAL order");
+    check(outcome.pendingRemirror[0].second == "b1",
+          "b1 comes FIRST - the order is what creates the transient conflict");
+
+    store.set_indexed_fields("people", {"email"});
+    store.set_unique_fields("people", {"email"});
+    // chunkSize=1 so each row gets its own transaction: with a batch of two the
+    // whole chunk would abort and be retried row by row anyway, but forcing the
+    // per-row shape makes the deferral the pass's own, not put_batch's.
+    auto res = freshStore.remirrorDocuments(outcome.pendingRemirror, /*chunkSize=*/1);
+
+    check(res.remirrored == 2,
+          "BOTH rows converged - b1 was deferred on the conflict and committed by "
+          "the retry once a1 had vacated the value");
+    check(res.failed == 0, "nothing was written off as failed");
+    check(mirrorDrift.load() == 0, "no drift bumped - a resolved conflict is not drift");
+
+    auto la = store.get("people", "a1");
+    auto lb = store.get("people", "b1");
+    check(la.has_value() && la->data()["email"] == "moved@example.com", "a1 moved off the value");
+    check(lb.has_value() && lb->data()["email"] == "shared@example.com", "b1 holds the value now");
+    auto who = store.index_lookup_eq("people", "email", nlohmann::json("shared@example.com"));
+    check(who.has_value() && who->size() == 1 && (*who)[0] == "b1",
+          "exactly one row holds the unique value, and it is the right one");
+
+    freshStore.stop();
+}
+
+// FINDING 3: the window is bounded by BYTES as well as by count, and a window
+// boundary is crossed at all (no prior test did - every fixture fitted inside
+// one chunk of 256, so the streaming structure round 5 introduced was
+// unpinned).
+//
+// chunkBytes=1 forces one row per window, which is the extreme the byte budget
+// can reach; every row must still converge, and the loop must terminate. A
+// document larger than the whole budget forms a window of one rather than
+// stalling, which is what guarantees progress.
+void test_remirror_windows_are_bounded_by_bytes_and_by_count() {
+    TmpEnv t("rel-remirror-window");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("rel-remirror-window-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    constexpr int kRows = 7;
+    for (int i = 0; i < kRows; ++i) {
+        Document d; d.id = "w" + std::to_string(i); d.collection = "widgets";
+        d.set_data({{"n", i}, {"pad", std::string(4096, 'x')}});
+        p.pm.logInsert("default:widgets", d);
+    }
+    p.pm.stop();
+
+    MemoryStore freshStore(MemoryStore::Config{});
+    freshStore.start();
+    std::atomic<bool> mirrorHealthy{true};
+    std::atomic<uint64_t> mirrorDrift{0};
+    freshStore.setDocumentStoreMirror(
+        [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
+        &mirrorHealthy, &mirrorDrift);
+
+    PersistenceManager::Config cfg2;
+    cfg2.dataDir = p.path;
+    PersistenceManager pm2(cfg2);
+    auto outcome = pm2.recover(freshStore);
+    check(outcome.pendingRemirror.size() == kRows, "all rows collected");
+
+    // Byte budget of 1 with a generous count cap: the BYTES must be what closes
+    // each window. Before this fix chunkBytes did not exist and the count cap
+    // alone put all seven in one window - on this repo's ~2.9 MB documents, 256
+    // of them is ~750 MB resident plus as much again in LMDB dirty pages, in
+    // the happy path, on the boot path.
+    //
+    // The OBSERVABLE is the number of committed LMDB transactions, read from
+    // the env's own last-txnid. Row counts alone cannot discriminate here (all
+    // seven rows converge either way, via put_batch instead of put), so
+    // counting commits is what actually pins that the window closed per row.
+    auto lastTxnId = [&]() -> uint64_t {
+        MDB_envinfo info;
+        mdb_env_info(t.env.raw(), &info);
+        return static_cast<uint64_t>(info.me_last_txnid);
+    };
+    const uint64_t txnBefore = lastTxnId();
+    auto res = freshStore.remirrorDocuments(outcome.pendingRemirror,
+                                           /*chunkSize=*/1024, /*chunkBytes=*/1);
+    const uint64_t commits = lastTxnId() - txnBefore;
+    check(res.remirrored == kRows,
+          "every row converged with a byte budget smaller than one document - "
+          "an oversized row forms a window of one rather than stalling");
+    check(res.failed == 0, "no failures");
+    check(commits >= kRows,
+          "the BYTE budget closed each window: one commit per row, not one "
+          "commit for all seven (which is what the count cap alone produced, "
+          "and what makes 256 rows of 2.9 MB a ~750 MB resident window)");
+    for (int i = 0; i < kRows; ++i) {
+        auto got = store.get("widgets", "w" + std::to_string(i));
+        check(got.has_value() && got->data()["n"] == i, "row survived its own window");
+    }
+
+    // And the count cap still closes a window on its own: two rows, chunkSize=1.
+    LmdbDocumentStore store2(t.env);
+    freshStore.setDocumentStoreMirror(
+        [&store2](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store2; },
+        &mirrorHealthy, &mirrorDrift);
+    auto res2 = freshStore.remirrorDocuments(outcome.pendingRemirror,
+                                             /*chunkSize=*/1,
+                                             /*chunkBytes=*/64ull * 1024 * 1024);
+    check(res2.remirrored == kRows, "count-bounded windows cover every row too");
+    check(res2.failed == 0, "no failures crossing count-bounded window boundaries");
+
+    freshStore.stop();
+}
+
+// FINDING 6: the cascade path builds every updatedDoc from the LMDB copy and
+// pushes it into MemoryStore. While the mirror is unhealthy or drifted LMDB may
+// be BEHIND - that is the entire reason reads fall back to MemoryStore - so the
+// cascade would overwrite MemoryStore's fresher child body with a stale one.
+// Real data loss, on a state this codebase treats as routine.
+void test_cascade_refuses_while_the_mirror_is_unhealthy_or_drifted() {
+    TmpEnv t("rel-cascade-gate");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("rel-cascade-gate-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+
+    std::atomic<bool> healthy{true};
+    std::atomic<uint64_t> drift{0};
+    mstore.setDocumentStoreMirror(
+        [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
+        &healthy, &drift);
+
+    RelationInfo rel;
+    rel.name = "default:exec_wf";
+    rel.child = "default:executions";
+    rel.childField = "workflowId";
+    rel.parent = "default:workflows";
+    rel.onDelete = OnDelete::Cascade;
+    std::string mgrErr;
+    check(rm.createRelation(rel, mgrErr), "declared the cascade relation");
+    store.set_relations("executions",
+                        {RelationRef{"exec_wf", "workflowId", "workflows", false, true}});
+
+    Document parent; parent.id = "wf-1"; parent.collection = "workflows";
+    parent.set_data({{"name", "wf-1"}});
+    store.put("workflows", "wf-1", parent);
+    mstore.loadDocument("default:workflows", parent);
+
+    Document child; child.id = "ex-1"; child.collection = "executions";
+    child.set_data({{"workflowId", "wf-1"}});
+    store.put("executions", "ex-1", child);
+    mstore.loadDocument("default:executions", child);
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 1, "one child indexed");
+
+    auto expectRefusal = [&](const char* what) {
+        bool refused = false;
+        std::string msg;
+        try {
+            executeCascade(rm, store, p.pm, mstore, cfgManager, "default:workflows", "wf-1", nullptr);
+        } catch (const CascadeBlocked& e) {
+            refused = true;
+            msg = e.what();
+        } catch (const std::exception& e) {
+            msg = std::string("wrong exception type: ") + e.what();
+        }
+        check(refused, what);
+        check(msg.find("LMDB mirror is") != std::string::npos,
+              "the refusal tells the operator WHY, not just that it failed");
+        // Nothing may have been touched anywhere: the gate runs before
+        // planCascade(), so before writeCascadeWal() and before any LMDB commit.
+        check(store.get("workflows", "wf-1").has_value(), "parent untouched in LMDB");
+        check(store.get("executions", "ex-1").has_value(), "child untouched in LMDB");
+        check(mstore.get("default:workflows", "wf-1").has_value(), "parent untouched in MemoryStore");
+        check(mstore.get("default:executions", "ex-1").has_value(), "child untouched in MemoryStore");
+    };
+
+    healthy.store(false);
+    expectRefusal("cascade refused while the mirror is UNHEALTHY");
+
+    healthy.store(true);
+    drift.store(1);
+    expectRefusal("cascade refused while the mirror has DRIFTED (health alone is not enough - "
+                  "the re-mirror pass bumps drift without flipping health, by design)");
+
+    // And with a clean mirror it proceeds, so the gate is not a blanket refusal.
+    drift.store(0);
+    bool parentExisted = false;
+    try {
+        parentExisted = executeCascade(rm, store, p.pm, mstore, cfgManager,
+                                       "default:workflows", "wf-1", nullptr);
+    } catch (const std::exception& e) {
+        check(false, "cascade threw with a healthy mirror");
+        std::cerr << "  (" << e.what() << ")\n";
+    }
+    check(parentExisted, "the cascade ran once the mirror was healthy and undrifted");
+    check(!store.get("executions", "ex-1").has_value(), "child cascaded away");
+    check(!store.get("workflows", "wf-1").has_value(), "parent deleted");
+
+    mstore.stop();
+}
+
+
+// FINDING 5: the gate that keeps T11's per-write Document deep copy off the
+// write path of every install that declares no rejecting constraint.
+//
+// Measured (this repo's shapes, same machine): the copy alone is 1.8 us at
+// 4 KB, 7.1 us at 64 KB and 364 us at 2.9 MB, all of it inside the
+// collection's unique_lock and all of it a fresh yyjson_mut_doc allocation.
+// The gate's own cost in the common case is one relaxed atomic load.
+//
+// The truth table is what matters, and one row is easy to get wrong: a
+// relation WITHOUT validateOnWrite maintains its reverse index on every put
+// but can never THROW, so it must not force a snapshot.
+void test_can_reject_writes_gates_the_undo_snapshot() {
+    TmpEnv t("rel-gate-undo");
+    LmdbDocumentStore store(t.env);
+
+    check(!store.can_reject_writes("widgets"),
+          "nothing declared anywhere: no snapshot needed (the case that used to "
+          "pay for the copy on every install)");
+
+    store.set_indexed_fields("widgets", {"status"});
+    check(!store.can_reject_writes("widgets"),
+          "a plain secondary index cannot reject a write, so still no snapshot");
+
+    store.set_relations("widgets",
+                        {RelationRef{"w_rel", "parentId", "parents", /*validateOnWrite=*/false,
+                                     /*relationsEnforced=*/true}});
+    check(!store.can_reject_writes("widgets"),
+          "a relation WITHOUT validate_on_write maintains the reverse index but "
+          "never throws - no snapshot");
+
+    store.set_relations("widgets",
+                        {RelationRef{"w_rel", "parentId", "parents", /*validateOnWrite=*/true,
+                                     /*relationsEnforced=*/true}});
+    check(store.can_reject_writes("widgets"),
+          "validate_on_write CAN reject (MissingParentReference) - snapshot needed");
+    check(!store.can_reject_writes("others"),
+          "and it is per collection, not process-wide: an unrelated collection "
+          "still pays nothing");
+
+    store.set_relations("widgets", {});
+    check(!store.can_reject_writes("widgets"), "dropping the relation drops the need");
+
+    store.set_unique_fields("widgets", {"email"});
+    check(store.can_reject_writes("widgets"),
+          "a unique field CAN reject (UniqueViolation) - snapshot needed");
+    check(!store.can_reject_writes("others"), "still per collection");
+    store.set_unique_fields("widgets", {});
+    check(!store.can_reject_writes("widgets"), "and dropping it drops the need again");
+
+    // MemoryStore's view of the same question, which is what the write paths
+    // actually call. Fails SAFE when no mirror is wired.
+    MemoryStore ms(MemoryStore::Config{});
+    ms.start();
+    check(ms.needsUndoSnapshot("default:widgets"),
+          "with NO mirror wired the answer is 'take the snapshot' - fail safe");
+    std::atomic<bool> healthy{true};
+    std::atomic<uint64_t> drift{0};
+    ms.setDocumentStoreMirror(
+        [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
+        &healthy, &drift);
+    check(!ms.needsUndoSnapshot("default:widgets"),
+          "wired, with nothing declared: no snapshot");
+    store.set_unique_fields("widgets", {"email"});
+    check(ms.needsUndoSnapshot("default:widgets"),
+          "wired, with a unique field declared: snapshot");
+    check(ms.needsUndoSnapshot("this is not a valid:collection:name"),
+          "an unparseable name fails safe too");
+    ms.stop();
+}
+
+
+// =========================================================================
+// v2.11.0 close-out — TTL EXPIRY MUST HANDLE CHILDREN LIKE A MANUAL DELETE.
+//
+// Operator's ruling. Before this, MemoryStore::expireDocuments() erased the
+// document and mirrored a DELETE with no relation enforcement whatsoever, so a
+// restrict-protected parent carrying a TTL silently vanished and orphaned every
+// child - strictly the worst of the three available behaviours.
+//
+// Every test below drives the REAL sweeper (MemoryStore::expireDocuments()) with
+// the REAL decision function (relations/relation_cascade.cpp's
+// ttlExpiryDecision(), which DatabaseService binds as the production hook), so
+// what is under test is the shipped path and not a re-statement of it.
+// =========================================================================
+
+// Residency probe. MemoryStore::get() deliberately hides an EXPIRED document
+// (Document::isExpired()), so "is it still there" cannot be asked with get()
+// when the whole point is that the document is past its TTL and still present.
+// getAllDocuments() does not filter.
+bool residentInMemory(MemoryStore& mstore, const std::string& collection,
+                      const std::string& id) {
+    for (const auto& d : mstore.getAllDocuments(collection)) {
+        if (d.id == id) return true;
+    }
+    return false;
+}
+
+// Install the production decision function as the sweeper's hook. Exactly what
+// DatabaseService::ttlExpiryRelationDecision() does, minus the per-project store
+// resolution (this fixture has one store) and the replication notifier.
+void installTtlHook(MemoryStore& mstore, RelationManager& rm, LmdbDocumentStore& store,
+                    PersistenceManager& pm, CollectionConfigManager& cfgManager,
+                    const smartbotic::database::CascadeNotifyFn& notify = nullptr) {
+    mstore.setTtlExpiryRelationHook(
+        [&rm, &store, &pm, &mstore, &cfgManager, notify](const std::string& coll,
+                                                          const std::string& id,
+                                                          bool firstAttempt) {
+            return ttlExpiryDecision(rm, store, pm, mstore, cfgManager, coll, id, notify,
+                                     firstAttempt);
+        });
+}
+
+// A parent carrying a TTL, plus one child, plus a declared relation. Returns
+// nothing; the caller owns every object so the fixtures stay explicit (this
+// file's established style).
+void seedTtlParentAndChild(MemoryStore& mstore, LmdbDocumentStore& store,
+                           RelationManager& rm, OnDelete policy,
+                           const std::string& childField = "workflowId",
+                           const nlohmann::json& childData = {{"workflowId", "wf-1"}}) {
+    RelationInfo rel;
+    rel.name = "default:exec_wf";
+    rel.child = "default:executions";
+    rel.childField = childField;
+    rel.parent = "default:workflows";
+    rel.onDelete = policy;
+    std::string mgrErr;
+    check(rm.createRelation(rel, mgrErr), "declared the relation under test");
+    store.set_relations("executions",
+                        {RelationRef{"exec_wf", childField, "workflows", false, true}});
+
+    // ⚠ expiresAt = 1 (one millisecond after the epoch) rather than "now minus
+    // something": expireDocuments() compares against currentTimeMs(), so 1 is
+    // unambiguously expired on any clock and the test cannot race the sweep.
+    Document parent;
+    parent.id = "wf-1";
+    parent.collection = "workflows";
+    parent.set_data({{"name", "wf-1"}});
+    parent.expiresAt = 1;
+    store.put("workflows", "wf-1", parent);
+    mstore.loadDocument("default:workflows", parent);
+
+    Document child;
+    child.id = "ex-1";
+    child.collection = "executions";
+    child.set_data(childData);
+    store.put("executions", "ex-1", child);
+    mstore.loadDocument("default:executions", child);
+}
+
+// restrict: a MANUAL delete fails, so the expiry must not happen. The document
+// is left in place, its expiry stays armed, and the block is signalled.
+//
+// Would fail before the fix on its first assertion: the pre-v2.11.0 sweeper
+// returned 1 and erased the parent, leaving ex-1 pointing at nothing.
+void test_ttl_restrict_blocks_the_expiry() {
+    TmpEnv t("ttl-restrict");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("ttl-restrict-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    // Small retry cadence so the test does not have to run 60 sweeps.
+    MemoryStore::Config cfg;
+    cfg.ttlBlockedRetrySweeps = 3;
+    MemoryStore mstore(cfg);
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+
+    seedTtlParentAndChild(mstore, store, rm, OnDelete::Restrict);
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 1, "one child indexed");
+    installTtlHook(mstore, rm, store, p.pm, cfgManager);
+
+    const uint64_t expired = mstore.expireDocuments();
+    check(expired == 0, "the restrict-protected parent was NOT expired");
+    check(residentInMemory(mstore, "default:workflows", "wf-1"),
+          "the parent is still in MemoryStore - it OUTLIVES its TTL, deliberately");
+    check(store.get("workflows", "wf-1").has_value(), "the parent is still in LMDB");
+    check(mstore.get("default:executions", "ex-1").has_value(), "the child was not orphaned");
+    check(mstore.getStats().ttlExpiryBlockedByRelation == 1,
+          "the skip is counted, so an operator can see a document is stuck past its TTL");
+    check(mstore.getStats().expiredCount == 0, "and it is not counted as expired");
+
+    check(mstore.ttlBlockedDocumentCount() == 1,
+          "and it is tracked as ONE stuck document - the gauge an operator wants");
+
+    // ⚠ The next sweep must NOT re-examine it (close-out review, finding 2). A
+    // blocked document costs one LMDB read txn + cursor scan + one log line every
+    // time it is examined, and at the 1s default that is a permanent flood plus a
+    // permanent index-lookup load. It is skipped for free until its retry is due.
+    const uint64_t again = mstore.expireDocuments();
+    check(again == 0, "still not expired on the next sweep");
+    check(mstore.getStats().ttlExpiryBlockedByRelation == 1,
+          "and the relation was NOT re-consulted - no second refusal event");
+    check(residentInMemory(mstore, "default:workflows", "wf-1"), "the parent is still there");
+
+    // The expiry must stay ARMED, so once the retry comes due AND the child is
+    // gone, it expires - the block is the relation's, not a permanent quarantine.
+    store.del("executions", "ex-1");
+    check(mstore.remove("default:executions", "ex-1"), "child removed");
+    uint64_t total = 0;
+    for (uint32_t i = 0; i < cfg.ttlBlockedRetrySweeps + 1; ++i) {
+        total += mstore.expireDocuments();
+    }
+    check(total == 1, "when the retry comes due, the parent finally expires");
+    check(!residentInMemory(mstore, "default:workflows", "wf-1"), "the parent is gone now");
+    check(mstore.ttlBlockedDocumentCount() == 0, "and it is no longer counted as stuck");
+
+    p.pm.stop();
+    mstore.stop();
+}
+
+// cascade with a SCALAR reference: children are deleted, exactly as
+// executeCascade does for a manual delete, and the WAL carries every mutation.
+//
+// Would fail before the fix: the old sweeper wrote no WAL entry for ex-1 at all
+// (it only mirrored a DELETE for the parent), so walHasDelete for the child is
+// the assertion that a hand-rolled, LMDB-only sweeper cascade cannot satisfy.
+void test_ttl_cascade_deletes_children_like_a_manual_delete() {
+    TmpEnv t("ttl-cascade");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("ttl-cascade-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+
+    seedTtlParentAndChild(mstore, store, rm, OnDelete::Cascade);
+
+    // The notify wiring: a TTL-driven cascade must drive replication and
+    // Subscribe events per mutation, or a follower never sees the child
+    // deletions and diverges permanently.
+    struct Notification { std::string collection, id; bool hadDoc; smartbotic::database::EventType et; };
+    std::vector<Notification> notifications;
+    auto notify = [&](const std::string& coll, const std::string& id,
+                      const std::optional<Document>& doc, smartbotic::database::EventType et) {
+        notifications.push_back({coll, id, doc.has_value(), et});
+    };
+    installTtlHook(mstore, rm, store, p.pm, cfgManager, notify);
+
+    const uint64_t expired = mstore.expireDocuments();
+    check(expired == 1, "the parent was expired");
+    check(!mstore.get("default:executions", "ex-1").has_value(),
+          "the child was CASCADED, not orphaned");
+    check(!store.get("executions", "ex-1").has_value(), "and it is gone from LMDB too");
+    check(!store.get("workflows", "wf-1").has_value(), "the parent is gone from LMDB");
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 0,
+          "the reverse-index posting went with it");
+    check(notifications.size() == 2,
+          "replication/events fired once per child plus once for the parent");
+
+    // Accounting (close-out review, minor 1): the parent is ONE expiry, and the
+    // one child is ONE delete. The cascade removes the parent through
+    // unloadDocument(), which counts a DELETE, so without compensating for that
+    // the parent was counted as both an expiry and a delete.
+    const auto st = mstore.getStats();
+    check(st.expiredCount == 1, "the parent counted as exactly one expiry");
+    check(st.deleteCount == 1,
+          "and the delete count covers the CHILD only - the parent is not counted twice");
+
+    // The property a hand-rolled sweeper cascade breaks: MemoryStore is rebuilt
+    // from snapshot + WAL, NEVER from LMDB, so a cascade whose child deletions
+    // exist only in LMDB has them RESURRECTED on the next boot.
+    p.pm.stop();
+    auto entries = replayRawWal(p.path);
+    check(walHasDelete(entries, "default:executions", "ex-1"),
+          "the WAL carries the CHILD's delete - without this the next boot resurrects it");
+    check(walHasDelete(entries, "default:workflows", "wf-1"),
+          "the WAL carries the parent's delete");
+
+    mstore.stop();
+}
+
+// set_null with a SCALAR reference: the field is nulled and the child KEPT.
+void test_ttl_set_null_nulls_the_scalar_and_keeps_the_child() {
+    TmpEnv t("ttl-setnull");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("ttl-setnull-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+
+    seedTtlParentAndChild(mstore, store, rm, OnDelete::SetNull);
+    installTtlHook(mstore, rm, store, p.pm, cfgManager);
+
+    const uint64_t expired = mstore.expireDocuments();
+    check(expired == 1, "the parent was expired");
+    auto child = store.get("executions", "ex-1");
+    check(child.has_value(), "the child SURVIVED - set_null keeps the document");
+    if (child) {
+        check(child->data().contains("workflowId") && child->data()["workflowId"].is_null(),
+              "and its reference is null, not deleted and not left dangling");
+    }
+    auto memChild = mstore.get("default:executions", "ex-1");
+    check(memChild.has_value(), "the child survived in MemoryStore too");
+    if (memChild) {
+        check(memChild->data()["workflowId"].is_null(), "with the same nulled reference");
+    }
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 0,
+          "the posting is gone, since the reference no longer names wf-1");
+
+    p.pm.stop();
+    auto entries = replayRawWal(p.path);
+    check(walHasUpdate(entries, "default:executions", "ex-1"),
+          "the child's UPDATE is in the WAL - the mutation survives a restart");
+    mstore.stop();
+}
+
+// An ARRAY-valued reference: the id is PULLED and the document KEPT, under
+// cascade as well as set_null. The place a MySQL mental model actively
+// misleads, and the same rule the request path follows - deleting a node
+// because one of its three credentials expired would be worse than useless.
+void test_ttl_array_reference_pulls_the_id_and_keeps_the_document() {
+    TmpEnv t("ttl-array");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("ttl-array-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+
+    // cascade, deliberately: the array rule must hold under the policy a
+    // MySQL-trained reader expects to delete the row.
+    seedTtlParentAndChild(mstore, store, rm, OnDelete::Cascade, "workflowIds",
+                          {{"workflowIds", {"wf-1", "wf-2"}}});
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 1,
+          "the array element is indexed as a posting");
+    installTtlHook(mstore, rm, store, p.pm, cfgManager);
+
+    const uint64_t expired = mstore.expireDocuments();
+    check(expired == 1, "the parent was expired");
+    auto child = store.get("executions", "ex-1");
+    check(child.has_value(), "the child SURVIVED - an array match never deletes the document");
+    if (child) {
+        const auto& ids = child->data()["workflowIds"];
+        check(ids.is_array() && ids.size() == 1, "exactly one id was pulled");
+        check(ids.is_array() && !ids.empty() && ids[0] == "wf-2",
+              "and it was the expiring parent's id, not the surviving one");
+    }
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 0, "posting for wf-1 gone");
+    check(store.relation_index_child_count("exec_wf", "wf-2") == 1, "posting for wf-2 intact");
+
+    p.pm.stop();
+    mstore.stop();
+}
+
+// no_action: the expiry proceeds and the reference is left dangling, which is
+// what it did before this change and must keep doing.
+void test_ttl_no_action_expires_and_leaves_the_reference_dangling() {
+    TmpEnv t("ttl-noaction");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("ttl-noaction-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+
+    seedTtlParentAndChild(mstore, store, rm, OnDelete::NoAction);
+    installTtlHook(mstore, rm, store, p.pm, cfgManager);
+
+    const uint64_t expired = mstore.expireDocuments();
+    check(expired == 1, "no_action does not block the expiry");
+    check(!mstore.get("default:workflows", "wf-1").has_value(), "the parent expired");
+    auto child = mstore.get("default:executions", "ex-1");
+    check(child.has_value(), "the child is untouched");
+    if (child) {
+        check(child->data()["workflowId"] == "wf-1",
+              "and its reference still names the gone parent - dangling, as declared");
+    }
+    check(mstore.getStats().ttlExpiryBlockedByRelation == 0, "nothing was blocked");
+
+    p.pm.stop();
+    mstore.stop();
+}
+
+// SCOPE: the common case - no relation names this collection as a parent - must
+// behave exactly as it did before, including the LMDB DELETE mirror and the
+// EXPIRE event. This is the regression guard for the two-phase rewrite of
+// expireDocuments(), which is what actually changed for every install.
+void test_ttl_with_no_relations_expires_exactly_as_before() {
+    TmpEnv t("ttl-plain");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("ttl-plain-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+
+    std::atomic<bool> healthy{true};
+    std::atomic<uint64_t> drift{0};
+    mstore.setDocumentStoreMirror(
+        [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
+        &healthy, &drift);
+
+    int expireEvents = 0;
+    mstore.setEventCallback([&](const smartbotic::database::DatabaseEvent& ev) {
+        if (ev.type == smartbotic::database::EventType::EXPIRE) ++expireEvents;
+    });
+
+    Document doc;
+    doc.id = "s-1";
+    doc.collection = "sessions";
+    doc.set_data({{"token", "abc"}});
+    doc.expiresAt = 1;
+    store.put("sessions", "s-1", doc);
+    mstore.loadDocument("default:sessions", doc);
+
+    // Hook installed, with the relation cache EMPTY - the early-return path.
+    installTtlHook(mstore, rm, store, p.pm, cfgManager);
+
+    const uint64_t expired = mstore.expireDocuments();
+    check(expired == 1, "an unrelated document still expires");
+    check(!mstore.get("default:sessions", "s-1").has_value(), "gone from MemoryStore");
+    check(!store.get("sessions", "s-1").has_value(),
+          "and the DELETE was mirrored to LMDB, as before");
+    check(expireEvents == 1, "exactly one EXPIRE event, as before");
+    check(mstore.getStats().expiredCount == 1, "counted as expired");
+    check(mstore.getStats().ttlExpiryBlockedByRelation == 0, "nothing blocked");
+
+    // A document whose TTL is in the future is not touched by any of this.
+    Document later;
+    later.id = "s-2";
+    later.collection = "sessions";
+    later.set_data({{"token", "def"}});
+    later.expiresAt = 1;
+    later.expiresAt = static_cast<uint64_t>(1) << 62;   // far future
+    mstore.loadDocument("default:sessions", later);
+    check(mstore.expireDocuments() == 0, "an unexpired document is left alone");
+    check(mstore.get("default:sessions", "s-2").has_value(), "and is still there");
+
+    p.pm.stop();
+    mstore.stop();
+}
+
+// THE ORDERING PROPERTY, observed rather than asserted from the outside: the
+// cascade's WAL entries are durable BEFORE the LMDB transaction commits.
+//
+// Induced honestly, with the v2.4.4 identity sentinel: the child sub-db's
+// sentinel is overwritten with the wrong name using raw LMDB, so
+// commitCascadeLmdb()'s open_for_write() throws and its WriteTxn aborts
+// UNWRITTEN. Everything before it has already happened. If the sweeper wrote
+// LMDB first (or hand-rolled its own cascade), the WAL would be empty here and
+// the child's deletion would be lost on the next boot.
+void test_ttl_cascade_wal_is_durable_before_the_lmdb_commit() {
+    TmpEnv t("ttl-wal-first");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("ttl-wal-first-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+
+    seedTtlParentAndChild(mstore, store, rm, OnDelete::Cascade);
+    installTtlHook(mstore, rm, store, p.pm, cfgManager);
+
+    // Overwrite the child sub-db's identity sentinel with a name that is not
+    // its own. open_for_write verifies it on every cached-handle reuse.
+    {
+        MDB_txn* txn = nullptr;
+        check(mdb_txn_begin(t.env.raw(), nullptr, 0, &txn) == 0, "tamper txn opened");
+        MDB_dbi dbi = 0;
+        check(mdb_dbi_open(txn, "executions", 0, &dbi) == 0, "child sub-db opened for tampering");
+        const std::string_view sentinel{"\0__subdb_identity__", 19};
+        MDB_val k{sentinel.size(), const_cast<char*>(sentinel.data())};
+        std::string wrong = "not_executions";
+        MDB_val v{wrong.size(), wrong.data()};
+        check(mdb_put(txn, dbi, &k, &v, 0) == 0, "sentinel overwritten");
+        check(mdb_txn_commit(txn) == 0, "tamper txn committed");
+    }
+
+    const uint64_t expired = mstore.expireDocuments();
+    check(expired == 0, "the sweeper did NOT expire the parent when the cascade faulted");
+    check(residentInMemory(mstore, "default:workflows", "wf-1"),
+          "the parent is still in MemoryStore - never expired behind a failed cascade");
+    check(store.get("workflows", "wf-1").has_value(),
+          "and still in LMDB - commitCascadeLmdb's WriteTxn aborted unwritten");
+    check(store.get("executions", "ex-1").has_value(), "the child is still in LMDB too");
+    check(mstore.getStats().ttlExpiryBlockedByRelation == 1, "the skip is counted");
+
+    // ...and yet the WAL already describes the whole cascade. That is the
+    // ordering: WAL fsynced, THEN the LMDB commit attempted.
+    p.pm.stop();
+    auto entries = replayRawWal(p.path);
+    check(walHasDelete(entries, "default:executions", "ex-1"),
+          "the child's DELETE is already durable in the WAL, before any LMDB commit");
+    check(walHasDelete(entries, "default:workflows", "wf-1"),
+          "so is the parent's - the next boot's replay applies the cascade regardless");
+
+    mstore.stop();
+}
+
+
+// =========================================================================
+// v2.11.0 close-out review, FINDING 1 — a concurrent TTL change must not have
+// its document (and its children) deleted by an in-flight sweep.
+//
+// Phase 1 collects candidates and releases every lock; the hook may then DELETE
+// the document and cascade its children, and the hook has no view of
+// `expiresAt`, so `Handled` cannot re-check the way `Proceed` does. An
+// update()/patch() that extends or clears the TTL in that window would
+// otherwise destroy a document that is no longer expired - and its children with
+// it. The pre-v2.11.0 loop held the collection lock across the whole erase and
+// had no such window, so this is a cost of the two-phase split.
+//
+// ⚠ HOW THIS IS INDUCED, and why it is honest rather than staged: a
+// single-threaded test cannot interleave a real writer, so the write is made to
+// land at exactly the wrong moment from INSIDE the sweep. Two documents expire
+// in the same sweep, wf-early before wf-late (the expiration index is keyed by
+// expiry time and walked in ascending order, so the order is deterministic).
+// The hook, while handling wf-early, extends wf-late's TTL - which is precisely
+// "a write landed after phase 1 collected wf-late and before the sweeper got to
+// it". Everything about wf-late's path after that point is the production path.
+void test_ttl_concurrent_ttl_extension_is_not_expired() {
+    TmpEnv t("ttl-race");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("ttl-race-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+
+    RelationInfo rel;
+    rel.name = "default:exec_wf";
+    rel.child = "default:executions";
+    rel.childField = "workflowId";
+    rel.parent = "default:workflows";
+    rel.onDelete = OnDelete::Cascade;
+    std::string mgrErr;
+    check(rm.createRelation(rel, mgrErr), "declared the cascade relation");
+    store.set_relations("executions",
+                        {RelationRef{"exec_wf", "workflowId", "workflows", false, true}});
+
+    const uint64_t farFuture = static_cast<uint64_t>(1) << 62;
+
+    auto seedParent = [&](const std::string& id, uint64_t expiresAt) {
+        Document d; d.id = id; d.collection = "workflows";
+        d.set_data({{"name", id}});
+        d.expiresAt = expiresAt;
+        store.put("workflows", id, d);
+        mstore.loadDocument("default:workflows", d);
+    };
+    auto seedChild = [&](const std::string& id, const std::string& parentId) {
+        Document d; d.id = id; d.collection = "executions";
+        d.set_data({{"workflowId", parentId}});
+        store.put("executions", id, d);
+        mstore.loadDocument("default:executions", d);
+    };
+
+    // expiresAt 1 sorts before 2, so wf-early is handled first in the sweep.
+    seedParent("wf-early", 1);
+    seedParent("wf-late", 2);
+    seedChild("ex-early", "wf-early");
+    seedChild("ex-late", "wf-late");
+
+    // The injected "concurrent" write: while the sweeper is busy expiring
+    // wf-early (a real cascade - WAL fsync plus an LMDB commit, which is what
+    // makes the window wide), wf-late's TTL is extended.
+    bool injected = false;
+    mstore.setTtlExpiryRelationHook(
+        [&](const std::string& coll, const std::string& id, bool firstAttempt) {
+            if (!injected && id == "wf-early") {
+                injected = true;
+                Document renewed;
+                renewed.id = "wf-late";
+                renewed.collection = "workflows";
+                renewed.set_data({{"name", "wf-late"}});
+                renewed.expiresAt = farFuture;   // TTL extended
+                mstore.loadDocument("default:workflows", renewed);
+            }
+            return ttlExpiryDecision(rm, store, p.pm, mstore, cfgManager, coll, id,
+                                     nullptr, firstAttempt);
+        });
+
+    const uint64_t expired = mstore.expireDocuments();
+
+    check(injected, "the injected concurrent TTL extension actually ran");
+    check(expired == 1, "exactly ONE document expired - wf-early only");
+
+    // wf-early: genuinely expired, cascaded as normal. Proves the sweep worked.
+    check(!residentInMemory(mstore, "default:workflows", "wf-early"), "wf-early expired");
+    check(!store.get("executions", "ex-early").has_value(), "its child cascaded away");
+
+    // wf-late: the whole point. Not expired, and - the data-loss half - its
+    // child was not cascaded either.
+    check(residentInMemory(mstore, "default:workflows", "wf-late"),
+          "wf-late was NOT expired - its TTL was extended after phase 1 collected it");
+    check(store.get("workflows", "wf-late").has_value(), "and it is intact in LMDB");
+    check(store.get("executions", "ex-late").has_value(),
+          "and ITS CHILD was not cascaded - this is the data-loss half of the race");
+    check(store.relation_index_child_count("exec_wf", "wf-late") == 1,
+          "the reverse-index posting survived too");
+    check(mstore.getStats().ttlExpiryBlockedByRelation == 0,
+          "and it was not recorded as relation-blocked - it simply is not expired");
+
+    p.pm.stop();
+    mstore.stop();
+}
+
+// =========================================================================
+// v2.11.0 close-out review, FINDING 2 — a relation-blocked document must not
+// consume the sweep's candidate budget.
+//
+// Phase 1 never erases a blocked document's expiration index entry (deliberately
+// - it has to stay armed so the block can lift), so with one shared budget those
+// documents refill the cap on every sweep. `collections_` iteration order is
+// stable, so every collection after them SILENTLY STOPS BEING SWEPT - and TTL'd
+// parents under `restrict` is exactly the combination this feature creates, so
+// that is the expected steady state, not an edge case.
+//
+// Budgets come from Config so this can be shown at 3 documents instead of 10001.
+void test_ttl_blocked_documents_do_not_starve_the_sweep() {
+    TmpEnv t("ttl-starve");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("ttl-starve-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    MemoryStore::Config cfg;
+    cfg.ttlMaxCandidatesPerSweep = 3;     // exactly filled by the blocked parents
+    cfg.ttlBlockedRetrySweeps = 100;      // far enough away not to interfere
+    MemoryStore mstore(cfg);
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+
+    std::atomic<bool> healthy{true};
+    std::atomic<uint64_t> drift{0};
+    mstore.setDocumentStoreMirror(
+        [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
+        &healthy, &drift);
+
+    RelationInfo rel;
+    rel.name = "default:exec_wf";
+    rel.child = "default:executions";
+    rel.childField = "workflowId";
+    rel.parent = "default:workflows";
+    rel.onDelete = OnDelete::Restrict;
+    std::string mgrErr;
+    check(rm.createRelation(rel, mgrErr), "declared the restrict relation");
+    store.set_relations("executions",
+                        {RelationRef{"exec_wf", "workflowId", "workflows", false, true}});
+
+    // Three restrict-blocked TTL'd parents. Their expiry times are the earliest,
+    // so they are collected first and fill the budget on the first sweep.
+    for (int i = 0; i < 3; ++i) {
+        const std::string pid = "wf-" + std::to_string(i);
+        Document parent; parent.id = pid; parent.collection = "workflows";
+        parent.set_data({{"name", pid}});
+        parent.expiresAt = 1 + static_cast<uint64_t>(i);
+        store.put("workflows", pid, parent);
+        mstore.loadDocument("default:workflows", parent);
+
+        Document child; child.id = "ex-" + std::to_string(i);
+        child.collection = "executions";
+        child.set_data({{"workflowId", pid}});
+        store.put("executions", child.id, child);
+        mstore.loadDocument("default:executions", child);
+    }
+
+    // The victim: a TTL'd document with NO children, expiring after the three
+    // blocked parents. Deliberately in the SAME collection, because that makes
+    // the ordering deterministic - `collections_` is an unordered_map, but a
+    // collection's expirationIndex is a std::map walked in ascending expiry
+    // order, so wf-0/1/2 (1, 2, 3) are always collected before wf-victim (100).
+    // Starvation inside one collection is the same defect as starvation across
+    // collections, and this way the test cannot pass or fail on hash order.
+    Document victim;
+    victim.id = "wf-victim";
+    victim.collection = "workflows";
+    victim.set_data({{"name", "wf-victim"}});
+    victim.expiresAt = 100;
+    store.put("workflows", "wf-victim", victim);
+    mstore.loadDocument("default:workflows", victim);
+
+    installTtlHook(mstore, rm, store, p.pm, cfgManager);
+
+    // Sweep 1: the three blocked parents are fresh, so they legitimately take
+    // the whole budget and the victim is not reached at all.
+    const uint64_t first = mstore.expireDocuments();
+    check(first == 0, "sweep 1 expired nothing - the budget went to the blocked parents");
+    check(mstore.getStats().ttlExpiryBlockedByRelation == 3, "all three parents were refused");
+    check(mstore.ttlBlockedDocumentCount() == 3, "and all three are tracked as stuck");
+    check(residentInMemory(mstore, "default:workflows", "wf-victim"),
+          "and the victim has not been reached yet");
+
+    // Sweep 2: the blocked parents are known and not due, so they must consume
+    // NOTHING and the victim finally gets in. With one shared budget it never
+    // would - the three blocked parents refill the cap on every sweep, forever.
+    uint64_t total = first;
+    total += mstore.expireDocuments();
+    check(total == 1,
+          "the childless document was expired despite three blocked parents filling "
+          "the budget - blocked documents no longer starve the sweep");
+    check(!residentInMemory(mstore, "default:workflows", "wf-victim"), "the victim is gone");
+    check(!store.get("workflows", "wf-victim").has_value(), "and its DELETE reached LMDB");
+    check(mstore.getStats().ttlExpiryBlockedByRelation == 3,
+          "and the blocked parents were not re-consulted - still three refusal events, "
+          "not three more per sweep");
+
+    // Ten more sweeps: still nothing re-examined, so the steady-state cost of a
+    // persistent block is zero rather than one LMDB scan + one WARN per document
+    // per second.
+    for (int i = 0; i < 10; ++i) mstore.expireDocuments();
+    check(mstore.getStats().ttlExpiryBlockedByRelation == 3,
+          "ten further sweeps consulted the relation zero times");
+    check(mstore.ttlBlockedDocumentCount() == 3, "and the three are still tracked as stuck");
+
+    p.pm.stop();
+    mstore.stop();
+}
+
+// Round-3 review, item 2 — dropping a blocked document's collection must not
+// leak its blocked-set entry. Phase 1 never revisits a key whose collection is
+// gone and dropCollection() knows nothing about the blocked set, so a leaked
+// entry would inflate ttlBlockedDocumentCount() - the gauge an operator reads -
+// for the rest of the process's life.
+void test_ttl_blocked_entry_is_released_when_the_collection_is_dropped() {
+    TmpEnv t("ttl-blocked-drop");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("ttl-blocked-drop-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    MemoryStore::Config cfg;
+    cfg.ttlBlockedRetrySweeps = 1;   // due again on the very next sweep
+    MemoryStore mstore(cfg);
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+
+    seedTtlParentAndChild(mstore, store, rm, OnDelete::Restrict);
+    installTtlHook(mstore, rm, store, p.pm, cfgManager);
+
+    check(mstore.expireDocuments() == 0, "the restrict-protected parent is blocked");
+    check(mstore.ttlBlockedDocumentCount() == 1, "and is counted as stuck");
+
+    // The collection goes away underneath it.
+    mstore.dropCollection("default:workflows");
+    check(mstore.expireDocuments() == 0, "nothing left to expire");
+    check(mstore.ttlBlockedDocumentCount() == 0,
+          "the blocked entry was released with the collection - the gauge does not "
+          "keep counting a document that no longer exists");
+
+    p.pm.stop();
+    mstore.stop();
+}
+
+// =========================================================================
+// v2.11.0 close-out — BOOT-PASS DRIFT MUST NOT DISABLE DESTRUCTIVE POLICIES.
+//
+// mirror_drift_count_ is never reset, and the post-replay re-mirror pass bumps
+// it for every row it cannot repair - a condition runPendingRemirror() treats as
+// advisory and which recurs on every boot, since the restart replays the same
+// failing row. Gated on the raw counter, ONE unrepairable row refused every
+// cascade/set_null delete and every CreateRelation for the process's life, with
+// an error message that told the operator to restart.
+// =========================================================================
+void test_boot_pass_drift_does_not_disable_the_cascade() {
+    TmpEnv t("rel-drift-baseline");
+    LmdbDocumentStore store(t.env);
+    TmpPersistence p("rel-drift-baseline-wal");
+    check(p.pm.start(), "persistence manager started");
+
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+    RelationManager rm(mstore);
+    rm.loadFromStore();
+    CollectionConfigManager cfgManager(mstore);
+
+    std::atomic<bool> healthy{true};
+    std::atomic<uint64_t> drift{0};
+    mstore.setDocumentStoreMirror(
+        [&store](std::string_view) -> smartbotic::db::storage::DocumentStore* { return &store; },
+        &healthy, &drift);
+
+    // The boot path: three rows the re-mirror pass could not write. Health is
+    // deliberately NOT flipped (the v2.8.1 lesson), so drift is the only signal.
+    drift.store(3);
+    check(mstore.mirrorDriftCount() == 3, "the raw counter still reports every stale row");
+    mstore.markMirrorDriftBaseline();       // what initialize() does last
+    check(mstore.mirrorDriftBaseline() == 3, "the boot path's drift is in the baseline");
+    check(mstore.mirrorDriftSinceBaseline() == 0,
+          "and nothing has drifted since READY");
+
+    seedTtlParentAndChild(mstore, store, rm, OnDelete::Cascade);
+
+    bool parentExisted = false;
+    try {
+        parentExisted = executeCascade(rm, store, p.pm, mstore, cfgManager,
+                                       "default:workflows", "wf-1", nullptr);
+    } catch (const std::exception& e) {
+        check(false, "the cascade was refused because of drift the BOOT PASS accrued");
+        std::cerr << "  (" << e.what() << ")\n";
+    }
+    check(parentExisted, "the cascade ran despite 3 rows the boot pass left stale");
+    check(!store.get("executions", "ex-1").has_value(), "and it actually cascaded");
+
+    // A LIVE write's drift still latches and still refuses - that half was
+    // right, and re-basing the gate must not have removed it.
+    drift.fetch_add(1);
+    check(mstore.mirrorDriftSinceBaseline() == 1, "post-READY drift is visible");
+    Document parent2;
+    parent2.id = "wf-2";
+    parent2.collection = "workflows";
+    parent2.set_data({{"name", "wf-2"}});
+    store.put("workflows", "wf-2", parent2);
+    mstore.loadDocument("default:workflows", parent2);
+    Document child2;
+    child2.id = "ex-2";
+    child2.collection = "executions";
+    child2.set_data({{"workflowId", "wf-2"}});
+    store.put("executions", "ex-2", child2);
+    mstore.loadDocument("default:executions", child2);
+
+    bool refused = false;
+    try {
+        executeCascade(rm, store, p.pm, mstore, cfgManager, "default:workflows", "wf-2", nullptr);
+    } catch (const CascadeBlocked&) {
+        refused = true;
+    }
+    check(refused, "drift accrued AFTER READY still refuses the cascade");
+    check(store.get("executions", "ex-2").has_value(), "and nothing was touched");
+
+    p.pm.stop();
+    mstore.stop();
+}
+
+// v2.11.0 close-out — runPendingRemirror() must RELEASE the id list.
+// RecoveryOutcome lives as DatabaseService::recovery_outcome_ for the process
+// lifetime; the pass runs exactly once and nothing reads the list afterwards, so
+// keeping it resident pinned ~100 bytes per replayed row (an install's whole
+// history under --recovery-mode=wal_only) for nothing.
+void test_pending_remirror_list_is_released_after_the_pass() {
+    TmpPersistence p("remirror-release");
+    MemoryStore mstore(MemoryStore::Config{});
+    mstore.start();
+
+    smartbotic::database::RecoveryOutcome outcome;
+    for (int i = 0; i < 500; ++i) {
+        outcome.pendingRemirror.emplace_back("default:widgets", "w-" + std::to_string(i));
+    }
+    check(outcome.pendingRemirror.size() == 500, "the list starts populated");
+
+    p.pm.runPendingRemirror(mstore, outcome);
+
+    check(outcome.pendingRemirror.empty(),
+          "the id list is cleared once the pass has run");
+    check(outcome.pendingRemirror.capacity() == 0,
+          "and its CAPACITY is released - clear() alone keeps the whole allocation");
+
+    mstore.stop();
+}
+
+int main() {
+    std::cout << "=== test_relation_enforcement ===\n";
+    test_index_follows_the_child_field();
+    test_undeclared_collection_maintains_nothing();
+    test_unrelated_update_leaves_the_posting_alone();
+    test_posting_visible_after_reopen();
+    test_restrict_blocks_and_names_the_blockers();
+    test_cascade_and_set_null_never_block_via_restrict_check();
+    test_relations_enforced_false_skips_the_check();
+    test_describe_delete_reports_restrict_and_no_action();
+    test_describe_delete_zero_children_does_not_block();
+    test_describe_delete_no_relations_declared();
+    test_cascade_deletes_scalar_children_wal_first();
+    test_array_reference_pulls_id_and_keeps_document();
+    test_crash_between_wal_and_commit_recovers();
+    test_cascade_refuses_when_grandchild_is_restrict_protected();
+    test_replayed_cascade_update_remirrors_to_lmdb_after_crash_window();
+    test_a_failing_row_does_not_stop_recovery();
+    test_reinsert_after_delete_converges_in_lmdb();
+    test_validate_on_write_rejects_missing_parent();
+    test_validate_on_write_accepts_existing_parent();
+    test_validate_on_write_false_allows_dangling_and_check_reports_it();
+    test_validate_on_write_never_rejects_absent_or_null();
+    test_validate_on_write_array_any_missing_rejects_whole_write();
+    test_validate_on_write_unrelated_update_not_rechecked();
+    test_validate_on_write_rolls_back_an_already_applied_sibling_relation();
+    test_validate_on_write_relations_enforced_false_is_the_escape_hatch();
+    test_validate_on_write_closes_the_stale_check_race();
+    test_remirror_maintains_the_secondary_index();
+    test_moved_unique_value_is_resolved_by_the_retry_pass();
+    test_remirror_windows_are_bounded_by_bytes_and_by_count();
+    test_cascade_refuses_while_the_mirror_is_unhealthy_or_drifted();
+    test_can_reject_writes_gates_the_undo_snapshot();
+    test_ttl_restrict_blocks_the_expiry();
+    test_ttl_cascade_deletes_children_like_a_manual_delete();
+    test_ttl_set_null_nulls_the_scalar_and_keeps_the_child();
+    test_ttl_array_reference_pulls_the_id_and_keeps_the_document();
+    test_ttl_no_action_expires_and_leaves_the_reference_dangling();
+    test_ttl_with_no_relations_expires_exactly_as_before();
+    test_ttl_cascade_wal_is_durable_before_the_lmdb_commit();
+    test_ttl_concurrent_ttl_extension_is_not_expired();
+    test_ttl_blocked_documents_do_not_starve_the_sweep();
+    test_ttl_blocked_entry_is_released_when_the_collection_is_dropped();
+    test_boot_pass_drift_does_not_disable_the_cascade();
+    test_pending_remirror_list_is_released_after_the_pass();
+
+    std::cout << "passed: " << g_pass << ", failed: " << g_fail << "\n";
+    return g_fail == 0 ? 0 : 1;
+}

+ 341 - 0
tests/test_relation_index.cpp

@@ -0,0 +1,341 @@
+// v2.11.0 T2 — relation reverse index tests.
+//
+// Storage-only: this exercises LmdbDocumentStore::relation_index_{add,remove,
+// child_count,children} directly. No document-write hooks (Task 3), no
+// enforcement (Task 4) are involved.
+//
+// test_index_created_at_runtime_is_visible_after_reopen is THE regression
+// test for this task: since v2.8.1 the read path serves only from the primed
+// dbi cache, so a sub-db created at runtime is invisible to every later read
+// in the process unless cacheCommittedDbi() runs after commit. Removing that
+// call from relation_index_add must fail this test.
+
+#include <algorithm>
+#include <atomic>
+#include <cstdio>
+#include <filesystem>
+#include <iostream>
+#include <string>
+#include <unistd.h>
+#include <vector>
+
+#include <nlohmann/json.hpp>
+
+#include "document.hpp"
+#include "storage/document_store_lmdb.hpp"
+#include "storage/lmdb_env.hpp"
+
+namespace fs = std::filesystem;
+
+using smartbotic::database::Document;
+using smartbotic::db::storage::LmdbDocumentStore;
+using smartbotic::db::storage::LmdbEnv;
+using smartbotic::db::storage::LmdbEnvOpts;
+
+namespace {
+
+int g_pass = 0;
+int g_fail = 0;
+
+void check(bool cond, const char* msg) {
+    if (cond) {
+        ++g_pass;
+    } else {
+        ++g_fail;
+        std::cerr << "FAIL: " << msg << "\n";
+    }
+}
+
+std::string make_tmpdir(const char* tag) {
+    static std::atomic<int> counter{0};
+    std::string path = "/tmp/relidx-test-" + std::to_string(::getpid()) + "-" +
+                       std::to_string(counter.fetch_add(1)) + "-" + tag;
+    std::error_code ec;
+    fs::remove_all(path, ec);
+    return path;
+}
+
+struct TmpEnv {
+    std::string path;
+    LmdbEnv env;
+    explicit TmpEnv(const char* tag)
+        : path(make_tmpdir(tag)),
+          env(LmdbEnvOpts{path, 64ULL << 20, 256, 126, false}) {}
+    ~TmpEnv() {
+        std::error_code ec;
+        fs::remove_all(path, ec);
+    }
+    TmpEnv(const TmpEnv&) = delete;
+    TmpEnv& operator=(const TmpEnv&) = delete;
+};
+
+// -------------------------------------------------------------------------
+
+// DUPSORT shape, per the re-validated design.
+void test_children_of_a_parent_are_a_dup_set() {
+    TmpEnv t("relidx");
+    LmdbDocumentStore store(t.env);
+
+    check(store.relation_index_add("r1", "wf-1", "exec-a"), "added a child");
+    check(store.relation_index_add("r1", "wf-1", "exec-b"), "and another");
+    check(store.relation_index_add("r1", "wf-2", "exec-c"), "under a second parent");
+
+    // THE operation restrict and DescribeDelete need: a count without reading
+    // the children. mdb_cursor_count makes it O(1)-ish.
+    check(store.relation_index_child_count("r1", "wf-1") == 2, "two children of wf-1");
+    check(store.relation_index_child_count("r1", "wf-2") == 1, "one child of wf-2");
+    check(store.relation_index_child_count("r1", "wf-none") == 0,
+          "an unreferenced parent has none, and that is not an error");
+
+    auto kids = store.relation_index_children("r1", "wf-1", 10);
+    std::sort(kids.begin(), kids.end());
+    check(kids == std::vector<std::string>{"exec-a", "exec-b"}, "children listed");
+
+    // Removing one pair must not remove the sibling.
+    check(store.relation_index_remove("r1", "wf-1", "exec-a"), "removed one pair");
+    check(store.relation_index_child_count("r1", "wf-1") == 1, "the sibling survives");
+
+    // Idempotent: re-adding the same pair is a no-op, not a duplicate.
+    store.relation_index_add("r1", "wf-1", "exec-b");
+    check(store.relation_index_child_count("r1", "wf-1") == 1, "no duplicate posting");
+
+    // A distinct relation is a distinct sub-db: same parent id, no crosstalk.
+    check(store.relation_index_child_count("r2", "wf-1") == 0,
+          "a different relation's index is independent");
+
+    // Removing a pair that was never present is reported, not thrown.
+    check(!store.relation_index_remove("r1", "wf-1", "exec-does-not-exist"),
+          "removing an absent pair returns false rather than throwing");
+    check(!store.relation_index_remove("r1", "wf-none", "exec-a"),
+          "removing under an absent parent returns false rather than throwing");
+}
+
+// The failure this repo is most likely to reproduce. Since v2.8.1 reads serve
+// only from the primed dbi cache, so a sub-db created at runtime is invisible
+// unless registered after commit - and relations would silently not enforce.
+void test_index_created_at_runtime_is_visible_after_reopen() {
+    const std::string path = make_tmpdir("relidx-visible");
+    const LmdbEnvOpts opts{path, 64ULL << 20, 256, 126, false};
+    {
+        LmdbEnv env(opts);
+        LmdbDocumentStore store(env);
+        store.relation_index_add("r1", "wf-1", "exec-a");
+        // Same process, no reopen: must be visible immediately.
+        check(store.relation_index_child_count("r1", "wf-1") == 1,
+              "visible in the process that created it - this is what "
+              "cacheCommittedDbi() buys");
+    }
+    LmdbEnv env2(opts);
+    LmdbDocumentStore store2(env2);      // priming runs in the constructor
+    check(store2.relation_index_child_count("r1", "wf-1") == 1,
+          "and after a restart, via prime_dbi_cache()");
+
+    std::error_code ec;
+    fs::remove_all(path, ec);
+}
+
+// v2.11.0 T7 — declaring a relation over a collection that ALREADY has rows
+// must index those pre-existing rows, not just ones written afterwards.
+// This is the exact bug the task brief calls out: without a bootstrap scan,
+// declaring a relation over populated data leaves every existing child
+// invisible to enforcement while looking like it worked.
+void test_build_relation_index_covers_pre_existing_rows() {
+    TmpEnv t("relbuild-basic");
+    LmdbDocumentStore store(t.env);
+
+    // Rows exist BEFORE the relation is ever declared - no set_relations()
+    // call has happened, so put() below does not touch the reverse index at
+    // all. This is exactly "declare a relation on a populated collection".
+    auto put = [&](const std::string& id, const nlohmann::json& data) {
+        Document d; d.id = id; d.collection = "executions"; d.set_data(data);
+        store.put("executions", id, d);
+    };
+    put("e1", {{"workflowId", "wf-1"}});
+    put("e2", {{"workflowId", "wf-1"}});
+    put("e3", {{"workflowId", "wf-2"}});
+    put("e4", {{"other", 1}});                 // no reference - must not count
+    put("e5", {{"workflowId", nullptr}});      // null - must not count
+
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 0,
+          "before the bootstrap scan, the reverse index knows nothing - this "
+          "is the bug: a declared relation would silently protect nothing");
+
+    const uint64_t indexed = store.build_relation_index("exec_wf", "executions", "workflowId");
+    check(indexed == 3, "3 rows had a resolvable reference (e4/e5 excluded)");
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 2,
+          "both pre-existing children of wf-1 are now indexed");
+    check(store.relation_index_child_count("exec_wf", "wf-2") == 1,
+          "and wf-2's child");
+
+    auto kids = store.relation_index_children("exec_wf", "wf-1", 10);
+    std::sort(kids.begin(), kids.end());
+    check(kids == std::vector<std::string>{"e1", "e2"}, "correct children listed");
+
+    // Idempotent: re-running the scan (e.g. re-declaring the relation) must
+    // not double the postings, the same MDB_NODUPDATA guarantee build_index
+    // already relies on.
+    const uint64_t reindexed = store.build_relation_index("exec_wf", "executions", "workflowId");
+    check(reindexed == 3, "the re-scan still visits the same 3 rows");
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 2,
+          "no duplicate postings from re-running the scan");
+    check(store.relation_index_child_count("exec_wf", "wf-2") == 1,
+          "no duplicate postings on wf-2 either");
+
+    check(store.relation_index_exists("exec_wf"),
+          "the sub-db now exists - relation_index_exists distinguishes this "
+          "from 'declared but never built'");
+    check(!store.relation_index_exists("never_declared"),
+          "an unrelated relation's sub-db was never created");
+}
+
+// An array-valued child field must contribute one posting per element, same
+// as the write-path maintainRelations() - the bootstrap scan must not
+// diverge from ongoing maintenance.
+void test_build_relation_index_handles_array_valued_field() {
+    TmpEnv t("relbuild-array");
+    LmdbDocumentStore store(t.env);
+
+    Document n; n.id = "n1"; n.collection = "nodes";
+    n.set_data({{"config", {{"credentialIds", {"c1", "c2"}}}}});
+    store.put("nodes", "n1", n);
+
+    const uint64_t indexed =
+        store.build_relation_index("node_creds", "nodes", "config.credentialIds");
+    check(indexed == 1, "one row contributed (it has two postings, one row)");
+    check(store.relation_index_child_count("node_creds", "c1") == 1, "array element 1");
+    check(store.relation_index_child_count("node_creds", "c2") == 1, "array element 2");
+}
+
+// A relation over a collection that does not exist yet (or is empty) must
+// not throw, and must not fabricate a sub-db that then confuses
+// relation_index_exists.
+void test_build_relation_index_over_absent_collection() {
+    TmpEnv t("relbuild-absent");
+    LmdbDocumentStore store(t.env);
+
+    const uint64_t indexed = store.build_relation_index("exec_wf", "executions", "workflowId");
+    check(indexed == 0, "nothing to index in a collection that was never written");
+}
+
+// v2.11.0 T7 — check_relation_dangling: a genuinely dangling reference (a
+// child row referencing a parent id that does not exist) must be reported,
+// with the correct count and sample child ids; a live reference must not be
+// flagged.
+void test_check_relation_dangling_finds_missing_parents() {
+    TmpEnv t("relcheck-basic");
+    LmdbDocumentStore store(t.env);
+
+    auto putParent = [&](const std::string& id) {
+        Document d; d.id = id; d.collection = "workflows"; d.set_data({{"name", id}});
+        store.put("workflows", id, d);
+    };
+    auto putChild = [&](const std::string& id, const std::string& wf) {
+        Document d; d.id = id; d.collection = "executions";
+        d.set_data({{"workflowId", wf}});
+        store.put("executions", id, d);
+    };
+
+    // wf-1 exists and is referenced - not dangling.
+    putParent("wf-1");
+    putChild("e1", "wf-1");
+    putChild("e2", "wf-1");
+
+    // wf-missing is referenced but was NEVER written as a parent - dangling.
+    putChild("e3", "wf-missing");
+    putChild("e4", "wf-missing");
+    putChild("e5", "wf-missing");
+
+    store.build_relation_index("exec_wf", "executions", "workflowId");
+
+    auto result = store.check_relation_dangling("exec_wf", "workflows", 100);
+    check(result.total == 1, "exactly one dangling parent id");
+    check(result.entries.size() == 1, "and it is reported (well under the cap)");
+    if (!result.entries.empty()) {
+        const auto& d = result.entries[0];
+        check(d.parentId == "wf-missing", "names the missing parent");
+        check(d.childCount == 3, "counts all three referencing children");
+        check(d.sampleChildIds.size() == 3, "samples all three (under the 5-sample cap)");
+        std::vector<std::string> sorted = d.sampleChildIds;
+        std::sort(sorted.begin(), sorted.end());
+        check(sorted == std::vector<std::string>{"e3", "e4", "e5"}, "correct sample ids");
+    }
+
+    // check_relation_dangling must mutate nothing - the index and parent
+    // collection are exactly as they were.
+    check(store.relation_index_child_count("exec_wf", "wf-missing") == 3,
+          "the check did not remove or alter the dangling postings");
+    check(store.count("workflows") == 1, "the check did not create a phantom parent row");
+}
+
+// A relation with no dangling references reports zero, and never invents
+// one for a value that legitimately exists.
+void test_check_relation_dangling_clean_relation() {
+    TmpEnv t("relcheck-clean");
+    LmdbDocumentStore store(t.env);
+
+    Document p; p.id = "wf-1"; p.collection = "workflows"; p.set_data({{"name", "wf-1"}});
+    store.put("workflows", "wf-1", p);
+    Document c; c.id = "e1"; c.collection = "executions";
+    c.set_data({{"workflowId", "wf-1"}});
+    store.put("executions", "e1", c);
+
+    store.build_relation_index("exec_wf", "executions", "workflowId");
+
+    auto result = store.check_relation_dangling("exec_wf", "workflows", 100);
+    check(result.total == 0, "no dangling references");
+    check(result.entries.empty(), "nothing reported");
+}
+
+// A relation that was declared but never built (no build_relation_index /
+// relation_index_add call at all) has no sub-db - check_relation_dangling
+// must report "nothing to say" rather than treating an absent index as
+// "everything is dangling".
+void test_check_relation_dangling_never_built_reports_nothing() {
+    TmpEnv t("relcheck-unbuilt");
+    LmdbDocumentStore store(t.env);
+
+    Document c; c.id = "e1"; c.collection = "executions";
+    c.set_data({{"workflowId", "wf-1"}});
+    store.put("executions", "e1", c);   // no relation declared/built at all
+
+    auto result = store.check_relation_dangling("exec_wf", "workflows", 100);
+    check(result.total == 0, "an unbuilt index has nothing to report - not a false positive");
+    check(result.entries.empty(), "nothing reported");
+}
+
+// The `maxResults` cap bounds the returned samples but `total` must still be
+// exact - a caller must never be able to mistake a capped list for the
+// complete one.
+void test_check_relation_dangling_respects_cap_but_total_is_exact() {
+    TmpEnv t("relcheck-capped");
+    LmdbDocumentStore store(t.env);
+
+    for (int i = 0; i < 5; ++i) {
+        Document c; c.id = "e" + std::to_string(i); c.collection = "executions";
+        c.set_data({{"workflowId", "wf-missing-" + std::to_string(i)}});
+        store.put("executions", "e" + std::to_string(i), c);
+    }
+    store.build_relation_index("exec_wf", "executions", "workflowId");
+
+    auto result = store.check_relation_dangling("exec_wf", "workflows", 2);
+    check(result.total == 5, "total counts every dangling parent, not just the capped sample");
+    check(result.entries.size() == 2, "entries is capped at maxResults");
+}
+
+}  // namespace
+
+int main() {
+    std::cout << "=== test_relation_index ===\n";
+    test_children_of_a_parent_are_a_dup_set();
+    test_index_created_at_runtime_is_visible_after_reopen();
+    test_build_relation_index_covers_pre_existing_rows();
+    test_build_relation_index_handles_array_valued_field();
+    test_build_relation_index_over_absent_collection();
+    test_check_relation_dangling_finds_missing_parents();
+    test_check_relation_dangling_clean_relation();
+    test_check_relation_dangling_never_built_reports_nothing();
+    test_check_relation_dangling_respects_cap_but_total_is_exact();
+
+    std::cout << "passed: " << g_pass << ", failed: " << g_fail << "\n";
+    return g_fail == 0 ? 0 : 1;
+}

+ 545 - 0
tests/test_relation_manager.cpp

@@ -0,0 +1,545 @@
+// Task 1 — RelationManager: the declaration registry for referential
+// integrity relations. No enforcement, no LMDB index yet (later tasks).
+//
+// Modeled on tests/test_view_manager_paging.cpp: relations are keyed by
+// their project-qualified name in a `_relations` system collection, and
+// loadFromStore() must page explicitly since Query::limit defaults to 100
+// and limit=0 returns nothing (not everything) — the same trap that has
+// already shipped as a bug in ViewManager, PolicyManager and
+// CollectionConfigManager.
+
+#include <iostream>
+#include <string>
+
+#include <nlohmann/json.hpp>
+
+#include <filesystem>
+#include <fstream>
+#include <unistd.h>
+
+#include "document.hpp"
+#include "memory_store.hpp"
+#include "migrations/migration_runner.hpp"
+#include "relations/relation_manager.hpp"
+#include "views/view_manager.hpp"
+
+using namespace smartbotic::database;
+
+namespace {
+
+int g_pass = 0;
+int g_fail = 0;
+
+void check(bool cond, const std::string& msg) {
+    if (cond) { ++g_pass; }
+    else { ++g_fail; std::cerr << "FAIL: " << msg << "\n"; }
+}
+
+struct Fixture {
+    MemoryStore store;
+    Fixture() : store(MemoryStore::Config{}) {
+        store.start();
+    }
+    ~Fixture() { store.stop(); }
+};
+
+void test_relations_are_project_scoped_and_survive_reload() {
+    Fixture f;                       // MemoryStore, started
+    RelationManager rm(f.store);
+    rm.loadFromStore();
+
+    RelationInfo a;
+    a.name = "default:exec_wf";
+    a.child = "default:executions";
+    a.childField = "workflowId";
+    a.parent = "default:workflows";
+    std::string err;
+    check(rm.createRelation(a, err), "created in default");
+
+    RelationInfo b = a;              // SAME bare name, different project
+    b.name = "acme:exec_wf";
+    b.child = "acme:executions";
+    b.parent = "acme:workflows";
+    check(rm.createRelation(b, err), "the same name in another project is allowed");
+
+    check(rm.listRelations("default").size() == 1, "listing is project-filtered");
+    check(rm.listRelations().size() == 2, "empty project lists everything");
+
+    RelationManager fresh(f.store);   // restart
+    fresh.loadFromStore();
+    check(fresh.getRelation("default:exec_wf").has_value(), "survives reload");
+    check(fresh.getRelation("acme:exec_wf").has_value(), "both survive");
+}
+
+void test_cross_project_relation_is_refused() {
+    Fixture f;
+    RelationManager rm(f.store);
+    RelationInfo r;
+    r.name = "default:bad";
+    r.child = "default:executions";
+    r.childField = "workflowId";
+    r.parent = "acme:workflows";      // different env - no txn spans two
+    std::string err;
+    check(!rm.createRelation(r, err), "a cross-project relation is refused");
+    check(err.find("project") != std::string::npos, "and says why");
+}
+
+void test_more_than_one_page_of_relations_loads() {
+    Fixture f;
+    RelationManager rm(f.store);
+    for (int i = 0; i < 250; ++i) {          // Query::limit defaults to 100
+        RelationInfo r;
+        char buf[32];
+        std::snprintf(buf, sizeof(buf), "default:r%03d", i);
+        r.name = buf;
+        r.child = "default:c";
+        r.childField = "p";
+        r.parent = "default:p";
+        std::string err;
+        rm.createRelation(r, err);
+    }
+    RelationManager fresh(f.store);
+    fresh.loadFromStore();
+    check(fresh.listRelations().size() == 250,
+          "all 250 load - a bare Query would stop at 100, as it did for views, "
+          "policies and collection configs");
+}
+
+// v2.11.0 T13 round 2 (review finding 3) — createRelation() refuses a
+// cross-project declaration (test_cross_project_relation_is_refused,
+// above), but loadFromStore() is the OTHER way a RelationInfo enters the
+// cache and did not re-check it. A legacy or hand-written `_relations`
+// document naming a cross-project parent must be skipped at load time too -
+// arming it would resolve the bare parent name inside the CHILD's own
+// project env (validate_on_write/RelationRef only ever resolve `parent`
+// against the child's project), silently checking the wrong collection.
+void test_load_skips_a_cross_project_relation_written_by_hand() {
+    Fixture f;
+
+    // Bypass createRelation()'s own guard entirely - write the raw document
+    // straight into `_relations`, the way a legacy record or a hand-edited
+    // one would exist on disk. `parent` names a DIFFERENT project than
+    // `child`, which createRelation() would refuse today.
+    nlohmann::json bad = {
+        {"name", "default:bad_cross"},
+        {"child", "default:executions"},
+        {"child_field", "workflowId"},
+        {"parent", "acme:workflows"},
+        {"on_delete", "restrict"},
+        {"validate_on_write", true},
+        {"created_at", 0},
+        {"updated_at", 0},
+    };
+    Document d;
+    d.id = "default:bad_cross";
+    d.collection = RelationManager::SYSTEM_COLLECTION;
+    d.set_data(bad);
+    f.store.insert(RelationManager::SYSTEM_COLLECTION, d);
+
+    // A well-formed, same-project relation alongside it, to confirm one bad
+    // record does not stop the rest of the load.
+    RelationInfo good;
+    good.name = "default:exec_wf";
+    good.child = "default:executions";
+    good.childField = "ownerId";
+    good.parent = "default:users";
+    {
+        RelationManager rm(f.store);
+        rm.loadFromStore();
+        std::string err;
+        check(rm.createRelation(good, err), "the well-formed sibling declares fine");
+    }
+
+    RelationManager fresh(f.store);
+    fresh.loadFromStore();
+    check(!fresh.getRelation("default:bad_cross").has_value(),
+          "the cross-project relation was skipped, not armed with the wrong parent");
+    check(fresh.getRelation("default:exec_wf").has_value(),
+          "the well-formed sibling still loaded - one bad record did not stop the rest");
+}
+
+}  // namespace
+
+
+// =========================================================================
+// v2.11.0 final review, finding 8 — the bare-vs-qualified bug class, made
+// unrepresentable at the RelationManager boundary rather than spot-fixed at
+// the caller.
+//
+// The live bug: armRelationsForChild() looked relations up under the caller's
+// RAW string while boot arming used the canonical "<project>:<collection>",
+// and set_relations() keys the storage map by the BARE name either way. So a
+// raw-gRPC caller naming "executions" instead of "default:executions" found
+// zero relations, and set_relations(bare, {}) then ERASED the entry the
+// canonical arming had filled - disarming both the reverse index and
+// validate_on_write for the life of the process, logged only as "re-armed 0
+// relation(s)", and silently repaired by a restart.
+//
+// This is the third occurrence of the class (v2.4.2 lost every view for two
+// releases, v2.4.5 gave every collection a phantom twin), so the fix is at the
+// one funnel every entry point already goes through.
+void test_bare_and_qualified_names_are_the_same_relation() {
+    Fixture f;
+    RelationManager rm(f.store);
+    rm.loadFromStore();
+
+    // Declared with an UNQUALIFIED name and unqualified endpoints, exactly as a
+    // raw-gRPC caller (or a hand-written migration) would.
+    RelationInfo bare;
+    bare.name = "exec_wf";
+    bare.child = "executions";
+    bare.childField = "workflowId";
+    bare.parent = "workflows";
+    std::string err;
+    check(rm.createRelation(bare, err), "a bare-named relation is accepted");
+
+    // It is STORED canonically, so everything downstream sees one form.
+    auto viaBare = rm.getRelation("exec_wf");
+    check(viaBare.has_value(), "found under the bare name the caller used");
+    check(viaBare && viaBare->name == "default:exec_wf",
+          "but it reports its CANONICAL name - the declaration was normalised, "
+          "not stored verbatim");
+    check(viaBare && viaBare->child == "default:executions", "child canonicalised too");
+    check(viaBare && viaBare->parent == "default:workflows", "parent canonicalised too");
+
+    auto viaQualified = rm.getRelation("default:exec_wf");
+    check(viaQualified.has_value(), "and found under the qualified name as well");
+
+    // A second declaration under the OTHER spelling is a DUPLICATE, not a new
+    // relation. Before the fix this created a second entry that armed the same
+    // bare collection and clobbered the first.
+    RelationInfo qualified = bare;
+    qualified.name = "default:exec_wf";
+    qualified.child = "default:executions";
+    qualified.parent = "default:workflows";
+    check(!rm.createRelation(qualified, err),
+          "declaring the qualified spelling of an existing bare relation is refused "
+          "as a duplicate");
+    check(err.find("already exists") != std::string::npos, "and says why");
+    check(rm.listRelations().size() == 1, "still exactly one relation, not two");
+
+    // THE LOOKUP THAT WAS BROKEN: armRelationsForChild's, and Delete's.
+    check(rm.relationsWithChild("executions").size() == 1,
+          "relationsWithChild finds it under the BARE collection name - this is "
+          "the lookup that returned zero and caused the disarm");
+    check(rm.relationsWithChild("default:executions").size() == 1,
+          "and under the qualified one");
+    check(rm.relationsWithParent("workflows").size() == 1,
+          "relationsWithParent likewise - a bare name here would report 'nobody's "
+          "parent' and permit every delete");
+    check(rm.relationsWithParent("default:workflows").size() == 1, "and qualified");
+
+    // Cross-project isolation is not weakened by canonicalisation.
+    check(rm.relationsWithChild("acme:executions").empty(),
+          "another project's same-named collection still matches nothing");
+
+    // Dropping by either spelling works, and actually removes the stored row.
+    check(rm.dropRelation("exec_wf", err), "droppable by the bare name");
+    check(!rm.getRelation("default:exec_wf").has_value(), "gone from the cache");
+    RelationManager reloaded(f.store);
+    reloaded.loadFromStore();
+    check(reloaded.listRelations().empty(),
+          "and gone from the store - dropping by the bare name really removed the "
+          "canonical document, it did not just clear the cache");
+}
+
+// finding 8, the migration half: a record already stored under a non-canonical
+// document id is re-keyed on load, in the store as well as in the cache.
+// Without the store half, dropRelation() (which removes by canonical name)
+// would leave the row behind and the relation would come back on the next boot.
+void test_load_rekeys_a_legacy_bare_named_record() {
+    Fixture f;
+    // Hand-write a record the way a pre-fix createRelation would have stored it.
+    Document d;
+    d.id = "legacy_rel";
+    d.set_data(nlohmann::json{
+        {"name", "legacy_rel"},
+        {"child", "executions"},
+        {"child_field", "workflowId"},
+        {"parent", "workflows"},
+        {"on_delete", "restrict"},
+        {"validate_on_write", false}});
+    f.store.createCollection(RelationManager::SYSTEM_COLLECTION, CollectionOptions{});
+    f.store.upsert(RelationManager::SYSTEM_COLLECTION, d);
+
+    RelationManager rm(f.store);
+    rm.loadFromStore();
+    auto got = rm.getRelation("default:legacy_rel");
+    check(got.has_value(), "the legacy record loaded");
+    check(got && got->name == "default:legacy_rel", "under its canonical name");
+
+    // The STORE was re-keyed, not just the cache.
+    check(!f.store.get(RelationManager::SYSTEM_COLLECTION, "legacy_rel").has_value(),
+          "the old bare-keyed document is gone");
+    check(f.store.get(RelationManager::SYSTEM_COLLECTION, "default:legacy_rel").has_value(),
+          "and a canonically-keyed one exists - so dropRelation(), which removes "
+          "by the canonical name, can actually remove it");
+
+    std::string err;
+    check(rm.dropRelation("default:legacy_rel", err), "and it drops");
+    RelationManager reloaded(f.store);
+    reloaded.loadFromStore();
+    check(reloaded.listRelations().empty(), "staying dropped across a reload");
+}
+
+// finding 9 — an unrecognised on_delete must be REFUSED, not coerced to
+// Restrict. There used to be two copies of the parser (one in
+// database_grpc_impl.cpp, one here) and both coerced silently. That failed safe
+// while cascade/set_null were inert; T12 made them destructive in the other
+// direction, so an operator who typed "Cascade" was told the relation was
+// created and believed cascade was armed while restrict was.
+void test_on_delete_is_validated_not_coerced() {
+    check(parseOnDelete("restrict").has_value(), "restrict parses");
+    check(parseOnDelete("cascade") == OnDelete::Cascade, "cascade parses");
+    check(parseOnDelete("set_null") == OnDelete::SetNull, "set_null parses");
+    check(parseOnDelete("no_action") == OnDelete::NoAction, "no_action parses");
+
+    // The typo cases that used to become Restrict silently.
+    check(!parseOnDelete("Cascade").has_value(), "'Cascade' is REFUSED, not coerced");
+    check(!parseOnDelete("CASCADE").has_value(), "'CASCADE' is refused");
+    check(!parseOnDelete("cascde").has_value(), "a misspelling is refused");
+    check(!parseOnDelete("setnull").has_value(), "'setnull' is refused");
+    check(!parseOnDelete("").has_value(),
+          "and the empty string is refused HERE - the RPC layer, not the parser, "
+          "is what maps absent to the documented default");
+
+    // Round-trips through the one renderer, so the two directions cannot drift.
+    for (auto v : {OnDelete::Restrict, OnDelete::Cascade, OnDelete::SetNull,
+                   OnDelete::NoAction}) {
+        check(parseOnDelete(onDeleteToString(v)) == v,
+              "onDeleteToString round-trips through parseOnDelete");
+    }
+
+    // A persisted record carrying a bad value falls back to Restrict (the safe
+    // direction: over-restricting refuses deletes, it never performs an
+    // unintended destructive one) rather than being dropped, which would remove
+    // protection entirely.
+    Fixture f;
+    Document d;
+    d.id = "default:bad_od";
+    d.set_data(nlohmann::json{
+        {"name", "default:bad_od"},
+        {"child", "default:executions"},
+        {"child_field", "workflowId"},
+        {"parent", "default:workflows"},
+        {"on_delete", "Cascade"}});
+    f.store.createCollection(RelationManager::SYSTEM_COLLECTION, CollectionOptions{});
+    f.store.upsert(RelationManager::SYSTEM_COLLECTION, d);
+
+    RelationManager rm(f.store);
+    rm.loadFromStore();
+    auto got = rm.getRelation("default:bad_od");
+    check(got.has_value(),
+          "a persisted record with a bad on_delete is still LOADED - dropping it "
+          "would silently remove protection");
+    check(got && got->onDelete == OnDelete::Restrict,
+          "and it falls back to restrict, the non-destructive direction");
+}
+
+
+// =========================================================================
+// v2.11.0 close-out — the `create_relation` MIGRATION OP.
+//
+// Declaring schema in migration files is how consumers ship views
+// (shadowman-cpp: /opt/shadowman/share/shadowman/migrations/json, callerai:
+// /etc/callerai/migrations), and until this op existed they could not declare a
+// relation at all. Modelled on create_view: same file shape, same idempotency,
+// and it goes through RelationManager::createRelation so the same-project rule
+// and the on_delete validation apply rather than being bypassed.
+// =========================================================================
+
+std::filesystem::path makeMigrationDir(const std::string& tag) {
+    auto dir = std::filesystem::temp_directory_path() /
+               ("mig-relation-" + tag + "-" + std::to_string(::getpid()));
+    std::filesystem::remove_all(dir);
+    std::filesystem::create_directories(dir);
+    return dir;
+}
+
+void writeMigration(const std::filesystem::path& dir, const std::string& file,
+                    const std::string& body) {
+    std::ofstream out(dir / file);
+    out << body;
+}
+
+void test_create_relation_migration_op_declares_and_is_idempotent() {
+    auto dir = makeMigrationDir("basic");
+    writeMigration(dir, "001_relations.json", R"({
+      "version": "001",
+      "name": "declare_exec_wf",
+      "operations": [
+        {"type": "create_collection", "collection": "workflows"},
+        {"type": "create_collection", "collection": "executions"},
+        {"type": "create_relation",
+         "name": "exec_wf",
+         "child": "executions",
+         "child_field": "workflowId",
+         "parent": "workflows",
+         "on_delete": "cascade",
+         "validate_on_write": true}
+      ]
+    })");
+
+    Fixture f;
+    ViewManager vm(f.store);
+    RelationManager rm(f.store);
+    MigrationRunner::Config cfg;
+    cfg.directory = dir;
+
+    {
+        MigrationRunner runner(f.store, vm, rm, cfg);
+        check(runner.runMigrations(), "the migration ran");
+    }
+
+    // Bare names in a migration file qualify to `default:`, exactly as every
+    // other collection name in a migration file does.
+    auto got = rm.getRelation("default:exec_wf");
+    check(got.has_value(), "the create_relation op DECLARED the relation");
+    if (got) {
+        check(got->child == "default:executions", "child is project-qualified");
+        check(got->childField == "workflowId", "child_field carried through");
+        check(got->parent == "default:workflows", "parent is project-qualified");
+        check(got->onDelete == OnDelete::Cascade, "on_delete was parsed, not defaulted");
+        check(got->validateOnWrite, "validate_on_write carried through");
+    }
+
+    // Idempotency has TWO layers and both matter. The runner skips an
+    // already-applied migration file, so re-running is a no-op at that level;
+    // the op itself must ALSO tolerate "already exists", which is what a
+    // consumer re-shipping the same declaration under a new version number
+    // hits.
+    {
+        MigrationRunner runner(f.store, vm, rm, cfg);
+        check(runner.runMigrations(), "re-running the same migrations still succeeds");
+    }
+    writeMigration(dir, "002_again.json", R"({
+      "version": "002",
+      "name": "declare_exec_wf_again",
+      "operations": [
+        {"type": "create_relation",
+         "name": "exec_wf", "child": "executions",
+         "child_field": "workflowId", "parent": "workflows",
+         "on_delete": "cascade"}
+      ]
+    })");
+    {
+        MigrationRunner runner(f.store, vm, rm, cfg);
+        check(runner.runMigrations(),
+              "a SECOND migration re-declaring the same relation succeeds - "
+              "'already exists' is not a failure on replay");
+    }
+    check(rm.listRelations("default").size() == 1,
+          "and it did not duplicate the declaration");
+
+    std::filesystem::remove_all(dir);
+}
+
+void test_create_relation_migration_op_validates() {
+    // A typo'd on_delete must FAIL the migration, not silently arm restrict.
+    // The reason it matters more here than at the RPC: a typo in a file that
+    // ships in a deb would otherwise be wrong on every install, forever.
+    {
+        auto dir = makeMigrationDir("typo");
+        writeMigration(dir, "001_typo.json", R"({
+          "version": "001", "name": "typo",
+          "operations": [
+            {"type": "create_relation", "name": "bad_rel", "child": "executions",
+             "child_field": "workflowId", "parent": "workflows",
+             "on_delete": "Cascade"}
+          ]
+        })");
+        Fixture f;
+        ViewManager vm(f.store);
+        RelationManager rm(f.store);
+        MigrationRunner::Config cfg;
+        cfg.directory = dir;
+        MigrationRunner runner(f.store, vm, rm, cfg);
+        check(!runner.runMigrations(), "an unrecognised on_delete FAILS the migration");
+        check(!rm.getRelation("default:bad_rel").has_value(),
+              "and nothing was declared - not coerced to restrict");
+        std::filesystem::remove_all(dir);
+    }
+
+    // A cross-project declaration must be refused by RelationManager, which is
+    // the whole point of routing through createRelation rather than writing the
+    // `_relations` record directly.
+    {
+        auto dir = makeMigrationDir("xproj");
+        writeMigration(dir, "001_xproj.json", R"({
+          "version": "001", "name": "xproj",
+          "operations": [
+            {"type": "create_relation", "name": "a:rel", "child": "a:executions",
+             "child_field": "workflowId", "parent": "b:workflows"}
+          ]
+        })");
+        Fixture f;
+        ViewManager vm(f.store);
+        RelationManager rm(f.store);
+        MigrationRunner::Config cfg;
+        cfg.directory = dir;
+        MigrationRunner runner(f.store, vm, rm, cfg);
+        check(!runner.runMigrations(),
+              "a cross-project relation is refused through the migration op too");
+        check(!rm.getRelation("a:rel").has_value(), "and nothing was declared");
+        std::filesystem::remove_all(dir);
+    }
+
+    // Missing required fields fail rather than declaring a half-relation.
+    {
+        auto dir = makeMigrationDir("missing");
+        writeMigration(dir, "001_missing.json", R"({
+          "version": "001", "name": "missing",
+          "operations": [
+            {"type": "create_relation", "name": "half_rel", "child": "executions"}
+          ]
+        })");
+        Fixture f;
+        ViewManager vm(f.store);
+        RelationManager rm(f.store);
+        MigrationRunner::Config cfg;
+        cfg.directory = dir;
+        MigrationRunner runner(f.store, vm, rm, cfg);
+        check(!runner.runMigrations(), "a create_relation missing child_field/parent fails");
+        check(!rm.getRelation("default:half_rel").has_value(), "and declares nothing");
+        std::filesystem::remove_all(dir);
+    }
+
+    // Absent on_delete means the documented default, and is NOT an error.
+    {
+        auto dir = makeMigrationDir("default-od");
+        writeMigration(dir, "001_default.json", R"({
+          "version": "001", "name": "defaulted",
+          "operations": [
+            {"type": "create_relation", "name": "def_rel", "child": "executions",
+             "child_field": "workflowId", "parent": "workflows"}
+          ]
+        })");
+        Fixture f;
+        ViewManager vm(f.store);
+        RelationManager rm(f.store);
+        MigrationRunner::Config cfg;
+        cfg.directory = dir;
+        MigrationRunner runner(f.store, vm, rm, cfg);
+        check(runner.runMigrations(), "an absent on_delete is accepted");
+        auto got = rm.getRelation("default:def_rel");
+        check(got.has_value(), "and the relation is declared");
+        check(got && got->onDelete == OnDelete::Restrict, "with restrict, the documented default");
+        check(got && !got->validateOnWrite, "and validate_on_write defaulting to false");
+        std::filesystem::remove_all(dir);
+    }
+}
+
+int main() {
+    std::cout << "=== test_relation_manager ===\n";
+    test_relations_are_project_scoped_and_survive_reload();
+    test_cross_project_relation_is_refused();
+    test_more_than_one_page_of_relations_loads();
+    test_load_skips_a_cross_project_relation_written_by_hand();
+    test_bare_and_qualified_names_are_the_same_relation();
+    test_load_rekeys_a_legacy_bare_named_record();
+    test_on_delete_is_validated_not_coerced();
+    test_create_relation_migration_op_declares_and_is_idempotent();
+    test_create_relation_migration_op_validates();
+    std::cout << "passed: " << g_pass << ", failed: " << g_fail << "\n";
+    return g_fail == 0 ? 0 : 1;
+}

+ 267 - 0
tests/test_subdb_identity.cpp

@@ -41,6 +41,7 @@ using smartbotic::db::storage::LmdbEnv;
 using smartbotic::db::storage::LmdbEnvOpts;
 using smartbotic::db::storage::read_subdb_identity;
 using smartbotic::db::storage::ReadTxn;
+using smartbotic::db::storage::RelationRef;
 using smartbotic::db::storage::verify_subdb_identity;
 using smartbotic::db::storage::write_subdb_identity;
 using smartbotic::db::storage::WriteTxn;
@@ -1716,6 +1717,270 @@ void test_index_values() {
           "values, and returning the array would misstate what the key means");
 }
 
+// v2.11.0 T10 — the whole point of the txn-accepting overloads: one
+// transaction spanning TWO DIFFERENT collections (plus an index sub-db) must
+// be all-or-nothing. Task 12's atomic cascade depends on this - a parent
+// delete and its children's cleanup have to share one commit/abort, and the
+// no-txn put()/del() each open and commit their own WriteTxn, so they cannot
+// compose into one atomic unit at all.
+//
+// Written BEFORE the overloads exist, per the brief: this must fail to
+// compile/link until put(WriteTxn&, ...) and friends are added.
+void test_txn_accepting_writes_are_atomic_across_collections() {
+    TmpEnv t("txn-atomic");
+    LmdbDocumentStore store(t.env);
+    store.set_indexed_fields("parents", {"name"});
+    // A relation declared on 'children' as the CHILD side, so put(wtxn, ...)
+    // on 'children' also maintains a reverse-index posting - the brief's "no
+    // relation entry survives" half needs one declared to be testable at
+    // all, and it shares the identical to_cache mechanism as the index path.
+    store.set_relations("children", {RelationRef{"par_child", "parentId"}});
+
+    Document pd;
+    pd.id = "p1";
+    pd.collection = "parents";
+    pd.set_data(nlohmann::json{{"name", "alice"}});
+
+    Document cd;
+    cd.id = "c1";
+    cd.collection = "children";
+    cd.set_data(nlohmann::json{{"parentId", "p1"}});
+
+    // --- Abort path: neither document, nor the index entry, nor the
+    // relation posting, may survive. ---
+    {
+        WriteTxn wtxn(t.env);
+        std::vector<std::pair<std::string, unsigned int>> to_cache;
+        store.put(wtxn, "parents", "p1", pd, to_cache);
+        store.put(wtxn, "children", "c1", cd, to_cache);
+        wtxn.abort();
+        // to_cache is deliberately NOT applied to the process-wide dbi cache -
+        // see the header comment: caching before commit is exactly the v2.8.0
+        // bug. Nothing here should call cacheCommittedDbi.
+    }
+    check(!store.get("parents", "p1").has_value(),
+          "aborted txn: the parent document does not exist");
+    check(!store.get("children", "c1").has_value(),
+          "aborted txn: the child document does not exist");
+
+    // --- The collection must not be poisoned by the aborted transaction: a
+    // later write, scan, count and get on EITHER collection must still work.
+    // This is precisely the v2.8.0 failure mode - caching a handle before
+    // commit leaves a closed handle behind an aborted caller transaction.
+    // It also DOUBLES as the only way to make the index/relation-rollback
+    // checks below non-vacuous: before either collection has ever committed
+    // successfully, index_lookup_eq()/relation_index_children() report
+    // nullopt/empty purely because their sub-db was never cached - that
+    // would pass whether or not the aborted posting actually rolled back.
+    // Writing a NEW row under the SAME key values the aborted write used
+    // (name="alice", parentId="p1") warms those caches for real, so a
+    // survived posting from the aborted write would show up alongside it. ---
+    bool ok_put = true;
+    try {
+        store.put("parents", "p2", pd);   // same name: "alice" as the aborted p1
+    } catch (const std::exception&) {
+        ok_put = false;
+    }
+    check(ok_put, "a later write to 'parents' after the abort still works");
+
+    bool ok_put2 = true;
+    try {
+        store.put("children", "c2", cd);  // same parentId: "p1" as the aborted c1
+    } catch (const std::exception&) {
+        ok_put2 = false;
+    }
+    check(ok_put2, "a later write to 'children' after the abort still works");
+
+    check(store.get("parents", "p2").has_value(), "get() on 'parents' works after the abort");
+    check(store.get("children", "c2").has_value(), "get() on 'children' works after the abort");
+
+    smartbotic::database::Query q;
+    q.limit = 10;
+    check(store.scan("parents", q).documents.size() == 1,
+          "scan() on 'parents' works after the abort and sees only the post-abort write");
+    check(store.scan("children", q).documents.size() == 1,
+          "scan() on 'children' works after the abort and sees only the post-abort write");
+    check(store.count("parents") == 1, "count() on 'parents' is right after the abort");
+    check(store.count("children") == 1, "count() on 'children' is right after the abort");
+
+    // Now the real proof: the index for "alice" holds ONLY p2, and the
+    // relation's postings for parent "p1" hold ONLY c2. If the aborted
+    // write's postings (p1 under "alice", c1 under "p1") had survived, either
+    // of these would report two ids instead of one - unlike the vacuous
+    // "nullopt/empty" checks a cold cache would also produce.
+    auto idx = store.index_lookup_eq("parents", "name", nlohmann::json("alice"));
+    check(idx.has_value() && idx->size() == 1 && (*idx)[0] == "p2",
+          "aborted txn: the index under 'alice' holds only the post-abort "
+          "write (p2) - the aborted p1 posting did not survive");
+    auto children_of_p1 = store.relation_index_children("par_child", "p1", 10);
+    check(children_of_p1.size() == 1 && children_of_p1[0] == "c2",
+          "aborted txn: parent 'p1' has only the post-abort child (c2) in the "
+          "relation index - the aborted c1 posting did not survive");
+
+    // --- Commit path: both documents AND the index entry land together,
+    // once the caller re-primes to pick up the handles it chose not to
+    // cache itself (see the next block for why that step is mandatory in
+    // real callers). ---
+    {
+        WriteTxn wtxn(t.env);
+        std::vector<std::pair<std::string, unsigned int>> to_cache;
+        Document pd2;
+        pd2.id = "p3";
+        pd2.collection = "parents";
+        pd2.set_data(nlohmann::json{{"name", "bob"}});
+        Document cd2;
+        cd2.id = "c3";
+        cd2.collection = "children";
+        cd2.set_data(nlohmann::json{{"parentId", "p3"}});
+        store.put(wtxn, "parents", "p3", pd2, to_cache);
+        store.put(wtxn, "children", "c3", cd2, to_cache);
+        wtxn.commit();
+        check(to_cache.size() >= 2,
+              "put(wtxn,...) reports every handle it opened (collection dbis, "
+              "plus the index dbi since 'parents' declares one) for the caller "
+              "to cache after its own commit");
+    }
+    store.prime_dbi_cache();  // stand-in for the caching step a real caller does
+    check(store.get("parents", "p3").has_value(), "committed txn: parent document exists");
+    check(store.get("children", "c3").has_value(), "committed txn: child document exists");
+    auto idx2 = store.index_lookup_eq("parents", "name", nlohmann::json("bob"));
+    check(idx2.has_value() && idx2->size() == 1 && (*idx2)[0] == "p3",
+          "committed txn: the index entry for the new parent is there");
+
+    // --- The contract that makes handle-return necessary: after a commit
+    // through the txn-accepting overload, the caller MUST cache the handles
+    // it was given, because reads are cache-only (see try_open_for_read's
+    // comment: a cache miss reads as "no such collection" by design, so
+    // that read transactions never have to call mdb_dbi_open). Skip that
+    // step and freshly-committed data becomes unreadable - not corrupted,
+    // just invisible - for the life of the process. Demonstrated on a
+    // brand-new collection name never touched before this point. ---
+    {
+        WriteTxn wtxn(t.env);
+        std::vector<std::pair<std::string, unsigned int>> to_cache;
+        Document nd;
+        nd.id = "n1";
+        nd.collection = "brandnew";
+        nd.set_data(nlohmann::json{{"x", 1}});
+        store.put(wtxn, "brandnew", "n1", nd, to_cache);
+        wtxn.commit();
+        check(!to_cache.empty(),
+              "put(wtxn,...) hands back the handle it opened for 'brandnew'");
+        // Deliberately do NOT cache it - that is the point of this block.
+    }
+    check(!store.get("brandnew", "n1").has_value(),
+          "without caching after commit, the freshly-committed row reads as "
+          "absent - a cold cache, not lost data (proven next)");
+    store.prime_dbi_cache();
+    check(store.get("brandnew", "n1").has_value(),
+          "after priming (which caches it), the same row is visible - "
+          "confirming the earlier miss was purely a cold cache");
+}
+
+// v2.11.0 T10 review round 2 — the vector equivalents of the two defects
+// fixed on the document path (put/del) were still present on put_vector/
+// del_vector: late handle-recording (Finding 2) and a same-transaction
+// existence-probe gap (Finding 3). Both are reachable under the Task 12
+// cascade design documented in the task report, which shares ONE to_cache
+// across put/del/put_vector/del_vector for a parent and its children in a
+// single transaction — dormant only because nothing but the self-contained
+// no-txn wrappers calls the txn-accepting vector overloads yet.
+void test_txn_accepting_vector_writes_are_atomic() {
+    TmpEnv t("txn-atomic-vec");
+    LmdbDocumentStore store(t.env);
+
+    // --- Abort path: an aborted put_vector must leave no vector behind.
+    // Proven the same non-vacuous way as the document test: get_vector()
+    // on a never-committed sub-db returns nullopt purely from a cold cache
+    // (try_open_for_read is cache-only), so the follow-up write under the
+    // SAME collection is what makes the absence-of-'a1' check real - if the
+    // aborted key had survived, get_vector("vecsA", "a1") would find it via
+    // the now-warm cache, not report absent. ---
+    {
+        WriteTxn wtxn(t.env);
+        std::vector<std::pair<std::string, unsigned int>> to_cache;
+        store.put_vector(wtxn, "vecsA", "a1", {1.0f, 2.0f, 3.0f}, to_cache);
+        wtxn.abort();
+    }
+    bool ok_put = true;
+    try {
+        store.put_vector("vecsA", "a2", {9.0f, 9.0f, 9.0f});  // warms the cache for real
+    } catch (const std::exception&) {
+        ok_put = false;
+    }
+    check(ok_put, "a later put_vector on 'vecsA' after the abort still works - "
+                  "not poisoned by the aborted transaction");
+    check(!store.get_vector("vecsA", "a1").has_value(),
+          "aborted txn: 'a1' does not exist, checked against a genuinely "
+          "warm cache (via 'a2'), not a cold-cache false negative");
+    auto v2 = store.get_vector("vecsA", "a2");
+    check(v2.has_value() && v2->size() == 3 && (*v2)[0] == 9.0f,
+          "the post-abort vector write is readable");
+
+    // --- Finding 2 (put_vector): the handle must be recorded in to_cache
+    // BEFORE mdb_put runs, not after - so a throw from mdb_put does not
+    // silently drop the collection's own handle from what the caller was
+    // told to cache. Reproduced with the same oversized-key technique as
+    // test_aborted_write_does_not_poison_the_collection: a key past LMDB's
+    // 511-byte limit makes mdb_put fail MDB_BAD_VALSIZE. ---
+    {
+        WriteTxn wtxn(t.env);
+        std::vector<std::pair<std::string, unsigned int>> to_cache;
+        const std::string huge_id(600, 'v');
+        bool threw = false;
+        try {
+            store.put_vector(wtxn, "vecsB", huge_id, {1.0f}, to_cache);
+        } catch (const std::exception&) {
+            threw = true;
+        }
+        check(threw, "an oversized key really does fail put_vector's mdb_put");
+        check(!to_cache.empty(),
+              "put_vector(wtxn,...) recorded the sub-db handle BEFORE the "
+              "failing mdb_put ran, not after - so to_cache is not silently "
+              "missing it");
+        wtxn.abort();
+    }
+    bool ok_put_b = true;
+    try {
+        store.put_vector("vecsB", "b1", {4.0f, 5.0f});
+    } catch (const std::exception&) {
+        ok_put_b = false;
+    }
+    check(ok_put_b, "'vecsB' is not poisoned by the failed put_vector either");
+    check(store.get_vector("vecsB", "b1").has_value(),
+          "and the post-failure vector write is readable");
+
+    // --- Finding 3 (del_vector): a vector sub-db opened by put_vector(wtxn,
+    // ...) earlier in the SAME still-uncommitted transaction must be
+    // deletable by del_vector(wtxn, ...) sharing that to_cache - cachedDbi()
+    // alone only sees already-committed sub-dbs, so without the to_cache
+    // fallback this would silently no-op instead of deleting. ---
+    bool existed = false;
+    {
+        WriteTxn wtxn(t.env);
+        std::vector<std::pair<std::string, unsigned int>> to_cache;
+        store.put_vector(wtxn, "vecsC", "c1", {7.0f, 7.0f}, to_cache);
+        existed = store.del_vector(wtxn, "vecsC", "c1", to_cache);
+        wtxn.commit();
+        for (const auto& [sub, d] : to_cache) {
+            // Stand-in for the caller's post-commit caching step - not
+            // strictly needed for this assertion (get_vector below would
+            // read as absent for a deleted key from a cold cache too, same
+            // as a genuinely-deleted one), but exercised for parity with
+            // how a real caller (Task 12) must use this.
+            (void)sub;
+            (void)d;
+        }
+    }
+    check(existed,
+          "del_vector(wtxn, ...) found and deleted 'c1' within the SAME "
+          "transaction that created it, rather than treating the "
+          "not-yet-committed sub-db as nonexistent and no-op'ing");
+    store.prime_dbi_cache();
+    check(!store.get_vector("vecsC", "c1").has_value(),
+          "'c1' is genuinely gone after the commit, not merely unreported");
+}
+
 }  // namespace
 
 int main() {
@@ -1729,6 +1994,8 @@ int main() {
     test_scan_limit_zero_reports_total();
     test_scan_fast_path_matches_general_path();
     test_aborted_write_does_not_poison_the_collection();
+    test_txn_accepting_writes_are_atomic_across_collections();
+    test_txn_accepting_vector_writes_are_atomic();
     test_filtered_scan_operator_matrix();
     test_concurrent_reads_do_not_rebind_cached_handles();
     test_existing_collection_readable_without_writing_first();

+ 73 - 1
tests/test_timestamp_precision.cpp

@@ -20,6 +20,7 @@
 #include <cstdint>
 #include <iostream>
 #include <nlohmann/json.hpp>
+#include <optional>
 #include <set>
 #include <string>
 
@@ -31,6 +32,18 @@ using smartbotic::database::MemoryStore;
 
 namespace {
 
+int g_pass = 0;
+int g_fail = 0;
+
+void check(bool cond, const char* msg) {
+    if (cond) {
+        ++g_pass;
+    } else {
+        ++g_fail;
+        std::cerr << "FAIL: " << msg << "\n";
+    }
+}
+
 // Anything below 10^15 fits comfortably in the ms-since-epoch range (≈ 2001..),
 // anything above is unambiguously ns-since-epoch in the current era.
 constexpr uint64_t NS_THRESHOLD = 1'000'000'000'000'000ULL;  // 10^15
@@ -49,6 +62,27 @@ Document makeDoc(const std::string& id = "") {
     return d;
 }
 
+// Mirrors the partial-update shape of DatabaseGrpcImpl::ConfigureCollection
+// (service/src/database_grpc_impl.cpp): start from the collection's current
+// config and overlay only the fields the caller actually set. An empty
+// precision string and a nullopt relations_enforced both mean "leave
+// unchanged" here, exactly as they do on the wire.
+void configureCollection(CollectionConfigManager& mgr,
+                          const std::string& collection,
+                          const std::string& precision,
+                          std::optional<bool> relationsEnforced) {
+    CollectionCfg cfg = mgr.configFor(collection);
+    if (!precision.empty()) {
+        cfg.timestampPrecision = precision;
+    }
+    if (relationsEnforced.has_value()) {
+        cfg.relationsEnforced = relationsEnforced.value();
+    }
+    std::string err;
+    bool ok = mgr.setConfig(collection, cfg, err);
+    check(ok, "setConfig should succeed for a partial-update configure call");
+}
+
 } // anonymous namespace
 
 void test_default_is_ns() {
@@ -195,6 +229,38 @@ void test_default_for_unknown_collection() {
     std::cout << "PASS: default config returned for unknown collection\n";
 }
 
+// v2.11.0 T8 — relations_enforced is a partial update, same shape as
+// versioning_enabled (v2.4.5). Declaring a relation names its child and
+// parent explicitly, so the declaration IS the opt-in: the default must be
+// enforced, and an unrelated configure call (e.g. precision-only) must never
+// silently disable it.
+void test_relations_enforced_is_a_partial_update() {
+    MemoryStore store(defaultConfig());
+    store.start();
+    CollectionConfigManager mgr(store);
+    store.setConfigManager(&mgr);
+
+    CollectionCfg cfg = mgr.configFor("c");
+    check(cfg.relationsEnforced,
+          "defaults to enforced - declaring a relation IS the opt-in, so a "
+          "declared constraint must not silently do nothing");
+
+    // Change ONLY precision. A plain proto3 bool defaults to false and would
+    // silently disable enforcement here - the exact trap v2.4.5 hit with
+    // versioning_enabled, which is why the field is `optional`.
+    configureCollection(mgr, "c", /*precision=*/"ms", /*relationsEnforced=*/std::nullopt);
+    check(mgr.configFor("c").relationsEnforced,
+          "a precision-only call leaves enforcement alone");
+
+    configureCollection(mgr, "c", "", /*relationsEnforced=*/false);
+    check(!mgr.configFor("c").relationsEnforced, "and it can be turned off");
+    check(mgr.configFor("c").timestampPrecision == "ms",
+          "without resetting precision");
+
+    store.stop();
+    std::cout << "PASS: relations_enforced is a partial update\n";
+}
+
 int main() {
     test_default_is_ns();
     test_ns_yields_unique_timestamps_in_tight_loop();
@@ -202,6 +268,12 @@ int main() {
     test_idempotent_configure();
     test_cache_reflects_configure_immediately();
     test_default_for_unknown_collection();
-    std::cout << "\nAll timestamp precision tests PASSED!\n";
+    test_relations_enforced_is_a_partial_update();
+
+    if (g_fail > 0) {
+        std::cerr << "\n" << g_fail << " check(s) FAILED (" << g_pass << " passed)\n";
+        return 1;
+    }
+    std::cout << "\nAll timestamp precision tests PASSED! (" << g_pass << " checks)\n";
     return 0;
 }

이 변경점에서 너무 많은 파일들이 변경되어 몇몇 파일들은 표시되지 않았습니다.