Quellcode durchsuchen

feat(relations): T13 validate_on_write closes the restrict race

A child write to a collection with a validateOnWrite=true relation is now
rejected if the referenced parent does not exist, checked via mdb_get on
the parent's own sub-db inside the child's own write transaction. Since
LMDB serialises writers, the child's transaction cannot begin until any
prior parent-delete transaction has committed, so there is no interval
between the check and the write for the parent to vanish in - closing the
restrict race the design documents as a known limitation.

- RelationRef gains `parent` (bare collection) and `validateOnWrite`,
  populated at both construction sites (armRelationsForChild,
  applyRelationDeclarations).
- New MissingParentReference exception, same treatment as UniqueViolation:
  rethrown untouched through applyDualWriteMirror and mirrorDocOrUndo (no
  mirror-health/drift impact, MemoryStore undoes its mutation), mapped to
  FAILED_PRECONDITION at all 8 gRPC handler sites, and swallowed-and-flipped
  on the replication-apply path for the same reason UniqueViolation is.
- Array-valued references: any missing parent id rejects the whole write
  (fail closed, no partial index state left behind). Only newly-introduced
  references (to_add) are checked, not unchanged ones. Absent/null
  references are never checked.
- 7 new tests in test_relation_enforcement.cpp, including one that proves
  the race is closed: a parent observed present by an earlier read, then
  deleted, then a child write referencing it is rejected against
  write-time state rather than the stale earlier read.

Verified by reverting the guard and confirming the 13 dependent assertions
fail; restored and reran clean.
fszontagh vor 1 Monat
Ursprung
Commit
452a2f9522

+ 30 - 1
service/src/database_grpc_impl.cpp

@@ -416,6 +416,12 @@ grpc::Status DatabaseGrpcImpl::Insert(
         // fault. MemoryStore has already undone its own mutation by the
         // time this propagates here (see mirrorDocOrUndo).
         return grpc::Status(grpc::StatusCode::ALREADY_EXISTS, e.what());
+    } catch (const smartbotic::db::storage::MissingParentReference& e) {
+        // v2.11.0 T13 — a validate_on_write rejection: the referenced
+        // parent does not exist. Not a server fault either, and not
+        // ALREADY_EXISTS - FAILED_PRECONDITION is the right status for "the
+        // system is not in a state this write requires".
+        return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, e.what());
     } catch (const std::exception& e) {
         return grpc::Status(grpc::StatusCode::INTERNAL, e.what());
     }
@@ -612,6 +618,9 @@ grpc::Status DatabaseGrpcImpl::Update(
     } catch (const smartbotic::db::storage::UniqueViolation& e) {
         // v2.11.0 T11 — see the Insert handler's note.
         return grpc::Status(grpc::StatusCode::ALREADY_EXISTS, e.what());
+    } catch (const smartbotic::db::storage::MissingParentReference& e) {
+        // v2.11.0 T13 — see the Insert handler's note.
+        return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, e.what());
     } catch (const std::exception& e) {
         return grpc::Status(grpc::StatusCode::INTERNAL, e.what());
     }
@@ -681,6 +690,9 @@ grpc::Status DatabaseGrpcImpl::PatchDocument(
     } catch (const smartbotic::db::storage::UniqueViolation& e) {
         // v2.11.0 T11 — see the Insert handler's note.
         return grpc::Status(grpc::StatusCode::ALREADY_EXISTS, e.what());
+    } catch (const smartbotic::db::storage::MissingParentReference& e) {
+        // v2.11.0 T13 — see the Insert handler's note.
+        return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, e.what());
     } catch (const std::exception& e) {
         return grpc::Status(grpc::StatusCode::INTERNAL, e.what());
     }
@@ -780,6 +792,9 @@ grpc::Status DatabaseGrpcImpl::Upsert(
     } catch (const smartbotic::db::storage::UniqueViolation& e) {
         // v2.11.0 T11 — see the Insert handler's note.
         return grpc::Status(grpc::StatusCode::ALREADY_EXISTS, e.what());
+    } catch (const smartbotic::db::storage::MissingParentReference& e) {
+        // v2.11.0 T13 — see the Insert handler's note.
+        return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, e.what());
     } catch (const std::exception& e) {
         return grpc::Status(grpc::StatusCode::INTERNAL, e.what());
     }
@@ -1111,6 +1126,8 @@ grpc::Status DatabaseGrpcImpl::RestoreVersion(
         return grpc::Status::OK;
     } catch (const smartbotic::db::storage::UniqueViolation& e) {
         return grpc::Status(grpc::StatusCode::ALREADY_EXISTS, e.what());
+    } catch (const smartbotic::db::storage::MissingParentReference& e) {
+        return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, e.what());
     } catch (const std::exception& e) {
         return grpc::Status(grpc::StatusCode::INTERNAL, e.what());
     }
@@ -1157,6 +1174,8 @@ grpc::Status DatabaseGrpcImpl::RestoreToDate(
         return grpc::Status::OK;
     } catch (const smartbotic::db::storage::UniqueViolation& e) {
         return grpc::Status(grpc::StatusCode::ALREADY_EXISTS, e.what());
+    } catch (const smartbotic::db::storage::MissingParentReference& e) {
+        return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, e.what());
     } catch (const std::exception& e) {
         return grpc::Status(grpc::StatusCode::INTERNAL, e.what());
     }
@@ -1676,6 +1695,8 @@ grpc::Status DatabaseGrpcImpl::SetAdd(
         return grpc::Status::OK;
     } catch (const smartbotic::db::storage::UniqueViolation& e) {
         return grpc::Status(grpc::StatusCode::ALREADY_EXISTS, e.what());
+    } catch (const smartbotic::db::storage::MissingParentReference& e) {
+        return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, e.what());
     } catch (const std::exception& e) {
         return grpc::Status(grpc::StatusCode::INTERNAL, e.what());
     }
@@ -1707,6 +1728,8 @@ grpc::Status DatabaseGrpcImpl::SetRemove(
         return grpc::Status::OK;
     } catch (const smartbotic::db::storage::UniqueViolation& e) {
         return grpc::Status(grpc::StatusCode::ALREADY_EXISTS, e.what());
+    } catch (const smartbotic::db::storage::MissingParentReference& e) {
+        return grpc::Status(grpc::StatusCode::FAILED_PRECONDITION, e.what());
     } catch (const std::exception& e) {
         return grpc::Status(grpc::StatusCode::INTERNAL, e.what());
     }
@@ -2842,7 +2865,13 @@ void DatabaseGrpcImpl::armRelationsForChild(const std::string& childQualified) {
         std::vector<smartbotic::db::storage::RelationRef> refs;
         for (const auto& rel : relation_manager_.relationsWithChild(childQualified)) {
             const auto rn = smartbotic::database::resolveCollection(rel.name);
-            refs.push_back(smartbotic::db::storage::RelationRef{rn.collection, rel.childField});
+            // v2.11.0 T13 — parent is resolved to its BARE collection name,
+            // same as name/childField above: relations never cross projects
+            // (relation_manager.hpp refuses it at declaration time), so the
+            // parent is guaranteed to live in this same project's env.
+            const auto pc = smartbotic::database::resolveCollection(rel.parent);
+            refs.push_back(smartbotic::db::storage::RelationRef{
+                rn.collection, rel.childField, pc.collection, rel.validateOnWrite});
         }
         lmdb->set_relations(rc.collection, refs);
         spdlog::info("v2.11 relations: re-armed {} relation(s) for child '{}'",

+ 25 - 2
service/src/database_service.cpp

@@ -404,9 +404,12 @@ void DatabaseService::applyRelationDeclarations() {
         try {
             const auto rn = resolveCollection(r.name);
             const auto rc = resolveCollection(r.child);
+            // v2.11.0 T13 — parent resolved to its bare name, same reasoning
+            // as armRelationsForChild's mirror of this construction.
+            const auto pc = resolveCollection(r.parent);
             const std::string key = rc.project + ":" + rc.collection;
-            byChild[key].push_back(
-                smartbotic::db::storage::RelationRef{rn.collection, r.childField});
+            byChild[key].push_back(smartbotic::db::storage::RelationRef{
+                rn.collection, r.childField, pc.collection, r.validateOnWrite});
             keyToProjectCollection[key] = {rc.project, rc.collection};
         } catch (const std::exception& e) {
             // Advisory per relation: one unparseable declaration must not
@@ -1377,6 +1380,26 @@ void DatabaseService::applyReplicatedEntry(const databasepb::ReplicationEntry& e
                                 entry.collection(), doc.id, e.what());
                             mirror_healthy_.store(false, std::memory_order_release);
                             mirror_drift_count_.fetch_add(1, std::memory_order_relaxed);
+                        } catch (const smartbotic::db::storage::MissingParentReference& e) {
+                            // v2.11.0 T13 — same deliberate swallow-and-flip as
+                            // the UniqueViolation catch just above, for the same
+                            // reason: loadDocument() has already committed this
+                            // row into MemoryStore, the origin already accepted
+                            // the write, and this follower may not even have the
+                            // same validate_on_write declaration (or the same
+                            // parent data) as the origin did at the time it
+                            // wrote. MemoryStore keeps the row; the mirror is
+                            // marked unhealthy on purpose so an operator sees
+                            // the drift and reads fall back to MemoryStore
+                            // rather than the two stores silently disagreeing.
+                            spdlog::error(
+                                "v2.11 replication: validate_on_write rejected applying "
+                                "{}/{}: {} - MemoryStore holds the row, LMDB does not; "
+                                "marking the mirror unhealthy so reads fall back instead "
+                                "of silently disagreeing with MemoryStore",
+                                entry.collection(), doc.id, e.what());
+                            mirror_healthy_.store(false, std::memory_order_release);
+                            mirror_drift_count_.fetch_add(1, std::memory_order_relaxed);
                         }
                     }
                 }

+ 7 - 0
service/src/memory_store.cpp

@@ -2830,6 +2830,13 @@ void MemoryStore::mirrorDocOrUndo(const std::string& collection, const std::stri
     } catch (const smartbotic::db::storage::UniqueViolation&) {
         undo();
         throw;
+    } catch (const smartbotic::db::storage::MissingParentReference&) {
+        // v2.11.0 T13 — same undo-then-rethrow treatment as UniqueViolation:
+        // a validate_on_write rejection is the caller's business, not a
+        // mirror fault, and MemoryStore must not keep the row it already
+        // wrote in-memory before the mirror ran.
+        undo();
+        throw;
     }
 }
 

+ 45 - 0
service/src/storage/document_store_lmdb.cpp

@@ -811,6 +811,51 @@ void LmdbDocumentStore::maintainRelations(
                             std::back_inserter(to_add));
         if (to_remove.empty() && to_add.empty()) continue;
 
+        // v2.11.0 T13 — validate_on_write: every id newly appearing in
+        // `to_add` must name a parent that already exists, checked with an
+        // mdb_get on the parent's own sub-db INSIDE this same write
+        // transaction. Only `to_add` is checked, not the full `new_ids` set -
+        // a reference that did not change this write already existed (or was
+        // already dangling) before this write and is not what this task
+        // closes the race for; re-validating it on every unrelated field
+        // update would also defeat the "unchanged reference costs no write"
+        // short-circuit above.
+        //
+        // Array-valued reference decision: ANY missing parent id rejects the
+        // WHOLE write, not just that element. Silently keeping the
+        // resolvable elements while dropping the unresolvable ones would
+        // discard data the caller explicitly supplied without telling them,
+        // and would make the same array partially "valid" depending on write
+        // order - the same fail-closed, all-or-nothing reasoning
+        // UniqueViolation already applies to the rest of the document.
+        //
+        // Absent/null references never reach here: extract_relation_ids /
+        // resolveFilterValue never put them in `new_ids`, so they can never
+        // land in `to_add`.
+        if (r.validateOnWrite && !to_add.empty()) {
+            // open_for_write, not try_open_for_read: this call happens
+            // inside our own write transaction (the only kind that may call
+            // mdb_dbi_open - see try_open_for_read's file comment), and
+            // passing MDB_CREATE is harmless here - if the parent collection
+            // truly has never been written, mdb_get below finds nothing for
+            // every id, this throws, and the whole transaction (including
+            // that dbi's creation) aborts with it. The parent's own handle
+            // is queued into `to_cache` like every other handle this
+            // function opens, so a later child write reuses the cached one
+            // once THIS write commits.
+            const unsigned int parent_dbi = open_for_write(wtxn, r.parent);
+            to_cache.emplace_back(r.parent, parent_dbi);
+            for (const auto& parentId : to_add) {
+                MDB_val pk = to_val(parentId);
+                MDB_val pv{0, nullptr};
+                const int prc = mdb_get(wtxn.raw(), parent_dbi, &pk, &pv);
+                if (prc == MDB_NOTFOUND) {
+                    throw MissingParentReference(r.name, r.childField, parentId);
+                }
+                if (prc != MDB_SUCCESS) throw_mdb(prc, "get (validate_on_write)");
+            }
+        }
+
         const std::string sub = relation_index_subdb(r.name);
         const unsigned int dbi = open_for_write(wtxn, sub, MDB_DUPSORT);
         to_cache.emplace_back(sub, dbi);

+ 37 - 0
service/src/storage/document_store_lmdb.hpp

@@ -46,6 +46,35 @@ public:
     std::string existing_id;
 };
 
+// v2.11.0 T13 — thrown when a write to a child collection introduces a
+// reference (via a relation declared with validateOnWrite=true) to a
+// parent id that does not exist. Checked with an mdb_get on the parent's own
+// sub-db INSIDE the child's write transaction (see maintainRelations) - not
+// before it, not after - which is what closes the restrict race: LMDB
+// serialises writers, so this transaction cannot begin until any transaction
+// that deleted the parent has already committed, and there is no interval
+// between the check and this write's own commit for the parent to vanish in.
+//
+// A distinct type, same reasoning as UniqueViolation: the gRPC layer answers
+// FAILED_PRECONDITION instead of INTERNAL, and applyDualWriteMirror /
+// mirrorDocOrUndo must rethrow it untouched rather than treat it as a mirror
+// fault - see UniqueViolation's comment on set_unique_fields for why that
+// distinction matters.
+class MissingParentReference : public std::runtime_error {
+public:
+    MissingParentReference(std::string relation, std::string childField,
+                            std::string parentId)
+        : std::runtime_error("relation '" + relation + "': field '" + childField +
+                             "' references parent id '" + parentId +
+                             "' which does not exist"),
+          relation(std::move(relation)),
+          childField(std::move(childField)),
+          parentId(std::move(parentId)) {}
+    std::string relation;
+    std::string childField;
+    std::string parentId;
+};
+
 // v2.11.0 T3 — a child-side relation reference, as the storage layer needs
 // it to maintain the reverse index on a write. Deliberately NOT
 // RelationInfo (relations/relation_manager.hpp): this layer stays free of
@@ -54,9 +83,17 @@ public:
 // project); `childField` is the dot-path on the CHILD document (the
 // collection this ref is declared against) that holds the parent id, or an
 // array of them.
+//
+// v2.11.0 T13 — `parent` is the BARE parent collection name (same project as
+// the child - see relation_manager.hpp's cross-project refusal), and
+// `validateOnWrite` mirrors RelationInfo::validateOnWrite. Both stay
+// optional-by-default (empty / false) so every pre-T13 aggregate-init call
+// site (`RelationRef{name, childField}`) keeps compiling unchanged.
 struct RelationRef {
     std::string name;
     std::string childField;
+    std::string parent;
+    bool validateOnWrite = false;
 };
 
 class LmdbDocumentStore : public DocumentStore {

+ 7 - 0
service/src/storage/dual_write_mirror.hpp

@@ -69,6 +69,13 @@ inline void applyDualWriteMirror(
         // can undo its own in-memory mutation, and so it eventually reaches
         // the gRPC handler as ALREADY_EXISTS rather than INTERNAL.
         throw;
+    } catch (const MissingParentReference&) {
+        // v2.11.0 T13 — same reasoning as the UniqueViolation catch above,
+        // for the same reason: a rejected write because its parent does not
+        // exist is the caller's business, not a mirror fault. Rethrown as-is
+        // so MemoryStore undoes its own mutation and the gRPC handler
+        // answers FAILED_PRECONDITION rather than INTERNAL.
+        throw;
     } catch (const std::exception& e) {
         spdlog::error("v2.0 mirror failed coll={} id={} op={}: {}",
                       collection, id, static_cast<int>(eventType), e.what());

+ 294 - 0
tests/test_relation_enforcement.cpp

@@ -60,6 +60,7 @@ using smartbotic::database::writeCascadeWal;
 using smartbotic::db::storage::LmdbDocumentStore;
 using smartbotic::db::storage::LmdbEnv;
 using smartbotic::db::storage::LmdbEnvOpts;
+using smartbotic::db::storage::MissingParentReference;
 using smartbotic::db::storage::RelationRef;
 
 namespace {
@@ -1220,6 +1221,292 @@ void test_reinsert_after_delete_converges_in_lmdb() {
     freshStore.stop();
 }
 
+// -------------------------------------------------------------------------
+// v2.11.0 T13 — validate_on_write.
+//
+// The mitigation for the restrict race documented in relation_cascade.hpp
+// and the plan's self-review: a `restrict` check on delete runs in its own
+// read, so a child insert racing that check can still commit after the
+// parent is gone (LMDB serialises the two transactions, but the loser just
+// commits second). validate_on_write closes it by re-checking the parent's
+// existence with an mdb_get on the parent's own sub-db INSIDE the child's
+// write transaction - see LmdbDocumentStore::maintainRelations. What makes
+// it a genuine fix rather than a narrower window: the check and the write
+// commit as one unit, so there is no interval between them for the parent
+// to vanish in.
+// -------------------------------------------------------------------------
+
+void test_validate_on_write_rejects_missing_parent() {
+    TmpEnv t("validate-missing");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions",
+                        {{"exec_wf", "workflowId", "workflows", true}});
+
+    Document d; d.id = "e1"; d.collection = "executions";
+    d.set_data({{"workflowId", "wf-ghost"}});
+
+    bool threw = false;
+    std::string what;
+    try {
+        store.put("executions", "e1", d);
+    } catch (const MissingParentReference& e) {
+        threw = true;
+        what = e.what();
+    }
+    check(threw, "validateOnWrite=true rejects a reference to a nonexistent parent");
+    check(what.find("exec_wf") != std::string::npos, "error names the relation");
+    check(what.find("workflowId") != std::string::npos, "error names the child field");
+    check(what.find("wf-ghost") != std::string::npos, "error names the missing parent id");
+
+    // A rejected write must leave no trace - not the document, not the
+    // reverse index posting.
+    check(!store.get("executions", "e1").has_value(),
+          "rejected insert left no document behind");
+    check(store.relation_index_child_count("exec_wf", "wf-ghost") == 0,
+          "rejected insert left no reverse-index posting behind");
+}
+
+void test_validate_on_write_accepts_existing_parent() {
+    TmpEnv t("validate-present");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions",
+                        {{"exec_wf", "workflowId", "workflows", true}});
+
+    Document w; w.id = "wf-1"; w.collection = "workflows";
+    w.set_data({{"name", "real workflow"}});
+    store.put("workflows", "wf-1", w);
+
+    Document d; d.id = "e1"; d.collection = "executions";
+    d.set_data({{"workflowId", "wf-1"}});
+    bool threw = false;
+    try {
+        store.put("executions", "e1", d);
+    } catch (const MissingParentReference&) {
+        threw = true;
+    }
+    check(!threw, "validateOnWrite=true accepts a reference to an existing parent");
+    check(store.get("executions", "e1").has_value(), "the child document was actually written");
+    check(store.relation_index_child_count("exec_wf", "wf-1") == 1,
+          "and the reverse index posting was written");
+}
+
+// With validateOnWrite=false (the default), the same missing-parent insert
+// succeeds and check_relation_dangling reports it - the brief's Step 1 case,
+// both halves.
+void test_validate_on_write_false_allows_dangling_and_check_reports_it() {
+    TmpEnv t("validate-off");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions",
+                        {{"exec_wf", "workflowId", "workflows", false}});
+
+    Document d; d.id = "e1"; d.collection = "executions";
+    d.set_data({{"workflowId", "wf-ghost"}});
+    bool threw = false;
+    try {
+        store.put("executions", "e1", d);
+    } catch (const MissingParentReference&) {
+        threw = true;
+    }
+    check(!threw, "validateOnWrite=false lets the insert through");
+    check(store.get("executions", "e1").has_value(), "the dangling child document was written");
+
+    auto result = store.check_relation_dangling("exec_wf", "workflows", 100);
+    check(result.total == 1, "relations check reports one dangling parent id");
+    check(result.entries.size() == 1 && result.entries[0].parentId == "wf-ghost",
+          "and it is the ghost id");
+    check(result.entries[0].childCount == 1 &&
+          !result.entries[0].sampleChildIds.empty() &&
+          result.entries[0].sampleChildIds[0] == "e1",
+          "naming the dangling child");
+}
+
+// Absent and null references are not references at all (same rule as T3's
+// index maintenance) - they must never be rejected, even under
+// validateOnWrite=true.
+void test_validate_on_write_never_rejects_absent_or_null() {
+    TmpEnv t("validate-absent-null");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions",
+                        {{"exec_wf", "workflowId", "workflows", true}});
+
+    Document d1; d1.id = "e1"; d1.collection = "executions";
+    d1.set_data({{"other", 1}});   // workflowId absent entirely
+    bool threw1 = false;
+    try {
+        store.put("executions", "e1", d1);
+    } catch (const MissingParentReference&) {
+        threw1 = true;
+    }
+    check(!threw1, "an absent reference field is never rejected");
+
+    Document d2; d2.id = "e2"; d2.collection = "executions";
+    d2.set_data({{"workflowId", nullptr}});
+    bool threw2 = false;
+    try {
+        store.put("executions", "e2", d2);
+    } catch (const MissingParentReference&) {
+        threw2 = true;
+    }
+    check(!threw2, "an explicit null reference is never rejected");
+}
+
+// Array-valued reference decision: ANY missing parent id rejects the WHOLE
+// write, not just that element - fail closed, and no partial state (some
+// elements resolvable, others not) is left behind.
+void test_validate_on_write_array_any_missing_rejects_whole_write() {
+    TmpEnv t("validate-array");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("nodes",
+                        {{"node_creds", "config.credentialIds", "credentials", true}});
+
+    Document c1; c1.id = "c1"; c1.collection = "credentials";
+    c1.set_data({{"name", "real cred"}});
+    store.put("credentials", "c1", c1);
+    // c2 is deliberately never created.
+
+    Document n; n.id = "n1"; n.collection = "nodes";
+    n.set_data({{"config", {{"credentialIds", {"c1", "c2"}}}}});
+    bool threw = false;
+    std::string what;
+    try {
+        store.put("nodes", "n1", n);
+    } catch (const MissingParentReference& e) {
+        threw = true;
+        what = e.what();
+    }
+    check(threw, "one missing element in an array reference rejects the whole write");
+    check(what.find("c2") != std::string::npos, "error names the missing element, not the valid one");
+    check(!store.get("nodes", "n1").has_value(),
+          "rejected write left no document behind");
+    check(store.relation_index_child_count("node_creds", "c1") == 0,
+          "rejected write left no posting for the VALID element either - "
+          "no partial index state from a rejected write");
+    check(store.relation_index_child_count("node_creds", "c2") == 0,
+          "and none for the missing one");
+
+    // With every element resolvable, the write goes through and every
+    // element gets its posting.
+    Document c2; c2.id = "c2"; c2.collection = "credentials";
+    c2.set_data({{"name", "the missing one, now created"}});
+    store.put("credentials", "c2", c2);
+    threw = false;
+    try {
+        store.put("nodes", "n1", n);
+    } catch (const MissingParentReference&) {
+        threw = true;
+    }
+    check(!threw, "once every element resolves, the write succeeds");
+    check(store.relation_index_child_count("node_creds", "c1") == 1, "posting for c1");
+    check(store.relation_index_child_count("node_creds", "c2") == 1, "posting for c2");
+}
+
+// An update that does not touch the reference field is not re-validated,
+// even if the previously-written reference has since gone dangling (e.g. the
+// parent was removed out from under a no_action/validateOnWrite=false-era
+// row, or validateOnWrite was turned on after the fact). Only NEWLY
+// introduced references (`to_add`) are checked - see maintainRelations'
+// comment for why: an unchanged reference already existed (or was already
+// dangling) before this write, and this task closes the race for writes
+// that introduce a reference, not for the mere fact that time has passed
+// since one was previously accepted.
+void test_validate_on_write_unrelated_update_not_rechecked() {
+    TmpEnv t("validate-unrelated-update");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions",
+                        {{"exec_wf", "workflowId", "workflows", true}});
+
+    Document w; w.id = "wf-1"; w.collection = "workflows";
+    w.set_data({{"name", "real"}});
+    store.put("workflows", "wf-1", w);
+
+    Document d1; d1.id = "e1"; d1.collection = "executions";
+    d1.set_data({{"workflowId", "wf-1"}, {"status", "running"}});
+    store.put("executions", "e1", d1);   // accepted: parent exists
+
+    store.del("workflows", "wf-1");      // parent now gone; reference dangles
+
+    // Rewrite e1 touching only `status` - workflowId is unchanged, so this
+    // must NOT re-validate it and must NOT throw.
+    Document d2; d2.id = "e1"; d2.collection = "executions";
+    d2.set_data({{"workflowId", "wf-1"}, {"status", "completed"}});
+    bool threw = false;
+    try {
+        store.put("executions", "e1", d2);
+    } catch (const MissingParentReference&) {
+        threw = true;
+    }
+    check(!threw, "an update that leaves the reference field unchanged is not re-validated");
+    check(store.get("executions", "e1")->data()["status"] == "completed",
+          "the update itself still applied");
+}
+
+// v2.11.0 T13 — proves the race is actually closed, not merely narrowed.
+//
+// Models the exact interleaving the plan describes: something (an
+// application-level pre-check, or the old restrict path's own read) observes
+// the parent present, and only AFTER that does the parent get deleted -
+// before the child's write actually lands. A stale check-then-act sequence
+// would let the child insert through anyway, because its answer was decided
+// against the state as of the check, not as of the write.
+//
+// LMDB is single-writer (see try_open_for_read's file comment and
+// document_store_lmdb.cpp's env setup): every write transaction begins only
+// after the previous one has fully committed, so the child's write
+// transaction here necessarily starts strictly after the parent-delete
+// transaction commits. Because validate_on_write's mdb_get runs INSIDE the
+// child's own write transaction rather than in a separate, earlier read, it
+// sees the parent's true state as of the write, not as of whatever was
+// observed before. That is the whole mechanism this task adds: no separate
+// transaction, no interval, nothing that can go stale.
+//
+// What this test does NOT establish: it does not exercise real multi-thread
+// scheduling or prove there is no OTHER race at the LMDB layer. It does not
+// need to - LMDB's single-writer guarantee means transaction ORDER is the
+// only thing that can vary under concurrency, never interleaving within a
+// transaction, so serialising the two operations in program order is the
+// honest, deterministic equivalent of "the delete's transaction commits
+// before the child insert's transaction begins," which is the only
+// interleaving the race actually depends on.
+void test_validate_on_write_closes_the_stale_check_race() {
+    TmpEnv t("validate-race");
+    LmdbDocumentStore store(t.env);
+    store.set_relations("executions",
+                        {{"exec_wf", "workflowId", "workflows", true}});
+
+    Document w; w.id = "wf-1"; w.collection = "workflows";
+    w.set_data({{"name", "about to be deleted"}});
+    store.put("workflows", "wf-1", w);
+
+    // The "stale check": some caller observes the parent present. This is
+    // exactly what a restrict check (or an application's own pre-flight
+    // lookup) does - a READ, complete and finished, before the write it is
+    // meant to gate.
+    check(store.get("workflows", "wf-1").has_value(),
+          "pre-check observes the parent present");
+
+    // The parent vanishes AFTER that check returned, in its own committed
+    // transaction - the race window a stale check cannot see across.
+    check(store.del("workflows", "wf-1"), "parent deleted after the check ran");
+
+    // The child write's OWN transaction begins only now, strictly after the
+    // delete's commit (LMDB's single-writer serialisation). If validation
+    // used the stale check's answer (or any state cached before this point)
+    // it would wrongly accept. It must instead re-read the parent's current
+    // state inside its own transaction and reject.
+    Document d; d.id = "e1"; d.collection = "executions";
+    d.set_data({{"workflowId", "wf-1"}});
+    bool threw = false;
+    try {
+        store.put("executions", "e1", d);
+    } catch (const MissingParentReference&) {
+        threw = true;
+    }
+    check(threw, "the child write is rejected against write-time state, "
+                "despite an earlier check having observed the parent present");
+    check(!store.get("executions", "e1").has_value(),
+          "no trace of the write the stale check would have allowed");
+}
+
 }  // namespace
 
 int main() {
@@ -1241,6 +1528,13 @@ int main() {
     test_replayed_cascade_update_remirrors_to_lmdb_after_crash_window();
     test_a_failing_row_does_not_stop_recovery();
     test_reinsert_after_delete_converges_in_lmdb();
+    test_validate_on_write_rejects_missing_parent();
+    test_validate_on_write_accepts_existing_parent();
+    test_validate_on_write_false_allows_dangling_and_check_reports_it();
+    test_validate_on_write_never_rejects_absent_or_null();
+    test_validate_on_write_array_any_missing_rejects_whole_write();
+    test_validate_on_write_unrelated_update_not_rechecked();
+    test_validate_on_write_closes_the_stale_check_race();
 
     std::cout << "passed: " << g_pass << ", failed: " << g_fail << "\n";
     return g_fail == 0 ? 0 : 1;