Sfoglia il codice sorgente

fix(count): serve Count from LMDB like every other read

Count called store_.count() unconditionally while Exists/Get/Find were
LMDB-first, so count() under-reported whenever MemoryStore and LMDB
diverged - observed 42 vs a Find returning 64 on the same collection right
after the v2.4.4 placement repair. Divergence is normal, not exceptional:
MemoryStore is a bounded cache subject to eviction, LMDB is the full
dataset.

Unfiltered counts read the sub-db entry count. Filtered counts go through
scan() with limit=0, which populates total_matched (computed before
pagination) without materialising or encoding any document. Same gate and
same error mapping as Find.

Tests: test_scan_limit_zero_reports_total pins the limit=0 contract the
filtered path depends on. 23 assertions, ctest 15/15.
fszontagh 1 mese fa
parent
commit
ddf0a7f3d2
2 ha cambiato i file con 76 aggiunte e 1 eliminazioni
  1. 41 1
      service/src/database_grpc_impl.cpp
  2. 35 0
      tests/test_subdb_identity.cpp

+ 41 - 1
service/src/database_grpc_impl.cpp

@@ -839,7 +839,47 @@ grpc::Status DatabaseGrpcImpl::Count(
         filters.push_back(std::move(f));
     }
 
-    uint64_t count = store_.count(targetCollection, filters);
+    // v2.4.4 — LMDB-first Count, same gate as Exists/Get/Find.
+    //
+    // Until now Count went to MemoryStore unconditionally while every other
+    // read was LMDB-first, so `count()` under-reported whenever the two
+    // diverged — a caller could Find 64 documents in a collection that
+    // Count called 42. Divergence is normal, not exceptional: MemoryStore is
+    // a bounded cache subject to eviction, LMDB is the full dataset.
+    //
+    // Unfiltered counts use the sub-db entry count directly. Filtered counts
+    // have to evaluate the predicate per document, so they go through scan()
+    // with limit=0 — that populates total_matched (computed before
+    // pagination) without materialising or encoding any document.
+    uint64_t count = 0;
+    bool used_lmdb = false;
+    if (!targetCollection.empty() && targetCollection[0] != '_'
+        && service_.mirrorHealthy() && service_.mirrorDriftCount() == 0) {
+        try {
+            const auto rc = smartbotic::database::resolveCollection(targetCollection);
+            if (auto* ds = service_.docStore(rc.project)) {
+                if (filters.empty()) {
+                    count = ds->count(rc.collection);
+                } else {
+                    Query q;
+                    q.filters = filters;
+                    q.limit = 0;
+                    count = ds->scan(rc.collection, q).total_matched;
+                }
+                used_lmdb = true;
+            }
+        } catch (const std::invalid_argument& e) {
+            return grpc::Status(grpc::StatusCode::INVALID_ARGUMENT, e.what());
+        } catch (const std::exception& e) {
+            spdlog::error("v2.4.4 Count LMDB failed coll={}: {}",
+                          targetCollection, e.what());
+            return grpc::Status(grpc::StatusCode::INTERNAL, e.what());
+        }
+    }
+    if (!used_lmdb) {
+        count = store_.count(targetCollection, filters);
+    }
+
     response->set_count(count);
     return grpc::Status::OK;
 }

+ 35 - 0
tests/test_subdb_identity.cpp

@@ -239,6 +239,40 @@ void test_vector_subdb_sentinel() {
           "vector sub-db is stamped with its prefixed name");
 }
 
+// v2.4.4 Count is LMDB-first. Unfiltered it uses count(); filtered it uses
+// scan() with limit=0 and reads total_matched. Pin that contract: total_matched
+// is computed BEFORE pagination, so limit=0 must still report the true total
+// while returning no documents.
+void test_scan_limit_zero_reports_total() {
+    TmpEnv t("counting");
+    LmdbDocumentStore store(t.env);
+    for (int i = 0; i < 5; ++i) {
+        Document d;
+        d.id = "id" + std::to_string(i);
+        d.collection = "things";
+        d.set_data(nlohmann::json{{"kind", i < 3 ? "alpha" : "beta"}});
+        store.put("things", d.id, d);
+    }
+
+    smartbotic::database::Query q;
+    q.limit = 0;
+    auto all = store.scan("things", q);
+    check(all.total_matched == 5, "limit=0 reports the full total");
+    check(all.documents.empty(), "limit=0 returns no documents");
+    check(store.count("things") == 5, "unfiltered count matches");
+
+    smartbotic::database::Query fq;
+    fq.limit = 0;
+    smartbotic::database::Filter f;
+    f.field = "kind";
+    f.op = smartbotic::database::FilterOp::EQ;
+    f.value = "alpha";
+    fq.filters.push_back(f);
+    auto filtered = store.scan("things", fq);
+    check(filtered.total_matched == 3,
+          "filtered limit=0 reports the matching total, not the collection size");
+}
+
 }  // namespace
 
 int main() {
@@ -249,6 +283,7 @@ int main() {
     test_identity_key_predicate();
     test_sentinel_invisible_through_store();
     test_vector_subdb_sentinel();
+    test_scan_limit_zero_reports_total();
 
     std::cout << "passed: " << g_pass << ", failed: " << g_fail << "\n";
     return g_fail == 0 ? 0 : 1;