|
|
@@ -23,6 +23,7 @@
|
|
|
#include "storage/document_store_lmdb.hpp"
|
|
|
|
|
|
#include <lmdb.h>
|
|
|
+#include <yyjson.h>
|
|
|
|
|
|
#include <algorithm>
|
|
|
#include <cctype>
|
|
|
@@ -104,6 +105,85 @@ smartbotic::database::Document decode_document(std::string_view bytes) {
|
|
|
// shared with MemoryStore so a shadow-read divergence can never be the
|
|
|
// result of two private copies drifting apart.
|
|
|
|
|
|
+
|
|
|
+// -------------------------------------------------------------------------
|
|
|
+// v2.8.0 — resolve one filter field straight out of a yyjson tree.
|
|
|
+//
|
|
|
+// A filtered scan used to build a full Document per row just to ask whether the
|
|
|
+// row matched. Over 10,124 real rows (193 MB) that was 3168ms against 369ms for
|
|
|
+// the yyjson parse alone - materialising was 88% of the work, and a filtered
|
|
|
+// query on that collection took ~2.5s to return ten documents. Views always add
|
|
|
+// their baked-in filters, so every view query paid it.
|
|
|
+//
|
|
|
+// This converts only the ONE value a filter asks about. The comparison logic
|
|
|
+// still lives in filter_eval, shared with MemoryStore's path.
|
|
|
+//
|
|
|
+// Field names map onto the stored JSON shape written by Document::toJson():
|
|
|
+// metadata at the top level, user fields under "data".
|
|
|
+// -------------------------------------------------------------------------
|
|
|
+std::optional<nlohmann::json> resolve_from_yyjson(yyjson_val* root,
|
|
|
+ const std::string& field) {
|
|
|
+ if (!root || !yyjson_is_obj(root)) return std::nullopt;
|
|
|
+
|
|
|
+ auto to_json = [](yyjson_val* v) -> std::optional<nlohmann::json> {
|
|
|
+ if (!v) return std::nullopt;
|
|
|
+ switch (yyjson_get_type(v)) {
|
|
|
+ case YYJSON_TYPE_NULL: return nlohmann::json(nullptr);
|
|
|
+ case YYJSON_TYPE_BOOL: return nlohmann::json(yyjson_get_bool(v));
|
|
|
+ case YYJSON_TYPE_NUM:
|
|
|
+ if (yyjson_is_real(v)) return nlohmann::json(yyjson_get_real(v));
|
|
|
+ if (yyjson_is_sint(v)) return nlohmann::json(yyjson_get_sint(v));
|
|
|
+ return nlohmann::json(yyjson_get_uint(v));
|
|
|
+ case YYJSON_TYPE_STR:
|
|
|
+ return nlohmann::json(std::string(yyjson_get_str(v),
|
|
|
+ yyjson_get_len(v)));
|
|
|
+ default: {
|
|
|
+ // Array or object: the filter needs the structure (CONTAINS, or a
|
|
|
+ // nested compare), so materialise just this subtree.
|
|
|
+ size_t len = 0;
|
|
|
+ char* raw = yyjson_val_write(v, 0, &len);
|
|
|
+ if (!raw) return std::nullopt;
|
|
|
+ std::optional<nlohmann::json> out;
|
|
|
+ try {
|
|
|
+ out = nlohmann::json::parse(std::string_view(raw, len));
|
|
|
+ } catch (const nlohmann::json::exception&) {
|
|
|
+ out = std::nullopt;
|
|
|
+ }
|
|
|
+ free(raw);
|
|
|
+ return out;
|
|
|
+ }
|
|
|
+ }
|
|
|
+ };
|
|
|
+
|
|
|
+ // Document metadata lives at the top level of the stored object, under the
|
|
|
+ // names Document::toJson() writes.
|
|
|
+ if (field == "_id") return to_json(yyjson_obj_get(root, "id"));
|
|
|
+ if (field == "_created_at") return to_json(yyjson_obj_get(root, "createdAt"));
|
|
|
+ if (field == "_updated_at") return to_json(yyjson_obj_get(root, "updatedAt"));
|
|
|
+ if (field == "_version") return to_json(yyjson_obj_get(root, "version"));
|
|
|
+
|
|
|
+ yyjson_val* data = yyjson_obj_get(root, "data");
|
|
|
+ if (!data) return std::nullopt;
|
|
|
+
|
|
|
+ if (field.find('.') == std::string::npos) {
|
|
|
+ return to_json(yyjson_obj_get(data, field.c_str()));
|
|
|
+ }
|
|
|
+ // Dotted path: descend objects only, matching getJsonPath's semantics
|
|
|
+ // (arrays are not indexed into).
|
|
|
+ yyjson_val* cur = data;
|
|
|
+ size_t pos = 0;
|
|
|
+ while (pos <= field.size()) {
|
|
|
+ const size_t dot = field.find('.', pos);
|
|
|
+ const std::string seg =
|
|
|
+ field.substr(pos, dot == std::string::npos ? std::string::npos : dot - pos);
|
|
|
+ if (!cur || !yyjson_is_obj(cur)) return std::nullopt;
|
|
|
+ cur = yyjson_obj_get(cur, seg.c_str());
|
|
|
+ if (dot == std::string::npos) break;
|
|
|
+ pos = dot + 1;
|
|
|
+ }
|
|
|
+ return to_json(cur);
|
|
|
+}
|
|
|
+
|
|
|
void sort_documents(std::vector<smartbotic::database::Document>& docs,
|
|
|
const smartbotic::database::Sort& sort) {
|
|
|
std::sort(docs.begin(), docs.end(),
|
|
|
@@ -442,7 +522,109 @@ ScanResult LmdbDocumentStore::scan(std::string_view collection,
|
|
|
}
|
|
|
// -------------------------------------------------------------------
|
|
|
|
|
|
+ // -------------------------------------------------------------------
|
|
|
+ // v2.8.0 — two passes: decide cheaply, materialise only what is returned.
|
|
|
+ //
|
|
|
+ // This used to build a full Document for EVERY row just to ask whether it
|
|
|
+ // matched. Over 10,124 real rows (193 MB) that was 3168ms against 369ms for
|
|
|
+ // the yyjson parse alone, so a filtered query took ~2.5s to return ten
|
|
|
+ // documents. Views always add their baked-in filters, so every view query
|
|
|
+ // paid it.
|
|
|
+ //
|
|
|
+ // Pass 1 parses each row with yyjson and evaluates the predicate through a
|
|
|
+ // field resolver, keeping only the ids that match (plus the sort key when
|
|
|
+ // sorting). Pass 2 decodes just the page. Sorting still needs a key for every
|
|
|
+ // match, but a key is one value rather than a whole document.
|
|
|
+ //
|
|
|
+ // A SEARCH filter scans every string in a document, so it cannot be answered
|
|
|
+ // from a per-field resolver; those queries keep the old row-at-a-time path.
|
|
|
+ // -------------------------------------------------------------------
|
|
|
+ const bool wholeDoc =
|
|
|
+ smartbotic::db::storage::filter_eval::needsWholeDocument(query.filters);
|
|
|
+ const bool sorting = query.sort && !query.sort->field.empty();
|
|
|
+
|
|
|
std::vector<smartbotic::database::Document> matches;
|
|
|
+
|
|
|
+ if (!wholeDoc) {
|
|
|
+ struct Hit {
|
|
|
+ std::string id;
|
|
|
+ nlohmann::json key; // sort key; null when not sorting
|
|
|
+ bool hasKey = false;
|
|
|
+ };
|
|
|
+ std::vector<Hit> hits;
|
|
|
+
|
|
|
+ int rc2 = mdb_cursor_get(cursor, &k, &v, MDB_FIRST);
|
|
|
+ while (rc2 == MDB_SUCCESS) {
|
|
|
+ if (!is_identity_key(to_sv(k))) {
|
|
|
+ const auto bytes = to_sv(v);
|
|
|
+ yyjson_doc* ydoc = yyjson_read(bytes.data(), bytes.size(), 0);
|
|
|
+ if (ydoc) {
|
|
|
+ yyjson_val* root = yyjson_doc_get_root(ydoc);
|
|
|
+ const bool ok =
|
|
|
+ smartbotic::db::storage::filter_eval::matchesFiltersResolved(
|
|
|
+ [root](const std::string& f) {
|
|
|
+ return resolve_from_yyjson(root, f);
|
|
|
+ },
|
|
|
+ query.filters);
|
|
|
+ if (ok) {
|
|
|
+ Hit h;
|
|
|
+ h.id = std::string(to_sv(k));
|
|
|
+ if (sorting) {
|
|
|
+ auto kv = resolve_from_yyjson(root, query.sort->field);
|
|
|
+ if (kv) { h.key = *kv; h.hasKey = true; }
|
|
|
+ }
|
|
|
+ hits.push_back(std::move(h));
|
|
|
+ }
|
|
|
+ yyjson_doc_free(ydoc);
|
|
|
+ }
|
|
|
+ // A row that will not parse cannot be matched; skipping it is the
|
|
|
+ // same outcome the old path reached by throwing on decode, minus
|
|
|
+ // failing the whole query for one bad row.
|
|
|
+ }
|
|
|
+ rc2 = mdb_cursor_get(cursor, &k, &v, MDB_NEXT);
|
|
|
+ }
|
|
|
+ if (rc2 != MDB_NOTFOUND && rc2 != MDB_SUCCESS) {
|
|
|
+ throw_mdb(rc2, "cursor_get");
|
|
|
+ }
|
|
|
+
|
|
|
+ if (sorting) {
|
|
|
+ const bool desc = query.sort->descending;
|
|
|
+ std::stable_sort(hits.begin(), hits.end(),
|
|
|
+ [desc](const Hit& a, const Hit& b) {
|
|
|
+ // Same ordering rules as sort_documents: missing values sort
|
|
|
+ // last ascending, and ties break on id for a stable page.
|
|
|
+ if (!a.hasKey && !b.hasKey) return desc ? a.id > b.id : a.id < b.id;
|
|
|
+ if (!a.hasKey) return desc;
|
|
|
+ if (!b.hasKey) return !desc;
|
|
|
+ if (a.key == b.key) return desc ? a.id > b.id : a.id < b.id;
|
|
|
+ const bool less = a.key < b.key;
|
|
|
+ return desc ? !less : less;
|
|
|
+ });
|
|
|
+ }
|
|
|
+
|
|
|
+ result.total_matched = hits.size();
|
|
|
+ const uint64_t start = std::min<uint64_t>(query.offset, hits.size());
|
|
|
+ const uint64_t end = std::min<uint64_t>(
|
|
|
+ static_cast<uint64_t>(query.offset) + query.limit, hits.size());
|
|
|
+ result.documents.reserve(end > start ? end - start : 0);
|
|
|
+ for (uint64_t i = start; i < end; ++i) {
|
|
|
+ MDB_val hk = to_val(hits[i].id);
|
|
|
+ MDB_val hv{0, nullptr};
|
|
|
+ if (mdb_get(rtxn.raw(), *dbi_opt, &hk, &hv) != MDB_SUCCESS) continue;
|
|
|
+ result.documents.push_back(decode_document(to_sv(hv)));
|
|
|
+ }
|
|
|
+ result.has_more = end < result.total_matched;
|
|
|
+
|
|
|
+ if (!query.projection.empty()) {
|
|
|
+ for (auto& d : result.documents) {
|
|
|
+ nlohmann::json data = d.data();
|
|
|
+ apply_projection_inplace(data, query.projection);
|
|
|
+ d.set_data(data);
|
|
|
+ }
|
|
|
+ }
|
|
|
+ return result;
|
|
|
+ }
|
|
|
+
|
|
|
int rc = mdb_cursor_get(cursor, &k, &v, MDB_FIRST);
|
|
|
while (rc == MDB_SUCCESS) {
|
|
|
// Skip the identity sentinel — it is not JSON and would throw in
|