Files
MultiforaDB/src/query.zig
Aleksey Shakhmatov 7482042f34 index/db/commands: fold duplicated index logic into single definitions
Cleanup pass over the secondary-index feature.

add_doc is now the one entry-commit path. create_index and
build_all_indexes each hand-rolled build -> check_unique -> reserve ->
insert, and had already drifted on whether multikey is set before or
after the unique check; add_doc gained an enforce_unique flag so the
rebuild path keeps its tolerate-and-warn behavior. reserve_for and
insert_entries are now the only way the engine touches Index.entries.

One definition each for: prefix comparison and the prefix binary
searches (prefix_order + std.sort), the cartesian-product odometer
(advance_choice), the spec pair list (write_spec builds on spec_pairs,
so the log format and the listIndexes reply share one schema), the _id
clause parser (plan_id reuses analyze_clause), key-pattern direction
(index.descending, which desc_dir already disagreed with on non-numeric
values), option truthiness (query.truthy), the E11000 message, and
index-removal-by-name (Collection.find_index/remove_index). Key-pattern
matching moved out of the dispatcher into index.find_by_key_pattern.

Dead or redundant: ParallelArraysError, the unread `dropped` counter,
insert_entries' discarded gpa, a third pass computing multikey, the
has_id/is_id_index flag pair, Plan.key_len (always lookup_keys[0].len,
now a method), first_match_consumed (now stages = stages[1..]).

Cheaper hot paths: remove_id compacts in one pass instead of an
orderedRemove per hit; Plan.search skips the sort/dedupe when neither
multikey nor multiple lookup keys can produce a repeat; the _id fast
path reuses one scratch key buffer (bson.write_serialized_value); the
plan loop uses the bound collection instead of re-resolving it through
two hash lookups per candidate.

Behavior is unchanged except that dropping plan_id's fixed 16-clause
buffer enables the _id fast path on filters that previously exceeded it.
2026-08-02 13:40:42 +03:00

972 lines
40 KiB
Zig

//! Query engine: filter matching (MongoDB query operators), sorting by
//! canonical BSON order, and projections. Includes a small backtracking
//! regex engine for $regex.
const std = @import("std");
const bson = @import("bson.zig");
// ---------------------------------------------------------------------------
// Filter matching
// ---------------------------------------------------------------------------
pub const QueryError = error{OutOfMemory};
pub fn matches(gpa: std.mem.Allocator, filter: *const bson.Document, doc: *const bson.Document) QueryError!bool {
for (filter.pairs) |p| {
if (p.key.len > 0 and p.key[0] == '$') {
if (!try match_top_level(gpa, p.key, p.value, doc)) return false;
} else {
if (!try field_matches(gpa, p.key, p.value, doc)) return false;
}
}
return true;
}
fn match_top_level(gpa: std.mem.Allocator, op: []const u8, value: bson.Value, doc: *const bson.Document) QueryError!bool {
if (std.mem.eql(u8, op, "$and") or std.mem.eql(u8, op, "$or")) {
const want_and = std.mem.eql(u8, op, "$and");
const filters = switch (value) {
.array => |arr| arr,
else => return false,
};
for (filters) |item| {
const f = switch (item) {
.doc => |pairs| pairs,
else => return false,
};
const matched = try matches(gpa, &.{ .arena = undefined, .pairs = f }, doc);
if (want_and and !matched) return false;
if (!want_and and matched) return true;
}
return want_and;
}
if (std.mem.eql(u8, op, "$nor")) {
const filters = switch (value) {
.array => |arr| arr,
else => return false,
};
for (filters) |item| {
const f = switch (item) {
.doc => |pairs| pairs,
else => return false,
};
if (try matches(gpa, &.{ .arena = undefined, .pairs = f }, doc)) return false;
}
return true;
}
return false;
}
/// True when every key is a `$`-prefixed operator. The one definition of
/// what an operator document looks like — filters, $elemMatch, $pull and
/// upsert-document construction all ask this question.
pub fn all_operator_keys(pairs: []const bson.Pair) bool {
if (pairs.len == 0) return false;
for (pairs) |p| {
if (p.key.len == 0 or p.key[0] != '$') return false;
}
return true;
}
/// The operator pairs of a `{$op: ...}` value, or null if it is not one. An
/// empty document counts as an (vacuously satisfied) operator document.
fn is_operator_doc(value: bson.Value) ?[]const bson.Pair {
const pairs = switch (value) {
.doc => |p| p,
else => return null,
};
if (pairs.len > 0 and !all_operator_keys(pairs)) return null;
return pairs;
}
fn field_matches(gpa: std.mem.Allocator, path: []const u8, expected: bson.Value, doc: *const bson.Document) QueryError!bool {
var candidates: std.ArrayListUnmanaged(bson.Value) = .empty;
defer candidates.deinit(gpa);
try collect_values(gpa, doc.pairs, path, &candidates, 0);
// MongoDB applies queries to array elements as well as the array itself.
// Index the snapshot length, re-reading items each iteration: appending
// may reallocate the buffer, which would invalidate a captured slice.
const direct_count = candidates.items.len;
var i: usize = 0;
while (i < direct_count) : (i += 1) {
const a = candidates.items[i];
if (a == .array) {
for (a.array) |elem| try candidates.append(gpa, elem);
}
}
if (is_operator_doc(expected)) |pairs| {
var options: []const u8 = "";
for (pairs) |p| {
if (std.mem.eql(u8, p.key, "$options")) {
if (p.value == .string) options = p.value.string;
}
}
for (pairs) |p| {
if (std.mem.eql(u8, p.key, "$options")) continue;
if (!try match_operator(gpa, p.key, p.value, candidates.items, options)) return false;
}
return true;
}
// Bare BSON regex value: {field: /re/} behaves like {$regex: "re"}.
if (expected == .regex) {
for (candidates.items) |a| {
if (a == .string and regex_match(expected.regex.pattern, expected.regex.options, a.string)) return true;
}
return false;
}
// Bare equality — matches if any candidate equals the expected value.
for (candidates.items) |actual| {
if (bson.compare(actual, expected) == .eq) return true;
}
return false;
}
fn match_operator(gpa: std.mem.Allocator, op: []const u8, value: bson.Value, actuals: []const bson.Value, regex_options: []const u8) QueryError!bool {
if (std.mem.eql(u8, op, "$eq")) {
for (actuals) |a| if (bson.compare(a, value) == .eq) return true;
return false;
}
if (std.mem.eql(u8, op, "$ne")) {
for (actuals) |a| if (bson.compare(a, value) == .eq) return false;
return true;
}
if (std.mem.eql(u8, op, "$gt") or std.mem.eql(u8, op, "$gte") or
std.mem.eql(u8, op, "$lt") or std.mem.eql(u8, op, "$lte"))
{
for (actuals) |a| {
const o = bson.compare(a, value);
if (std.mem.eql(u8, op, "$gt") and o == .gt) return true;
if (std.mem.eql(u8, op, "$gte") and o != .lt) return true;
if (std.mem.eql(u8, op, "$lt") and o == .lt) return true;
if (std.mem.eql(u8, op, "$lte") and o != .gt) return true;
}
return false;
}
if (std.mem.eql(u8, op, "$in") or std.mem.eql(u8, op, "$nin")) {
const members = switch (value) {
.array => |arr| arr,
else => return false,
};
const want_in = std.mem.eql(u8, op, "$in");
for (actuals) |a| {
for (members) |m| {
if (bson.compare(a, m) == .eq) return want_in;
}
}
return !want_in;
}
if (std.mem.eql(u8, op, "$exists")) {
const want = switch (value) {
.bool => |b| b,
else => return false,
};
return (actuals.len > 0) == want;
}
if (std.mem.eql(u8, op, "$regex")) {
const pattern = switch (value) {
.string => |s| s,
.doc => |pairs| blk: {
const pat = bson.get_pair(pairs, "$regex") orelse return false;
break :blk switch (pat) {
.string => |s| s,
else => return false,
};
},
else => return false,
};
for (actuals) |a| {
if (a == .string and regex_match(pattern, regex_options, a.string)) return true;
}
return false;
}
if (std.mem.eql(u8, op, "$not")) {
const pairs = is_operator_doc(value) orelse {
// $not with a bare value means $ne-ish semantics; treat as
// "not equal to this regex or value".
if (value == .string) {
for (actuals) |a| {
if (a == .string and regex_match(value.string, "", a.string)) return false;
}
return true;
}
return false;
};
for (pairs) |p| {
if (try match_operator(gpa, p.key, p.value, actuals, regex_options)) return false;
}
return true;
}
if (std.mem.eql(u8, op, "$size")) {
const want = switch (value) {
.int32 => |i| i,
.int64 => |i| @as(i32, @intCast(i)),
else => return false,
};
for (actuals) |a| {
if (a == .array and a.array.len == @as(usize, @intCast(want))) return true;
}
return false;
}
if (std.mem.eql(u8, op, "$all")) {
const members = switch (value) {
.array => |arr| arr,
else => return false,
};
outer: for (members) |m| {
for (actuals) |a| {
if (a == .array) {
for (a.array) |elem| {
if (bson.compare(elem, m) == .eq) continue :outer;
}
}
}
return false;
}
return true;
}
if (std.mem.eql(u8, op, "$elemMatch")) {
const operand = switch (value) {
.doc => |pairs| pairs,
else => return false,
};
const all_operators = all_operator_keys(operand);
for (actuals) |a| {
if (a != .array) continue;
for (a.array) |elem| {
if (all_operators) {
var single: [1]bson.Value = .{elem};
var ok = true;
for (operand) |p| {
if (!try match_operator(gpa, p.key, p.value, single[0..], "")) {
ok = false;
break;
}
}
if (ok) return true;
} else {
switch (elem) {
.doc => |pairs| {
if (try matches(gpa, &.{ .arena = undefined, .pairs = operand }, &.{ .arena = undefined, .pairs = pairs })) return true;
},
else => {},
}
}
}
}
return false;
}
return false;
}
/// Collect values reachable at `path` (dot-separated), descending into
/// documents and, per MongoDB multikey semantics, into arrays of documents.
/// Appends into `out`; on OOM, collection stops early (the engine is
/// already failing at that point). Public because index entry generation
/// must mirror field_matches exactly (src/index.zig).
pub fn collect_values(gpa: std.mem.Allocator, pairs: []const bson.Pair, path: []const u8, out: *std.ArrayListUnmanaged(bson.Value), depth: usize) QueryError!void {
var it = std.mem.splitScalar(u8, path, '.');
const first = it.next() orelse return;
for (pairs) |p| {
if (!std.mem.eql(u8, p.key, first)) continue;
const rest = it.rest();
if (rest.len == 0) {
if (depth < 8) {
try out.append(gpa, p.value);
}
} else {
try collect_from_value(gpa, p.value, rest, out, depth + 1);
}
}
}
fn collect_from_value(gpa: std.mem.Allocator, v: bson.Value, path: []const u8, out: *std.ArrayListUnmanaged(bson.Value), depth: usize) QueryError!void {
if (depth > 8) return;
switch (v) {
.doc => |pairs| try collect_values(gpa, pairs, path, out, depth),
.array => |items| {
// Numeric first segment: address an element by index ("tags.0").
var pit = std.mem.splitScalar(u8, path, '.');
const seg = pit.next() orelse return;
if (std.fmt.parseInt(usize, seg, 10)) |idx| {
if (idx < items.len) {
const rest = pit.rest();
if (rest.len == 0) {
if (depth < 8) try out.append(gpa, items[idx]);
} else {
try collect_from_value(gpa, items[idx], rest, out, depth + 1);
}
}
return;
} else |_| {}
// Multikey semantics: descend into embedded documents of the array.
for (items) |item| {
switch (item) {
.doc => try collect_values(gpa, item.doc, path, out, depth),
else => {},
}
}
},
else => return,
}
}
// ---------------------------------------------------------------------------
// Regex (subset): ^ $ . * + ? [...] ( ) | and escaped literals
// ---------------------------------------------------------------------------
pub fn regex_match(pattern: []const u8, options: []const u8, text: []const u8) bool {
// Bound recursion depth: deeply nested groups would overflow the stack.
var depth: usize = 0;
var max_depth: usize = 0;
var ri: usize = 0;
while (ri < pattern.len) : (ri += 1) {
if (pattern[ri] == '\\') {
ri += 1;
continue;
}
if (pattern[ri] == '(') {
depth += 1;
max_depth = @max(max_depth, depth);
} else if (pattern[ri] == ')') {
depth -|= 1;
}
}
if (max_depth > 256) return false;
const case_insensitive = std.mem.indexOfScalar(u8, options, 'i') != null;
const dot_all = std.mem.indexOfScalar(u8, options, 's') != null;
const anchored = pattern.len > 0 and pattern[0] == '^';
const start_pattern = if (anchored) pattern[1..] else pattern;
var p: usize = 0;
if (anchored) {
if (match_here(start_pattern, &p, text, 0, case_insensitive, dot_all) == null) return false;
return p == start_pattern.len;
}
var t: usize = 0;
while (t <= text.len) : (t += 1) {
p = 0;
if (match_here(start_pattern, &p, text, t, case_insensitive, dot_all) != null and p == start_pattern.len) return true;
}
return false;
}
/// Match `pattern[p..]` against `text[t..]`, returning the new text
/// position on success (null on failure). Backtracks via recursion.
fn match_here(pattern: []const u8, p: *usize, text: []const u8, t: usize, ci: bool, dot_all: bool) ?usize {
var pos = t;
while (p.* < pattern.len) {
const c = pattern[p.*];
switch (c) {
'$' => {
if (p.* + 1 == pattern.len) {
p.* += 1;
return if (pos == text.len) pos else null;
}
if (pos >= text.len) return null;
if (!chars_eq(c, text[pos], ci)) return null;
p.* += 1;
pos += 1;
},
'^' => {
if (pos != 0) return null;
p.* += 1;
},
'(' => {
const end = find_group_end(pattern, p.*) orelse return null;
const inner = pattern[p.* + 1 .. end - 1];
var parts: [8][]const u8 = undefined;
var nparts: usize = 0;
var seg_start: usize = 0;
var depth: usize = 0;
var i: usize = 0;
while (i < inner.len) : (i += 1) {
const ic = inner[i];
if (ic == '\\') {
i += 1;
continue;
}
if (ic == '(') depth +|= 1;
if (ic == ')') depth -|= 1;
if (ic == '|' and depth == 0) {
if (nparts < parts.len) parts[nparts] = inner[seg_start..i];
nparts += 1;
seg_start = i + 1;
}
}
if (nparts < parts.len) parts[nparts] = inner[seg_start..];
nparts += 1;
if (nparts == 1) {
var gp: usize = 0;
const after = match_here(parts[0], &gp, text, pos, ci, dot_all) orelse return null;
if (gp != parts[0].len) return null;
p.* = end;
pos = after;
} else {
var matched = false;
for (parts[0..nparts]) |part| {
var gp: usize = 0;
const after = match_here(part, &gp, text, pos, ci, dot_all) orelse continue;
if (gp == part.len) {
p.* = end;
pos = after;
matched = true;
break;
}
}
if (!matched) return null;
}
},
')' => return null, // unbalanced
'\\' => {
if (p.* + 1 >= pattern.len) return null;
const lit = pattern[p.* + 1];
if (pos >= text.len or !chars_eq(lit, text[pos], ci)) return null;
p.* += 2;
pos += 1;
},
else => {
var element_end: usize = undefined;
if (c == '[') {
var close = p.* + 1;
if (close < pattern.len and pattern[close] == '^') close += 1;
while (close < pattern.len and pattern[close] != ']') close += 1;
if (close >= pattern.len) return null;
element_end = close + 1;
} else {
element_end = p.* + 1;
}
const element = pattern[p.*..element_end];
var q_end = element_end;
var min: usize = 1;
var max: usize = 1;
if (element_end < pattern.len and (pattern[element_end] == '*' or pattern[element_end] == '+' or pattern[element_end] == '?')) {
switch (pattern[element_end]) {
'*' => {
min = 0;
max = std.math.maxInt(usize);
},
'+' => {
min = 1;
max = std.math.maxInt(usize);
},
'?' => {
min = 0;
},
else => {},
}
q_end = element_end + 1;
}
// Greedy: consume as many as possible, then backtrack.
var consumed: usize = 0;
var pos_cur = pos;
while (max == std.math.maxInt(usize) or consumed < max) {
if (element_matches(element, text, pos_cur, ci, dot_all)) {
pos_cur += 1;
consumed += 1;
} else break;
}
var attempt = consumed;
while (attempt >= min) : (attempt -= 1) {
p.* = q_end;
if (match_here(pattern, p, text, pos_cur - (consumed - attempt), ci, dot_all)) |after| {
return after;
}
if (attempt == 0) break;
}
return null;
},
}
}
return pos;
}
fn chars_eq(a: u8, b: u8, ci: bool) bool {
if (ci) return std.ascii.toLower(a) == std.ascii.toLower(b);
return a == b;
}
fn in_range(lo: u8, hi: u8, c: u8, ci: bool) bool {
if (ci) {
const l = std.ascii.toLower(c);
return l >= std.ascii.toLower(lo) and l <= std.ascii.toLower(hi);
}
return c >= lo and c <= hi;
}
fn find_group_end(pattern: []const u8, open: usize) ?usize {
var depth: usize = 1;
var i = open + 1;
while (i < pattern.len) : (i += 1) {
if (pattern[i] == '\\') {
i += 1;
continue;
}
if (pattern[i] == '(') depth += 1;
if (pattern[i] == ')') {
depth -= 1;
if (depth == 0) return i + 1;
}
}
return null;
}
fn element_matches(element: []const u8, text: []const u8, t: usize, ci: bool, dot_all: bool) bool {
if (t >= text.len) return false;
if (element[0] == '.') {
return dot_all or text[t] != '\n';
}
if (element[0] == '[') {
var negated = false;
var i: usize = 1;
if (i < element.len and element[i] == '^') {
negated = true;
i += 1;
}
var matched = false;
while (i < element.len and element[i] != ']') {
if (i + 2 < element.len and element[i + 1] == '-') {
if (in_range(element[i], element[i + 2], text[t], ci)) matched = true;
i += 3;
} else {
if (chars_eq(element[i], text[t], ci)) matched = true;
i += 1;
}
}
return matched != negated;
}
return chars_eq(element[0], text[t], ci);
}
// ---------------------------------------------------------------------------
// Sort
// ---------------------------------------------------------------------------
pub const SortKey = struct {
path: []const u8,
descending: bool,
};
/// Sort `docs` in place by `keys`. Candidate values are collected up front
/// (allocations happen before the sort), so the comparator itself is pure
/// and cannot fail — OOM during collection propagates as QueryError.
pub fn sort_docs(arena: std.mem.Allocator, docs: []*const bson.Document, keys: []const SortKey) QueryError!void {
if (keys.len == 0 or docs.len < 2) return;
const SortedDoc = struct {
doc: *const bson.Document,
values: [][]const bson.Value,
};
const entries = try arena.alloc(SortedDoc, docs.len);
for (docs, 0..) |d, i| {
const values = try arena.alloc([]const bson.Value, keys.len);
for (keys, 0..) |k, ki| {
var list: std.ArrayListUnmanaged(bson.Value) = .empty;
try collect_values(arena, d.pairs, k.path, &list, 0);
values[ki] = list.items;
}
entries[i] = .{ .doc = d, .values = values };
}
const Ctx = struct {
keys: []const SortKey,
fn lessThan(ctx: @This(), a: SortedDoc, b: SortedDoc) bool {
for (ctx.keys, 0..) |k, ki| {
const aval: bson.Value = if (a.values[ki].len > 0) a.values[ki][0] else .null;
const bval: bson.Value = if (b.values[ki].len > 0) b.values[ki][0] else .null;
const o = bson.compare(aval, bval);
if (o != .eq) return if (k.descending) o == .gt else o == .lt;
}
return false;
}
};
std.mem.sort(SortedDoc, entries, Ctx{ .keys = keys }, Ctx.lessThan);
for (entries, 0..) |e, i| docs[i] = e.doc;
}
// ---------------------------------------------------------------------------
// Projection
// ---------------------------------------------------------------------------
pub const ProjectionError = std.mem.Allocator.Error;
/// Apply a projection document, writing resulting pairs into `out` (which
/// should use the caller's arena so strings are owned).
pub fn project(arena: std.mem.Allocator, doc: *const bson.Document, proj: *const bson.Document, out: *std.ArrayListUnmanaged(bson.Pair)) ProjectionError!void {
var inclusion: ?bool = null;
var non_id_count: usize = 0;
for (proj.pairs) |p| {
if (std.mem.eql(u8, p.key, "_id")) continue;
non_id_count += 1;
const flag = truthy(p.value);
inclusion = if (inclusion == null) flag else inclusion;
}
// {_id: 0} alone means "drop _id, keep everything else".
const include = if (non_id_count > 0) (inclusion orelse true) else false;
if (include) {
// Inclusion list: _id unless excluded, plus listed paths.
var include_id = true;
if (bson.get_pair(proj.pairs, "_id")) |idv| {
include_id = truthy(idv);
}
if (include_id) {
if (bson.get_pair(doc.pairs, "_id")) |idv| {
try out.append(arena, .{ .key = "_id", .value = try bson.copy_value(arena, idv) });
}
}
for (proj.pairs) |p| {
if (std.mem.eql(u8, p.key, "_id")) continue;
if (!truthy(p.value)) continue;
try project_path(arena, doc.pairs, p.key, out);
}
} else {
// Exclusion: copy everything except excluded paths (and _id if set).
for (doc.pairs) |p| {
if (std.mem.eql(u8, p.key, "_id")) {
var excluded = false;
if (bson.get_pair(proj.pairs, "_id")) |idv| excluded = !truthy(idv);
if (excluded) continue;
}
if (is_excluded(proj, p.key)) continue;
if (p.value == .doc and has_deeper_exclusion(proj, p.key)) {
const filtered = try exclude_doc(arena, p.value.doc, proj, p.key);
try out.append(arena, .{ .key = try arena.dupe(u8, p.key), .value = .{ .doc = filtered } });
} else {
try out.append(arena, .{ .key = try arena.dupe(u8, p.key), .value = try bson.copy_value(arena, p.value) });
}
}
}
}
/// Recursively apply exclusions to a nested document given the parent path.
fn exclude_doc(arena: std.mem.Allocator, pairs: []const bson.Pair, proj: *const bson.Document, parent: []const u8) ProjectionError![]const bson.Pair {
var out: std.ArrayListUnmanaged(bson.Pair) = .empty;
errdefer out.deinit(arena);
for (pairs) |p| {
const full = if (parent.len > 0) try std.fmt.allocPrint(arena, "{s}.{s}", .{ parent, p.key }) else p.key;
if (is_excluded(proj, full)) continue;
if (p.value == .doc and has_deeper_exclusion(proj, full)) {
const filtered = try exclude_doc(arena, p.value.doc, proj, full);
try out.append(arena, .{ .key = try arena.dupe(u8, p.key), .value = .{ .doc = filtered } });
} else {
try out.append(arena, .{ .key = try arena.dupe(u8, p.key), .value = try bson.copy_value(arena, p.value) });
}
}
return out.toOwnedSlice(arena);
}
fn is_excluded(proj: *const bson.Document, key: []const u8) bool {
return bson.get_pair(proj.pairs, key) != null;
}
fn has_deeper_exclusion(proj: *const bson.Document, key: []const u8) bool {
for (proj.pairs) |pp| {
if (is_prefix_or_equal(key, pp.key)) return true;
}
return false;
}
/// MongoDB's truthiness for an option/flag value (projection flags, index
/// spec options).
pub fn truthy(v: bson.Value) bool {
return switch (v) {
.bool => |b| b,
.int32 => |i| i != 0,
.int64 => |i| i != 0,
.double => |d| d != 0,
else => false,
};
}
/// Include a dotted path (e.g. "a.b.c"), creating nested documents as needed.
fn project_path(arena: std.mem.Allocator, pairs: []const bson.Pair, path: []const u8, out: *std.ArrayListUnmanaged(bson.Pair)) ProjectionError!void {
var it = std.mem.splitScalar(u8, path, '.');
const first = it.next() orelse return;
const rest = it.rest();
// Does the doc have this top-level field?
const field = bson.get_pair(pairs, first);
if (rest.len == 0) {
if (field) |f| {
try out.append(arena, .{ .key = try arena.dupe(u8, first), .value = try bson.copy_value(arena, f) });
}
return;
}
if (field) |f| {
switch (f) {
.doc => |sub| {
var nested: std.ArrayListUnmanaged(bson.Pair) = .empty;
errdefer nested.deinit(arena);
try project_path(arena, sub, rest, &nested);
if (nested.items.len > 0) {
try out.append(arena, .{ .key = try arena.dupe(u8, first), .value = .{ .doc = try nested.toOwnedSlice(arena) } });
}
},
else => {},
}
}
}
fn is_prefix_or_equal(prefix: []const u8, key: []const u8) bool {
if (std.mem.eql(u8, prefix, key)) return true;
if (prefix.len < key.len and std.mem.eql(u8, prefix, key[0..prefix.len])) {
return key[prefix.len] == '.';
}
return false;
}
// ---------------------------------------------------------------------------
// Tests
// ---------------------------------------------------------------------------
const testing = std.testing;
fn doc_of(pairs: []const bson.Pair) bson.Document {
return .{ .arena = undefined, .pairs = pairs };
}
test "basic filters" {
const d = doc_of(&.{
.{ .key = "_id", .value = .{ .int32 = 1 } },
.{ .key = "age", .value = .{ .int32 = 30 } },
.{ .key = "name", .value = .{ .string = "alice" } },
.{ .key = "tags", .value = .{ .array = &.{ .{ .string = "a" }, .{ .string = "b" } } } },
});
try testing.expect(try matches(testing.allocator, &doc_of(&.{}), &d));
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "age", .value = .{ .int32 = 30 } }}), &d));
try testing.expect(!try matches(testing.allocator, &doc_of(&.{.{ .key = "age", .value = .{ .int32 = 31 } }}), &d));
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "age", .value = .{ .doc = &.{.{ .key = "$gt", .value = .{ .int32 = 20 } }} } }}), &d));
try testing.expect(!try matches(testing.allocator, &doc_of(&.{.{ .key = "age", .value = .{ .doc = &.{.{ .key = "$gt", .value = .{ .int32 = 30 } }} } }}), &d));
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "age", .value = .{ .doc = &.{.{ .key = "$in", .value = .{ .array = &.{ .{ .int32 = 29 }, .{ .int32 = 30 } } } }} } }}), &d));
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "missing", .value = .{ .doc = &.{.{ .key = "$exists", .value = .{ .bool = false } }} } }}), &d));
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "tags", .value = .{ .string = "b" } }}), &d));
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "tags", .value = .{ .doc = &.{.{ .key = "$size", .value = .{ .int32 = 2 } }} } }}), &d));
}
test "regex filter" {
const d = doc_of(&.{.{ .key = "name", .value = .{ .string = "Alice Smith" } }});
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "name", .value = .{ .doc = &.{
.{ .key = "$regex", .value = .{ .string = "^al" } },
.{ .key = "$options", .value = .{ .string = "i" } },
} } }}), &d));
try testing.expect(!try matches(testing.allocator, &doc_of(&.{.{ .key = "name", .value = .{ .doc = &.{.{ .key = "$regex", .value = .{ .string = "^X" } }} } }}), &d));
try testing.expect(!try matches(testing.allocator, &doc_of(&.{.{ .key = "name", .value = .{ .doc = &.{.{ .key = "$regex", .value = .{ .string = "sm[i]th$" } }} } }}), &d));
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "name", .value = .{ .doc = &.{.{ .key = "$regex", .value = .{ .string = "th$" } }} } }}), &d));
}
test "regex engine basics" {
try testing.expect(regex_match("abc", "", "xabcx"));
try testing.expect(regex_match("^abc", "", "abc"));
try testing.expect(!regex_match("^abc", "", "xabc"));
try testing.expect(regex_match("a.c", "", "abc"));
try testing.expect(regex_match("a*b", "", "aaab"));
try testing.expect(regex_match("a+b", "", "aab"));
try testing.expect(regex_match("colou?r", "", "color"));
try testing.expect(regex_match("colou?r", "", "colour"));
try testing.expect(regex_match("[0-9]+", "", "abc123def"));
try testing.expect(!regex_match("^[0-9]+$", "", "abc123"));
try testing.expect(regex_match("(ab|cd)e", "", "cde"));
try testing.expect(regex_match("a\\.b", "", "a.b"));
try testing.expect(regex_match("^foo$", "", "foo"));
try testing.expect(!regex_match("^foo$", "", "foobar"));
try testing.expect(regex_match("hello", "i", "HELLO"));
try testing.expect(regex_match("^a+$", "", "aaaa"));
try testing.expect(!regex_match("^a+$", "", "aaab"));
}
test "regex hostile input does not crash" {
// Escaped paren inside a group used to underflow the alternation scan.
// Pattern is: "(" ++ "\" ++ "))" — a group containing an escaped ')'.
const hostile = "(" ++ "\\" ++ "))";
try testing.expect(regex_match(hostile, "", ")"));
try testing.expect(!regex_match(hostile, "", "x"));
try testing.expect(regex_match("a" ++ "\\" ++ "|b", "", "a|b"));
// Deeply nested groups must be rejected, not blow the stack.
const deep = "(" ** 300 ++ "x" ++ ")" ** 300;
try testing.expect(!regex_match(deep, "", "x"));
}
test "long array values are not truncated" {
var items: [30]bson.Value = undefined;
for (0..30) |i| items[i] = .{ .int32 = @intCast(i) };
const d = doc_of(&.{.{ .key = "tags", .value = .{ .array = &items } }});
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "tags", .value = .{ .int32 = 29 } }}), &d));
try testing.expect(!try matches(testing.allocator, &doc_of(&.{.{ .key = "tags", .value = .{ .int32 = 30 } }}), &d));
}
test "multikey path with several array candidates does not use-after-free" {
// Two array candidates at the path; flattening the first appends past
// the initial capacity, which used to realloc the buffer while a
// captured slice was still being iterated.
var big: [20]bson.Value = undefined;
for (0..20) |j| big[j] = .{ .int32 = @intCast(j) };
const d = doc_of(&.{.{ .key = "items", .value = .{ .array = &.{
.{ .doc = &.{.{ .key = "tags", .value = .{ .array = &big } }} },
.{ .doc = &.{.{ .key = "tags", .value = .{ .array = &.{.{ .string = "needle" }} } }} },
} } }});
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "items.tags", .value = .{ .string = "needle" } }}), &d));
// And the flattened long array is searched, not just the first element.
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "items.tags", .value = .{ .int32 = 19 } }}), &d));
}
test "OOM during value collection propagates, not a false match" {
// A tiny FixedBufferAllocator makes the candidate collection fail; the
// error must surface instead of leaving an empty candidate list, which
// would make negating operators like $ne report a match.
const d = doc_of(&.{.{ .key = "x", .value = .{ .int32 = 5 } }});
const ne = doc_of(&.{.{ .key = "x", .value = .{ .doc = &.{.{ .key = "$ne", .value = .{ .int32 = 5 } }} } }});
var buf: [16]u8 = undefined;
var fba = std.heap.FixedBufferAllocator.init(&buf);
try testing.expectError(error.OutOfMemory, matches(fba.allocator(), &ne, &d));
const ex = doc_of(&.{.{ .key = "x", .value = .{ .doc = &.{.{ .key = "$exists", .value = .{ .bool = false } }} } }});
try testing.expectError(error.OutOfMemory, matches(fba.allocator(), &ex, &d));
}
test "documents compare by field name too" {
try testing.expectEqual(std.math.Order.lt, bson.compare(
.{ .doc = &.{.{ .key = "a", .value = .{ .int32 = 1 } }} },
.{ .doc = &.{.{ .key = "b", .value = .{ .int32 = 1 } }} },
));
}
test "dot path filters" {
const d = doc_of(&.{
.{ .key = "user", .value = .{ .doc = &.{
.{ .key = "profile", .value = .{ .doc = &.{.{ .key = "age", .value = .{ .int32 = 25 } }} } },
} } },
.{ .key = "items", .value = .{ .array = &.{
.{ .doc = &.{.{ .key = "sku", .value = .{ .string = "x" } }} },
.{ .doc = &.{.{ .key = "sku", .value = .{ .string = "y" } }} },
} } },
});
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "user.profile.age", .value = .{ .int32 = 25 } }}), &d));
try testing.expect(!try matches(testing.allocator, &doc_of(&.{.{ .key = "user.profile.age", .value = .{ .int32 = 26 } }}), &d));
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "items.sku", .value = .{ .string = "y" } }}), &d));
}
test "array index dot path and bare regex value" {
const d = doc_of(&.{
.{ .key = "tags", .value = .{ .array = &.{
.{ .string = "c" },
.{ .string = "a" },
} } },
.{ .key = "name", .value = .{ .string = "carol" } },
});
// tags.0 addresses the first element.
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "tags.0", .value = .{ .string = "c" } }}), &d));
try testing.expect(!try matches(testing.allocator, &doc_of(&.{.{ .key = "tags.1", .value = .{ .string = "c" } }}), &d));
try testing.expect(!try matches(testing.allocator, &doc_of(&.{.{ .key = "tags.5", .value = .{ .string = "c" } }}), &d));
// Bare BSON regex value behaves like {$regex: "re"}.
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "name", .value = .{ .regex = .{ .pattern = "^c", .options = "" } } }}), &d));
try testing.expect(!try matches(testing.allocator, &doc_of(&.{.{ .key = "name", .value = .{ .regex = .{ .pattern = "^z", .options = "" } } }}), &d));
}
test "sort compares by BSON order" {
const a = doc_of(&.{ .{ .key = "n", .value = .{ .int32 = 2 } }, .{ .key = "x", .value = .{ .string = "a" } } });
const b = doc_of(&.{ .{ .key = "n", .value = .{ .int32 = 10 } }, .{ .key = "x", .value = .{ .string = "b" } } });
var arena = std.heap.ArenaAllocator.init(testing.allocator);
defer arena.deinit();
var docs = [_]*const bson.Document{ &b, &a };
const asc = [_]SortKey{.{ .path = "n", .descending = false }};
try sort_docs(arena.allocator(), &docs, &asc);
try testing.expect(docs[0] == &a);
try testing.expect(docs[1] == &b);
var docs2 = [_]*const bson.Document{ &a, &b };
const desc = [_]SortKey{.{ .path = "n", .descending = true }};
try sort_docs(arena.allocator(), &docs2, &desc);
try testing.expect(docs2[0] == &b);
const missing = [_]SortKey{.{ .path = "zz", .descending = false }};
try sort_docs(arena.allocator(), &docs2, &missing);
try testing.expect(docs2[0] == &b); // stable-ish: order untouched by missing key
}
test "projection inclusion and exclusion" {
const d = doc_of(&.{
.{ .key = "_id", .value = .{ .int32 = 1 } },
.{ .key = "a", .value = .{ .int32 = 1 } },
.{ .key = "b", .value = .{ .int32 = 2 } },
.{ .key = "nested", .value = .{ .doc = &.{
.{ .key = "x", .value = .{ .int32 = 3 } },
.{ .key = "y", .value = .{ .int32 = 4 } },
} } },
});
var arena = std.heap.ArenaAllocator.init(testing.allocator);
defer arena.deinit();
var out: std.ArrayListUnmanaged(bson.Pair) = .empty;
defer out.deinit(arena.allocator());
try project(arena.allocator(), &d, &doc_of(&.{
.{ .key = "a", .value = .{ .int32 = 1 } },
.{ .key = "nested.x", .value = .{ .int32 = 1 } },
}), &out);
try testing.expectEqual(@as(usize, 3), out.items.len);
try testing.expect(bson.get_pair(out.items, "a") != null);
try testing.expect(bson.get_pair(out.items, "b") == null);
const nx = bson.get_pair(out.items, "nested").?;
try testing.expectEqual(@as(usize, 1), nx.doc.len);
try testing.expectEqualStrings("x", nx.doc[0].key);
// {_id: 0} alone: everything except _id.
out.clearRetainingCapacity();
try project(arena.allocator(), &d, &doc_of(&.{.{ .key = "_id", .value = .{ .int32 = 0 } }}), &out);
try testing.expect(bson.get_pair(out.items, "_id") == null);
try testing.expect(bson.get_pair(out.items, "a") != null);
try testing.expect(bson.get_pair(out.items, "b") != null);
// Mixed inclusion with _id: 0: only listed fields, no _id.
out.clearRetainingCapacity();
try project(arena.allocator(), &d, &doc_of(&.{
.{ .key = "_id", .value = .{ .int32 = 0 } },
.{ .key = "a", .value = .{ .int32 = 1 } },
}), &out);
try testing.expect(bson.get_pair(out.items, "_id") == null);
try testing.expect(bson.get_pair(out.items, "a") != null);
try testing.expect(bson.get_pair(out.items, "b") == null);
out.clearRetainingCapacity();
try project(arena.allocator(), &d, &doc_of(&.{
.{ .key = "b", .value = .{ .int32 = 0 } },
.{ .key = "nested.x", .value = .{ .int32 = 0 } },
}), &out);
try testing.expect(bson.get_pair(out.items, "a") != null);
try testing.expect(bson.get_pair(out.items, "b") == null);
const n = bson.get_pair(out.items, "nested").?;
try testing.expectEqual(@as(usize, 1), n.doc.len);
try testing.expectEqualStrings("y", n.doc[0].key);
}
test "and/or filters" {
const d = doc_of(&.{ .{ .key = "a", .value = .{ .int32 = 1 } }, .{ .key = "b", .value = .{ .int32 = 2 } } });
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "$and", .value = .{ .array = &.{
.{ .doc = &.{.{ .key = "a", .value = .{ .int32 = 1 } }} },
.{ .doc = &.{.{ .key = "b", .value = .{ .int32 = 2 } }} },
} } }}), &d));
try testing.expect(try matches(testing.allocator, &doc_of(&.{.{ .key = "$or", .value = .{ .array = &.{
.{ .doc = &.{.{ .key = "a", .value = .{ .int32 = 99 } }} },
.{ .doc = &.{.{ .key = "b", .value = .{ .int32 = 2 } }} },
} } }}), &d));
try testing.expect(!try matches(testing.allocator, &doc_of(&.{.{ .key = "$or", .value = .{ .array = &.{
.{ .doc = &.{.{ .key = "a", .value = .{ .int32 = 99 } }} },
.{ .doc = &.{.{ .key = "b", .value = .{ .int32 = 99 } }} },
} } }}), &d));
}
/// Public single-value operator matcher, used by $pull and $elemMatch.
pub fn value_matches_operator(gpa: std.mem.Allocator, op: []const u8, value: bson.Value, actual: bson.Value) QueryError!bool {
var single: [1]bson.Value = .{actual};
return match_operator(gpa, op, value, single[0..], "");
}