//! Database engine over an mmap'd data file with the append-only log in front //! of it as the write-ahead log. Maps db -> collection -> `_id_` B+tree -> //! absolute slab offset; documents, tree pages and overflow records all live in //! the data file, so resident memory is the working set rather than the size of //! the database. All mutations are logged and synced before they become visible, //! so a crash never loses a committed write, and a checkpoint publishes the data //! file and truncates the log so an open does not replay everything ever //! written. Callers must hold the write lock (`lock`) around any command that //! mutates state, and the read lock (`lock_read`) around read-only commands so //! reads overlap with each other. //! //! Two invariants the rest of this file depends on. A checkpoint never renumbers //! slab offsets, because index leaves hold them physically -- only `compact` //! moves documents, and it rebuilds every index in the same pass. And an index //! must never under-approximate: it generates candidates and the full filter is //! re-applied to those, so a missing entry is a missing query result that //! nothing else detects (see `assert_indexes_cover_every_document`). const std = @import("std"); const builtin = @import("builtin"); const bson = @import("bson.zig"); const storage = @import("storage.zig"); const index = @import("index.zig"); const pgr = @import("pager.zig"); const cursor = @import("cursor.zig"); // Always active, including in the default ReleaseFast build -- see assert.zig // for why std.debug.assert is the wrong tool for these invariants. const assert = @import("assert.zig").assert; // For the durability invariants: a panic carries no expression text, so the // message is all an operator gets. const assert_msg = @import("assert.zig").assert_msg; /// Pages in a standard slab extent: 8 MiB, as the old in-memory segments were. /// Slack is bounded by one extent per collection. const slab_extent_pages: u32 = (8 * 1024 * 1024) / pgr.page_size; /// Shortest run worth taking off the free list for a slab, 1 MiB. Below this /// the extent is exhausted after a few documents, and every exhaustion writes /// off whatever is left of the one before it. const slab_run_min_pages: u32 = slab_extent_pages / 8; /// Pages a slab allocation of `len` bytes needs at minimum. fn pages_for(len: usize) u32 { return @intCast((len + pgr.page_size - 1) / pgr.page_size); } const LogKind = enum { upsert, delete, index_create, index_drop }; /// Dead bytes in one `map_align` window. The window is the unit of reclamation /// -- a whole system page is the smallest thing `mark_appendable` and /// `protect_stable` can hand back -- so a counter never exceeds `map_align`, /// and the width follows from that rather than being chosen. /// /// Two bytes per window on the usual platforms. That is the entire memory cost /// of knowing where a collection's garbage is: 2.7 MB for a 21 GB slab on /// 16 KiB pages, 10.6 MB on 4 KiB ones. The alternative shapes -- an interval /// set, a free-run list -- cost memory proportional to the number of *dead /// documents*, which for 200-byte documents at that scale is gigabytes, and /// would make `evict_doc` allocate after the write is already committed. const WindowDead = if (pgr.map_align <= std.math.maxInt(u16)) u16 else u32; /// A run of pages a collection's slab owns, plus where its garbage is. /// /// This replaced a bare `pgr.Extent` because an extent can only be given back /// whole, and a churning collection almost never empties one. A run is split /// instead: the windows inside it with nothing live left go to the pager and /// the run becomes two shorter ones. So the list is kept sorted by page number, /// which makes `run_of` a binary search and the sortedness itself an assert -- /// allocation order stopped being meaningful once a run could be recycled to a /// *lower* address than one already owned. const SlabRun = struct { first: u32, pages: u32, /// The run's start, rounded up to `map_align`: the first offset that begins /// a whole window. `alloc_pages` works in 4 KiB pages, so a run need not /// start on a system page. window_first: u64, /// Dead bytes per window, `dead[i]` covering /// `[window_first + i*map_align, +map_align)`. `map_align` means the window /// holds nothing live and can be handed back. dead: []WindowDead, /// Windows wholly inside the pages `[first, first+pages)`. The bytes /// outside them -- below `window_first`, and the tail after the last whole /// window -- are real slab that documents do live in; their garbage is /// counted in `Collection.dead_unlocated` instead, because it can never be /// reclaimed on its own. fn window_count(first: u32, pages: u32) usize { const from = @as(u64, first) << pgr.page_shift; const to = from + (@as(u64, pages) << pgr.page_shift); const wf = std.mem.alignForward(u64, from, pgr.map_align); const we = std.mem.alignBackward(u64, to, pgr.map_align); return if (we > wf) @intCast((we - wf) / pgr.map_align) else 0; } fn start(self: SlabRun) u64 { return @as(u64, self.first) << pgr.page_shift; } fn end(self: SlabRun) u64 { return (@as(u64, self.first) + self.pages) << pgr.page_shift; } /// One past the last byte covered by a window counter. fn window_end(self: SlabRun) u64 { return self.window_first + self.dead.len * pgr.map_align; } }; pub const Collection = struct { /// Documents live as canonical BSON bytes in the data file, in extents this /// collection owns; the map holds each document's offset. Those are /// *absolute file offsets* now, which is what makes doc_bytes a single add /// rather than a binary search over segment starts -- and what removes the /// dangling-pointer hazard the old segment list had, since the mapping's /// base never moves. /// /// Removed documents leave garbage bytes until a rebuild rewrites them. A /// checkpoint must never renumber these offsets: every index leaf holds one /// (PLAN amendment A3). /// Live documents. The `_id_` index is the lookup now, so this is only a /// count -- kept because the compaction trigger and `collStats` want it and /// the tree cannot answer it in O(1). doc_count: u64, /// The data file this collection's documents live in. pager: *pgr.Pager, /// Page runs owned by this collection's slab, sorted by page number, each /// carrying the map of where its dead bytes are. See `SlabRun`. slab_runs: std.ArrayListUnmanaged(SlabRun), /// Absolute file offset of the next document write, and the end of the /// extent it falls in. slab_tail: u64, slab_end: u64, /// Slab this collection has consumed and not yet given back. `slab_tail` /// cannot answer that -- it is an absolute file offset, so it jumps forward /// whenever a fresh extent is taken. /// /// It used to mean "bytes ever appended since the last rebuild", which was /// the same thing while a rebuild was the only way to get slab back. Window /// reclamation subtracts from it, and that is what keeps /// `slab_used - live_bytes` equal to the garbage the collection still has /// -- with no new persistent field, since both halves are already in the /// catalog. slab_used: u64, /// This collection's outstanding page promise, for the document slab. Per /// collection because concurrent writers must not release each other's -- /// see `pager.Reservation`. hold: pgr.Reservation, /// Of those bytes, the ones still reachable. `slab_used - live_bytes` is /// this collection's slab garbage, which only a rebuild reclaims. Kept per /// collection so dropping one can move the right amount from the engine's /// live total to its dead total. live_bytes: u64, /// Garbage this collection knows it has but cannot place in a window: the /// edges of a run that fall outside any whole window, and -- the larger /// share -- everything that died before the last restart, since the window /// map is not persisted. /// /// It exists to keep one identity exact: /// /// sum of every window counter + dead_unlocated == slab_used - live_bytes /// /// Without it the two halves of the accounting would drift apart at every /// open, and there would be no assert that could tell drift from a lost /// update. What it costs is only that garbage from before a restart is not /// reclaimed window-wise; it still arms compaction like any other. dead_unlocated: u64, /// Windows whose counter has reached `map_align`, i.e. how much there is /// for the next checkpoint to give back. /// /// It exists so that a checkpoint costs nothing on a collection with /// nothing to reclaim. Scanning would otherwise be O(slab) per checkpoint /// whatever the workload -- and the workload this design is *known* not to /// help, small documents on large system pages, is exactly the one that /// would pay that for no return. full_windows: u32, /// Slab handed back to the pager by window reclamation, cumulative for the /// life of the process. Purely an observation: it is what distinguishes /// "the ratio improved because reclamation worked" from "the ratio improved /// for some other reason", which is the only way to read the churn gate. reclaimed_bytes: u64, /// Secondary indexes (persisted through the log). Heap-allocated, so an /// `*Index` handed out by `find_index` or `create_index` stays valid when /// a sibling index is dropped. Held by value, `orderedRemove` memmoved the /// whole ~5 KB struct and every live pointer into the list -- a query /// plan's `index` field, or a slice into an index's promoted-key buffer -- /// silently aimed at a different index or past the end. Nothing exercised /// that concurrently yet; the mmap work makes it worse, since an Index /// will own a mapping. indexes: std.ArrayListUnmanaged(*index.Index), /// Guards this collection's docs/slab/indexes. Writers take it /// exclusive, readers shared; never held while taking the catalog lock, /// and never more than one collection lock at a time. lock: std.Io.RwLock = .init, /// The secondary index that rejected the most recent unique write /// (duplicate-key error path); per-collection so concurrent writers on /// other collections cannot clobber it mid-command. dup_index: ?[]const u8 = null, /// The implicit _id_ index: every document has an _id and it is not /// sparse, so entry count equals document count and a full scan of it /// cannot miss a document — which is what the sort planner's full-scan /// plan relies on. Kept out of `indexes` so the listing/drop commands /// and the log format are unchanged (it is rebuilt on open like /// everything else). `bson.encode_key` keys are canonical, so it also /// replaces the old serialization-guarded docs-map fast path for /// integer/string/etc. _id lookups. id_index: index.Index, /// Identity-and-layout token for open cursors. Drawn from /// `Engine.layout_epoch_seq`, so it is unique across the engine's life and /// bumped again by every rebuild. /// /// It answers two questions a cursor cannot answer any other way. A rebuild /// moves every document, so a saved slab offset (or a saved index anchor's /// offset) is stale -- and the keys surviving unchanged makes that *worse*, /// because a lookup then succeeds and quietly resolves to the wrong bytes. /// And a cursor holds namespace *strings*, not a `*Collection`, so a /// drop-and-recreate under the same name would otherwise be invisible to it; /// drawing from an engine-wide sequence rather than starting each collection /// at zero is what makes the recreated one compare unequal. layout_epoch: u64, fn init(gpa: std.mem.Allocator, pager: *pgr.Pager, layout_epoch: u64) !Collection { var self: Collection = .{ .doc_count = 0, .pager = pager, .slab_runs = .empty, .slab_tail = 0, .slab_end = 0, .slab_used = 0, .live_bytes = 0, .dead_unlocated = 0, .full_windows = 0, .reclaimed_bytes = 0, .hold = .{}, .indexes = .empty, .id_index = undefined, .layout_epoch = layout_epoch, }; const keys = [_]index.IndexKey{.{ .path = "_id", .descending = false }}; // unique: the tree, not the docs map, is what enforces _id uniqueness // now (PLAN A3/A4). It is keyed on bson.encode_key, which is canonical // where serialize_value is not, so int32 1 / int64 1 / double 1.0 // collide as they do in MongoDB -- see the migration note in // apply_record. self.id_index = try index.Index.init(gpa, pager, "_id_", &keys, true, false, null); return self; } /// The secondary index with this name, or null. The single by-name /// lookup: index lifetime (who calls Index.deinit, and when) is decided /// here rather than at each caller. pub fn find_index(self: *Collection, name: []const u8) ?*index.Index { for (self.indexes.items) |ix| { if (std.mem.eql(u8, ix.name, name)) return ix; } return null; } /// Take ownership of a page run, keeping `slab_runs` sorted by page number. /// The only place a run enters the list, so the sort order and the window /// map are established together and cannot disagree. fn insert_run(self: *Collection, gpa: std.mem.Allocator, first: u32, pages: u32) !void { const dead = try gpa.alloc(WindowDead, SlabRun.window_count(first, pages)); errdefer gpa.free(dead); @memset(dead, 0); var at: usize = 0; while (at < self.slab_runs.items.len and self.slab_runs.items[at].first < first) at += 1; // A recycled run must not overlap one this collection already owns: // that would be the pager handing out pages twice, and the symptom // would be a document quietly overwritten rather than anything failing. if (at > 0) { const prev = self.slab_runs.items[at - 1]; assert_msg(prev.first + prev.pages <= first, "a slab run overlaps the one below it"); } if (at < self.slab_runs.items.len) { assert_msg(first + pages <= self.slab_runs.items[at].first, "a slab run overlaps the one above it"); } try self.slab_runs.insert(gpa, at, .{ .first = first, .pages = pages, .window_first = std.mem.alignForward(u64, @as(u64, first) << pgr.page_shift, pgr.map_align), .dead = dead, }); } /// The run holding `off`, or null if no run does. Binary search, which the /// sorted list is for: `mark_dead` runs once per evicted document, and a /// collection with a fragmented slab can own thousands of runs. fn run_of(self: *const Collection, off: u64) ?usize { const page: u32 = @intCast(off >> pgr.page_shift); var lo: usize = 0; var hi: usize = self.slab_runs.items.len; while (lo < hi) { const mid = lo + (hi - lo) / 2; const r = self.slab_runs.items[mid]; if (page < r.first) { hi = mid; } else if (page >= r.first + r.pages) { lo = mid + 1; } else { return mid; } } return null; } /// Record that `[off, off+len)` of slab is garbage. /// /// Infallible, and that is the constraint the whole representation was /// chosen around: the two callers are `evict_doc`, which runs after the /// log record is already durable, and the appender's skip accounting. An /// allocation here would be a failure with nowhere to report it. /// /// Bytes that fall outside a whole window -- the head of a run before its /// first window boundary, and the tail after its last -- go to /// `dead_unlocated`. They are not lost, only unreclaimable on their own. fn mark_dead(self: *Collection, off: u64, len: u64) void { if (len == 0) return; const ri = self.run_of(off) orelse { assert_msg(false, "dead slab bytes fall outside every run the collection owns"); unreachable; }; const r = &self.slab_runs.items[ri]; const stop = off + len; // A document is written inside one extent by construction // (`slab_reserve` never lets an append cross `slab_end`), so a dead // range that crosses a run boundary means an offset from a different // layout -- a stale index entry, which is the failure the layout epoch // exists to prevent. assert_msg(stop <= r.end(), "a dead slab range crosses the end of the run holding it"); var pos = off; if (pos < r.window_first) { const n = @min(stop, r.window_first) - pos; self.dead_unlocated += n; pos += n; } const win_end = r.window_end(); while (pos < stop and pos < win_end) { const w: usize = @intCast((pos - r.window_first) / pgr.map_align); const w_end = r.window_first + (w + 1) * pgr.map_align; const n = @min(stop, w_end) - pos; // A window cannot hold more dead bytes than it has bytes. Tripping // this means the same range was marked twice -- a double eviction, // or a recycled offset marked against the previous owner's map. assert_msg(r.dead[w] + n <= pgr.map_align, "a slab window holds more dead bytes than it has"); const was_full = r.dead[w] == pgr.map_align; r.dead[w] += @intCast(n); if (!was_full and r.dead[w] == pgr.map_align) self.full_windows += 1; pos += n; } if (pos < stop) self.dead_unlocated += stop - pos; } /// Garbage this collection has placed in windows. Walks every window, so it /// belongs to the reclamation scan and to tests, not to a hot path. fn dead_located(self: *const Collection) u64 { var sum: u64 = 0; for (self.slab_runs.items) |r| { for (r.dead) |d| sum += d; } return sum; } /// Drop the window maps and the run list. The pages themselves are the /// caller's business -- a drop hands them to the pager, a rebuild has /// already done so. fn free_runs(self: *Collection, gpa: std.mem.Allocator) void { for (self.slab_runs.items) |r| gpa.free(r.dead); self.slab_runs.clearRetainingCapacity(); self.full_windows = 0; } /// One piece of a run that survives reclamation, with a window map of its /// own copied out of the original. /// /// Every kept piece gets a fresh array, including a run nothing was taken /// from. Moving the original array instead would save a copy and make the /// failure path have to know which arrays it still owns -- the version that /// tried it had a double free in the out-of-memory case, which is the one /// case nothing exercises. fn keep_piece( out: *std.ArrayListUnmanaged(SlabRun), gpa: std.mem.Allocator, r: SlabRun, p0: u32, p1: u32, ) !void { const wf = std.mem.alignForward(u64, @as(u64, p0) << pgr.page_shift, pgr.map_align); const we = std.mem.alignBackward(u64, @as(u64, p1) << pgr.page_shift, pgr.map_align); const count: usize = if (we > wf) @intCast((we - wf) / pgr.map_align) else 0; const dead = try gpa.alloc(WindowDead, count); errdefer gpa.free(dead); // A piece boundary is either the run's own start/end or a window // boundary, so the piece's windows line up with a contiguous stretch of // the original's and the counters can be copied rather than rebuilt. const base: usize = @intCast((wf - r.window_first) / pgr.map_align); @memcpy(dead, r.dead[base..][0..count]); try out.append(gpa, .{ .first = p0, .pages = p1 - p0, .window_first = wf, .dead = dead }); } /// Give back every window with nothing live left in it, splitting the runs /// around what is kept. Returns the bytes handed to the pager. /// /// Counting is the whole test: a window reaches `map_align` dead only once /// every document with a byte in it has been through `evict_doc`, which /// removes its index entries before marking it. So "no live bytes" and "no /// reference to these bytes" are the same statement, and nothing has to be /// scanned to establish it. /// /// Fallible, and arranged so a failure changes nothing: the replacement /// list is built whole before the old one is touched. The garbage simply /// stays and the next checkpoint tries again. fn reclaim_windows(self: *Collection, gpa: std.mem.Allocator) !u64 { assert_msg( self.slab_used >= self.live_bytes, "a collection cannot hold more live bytes than it ever appended", ); // The identity, checked where every window is being walked anyway. assert_msg( self.dead_located() + self.dead_unlocated == self.slab_used - self.live_bytes, "the collection's placed and unplaced garbage must add up to its garbage", ); var out: std.ArrayListUnmanaged(SlabRun) = .empty; errdefer { for (out.items) |p| gpa.free(p.dead); out.deinit(gpa); } var give: std.ArrayListUnmanaged(pgr.Extent) = .empty; defer give.deinit(gpa); var freed: u64 = 0; for (self.slab_runs.items) |r| { var keep_from = r.first; var i: usize = 0; while (i < r.dead.len) { if (r.dead[i] != pgr.map_align) { i += 1; continue; } var j = i + 1; while (j < r.dead.len and r.dead[j] == pgr.map_align) j += 1; const from = r.window_first + i * pgr.map_align; const to = r.window_first + j * pgr.map_align; // The appender's own extent is off limits, and not by // filtering: bytes above the cursor have never been written, so // no window covering them can have reached `map_align` dead. // Tripping this means a range was marked dead twice. assert_msg( to <= self.slab_tail or from >= self.slab_end, "reclaiming a slab window the append cursor is still walking", ); const p_from: u32 = @intCast(from >> pgr.page_shift); const p_to: u32 = @intCast(to >> pgr.page_shift); if (p_from > keep_from) try keep_piece(&out, gpa, r, keep_from, p_from); try give.append(gpa, .{ .first = p_from, .pages = p_to - p_from }); freed += to - from; keep_from = p_to; i = j; } if (keep_from < r.first + r.pages) { try keep_piece(&out, gpa, r, keep_from, r.first + r.pages); } } if (freed == 0) { for (out.items) |p| gpa.free(p.dead); out.deinit(gpa); // Every full window was given back or there were none, so nothing // is left for the next checkpoint to find. self.full_windows = 0; return 0; } // Past the last fallible step: swap the list in, then hand the pages // over. A `free_pages` that fails here leaks the run -- it is no longer // the collection's and not yet the pager's -- which costs space and // nothing else. The other order would leave the same pages owned twice. for (self.slab_runs.items) |r| gpa.free(r.dead); self.slab_runs.deinit(gpa); self.slab_runs = out; self.full_windows = 0; for (give.items) |e| self.pager.free_pages(e.first, e.pages) catch {}; assert_msg(self.slab_used >= self.live_bytes + freed, "reclaiming more slab than the collection has"); self.slab_used -= freed; self.reclaimed_bytes += freed; return freed; } /// Append `bytes` to the slab, returning its flat offset. The last /// segment holds up to `slab_segment_size`; a full one starts the next. /// Make room for a document of `len` bytes, so the append that follows /// cannot fail. /// /// Separated from the append because the append runs *after* the log /// record is durable, where failure has nowhere to go: the write is already /// committed and reporting an error for it would be a lie the next open /// contradicts. Reserving first keeps the fallible half before the log. /// /// Returns the slab bytes it wrote off along the way, for the caller to /// charge to the engine's dead total. See `note_skip`. fn slab_reserve(self: *Collection, gpa: std.mem.Allocator, len: usize) !u64 { // A checkpoint can land in the middle of an extent, which freezes the // page the tail points into. Appending there would store inside the // durable image, so abandon the rest of the extent and start a fresh // one. The waste is bounded by one extent per collection per checkpoint. // // `is_unpublished_at` rather than a comparison against the stable mark: // an extent recycled off the free list starts *below* the mark and is // still writable. Asking the mark meant every recycled extent was thrown // away after one document, so churn never reused anything. // // Room is checked from the *rounded-up* cursor rather than the cursor // itself, so the round-up `slab_append` may have to do is guaranteed to // fit. Without that the append's own re-check could discover it needs a // fresh extent, which is fallible, after the log record is already // durable. Costs under one system page per extent. if (self.pager.is_unpublished_at(self.slab_tail) and self.appendable_end(len)) return 0; // The page holding the tail is frozen, but the *rest* of the extent is // not: nothing above the live cursor is referenced by the image or by an // index. So skip to the next system page and keep the extent, instead of // throwing away what is left of 8 MiB. // // This is what the plan called for ("append cursors are rounded up to the // system page size at each checkpoint") and it matters more than it // sounds: abandoning the extent costs ~8 MiB per collection per // checkpoint, and a pure-insert workload generates no garbage, so // compaction never fires and nothing ever gives it back. Measured at 40 // collections: the data file reached 11.8x the live data and grew by // ~335 MB per checkpoint, heading for DatabaseTooLarge at around 6 GB of // real data. if (self.appendable_end(len)) { const resumed = std.mem.alignForward(u64, self.slab_tail, pgr.map_align); const skipped = self.note_skip(resumed - self.slab_tail); self.pager.mark_appendable(resumed, self.slab_end); self.slab_tail = resumed; return skipped; } // Nothing will ever be written between the cursor and the end of the // extent this collection is walking away from -- the extent stays // allocated to it and every byte above the cursor is unreachable. That // is garbage, in the whole 8 MiB, and the reservation is the only place // that knows about it. assert_msg(self.slab_tail <= self.slab_end, "the slab cursor is past the end of its extent"); const skipped = self.note_skip(self.slab_end - self.slab_tail); // A document larger than the standard extent gets one of its own; BSON // reaches 16 MB and the extent is 8 MiB. const want_pages: u32 = @max(slab_extent_pages, pages_for(len)); try self.pager.reserve_pages(&self.hold, want_pages); // Off the free list first, or window reclamation is decorative: the // pages come back, nothing asks for them in a shape they arrive in, and // the file grows by the whole write volume anyway. A floor of 1 MiB, // because a shorter extent is exhausted after a handful of documents // and every exhaustion abandons what is left of it -- and because the // floor is what makes trimming a larger run harmless. const min_pages: u32 = @min(want_pages, @max(pages_for(len), slab_run_min_pages)); const run = self.pager.alloc_slab_run(&self.hold, min_pages, want_pages) orelse pgr.Extent{ .first = self.pager.alloc_pages_assume_reserved(&self.hold, want_pages), .pages = want_pages, }; try self.insert_run(gpa, run.first, run.pages); self.slab_tail = @as(u64, run.first) << pgr.page_shift; self.slab_end = self.slab_tail + (@as(u64, run.pages) << pgr.page_shift); return skipped; } /// Count `bytes` of slab that no document will ever occupy, and hand the /// same number back so the caller can charge the engine's dead total. /// /// The appender skips slab in two places -- rounding the cursor up to a /// system page after a checkpoint froze the page it pointed into, and /// abandoning the tail of an extent that no longer has room. Neither used to /// be counted anywhere: not in `slab_used`, which only ever grew by a /// document's length, and so not in the engine's `dead_bytes` either. It is /// garbage all the same -- only a rebuild gets it back -- and it was /// invisible to the trigger that decides whether a rebuild is worth doing. /// /// Two collections churning against a checkpoint every 32 MiB skip up to a /// system page each per checkpoint, and an abandoned extent tail can be /// most of 8 MiB. Counted here, that garbage arms compaction like any other. /// /// Skipped slab always starts at the cursor -- all three callers write off /// the bytes in front of it and then move it -- so this is also where the /// window map learns about it. That matters more for the abandoned tail /// than for the round-up: most of 8 MiB of a run is whole windows, dead on /// arrival, and reclaiming them is free. fn note_skip(self: *Collection, bytes: u64) u64 { self.slab_used += bytes; self.mark_dead(self.slab_tail, bytes); return bytes; } /// Whether a document of `len` bytes fits in this extent even if the cursor /// first has to be rounded up to a system page. The reservation and the /// append both ask this, so they agree on what "there is room" means. fn appendable_end(self: *const Collection, len: usize) bool { return std.mem.alignForward(u64, self.slab_tail, pgr.map_align) + len <= self.slab_end; } /// Where a document landed, and what the landing cost besides its own /// length. `skipped` is `note_skip`'s tally for this append. const Appended = struct { off: u64, skipped: u64 }; /// Copy `bytes` into the slab and return its absolute file offset. /// Infallible: slab_reserve must have run for at least this many bytes. fn slab_append(self: *Collection, bytes: []const u8) Appended { // The cursor was checked in `slab_reserve`, but a checkpoint can have // published since -- the reservation runs before the log append and this // runs after it, with an fsync in between. `publish` clears the whole // unpublished set, so a cursor that was writable then can be inside the // frozen image now, and the copy below would store into it: a bus error // where the protection is compiled in, and a silent overwrite of durable // data in ReleaseFast, where it is not. // // Re-arming is infallible because `slab_reserve` measured its room from // the rounded-up cursor. The pager's append lock holds off the next // publish for the rest of this function, so the answer stays true. self.pager.lock_append(); defer self.pager.unlock_append(); var skipped: u64 = 0; if (!self.pager.is_unpublished_at(self.slab_tail)) { const resumed = std.mem.alignForward(u64, self.slab_tail, pgr.map_align); skipped = self.note_skip(resumed - self.slab_tail); self.pager.mark_appendable(resumed, self.slab_end); self.slab_tail = resumed; } assert_msg( self.slab_tail + bytes.len <= self.slab_end, "document append overran the slab reservation", ); const off = self.slab_tail; @memcpy(self.pager.bytes_mut(off, bytes.len), bytes); self.slab_tail += bytes.len; self.slab_used += bytes.len; // Here rather than at the call site: a rebuild appends through this same // path, and its copies are live by definition. self.live_bytes += bytes.len; return .{ .off = off, .skipped = skipped }; } /// The canonical bytes of the document stored at `off` — a slice into a /// segment, stable until the collection is freed or rebuilt. pub fn doc_bytes(self: *const Collection, off: u64) []const u8 { // An absolute file offset, so this is base + off. The length comes from // the document's own BSON int32 prefix, as it always has. const len: usize = std.mem.readInt(u32, self.pager.bytes(off, 4)[0..4], .little); return self.pager.bytes(off, len); } /// Remove and free the index with this name. Returns whether it existed. /// `orderedRemove` now moves 8-byte pointers rather than whole Index /// structs, so the surviving indexes do not move and pointers to them stay /// valid; only the removed one dies, here. fn remove_index(self: *Collection, gpa: std.mem.Allocator, name: []const u8) bool { for (self.indexes.items, 0..) |ix, i| { if (std.mem.eql(u8, ix.name, name)) { _ = self.indexes.orderedRemove(i); ix.deinit(gpa); gpa.destroy(ix); return true; } } return false; } }; pub const Db = struct { /// Collections are heap-allocated so their addresses are stable while a /// command holds a collection lock — the map may reallocate under the /// catalog lock, but the pointers it holds do not move. collections: std.StringHashMapUnmanaged(*Collection), }; pub const Engine = struct { gpa: std.mem.Allocator, io: std.Io, // Legacy whole-engine lock, used by the unit tests' explicit // lock()/lock_read() calls. The server uses the finer-grained locks // below: catalog (maps), per-collection (docs/slab/indexes), and // log_lock (append + commit). rwlock: std.Io.RwLock, /// Guards the dbs/collections maps. Commands hold it shared for their /// whole duration so a concurrent DDL cannot mutate the maps under /// them; DDL takes it exclusive. catalog_lock: std.Io.RwLock = .init, /// Serializes log appends, seals and the commit sync. log_lock: std.Io.Mutex = .init, /// Serializes commit decisions; the group-commit leader holds it while /// sealing and syncing. commit_lock: std.Io.Mutex = .init, /// Sequence number covered by the last completed commit. A seq rather /// than a file position: an append leaves its bytes in the log's open /// block without moving end_pos, so a position comparison would call /// buffered-but-unwritten records durable. committed_seq: u64 = 0, /// Writers increment before appending and decrement after; the commit /// leader waits for this to reach zero so its seal covers every append /// in flight, coalescing many writers' fsyncs into one. pending_appends: std.atomic.Value(u32) = std.atomic.Value(u32).init(0), committing: bool = false, commit_done: std.Io.Condition = std.Io.Condition.init, /// Set when the garbage ratio crosses the compaction threshold; the /// write command's epilogue runs compact after releasing its locks. /// Atomic because it is set under a *collection* lock (see `note_compact`) /// but read by the epilogue holding no lock at all. compact_pending: std.atomic.Value(bool) = std.atomic.Value(bool).init(false), /// Set while a compaction runs, so only one runs at a time. Compactions /// share one tmp path and each ends in a rename onto the log, so two at /// once would publish one compaction's half-written file as the database. compacting: std.atomic.Value(bool) = std.atomic.Value(bool).init(false), /// Collections rewritten by a rebuild since the process started. Reported by /// `serverStatus`, because "the ratio improved" and "the ratio improved /// because reclamation worked rather than because a rebuild ran" are /// different results and no ratio distinguishes them. Under `counter_lock`. compactions: u64 = 0, log: storage.Log, /// The data file: documents live here, and the B+tree arenas follow. /// /// Heap-allocated because `open` builds an Engine on the stack and returns /// it by value: every Collection holds a `*Pager`, and those were taken /// during replay, before the move. They all dangled -- which surfaced as a /// corrupt docs hashmap on the *second* engine in a test, not as anything /// resembling its cause. pager: *pgr.Pager, dbs: std.StringHashMapUnmanaged(Db), seq: u64, /// Floor for the compaction trigger. The real trigger also scales with /// the live data size — see `note_compact`. compact_threshold: u64, /// Documents currently resident across every collection, and documents /// superseded or deleted since the last compaction. Their ratio is the /// share of the log that is garbage, which is what decides whether a /// rewrite is worth doing — see `note_compact`. live_docs: u64 = 0, dead_docs: u64 = 0, /// Hands out `Collection.layout_epoch` values. Monotonic and never reset, so /// no two collection instances -- including a drop followed by a recreate /// under the same name -- ever share one. layout_epoch_seq: u64 = 0, /// Open cursors. Lives on the engine rather than the server because the C /// API seam (PLAN D1) lists cursor iteration, and because the unit tests /// build an Engine with no server at all. Its mutex is a leaf: see /// `cursor.Store`. cursors: cursor.Store, /// The same question in bytes, about the *data file* rather than the log. /// Once a checkpoint truncates the log, the log no longer holds the garbage /// -- the doc slab does, and only a rebuild reclaims it. These are what /// `note_compact` gates on; counting documents would let one collection of /// 16 KiB documents and one of 40 B documents look identical. live_bytes: u64 = 0, dead_bytes: u64 = 0, /// Guards the four counters above -- `live_docs`, `dead_docs`, /// `live_bytes`, `dead_bytes` -- and nothing else. /// /// They are the only engine-wide mutable state a writer touches while /// holding nothing but its own collection's lock, so two writers on /// different collections reach them with no lock in common. The lost update /// that allows is the smaller half of the problem. The larger half is that a /// reader had no way to see them consistently with the per-collection totals /// they are supposed to equal: `checkpoint` reads each collection's counters /// under that collection's lock -- which orders it against that /// collection's writer -- and then read these with no lock at all, so it /// could see a total that predated a write it had just serialized. Its own /// assertion then aborted the server, correctly, about a database that was /// consistent. /// /// A leaf: nothing else is taken while it is held, and it is never held /// across an append, an fsync, or an allocation. counter_lock: std.Io.Mutex = .init, /// The checkpoint's own page promise, for the catalog and free-list pages it /// writes. Separate from any collection's for the same reason those are /// separate from each other. hold: pgr.Reservation = .{}, /// Set when the log has grown enough since the last checkpoint to be worth /// reclaiming. Read by the write epilogue and the TTL monitor, both of which /// run without holding a collection lock. checkpoint_pending: std.atomic.Value(bool) = std.atomic.Value(bool).init(false), /// Log bytes that trigger a checkpoint. Distinct from the compaction /// threshold: compaction is about the *garbage share* of the data, a /// checkpoint is about how much replay an open would otherwise have to do. checkpoint_threshold: u64 = 32 * 1024 * 1024, /// Whether replay must maintain index entries as it goes. /// /// A full replay does not: it puts documents in place and lets /// `build_all_indexes` bulk-pack every index afterwards, which is O(n log n) /// once instead of per record. After a checkpoint that is wrong -- the /// indexes arrive already populated, `rebuild_index` skips a non-empty one by /// design, and the records replayed on top would be invisible to every /// index. The symptom was a document present in the collection and missing /// from `_id_`, which after the hashmap goes away means simply missing. replay_maintains_indexes: bool = false, /// Set to the failing index's own stable name when an upsert is /// rejected by a unique secondary index (error.DuplicateKeyIndex). The /// command reads it while still holding the write lock. dup_index: ?[]const u8 = null, /// The registry an embedded caller gets without configuring anything; the /// CLI replaces it through `reconfigure_cursors`. fn default_cursor_store(gpa: std.mem.Allocator, io: std.Io) !cursor.Store { return cursor.Store.init(gpa, io, cursor.default_capacity, cursor.default_idle_timeout_ms); } pub fn open(gpa: std.mem.Allocator, io: std.Io, path: []const u8) !Engine { var log = try storage.Log.open(gpa, io, path); errdefer log.close(); // The data file sits beside the log and is *kept*: a valid watermark in // it means most of the log never has to be replayed. const data_path = try std.fmt.allocPrint(gpa, "{s}.data", .{log.path}); defer gpa.free(data_path); const pager_box = try gpa.create(pgr.Pager); errdefer gpa.destroy(pager_box); pager_box.* = try pgr.Pager.open(gpa, io, data_path, .{}); var engine = Engine{ .gpa = gpa, .io = io, .rwlock = .init, .log = log, .pager = pager_box, .dbs = .empty, .seq = 0, .compact_threshold = 16 * 1024 * 1024, .cursors = try default_cursor_store(gpa, io), }; errdefer { engine.cursors.deinit(); engine.pager.deinit(); engine.dbs.deinit(gpa); } // A checkpoint, if the data file has one, decides where replay starts. // Nothing below the watermark needs re-applying: the data file already // holds its effect. var replay_from: u64 = 0; if (engine.pager.loaded.generation != 0) { engine.read_catalog() catch |err| { // The image is unusable but the log is not. Warn, drop // everything loaded, and fall back to a full replay -- the // database must always open. std.debug.print( "multiforadb: WARNING: data file catalog unreadable ({s}); " ++ "replaying the log in full\n", .{@errorName(err)}, ); engine.reset_after_failed_catalog(); replay_from = 0; }; if (replay_from == 0 and engine.dbs.count() > 0) { replay_from = engine.pager.loaded.seq; engine.replay_maintains_indexes = true; engine.seq = replay_from; engine.committed_seq = replay_from; engine.live_docs = engine.pager.loaded.live_docs; // `dead_bytes` is *not* restored from the watermark. It is // derived, not stored: `read_catalog` has already summed // `slab_used - live_bytes` over the collections the catalog // still lists. The watermark's copy is a hint for anything // inspecting the header without parsing the catalog, and it // would be wrong here in one specific way -- a collection // dropped after the last checkpoint takes its garbage with it, // and the hint would keep charging the engine for it. } } try engine.log.replay(&engine, apply_record, replay_from); // Replay registers empty indexes; build them from the live docs // once replay completes (order-independent). A checkpointed open finds // them already populated, and the guard in rebuild_index skips them. try engine.build_all_indexes(); engine.assert_indexes_cover_every_document(); // Everything replayed is durable by definition -- it was read back off // the log -- so the commit watermark starts level with the sequence. engine.committed_seq = engine.seq; return engine; } /// Replace the cursor registry with one of a different shape. Only legal /// before the server starts accepting connections, because it drops every /// cursor -- asserted rather than left to the comment, since the method is /// public and a later caller would otherwise get silent data loss. pub fn reconfigure_cursors(self: *Engine, capacity: u32, idle_timeout_ms: i64) !void { assert_msg(self.cursors.live == 0, "reconfigured the cursor registry with cursors open"); const fresh = try cursor.Store.init(self.gpa, self.io, capacity, idle_timeout_ms); self.cursors.deinit(); self.cursors = fresh; } pub fn deinit(self: *Engine) void { var db_it = self.dbs.iterator(); while (db_it.next()) |db_entry| { self.free_db(db_entry.value_ptr); self.gpa.free(db_entry.key_ptr.*); } self.dbs.deinit(self.gpa); // Before the pager: a cursor's arena is its own, but freeing cursors // first keeps the teardown order the same as the construction order // reversed, which is the only order that stays obviously correct as // cursors grow to hold more. self.cursors.deinit(); self.pager.deinit(); self.gpa.destroy(self.pager); self.log.close(); } /// Slab a reservation wrote off, on the engine's books. Its own acquisition /// rather than the reservation's caller adding to the field: skips cluster /// at a checkpoint -- `publish` freezes every collection's append cursor at /// once, so the next write to each of them skips -- which is precisely when /// several writers reach this counter at the same moment. fn count_slab_skip(self: *Engine, bytes: u64) void { if (bytes == 0) return; self.counter_lock.lockUncancelable(self.io); defer self.counter_lock.unlock(self.io); self.dead_bytes += bytes; } /// A snapshot of the four counters, taken together. Both readers reason /// about a *relation* -- the compaction trigger about the ratio of two of /// them, the checkpoint about how they compare to the sum over collections /// -- so reading them one at a time would be comparing two moments. const Counters = struct { live_docs: u64, dead_docs: u64, live_bytes: u64, dead_bytes: u64 }; fn counters(self: *Engine) Counters { self.counter_lock.lockUncancelable(self.io); defer self.counter_lock.unlock(self.io); return .{ .live_docs = self.live_docs, .dead_docs = self.dead_docs, .live_bytes = self.live_bytes, .dead_bytes = self.dead_bytes, }; } /// Free every document in a collection along with its owned _id keys /// and secondary indexes (whose entries alias the documents — freed /// first). fn free_collection(self: *Engine, coll: *Collection) void { self.counter_lock.lockUncancelable(self.io); // Dropping a collection turns all of its records into garbage. The // engine's live count includes every collection's documents, so it can // never be smaller than this one's -- and a u64 underflow here would // read as an astronomically large live count, permanently suppressing // compaction rather than crashing. assert_msg(self.live_docs >= coll.doc_count, "dropping a collection would underflow the engine's live count"); self.live_docs -= coll.doc_count; self.dead_docs += coll.doc_count; assert_msg(self.live_bytes >= coll.live_bytes, "dropping a collection would underflow the engine's live bytes"); self.live_bytes -= coll.live_bytes; // A drop *reclaims*, it does not deaden. The loop below hands every page // this collection owned back to the pager, so its live bytes are not // garbage waiting for a rebuild -- they are already gone. Adding them to // `dead_bytes` armed a compaction for space that had just been returned, // and a rebuild costs a full copy of every *other* collection. // // Its garbage goes the other way, for the same reason: the bytes this // collection had already lost to eviction were counted in `dead_bytes` // when they died, and those pages are being freed too. That keeps // `dead_bytes` exactly the sum of `slab_used - live_bytes` over the // collections that still exist, which is what `read_catalog` recomputes // on open and what `write_catalog` asserts. assert_msg( coll.slab_used >= coll.live_bytes, "a collection cannot hold more live bytes than it ever appended", ); const coll_dead = coll.slab_used - coll.live_bytes; assert_msg( self.dead_bytes >= coll_dead, "dropping a collection would underflow the engine's dead bytes", ); self.dead_bytes -= coll_dead; self.counter_lock.unlock(self.io); coll.id_index.deinit(self.gpa); for (coll.indexes.items) |ix| { ix.deinit(self.gpa); self.gpa.destroy(ix); } coll.indexes.deinit(self.gpa); // Give the slab's pages back. They become reusable two generations // later, so a fallback to the previous image still finds them intact. for (coll.slab_runs.items) |r| { self.pager.free_pages(r.first, r.pages) catch {}; } coll.free_runs(self.gpa); coll.slab_runs.deinit(self.gpa); self.gpa.destroy(coll); } /// Free every collection in a database along with its owned name keys. fn free_db(self: *Engine, db: *Db) void { var coll_it = db.collections.iterator(); while (coll_it.next()) |coll_entry| { self.free_collection(coll_entry.value_ptr.*); self.gpa.free(coll_entry.key_ptr.*); } db.collections.deinit(self.gpa); } /// Drop the document stored under `id_key`, freeing it and its key. /// No-op when the id is absent. This is the single chokepoint where a /// document dies, so index entries are removed here, keyed by the slab /// offset the map hands back. It used to matter that the map key was still /// alive at this point, because entries aliased it; entries carry an /// offset now, so that constraint is gone. /// /// The document itself is handed to the index: entries are located by /// regenerating them from it, which is far cheaper than scanning. /// `id_enc` is `bson.encode_key` of the document's `_id` -- the canonical /// encoding, which is what the `_id_` index is keyed on. /// Copy a new document's bytes into the collection's slab and count them as /// live at both levels. `Collection.slab_append` maintains the collection's /// own total (a rebuild appends through it too, and its copies are live by /// definition); the engine's total only moves when a document actually /// becomes live, which a rebuild's copies do not. /// Drop the unclaimed part of every promise a write to this collection took: /// the slab's and one per index. Called after the write is published, under /// the same collection lock the reservations were taken under. fn release_write_reservations(self: *Engine, coll: *Collection) void { self.pager.release_reservation(&coll.hold); self.pager.release_reservation(&coll.id_index.hold); for (coll.indexes.items) |ix| self.pager.release_reservation(&ix.hold); } /// A document becoming live: its bytes into the slab, and every engine /// counter that describes. The document count moved here from the two call /// sites so that one document costs one acquisition of `counter_lock` and /// leaves the totals agreeing at every moment a reader could look. fn publish_doc_bytes(self: *Engine, coll: *Collection, bytes: []const u8) u64 { const appended = coll.slab_append(bytes); self.counter_lock.lockUncancelable(self.io); defer self.counter_lock.unlock(self.io); self.live_bytes += bytes.len; self.live_docs += 1; // Slab the append had to write off. Charged here, under the same // collection lock the append ran under, so a checkpoint's catalog walk // never sees the collection's total moved and the engine's not. self.dead_bytes += appended.skipped; return appended.off; } fn evict_doc(self: *Engine, coll: *Collection, id_enc: []const u8) void { const off = coll.id_index.lookup_exact(id_enc) orelse return; // Resolve the bytes before any mutation; the slab is untouched by // index removal, so the slice is safe for the call. const old_bytes = coll.doc_bytes(off); coll.id_index.remove_doc(self.gpa, old_bytes, off); for (coll.indexes.items) |ix| ix.remove_doc(self.gpa, old_bytes, off); // This document's log record (and its slab bytes) just became garbage. { self.counter_lock.lockUncancelable(self.io); defer self.counter_lock.unlock(self.io); assert_msg(self.live_docs >= 1, "evicting a document would underflow the engine's live count"); self.live_docs -= 1; self.dead_docs += 1; assert_msg(self.live_bytes >= old_bytes.len, "evicting a document would underflow the engine's live bytes"); self.live_bytes -= old_bytes.len; self.dead_bytes += old_bytes.len; } assert_msg(coll.doc_count >= 1, "evicting a document would underflow the collection's count"); coll.doc_count -= 1; assert_msg(coll.live_bytes >= old_bytes.len, "evicting a document would underflow the collection's live bytes"); coll.live_bytes -= old_bytes.len; // Last, and after every index entry naming these bytes is gone. That // ordering is what makes window reclamation safe to do by counting // alone: a window only reaches `map_align` dead once every document // touching it has been through here, so nothing reachable is inside it. coll.mark_dead(off, old_bytes.len); } /// `bson.encode_key` of a stored document's `_id`, owned by the caller. fn id_enc_of(gpa: std.mem.Allocator, doc_bytes: []const u8) ![]u8 { var arena = std.heap.ArenaAllocator.init(gpa); defer arena.deinit(); const id_value = (try bson.get_at(arena.allocator(), doc_bytes, "_id")) orelse return error.MissingId; var enc: std.ArrayListUnmanaged(u8) = .empty; errdefer enc.deinit(gpa); try bson.encode_key(id_value, gpa, &enc); return enc.toOwnedSlice(gpa); } // -- commands (callers must hold the matching lock) --------------------- /// Exclusive lock: for commands that mutate the engine. pub fn lock(self: *Engine) !void { try self.rwlock.lock(self.io); } pub fn unlock(self: *Engine) void { self.rwlock.unlock(self.io); } /// Shared lock: for read-only commands (find, count, aggregate, list*). /// Multiple readers may hold it simultaneously; writers wait for them. pub fn lock_read(self: *Engine) !void { try self.rwlock.lockShared(self.io); } pub fn unlock_read(self: *Engine) void { self.rwlock.unlockShared(self.io); } // -- per-collection locking (the server's command dispatch) ------------ /// Lock the catalog for a command's duration. pub fn lock_catalog(self: *Engine, exclusive: bool) !void { if (exclusive) { try self.catalog_lock.lock(self.io); } else { try self.catalog_lock.lockShared(self.io); } } pub fn unlock_catalog(self: *Engine, exclusive: bool) void { if (exclusive) self.catalog_lock.unlock(self.io) else self.catalog_lock.unlockShared(self.io); } /// With the catalog lock held, resolve the target collection and take /// its lock. When the collection is missing and `create` is set, the /// catalog lock is upgraded to exclusive to create it (then restored to /// shared); the collection lock is acquired before the exclusive catalog /// lock is dropped, so a concurrent drop can never free it underneath. /// Returns null when the collection does not exist (and create is off). pub fn lock_collection( self: *Engine, db_name: []const u8, coll_name: []const u8, write: bool, create: bool, ) !?*Collection { var coll = self.get_collection(db_name, coll_name); if (coll == null and create) { self.catalog_lock.unlockShared(self.io); try self.catalog_lock.lock(self.io); coll = try self.get_or_create_collection(db_name, coll_name); try self.lock_one(coll.?, write); self.catalog_lock.unlock(self.io); try self.catalog_lock.lockShared(self.io); return coll; } if (coll) |c| try self.lock_one(c, write); return coll; } fn lock_one(self: *Engine, coll: *Collection, write: bool) !void { if (write) { try coll.lock.lock(self.io); } else { try coll.lock.lockShared(self.io); } } pub fn unlock_collection(self: *Engine, coll: *Collection, write: bool) void { if (write) coll.lock.unlock(self.io) else coll.lock.unlockShared(self.io); } /// Ensure this command's appends are durable. The commit leader waits /// for writers mid-append to finish, then seals and syncs once, covering /// every append in flight — followers that arrived during the leader's /// commit find their records already covered and return without a sync /// of their own. Every acknowledged write is fsynced before its reply, /// so the crash guarantees are unchanged. pub fn commit(self: *Engine) !void { try self.commit_lock.lock(self.io); defer self.commit_lock.unlock(self.io); // A commit can never have sealed more than was ever appended. assert_msg(self.committed_seq <= self.seq, "commit claims to have sealed more than was appended"); // Everything this command appended is at or below the current seq. // Read it before waiting, so a leader that sealed before this // command's appends cannot be mistaken for one that covered them. try self.log_lock.lock(self.io); const want = self.seq; self.log_lock.unlock(self.io); // A commit is in flight; wait for it, then check whether the // leader's seal covered this writer's append. // // The error propagates rather than being swallowed: this function // returning success is what tells the caller its write is on disk, so // reporting success after a failed wait acknowledges a write that was // never synced. A canceled connection has no reason to wait out // another writer's commit, so cancelable is right here. while (self.committing) try self.commit_done.wait(self.io, &self.commit_lock); if (self.committed_seq >= want) { return; // a concurrent commit already synced this writer's records } // Become the leader: the flag is set before the wait below, so any // commit that arrives during it waits as a follower. self.committing = true; var done = false; defer { if (!done) { self.committing = false; // Broadcast: followers sleeping on `committing` all need to // re-check it, not just one of them. self.commit_done.broadcast(self.io); } } // Wait for writers mid-append to finish so the seal covers them. // // Uncancelable: `committing` is set, so every other writer is now // parked behind this leader. Abandoning the commit here would strand // them for a full extra round trip, and the drain is bounded anyway -- // an in-flight append only holds log_lock long enough to buffer its // record. Finishing is strictly better than bailing out. while (self.pending_appends.load(.acquire) > 0) { self.commit_done.waitUncancelable(self.io, &self.commit_lock); } // The drain is what makes the seal below cover every append in flight. // Not asserted as pending_appends == 0 here: a new append can start // at any moment (it increments without commit_lock), so a fresh // writer can be in flight between the drain's last check and this // point. The seal still covers every append that wrote bytes before // the sync — appends serialize with it on log_lock — and any append // that starts after it is sealed by its own commit. try self.log_lock.lock(self.io); defer self.log_lock.unlock(self.io); try self.log.sync(); // Appends drained above, so the seal covered every record written so // far -- including any that arrived while this leader waited. self.committed_seq = self.seq; // This writer's own records are now durable: the postcondition the // caller relies on before it acknowledges the write. Paired with the // same check in the dispatch epilogue (see commands.zig). assert_msg(self.committed_seq >= want, "commit returning success without sealing this writer's records"); done = true; self.committing = false; // Broadcast: every follower waiting on `committing` must wake to see // it cleared — a single signal would wake only one and strand the // rest. self.commit_done.broadcast(self.io); } /// Log an append (and its seq increment) under the log lock, marking /// the append as in flight so a commit leader's seal covers it. fn log_append( self: *Engine, comptime kind: LogKind, db: []const u8, coll: []const u8, doc: []const u8, ) !void { _ = self.pending_appends.fetchAdd(1, .acq_rel); defer { // The increment above pairs with this decrement on every return // path, so the count can never be zero here. assert_msg(self.pending_appends.load(.acquire) > 0, "log_append decrementing an already-zero in-flight count"); _ = self.pending_appends.fetchSub(1, .acq_rel); // Wake a commit leader waiting for in-flight appends. The // signal must be delivered while holding commit_lock: a leader // between its pending_appends check and its wait() still holds // the lock, so a signal here can never land in that window and // be lost (which froze every writer once a few connections // committed concurrently). The leader's wait() releases the // lock, so this lock only blocks until it starts waiting. // Broadcast rather than signal: if a follower waiting on // `committing` snatches the single wakeup, the leader would // sleep forever even with pending_appends back at zero. // // lockUncancelable, not lock: this is a cleanup path, and the // cancelable variant can fail. Swallowing that failure and // unlocking anyway would release a mutex we never took, which // Mutex.unlock treats as `unreachable` -- a panic in ReleaseSafe // and silent memory corruption in the default ReleaseFast build. self.commit_lock.lockUncancelable(self.io); self.commit_done.broadcast(self.io); self.commit_lock.unlock(self.io); } try self.log_lock.lock(self.io); defer self.log_lock.unlock(self.io); self.seq += 1; // Seqs start at 1 and only ever increase; 0 means "nothing appended", // which is what committed_seq is compared against. assert_msg(self.seq > 0, "log_append produced a zero seq"); switch (kind) { .upsert => try self.log.append_upsert(db, coll, doc, self.seq), .delete => try self.log.append_delete(db, coll, doc, self.seq), .index_create => try self.log.append_index_create(db, coll, doc, self.seq), .index_drop => try self.log.append_index_drop(db, coll, doc, self.seq), } } /// Insert a document. Fails with error.DuplicateKey if the _id exists. /// Generates an ObjectId _id when absent. pub fn insert( self: *Engine, db_name: []const u8, coll_name: []const u8, doc: *const bson.Document, oid_gen: *bson.ObjectIdGen, ) !void { _ = try self.upsert(db_name, coll_name, doc, oid_gen, .insert); } /// Whether a write changed anything. A replace whose result is byte-identical /// to what is stored is not an error and not a write: MongoDB reports it as /// matched but not modified, and writes no oplog entry for it. pub const Written = enum { modified, unchanged }; /// Insert or replace a document by _id (upsert without existence check). /// Returns `.unchanged` when the stored document already had these exact /// bytes -- see `Written`. pub fn replace( self: *Engine, db_name: []const u8, coll_name: []const u8, doc: *const bson.Document, oid_gen: *bson.ObjectIdGen, ) !Written { return self.upsert(db_name, coll_name, doc, oid_gen, .replace); } /// One document's built entries for one index, tracked so a failure /// anywhere before the log append frees them all. const Built = struct { built: index.BuiltEntries, ix: *index.Index, }; /// Shared body of `insert` and `replace`: they differ only in how an /// existing _id is treated. Logs (and syncs) the new document before it /// becomes visible in memory. fn upsert( self: *Engine, db_name: []const u8, coll_name: []const u8, doc: *const bson.Document, oid_gen: *bson.ObjectIdGen, mode: enum { insert, replace }, ) !Written { const coll = try self.get_or_create_collection(db_name, coll_name); const doc_bytes = try self.serialize_with_id(doc, oid_gen); defer self.gpa.free(doc_bytes); // A document _id materializes a spine; free it right after the key // is serialized. var id_arena = std.heap.ArenaAllocator.init(self.gpa); defer id_arena.deinit(); const id_value = (try bson.get_at(id_arena.allocator(), doc_bytes, "_id")) orelse unreachable; // The canonical encoding, because that is what `_id_` is keyed on. It // used to be serialize_value, for a hashmap that no longer exists. var id_enc_list: std.ArrayListUnmanaged(u8) = .empty; defer id_enc_list.deinit(self.gpa); try bson.encode_key(id_value, self.gpa, &id_enc_list); const id_enc = id_enc_list.items; coll.dup_index = null; // 1. Build entries for every index. ParallelArrays escapes here, // before anything is logged or mutated. var built_list: std.ArrayListUnmanaged(Built) = .empty; defer { for (built_list.items) |*b| b.built.deinit(self.gpa); built_list.deinit(self.gpa); } { // The implicit _id_ index, through the same protocol: reserved // before the log append, inserted infallibly after it. Built // *first* so it is checked first below -- MongoDB reports _id_ // when a write violates both it and a unique secondary. var built = try coll.id_index.build_entries(self.gpa, doc_bytes); built_list.append(self.gpa, .{ .built = built, .ix = &coll.id_index }) catch |err| { built.deinit(self.gpa); return err; }; } for (coll.indexes.items) |ix| { var built = try ix.build_entries(self.gpa, doc_bytes); built_list.append(self.gpa, .{ .built = built, .ix = ix }) catch |err| { built.deinit(self.gpa); return err; }; } // A replace that would store the same bytes is not a write at all. It // has to be decided here -- after the document is serialized, so the // comparison is against what would actually be stored, and before the // log append, so a no-op costs no log record, no fsync, no slab bytes // and no garbage. `nModified` is the visible half of this: MongoDB // counts a document as modified only if the update altered it, so // `$set: {x: 11}` on a document already holding `x: 11` is matched and // not modified. if (mode == .replace) { if (coll.id_index.lookup_exact(id_enc)) |old_off| { if (std.mem.eql(u8, coll.doc_bytes(old_off), doc_bytes)) return .unchanged; } } // 2. Unique-index checks, _id_ included; a rejected write never // reaches the log. `_id` uniqueness used to be a `docs.contains` // probe here, which the docs map will not be around to answer // (PLAN amendment A3) -- and the tree answers it better, since it // is keyed on the canonical encode_key rather than serialize_value // (A4). Exclude-self is null for an insert: the document has no // entries yet, and passing its offset would hide precisely the // same-_id collision this must catch. For a replace it is the // document's *current* slab offset, since that is what its existing // entries carry -- the new offset does not exist yet. const exclude: ?u64 = if (mode == .replace) coll.id_index.lookup_exact(id_enc) else null; for (built_list.items) |*b| { if (!b.ix.unique) continue; b.ix.check_unique(b.built.entries.items, exclude) catch { // The implicit index keeps its own error identity, so // commands.zig renders E11000 with index "_id_" exactly as // before and needs no change; `dup_index` stays null, which is // what that rendering treats as "the _id_ index". if (b.ix == &coll.id_index) return error.DuplicateKey; coll.dup_index = b.ix.name; return error.DuplicateKeyIndex; }; } // 4. Reserve everything the publish step needs -- tree capacity and // slab room -- as the last fallible work, so nothing after the log // append can fail. The slab reservation used to be absent because // appending to an in-memory ArrayList was the only failure mode; a // file-backed slab can also fail on growth, and failing *after* the // record is durable would report an error for a write the next open // would produce anyway. for (built_list.items) |*b| { try b.ix.reserve_for(self.gpa, b.built.entries.items); } self.count_slab_skip(try coll.slab_reserve(self.gpa, doc_bytes.len)); // 5. Log (and sync) before anything becomes visible. The append // takes the log lock; durability (fsync) is the command's commit. try self.log_append(.upsert, db_name, coll_name, doc_bytes); // 6. Replace drops the old document (and its index entries). if (mode == .replace) self.evict_doc(coll, id_enc); // 7. Publish the document and its entries: copy the bytes into the // slab and record the offset. Infallible from here. const off = self.publish_doc_bytes(coll, doc_bytes); coll.doc_count += 1; for (built_list.items) |*b| { if (b.built.multikey) b.ix.multikey = true; b.ix.insert_entries(&b.built, off); } // The write is published; anything the reservations above did not claim // is dead. Leaving it promised would grow the file on every write. Every // consumer this upsert reserved through, and only those: another // collection may be mid-write on another thread. self.release_write_reservations(coll); self.note_compact(); self.note_checkpoint(); return .modified; } /// Remove a document by its `_id` value. Returns true if it existed. /// The serialized-key encoding stays private to the engine. pub fn remove_by_id( self: *Engine, db_name: []const u8, coll_name: []const u8, id: bson.Value, ) !bool { var enc: std.ArrayListUnmanaged(u8) = .empty; defer enc.deinit(self.gpa); try bson.encode_key(id, self.gpa, &enc); return self.remove(db_name, coll_name, enc.items); } /// `id_enc` is `bson.encode_key` of the `_id`. fn remove(self: *Engine, db_name: []const u8, coll_name: []const u8, id_enc: []const u8) !bool { const db = self.dbs.get(db_name) orelse return false; const coll = db.collections.get(coll_name) orelse return false; const off = coll.id_index.lookup_exact(id_enc) orelse return false; // Log (and sync) the delete before removing it from memory, so the // log always describes at least as much as the in-memory state. // Replay only reads _id out of a delete record, so log just that // rather than a copy of the whole document. const id_bytes = coll.doc_bytes(off); var id_arena = std.heap.ArenaAllocator.init(self.gpa); defer id_arena.deinit(); const id_pairs = [_]bson.Pair{.{ .key = "_id", .value = (try bson.get_at(id_arena.allocator(), id_bytes, "_id")) orelse unreachable }}; var id_doc: std.ArrayListUnmanaged(u8) = .empty; defer id_doc.deinit(self.gpa); try bson.write_doc(&id_pairs, self.gpa, &id_doc); try self.log_append(.delete, db_name, coll_name, id_doc.items); self.evict_doc(coll, id_enc); // Deletes grow the log too. Without this a delete-heavy workload // never compacts, because only upsert and ttl_sweep used to check. self.note_compact(); return true; } pub fn get_collection(self: *Engine, db_name: []const u8, coll_name: []const u8) ?*Collection { const db = self.dbs.get(db_name) orelse return null; return db.collections.get(coll_name); } pub fn get_doc( self: *Engine, db_name: []const u8, coll_name: []const u8, /// `bson.encode_key` of the `_id`. id_enc: []const u8, ) ?[]const u8 { const coll = self.get_collection(db_name, coll_name) orelse return null; const off = coll.id_index.lookup_exact(id_enc) orelse return null; return coll.doc_bytes(off); } pub fn drop_collection(self: *Engine, db_name: []const u8, coll_name: []const u8) !bool { const db = self.dbs.getPtr(db_name) orelse return false; const removed = db.collections.fetchRemove(coll_name) orelse return false; self.free_collection(removed.value); self.gpa.free(removed.key); // A cursor on this namespace is already safe -- it holds names, so its // next getMore finds nothing to lock -- but reaping here frees the slots // now instead of at the idle timeout, and keeps the open-cursor metric // describing cursors that can still return something. _ = self.cursors.kill_namespace(self.io, db_name, coll_name); return true; } pub fn drop_database(self: *Engine, db_name: []const u8) !bool { var removed = self.dbs.fetchRemove(db_name) orelse return false; self.free_db(&removed.value); self.gpa.free(removed.key); _ = self.cursors.kill_namespace(self.io, db_name, null); return true; } /// Build and register a secondary index from a spec document /// ({key, name, unique?, sparse?}). The create record is written only /// after the index builds over the existing documents and passes /// uniqueness, so a rejected create persists nothing. Returns the new /// index (or the existing one when the spec matches — idempotent). pub fn create_index( self: *Engine, db_name: []const u8, coll_name: []const u8, spec_doc: *const bson.Document, ) !*index.Index { const coll = try self.get_or_create_collection(db_name, coll_name); const parsed = try index.parse_spec(self.gpa, self.pager, spec_doc); // Boxed before anything is built into it, so publishing is a pointer // append rather than a struct copy. An Index will own a mapping once // the arena is file-backed, and copying one then would duplicate that // ownership. const ix = self.gpa.create(index.Index) catch |err| { var dead = parsed; dead.deinit(self.gpa); return err; }; ix.* = parsed; var committed = false; // Runs on every return path (including the idempotent no-op): the // parsed spec is only owned by the collection once committed. defer if (!committed) { ix.deinit(self.gpa); self.gpa.destroy(ix); }; if (coll.find_index(ix.name)) |existing| { if (index.Index.spec_equal(existing, ix)) return existing; return error.IndexOptionsConflict; } // Build entries over the existing documents (the index is not // exposed until the end, so mutating it is safe). Entries are // appended unsorted and ordered once at the end — inserting each // document into a sorted array memmoves the tail every time, which // is what made this quadratic. On any failure the deferred // ix.deinit frees every appended key. Nothing is persisted. var doc_it = coll.id_index.iter(); while (doc_it.next()) |entry| { try ix.append_doc_entries(self.gpa, coll.doc_bytes(entry.off), entry.off); } _ = try ix.finish_bulk(self.gpa, true); self.pager.release_reservation(&ix.hold); // Reserve the collection slot, then persist and publish. try coll.indexes.ensureUnusedCapacity(self.gpa, 1); var spec_bytes: std.ArrayListUnmanaged(u8) = .empty; defer spec_bytes.deinit(self.gpa); try ix.write_spec(self.gpa, &spec_bytes); try self.log_append(.index_create, db_name, coll_name, spec_bytes.items); coll.indexes.appendAssumeCapacity(ix); committed = true; return ix; } /// Remove a secondary index by name, persisting a drop record first. /// Returns false when no such index exists. pub fn drop_index( self: *Engine, db_name: []const u8, coll_name: []const u8, index_name: []const u8, ) !bool { const db = self.dbs.get(db_name) orelse return false; const coll = db.collections.get(coll_name) orelse return false; if (coll.find_index(index_name) == null) return false; const name_pairs = [_]bson.Pair{.{ .key = "name", .value = .{ .string = index_name } }}; var name_doc: std.ArrayListUnmanaged(u8) = .empty; defer name_doc.deinit(self.gpa); try bson.write_doc(&name_pairs, self.gpa, &name_doc); try self.log_append(.index_drop, db_name, coll_name, name_doc.items); _ = coll.remove_index(self.gpa, index_name); return true; } /// Delete every document expired as of `now_ms` (Unix milliseconds) /// under some TTL index, and return how many were deleted. Callers must /// hold the write lock; the server's monitor coroutine (src/server.zig) /// is the only caller in production, tests call it with a fixed clock. /// /// Each expiry goes through `remove`, so it is logged and fsynced like /// any other delete and survives a restart. Expiry is therefore coarse /// by design (as in MongoDB): an expired document stays visible until /// the next sweep. pub fn ttl_sweep(self: *Engine, now_ms: i64) !usize { var deleted: usize = 0; // Catalog lock for the whole sweep (the collection pointers stay // valid); each collection is swept under its own write lock, one at // a time, never two at once. try self.catalog_lock.lockShared(self.io); defer self.catalog_lock.unlockShared(self.io); var db_it = self.dbs.iterator(); while (db_it.next()) |db_entry| { var coll_it = db_entry.value_ptr.collections.iterator(); while (coll_it.next()) |coll_entry| { const coll = coll_entry.value_ptr.*; deleted += try self.ttl_sweep_coll(coll, now_ms, db_entry.key_ptr.*, coll_entry.key_ptr.*); } } // A TTL-only workload never reaches the threshold check in `upsert`, // so the log would otherwise grow without bound. if (deleted > 0) self.note_compact(); return deleted; } /// Sweep one collection under its write lock; the lock is released on /// every return path. Returns how many documents were removed. fn ttl_sweep_coll( self: *Engine, coll: *Collection, now_ms: i64, db_name: []const u8, coll_name: []const u8, ) !usize { try coll.lock.lock(self.io); defer coll.lock.unlock(self.io); // Offsets, collected before any removal. They are values, so unlike // the id slices this used to dupe -- which aliased a docs-map key that // `remove` would free out from under the rest of the batch -- there is // nothing to own here. Collect-then-remove still matters, because the // iterator below aliases tree pages that removal reshapes. var offs: std.ArrayListUnmanaged(u64) = .empty; defer offs.deinit(self.gpa); for (coll.indexes.items) |ix| { const ttl = ix.ttl orelse continue; const cutoff: i128 = @as(i128, now_ms) - @as(i128, ttl) * 1000; // bson compare order ranks datetime above null, numbers // and strings and below only timestamp and maxKey, so // datetimes form a contiguous band in the encoded key // order: seek the minimum datetime and stop when the // leading type changes or the cutoff is passed. const min_dt = [_]u8{ bson.encoded_datetime_tag, 0, 0, 0, 0, 0, 0, 0, 0 }; var it = ix.seek(&min_dt); while (it.next()) |e| { const ms = bson.encoded_leading_datetime(e.key) orelse break; if (@as(i128, ms) > cutoff) break; try offs.append(self.gpa, e.off); } } if (offs.items.len == 0) return 0; // One document can be expired by several entries (an array // of dates) or by several TTL indexes. std.mem.sort(u64, offs.items, {}, std.sort.asc(u64)); var w: usize = 1; for (offs.items[1..]) |off| { if (off != offs.items[w - 1]) { offs.items[w] = off; w += 1; } } offs.items.len = w; // `remove` works by _id, so recover each one from the document its // offset names. get_at materializes a spine, hence the arena; the slab // is untouched by the removals, so the bytes stay valid throughout. var arena = std.heap.ArenaAllocator.init(self.gpa); defer arena.deinit(); var removed: usize = 0; for (offs.items) |off| { const id_value = (try bson.get_at(arena.allocator(), coll.doc_bytes(off), "_id")) orelse continue; var enc: std.ArrayListUnmanaged(u8) = .empty; try bson.encode_key(id_value, arena.allocator(), &enc); if (try self.remove(db_name, coll_name, enc.items)) removed += 1; } return removed; } pub fn database_names(self: *Engine, out: *std.ArrayListUnmanaged([]const u8)) !void { var it = self.dbs.iterator(); while (it.next()) |entry| try out.append(self.gpa, entry.key_ptr.*); } pub fn collection_names( self: *Engine, db_name: []const u8, out: *std.ArrayListUnmanaged([]const u8), ) !void { const db = self.dbs.get(db_name) orelse return; var it = db.collections.iterator(); while (it.next()) |entry| try out.append(self.gpa, entry.key_ptr.*); } // -- internals ----------------------------------------------------------- pub fn get_or_create_collection( self: *Engine, db_name: []const u8, coll_name: []const u8, ) !*Collection { const db = self.dbs.getPtr(db_name) orelse { const db_key = try self.gpa.dupe(u8, db_name); errdefer self.gpa.free(db_key); try self.dbs.put(self.gpa, db_key, .{ .collections = .empty }); return self.get_or_create_collection(db_name, coll_name); }; if (db.collections.get(coll_name)) |coll| return coll; const coll_key = try self.gpa.dupe(u8, coll_name); errdefer self.gpa.free(coll_key); const new_coll = try self.gpa.create(Collection); errdefer self.gpa.destroy(new_coll); self.layout_epoch_seq += 1; new_coll.* = try Collection.init(self.gpa, self.pager, self.layout_epoch_seq); errdefer new_coll.id_index.deinit(self.gpa); try db.collections.put(self.gpa, coll_key, new_coll); return new_coll; } /// Deep-copy a document into engine-owned storage, prepending a /// generated ObjectId `_id` when absent. /// The canonical bytes of `doc`, with an ObjectId `_id` generated when /// absent. The result is owned by the caller. fn serialize_with_id( self: *Engine, doc: *const bson.Document, oid_gen: *bson.ObjectIdGen, ) ![]u8 { // `_id` first, always -- generated here, or moved if the client put it // later. MongoDB stores it first whatever order it arrives in, and the // Node driver arrives in the other order: it fills a missing `_id` by // assigning the property, which in JavaScript appends it, so an // `insertOne({name, age})` reaches us as `{name, age, _id}`. // // Two things depend on this beyond field order in results. A replacement // keeps `_id` at the front, so storing it elsewhere made replacing a // document with itself a byte-level change and therefore a write. And // the position is part of the stored bytes, so it has to be settled once, // here, rather than by every reader. if (doc.pairs.len > 0 and std.mem.eql(u8, doc.pairs[0].key, "_id")) { return serialize_doc(self.gpa, doc); } var pairs: std.ArrayListUnmanaged(bson.Pair) = .empty; defer pairs.deinit(self.gpa); if (doc.get("_id")) |id| { try pairs.append(self.gpa, .{ .key = "_id", .value = id }); for (doc.pairs) |p| { if (std.mem.eql(u8, p.key, "_id")) continue; try pairs.append(self.gpa, p); } } else { const oid = oid_gen.new(self.io); try pairs.append(self.gpa, .{ .key = "_id", .value = .{ .object_id = oid } }); try pairs.appendSlice(self.gpa, doc.pairs); } var out: std.ArrayListUnmanaged(u8) = .empty; defer out.deinit(self.gpa); try bson.write_doc(pairs.items, self.gpa, &out); return out.toOwnedSlice(self.gpa); } /// Keep the log file at roughly 1.5x the live data, rather than /// compacting every fixed number of appended bytes. /// /// A fixed byte trigger makes total rewrite traffic quadratic: a 1 GB /// dataset with a 16 MiB threshold compacts ~64 times, rewriting 1 GB /// each time. Triggering on file size relative to the live size makes /// successive compactions geometric, so the total bytes rewritten over /// the life of the log is O(n) rather than O(n²) — and it bounds the /// disk footprint directly, which is what the threshold is really for. /// /// The other half of the problem is the opposite workload: a pure bulk /// insert has no garbage at all, so every compaction rewrites a /// perfectly compact file for nothing. `compact` reports how much it /// reclaimed; when that is little, we back the baseline off /// multiplicatively so a garbage-free log is left alone. /// Called at the end of a write command's in-lock section: when the /// garbage share crosses the threshold, record that a compaction is /// wanted. It runs in the command epilogue, after the collection lock is /// released — never inline, since compact takes the collection locks /// itself and would deadlock against the caller's. /// Arm a checkpoint when the log has grown past the threshold. Cheap enough /// to call on every write: one relaxed load and a compare. fn note_checkpoint(self: *Engine) void { if (self.log.log_bytes < self.checkpoint_threshold) return; self.checkpoint_pending.store(true, .release); } /// Claim a pending checkpoint, if there is one. pub fn take_checkpoint(self: *Engine) bool { return self.checkpoint_pending.swap(false, .acq_rel); } fn note_compact(self: *Engine) void { // Garbage is measured in the *data file*, not the log. This used to read // `log.data_bytes`, which was the right question while the log was the // only copy of the data -- but a checkpoint now truncates the log and // `truncate_to_header` resets that counter, so the first gate stopped // being reachable and compaction silently never fired again. The churn // gate caught it: 50% churn over three rounds left the data file at 4.1x // the live data, with a trigger that had been dead since the log started // being reclaimed. // // One snapshot, because the second gate is a ratio: reading the two // totals separately compares a live figure from one moment against a // dead figure from another, and under concurrent writers that is how a // rebuild fires on a database that does not want one. const c = self.counters(); // Absolute volume first: a rewrite costs a full copy of the live data, // so it is not worth doing for a few kilobytes however bad the ratio. if (c.dead_bytes < self.compact_threshold) return; // Then the share, dead / (live + dead), firing at ~20%: the file stays // near 1.25x the live data and each rebuild is paid for by the space it // reclaims. Bytes rather than document counts, because a rewrite copies // bytes -- 100k evicted 40 B documents are not worth the same rebuild as // 100k evicted 16 KiB ones. if (c.dead_bytes * 4 < c.live_bytes) return; self.compact_pending.store(true, .release); } /// Whether a compaction is wanted; clears the flag atomically so only one /// of several concurrent writers takes the request. `compact` excludes /// itself besides, so a caller that wins here still yields to a rewrite /// already in progress. pub fn take_compact(self: *Engine) bool { return self.compact_pending.swap(false, .acq_rel); } /// Re-arm the compaction request. For a caller that took the request but /// could not carry it out (a failed or abandoned rewrite), so the garbage /// is reconsidered by a later, quieter epilogue instead of being forgotten. pub fn request_compact(self: *Engine) void { self.compact_pending.store(true, .release); } /// Rewrite the log with only live documents, atomically swapping the /// file. Runs outside every collection lock (called from a write /// command's epilogue). The snapshot takes the collection locks one at /// a time without the log lock — so a concurrent writer can always /// finish its append — then takes the log lock and retries until no /// writer appended during the snapshot (detected via the record seq), /// which makes the snapshot consistent with what the log contains. /// /// Only one compaction runs at a time; a second caller returns immediately. /// One document's new home during a rebuild: its docs-map key and the offset /// it was copied to. Named rather than anonymous so the rebuild and the /// repack agree on the type. const Moved = struct { off: u64 }; /// Reclaim what a checkpoint cannot: dead document bytes and abandoned /// pages. /// /// A checkpoint publishes the structures where they already are. It cannot /// move a document, because every index leaf holds that document's physical /// offset -- so reclaiming a replaced document's bytes means rewriting the /// documents *and* every index that points at them, together, which is what /// this does. /// /// It used to rewrite the *log* instead: re-emit every live document into a /// fresh log and rename it over the old one. That is now the wrong shape /// twice over. The log is no longer where the data lives, and a re-emitted /// record carries a sequence that a later watermark can cover, which would /// make the next open skip it (PLAN section 4). The log is simply truncated /// by the checkpoint at the end. pub fn compact(self: *Engine) !void { // Claim it, or leave it to the one already running. A caller that loses // this race has nothing to do: the winner's rebuild covers its garbage. if (self.compacting.swap(true, .acq_rel)) return; defer self.compacting.store(false, .release); // Reclamation first, because it is the cheap half of the same job: a // checkpoint hands back whole windows for the cost of one publish, // where a rebuild copies every live byte in the database. Whatever it // takes, the per-collection gate below no longer sees, so a collection // whose garbage was all in empty windows is not rewritten at all. // // This is not a refinement, it is what makes reclamation reachable // under a delete-heavy workload. A checkpoint is otherwise armed by log // volume, and a delete logs only an `_id` -- so deleting half a 190 MB // collection moves the log by a couple of megabytes and no checkpoint // runs, while the garbage sails past the rebuild threshold. Measured // with the churn harness: six rounds, six rebuilds, 1 MB reclaimed. try self.checkpoint(); try self.catalog_lock.lockShared(self.io); var rebuild_err: ?anyerror = null; var db_it = self.dbs.iterator(); outer: while (db_it.next()) |db_entry| { var coll_it = db_entry.value_ptr.collections.iterator(); while (coll_it.next()) |coll_entry| { self.rebuild_collection(coll_entry.value_ptr.*) catch |err| { rebuild_err = err; break :outer; }; } } // Before releasing the catalog, and whether the walk finished or gave up // part way: a rebuild resets a collection's totals, so an incremental // `dead_bytes` now describes collections that no longer exist in that // shape. It used to be zeroed here, which was true only if a repack // leaves nothing behind -- it does not. A checkpoint landing mid-rebuild // forces the copy's own append cursor up to a system page, and those // skipped bytes are as dead as the ones being reclaimed. const dead_after = self.sum_dead_bytes(); self.catalog_lock.unlockShared(self.io); self.counter_lock.lockUncancelable(self.io); self.dead_bytes = dead_after; self.dead_docs = 0; self.counter_lock.unlock(self.io); if (rebuild_err) |err| return err; // Publish the rebuilt layout, which is also what reclaims the log. Until // this lands the old pages are still referenced by the previous // watermark, so a crash mid-rebuild simply loses the rebuild. try self.checkpoint(); // And again, to walk the pages the rebuild just abandoned the rest of the // way down the free list: one publish moves them from `pending` to // `hold`, a second from `hold` to `ready`. Without this the space a // rebuild reclaims is not reusable until two unrelated checkpoints have // happened, so the next rebuild grows the file instead of reusing it -- // measured as ~1.2x of extra steady-state size under sustained churn. // // Safe for the same reason the delay exists: what the second publish // releases is the pages the *pre-rebuild* image referenced, and that // image is no longer the fallback -- the first publish made the rebuilt // one current and the one before it the fallback. Both remain intact. try self.checkpoint(); } /// Whether rewriting this collection would pay for itself. The caller holds /// its lock. /// /// A rebuild copies a collection's live bytes to reclaim its dead ones, so /// the one thing it must not do is copy a collection that has none. It used /// to: `compact` walked every collection unconditionally, so garbage in one /// paid for a full copy of the other thirty-nine. /// /// The share is the same one `note_compact` applies to the engine's totals, /// and that is what keeps the two from disagreeing. If no collection passes /// this test then `dead_i < live_i / 4` for every one of them, so /// `sum(dead) < sum(live) / 4` and the engine's trigger could not have fired /// either. So a compaction that runs always rebuilds at least one /// collection, and cannot spin re-arming itself over garbage no rebuild will /// take. An absolute floor per collection would break exactly that: forty /// collections each under the floor can sum to well over it. fn wants_rebuild(coll: *const Collection) bool { assert_msg( coll.slab_used >= coll.live_bytes, "a collection cannot hold more live bytes than it ever appended", ); const dead = coll.slab_used - coll.live_bytes; if (dead == 0) return false; return dead * 4 >= coll.live_bytes; } /// Copy one collection's live documents into fresh extents and rebuild every /// index against the new offsets. /// /// Documents and indexes have to move together: an index leaf holds a /// physical offset, so a document that moves without its indexes being /// rebuilt is a stale entry pointing at whatever now occupies those bytes. /// fn rebuild_collection(self: *Engine, coll: *Collection) !void { try coll.lock.lock(self.io); defer coll.lock.unlock(self.io); if (!wants_rebuild(coll)) return; var old_extents = try self.gpa.alloc(pgr.Extent, coll.slab_runs.items.len); defer self.gpa.free(old_extents); for (coll.slab_runs.items, 0..) |r, i| old_extents[i] = .{ .first = r.first, .pages = r.pages }; // Fresh slab. The old extents stay allocated until the free list // releases them, two generations on. coll.free_runs(self.gpa); coll.slab_tail = 0; coll.slab_end = 0; coll.slab_used = 0; coll.live_bytes = 0; // The rebuild is the one place the located and unlocated halves are // both reset: every byte it copies is live, so a fresh slab has no // garbage to place. Anything the copy skips is marked as it happens. coll.dead_unlocated = 0; // Walk in _id order, which is also the order the new slab ends up in -- // so a later scan reads it sequentially. var moved: std.ArrayListUnmanaged(Moved) = .empty; defer moved.deinit(self.gpa); try moved.ensureTotalCapacity(self.gpa, @intCast(coll.doc_count)); // The `_id_` tree is the enumeration, and it is in key order -- so the // new slab ends up ordered and a later scan reads it sequentially. var it = coll.id_index.iter(); while (it.next()) |entry| { const bytes = doc_bytes_in(self.pager, entry.off); // The skips are dropped rather than charged: this collection's // totals were reset above and `compact` recomputes the engine's from // what the rebuild leaves behind. _ = try coll.slab_reserve(self.gpa, bytes.len); const appended = coll.slab_append(bytes); self.pager.release_reservation(&coll.hold); try moved.append(self.gpa, .{ .off = appended.off }); } // Republish the offsets. // Rebuild every index from the new offsets, bulk-packed. try self.repack_index(coll, &coll.id_index, moved.items); for (coll.indexes.items) |ix| try self.repack_index(coll, ix, moved.items); for (old_extents) |e| try self.pager.free_pages(e.first, e.pages); // Every document has moved, so every offset an open cursor is holding // now names different bytes. Bumped last, after the rebuild can no // longer fail: a cursor invalidated by a rebuild that then errored out // would have been invalidated for nothing. self.layout_epoch_seq += 1; coll.layout_epoch = self.layout_epoch_seq; self.counter_lock.lockUncancelable(self.io); self.compactions += 1; self.counter_lock.unlock(self.io); } /// What the slab is doing, for `serverStatus`. Collections under the same /// catalog-then-collection order everything else uses. /// /// It exists because the milestone's own gate cannot be read without it. A /// steady-state size ratio can look healthy while reclamation does nothing /// -- the file grows, a rebuild periodically halves it, and the average /// comes out fine. `reclaimed_bytes` climbing while `alloc_tail` stays put /// is the shape that says the free list is load-bearing; either one alone /// says very little. pub const SlabStats = struct { live_bytes: u64 = 0, dead_bytes: u64 = 0, slab_bytes: u64 = 0, reclaimed_bytes: u64 = 0, slab_runs: u64 = 0, free_ready_pages: u32 = 0, alloc_tail: u32 = 0, compactions: u64 = 0, }; pub fn slab_stats(self: *Engine) SlabStats { var out: SlabStats = .{}; self.catalog_lock.lockSharedUncancelable(self.io); var db_it = self.dbs.iterator(); while (db_it.next()) |db_entry| { var coll_it = db_entry.value_ptr.collections.iterator(); while (coll_it.next()) |ce| { const coll = ce.value_ptr.*; coll.lock.lockSharedUncancelable(self.io); defer coll.lock.unlockShared(self.io); out.live_bytes += coll.live_bytes; out.slab_bytes += coll.slab_used; out.reclaimed_bytes += coll.reclaimed_bytes; out.slab_runs += coll.slab_runs.items.len; } } self.catalog_lock.unlockShared(self.io); // From the collections rather than the engine's running total, so this // is the same figure `write_catalog` asserts against rather than a // second opinion about it. assert_msg(out.slab_bytes >= out.live_bytes, "the slab cannot hold more live bytes than it has"); out.dead_bytes = out.slab_bytes - out.live_bytes; out.free_ready_pages = self.pager.free_ready_pages(); out.alloc_tail = self.pager.alloc_tail; self.counter_lock.lockUncancelable(self.io); out.compactions = self.compactions; self.counter_lock.unlock(self.io); return out; } /// The engine's dead-byte total, recomputed from the collections that /// exist. Everywhere else the counter is incremental; a rebuild is the one /// place that has to reset it, and the answer after a rebuild is not zero -- /// the copying skips slab of its own whenever a checkpoint lands mid-walk. /// /// Each collection is read under its own lock, catalog then collection: the /// order `compact` and `write_catalog` both use. Uncancelable, because the /// caller has already rewritten the collections and the counter describing /// them cannot be left behind. fn sum_dead_bytes(self: *Engine) u64 { var sum: u64 = 0; var db_it = self.dbs.iterator(); while (db_it.next()) |db_entry| { var coll_it = db_entry.value_ptr.collections.iterator(); while (coll_it.next()) |ce| { const coll = ce.value_ptr.*; coll.lock.lockSharedUncancelable(self.io); defer coll.lock.unlockShared(self.io); assert_msg( coll.slab_used >= coll.live_bytes, "a collection cannot hold more live bytes than it ever appended", ); sum += coll.slab_used - coll.live_bytes; } } return sum; } fn repack_index( self: *Engine, coll: *Collection, ix: *index.Index, moved: []const Moved, ) !void { _ = coll; ix.reset_tree(self.gpa) catch |err| return err; for (moved) |m| { ix.append_doc_entries(self.gpa, doc_bytes_in(self.pager, m.off), m.off) catch |err| switch (err) { error.ParallelArrays => continue, else => return err, }; } // Duplicates are tolerated here for the same reason they are on open: // refusing would make a maintenance task able to take the database down. _ = ix.finish_bulk(self.gpa, false) catch |err| return err; self.pager.release_reservation(&ix.hold); } /// Re-emit one collection's index specs and documents into the compacted /// log, under the collection's write lock (released on every return /// path, including errors). /// Rebuild every empty index from the live documents. Runs after replay /// completes, so it is order-independent: a create record, the documents /// it indexes, and any drop record all replay first. A duplicate under a /// unique index logs a loud warning and keeps the index (still correct /// as a candidate generator; future writes are still enforced) — the /// database always opens, leaving dropIndexes as an in-band recovery /// path. /// Every document is reachable through every index that is supposed to cover /// it, checked once at the end of an open. /// /// An index that is merely *incomplete* is the worst failure this engine can /// have, because nothing reports it: the index only generates candidates and /// the full filter is re-applied to those, so a missing entry is a missing /// query result and every other check still passes. That is exactly how the /// replay-time `createIndex` bug survived -- `countDocuments` was right, /// `find({})` was right, and only `find({k: v})` was quietly short. /// /// `_id_` is exact: one entry per document, always. A secondary index is /// checked only when its shape makes the count exact -- `sparse` omits /// documents missing the key, and `multikey` contributes several entries for /// one document -- so those are compared as a lower bound instead of an /// equality. Debug and ReleaseSafe only; an open is not a hot path, but a /// full index walk per collection is not free either. fn assert_indexes_cover_every_document(self: *Engine) void { if (builtin.mode == .ReleaseFast or builtin.mode == .ReleaseSmall) return; var db_it = self.dbs.iterator(); while (db_it.next()) |db_entry| { var coll_it = db_entry.value_ptr.collections.iterator(); while (coll_it.next()) |coll_entry| { const coll = coll_entry.value_ptr.*; assert_msg( coll.id_index.count() == coll.doc_count, "the _id_ index must hold exactly one entry per document after an open", ); assert_msg( coll.id_index.unreachable_key_count() == 0, "every _id_ entry must be findable by descent, not only by iteration", ); for (coll.indexes.items) |ix| { // Reachability applies to every index whatever its shape: an // entry in the leaf chain that a descent cannot find is a // query result that silently goes missing. if (ix.unreachable_key_count() != 0) { ix.dbg_root(); @panic("unreachable index entries"); } if (ix.sparse or ix.multikey) continue; assert_msg( ix.count() >= coll.doc_count, "a non-sparse index must cover every document after an open", ); } } } } fn build_all_indexes(self: *Engine) !void { var db_it = self.dbs.iterator(); while (db_it.next()) |db_entry| { var coll_it = db_entry.value_ptr.collections.iterator(); while (coll_it.next()) |coll_entry| { for (coll_entry.value_ptr.*.indexes.items) |ix| { try self.rebuild_index(coll_entry.value_ptr.*, ix); } try self.rebuild_index(coll_entry.value_ptr.*, &coll_entry.value_ptr.*.id_index); } } } /// Rebuild one index from the live documents. Runs after replay, so it /// is order-independent; indexes already holding entries (maintained /// live) are skipped defensively. A duplicate under a unique index logs /// a loud warning and keeps the index (still correct as a candidate /// generator; future writes are still enforced) — the database always /// opens, leaving dropIndexes as an in-band recovery path. fn rebuild_index(self: *Engine, coll: *Collection, ix: *index.Index) !void { if (ix.count() > 0) return; // defensive // The `_id_` tree is the enumeration of live documents now. It is also // ordered, so this reads the slab sequentially where the hashmap read it // in hash order. var doc_it = coll.id_index.iter(); while (doc_it.next()) |entry| { ix.append_doc_entries(self.gpa, coll.doc_bytes(entry.off), entry.off) catch |err| switch (err) { error.ParallelArrays => { std.debug.print( "multiforadb: WARNING: index '{s}' cannot index an existing " ++ "document; entry skipped\n", .{ ix.name, }, ); continue; }, else => return err, }; } // Tolerated, not enforced: the database must always open. defer self.pager.release_reservation(&ix.hold); if (try ix.finish_bulk(self.gpa, false)) { std.debug.print( "multiforadb: WARNING: unique index '{s}' has duplicate keys in existing " ++ "data; duplicates not enforced for existing documents\n", .{ ix.name, }, ); } } // -- checkpoint --------------------------------------------------------- /// The catalog: everything about where the engine's structures live that is /// not recoverable by looking at the pages themselves. /// /// Written wholesale into freshly allocated pages at every checkpoint, never /// mutated in place, so the previous copy stays intact and referenced by the /// previous watermark until the new one switches over. That is what makes it /// untearable, and it is why there is no incremental catalog update path. /// /// Format (little-endian throughout): /// u32 magic "MFCT", u32 version, u64 live_docs, u32 db_count /// per db: u32 name_len, name, u32 coll_count /// per coll: u32 name_len, name, u64 slab_tail, u64 slab_end, /// u32 extent_count, (u32 first, u32 pages)*, u32 index_count /// index 0 is always the implicit _id_ /// per index: u32 name_len, name, u32 key_count, /// (u32 path_len, path, u8 descending)*, /// u8 flags(unique|sparse|multikey|has_ttl), i64 ttl, /// u32 root, u32 first_leaf, u32 leaf_count, u32 depth, /// u64 entry_count, u64 ovf_tail, u64 ovf_end, /// u32 ovf_extent_count, (u32 first, u32 pages)*, /// u32 node_count, (u32 page)* /// u64 xxhash3 over everything above const catalog_magic: u32 = 0x4D464354; // "MFCT" const catalog_version: u32 = 1; /// What a catalog walk observed, for the caller to check the engine's own /// running totals against. const CatalogSums = struct { live: u64, dead: u64 }; /// Serialize the catalog and return the byte totals it observed. /// /// The engine's own totals are by definition the sums over collections, and /// `read_catalog` rebuilds them that way, so a divergence means some path /// published or evicted bytes at one level and not the other -- with a /// compaction trigger that fires never or always as the visible symptom. /// The check is worth making and this is where every collection is walked /// anyway, but it cannot be made *here*: the sums are accumulated across /// collections over time while the engine's totals move under them, so a /// writer landing mid-walk would trip it on a database that is perfectly /// consistent. The caller asserts them after the `seq` check has established /// that no writer landed at all. fn write_catalog(self: *Engine, out: *std.ArrayListUnmanaged(u8)) !CatalogSums { const gpa = self.gpa; var live_sum: u64 = 0; var dead_sum: u64 = 0; try put_u32(gpa, out, catalog_magic); try put_u32(gpa, out, catalog_version); try put_u64(gpa, out, self.counters().live_docs); try put_u32(gpa, out, @intCast(self.dbs.count())); var db_it = self.dbs.iterator(); while (db_it.next()) |db_entry| { try put_bytes(gpa, out, db_entry.key_ptr.*); const colls = &db_entry.value_ptr.collections; try put_u32(gpa, out, @intCast(colls.count())); var coll_it = colls.iterator(); while (coll_it.next()) |ce| { const coll = ce.value_ptr.*; // Everything below this line is written by a collection's own // writer under its own lock, and `slab_runs` is an ArrayList // that `slab_reserve` appends to -- so reading it under only the // shared catalog lock could walk a slice a concurrent append had // already reallocated. Lock order is catalog then collection, // the same order `compact` uses. try coll.lock.lockShared(self.io); defer coll.lock.unlockShared(self.io); try put_bytes(gpa, out, ce.key_ptr.*); try put_u64(gpa, out, coll.slab_tail); try put_u64(gpa, out, coll.slab_end); try put_u64(gpa, out, coll.slab_used); try put_u64(gpa, out, coll.live_bytes); live_sum += coll.live_bytes; assert_msg( coll.live_bytes <= coll.slab_used, "a collection cannot hold more live bytes than it ever appended", ); dead_sum += coll.slab_used - coll.live_bytes; // Runs, not extents, but the same two u32s: only the window map // is new and it is deliberately not persisted (see // `dead_unlocated`), so `catalog_version` stays 1 and there is // no second read path to keep working. try put_u32(gpa, out, @intCast(coll.slab_runs.items.len)); for (coll.slab_runs.items) |r| { try put_u32(gpa, out, r.first); try put_u32(gpa, out, r.pages); } try put_u32(gpa, out, @intCast(coll.indexes.items.len + 1)); try write_index_catalog(gpa, out, &coll.id_index); for (coll.indexes.items) |ix| try write_index_catalog(gpa, out, ix); } } try put_u64(gpa, out, std.hash.XxHash3.hash(0, out.items)); return .{ .live = live_sum, .dead = dead_sum }; } fn write_index_catalog( gpa: std.mem.Allocator, out: *std.ArrayListUnmanaged(u8), ix: *const index.Index, ) !void { try put_bytes(gpa, out, ix.name); try put_u32(gpa, out, @intCast(ix.keys.len)); for (ix.keys) |k| { try put_bytes(gpa, out, k.path); try out.append(gpa, @intFromBool(k.descending)); } var flags: u8 = 0; if (ix.unique) flags |= 1; if (ix.sparse) flags |= 2; if (ix.multikey) flags |= 4; if (ix.ttl != null) flags |= 8; try out.append(gpa, flags); try put_u64(gpa, out, @bitCast(ix.ttl orelse 0)); try put_u32(gpa, out, ix.root); try put_u32(gpa, out, ix.first_leaf); try put_u32(gpa, out, ix.leaf_count); try put_u32(gpa, out, ix.depth); try put_u64(gpa, out, ix.entry_count); try put_u64(gpa, out, ix.ovf_tail); try put_u64(gpa, out, ix.ovf_end); try put_u32(gpa, out, @intCast(ix.ovf_extents.items.len)); for (ix.ovf_extents.items) |e| { try put_u32(gpa, out, e.first); try put_u32(gpa, out, e.pages); } try put_u32(gpa, out, @intCast(ix.node_pages.items.len)); for (ix.node_pages.items) |pg| try put_u32(gpa, out, pg); } /// Rebuild the catalog from the data file. On any inconsistency this returns /// an error and the caller falls back to a full replay. fn read_catalog(self: *Engine) !void { const wm = self.pager.loaded; if (wm.catalog_len == 0) return error.NoCatalog; const buf = self.pager.bytes(@as(u64, wm.catalog_page) << pgr.page_shift, @intCast(wm.catalog_len)); if (buf.len < 8) return error.CorruptCatalog; const body = buf[0 .. buf.len - 8]; if (std.hash.XxHash3.hash(0, body) != std.mem.readInt(u64, buf[buf.len - 8 ..][0..8], .little)) { return error.CorruptCatalog; } var r: Reader = .{ .b = body }; if (try r.read_u32() != catalog_magic) return error.CorruptCatalog; if (try r.read_u32() != catalog_version) return error.CorruptCatalog; self.live_docs = try r.read_u64(); const ndbs = try r.read_u32(); var d: u32 = 0; while (d < ndbs) : (d += 1) { const db_name = try r.read_bytes(); const ncolls = try r.read_u32(); var c: u32 = 0; while (c < ncolls) : (c += 1) { const coll_name = try r.read_bytes(); const coll = try self.get_or_create_collection(db_name, coll_name); coll.slab_tail = try r.read_u64(); coll.slab_end = try r.read_u64(); coll.slab_used = try r.read_u64(); coll.live_bytes = try r.read_u64(); // Both engine totals are sums over collections rather than // separately stored fields, so neither can disagree with the // catalog. `slab_used - live_bytes` is this collection's slab // garbage by definition -- bytes it appended and no longer // reaches -- which is exactly what the rebuild trigger counts. if (coll.slab_used < coll.live_bytes) return error.CorruptCatalog; self.live_bytes += coll.live_bytes; self.dead_bytes += coll.slab_used - coll.live_bytes; // An open knows *that* the collection has garbage but not // *where*: the window map is rebuilt empty, and the whole // amount starts out unlocated. The consequence runs one way -- // a forgotten dead byte is a window that is not handed back, // never a live window that is. coll.dead_unlocated = coll.slab_used - coll.live_bytes; const nex = try r.read_u32(); var e: u32 = 0; while (e < nex) : (e += 1) { const first = try r.read_u32(); const pages = try r.read_u32(); // Sorted insert rather than append: a catalog written // before runs were address-ordered holds them in // allocation order, and `run_of` is a binary search. try coll.insert_run(self.gpa, first, pages); } const nix = try r.read_u32(); // Index 0 is the implicit _id_, already created by // get_or_create_collection; the rest are registered here. try read_index_catalog(self.gpa, &r, &coll.id_index); var i: u32 = 1; while (i < nix) : (i += 1) { const boxed = try self.gpa.create(index.Index); errdefer self.gpa.destroy(boxed); boxed.* = try index.Index.init(self.gpa, self.pager, "", &.{}, false, false, null); read_index_catalog(self.gpa, &r, boxed) catch |err| { boxed.deinit(self.gpa); self.gpa.destroy(boxed); return err; }; try coll.indexes.append(self.gpa, boxed); } // Nothing else to rebuild. The `_id_` tree *is* the lookup, and // it is already in the file -- which is the whole reason an open // no longer touches a single document page. The version of this // that kept a hashmap had to read every document here to recover // its `_id`, and that alone faulted the entire database in. coll.doc_count = coll.id_index.count(); } } } /// Replace an index's identity and tree position from the catalog. The index /// arrives freshly initialised, so its own two starting pages are discarded /// in favour of what was published. fn read_index_catalog( gpa: std.mem.Allocator, r: *Reader, ix: *index.Index, ) !void { const name = try r.read_bytes(); const nkeys = try r.read_u32(); if (nkeys == 0 or nkeys > index.max_index_keys) return error.CorruptCatalog; var keys = try gpa.alloc(index.IndexKey, nkeys); var built: usize = 0; errdefer { for (keys[0..built]) |k| gpa.free(k.path); gpa.free(keys); } while (built < nkeys) : (built += 1) { const path = try r.read_bytes(); keys[built] = .{ .path = try gpa.dupe(u8, path), .descending = (try r.read_byte()) != 0 }; } const flags = try r.read_byte(); const ttl_raw: i64 = @bitCast(try r.read_u64()); const new_name = try gpa.dupe(u8, name); errdefer gpa.free(new_name); // Swap in the published identity, freeing what init made. for (ix.keys) |k| gpa.free(k.path); gpa.free(ix.keys); gpa.free(ix.name); ix.name = new_name; ix.keys = keys; ix.unique = flags & 1 != 0; ix.sparse = flags & 2 != 0; ix.multikey = flags & 4 != 0; ix.ttl = if (flags & 8 != 0) ttl_raw else null; ix.root = try r.read_u32(); ix.first_leaf = try r.read_u32(); ix.leaf_count = try r.read_u32(); ix.depth = try r.read_u32(); ix.entry_count = @intCast(try r.read_u64()); ix.ovf_tail = try r.read_u64(); ix.ovf_end = try r.read_u64(); const novf = try r.read_u32(); var o: u32 = 0; while (o < novf) : (o += 1) { const first = try r.read_u32(); const pages = try r.read_u32(); try ix.ovf_extents.append(gpa, .{ .first = first, .pages = pages }); } const nnodes = try r.read_u32(); if (nnodes < 2) return error.CorruptCatalog; ix.node_pages.clearRetainingCapacity(); try ix.node_pages.ensureTotalCapacity(gpa, nnodes); var n: u32 = 0; while (n < nnodes) : (n += 1) ix.node_pages.appendAssumeCapacity(try r.read_u32()); } /// Add a replayed document's entries to every index that is already /// populated. Best effort and infallible: replay must not refuse to start, /// and an index that cannot key this document is reported and left alone -- /// the same tolerance `rebuild_index` has always had. fn index_doc_on_replay(self: *Engine, coll: *Collection, doc_bytes: []const u8, off: u64) void { self.index_one(&coll.id_index, doc_bytes, off); for (coll.indexes.items) |ix| self.index_one(ix, doc_bytes, off); } fn index_one(self: *Engine, ix: *index.Index, doc_bytes: []const u8, off: u64) void { var built = ix.build_entries(self.gpa, doc_bytes) catch return; defer built.deinit(self.gpa); if (built.multikey) ix.multikey = true; ix.reserve_for(self.gpa, built.entries.items) catch return; ix.insert_entries(&built, off); } fn rebuild_docs_map(self: *Engine, coll: *Collection) !void { var arena = std.heap.ArenaAllocator.init(self.gpa); defer arena.deinit(); var it = coll.id_index.iter(); while (it.next()) |e| { _ = arena.reset(.retain_capacity); const id_value = (try bson.get_at(arena.allocator(), coll.doc_bytes(e.off), "_id")) orelse continue; const id_key = try bson.serialize_value(self.gpa, id_value); errdefer self.gpa.free(id_key); try coll.docs.put(self.gpa, id_key, e.off); } } /// Discard everything a failed catalog load put in place, so the caller can /// replay the log into a clean engine. fn reset_after_failed_catalog(self: *Engine) void { var db_it = self.dbs.iterator(); while (db_it.next()) |db_entry| { self.free_db(db_entry.value_ptr); self.gpa.free(db_entry.key_ptr.*); } self.dbs.clearRetainingCapacity(); self.live_docs = 0; self.dead_docs = 0; self.live_bytes = 0; self.dead_bytes = 0; self.seq = 0; self.committed_seq = 0; } /// Hand back every slab window with nothing live left in it, across every /// collection. The first phase of a checkpoint. /// /// Inside the checkpoint rather than a hook after it, and that placement is /// the whole safety argument. Reclamation changes two things: it splits /// `slab_runs`, which the catalog describes, and it calls /// `pager.free_pages`, which the free list describes. The checkpoint's /// single `publish` makes both durable together, so a crash before it /// leaves the old catalog and the old free list -- no reclamation happened /// -- and a crash after leaves both describing the new ownership. There is /// no order in between to get wrong, and no new record type or replay path. /// /// Batched with the checkpoint for a second reason: the write path pays /// only for a counter update, and the cadence of actually returning pages /// is the checkpoint threshold rather than per-delete. /// /// Catalog shared, then each collection exclusive, one at a time -- the /// order `compact` and `write_catalog` both use. fn reclaim_slabs(self: *Engine) void { self.catalog_lock.lockSharedUncancelable(self.io); defer self.catalog_lock.unlockShared(self.io); var db_it = self.dbs.iterator(); while (db_it.next()) |db_entry| { var coll_it = db_entry.value_ptr.collections.iterator(); while (coll_it.next()) |ce| self.reclaim_collection(ce.value_ptr.*); } } fn reclaim_collection(self: *Engine, coll: *Collection) void { coll.lock.lockUncancelable(self.io); defer coll.lock.unlock(self.io); // The common case, and the reason this is affordable at every // checkpoint: a collection with no full window is not scanned at all. if (coll.full_windows == 0) return; // Out of memory here means the garbage stays where it is. Nothing is // lost and the next checkpoint tries again. const freed = coll.reclaim_windows(self.gpa) catch return; if (freed == 0) return; self.counter_lock.lockUncancelable(self.io); assert_msg(self.dead_bytes >= freed, "reclaiming more slab than the engine counts as dead"); self.dead_bytes -= freed; self.counter_lock.unlock(self.io); // A cursor holding slab offsets is now holding some that name pages // this collection no longer owns -- and reading them would succeed, // since the pages are only on the free list, so the answer would be // plausible garbage rather than an error. `cursor_still_valid` compares // this epoch and kills such a cursor with QueryPlanKilled, which is // what a rebuild already does to it. // // Only when something was actually given back: a collection that // reclaimed nothing must not have its cursors killed on the cadence of // the checkpoint. self.layout_epoch_seq += 1; coll.layout_epoch = self.layout_epoch_seq; } /// Publish the current state as a checkpoint. /// /// The watermark equals the sequence the log has already made durable, never /// more: `commit` first, then snapshot, and the snapshot is validated against /// an unchanged `seq` under the log lock -- the same bounded-retry shape /// compaction has always used. That is the crash-recovery invariant (PLAN /// D6) reduced to an ordering. pub fn checkpoint(self: *Engine) !void { try self.commit(); self.reclaim_slabs(); var buf: std.ArrayListUnmanaged(u8) = .empty; defer buf.deinit(self.gpa); var attempt: usize = 0; const attempt_max = 8; while (attempt < attempt_max) : (attempt += 1) { buf.clearRetainingCapacity(); try self.catalog_lock.lockShared(self.io); const snapshot_seq = self.seq; const before = self.counters(); const sums = self.write_catalog(&buf) catch |err| { self.catalog_lock.unlockShared(self.io); return err; }; self.catalog_lock.unlockShared(self.io); try self.log_lock.lock(self.io); if (self.seq != snapshot_seq) { // A writer landed mid-snapshot; the catalog describes a state // that no longer matches the log. Retry rather than publish it. self.log_lock.unlock(self.io); continue; } if (snapshot_seq > self.committed_seq) { // A writer appended before the snapshot and its commit has not // landed yet -- it is between `insert` and `commit`, or inside // one, waiting on the leader's fsync. The seq check above does // not catch this: nothing appended *during* the walk, the // append was already there when it started. // // Publishing here would claim durability for a record that is // still in the log's buffer, and the truncation that follows a // checkpoint would then throw it away. That is the one thing // the whole watermark ordering exists to prevent (PLAN D6), and // it used to be an assertion -- so the failure mode was a // server abort under exactly the load that makes checkpoints // frequent. Reproduced in seconds by four writers against a // checkpoint loop, and the window is as wide as an fsync. // // Seal it and take the snapshot again rather than spinning: // `commit` covers every append made so far, so one more round // is enough. Outside `log_lock`, which `commit` takes itself. self.log_lock.unlock(self.io); try self.commit(); continue; } // Only when the walk was quiet. An unchanged `seq` is not enough on // its own: a writer bumps it when it appends the log record and // updates the byte counters afterwards, so it can be past the seq // the snapshot captured and still be about to move `live_bytes` // under a collection the walk has already been through. Requiring // the engine total to be unmoved across the whole walk closes that, // at the cost of skipping the check under sustained writes -- which // is the right trade, because what it guards against is a code path // that updates one level and not the other, and that is // deterministic wherever it exists. // // Both reads go through `counters`, which is the only way the // comparison means anything: the sums were gathered under each // collection's lock, which orders them against that collection's // writer, and an unsynchronized read of the engine's own totals is // ordered against nothing at all -- so it could return a figure from // before a write the walk had already serialized, and the assertion // would abort a server whose accounting was correct. const after = self.counters(); if (after.live_bytes == before.live_bytes) assert_msg( sums.live == before.live_bytes, "the engine's live-byte total must equal the sum over collections", ); // The same argument, for the total the rebuild trigger reads. This // is what makes a drop's accounting checkable: charge the engine for // a dropped collection's bytes and the two sides part company here. if (after.dead_bytes == before.dead_bytes) assert_msg( sums.dead == before.dead_bytes, "the engine's dead-byte total must equal the sum over collections", ); const pages: u32 = @intCast((buf.items.len + pgr.page_size - 1) / pgr.page_size); const first = self.pager.alloc_pages(pages) catch |err| { self.log_lock.unlock(self.io); return err; }; @memcpy(self.pager.bytes_mut(@as(u64, first) << pgr.page_shift, buf.items.len), buf.items); // The invariant, on the line that would break it: `log_lock` has // been held since the check above and `committed_seq` only grows. assert_msg( snapshot_seq <= self.committed_seq, "checkpoint watermark past the durable log tail", ); self.pager.publish(.{ .seq = snapshot_seq, .catalog_page = first, .catalog_len = buf.items.len, .live_docs = after.live_docs, .dead_bytes = after.dead_bytes, }) catch |err| { self.log_lock.unlock(self.io); return err; }; self.pager.release_reservation(&self.hold); // The watermark is durable, so every record it covers is now // redundant. Strictly after the publish: the other order loses data // if a crash lands between them. self.log.truncate_to_header() catch |err| { // A failed truncation wastes space and costs replay time on the // next open; it does not lose anything, because the records are // still there and still above no watermark. Not worth failing // the checkpoint that already succeeded. std.debug.print("multiforadb: WARNING: log truncation failed: {s}\n", .{@errorName(err)}); }; self.committed_seq = snapshot_seq; self.log_lock.unlock(self.io); return; } std.debug.print("multiforadb: WARNING: checkpoint gave up after {d} attempts under sustained writes\n", .{attempt_max}); } /// Register an (empty) index from a persisted spec document. A repeated /// create record for the same name is an idempotent no-op. /// Register an index from a logged spec. Returns the new index, or null when /// one of that name was already present (a re-registration is a no-op, not an /// error). The caller needs the pointer because an index registered during /// replay may have to be built over documents that replay will never see. fn register_index_from_spec( self: *Engine, coll: *Collection, spec_doc: *const bson.Document, ) !?*index.Index { const parsed = try index.parse_spec(self.gpa, self.pager, spec_doc); const ix = self.gpa.create(index.Index) catch |err| { var dead = parsed; dead.deinit(self.gpa); return err; }; ix.* = parsed; var committed = false; defer if (!committed) { ix.deinit(self.gpa); self.gpa.destroy(ix); }; if (coll.find_index(ix.name) != null) return null; try coll.indexes.append(self.gpa, ix); committed = true; return ix; } }; fn parent_dir(path: []const u8) []const u8 { const last = std.mem.lastIndexOfScalar(u8, path, '/') orelse return "."; if (last == 0) return "/"; return path[0..last]; } fn serialize_doc(gpa: std.mem.Allocator, doc: *const bson.Document) ![]u8 { var out: std.ArrayListUnmanaged(u8) = .empty; errdefer out.deinit(gpa); try doc.to_bytes(gpa, &out); return out.toOwnedSlice(gpa); } fn apply_record(ctx: *anyopaque, record: storage.Record, doc: *bson.Document) anyerror!void { const self: *Engine = @ptrCast(@alignCast(ctx)); // The document is transient: only its canonical bytes are stored in the // collection slab. Always owned by this frame. defer { doc.deinit(); self.gpa.destroy(doc); } const coll = self.get_or_create_collection(record.db, record.coll) catch return; // Index records carry no _id — handle them before the lookup. Replay // registers indexes empty; Engine.open builds them from the live docs // after replay completes. switch (record.type) { storage.record_type_index_create => { const registered = self.register_index_from_spec(coll, doc) catch |err| { std.debug.print("multiforadb: index create record failed to apply: {s}\n", .{ @errorName(err), }); return; }; // A checkpointed open replays only what the watermark does not cover, // so the documents already in the image never reach this index. Build // it over them now, which is what the live `create_index` command // does with pre-existing documents. // // Leaving it to `build_all_indexes` does not work and fails silently: // the next upsert in the log puts one entry in, and a non-empty index // is skipped by the `count() > 0` guard there -- so the index ends up // holding the documents logged after its creation and none of the // ones logged before, which is an index that under-approximates. // // Only for a maintaining replay. A full replay leaves every secondary // index empty on purpose and `build_all_indexes` fills them in one // pass at the end, which is cheaper than one pass per index here. if (self.replay_maintains_indexes) { if (registered) |ix| self.rebuild_index(coll, ix) catch |err| { // The database must always open (ground rule 4). A failure // here leaves the index short, so say so rather than leaving // a query to be quietly wrong about it. std.debug.print( "multiforadb: WARNING: index '{s}' could not be built over existing " ++ "documents during replay: {s}; drop and re-create it\n", .{ ix.name, @errorName(err) }, ); }; } return; }, storage.record_type_index_drop => { const name_value = doc.get("name") orelse return; const name = switch (name_value) { .string => |s| s, else => return, }; _ = coll.remove_index(self.gpa, name); return; }, else => {}, } const id_value = doc.get("_id") orelse { std.debug.print("multiforadb: log record without _id, skipping\n", .{}); return; }; var id_enc_list: std.ArrayListUnmanaged(u8) = .empty; defer id_enc_list.deinit(self.gpa); bson.encode_key(id_value, self.gpa, &id_enc_list) catch return; const id_enc = id_enc_list.items; // Engine.seq used to restart at 0 on every open, which was harmless while // the log was always replayed in full and fatal the moment a watermark // exists: the first append after an open would reuse a sequence at or below // it, and the *next* open would discard that record as already-checkpointed. self.seq = @max(self.seq, record.seq); switch (record.type) { storage.record_type_upsert => { // A record that supersedes one already present. Worth reporting when // the two `_id`s are not byte-identical, because that means the // database predates `_id_` being canonical and held two documents // whose ids compare equal -- int32 1 and int64 1, say. One of them is // being dropped here, which is MongoDB's semantics but is also silent // data loss for an existing file (PLAN amendment A4). if (coll.id_index.lookup_exact(id_enc)) |old_off| { warn_on_equal_id_collision(self.gpa, coll, old_off, doc, record); } self.evict_doc(coll, id_enc); const doc_bytes = try serialize_doc(self.gpa, doc); defer self.gpa.free(doc_bytes); self.count_slab_skip(try coll.slab_reserve(self.gpa, doc_bytes.len)); const off = self.publish_doc_bytes(coll, doc_bytes); coll.doc_count += 1; // The `_id_` entry is added *now*, not after replay: it is the only // way the next record can find this document to supersede it. The // secondaries can still wait for the bulk build, unless this open // came from a checkpoint that already populated them. if (self.replay_maintains_indexes) { self.index_doc_on_replay(coll, doc_bytes, off); } else { self.index_one(&coll.id_index, doc_bytes, off); } self.release_write_reservations(coll); // The _id_ entry is added after replay, in build_all_indexes, // together with the secondary indexes. // // Which is why making _id_ unique cannot lose a document here: // eviction above goes through the docs map, keyed on // serialize_value, so a database holding both {_id: int32 1} and // {_id: int64 1} keeps both. The bulk build then finds duplicate // canonical keys, tolerates them and warns (rule: the database // must always open). The commit that drops the docs map is where // that stops being true -- see PLAN amendment A4. }, storage.record_type_delete => self.evict_doc(coll, id_enc), else => {}, } } // --------------------------------------------------------------------------- // Tests // --------------------------------------------------------------------------- const testing = std.testing; const TmpLog = storage.TmpLog; fn test_env(threaded: *std.Io.Threaded) struct { io: std.Io, gen: bson.ObjectIdGen } { const io = threaded.io(); const gen = bson.ObjectIdGen.init(io); return .{ .io = io, .gen = gen }; } fn make_doc(gpa: std.mem.Allocator, id: i32, name: []const u8) !bson.Document { var arena = std.heap.ArenaAllocator.init(gpa); errdefer arena.deinit(); const pairs = try arena.allocator().alloc(bson.Pair, 2); pairs[0] = .{ .key = try arena.allocator().dupe(u8, "_id"), .value = .{ .int32 = id } }; pairs[1] = .{ .key = try arena.allocator().dupe(u8, "name"), .value = .{ .string = try arena.allocator().dupe(u8, name) } }; return .{ .arena = arena, .pairs = pairs }; } test "insert, query, remove" { var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); var d1 = try make_doc(gpa, 1, "alice"); defer d1.deinit(); var d2 = try make_doc(gpa, 2, "bob"); defer d2.deinit(); try engine.lock(); try engine.insert("app", "users", &d1, &env.gen); try engine.insert("app", "users", &d2, &env.gen); engine.unlock(); // duplicate key var d3 = try make_doc(gpa, 1, "alice2"); defer d3.deinit(); try engine.lock(); try testing.expectError(error.DuplicateKey, engine.insert("app", "users", &d3, &env.gen)); engine.unlock(); // find by id const id_key = try id_key_for(gpa, bson.Value{ .int32 = 2 }); defer gpa.free(id_key); try engine.lock(); const found = engine.get_doc("app", "users", id_key).?; try testing.expectEqualStrings("bob", (try bson.get_at(gpa, found, "name")).?.string); const removed = try engine.remove_by_id("app", "users", .{ .int32 = 2 }); try testing.expect(removed); engine.unlock(); } test "live/dead doc accounting drives compaction" { var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); // Keep compaction from firing and resetting dead_docs mid-test. engine.compact_threshold = std.math.maxInt(u64); var d1 = try make_doc(gpa, 1, "alice"); defer d1.deinit(); var d2 = try make_doc(gpa, 2, "bob"); defer d2.deinit(); try engine.lock(); defer engine.unlock(); try engine.insert("app", "users", &d1, &env.gen); try engine.insert("app", "users", &d2, &env.gen); try testing.expectEqual(@as(u64, 2), engine.live_docs); try testing.expectEqual(@as(u64, 0), engine.dead_docs); // A replace supersedes one record: live is unchanged, garbage grows. var d1b = try make_doc(gpa, 1, "alice2"); defer d1b.deinit(); _ = try engine.replace("app", "users", &d1b, &env.gen); try testing.expectEqual(@as(u64, 2), engine.live_docs); try testing.expectEqual(@as(u64, 1), engine.dead_docs); // A delete drops a live doc and leaves its record behind as garbage. try testing.expect(try engine.remove_by_id("app", "users", .{ .int32 = 2 })); try testing.expectEqual(@as(u64, 1), engine.live_docs); try testing.expectEqual(@as(u64, 2), engine.dead_docs); // Removing something absent must not move either counter. try testing.expect(!try engine.remove_by_id("app", "users", .{ .int32 = 99 })); try testing.expectEqual(@as(u64, 1), engine.live_docs); try testing.expectEqual(@as(u64, 2), engine.dead_docs); // Dropping the collection accounts for everything it still held, and // must leave live_docs at zero rather than wrapping. try testing.expect(try engine.drop_collection("app", "users")); try testing.expectEqual(@as(u64, 0), engine.live_docs); try testing.expectEqual(@as(u64, 3), engine.dead_docs); } test "compaction reclaims garbage but leaves a garbage-free log alone" { var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = 4096; // small enough to be crossed here try engine.lock(); defer engine.unlock(); // Pure inserts produce no garbage, so the log must never be rewritten. for (0..200) |i| { var d = try make_doc(gpa, @intCast(i), "x"); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } try engine.commit(); try testing.expectEqual(@as(u64, 0), engine.dead_docs); // The compaction threshold counts data volume (uncompressed bytes), not // the compressed on-disk size. const after_insert = engine.log.data_bytes; try testing.expect(after_insert > engine.compact_threshold); // Rewriting every document makes the log mostly garbage; compaction // must fire and bring the file back down near the live size. for (0..200) |round| { for (0..200) |i| { var d = try make_doc(gpa, @intCast(i), if (round % 2 == 0) "yy" else "z"); defer d.deinit(); _ = try engine.replace("app", "c", &d, &env.gen); } try engine.commit(); if (engine.log.data_bytes >= after_insert * 2) break; } // The server runs compaction in a write command's epilogue; the tests // drive it directly. if (engine.take_compact()) try engine.compact(); try engine.commit(); try testing.expectEqual(@as(u64, 200), engine.live_docs); // Bounded well below the ~40x of record bytes those rewrites wrote. try testing.expect(engine.log.data_bytes < after_insert * 2); } test "a secondary index stays reachable across checkpoints, churn and a rebuild" { // The one path the index unit tests cannot reach: copy-on-write. `test_pager` // never publishes a watermark, so `stable_pages` is 0 there and every page is // writable in place -- no node page is ever relocated. Through the engine a // checkpoint makes the whole image stable, so the next tree mutation copies // each node it touches to a fresh page and rewrites the id->page slot. // // Few distinct keys on purpose: ten values over thousands of documents means // each value spans many leaves and most interior separators are duplicates, // which is the shape the crash fuzzer fails on. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = 64 * 1024; // rebuild often, like --heavy try engine.lock(); defer engine.unlock(); var spec = try index_spec(gpa, "k", "k_1", false, false, null); defer spec.deinit(); _ = try engine.create_index("app", "c", &spec); const n_keys: i32 = 10; const n: i32 = 1200; var id: i32 = 0; while (id < n) : (id += 1) { var d = try make_keyed(gpa, id, @mod(id, n_keys)); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); // Checkpoint, churn and rebuild interleaved with the writes, so tree // mutations land on pages the last checkpoint froze. if (@mod(id, 150) == 0) { try engine.commit(); try engine.checkpoint(); } if (@mod(id, 7) == 0 and id > 20) { _ = try engine.remove_by_id("app", "c", .{ .int32 = id - 20 }); } if (engine.take_compact()) try engine.compact(); } try engine.commit(); try engine.checkpoint(); const coll = engine.get_collection("app", "c").?; const ix = coll.find_index("k_1").?; // Every entry the leaf chain holds must also be findable by descending from // the root, which is the only way a query reaches it. try testing.expectEqual(@as(u32, 0), ix.unreachable_key_count()); try testing.expectEqual(@as(u32, 0), coll.id_index.unreachable_key_count()); // And per key, the index must agree with a scan of the documents. var k: i32 = 0; while (k < n_keys) : (k += 1) { var want: usize = 0; var scan = coll.id_index.iter(); while (scan.next()) |e| { const kv = try bson.get_at(gpa, coll.doc_bytes(e.off), "k"); if (kv) |v| if (v.int32 == k) { want += 1; }; } var out: std.ArrayListUnmanaged(u64) = .empty; defer out.deinit(gpa); try ix.lookup_eq(gpa, &.{.{ .int32 = k }}, &out); testing.expectEqual(want, out.items.len) catch |err| { std.debug.print(" key {d}: index {d}, scan {d}, entry_count {d}\n", .{ k, out.items.len, want, ix.count(), }); return err; }; } } test "an index created after the checkpoint indexes the documents that predate it" { // Found by tests/fuzz/crash-fuzz.js in --heavy mode, roughly once per 700 // crash/reopen cycles, as `find({k:v})` returning nothing for a key that has // documents. `countDocuments` and `find({})` were right, so the documents // were there and only the index's answer about them was wrong -- an index // that under-approximates, which is silent by construction: the index only // generates candidates and the full filter is re-applied to those, so a // missing entry is a missing result and nothing complains. // // The sequence needs three things at once: a checkpoint, a `createIndex` // logged after it, and a write after that. // // 1. documents exist and a checkpoint puts them in the durable image // 2. createIndex is logged *after* the watermark // 3. another document is written, also after the watermark // // On reopen the catalog restores step 1's documents but not the index, so // replay starts at the watermark and never sees them. Replay registers the // index empty and -- because a checkpointed open maintains indexes as it // replays -- step 3's document goes in. The index is now non-empty and // incomplete, so `rebuild_index`'s `count() > 0` guard skips it and step 1's // documents are never indexed at all. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); try engine.lock(); // 1. Two documents, made durable in the data file. var d1 = try make_user(gpa, 1, "a@x.io"); defer d1.deinit(); var d2 = try make_user(gpa, 2, "b@x.io"); defer d2.deinit(); try engine.insert("app", "users", &d1, &env.gen); try engine.insert("app", "users", &d2, &env.gen); try engine.commit(); try engine.checkpoint(); // 2. The index arrives after the watermark. var spec = try index_spec(gpa, "email", "email_1", false, false, null); defer spec.deinit(); _ = try engine.create_index("app", "users", &spec); // 3. And a write after that, which is what makes the index non-empty on // replay and so hides the two documents behind the `count() > 0` guard. var d3 = try make_user(gpa, 3, "c@x.io"); defer d3.deinit(); try engine.insert("app", "users", &d3, &env.gen); try engine.commit(); engine.unlock(); // No second checkpoint: the watermark still predates the createIndex. } var engine2 = try Engine.open(gpa, io, tmp.path); defer engine2.deinit(); const coll = engine2.get_collection("app", "users").?; const ix = coll.find_index("email_1").?; // One entry per document. Under-approximation is the whole failure mode, so // the count is the assertion that matters. try testing.expectEqual(@as(u64, 3), coll.doc_count); try testing.expectEqual(@as(usize, 3), ix.count()); // And every entry resolves to a document whose email re-encodes to its key, // so the entries are the right ones and not merely the right number. var seen: [3]bool = .{ false, false, false }; var it = ix.iter(); while (it.next()) |entry| { const doc_id = (try bson.get_at(gpa, coll.doc_bytes(entry.off), "_id")).?; const idx: usize = @intCast(doc_id.int32 - 1); try testing.expect(idx < seen.len); try testing.expect(!seen[idx]); seen[idx] = true; } try testing.expect(seen[0] and seen[1] and seen[2]); } test "reopening without a checkpoint reuses the data file instead of appending to it" { // Mutation check: delete the `loaded.generation == 0` reset of `alloc_tail` // in `Pager.open`. Red -- each reopen starts allocating above the previous // file end, so the file grows by a slab extent every time and no watermark // exists to put the abandoned copy on a free list. Unbounded, and it needs no // crash: a database too small to reach the checkpoint threshold never // publishes a watermark, so every clean reopen took that path. Measured // through the server at 20 documents per cycle: +17 MB per reopen. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); // Three rounds of open, write a little, close -- with no checkpoint, so the // data file never gets a watermark and replay rebuilds everything each time. var tails: [3]u32 = undefined; for (0..3) |round| { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); try testing.expectEqual(@as(u64, 0), engine.pager.loaded.generation); try engine.lock(); for (0..5) |i| { var d = try make_doc(gpa, @intCast(round * 10 + i), "x"); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } try engine.commit(); engine.unlock(); tails[round] = engine.pager.alloc_tail; // Every document written so far is readable, so reuse is not data loss. try testing.expectEqual(@as(u64, (round + 1) * 5), engine.live_docs); } // The third open must not have allocated a third copy of the arena. Reuse // makes the tail essentially flat; appending makes it grow by a slab extent // (2048 pages) per round. try testing.expect(tails[2] < tails[0] + slab_extent_pages); try testing.expect(tails[1] < tails[0] + slab_extent_pages); } test "an append after a checkpoint keeps its extent instead of abandoning it" { // Mutation check: delete the `resumed` branch in `slab_reserve`. Red on the // extent count -- every checkpoint would take a fresh 8 MiB extent per // collection and leave the old one's remaining space stranded, reclaimable // only by a rebuild. A pure-insert workload produces no garbage, so no // rebuild is ever triggered and nothing gives it back: measured at 40 // collections, the data file reached 11.8x the live data and grew ~335 MB per // checkpoint, on course for DatabaseTooLarge at ~6 GB of real data. // // Second mutation: round `resumed` to `pgr.page_size` instead of // `pgr.map_align`. Red on the frozen-page assertion below on any host whose // system page is larger than 4 KiB (16 KiB on Apple Silicon) -- a 4 KiB store // dirties the whole system page, so a torn writeback would take the published // bytes sharing it. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = std.math.maxInt(u64); // no rebuild may intervene try engine.lock(); defer engine.unlock(); var first = try make_doc(gpa, 1, "alice"); defer first.deinit(); try engine.insert("app", "users", &first, &env.gen); try engine.commit(); const coll = engine.get_collection("app", "users").?; try testing.expectEqual(@as(usize, 1), coll.slab_runs.items.len); const extent_start = coll.slab_runs.items[0].start(); try engine.checkpoint(); const tail_at_checkpoint = coll.slab_tail; try testing.expect(tail_at_checkpoint > extent_start); const tail_before = engine.pager.alloc_tail; // The next write must land in the same extent, past the frozen page. var second = try make_doc(gpa, 2, "bob"); defer second.deinit(); try engine.insert("app", "users", &second, &env.gen); try engine.commit(); try testing.expectEqual(@as(usize, 1), coll.slab_runs.items.len); // A page or two for the tree's copy-on-write is expected; a whole slab // extent is the regression this guards against. try testing.expect(engine.pager.alloc_tail < tail_before + slab_extent_pages); try testing.expect(coll.slab_tail > tail_at_checkpoint); // The document itself landed on the next system-page boundary past the // frozen tail -- checked at the offset the index recorded, since `slab_tail` // has already advanced past it by the document's length. const bob_enc = try id_key_for(gpa, bson.Value{ .int32 = 2 }); defer gpa.free(bob_enc); const bob_off = coll.id_index.lookup_exact(bob_enc).?; try testing.expectEqual(std.mem.alignForward(u64, tail_at_checkpoint, pgr.map_align), bob_off); // And the page holding the last published byte is still frozen, so the // resumed append cannot have shared a page with the durable image. try testing.expect(!engine.pager.is_unpublished_at(tail_at_checkpoint - 1)); // Both documents readable, and the first one -- which lives below the // checkpoint's tail -- unharmed. try testing.expectEqual(@as(u64, 2), engine.live_docs); for ([_]i32{ 1, 2 }) |id| { const id_enc = try id_key_for(gpa, bson.Value{ .int32 = id }); defer gpa.free(id_enc); const off = coll.id_index.lookup_exact(id_enc).?; const name = try bson.get_at(gpa, coll.doc_bytes(off), "name"); try testing.expectEqualStrings(if (id == 1) "alice" else "bob", name.?.string); } } /// A document of roughly `size` bytes, so a test can fill extents without /// writing tens of thousands of records. fn make_padded(gpa: std.mem.Allocator, id: i32, size: usize) !bson.Document { var arena = std.heap.ArenaAllocator.init(gpa); errdefer arena.deinit(); const a = arena.allocator(); const pad = try a.alloc(u8, size); @memset(pad, 'x'); const pairs = try a.alloc(bson.Pair, 2); pairs[0] = .{ .key = try a.dupe(u8, "_id"), .value = .{ .int32 = id } }; pairs[1] = .{ .key = try a.dupe(u8, "pad"), .value = .{ .string = pad } }; return .{ .arena = arena, .pairs = pairs }; } /// Every dead byte the collection knows about, placed or not. fn dead_total(coll: *const Collection) u64 { return coll.dead_located() + coll.dead_unlocated; } test "every dead slab byte is counted in exactly one place" { // The identity the window map rests on: // // sum of window counters + dead_unlocated == slab_used - live_bytes // // The left side is where the garbage is, the right side is how much there // is; reclamation reads the first and the compaction trigger reads the // second, so a drift between them is a rebuild that fires on a clean // database or a window that is handed back with a document in it. // // Both kinds of death are exercised: evicted documents, and the slab the // appender writes off when a checkpoint freezes the page its cursor points // into. // // Mutation check: drop the `mark_dead` call from `note_skip`, or the // `mark_dead` call from `evict_doc` -- each removes one of the two ways // slab dies and the sides part company by that amount. The run *edges* are // covered separately, by the test below, because whether a run's start is // `map_align`-aligned is up to the allocator and not something an // engine-level test can arrange. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = std.math.maxInt(u64); // no rebuild may intervene try engine.lock(); defer engine.unlock(); var i: i32 = 0; while (i < 40) : (i += 1) { var d = try make_padded(gpa, i, 6000); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); // A checkpoint every few documents, so the appender keeps having to // round its cursor up to a system page and skipping the bytes between. if (@mod(i, 7) == 6) { try engine.commit(); try engine.checkpoint(); } } try engine.commit(); const coll = engine.get_collection("app", "c").?; // Some skipping must actually have happened, or the test proves only the // easy half. 6000-byte documents never land flush against a page boundary. try testing.expect(coll.slab_used > coll.live_bytes); try testing.expectEqual(coll.slab_used - coll.live_bytes, dead_total(coll)); // Now the other kind: evictions. i = 0; while (i < 40) : (i += 2) { try testing.expect(try engine.remove_by_id("app", "c", .{ .int32 = i })); } try engine.commit(); try testing.expectEqual(coll.slab_used - coll.live_bytes, dead_total(coll)); // And the evicted bytes are mostly placeable: 6000-byte documents are far // smaller than a window, so they fall inside one rather than off its edge. try testing.expect(coll.dead_located() > coll.dead_unlocated); } test "dead bytes outside a whole window are counted but not placed" { // A run is allocated in 4 KiB pages but reclaimed in `map_align` windows, // so unless the allocator happens to hand back an aligned run there is a // head below its first window boundary and a tail above its last. Bytes // that die there can never be given back on their own -- but they are // still garbage, and if they were simply dropped the amount of garbage the // collection reports would fall short of the amount it has, which is a // compaction that never fires. // // Driven against `mark_dead` directly: the alignment of a real slab extent // is the allocator's business and an engine-level test cannot arrange for // an unaligned one. // // Mutation checks, each red on its own: drop the trailing `if (pos < // stop)` and the tail bytes go uncounted; drop the leading `if (pos < // r.window_first)` and the window index underflows instead, which takes // down half the suite. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, env.io, tmp.path); defer engine.deinit(); try engine.lock(); defer engine.unlock(); var d = try make_doc(gpa, 1, "alice"); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); const coll = engine.get_collection("app", "c").?; // A run deliberately starting one 4 KiB page past a window boundary, long // enough to hold two whole windows plus a partial one at each end. const pages_per_window: u32 = @intCast(pgr.map_align / pgr.page_size); if (pages_per_window < 2) return error.SkipZigTest; // no edges to test const owned = coll.slab_runs.items[0]; const aligned = std.mem.alignForward(u32, owned.first + owned.pages + 8, pages_per_window); try coll.insert_run(gpa, aligned + 1, 3 * pages_per_window); const ri = coll.run_of(@as(u64, aligned + 1) << pgr.page_shift).?; const r = coll.slab_runs.items[ri]; try testing.expectEqual(@as(usize, 2), r.dead.len); try testing.expect(r.window_first > r.start()); try testing.expect(r.window_end() < r.end()); const before = coll.dead_unlocated; // The head, one whole window, and the tail. coll.mark_dead(r.start(), r.window_first - r.start()); coll.mark_dead(r.window_first, pgr.map_align); coll.mark_dead(r.window_end(), r.end() - r.window_end()); try testing.expectEqual(@as(u64, pgr.map_align), coll.dead_located()); try testing.expectEqual( before + (r.window_first - r.start()) + (r.end() - r.window_end()), coll.dead_unlocated, ); // The whole window is full and the one beside it untouched: the head and // tail bytes did not leak into a counter that would hand a window back. try testing.expectEqual(@as(WindowDead, pgr.map_align), r.dead[0]); try testing.expectEqual(@as(WindowDead, 0), r.dead[1]); } test "a slab run holds whole documents" { // `mark_dead` charges a document to the run holding its first byte and // asserts the rest is in the same run. That is only sound because // `slab_reserve` never lets an append cross `slab_end` -- so check it // against a collection that owns several runs, including one taken for a // single oversized document. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = std.math.maxInt(u64); try engine.lock(); defer engine.unlock(); var small = try make_padded(gpa, 1, 1000); defer small.deinit(); try engine.insert("app", "c", &small, &env.gen); // Larger than the standard 8 MiB extent, so it gets a run of its own and // the next document forces a third. var huge = try make_padded(gpa, 2, 9 * 1024 * 1024); defer huge.deinit(); try engine.insert("app", "c", &huge, &env.gen); var after = try make_padded(gpa, 3, 1000); defer after.deinit(); try engine.insert("app", "c", &after, &env.gen); try engine.commit(); const coll = engine.get_collection("app", "c").?; try testing.expect(coll.slab_runs.items.len >= 2); // Sorted by page number, and non-overlapping. for (coll.slab_runs.items[1..], 0..) |r, k| { const prev = coll.slab_runs.items[k]; try testing.expect(prev.first + prev.pages <= r.first); } var it = coll.id_index.iter(); while (it.next()) |entry| { const ri = coll.run_of(entry.off) orelse return error.TestUnexpectedResult; const r = coll.slab_runs.items[ri]; try testing.expect(entry.off + coll.doc_bytes(entry.off).len <= r.end()); } } test "a slab run recycled to a lower address keeps the list sorted" { // Runs used to be held in allocation order, which was fine while an extent // could only be appended. Reclamation makes a recycled run arrive at an // address *below* one the collection already owns, and `run_of` is a binary // search -- so the list has to be ordered by page, not by age. // // Exercised on the structure directly: producing a lower-addressed // allocation through the engine needs the free list to be primed, which is // 3.3's business, and this invariant should hold before then. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, env.io, tmp.path); defer engine.deinit(); try engine.lock(); defer engine.unlock(); var d = try make_doc(gpa, 1, "alice"); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); const coll = engine.get_collection("app", "c").?; // Three more runs, arriving out of order and clear of the one the insert // took. They are never written to, so no pages need to exist. const base: u32 = coll.slab_runs.items[0].first + coll.slab_runs.items[0].pages + 16; try coll.insert_run(gpa, base + 200, 8); try coll.insert_run(gpa, base, 8); try coll.insert_run(gpa, base + 100, 8); try testing.expectEqual(@as(usize, 4), coll.slab_runs.items.len); for (coll.slab_runs.items[1..], 0..) |r, k| { try testing.expect(coll.slab_runs.items[k].first < r.first); } // And every one of them is findable at its own address, which is the point // of the ordering. for ([_]u32{ base, base + 100, base + 200 }) |first| { const off = @as(u64, first) << pgr.page_shift; const ri = coll.run_of(off) orelse return error.TestUnexpectedResult; try testing.expectEqual(first, coll.slab_runs.items[ri].first); } try testing.expect(coll.run_of(@as(u64, base + 8) << pgr.page_shift) == null); } test "a restart forgets where the garbage is, not that there is any" { // The window map is deliberately not persisted: it would be a new catalog // field, a new version, and a second read path, to save re-deriving // something the collection can live without. What must survive is the // *amount*, because that is what arms compaction -- so an open puts the // whole of it into `dead_unlocated` and the identity still holds with every // window counter at zero. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var dead_before: u64 = 0; { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = std.math.maxInt(u64); try engine.lock(); defer engine.unlock(); var i: i32 = 0; while (i < 20) : (i += 1) { var d = try make_padded(gpa, i, 4000); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } i = 0; while (i < 20) : (i += 2) { _ = try engine.remove_by_id("app", "c", .{ .int32 = i }); } try engine.commit(); const coll = engine.get_collection("app", "c").?; try testing.expect(coll.dead_located() > 0); dead_before = coll.slab_used - coll.live_bytes; try engine.checkpoint(); } var engine2 = try Engine.open(gpa, io, tmp.path); defer engine2.deinit(); const coll = engine2.get_collection("app", "c").?; try testing.expectEqual(dead_before, coll.slab_used - coll.live_bytes); try testing.expectEqual(@as(u64, 0), coll.dead_located()); try testing.expectEqual(dead_before, coll.dead_unlocated); try testing.expectEqual(dead_before, dead_total(coll)); // The runs came back too, and in a shape `run_of` can use. var it = coll.id_index.iter(); while (it.next()) |entry| try testing.expect(coll.run_of(entry.off) != null); } /// Whether `page` falls in one of the pager's free-list generations. fn in_extents(list: []const pgr.Extent, page: u32) bool { for (list) |e| if (page >= e.first and page < e.first + e.pages) return true; return false; } test "a slab window with one live document in it is never given back" { // The load-bearing test of window reclamation. A window goes back when its // counter reaches `map_align`, which is a statement about *bytes*, not // about documents -- so the thing that must never happen is a window handed // to the pager while a document still sits in it. The document would still // read, because a freed page is only on a list, so the failure would be // silent until the pages were handed out again and overwritten. // // Mutation check: relax the fullness test in `reclaim_windows` to // `r.dead[i] + 2048 < pgr.map_align`, so a window 2 KiB short of empty // qualifies. On its own that aborts the whole suite on the append-cursor // assertion instead -- the appender's own window is the first thing a // loosened test reaches, which is worth knowing. Drop that assertion too // and this is the test that goes red, alone. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = std.math.maxInt(u64); // no rebuild may intervene try engine.lock(); defer engine.unlock(); var i: i32 = 0; while (i < 200) : (i += 1) { var d = try make_padded(gpa, i, 2000); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } try engine.commit(); const coll = engine.get_collection("app", "c").?; // Everything dies but one document, roughly in the middle of the slab. i = 0; while (i < 200) : (i += 1) { if (i == 100) continue; try testing.expect(try engine.remove_by_id("app", "c", .{ .int32 = i })); } try engine.commit(); const survivor_enc = try id_key_for(gpa, bson.Value{ .int32 = 100 }); defer gpa.free(survivor_enc); const survivor = coll.id_index.lookup_exact(survivor_enc).?; try engine.checkpoint(); // Most of the slab went back... try testing.expect(coll.reclaimed_bytes > 100 * 2000); // ...but not the window the survivor is in, and it still reads. try testing.expect(coll.run_of(survivor) != null); try testing.expect(std.mem.indexOf(u8, coll.doc_bytes(survivor), "xxxx") != null); // The accounting followed the pages: what is left is what is left. try testing.expectEqual(coll.slab_used - coll.live_bytes, engine.dead_bytes); try testing.expectEqual( coll.slab_used - coll.live_bytes, coll.dead_located() + coll.dead_unlocated, ); // And once the survivor is gone, its window goes too. try testing.expect(try engine.remove_by_id("app", "c", .{ .int32 = 100 })); try engine.commit(); try engine.checkpoint(); try testing.expect(coll.run_of(survivor) == null); } test "a reclaimed slab window is not reusable until two publishes later" { // Reclamation hands pages to `free_pages`, which withholds them for two // generations -- and it has to, because the image one generation back is // still the fallback a crash would open, and its catalog still claims them. // Handing them straight out would let a write land on pages the recovery // path is about to read as documents. // // Asserted on the pager's own lists rather than on `free_ready_pages()`, // whose total also moves for copy-on-write victims and the catalog stream. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = std.math.maxInt(u64); try engine.lock(); defer engine.unlock(); var i: i32 = 0; while (i < 200) : (i += 1) { var d = try make_padded(gpa, i, 2000); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } try engine.commit(); const coll = engine.get_collection("app", "c").?; const owned_before = coll.slab_runs.items[0]; i = 0; while (i < 200) : (i += 1) _ = try engine.remove_by_id("app", "c", .{ .int32 = i }); try engine.commit(); try engine.checkpoint(); try testing.expect(coll.reclaimed_bytes > 0); // A page from the first window given back: the run's first window is all // dead now, so its first page is no longer the collection's. const gone = @as(u32, @intCast(owned_before.window_first >> pgr.page_shift)); try testing.expect(coll.run_of(@as(u64, gone) << pgr.page_shift) == null); try testing.expect(gone >= owned_before.first); // One publish has happened, so it is held, not ready. try testing.expect(!in_extents(engine.pager.free_ready.items, gone)); try testing.expect(in_extents(engine.pager.free_hold.items, gone)); // The second publish is what makes it allocatable. try engine.checkpoint(); try testing.expect(in_extents(engine.pager.free_ready.items, gone)); } test "a churning collection reuses its slab instead of growing the file" { // The one that decides whether any of this was worth doing. Reclamation can // be working perfectly -- windows counted, pages handed back, // `reclaimed_bytes` climbing -- and the file still grow by the whole write // volume, because nothing asks for the pages in the shape they come back // in. That is what `alloc_slab_run` is for, and this is what says so. // // Delete-and-refill in rounds, with checkpoints per round so reclamation // gets to run and what it frees becomes allocatable. The first round has to // grow the file; the ones after it must not. // // Measured here: 2068 pages after the first round, 2083 after three more of // the same volume -- 15 pages of growth against 4800 pages written. // // Mutation check: raise `slab_run_min_pages` to a whole extent, so no // reclaimed run is ever long enough to qualify and every extent request // bumps the tail. Red -- which is also the measurement that says the floor // has to stay well below an extent to be any use. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = std.math.maxInt(u64); // reuse, not rebuild try engine.lock(); defer engine.unlock(); // Documents a whole window wide, which is the case the design is built for // -- see the note on small documents in the churn gate. const doc_size = pgr.map_align; const per_round = 400; var round: i32 = 0; var tail_after_first: u32 = 0; while (round < 4) : (round += 1) { var i: i32 = 0; while (i < per_round) : (i += 1) { var d = try make_padded(gpa, round * per_round + i, doc_size); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } try engine.commit(); i = 0; while (i < per_round) : (i += 1) { _ = try engine.remove_by_id("app", "c", .{ .int32 = round * per_round + i }); } try engine.commit(); // Two, so what this round freed is allocatable in the next one. try engine.checkpoint(); try engine.checkpoint(); if (round == 0) tail_after_first = engine.pager.alloc_tail; } const coll = engine.get_collection("app", "c").?; try testing.expect(coll.reclaimed_bytes > 0); // Three more rounds of the same volume after the first. Anything left is // fragmentation the windows could not cover, not the write volume. const grew = engine.pager.alloc_tail - tail_after_first; try testing.expect(grew < 3 * per_round * doc_size / pgr.page_size / 4); } test "a rebuild copies only the collections that have garbage" { // `compact` walked every collection unconditionally, so garbage in one paid // for a full copy of all the others -- and a copy is not free even when it // reclaims nothing: it rewrites every document and every index, and it // bumps the layout epoch, which kills every open cursor on a collection // that had no reason to be touched. // // The epoch is the observable, and it is also the user-visible harm: the // clean collection's cursors survive. // // Mutation check: delete the `wants_rebuild` guard. Red -- the clean // collection's epoch moves too. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = std.math.maxInt(u64); // rebuild only when told to try engine.lock(); var i: i32 = 0; while (i < 40) : (i += 1) { var d = try make_padded(gpa, i, 3000); defer d.deinit(); try engine.insert("app", "dirty", &d, &env.gen); var c = try make_padded(gpa, i, 3000); defer c.deinit(); try engine.insert("app", "clean", &c, &env.gen); } // Only one of them loses anything. i = 0; while (i < 40) : (i += 2) _ = try engine.remove_by_id("app", "dirty", .{ .int32 = i }); try engine.commit(); const dirty = engine.get_collection("app", "dirty").?; const clean = engine.get_collection("app", "clean").?; const dirty_epoch = dirty.layout_epoch; const clean_epoch = clean.layout_epoch; const clean_used = clean.slab_used; engine.unlock(); try engine.compact(); try testing.expect(dirty.layout_epoch != dirty_epoch); try testing.expectEqual(clean_epoch, clean.layout_epoch); // And it was not rewritten: a repack starts the slab over, so its byte // count would not survive one unchanged. try testing.expectEqual(clean_used, clean.slab_used); try testing.expectEqual(@as(u64, 0), clean.slab_used - clean.live_bytes); } test "a replace that changes nothing is not a write" { // Mutation check: delete the byte comparison in `upsert`'s `.replace` arm. // Red on all three: the log grows, the document is superseded so the engine // counts garbage that does not exist, and `replace` claims `.modified` -- // which is what `nModified` reports to the client. // // MongoDB counts a document as modified only if the update altered it, and // writes no oplog entry when it did not. `$set: {x: 11}` on a document // already holding `x: 11` is matched and not modified. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = std.math.maxInt(u64); try engine.lock(); defer engine.unlock(); var d = try make_doc(gpa, 1, "alice"); defer d.deinit(); try engine.insert("app", "users", &d, &env.gen); try engine.commit(); const log_after_insert = engine.log.data_bytes; // The same document again, byte for byte. var same = try make_doc(gpa, 1, "alice"); defer same.deinit(); try testing.expectEqual(Engine.Written.unchanged, try engine.replace("app", "users", &same, &env.gen)); try engine.commit(); try testing.expectEqual(log_after_insert, engine.log.data_bytes); try testing.expectEqual(@as(u64, 0), engine.dead_docs); try testing.expectEqual(@as(u64, 0), engine.dead_bytes); try testing.expectEqual(@as(u64, 1), engine.live_docs); // A different one is a write, and is reported as one. var changed = try make_doc(gpa, 1, "bob"); defer changed.deinit(); try testing.expectEqual(Engine.Written.modified, try engine.replace("app", "users", &changed, &env.gen)); try engine.commit(); try testing.expect(engine.log.data_bytes > log_after_insert); try testing.expectEqual(@as(u64, 1), engine.dead_docs); try testing.expectEqual(@as(u64, 1), engine.live_docs); // And the skipped write left the document readable and correctly indexed. const id_enc = try id_key_for(gpa, bson.Value{ .int32 = 1 }); defer gpa.free(id_enc); const coll = engine.get_collection("app", "users").?; const off = coll.id_index.lookup_exact(id_enc).?; const stored = try bson.get_at(gpa, coll.doc_bytes(off), "name"); try testing.expectEqualStrings("bob", stored.?.string); } test "compaction still triggers after a checkpoint has truncated the log" { // Mutation: gate `note_compact` on `self.log.data_bytes` (what it read // before the checkpoint existed) instead of `self.dead_bytes`. Red, because // `truncate_to_header` zeroes that counter at every checkpoint -- the trigger // then never fires and the doc slab grows without bound. This test exists // because the churn gate measured exactly that: 4.1x live data. // // Second mutation: drop the `dead_bytes` accumulation from `read_catalog`. // Red in "reopen carries the garbage counter across a restart", which is // where the counter has to survive a restart. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); // 200 documents of a few dozen bytes each, so the volume gate has to be // small enough for their garbage to clear it. engine.compact_threshold = 1024; try engine.lock(); defer engine.unlock(); for (0..200) |i| { var d = try make_doc(gpa, @intCast(i), "x"); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } try engine.commit(); // A checkpoint, which is what truncates the log. From here on the log says // nothing about how much garbage the database holds. try engine.checkpoint(); try testing.expectEqual(@as(u64, 0), engine.log.data_bytes); const live_after_load = engine.live_bytes; try testing.expect(live_after_load > 0); // Now make garbage. Every replace supersedes a document, so its slab bytes // are dead: the data file holds them and only a rebuild reclaims them. _ = engine.take_compact(); for (0..200) |i| { var d = try make_doc(gpa, @intCast(i), "yy"); defer d.deinit(); _ = try engine.replace("app", "c", &d, &env.gen); } try engine.commit(); try testing.expectEqual(live_after_load + 200, engine.live_bytes); // Every superseded document, plus what the first replace after the // checkpoint had to skip: that checkpoint froze the page the append cursor // pointed into, so the cursor moved up to the next system page and the gap // it stepped over is garbage too. Both are in the same total, which is the // point -- the trigger reads one number. const superseded = live_after_load; try testing.expect(engine.dead_bytes >= superseded); try testing.expect(engine.dead_bytes - superseded < pgr.map_align); try testing.expect(engine.dead_bytes >= engine.compact_threshold); try testing.expect(engine.take_compact()); // And the rebuild actually clears the garbage it was called for. try engine.compact(); try testing.expectEqual(@as(u64, 0), engine.dead_bytes); try testing.expectEqual(@as(u64, 200), engine.live_docs); // The engine's live total is the sum over collections, both after a rebuild // and after the catalog round trip below. const coll = engine.get_collection("app", "c").?; try testing.expectEqual(engine.live_bytes, coll.live_bytes); try testing.expectEqual(coll.slab_used, coll.live_bytes); } test "a rebuild leaves the space it reclaimed ready to reuse" { // Mutation check: delete the second `checkpoint()` at the end of `compact`. // Red -- one publish only moves the abandoned extents from `pending` to // `hold`, so nothing is reusable and the next rebuild grows the file instead. // Measured on the churn gate as ~1.2x of extra steady-state size (3.58x live // data against 2.47x) under sustained update churn. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = 1024; try engine.lock(); defer engine.unlock(); for (0..300) |i| { var d = try make_doc(gpa, @intCast(i), "x"); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } for (0..300) |i| { var d = try make_doc(gpa, @intCast(i), "yy"); defer d.deinit(); _ = try engine.replace("app", "c", &d, &env.gen); } try engine.commit(); try testing.expect(engine.take_compact()); try engine.compact(); // The rebuild abandoned the old slab extent and every page the old trees // occupied. Those must be handed back, not merely queued. try testing.expect(engine.pager.free_ready_pages() > 0); // And the next allocation actually uses them rather than the tail. const tail_before = engine.pager.alloc_tail; for (0..300) |i| { var d = try make_doc(gpa, @intCast(i), "zzz"); defer d.deinit(); _ = try engine.replace("app", "c", &d, &env.gen); } try engine.commit(); try testing.expect(engine.pager.alloc_tail < tail_before + engine.pager.free_ready_pages() + 64); try testing.expectEqual(@as(u64, 300), engine.live_docs); } test "reopen carries the garbage counter across a restart" { // It carries it by *recomputing* it: `read_catalog` sums // `slab_used - live_bytes` over the collections the catalog lists, rather // than trusting the watermark's cached copy. That is a stronger claim than // the hint was -- a collection dropped since the last checkpoint is simply // not in the sum, where the hint kept charging the engine for it. // // Mutation: drop the accumulation in `read_catalog`; `engine2.dead_bytes` // reads zero below. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var dead_before: u64 = 0; var live_before: u64 = 0; { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = std.math.maxInt(u64); // never rebuild here try engine.lock(); for (0..100) |i| { var d = try make_doc(gpa, @intCast(i), "x"); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } for (0..50) |i| { var d = try make_doc(gpa, @intCast(i), "yy"); defer d.deinit(); _ = try engine.replace("app", "c", &d, &env.gen); } try engine.commit(); try engine.checkpoint(); dead_before = engine.dead_bytes; live_before = engine.live_bytes; engine.unlock(); try testing.expect(dead_before > 0); } var engine2 = try Engine.open(gpa, io, tmp.path); defer engine2.deinit(); try testing.expectEqual(dead_before, engine2.dead_bytes); try testing.expectEqual(live_before, engine2.live_bytes); // And it is the sum over collections on both sides of the restart, not a // number kept beside them. const reopened = engine2.get_collection("app", "c").?; try testing.expectEqual(reopened.slab_used - reopened.live_bytes, engine2.dead_bytes); } test "the slab counts what the appender skips" { // `slab_used` only ever grew by a document's length, so the two places the // appender writes slab off went uncounted: the gap left when a checkpoint // freezes the page the cursor points into and the cursor moves up to the // next system page, and the tail of an extent abandoned for a document that // no longer fits. Real garbage -- only a rebuild gets it back -- and // invisible to the trigger that decides whether a rebuild is worth doing. // // Mutation: drop either `note_skip` call in `slab_reserve`. Red on the // matching half below. Dropping the one in `slab_append` is not covered // here: it needs a publish between the reservation and the append, which is // what "a checkpoint runs alongside writers on several collections" // arranges, and the accounting assertion in `checkpoint` is what catches it. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = std.math.maxInt(u64); // no rebuild mid-test try engine.lock(); defer engine.unlock(); for (0..200) |i| { var d = try make_doc(gpa, @intCast(i), "x"); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } try engine.commit(); const coll = engine.get_collection("app", "c").?; // Nothing is dead yet: every document inserted is still live and the slab // has been walked straight through. try testing.expectEqual(@as(u64, 0), engine.dead_bytes); try testing.expectEqual(coll.slab_used, coll.live_bytes); // 1. The round-up. The checkpoint freezes the page the cursor is in, so the // next append resumes at the next system page and the bytes in between // are never written. try engine.checkpoint(); const tail_before = coll.slab_tail; const gap = std.mem.alignForward(u64, tail_before, pgr.map_align) - tail_before; try testing.expect(gap > 0); { var d = try make_doc(gpa, 1000, "x"); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } try engine.commit(); try testing.expectEqual(gap, engine.dead_bytes); try testing.expectEqual(coll.slab_used - coll.live_bytes, engine.dead_bytes); // 2. The abandoned tail. A document bigger than the standard extent takes // one of its own, and everything left in the extent being walked away // from is unreachable -- the extent stays allocated to this collection. const abandoned = coll.slab_end - coll.slab_tail; try testing.expect(abandoned > 4 * 1024 * 1024); { const big = try gpa.alloc(u8, 9 * 1024 * 1024); defer gpa.free(big); @memset(big, 'z'); var d = try make_doc(gpa, 1001, big); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } try engine.commit(); try testing.expectEqual(gap + abandoned, engine.dead_bytes); try testing.expectEqual(coll.slab_used - coll.live_bytes, engine.dead_bytes); // And the next checkpoint gives most of it straight back. An abandoned // extent tail is whole windows with nothing live in them, which is exactly // what window reclamation is for -- so counting it was not bookkeeping for // its own sake, it is what made this reclaimable at all. // // What stays is the edges: the round-up gap, which shares its window with // the live documents below it, and the bytes of the run outside any whole // window. try engine.checkpoint(); try testing.expect(coll.reclaimed_bytes > 4 * 1024 * 1024); try testing.expectEqual(gap + abandoned - coll.reclaimed_bytes, engine.dead_bytes); try testing.expectEqual(coll.slab_used - coll.live_bytes, engine.dead_bytes); try testing.expect(engine.dead_bytes < gap + 2 * pgr.map_align); } test "a rebuild leaves behind what its own copying skipped" { // `compact` used to set `dead_bytes = 0`, on the reasoning that a repack // holds only live bytes. It does not: the repack appends through the same // slab, so it abandons an extent tail whenever the next document no longer // fits. Zeroing the counter there broke the identity the checkpoint asserts // -- and understated the garbage, so a collection that fragments on every // rebuild would never be rebuilt again. // // Mutation: put `self.dead_bytes = 0;` back. Red below, and the checkpoint // inside `compact` aborts on the accounting assertion before it gets there. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = std.math.maxInt(u64); // rebuild only when told to try engine.lock(); defer engine.unlock(); // Documents big enough that two of them do not fit in one 8 MiB extent, so // the copying itself has to abandon a tail. const big = try gpa.alloc(u8, 5 * 1024 * 1024); defer gpa.free(big); @memset(big, 'z'); for (0..3) |i| { var d = try make_doc(gpa, @intCast(i), big); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } try engine.commit(); // One of them becomes garbage, which is what the rebuild is for. try testing.expect(try engine.remove_by_id("app", "c", .{ .int32 = 1 })); try engine.commit(); try engine.compact(); const coll = engine.get_collection("app", "c").?; try testing.expectEqual(@as(u64, 2), coll.doc_count); // The deleted document is gone from the slab, and the gap the copy left // between the two survivors is now mostly gone too -- `compact` ends in a // checkpoint, and a checkpoint reclaims whole windows. What survives is the // edges of that gap, which is a smaller number than this test used to // assert but the same statement: the counter is *not* zeroed, and it equals // what the collection actually has. try testing.expectEqual(coll.slab_used - coll.live_bytes, engine.dead_bytes); try testing.expect(coll.reclaimed_bytes > 2 * 1024 * 1024); try testing.expect(engine.dead_bytes > 0); try testing.expect(engine.dead_bytes < 4 * pgr.map_align); } test "dropping a collection does not arm compaction" { // `free_collection` charged the engine's `dead_bytes` with the dropped // collection's *live* bytes, having just handed every page it owned back to // the pager on the following line. A drop of a large collection therefore // asked for a rebuild -- a full copy of every collection that was left -- // to reclaim space that had already been reclaimed. Its own garbage was // wrong the other way: it stayed on the engine's books after the pages // holding it were gone. // // Mutation: restore `self.dead_bytes += coll.live_bytes;`, or delete the // subtraction of `coll.slab_used - coll.live_bytes`. Either one is red on // the equality below, and the first also re-arms the trigger. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = std.math.maxInt(u64); // no rebuild while setting up try engine.lock(); defer engine.unlock(); // A small collection that survives, and a large one that does not. Both // hold garbage, so the drop has to keep one collection's and discard the // other's. for (0..40) |i| { var d = try make_doc(gpa, @intCast(i), "x"); defer d.deinit(); try engine.insert("app", "keep", &d, &env.gen); } for (0..200) |i| { var d = try make_doc(gpa, @intCast(i), "x"); defer d.deinit(); try engine.insert("app", "gone", &d, &env.gen); } for (0..10) |i| { var d = try make_doc(gpa, @intCast(i), "yy"); defer d.deinit(); _ = try engine.replace("app", "keep", &d, &env.gen); } for (0..200) |i| { var d = try make_doc(gpa, @intCast(i), "yy"); defer d.deinit(); _ = try engine.replace("app", "gone", &d, &env.gen); } try engine.commit(); const keep = engine.get_collection("app", "keep").?; const keep_dead = keep.slab_used - keep.live_bytes; const keep_live = keep.live_bytes; try testing.expect(keep_dead > 0); try testing.expect(engine.dead_bytes > keep_dead); // Arm the trigger for what the two of them hold between them. engine.compact_threshold = keep_dead + 1; engine.note_compact(); try testing.expect(engine.take_compact()); try testing.expect(try engine.drop_collection("app", "gone")); try testing.expectEqual(keep_dead, engine.dead_bytes); try testing.expectEqual(keep_live, engine.live_bytes); // The next write reconsiders the trigger and finds nothing worth a rebuild: // what the drop reclaimed is not garbage, it is free. engine.note_compact(); try testing.expect(!engine.take_compact()); // The catalog agrees, which is what the reopened engine will read. try engine.checkpoint(); try testing.expectEqual(keep_dead, engine.dead_bytes); } test "reopen replays log" { var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); var d1 = try make_doc(gpa, 1, "alice"); defer d1.deinit(); var d2 = try make_doc(gpa, 2, "bob"); defer d2.deinit(); try engine.lock(); try engine.insert("app", "users", &d1, &env.gen); try engine.insert("app", "users", &d2, &env.gen); _ = try engine.remove_by_id("app", "users", .{ .int32 = 2 }); engine.unlock(); } var engine2 = try Engine.open(gpa, io, tmp.path); defer engine2.deinit(); try engine2.lock(); const id_key = try id_key_for(gpa, bson.Value{ .int32 = 2 }); defer gpa.free(id_key); try testing.expect(engine2.get_doc("app", "users", id_key) == null); const id_key1 = try id_key_for(gpa, bson.Value{ .int32 = 1 }); defer gpa.free(id_key1); try testing.expectEqualStrings("alice", (try bson.get_at(gpa, engine2.get_doc("app", "users", id_key1).?, "name")).?.string); engine2.unlock(); } test "auto _id generation survives reopen" { var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); var doc = try make_doc(gpa, 0, "no-id-here"); defer doc.deinit(); // strip _id const stripped = doc.pairs[1..]; var arena = std.heap.ArenaAllocator.init(gpa); defer arena.deinit(); var d2 = try bson.Document.alloc(gpa, try arena.allocator().dupe(bson.Pair, stripped)); defer d2.deinit(); try engine.lock(); try engine.insert("app", "no_ids", &d2, &env.gen); engine.unlock(); } var engine2 = try Engine.open(gpa, io, tmp.path); defer engine2.deinit(); const coll = engine2.get_collection("app", "no_ids").?; var it = coll.id_index.iter(); var count: usize = 0; while (it.next()) |entry| { count += 1; const b = coll.doc_bytes(entry.off); try testing.expect((try bson.get_at(gpa, b, "_id")).?.object_id.len == 12); } try testing.expectEqual(@as(usize, 1), count); } test "compaction rewrites log and keeps data" { var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); { var engine = try Engine.open(gpa, io, tmp.path); engine.compact_threshold = 1; // always compact defer engine.deinit(); var docs: [4]bson.Document = undefined; defer for (&docs) |*d| d.deinit(); try engine.lock(); for (0..4) |i| { docs[i] = try make_doc(gpa, @intCast(i + 1), "user-{d}"); try engine.insert("app", "users", &docs[i], &env.gen); } try engine.commit(); // threshold 1 makes every write want a compaction; run it. if (engine.take_compact()) try engine.compact(); engine.unlock(); } // Reopen after compaction and keep writing: with the log reopened at // end_pos 0, appends would clobber the compacted records. var engine2 = try Engine.open(gpa, io, tmp.path); defer engine2.deinit(); try engine2.lock(); var extra = try make_doc(gpa, 5, "eve"); defer extra.deinit(); try engine2.insert("app", "users", &extra, &env.gen); try engine2.commit(); engine2.unlock(); var engine3 = try Engine.open(gpa, io, tmp.path); defer engine3.deinit(); try engine3.lock(); for (1..6) |i| { const id_key = try id_key_for(gpa, bson.Value{ .int32 = @intCast(i) }); defer gpa.free(id_key); try testing.expect(engine3.get_doc("app", "users", id_key) != null); } engine3.unlock(); } test "a checkpoint runs alongside writers on several collections" { // `write_catalog` reads each collection's slab extents, indexes and byte // counters while holding only the *shared catalog* lock -- and a writer // holds that same lock shared, taking the collection's lock exclusively. // So the snapshot walked structures its owner was free to mutate, and // `slab_runs` is an ArrayList a new extent inserts into: a reallocation // mid-walk leaves the serializer reading freed memory. // // Several collections rather than one, because the interesting overlap is a // writer on collection B while the catalog is serializing collection A. // // Mutation check: drop the `lockShared` from `write_catalog`'s collection // loop. Not reliably red -- a data race never is -- but it runs under // ReleaseSafe, where the reads it makes are bounds-checked. const gpa = testing.allocator; var threaded: std.Io.Threaded = .init(gpa, .{}); defer threaded.deinit(); const io = threaded.io(); var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); const colls = [_][]const u8{ "a", "b", "c", "d" }; const per_coll: i32 = 150; var done = std.atomic.Value(usize).init(colls.len); const Worker = struct { fn writer( e: *Engine, name: []const u8, left: *std.atomic.Value(usize), alloc: std.mem.Allocator, ) error{Canceled}!void { defer _ = left.fetchSub(1, .release); for (1..per_coll + 1) |i| { var doc = make_doc(alloc, @intCast(i), "user") catch return error.Canceled; defer doc.deinit(); { e.lock_catalog(false) catch return error.Canceled; defer e.unlock_catalog(false); const coll = (e.lock_collection("app", name, true, true) catch return error.Canceled) orelse return error.Canceled; defer e.unlock_collection(coll, true); e.insert("app", name, &doc, undefined) catch return error.Canceled; } // As the dispatch epilogue does (commands.zig): the append bumps // `seq`, the commit is what makes it durable, and a checkpoint // may only describe what is durable. e.commit() catch return error.Canceled; } } fn checkpointer(e: *Engine, left: *std.atomic.Value(usize)) error{Canceled}!void { while (left.load(.acquire) > 0) { // Errors are the point of the retry loop inside `checkpoint`, // not a failure of this test; a checkpoint that gives up under // sustained writes has still not corrupted anything. e.checkpoint() catch {}; } } }; var group: std.Io.Group = .init; defer group.cancel(io); for (colls) |name| group.async(io, Worker.writer, .{ &engine, name, &done, gpa }); group.async(io, Worker.checkpointer, .{ &engine, &done }); try group.await(io); // Every write is still there, and the catalog the checkpoints wrote agrees // with the engine -- the second half is what `write_catalog`'s own assertion // checks on the way through. try engine.checkpoint(); try engine.lock_read(); defer engine.unlock_read(); var live_sum: u64 = 0; var dead_sum: u64 = 0; var docs_sum: u64 = 0; for (colls) |name| { const coll = engine.get_collection("app", name) orelse return error.TestUnexpectedResult; try testing.expectEqual(@as(usize, @intCast(per_coll)), coll.id_index.count()); live_sum += coll.live_bytes; dead_sum += coll.slab_used - coll.live_bytes; docs_sum += coll.doc_count; } // The engine's counters are the sums over collections, checked once // everything is quiet rather than left to `checkpoint`'s own assertion -- // which only runs on a checkpoint that happened to fall in a gap between // writes, so under sustained load it can go a whole run without firing. // Every writer here held nothing but its own collection's lock while moving // these, so a lost update lands exactly as a mismatch on one of these three. try testing.expectEqual(live_sum, engine.live_bytes); try testing.expectEqual(dead_sum, engine.dead_bytes); try testing.expectEqual(docs_sum, engine.live_docs); } test "concurrent readers and writers on a threaded Io" { // Real worker threads: writers hold the exclusive lock, readers the // shared lock. Proves the RwLock split keeps committed writes visible // to concurrent readers and never corrupts the maps. const gpa = testing.allocator; var threaded: std.Io.Threaded = .init(gpa, .{}); defer threaded.deinit(); const io = threaded.io(); var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); const writers = 4; const readers = 4; const per_writer: i32 = 200; const total: i32 = writers * per_writer; var next_id = std.atomic.Value(i32).init(1); var remaining = std.atomic.Value(usize).init(@intCast(total)); const Worker = struct { fn writer( e: *Engine, id_counter: *std.atomic.Value(i32), pending: *std.atomic.Value(usize), alloc: std.mem.Allocator, ) error{Canceled}!void { while (true) { const id = id_counter.fetchAdd(1, .monotonic); if (id > total) return; var doc = make_doc(alloc, id, "user") catch return error.Canceled; defer doc.deinit(); e.lock() catch return error.Canceled; defer e.unlock(); e.insert("app", "users", &doc, undefined) catch return error.Canceled; _ = pending.fetchSub(1, .monotonic); } } fn reader(e: *Engine, pending: *std.atomic.Value(usize)) error{Canceled}!void { while (pending.load(.acquire) > 0) { e.lock_read() catch return error.Canceled; defer e.unlock_read(); if (e.get_collection("app", "users")) |coll| { var n: usize = 0; var it = coll.id_index.iter(); while (it.next()) |_| n += 1; // A reader must never observe more docs than can exist. if (n > @as(usize, @intCast(total))) return error.Canceled; } } } }; var group: std.Io.Group = .init; defer group.cancel(io); for (0..readers) |_| group.async(io, Worker.reader, .{ &engine, &remaining }); for (0..writers) |_| group.async(io, Worker.writer, .{ &engine, &next_id, &remaining, gpa }); try group.await(io); // Every committed write must be visible once all writers finish. try engine.lock_read(); defer engine.unlock_read(); const coll = engine.get_collection("app", "users") orelse return error.TestUnexpectedResult; try testing.expectEqual(@as(usize, @intCast(total)), coll.id_index.count()); for (1..total + 1) |i| { const id_key = try id_key_for(gpa, bson.Value{ .int32 = @intCast(i) }); defer gpa.free(id_key); try testing.expect(engine.get_doc("app", "users", id_key) != null); } } test "compact yields to a compaction already in flight" { // The guard's contract, checked deterministically. Two compactions at once // share one tmp path and each ends in a rename onto the log, so the second // truncates and rewrites the file the first is about to publish -- and the // first then renames whatever the second left there over the live log. // // The real interleaving is hard to force: `compact_snapshot_coll` holds each // collection's write lock while writing its snapshot, so two compactions // serialize there, and an insert cannot re-arm `compact_pending` while that // lock is held either. The overlap window is only between the end of one // snapshot and its rename. Rather than race for it, drive the flag directly // and pin what the guard promises. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = 1; // Leave real garbage behind, so a compaction that ran would be visible: // `compact` resets dead_docs to zero and nothing else does. try engine.lock(); var doc = try make_doc(gpa, 1, "alice"); defer doc.deinit(); try engine.insert("app", "users", &doc, &env.gen); var doc2 = try make_doc(gpa, 1, "alice-again"); defer doc2.deinit(); // A replace supersedes the first record, leaving it behind as garbage. _ = try engine.replace("app", "users", &doc2, &env.gen); try engine.commit(); engine.unlock(); try testing.expect(engine.dead_docs > 0); const dead_before = engine.dead_docs; // With a compaction "in flight", compact must return without rewriting. engine.compacting.store(true, .release); try engine.compact(); try testing.expectEqual(dead_before, engine.dead_docs); // With the slot free, the same call does the work -- proving the assertion // above came from the guard and not from there being nothing to do. engine.compacting.store(false, .release); try engine.compact(); try testing.expectEqual(@as(u64, 0), engine.dead_docs); } test "concurrent writers compacting: the log survives a reopen" { // Real worker threads driving compaction while other writers append, each // following the lock sequence the server's dispatch uses (catalog -> // collection -> release both -> commit -> compact). Every other compaction // test is single-threaded, so this is the only coverage of the whole write // path under genuine contention. // // What it proves: concurrent compaction leaves a log that replays to // exactly the right documents. The count is checked exactly -- too few // means a rewrite was published half-written, too many means a stale tmp // tail was replayed as live data. // // What it does not prove: that either specific race is fixed. Both windows // are too narrow to hit reliably (see the test above), and this test passes // with the `compacting` guard removed. It is a smoke test for the path, not // a regression test for the guard; the deterministic tests above and in // storage.zig are what pin those two invariants. const gpa = testing.allocator; var threaded: std.Io.Threaded = .init(gpa, .{}); defer threaded.deinit(); const io = threaded.io(); var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); const writers = 4; const per_writer: i32 = 60; const total: i32 = writers * per_writer; { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); // Every write wants a compaction, so writers pile into compact() with // maximum overlap -- the point of the test. engine.compact_threshold = 1; var next_id = std.atomic.Value(i32).init(1); const Worker = struct { /// The in-lock half of a write command: the collection lock is /// taken under the catalog lock, and both are released on return -- /// so the caller's commit runs holding neither, exactly as the /// server's dispatch epilogue does. fn insert_locked(e: *Engine, doc: *bson.Document) !void { try e.lock_catalog(false); defer e.unlock_catalog(false); const coll = try e.lock_collection("app", "users", true, true); if (coll) |c| { defer e.unlock_collection(c, true); try e.insert("app", "users", doc, undefined); } } fn writer( e: *Engine, id_counter: *std.atomic.Value(i32), alloc: std.mem.Allocator, ) error{Canceled}!void { while (true) { const id = id_counter.fetchAdd(1, .monotonic); if (id > total) return; var doc = make_doc(alloc, id, "user") catch return error.Canceled; defer doc.deinit(); // Ids come from the shared counter, so no insert here can // legitimately fail; any error is a real defect. insert_locked(e, &doc) catch return error.Canceled; e.commit() catch return error.Canceled; if (e.take_compact()) e.compact() catch return error.Canceled; } } }; var group: std.Io.Group = .init; defer group.cancel(io); for (0..writers) |_| group.async(io, Worker.writer, .{ &engine, &next_id, gpa }); try group.await(io); } // Reopen from disk: this replays the log that compaction left behind, which // is the only place the races above are observable. var reopened = try Engine.open(gpa, io, tmp.path); defer reopened.deinit(); try reopened.lock_read(); defer reopened.unlock_read(); const coll = reopened.get_collection("app", "users") orelse return error.TestUnexpectedResult; try testing.expectEqual(@as(usize, @intCast(total)), coll.id_index.count()); for (1..total + 1) |i| { const id_key = try id_key_for(gpa, bson.Value{ .int32 = @intCast(i) }); defer gpa.free(id_key); try testing.expect(reopened.get_doc("app", "users", id_key) != null); } } // -- index tests ----------------------------------------------------------- /// A spec document for a single-path index, built by serializing and /// re-parsing so the pairs are arena-owned. fn index_spec( gpa: std.mem.Allocator, path: []const u8, name: []const u8, unique: bool, sparse: bool, ttl: ?i64, ) !bson.Document { var out: std.ArrayListUnmanaged(u8) = .empty; defer out.deinit(gpa); var pairs: std.ArrayListUnmanaged(bson.Pair) = .empty; defer pairs.deinit(gpa); try pairs.appendSlice(gpa, &.{ .{ .key = "key", .value = .{ .doc = &.{.{ .key = path, .value = .{ .int32 = 1 } }} } }, .{ .key = "name", .value = .{ .string = name } }, .{ .key = "unique", .value = .{ .bool = unique } }, .{ .key = "sparse", .value = .{ .bool = sparse } }, }); if (ttl) |secs| try pairs.append(gpa, .{ .key = "expireAfterSeconds", .value = .{ .int64 = secs } }); try bson.write_doc(pairs.items, gpa, &out); return bson.Document.parse(gpa, out.items); } /// Number of entries the named index has for a single-value equality key. fn index_count( gpa: std.mem.Allocator, engine: *Engine, db_name: []const u8, coll_name: []const u8, name: []const u8, key_value: bson.Value, ) !usize { const coll = engine.get_collection(db_name, coll_name) orelse return 0; for (coll.indexes.items) |ix| { if (std.mem.eql(u8, ix.name, name)) { var out: std.ArrayListUnmanaged(u64) = .empty; defer out.deinit(gpa); try ix.lookup_eq(gpa, &.{key_value}, &out); return out.items.len; } } return 0; } fn make_keyed(gpa: std.mem.Allocator, id: i32, k: i32) !bson.Document { var arena = std.heap.ArenaAllocator.init(gpa); errdefer arena.deinit(); const pairs = try arena.allocator().alloc(bson.Pair, 2); pairs[0] = .{ .key = try arena.allocator().dupe(u8, "_id"), .value = .{ .int32 = id } }; pairs[1] = .{ .key = try arena.allocator().dupe(u8, "k"), .value = .{ .int32 = k } }; return .{ .arena = arena, .pairs = pairs }; } fn make_user(gpa: std.mem.Allocator, id: i32, email: []const u8) !bson.Document { var arena = std.heap.ArenaAllocator.init(gpa); errdefer arena.deinit(); const pairs = try arena.allocator().alloc(bson.Pair, 2); pairs[0] = .{ .key = try arena.allocator().dupe(u8, "_id"), .value = .{ .int32 = id } }; pairs[1] = .{ .key = try arena.allocator().dupe(u8, "email"), .value = .{ .string = try arena.allocator().dupe(u8, email) } }; return .{ .arena = arena, .pairs = pairs }; } test "unique index enforced on insert, replace, and upsert-conflict" { var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); var spec = try index_spec(gpa, "email", "email_1", true, false, null); defer spec.deinit(); try engine.lock(); _ = try engine.create_index("app", "users", &spec); var d1 = try make_user(gpa, 1, "a@x.io"); defer d1.deinit(); try engine.insert("app", "users", &d1, &env.gen); // A second doc with the same email is rejected and never logged. var d2 = try make_user(gpa, 2, "a@x.io"); defer d2.deinit(); try testing.expectError(error.DuplicateKeyIndex, engine.insert("app", "users", &d2, &env.gen)); try testing.expectEqualStrings("email_1", engine.get_collection("app", "users").?.dup_index.?); // A replace that keeps its own email is fine (own entries excluded). var d1b = try make_user(gpa, 1, "a@x.io"); defer d1b.deinit(); _ = try engine.replace("app", "users", &d1b, &env.gen); try testing.expectEqual(@as(usize, 1), try index_count(gpa, &engine, "app", "users", "email_1", .{ .string = "a@x.io" })); // An update that would collide is rejected. var d2b = try make_user(gpa, 2, "a@x.io"); defer d2b.deinit(); try testing.expectError(error.DuplicateKeyIndex, engine.replace("app", "users", &d2b, &env.gen)); // A different email still inserts. var d3 = try make_user(gpa, 3, "b@x.io"); defer d3.deinit(); try engine.insert("app", "users", &d3, &env.gen); engine.unlock(); } test "index maintained across update and delete" { var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); var spec = try index_spec(gpa, "a", "a_1", false, false, null); defer spec.deinit(); try engine.lock(); _ = try engine.create_index("app", "items", &spec); var d1 = try doc_with_a(gpa, 1, 10); defer d1.deinit(); var d2 = try doc_with_a(gpa, 2, 20); defer d2.deinit(); try engine.insert("app", "items", &d1, &env.gen); try engine.insert("app", "items", &d2, &env.gen); try testing.expectEqual(@as(usize, 1), try index_count(gpa, &engine, "app", "items", "a_1", .{ .int32 = 20 })); // Replace doc 1 with a new value: old entry gone, new entry present. var d1b = try doc_with_a(gpa, 1, 30); defer d1b.deinit(); _ = try engine.replace("app", "items", &d1b, &env.gen); try testing.expectEqual(@as(usize, 0), try index_count(gpa, &engine, "app", "items", "a_1", .{ .int32 = 10 })); try testing.expectEqual(@as(usize, 1), try index_count(gpa, &engine, "app", "items", "a_1", .{ .int32 = 30 })); // Delete doc 2: its entry is removed. _ = try engine.remove_by_id("app", "items", .{ .int32 = 2 }); try testing.expectEqual(@as(usize, 0), try index_count(gpa, &engine, "app", "items", "a_1", .{ .int32 = 20 })); engine.unlock(); } /// A document with an integer `a` field (on top of _id + name). fn doc_with_a(gpa: std.mem.Allocator, id: i32, a: i32) !bson.Document { var arena = std.heap.ArenaAllocator.init(gpa); errdefer arena.deinit(); const pairs = try arena.allocator().alloc(bson.Pair, 3); pairs[0] = .{ .key = try arena.allocator().dupe(u8, "_id"), .value = .{ .int32 = id } }; pairs[1] = .{ .key = try arena.allocator().dupe(u8, "name"), .value = .{ .string = try arena.allocator().dupe(u8, "x") } }; pairs[2] = .{ .key = try arena.allocator().dupe(u8, "a"), .value = .{ .int32 = a } }; return .{ .arena = arena, .pairs = pairs }; } test "index survives reopen and compaction" { var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); engine.compact_threshold = 1; // every write compacts var spec = try index_spec(gpa, "email", "email_1", false, false, null); defer spec.deinit(); try engine.lock(); _ = try engine.create_index("app", "users", &spec); var d1 = try make_user(gpa, 1, "a@x.io"); defer d1.deinit(); var d2 = try make_user(gpa, 2, "b@x.io"); defer d2.deinit(); try engine.insert("app", "users", &d1, &env.gen); try engine.insert("app", "users", &d2, &env.gen); engine.unlock(); } // Reopen: the index (rebuilt from the compacted log) still finds docs. var engine2 = try Engine.open(gpa, io, tmp.path); defer engine2.deinit(); try engine2.lock(); try testing.expectEqual(@as(usize, 1), try index_count(gpa, &engine2, "app", "users", "email_1", .{ .string = "b@x.io" })); engine2.unlock(); } test "a checkpoint lets the next open skip the log it covers" { // The point of the whole milestone: an open that finds a watermark loads the // data file and replays only what happened after it, instead of rebuilding // everything from the log. // // Mutation checks, red: publishing a watermark seq of 0; and removing the // index maintenance in apply_record, which leaves a replayed document // present in the collection and absent from `_id_` -- which, once the // hashmap goes, means simply absent. // // Not covered, and worth stating rather than implying: removing // `self.seq = @max(self.seq, record.seq)` from apply_record leaves this // green. The sequence is seeded from the watermark on a checkpointed open, // so it only drifts by the records replayed on top -- and every sequence // reachable from here has the catalog carrying those same records, which // masks the drift. Observing it needs a crash between a duplicate-sequence // append and the checkpoint that would have captured it, which is the // crash-injection harness's job, not this test's. The line stays because a // log whose sequences are not monotonic has no total order, and // `committed_seq <= seq` is asserted on every commit. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); const data_path = try std.fmt.allocPrint(gpa, "{s}.data", .{tmp.path}); defer gpa.free(data_path); defer std.Io.Dir.cwd().deleteFile(io, data_path) catch {}; { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); try engine.lock(); var i: i32 = 0; while (i < 40) : (i += 1) { var d = try make_user(gpa, i, "a@x.io"); defer d.deinit(); try engine.insert("app", "users", &d, &env.gen); } engine.unlock(); try engine.checkpoint(); try testing.expect(engine.pager.loaded.generation >= 1); try testing.expectEqual(engine.seq, engine.pager.loaded.seq); // Writes after the checkpoint are the ones a reopen must replay. try engine.lock(); i = 100; while (i < 105) : (i += 1) { var d = try make_user(gpa, i, "b@x.io"); defer d.deinit(); try engine.insert("app", "users", &d, &env.gen); } engine.unlock(); } // Reopen: the checkpoint is loaded, so only the five later records apply. { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); try engine.lock(); defer engine.unlock(); try testing.expect(engine.pager.loaded.generation >= 1); const coll = engine.get_collection("app", "users").?; try testing.expectEqual(@as(usize, 45), coll.id_index.count()); try testing.expectEqual(@as(usize, 45), coll.id_index.count()); // And the sequence continued from the watermark rather than restarting. try testing.expect(engine.seq >= engine.pager.loaded.seq); } // A second reopen, to catch a sequence that restarted: the writes made after // the first reopen must survive it. { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); try engine.lock(); var d = try make_user(gpa, 500, "c@x.io"); defer d.deinit(); try engine.insert("app", "users", &d, &env.gen); engine.unlock(); try engine.checkpoint(); } { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); try engine.lock(); defer engine.unlock(); const coll = engine.get_collection("app", "users").?; try testing.expectEqual(@as(usize, 46), coll.id_index.count()); const id_key = try id_key_for(gpa, .{ .int32 = 500 }); defer gpa.free(id_key); try testing.expect(engine.get_doc("app", "users", id_key) != null); } } test "a checkpoint reclaims the log and the data survives" { // The payoff of a lagging checkpoint: once the data file holds the effect of // a record, the record is redundant and the log can be reclaimed. Without // this the log only ever grows and every open pays for every write ever made. // // Mutation check, red: skipping the truncation. // // Not covered: moving the truncation *before* the publish. That is still // correct in the absence of a crash -- the publish follows immediately -- and // the hazard is precisely a crash landing between the two, with the records // gone from the log and not yet in any image. Catching it needs process-level // crash injection, which the milestone's gates cover; an in-process test // cannot express "stop here and die". The order stays because it is the // whole reason a lagging checkpoint is safe. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); const data_path = try std.fmt.allocPrint(gpa, "{s}.data", .{tmp.path}); defer gpa.free(data_path); defer std.Io.Dir.cwd().deleteFile(io, data_path) catch {}; var log_after_checkpoint: u64 = 0; { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); try engine.lock(); var i: i32 = 0; while (i < 200) : (i += 1) { var d = try make_user(gpa, i, "a@x.io"); defer d.deinit(); try engine.insert("app", "users", &d, &env.gen); } engine.unlock(); // Commit first, so the records are actually on disk: appends buffer in // the log's open block, and only a commit seals and writes it. Measuring // before that reads a file that is still just its header. try engine.commit(); const before = try engine.log.file.length(io); try testing.expect(before > storage.file_header_len); try engine.checkpoint(); log_after_checkpoint = try engine.log.file.length(io); // The log is back to just its header. try testing.expect(log_after_checkpoint < before); try testing.expectEqual(@as(u64, storage.file_header_len), log_after_checkpoint); // And writing still works afterwards, at a sequence above the watermark. try engine.lock(); var d = try make_user(gpa, 999, "z@x.io"); defer d.deinit(); try engine.insert("app", "users", &d, &env.gen); engine.unlock(); try testing.expect(engine.seq > engine.pager.loaded.seq); } // Everything is still there after a reopen: 200 from the image, 1 from the // log records written after the truncation. { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); try engine.lock(); defer engine.unlock(); const coll = engine.get_collection("app", "users").?; try testing.expectEqual(@as(usize, 201), coll.id_index.count()); try testing.expectEqual(@as(usize, 201), coll.id_index.count()); const id_key = try id_key_for(gpa, .{ .int32 = 999 }); defer gpa.free(id_key); try testing.expect(engine.get_doc("app", "users", id_key) != null); const first_key = try id_key_for(gpa, .{ .int32 = 0 }); defer gpa.free(first_key); try testing.expect(engine.get_doc("app", "users", first_key) != null); } } test "a rebuild reclaims dead document bytes and keeps every index valid" { // What a checkpoint cannot do. A replaced document leaves its old bytes // behind, and they cannot be reclaimed in place because every index leaf // holds a physical offset -- so the rebuild has to move the documents *and* // repack the indexes against the new offsets, together. // // The assertion that matters is not the size but the second half: after the // rebuild every document is still findable through both the _id_ index and a // secondary one. A rebuild that moved documents and left one stale entry // behind would shrink the file and return wrong answers. // // Mutation checks: skip the repack and the lookups go red (stale offsets); // skip the slab reset and the file never shrinks. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); const data_path = try std.fmt.allocPrint(gpa, "{s}.data", .{tmp.path}); defer gpa.free(data_path); defer std.Io.Dir.cwd().deleteFile(io, data_path) catch {}; var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); try engine.lock(); var spec = try index_spec(gpa, "email", "email_1", false, false, null); defer spec.deinit(); _ = try engine.create_index("app", "users", &spec); var i: i32 = 0; while (i < 60) : (i += 1) { var d = try make_user(gpa, i, "a@x.io"); defer d.deinit(); try engine.insert("app", "users", &d, &env.gen); } // Replace every one of them, which is what makes the old bytes garbage. i = 0; while (i < 60) : (i += 1) { var d = try make_user(gpa, i, "b@x.io"); defer d.deinit(); _ = try engine.replace("app", "users", &d, &env.gen); } engine.unlock(); const coll = engine.get_collection("app", "users").?; const before_used = coll.slab_used; try engine.compact(); // 60 live documents occupy less than the 120 writes that produced them. try testing.expect(coll.slab_used < before_used); try testing.expect(coll.slab_used > 0); try testing.expectEqual(@as(usize, 60), coll.id_index.count()); try testing.expectEqual(@as(usize, 60), coll.id_index.count()); // The assertions that matter. Content alone proves nothing here: the old // extents are only handed to the free list, not overwritten, so a stale // offset still reads a plausible document. What distinguishes a repacked // index from a stale one is *where* the offset points -- every live offset // must fall inside an extent the collection currently owns. try engine.lock(); defer engine.unlock(); i = 0; while (i < 60) : (i += 1) { const id_key = try id_key_for(gpa, .{ .int32 = i }); defer gpa.free(id_key); var enc: std.ArrayListUnmanaged(u8) = .empty; defer enc.deinit(gpa); try bson.encode_key(.{ .int32 = i }, gpa, &enc); const from_index = coll.id_index.lookup_exact(enc.items) orelse return error.TestUnexpectedResult; // The offset must be in the new slab. With the hashmap gone this is the // whole check: there is no second structure to disagree with, so a // rebuild that left stale offsets behind shows up here and nowhere else. try testing.expect(offset_in_extents(coll, from_index)); try testing.expect(std.mem.indexOf(u8, coll.doc_bytes(from_index), "b@x.io") != null); } // The secondary index too, by the same standard. var found: std.ArrayListUnmanaged(u64) = .empty; defer found.deinit(gpa); const email_ix = coll.find_index("email_1").?; try email_ix.lookup_eq(gpa, &.{.{ .string = "b@x.io" }}, &found); try testing.expectEqual(@as(usize, 60), found.items.len); for (found.items) |off| try testing.expect(offset_in_extents(coll, off)); } test "index drop survives reopen" { var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); var spec = try index_spec(gpa, "email", "email_1", false, false, null); defer spec.deinit(); try engine.lock(); _ = try engine.create_index("app", "users", &spec); var d1 = try make_user(gpa, 1, "a@x.io"); defer d1.deinit(); try engine.insert("app", "users", &d1, &env.gen); try testing.expect(try engine.drop_index("app", "users", "email_1")); engine.unlock(); } var engine2 = try Engine.open(gpa, io, tmp.path); defer engine2.deinit(); try engine2.lock(); try testing.expectEqual(@as(usize, 0), engine2.get_collection("app", "users").?.indexes.items.len); engine2.unlock(); } test "dropping an index does not move its siblings" { // Indexes used to be stored by value, so `orderedRemove` memmoved the // whole list and every `*Index` already handed out -- notably a query // plan's `index` field -- silently referred to a *different* index // afterwards. Nothing caught it: the collection's own bookkeeping stayed // consistent, so only a caller holding a pointer across a drop would see // it, and none of the tests did. // // Mutation check: restore `indexes` to ArrayListUnmanaged(index.Index) // (with the by-value append/remove that goes with it) and the b_1 // assertion below reads "c_1", because slot 1 now holds what used to be in // slot 2. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); const env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); try engine.lock(); defer engine.unlock(); for ([_][]const u8{ "a", "b", "c" }) |field| { const name = try std.fmt.allocPrint(gpa, "{s}_1", .{field}); defer gpa.free(name); var spec = try index_spec(gpa, field, name, false, false, null); defer spec.deinit(); _ = try engine.create_index("app", "users", &spec); } const coll = engine.get_collection("app", "users").?; // Hold pointers across the drop, which is the whole point. const b_ix = coll.find_index("b_1").?; const c_ix = coll.find_index("c_1").?; try testing.expect(try engine.drop_index("app", "users", "a_1")); try testing.expectEqual(@as(usize, 2), coll.indexes.items.len); try testing.expectEqualStrings("b_1", b_ix.name); try testing.expectEqualStrings("c_1", c_ix.name); // And they are still the collection's own indexes, not detached copies. try testing.expectEqual(b_ix, coll.find_index("b_1").?); try testing.expectEqual(c_ix, coll.find_index("c_1").?); } test "drop_collection frees indexes; log without index records replays" { var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); var spec = try index_spec(gpa, "email", "email_1", false, false, null); defer spec.deinit(); try engine.lock(); _ = try engine.create_index("app", "users", &spec); var d1 = try make_user(gpa, 1, "a@x.io"); defer d1.deinit(); try engine.insert("app", "users", &d1, &env.gen); // Dropped in memory; free_collection releases the index memory // (verified by testing.allocator at engine.deinit). try testing.expect(try engine.drop_collection("app", "users")); try testing.expect(engine.get_collection("app", "users") == null); // A log that only ever contained plain upserts replays fine. var d2 = try make_doc(gpa, 2, "bob"); defer d2.deinit(); try engine.insert("app", "plain", &d2, &env.gen); engine.unlock(); } var engine2 = try Engine.open(gpa, io, tmp.path); defer engine2.deinit(); try engine2.lock(); const id_key = try id_key_for(gpa, bson.Value{ .int32 = 2 }); defer gpa.free(id_key); try testing.expect(engine2.get_doc("app", "plain", id_key) != null); // Pre-existing limitation (documented in the README): drop_collection // writes no log record, so the collection and its index resurrect. const users = engine2.get_collection("app", "users").?; try testing.expectEqual(@as(usize, 1), users.indexes.items.len); try testing.expectEqual(@as(usize, 1), try index_count(gpa, &engine2, "app", "users", "email_1", .{ .string = "a@x.io" })); engine2.unlock(); } /// A document with an `expireAt` field of any type (omitted when null). fn doc_with_expire(gpa: std.mem.Allocator, id: i32, expire: ?bson.Value) !bson.Document { var arena = std.heap.ArenaAllocator.init(gpa); errdefer arena.deinit(); const n: usize = if (expire == null) 1 else 2; const pairs = try arena.allocator().alloc(bson.Pair, n); pairs[0] = .{ .key = try arena.allocator().dupe(u8, "_id"), .value = .{ .int32 = id } }; if (expire) |v| { const value = switch (v) { .string => |s| bson.Value{ .string = try arena.allocator().dupe(u8, s) }, else => v, }; pairs[1] = .{ .key = try arena.allocator().dupe(u8, "expireAt"), .value = value }; } return .{ .arena = arena, .pairs = pairs }; } test "ttl_sweep deletes expired documents and the deletion survives reopen" { var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; // A fixed clock: the sweep takes `now` as a parameter precisely so the // test does not depend on the wall clock. const now_ms: i64 = 1_700_000_000_000; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); { var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); var spec = try index_spec(gpa, "expireAt", "expireAt_1", false, false, 60); defer spec.deinit(); try engine.lock(); defer engine.unlock(); _ = try engine.create_index("app", "sessions", &spec); const docs = [_]struct { id: i32, expire: ?bson.Value }{ .{ .id = 1, .expire = .{ .datetime = now_ms - 120_000 } }, // long expired .{ .id = 2, .expire = .{ .datetime = now_ms - 60_000 } }, // exactly at the cutoff .{ .id = 3, .expire = .{ .datetime = now_ms - 30_000 } }, // not yet .{ .id = 4, .expire = .{ .datetime = now_ms + 3_600_000 } }, // future .{ .id = 5, .expire = .{ .string = "tomorrow" } }, // not a date: never expires .{ .id = 6, .expire = null }, // no field: indexed as null }; for (docs) |d| { var doc = try doc_with_expire(gpa, d.id, d.expire); defer doc.deinit(); try engine.insert("app", "sessions", &doc, &env.gen); } const coll = engine.get_collection("app", "sessions").?; try testing.expectEqual(@as(usize, 6), coll.id_index.count()); try testing.expectEqual(@as(usize, 6), coll.indexes.items[0].count()); // The cutoff is inclusive: doc 2 goes with doc 1. try testing.expectEqual(@as(usize, 2), try engine.ttl_sweep(now_ms)); try testing.expectEqual(@as(usize, 4), coll.id_index.count()); try testing.expectEqual(@as(usize, 4), coll.indexes.items[0].count()); try testing.expectEqual(@as(usize, 0), try index_count(gpa, &engine, "app", "sessions", "expireAt_1", .{ .datetime = now_ms - 120_000 })); // The string and the missing field are untouched by any sweep. try testing.expectEqual(@as(usize, 1), try index_count(gpa, &engine, "app", "sessions", "expireAt_1", .{ .string = "tomorrow" })); try testing.expectEqual(@as(usize, 1), try index_count(gpa, &engine, "app", "sessions", "expireAt_1", .null)); // Idempotent: nothing else is expired at the same instant. try testing.expectEqual(@as(usize, 0), try engine.ttl_sweep(now_ms)); // An hour later doc 3 has expired too; doc 4 still has not. try testing.expectEqual(@as(usize, 1), try engine.ttl_sweep(now_ms + 3_000_000)); try testing.expectEqual(@as(usize, 3), coll.id_index.count()); } // Sweeps go through `remove`, so they are logged: the deletions hold // across a restart, and the TTL index comes back with its expiry. var engine2 = try Engine.open(gpa, io, tmp.path); defer engine2.deinit(); try engine2.lock(); defer engine2.unlock(); const coll = engine2.get_collection("app", "sessions").?; try testing.expectEqual(@as(usize, 3), coll.id_index.count()); try testing.expectEqual(@as(usize, 1), coll.indexes.items.len); try testing.expectEqual(@as(?i64, 60), coll.indexes.items[0].ttl); try testing.expectEqual(@as(usize, 3), coll.indexes.items[0].count()); for ([_]i32{ 1, 2, 3 }) |id| { const id_key = try id_key_for(gpa, bson.Value{ .int32 = id }); defer gpa.free(id_key); try testing.expect(engine2.get_doc("app", "sessions", id_key) == null); } const alive = try id_key_for(gpa, bson.Value{ .int32 = 4 }); defer gpa.free(alive); try testing.expect(engine2.get_doc("app", "sessions", alive) != null); } test "ttl_sweep spans collections and several TTL indexes on one collection" { var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; const now_ms: i64 = 1_700_000_000_000; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); try engine.lock(); defer engine.unlock(); // Two TTL indexes over the same collection (MongoDB allows this): one // document is expired by both, and must only be deleted once. var spec_a = try index_spec(gpa, "expireAt", "expireAt_1", false, false, 60); defer spec_a.deinit(); var spec_b = try index_spec(gpa, "seenAt", "seenAt_1", false, false, 10); defer spec_b.deinit(); _ = try engine.create_index("app", "sessions", &spec_a); _ = try engine.create_index("app", "sessions", &spec_b); var both = try bson.Document.alloc(gpa, &.{ .{ .key = "_id", .value = .{ .int32 = 1 } }, .{ .key = "expireAt", .value = .{ .datetime = now_ms - 120_000 } }, .{ .key = "seenAt", .value = .{ .datetime = now_ms - 120_000 } }, }); defer both.deinit(); try engine.insert("app", "sessions", &both, &env.gen); // A second collection with its own TTL index, and a plain collection // that no sweep may touch. var spec_c = try index_spec(gpa, "at", "at_1", false, false, 0); defer spec_c.deinit(); _ = try engine.create_index("app", "events", &spec_c); var ev = try bson.Document.alloc(gpa, &.{ .{ .key = "_id", .value = .{ .int32 = 2 } }, // expireAfterSeconds 0: expires at exactly the stored instant. .{ .key = "at", .value = .{ .datetime = now_ms } }, }); defer ev.deinit(); try engine.insert("app", "events", &ev, &env.gen); var plain = try make_doc(gpa, 3, "keep"); defer plain.deinit(); try engine.insert("other", "plain", &plain, &env.gen); try testing.expectEqual(@as(usize, 2), try engine.ttl_sweep(now_ms)); try testing.expectEqual(@as(usize, 0), engine.get_collection("app", "sessions").?.doc_count); try testing.expectEqual(@as(usize, 0), engine.get_collection("app", "events").?.doc_count); try testing.expectEqual(@as(usize, 1), engine.get_collection("other", "plain").?.doc_count); } /// Report a replay eviction whose `_id` bytes differ from the incoming record's. /// /// `_id_` is keyed on the canonical `bson.encode_key`, under which int32 1, /// int64 1 and double 1.0 are one key -- as they are in MongoDB. A database /// written before that could legitimately hold two such documents, and replaying /// it now drops one. That is the intended semantics and a one-way migration, so /// it has to be said out loud rather than discovered. /// /// Best effort by design: this runs during replay, where nothing may refuse to /// start. fn warn_on_equal_id_collision( gpa: std.mem.Allocator, coll: *const Collection, old_off: u64, doc: *const bson.Document, record: storage.Record, ) void { var arena = std.heap.ArenaAllocator.init(gpa); defer arena.deinit(); const a = arena.allocator(); const old_id = (bson.get_at(a, coll.doc_bytes(old_off), "_id") catch return) orelse return; const new_id = doc.get("_id") orelse return; const old_bytes = bson.serialize_value(a, old_id) catch return; const new_bytes = bson.serialize_value(a, new_id) catch return; if (std.mem.eql(u8, old_bytes, new_bytes)) return; // an ordinary replace std.debug.print( "multiforadb: WARNING: {s}.{s} holds two documents whose _id values compare " ++ "equal but were stored differently; keeping the later one. This is a " ++ "one-way migration -- _id uniqueness is canonical now, as in MongoDB.\n", .{ record.db, record.coll }, ); } /// The canonical `_id` key the engine looks documents up by, owned by the caller. /// Tests used `bson.serialize_value` when a hashmap keyed on it; `_id_` is keyed /// on `bson.encode_key`, which is the canonical encoding. fn id_key_for(gpa: std.mem.Allocator, v: bson.Value) ![]u8 { var enc: std.ArrayListUnmanaged(u8) = .empty; errdefer enc.deinit(gpa); try bson.encode_key(v, gpa, &enc); return enc.toOwnedSlice(gpa); } /// Whether an offset falls inside one of the collection's current slab extents. /// After a rebuild every live offset must, and that is what tells a repacked /// index from one still holding pre-rebuild offsets -- the old bytes are on the /// free list rather than overwritten, so reading them still succeeds. fn offset_in_extents(coll: *const Collection, off: u64) bool { return coll.run_of(off) != null; } /// Document bytes at an absolute file offset, without needing the Collection. /// The rebuild works from offsets while the collection's own slab cursors are /// being replaced under it. fn doc_bytes_in(pager: *const pgr.Pager, off: u64) []const u8 { const len: usize = std.mem.readInt(u32, pager.bytes(off, 4)[0..4], .little); return pager.bytes(off, len); } // --------------------------------------------------------------------------- // Catalog encoding helpers // --------------------------------------------------------------------------- fn put_u32(gpa: std.mem.Allocator, out: *std.ArrayListUnmanaged(u8), v: u32) !void { var b: [4]u8 = undefined; std.mem.writeInt(u32, &b, v, .little); try out.appendSlice(gpa, &b); } fn put_u64(gpa: std.mem.Allocator, out: *std.ArrayListUnmanaged(u8), v: u64) !void { var b: [8]u8 = undefined; std.mem.writeInt(u64, &b, v, .little); try out.appendSlice(gpa, &b); } fn put_bytes(gpa: std.mem.Allocator, out: *std.ArrayListUnmanaged(u8), v: []const u8) !void { try put_u32(gpa, out, @intCast(v.len)); try out.appendSlice(gpa, v); } /// A bounds-checked cursor over the catalog. Every read is checked because the /// bytes come off disk: a truncated or scrambled catalog must produce an error /// the caller can fall back from, never a read past the end. const Reader = struct { b: []const u8, at: usize = 0, fn take(self: *Reader, n: usize) ![]const u8 { if (self.at + n > self.b.len) return error.CorruptCatalog; defer self.at += n; return self.b[self.at..][0..n]; } fn read_byte(self: *Reader) !u8 { return (try self.take(1))[0]; } fn read_u32(self: *Reader) !u32 { return std.mem.readInt(u32, (try self.take(4))[0..4], .little); } fn read_u64(self: *Reader) !u64 { return std.mem.readInt(u64, (try self.take(8))[0..8], .little); } fn read_bytes(self: *Reader) ![]const u8 { const n = try self.read_u32(); return self.take(n); } }; test "the epochs that invalidate a cursor move exactly when they must" { // Three separate promises, each one load-bearing for an open cursor: // // - a rebuild moves every document, so a saved slab offset is stale; // - a drop-and-recreate under the same name is a different collection, // which a cursor holding only namespace strings cannot otherwise see; // - `Index.reset_tree` re-creates node ids 0 and 1 as different nodes, so a // saved (leaf, slot) position becomes valid-and-wrong rather than absent. // // A cursor's whole safety story is these three bumps, so assert them here // rather than inferring them from cursor behaviour later. var threaded: std.Io.Threaded = .init_single_threaded; defer threaded.deinit(); var env = test_env(&threaded); const io = env.io; const gpa = testing.allocator; var tmp = try TmpLog.init(gpa); defer tmp.deinit(gpa); var engine = try Engine.open(gpa, io, tmp.path); defer engine.deinit(); try engine.lock(); var i: i32 = 0; while (i < 40) : (i += 1) { var d = try make_doc(gpa, i, "payload"); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } // Garbage, so the collection is worth rewriting: `compact` skips a // collection with nothing to reclaim. i = 0; while (i < 20) : (i += 1) _ = try engine.remove_by_id("app", "c", .{ .int32 = i }); try engine.commit(); const before = engine.get_collection("app", "c").?.layout_epoch; engine.unlock(); try testing.expect(before != 0); // A rebuild moves documents, so the epoch must move with them. try engine.compact(); try engine.lock(); const after_rebuild = engine.get_collection("app", "c").?.layout_epoch; engine.unlock(); try testing.expect(after_rebuild != before); // A recreated collection must not be mistaken for the one that was // dropped. Starting each collection's epoch at zero would fail here. try engine.lock(); try testing.expect(try engine.drop_collection("app", "c")); var fresh_doc = try make_doc(gpa, 1, "fresh"); defer fresh_doc.deinit(); try engine.insert("app", "c", &fresh_doc, &env.gen); const after_recreate = engine.get_collection("app", "c").?.layout_epoch; engine.unlock(); try testing.expect(after_recreate != after_rebuild); try testing.expect(after_recreate != before); // Reclamation does not move a live document, so a cursor's *live* offsets // stay good -- but the pages it gives back can be handed out again, and a // cursor's saved offset list may name one of them. Same remedy, and the // same token. // // Both halves matter. A checkpoint that reclaims nothing must leave the // epoch alone, or every open cursor on a busy collection dies on the // checkpoint cadence for nothing. try engine.lock(); var j: i32 = 0; while (j < 200) : (j += 1) { var d = try make_padded(gpa, 1000 + j, 2000); defer d.deinit(); try engine.insert("app", "c", &d, &env.gen); } try engine.commit(); const quiet_before = engine.get_collection("app", "c").?.layout_epoch; try engine.checkpoint(); try testing.expectEqual(quiet_before, engine.get_collection("app", "c").?.layout_epoch); j = 0; while (j < 200) : (j += 1) _ = try engine.remove_by_id("app", "c", .{ .int32 = 1000 + j }); try engine.commit(); try engine.checkpoint(); const after_reclaim = engine.get_collection("app", "c").?.layout_epoch; try testing.expect(engine.get_collection("app", "c").?.reclaimed_bytes > 0); try testing.expect(after_reclaim != quiet_before); // And the index-level token, which guards the position hint. const coll = engine.get_collection("app", "c").?; const index_before = coll.id_index.epoch; try coll.id_index.reset_tree(gpa); try testing.expect(coll.id_index.epoch != index_before); engine.unlock(); }