PLAN D7's six items, with the numbers and the command that reproduces each in
tests/e2e/results/m0-gates.txt. Unit tests green in both optimize modes, the
whole e2e matrix green, the spec scorecard byte-identical at 131/161/195, and
the large smoke run at the scale D7.3 asked for:
21.47 GB collection (1,310,720 x 16 KiB)
data file 21.75 GB (+1.3% over the documents)
log after the load 2.5 MB (checkpoints reclaim it)
kill -9 then reopen 0.5 s (0.5 s at 4 GB too -- flat)
RSS after reopen 237 MB (1.1% of the data)
count after restart 1,310,720 last document byte-intact
acked writes after kill 200/200
That is the milestone's claim, measured: an open costs the working set rather
than the size of the database. Before M0 the same measurement was 523 MB
resident for a 512 MB database, because recovering each document's `_id` meant
reading every document at open.
Two gates need reading rather than a tick, and m0-gates.txt says so where a
reader would otherwise take a tick for granted.
The churn gate settles at 1.65x live data (delete-heavy) to 2.47x
(update-heavy), flat, above the ~1.3x amendment A2 hoped for. Rebuild-only
reclamation cannot reach that: it needs a whole second copy of the live data
before the first can be freed. The gate existed to decide whether doc-level
free lists are needed after M0, and that is the answer.
Benchmark parity holds for every read and latency row inside the run-to-run
spread, and bulk insert regresses 24% (732 -> 555 MB/s), reproducibly across
three runs. Risk 1 as written: document bytes now reach the disk uncompressed
on top of the LZ4 log. createIndex improves 62% from the same change.
Three measurement bugs fixed while running the gates, because each would have
put a false number in the README:
- `compare-run.sh` measured "db on disk" as `du` of the log alone against
`du` of mongod's whole dbpath. It reported 20 MB for a 1 GB collection --
the documents had moved to <db>.data. Honest figure, measured: 914 MB of
allocated blocks against mongod's compressed 85 MB.
- `big.js` counted "compaction events" as "the log shrank", which is a
*checkpoint* now. It claimed 12 compaction rewrites during a pure insert
load, which has no garbage to compact.
- `big.js` labelled peak RSS "in-memory engine: docs live in RAM" and its
summary said the collection was held "fully in RAM". Both were true of the
engine this milestone replaced.
README: the storage section described an all-in-RAM engine; the comparison
table mixed one old run's body with three new rows; and `findOne({_id})` was
documented as a full scan for integer ids, which the ordered `_id_` index made
false (2 ms against 55 s for a scan of the same 21.5 GB collection). The table
is now best-of-three for both servers, with the measured variance stated, since
two runs of the same binary moved the sub-10 ms rows by 27-51%.
135 lines
6.9 KiB
Bash
135 lines
6.9 KiB
Bash
#!/bin/bash
|
||
# Compare multiforadb against a real MongoDB with the same workload, same driver.
|
||
#
|
||
# bash tests/e2e/compare-run.sh [size] [doc-size]
|
||
# (defaults: 1g, 16k)
|
||
#
|
||
# Starts mongod on :27018 and multiforadb on :27019, runs compare.js against
|
||
# each (durable writes: multiforadb fsyncs per doc, mongod ack'd with j:true),
|
||
# measures kill -9 reopen time for both, and prints a side-by-side table.
|
||
set -u
|
||
cd "$(dirname "$0")/../.."
|
||
SIZE="${1:-1g}"
|
||
DOC="${2:-16k}"
|
||
echo "comparing multiforadb vs mongodb — dataset ${SIZE}, docs ~${DOC}"
|
||
|
||
CMPDIR=/tmp/mongo-cmp
|
||
mkdir -p "$CMPDIR/mongod"
|
||
MFDB_LOG="$CMPDIR/mfdb.log"
|
||
MFDB_OUT="$CMPDIR/mfdb-srv.out"
|
||
MD_OUT="$CMPDIR/md-srv.out"
|
||
MFDB_PORT=27019
|
||
MD_PORT=27018
|
||
rm -f "$MFDB_LOG" "$MFDB_LOG.data"
|
||
rm -rf "$CMPDIR/mongod" && mkdir -p "$CMPDIR/mongod"
|
||
|
||
# ---- MongoDB -------------------------------------------------------------
|
||
echo; echo "### mongod (MongoDB $(mongod --version | grep -oE 'v[0-9.]+' | head -1))"
|
||
mongod --dbpath "$CMPDIR/mongod" --port $MD_PORT --bind_ip 127.0.0.1 \
|
||
--quiet >"$MD_OUT" 2>&1 &
|
||
MD_PID=$!
|
||
# Poll until mongod answers; a fixed sleep is flaky right after other
|
||
# benchmark phases have warmed the machine.
|
||
wait_ready() { # $1 = url
|
||
for _ in $(seq 1 90); do
|
||
NODE_PATH="tests/e2e/node_modules" node -e "require('mongodb').MongoClient.connect(process.argv[1],{serverSelectionTimeoutMS:800}).then(c=>c.close().then(()=>process.exit(0))).catch(()=>process.exit(1))" "$1" 2>/dev/null \
|
||
&& return 0
|
||
sleep 1
|
||
done
|
||
return 1
|
||
}
|
||
wait_ready "mongodb://127.0.0.1:$MD_PORT" || { echo "mongod never became ready" >&2; exit 1; }
|
||
node tests/e2e/compare.js --url "mongodb://127.0.0.1:$MD_PORT" --label mongodb --size "$SIZE" --doc-size "$DOC" \
|
||
> "$CMPDIR/mongo-report.txt" 2>&1 || { echo "mongodb bench failed:"; tail -5 "$CMPDIR/mongo-report.txt"; }
|
||
MD_RSS=$(ps -o rss= -p $MD_PID | awk '{printf "%.0f", $1/1024}')
|
||
MD_DISK=$(du -sm "$CMPDIR/mongod" | awk '{print $1}')
|
||
|
||
echo; echo "### mongod kill -9 + reopen"
|
||
kill -9 $MD_PID; wait $MD_PID 2>/dev/null
|
||
MD_REOPEN=$(cd tests/e2e && node -e '
|
||
const { spawn } = require("child_process");
|
||
const { MongoClient } = require("mongodb");
|
||
const t0 = Date.now();
|
||
const p = spawn("mongod", ["--dbpath","/tmp/mongo-cmp/mongod","--port","27018","--bind_ip","127.0.0.1","--quiet"], {stdio:"ignore"});
|
||
const poll = async () => {
|
||
const c = new MongoClient("mongodb://127.0.0.1:27018", {serverSelectionTimeoutMS: 800});
|
||
try { await c.connect(); await c.db("admin").command({ping:1}); await c.close(); p.kill("SIGKILL"); console.log(((Date.now()-t0)/1000).toFixed(1)); }
|
||
catch { try { await c.close(); } catch {}; setTimeout(poll, 200); }
|
||
};
|
||
setTimeout(poll, 300);
|
||
')
|
||
kill -9 $MD_PID 2>/dev/null
|
||
|
||
# ---- multiforadb ----------------------------------------------------------
|
||
echo; echo "### multiforadb (recommended config: --compact-threshold 1g)"
|
||
# Debug is ~10-200x slower (see the README's perf section) — the comparison
|
||
# must use the optimized build.
|
||
zig build -Doptimize=ReleaseFast 2>&1 | grep -c "^error" | grep -q "^0" || { echo "build failed"; exit 1; }
|
||
./zig-out/bin/multiforadb --port $MFDB_PORT --db "$MFDB_LOG" --compact-threshold 1g >"$MFDB_OUT" 2>&1 &
|
||
MFDB_PID=$!
|
||
wait_ready "mongodb://127.0.0.1:$MFDB_PORT" || { echo "multiforadb never became ready" >&2; exit 1; }
|
||
node tests/e2e/compare.js --url "mongodb://127.0.0.1:$MFDB_PORT" --label multiforadb --size "$SIZE" --doc-size "$DOC" \
|
||
> "$CMPDIR/mfdb-report.txt" 2>&1 || { echo "multiforadb bench failed:"; tail -5 "$CMPDIR/mfdb-report.txt"; }
|
||
MFDB_RSS=$(ps -o rss= -p $MFDB_PID | awk '{printf "%.0f", $1/1024}')
|
||
# Both files. The documents live in "$MFDB_LOG".data since the mmap foundation
|
||
# landed, and the log is truncated at every checkpoint -- so measuring the log
|
||
# alone reported 20 MB for a 1 GB collection, against a `du` over mongod's whole
|
||
# dbpath. `du` rather than `ls`: the data file is grown with setLength and is
|
||
# sparse until written, and allocated blocks are what actually costs disk.
|
||
MFDB_DISK=$(du -scm "$MFDB_LOG" "$MFDB_LOG.data" 2>/dev/null | tail -1 | awk '{print $1}')
|
||
|
||
echo; echo "### multiforadb kill -9 + reopen (replay)"
|
||
kill -9 $MFDB_PID; wait $MFDB_PID 2>/dev/null
|
||
MFDB_REOPEN=$(cd tests/e2e && node -e '
|
||
const { spawn } = require("child_process");
|
||
const { MongoClient } = require("mongodb");
|
||
const t0 = Date.now();
|
||
const p = spawn("../../zig-out/bin/multiforadb", ["--port","27019","--db","/tmp/mongo-cmp/mfdb.log","--compact-threshold","1g"], {stdio:"ignore"});
|
||
const poll = async () => {
|
||
const c = new MongoClient("mongodb://127.0.0.1:27019", {serverSelectionTimeoutMS: 800});
|
||
try { await c.connect(); await c.db("admin").command({ping:1}); await c.close(); p.kill("SIGKILL"); console.log(((Date.now()-t0)/1000).toFixed(1)); }
|
||
catch { try { await c.close(); } catch {}; setTimeout(poll, 200); }
|
||
};
|
||
setTimeout(poll, 300);
|
||
')
|
||
kill -9 $MFDB_PID 2>/dev/null
|
||
|
||
# ---- side by side ----------------------------------------------------------
|
||
echo; echo "### side by side — ${SIZE} dataset, ~${DOC} docs"
|
||
cat > "$CMPDIR/meta.json" <<EOF
|
||
{"mfdb_rss_mb": "$MFDB_RSS", "md_rss_mb": "$MD_RSS", "mfdb_reopen": "${MFDB_REOPEN}s", "md_reopen": "${MD_REOPEN}s", "mfdb_disk_mb": "${MFDB_DISK}MB", "md_disk_mb": "${MD_DISK}MB"}
|
||
EOF
|
||
node -e '
|
||
const fs = require("fs");
|
||
const read = (p) => {
|
||
const m = {};
|
||
if (!fs.existsSync(p)) return m;
|
||
for (const line of fs.readFileSync(p, "utf8").split("\n")) {
|
||
const i = line.indexOf("\t");
|
||
if (i > 0) m[line.slice(0, i)] = line.slice(i + 1).replace(/\t.*$/, "");
|
||
}
|
||
return m;
|
||
};
|
||
const a = read("/tmp/mongo-cmp/mfdb-report.txt");
|
||
const b = read("/tmp/mongo-cmp/mongo-report.txt");
|
||
const meta = JSON.parse(fs.readFileSync("/tmp/mongo-cmp/meta.json", "utf8"));
|
||
const keys = ["insertOne (sequential) ×200","bulk insert throughput","docs loaded","createIndex({k: 1})",
|
||
"countDocuments({})","findOne({_id: <ObjectId>})","findOne({k: 500}) (indexed)","find({p: {$gte,$lt}}).count() (scan)",
|
||
"find({}).sort({_id:-1}).limit(20)","find({}, {proj}).limit(1000)","aggregate $group by k",
|
||
"updateOne({_id}) ×50","updateMany({k: 7}, {$inc})","deleteOne({_id}) + insertOne","node client RSS"];
|
||
const col = (v) => String(v).padEnd(22);
|
||
console.log(`${"benchmark".padEnd(42)} ${col("multiforadb")} ${col("mongodb")} ratio`);
|
||
for (const k of keys) {
|
||
const av = a[k] || "—", bv = b[k] || "—";
|
||
const ar = parseFloat(av), br = parseFloat(bv);
|
||
const ratio = isFinite(ar) && isFinite(br) && ar > 0 && br > 0 ? (ar / br).toFixed(1) + "x" : "";
|
||
console.log(`${k.padEnd(42)} ${col(av)} ${col(bv)} ${ratio}`);
|
||
}
|
||
console.log(`${`server RSS`.padEnd(42)} ${col(meta.mfdb_rss_mb + " MB")} ${col(meta.md_rss_mb + " MB")}`);
|
||
console.log(`${`kill -9 reopen`.padEnd(42)} ${col(meta.mfdb_reopen)} ${col(meta.md_reopen)}`);
|
||
console.log(`${`db on disk`.padEnd(42)} ${col(meta.mfdb_disk_mb)} ${col(meta.md_disk_mb)}`);
|
||
'
|
||
|
||
pkill -9 -f "multiforadb --port $MFDB_PORT" 2>/dev/null
|
||
echo; echo "done — reports: $CMPDIR/mfdb-report.txt, $CMPDIR/mongo-report.txt"
|