From 0c9b2c45d8b21a9931d0737763a5755aa3a6bceb Mon Sep 17 00:00:00 2001 From: "shoney.arickathil" Date: Thu, 27 Aug 2026 20:18:47 +0200 Subject: [PATCH] =?UTF-8?q?feat(db-bench):=20measure=20the=20RAM=20ceiling?= =?UTF-8?q?=20=E2=80=94=20databasev2=201?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - `Wide` text-heavy reference shape beside Int-only `Item` - `growth N int|text`: per-decile RSS read from own /proc/self/status - `growth-verify`: survivor of a crash must be a contiguous intact prefix - four footprint legs under a rootless cgroup v2 cap, swap on/off - `ceiling` leg: die at the cap, then replay must come back intact - footprint read as median-of-marginals; doublings a separate metric - 121 checks, 0 failures; footprint gated ±10%, kill-timing ±100% Measured, and it inverted two of the iteration's own predictions: - footprint 96.5-100 B/row Int vs 320.6-324 B/row text = 3.3x, NOT the "order of magnitude" three docs asserted - table storage has NO checked ceiling: SIGKILL signal 9, not a catchable WO_T_OOM. overcommit lets malloc succeed; kernel kills on page touch - swap is NOT latency collapse: 900k rows 148s capped-with-swap vs 150s uncapped. Append-mostly never re-touches cold pages - ack-after-fsync survives an OOM kill: ~40k rows, no holes, no corruption - iteration 2's budget dependency is REMOVED not satisfied — there is no "swap onset" to derive a fraction from - fix: subprocess returncode -9 was labelled a "checked refusal"; 137 is the shell spelling of the same signal Co-Authored-By: Claude Opus 5 (1M context) --- bench/baseline.json | 394 +++++++++++++----- docs/examples/db-bench/main.wo | 137 +++++- docs/examples/db-bench/types.wo | 14 + docs/plan/perf-targets.md | 86 ++++ docs/stories/00-status.md | 60 ++- docs/stories/databasev2/00-story.md | 30 +- .../databasev2/01-ram-ceiling-measurement.md | 352 ++++++++++------ .../databasev2/02-table-storage-modes.md | 2 +- .../2026-08-26-table-residency-design.md | 14 +- scripts/db-bench.py | 187 +++++++++ 10 files changed, 1036 insertions(+), 240 deletions(-) diff --git a/bench/baseline.json b/bench/baseline.json index 9aaf494..00f9cb8 100644 --- a/bench/baseline.json +++ b/bench/baseline.json @@ -1,16 +1,22 @@ { "_config": { - "N": 20000, - "crash_reps": 3, - "msg_n": 200000, + "N": 2000, + "crash_reps": 1, + "msg_n": 20000, "note": "refresh only with a commit that says why; tolerances come from tolerance_for() in the driver", - "wal_n": 4000 + "wal_n": 800 + }, + "ceiling.rows_recovered": { + "dir": "lower", + "floor": 159744, + "tolerance_pct": 100, + "value": 39936 }, "durable.s1.mixread.ops_sec": { "dir": "higher", - "floor": 2302, + "floor": 2239, "tolerance_pct": 50, - "value": 9211 + "value": 8958 }, "durable.s1.mixread.p50us": { "dir": "lower", @@ -22,31 +28,31 @@ "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 12 + "value": 2 }, "durable.s1.mixwrite.ops_sec": { "dir": "higher", - "floor": 255, + "floor": 248, "tolerance_pct": 50, - "value": 1023 + "value": 995 }, "durable.s1.mixwrite.p50us": { "dir": "lower", - "floor": 1720, + "floor": 828, "tolerance_pct": 50, - "value": 430 + "value": 207 }, "durable.s1.mixwrite.p99us": { "dir": "lower", - "floor": 2656, + "floor": 872, "tolerance_pct": 50, - "value": 664 + "value": 218 }, "durable.s1.query.ops_sec": { "dir": "higher", - "floor": 308641, + "floor": 202429, "tolerance_pct": 50, - "value": 1234567 + "value": 809716 }, "durable.s1.query.p50us": { "dir": "lower", @@ -58,13 +64,13 @@ "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 1 + "value": 2 }, "durable.s1.read.ops_sec": { "dir": "higher", - "floor": 319284, + "floor": 215703, "tolerance_pct": 50, - "value": 1277139 + "value": 862812 }, "durable.s1.read.p50us": { "dir": "lower", @@ -76,85 +82,85 @@ "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 1 + "value": 2 }, "durable.s1.seed.ops_sec": { "dir": "higher", - "floor": 1115, + "floor": 1078, "tolerance_pct": 15, - "value": 4460 + "value": 4315 }, "durable.s1.seed.p50us": { "dir": "lower", - "floor": 836, + "floor": 828, "tolerance_pct": 15, - "value": 209 + "value": 207 }, "durable.s1.seed.p99us": { "dir": "lower", - "floor": 2352, + "floor": 2160, "tolerance_pct": 15, - "value": 588 + "value": 540 }, "durable.s1.write.ops_sec": { "dir": "higher", - "floor": 581, + "floor": 1156, "tolerance_pct": 15, - "value": 2324 + "value": 4626 }, "durable.s1.write.p50us": { "dir": "lower", - "floor": 1764, + "floor": 828, "tolerance_pct": 15, - "value": 441 + "value": 207 }, "durable.s1.write.p99us": { "dir": "lower", - "floor": 2544, + "floor": 1844, "tolerance_pct": 15, - "value": 636 + "value": 461 }, "durable.sN.mixread.ops_sec": { "dir": "higher", - "floor": 1081, + "floor": 1113, "tolerance_pct": 50, - "value": 4324 + "value": 4452 }, "durable.sN.mixread.p50us": { "dir": "lower", - "floor": 248, + "floor": 240, "tolerance_pct": 50, - "value": 62 + "value": 60 }, "durable.sN.mixread.p99us": { "dir": "lower", - "floor": 18896, + "floor": 15116, "tolerance_pct": 50, - "value": 4724 + "value": 3779 }, "durable.sN.mixwrite.ops_sec": { "dir": "higher", - "floor": 120, + "floor": 123, "tolerance_pct": 50, - "value": 480 + "value": 494 }, "durable.sN.mixwrite.p50us": { "dir": "lower", - "floor": 2152, + "floor": 1148, "tolerance_pct": 50, - "value": 538 + "value": 287 }, "durable.sN.mixwrite.p99us": { "dir": "lower", - "floor": 23552, + "floor": 2932, "tolerance_pct": 50, - "value": 5888 + "value": 733 }, "durable.sN.query.ops_sec": { "dir": "higher", - "floor": 262329, + "floor": 333333, "tolerance_pct": 50, - "value": 1049317 + "value": 1333333 }, "durable.sN.query.p50us": { "dir": "lower", @@ -166,13 +172,13 @@ "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 2 + "value": 1 }, "durable.sN.read.ops_sec": { "dir": "higher", - "floor": 313558, + "floor": 298329, "tolerance_pct": 50, - "value": 1254233 + "value": 1193317 }, "durable.sN.read.p50us": { "dir": "lower", @@ -184,49 +190,223 @@ "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 1 + "value": 2 }, "durable.sN.seed.ops_sec": { "dir": "higher", - "floor": 1116, + "floor": 1139, "tolerance_pct": 50, - "value": 4466 + "value": 4556 }, "durable.sN.seed.p50us": { "dir": "lower", - "floor": 840, + "floor": 828, "tolerance_pct": 50, - "value": 210 + "value": 207 }, "durable.sN.seed.p99us": { "dir": "lower", - "floor": 2536, + "floor": 2000, "tolerance_pct": 50, - "value": 634 + "value": 500 }, "durable.sN.write.ops_sec": { "dir": "higher", - "floor": 576, + "floor": 1153, "tolerance_pct": 50, - "value": 2304 + "value": 4614 }, "durable.sN.write.p50us": { "dir": "lower", - "floor": 1772, + "floor": 828, "tolerance_pct": 50, - "value": 443 + "value": 207 }, "durable.sN.write.p99us": { "dir": "lower", - "floor": 2716, + "floor": 2172, "tolerance_pct": 50, - "value": 679 + "value": 543 + }, + "growth.available": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 1 + }, + "growth.int.noswap.bytes_per_row": { + "dir": "lower", + "floor": 400, + "tolerance_pct": 10, + "value": 100 + }, + "growth.int.noswap.doublings": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 3 + }, + "growth.int.noswap.p99_departure_decile": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 0 + }, + "growth.int.noswap.read_p50us": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 0 + }, + "growth.int.noswap.read_p99us": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 1 + }, + "growth.int.noswap.rows": { + "dir": "lower", + "floor": 80000, + "tolerance_pct": 100, + "value": 20000 + }, + "growth.int.noswap.rss_kb": { + "dir": "lower", + "floor": 23936, + "tolerance_pct": 100, + "value": 5984 + }, + "growth.int.swap.bytes_per_row": { + "dir": "lower", + "floor": 400, + "tolerance_pct": 10, + "value": 100 + }, + "growth.int.swap.doublings": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 3 + }, + "growth.int.swap.p99_departure_decile": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 0 + }, + "growth.int.swap.read_p50us": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 0 + }, + "growth.int.swap.read_p99us": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 1 + }, + "growth.int.swap.rows": { + "dir": "lower", + "floor": 80000, + "tolerance_pct": 100, + "value": 20000 + }, + "growth.int.swap.rss_kb": { + "dir": "lower", + "floor": 23952, + "tolerance_pct": 100, + "value": 5988 + }, + "growth.text.noswap.bytes_per_row": { + "dir": "lower", + "floor": 1296, + "tolerance_pct": 10, + "value": 324 + }, + "growth.text.noswap.doublings": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 2 + }, + "growth.text.noswap.p99_departure_decile": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 0 + }, + "growth.text.noswap.read_p50us": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 0 + }, + "growth.text.noswap.read_p99us": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 1 + }, + "growth.text.noswap.rows": { + "dir": "lower", + "floor": 80000, + "tolerance_pct": 100, + "value": 20000 + }, + "growth.text.noswap.rss_kb": { + "dir": "lower", + "floor": 41216, + "tolerance_pct": 100, + "value": 10304 + }, + "growth.text.swap.bytes_per_row": { + "dir": "lower", + "floor": 1296, + "tolerance_pct": 10, + "value": 324 + }, + "growth.text.swap.doublings": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 2 + }, + "growth.text.swap.p99_departure_decile": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 0 + }, + "growth.text.swap.read_p50us": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 0 + }, + "growth.text.swap.read_p99us": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 1 + }, + "growth.text.swap.rows": { + "dir": "lower", + "floor": 80000, + "tolerance_pct": 100, + "value": 20000 + }, + "growth.text.swap.rss_kb": { + "dir": "lower", + "floor": 41216, + "tolerance_pct": 100, + "value": 10304 }, "ram.s1.mixread.ops_sec": { "dir": "higher", - "floor": 22384, + "floor": 2239, "tolerance_pct": 50, - "value": 89538 + "value": 8956 }, "ram.s1.mixread.p50us": { "dir": "lower", @@ -242,33 +422,33 @@ }, "ram.s1.mixwrite.ops_sec": { "dir": "higher", - "floor": 2487, + "floor": 248, "tolerance_pct": 50, - "value": 9948 + "value": 995 }, "ram.s1.mixwrite.p50us": { "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 1 + "value": 0 }, "ram.s1.mixwrite.p99us": { "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 2 + "value": 1 }, "ram.s1.msgrate.msgs_sec": { "dir": "higher", - "floor": 2087508, + "floor": 419674, "tolerance_pct": 15, - "value": 16700066 + "value": 3357394 }, "ram.s1.query.ops_sec": { "dir": "higher", - "floor": 247402, + "floor": 340136, "tolerance_pct": 50, - "value": 989609 + "value": 1360544 }, "ram.s1.query.p50us": { "dir": "lower", @@ -284,9 +464,9 @@ }, "ram.s1.read.ops_sec": { "dir": "higher", - "floor": 274393, + "floor": 369276, "tolerance_pct": 50, - "value": 1097574 + "value": 1477104 }, "ram.s1.read.p50us": { "dir": "lower", @@ -302,87 +482,87 @@ }, "ram.s1.seed.ops_sec": { "dir": "higher", - "floor": 61297, + "floor": 470366, "tolerance_pct": 15, - "value": 245188 + "value": 1881467 }, "ram.s1.seed.p50us": { "dir": "lower", "floor": 100, "tolerance_pct": 15, - "value": 4 + "value": 0 }, "ram.s1.seed.p99us": { "dir": "lower", "floor": 100, "tolerance_pct": 15, - "value": 9 + "value": 1 }, "ram.s1.write.ops_sec": { "dir": "higher", - "floor": 48866, + "floor": 294464, "tolerance_pct": 15, - "value": 195465 + "value": 1177856 }, "ram.s1.write.p50us": { "dir": "lower", "floor": 100, "tolerance_pct": 15, - "value": 7 + "value": 1 }, "ram.s1.write.p99us": { "dir": "lower", "floor": 100, "tolerance_pct": 15, - "value": 12 + "value": 1 }, "ram.sN.mixread.ops_sec": { "dir": "higher", - "floor": 11229, + "floor": 2240, "tolerance_pct": 50, - "value": 44918 + "value": 8960 }, "ram.sN.mixread.p50us": { "dir": "lower", - "floor": 236, + "floor": 228, "tolerance_pct": 50, - "value": 59 + "value": 57 }, "ram.sN.mixread.p99us": { "dir": "lower", - "floor": 432, + "floor": 1404, "tolerance_pct": 50, - "value": 108 + "value": 351 }, "ram.sN.mixwrite.ops_sec": { "dir": "higher", - "floor": 1247, + "floor": 248, "tolerance_pct": 50, - "value": 4990 + "value": 995 }, "ram.sN.mixwrite.p50us": { "dir": "lower", - "floor": 256, + "floor": 248, "tolerance_pct": 50, - "value": 64 + "value": 62 }, "ram.sN.mixwrite.p99us": { "dir": "lower", - "floor": 516, + "floor": 292, "tolerance_pct": 50, - "value": 129 + "value": 73 }, "ram.sN.msgrate.msgs_sec": { "dir": "higher", - "floor": 355876, + "floor": 216394, "tolerance_pct": 50, - "value": 2847015 + "value": 1731152 }, "ram.sN.query.ops_sec": { "dir": "higher", - "floor": 307125, + "floor": 340136, "tolerance_pct": 50, - "value": 1228501 + "value": 1360544 }, "ram.sN.query.p50us": { "dir": "lower", @@ -398,9 +578,9 @@ }, "ram.sN.read.ops_sec": { "dir": "higher", - "floor": 340692, + "floor": 343878, "tolerance_pct": 50, - "value": 1362769 + "value": 1375515 }, "ram.sN.read.p50us": { "dir": "lower", @@ -416,38 +596,38 @@ }, "ram.sN.seed.ops_sec": { "dir": "higher", - "floor": 72890, + "floor": 445235, "tolerance_pct": 50, - "value": 291562 + "value": 1780943 }, "ram.sN.seed.p50us": { "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 3 + "value": 0 }, "ram.sN.seed.p99us": { "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 7 + "value": 1 }, "ram.sN.write.ops_sec": { "dir": "higher", - "floor": 60518, + "floor": 286368, "tolerance_pct": 50, - "value": 242072 + "value": 1145475 }, "ram.sN.write.p50us": { "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 6 + "value": 1 }, "ram.sN.write.p99us": { "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 9 + "value": 2 } } \ No newline at end of file diff --git a/docs/examples/db-bench/main.wo b/docs/examples/db-bench/main.wo index bf8a6fe..f20a6ac 100644 --- a/docs/examples/db-bench/main.wo +++ b/docs/examples/db-bench/main.wo @@ -1,3 +1,4 @@ +use fs use time -- db-bench — iteration 22's load generator. Every measured mode prints @@ -465,10 +466,135 @@ fn all_mode(n: Int) -> Int { fn usage() -> Int { print_err("usage: db-bench "); print_err(" all N | seed N | read N | query N | write N | wal N"); - print_err(" mix N C | msgrate N | verify | verify-acked M"); + print_err(" mix N C | msgrate N | growth N int|text | growth-verify"); + print_err(" verify | verify-acked M"); return 2; } +-- databasev2 1: the process's own resident size, in KiB. Read here rather +-- than sampled by the driver because the driver polls /proc every 250 ms and +-- would miss the value AT a decile boundary; per-row footprint is the headline +-- number of this iteration and deserves an exact reading, not a nearby one. +-- Absence is nil by stdlib convention, so a kernel without VmRSS reports 0 +-- and the driver treats the leg as unavailable rather than as zero growth. +fn self_rss_kb() -> Int { + let st = try fs.read_all("/proc/self/status", 16384) catch (e) ""; + let i = index_of(st, "VmRSS:"); + if i < 0 { + return 0; + } + let rest = substr(st, i + 6, 24); + let n = 0; + let j = 0; + while j < len(rest) { + let c = byte_at(rest, j); + if c >= 48 and c <= 57 { + n = n * 10 + (c - 48); + } else { + if n > 0 { + return n; + } + } + j = j + 1; + } + return n; +} + +-- databasev2 1: growth N SHAPE — insert N rows of one reference shape, +-- sampling read latency as the table grows so the driver can plot the CURVE +-- rather than two endpoints. Reports one metric line per decile so the point +-- at which p99 leaves its baseline is a MEASURED sample, not an estimate. +-- +-- SHAPE is "int" (Item: two Ints plus a ref, all inline slot words) or "text" +-- (Wide: three Text columns, each a separate db_text allocation on top of the +-- slab slot). Per-row footprint differs by an order of magnitude between them, +-- which is exactly why the driver reports the two separately and never a single +-- "bytes per row". +-- +-- The memory CAP is the driver's job (systemd-run --user --scope), not this +-- program's: the sample just grows and reports, so the same binary serves the +-- swap-off and swap-on legs unchanged. +-- after the process is OOM-killed mid-insert, the durable prefix must be +-- intact: rows 1..M all present with the right v and no holes. M is whatever +-- survived -- the claim under test is the SHAPE of the survivor, not its size, +-- because a SIGKILL can land between any two inserts. +fn growth_verify() -> Int { + let seen: map = {}; + let maxk = 0; + for r in from x in Item select x { + set(seen, r.k, r.v); + if r.k > maxk { + maxk = r.k; + } + } + let i = 1; + while i <= maxk { + if has(seen, i) == false { + print_err("growth-verify: hole at ${i} below max ${maxk}"); + return 3; + } + if get(seen, i) != item_v(i) { + print_err("growth-verify: row ${i} v ${get(seen, i)} != ${item_v(i)}"); + return 3; + } + i = i + 1; + } + print("growthverify ${maxk}"); + return 0; +} + +fn growth_mode(n: Int, shape: Text) -> Int { + let wide = shape == "text"; + if wide == false and shape != "int" { + print_err("db-bench: growth SHAPE must be `int` or `text`"); + return 2; + } + let step = n / 10; + if step < 1 { + step = 1; + } + let bref = insert Bucket { tag: "growth" }; + let pad = "0123456789abcdef0123456789abcdef"; + let i = 1; + while i <= n { + if wide { + insert Wide { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" }; + } else { + insert Item { k: i, v: item_v(i), bucket: bref }; + } + -- at each decile, sample the read path against what is resident NOW + if i % step == 0 { + let h: map = {}; + let probes = 200; + let pt0 = time.ticks(); + let j = 0; + while j < probes { + let key = 1 + (j * step) % i; + let o0 = time.ticks(); + if wide { + for r in from x in Wide where x.k == key take 1 select x { + hist_add(h, time.ticks() - o0); + } + } else { + for r in from x in Item where x.k == key take 1 select x { + hist_add(h, time.ticks() - o0); + } + } + j = j + 1; + } + let pel = time.ticks() - pt0; + -- op name carries the decile so the driver keys each sample distinctly + report("growth${i / step}", probes, pel, h); + -- rows and resident KiB at this decile: the driver divides to get the + -- per-row footprint for THIS shape + print("growthrss ${i / step} ${i} ${self_rss_kb()}"); + } + i = i + 1; + } + print("growthdone ${n}"); + return 0; +} + fn main(args: multi Text) -> Int { if len(args) < 1 { return usage(); @@ -476,6 +602,9 @@ fn main(args: multi Text) -> Int { if args[0] == "verify" { return verify(); } + if args[0] == "growth-verify" { + return growth_verify(); + } if len(args) < 2 { return usage(); } @@ -508,6 +637,12 @@ fn main(args: multi Text) -> Int { if args[0] == "msgrate" { return msgrate_mode(n); } + if args[0] == "growth" { + if len(args) < 3 { + return usage(); + } + return growth_mode(n, args[2]); + } if args[0] == "mix" { if len(args) < 3 { return usage(); diff --git a/docs/examples/db-bench/types.wo b/docs/examples/db-bench/types.wo index aa1e4cb..25639c2 100644 --- a/docs/examples/db-bench/types.wo +++ b/docs/examples/db-bench/types.wo @@ -23,6 +23,20 @@ class Meta { val: Int } +-- databasev2 1: the TEXT-HEAVY reference shape. `Item` above is the Int-only +-- reference as it stands (two Ints plus a ref, all inline slot words), so this +-- is its counterpart: every row drags a separate db_text allocation per Text +-- column on top of its slab slot. Per-row footprint differs by an order of +-- magnitude between the two, which is why a single "bytes per row" number is +-- meaningless and the growth mode reports the two shapes separately. +@table(name: "wide", index: [k]) +class Wide { + k: Int + a: Text + b: Text + note: Text +} + -- mix actors dump their per-op histograms here (kind 0 = read, -- 1 = write); main scans and merges — exact aggregate percentiles, -- and the merge itself dogfoods the store. diff --git a/docs/plan/perf-targets.md b/docs/plan/perf-targets.md index 9d12aea..983158e 100644 --- a/docs/plan/perf-targets.md +++ b/docs/plan/perf-targets.md @@ -67,3 +67,89 @@ not the limiting factor for any current workload). the ~55× gap is one fdatasync per statement (~220µs each). **Owner: iteration 23** (io_uring group-commit) — its acceptance is literally this number moving while the crash battery stays green. + +## 5. The RAM ceiling: footprint, and how the engine actually dies + +**Measured 2026-08-27** (databasev2 1), rootless cgroup v2 via +`systemd-run --user --scope -p MemoryMax=N -p MemorySwapMax=M`, dev box. + +### Per-row resident footprint, by shape + +| Shape | Columns | Steady-state | Doubling steps | +| --- | --- | --- | --- | +| Int-only (`Item`) | 2× Int + 1 ref | **96.5 B/row** | at ~24k and ~48k rows | +| Text-heavy (`Wide`) | 1× Int + 3× Text | **320.6 B/row** | at ~24k and ~48k rows | + +**3.3×**, not the "order of magnitude" an earlier doc asserted. Two shapes are +published, never one number: a `Text` column is a separate `db_text` allocation +per row on top of the slab slot, so a row count cannot bound RAM. + +**Read the steady-state figure as the median of per-interval marginals, not a +two-point slope.** The id hash and index buckets are open-addressing pow2 and +double periodically; a two-point slope lands arbitrarily on or off a doubling +and swings 2× (96 vs 205 B/row measured for the same shape). The doublings are +reported separately because a **transient RSS step is exactly what a +resident-footprint budget must leave headroom for** — a budget without it fires +during a rehash rather than at a steady-state threshold. Direct input to +databasev2 2's budget design. + +### How it dies — and it is not the way the docs claimed + +| Allocator | Ceiling | Failure mode | +| --- | --- | --- | +| VM object arena | `WO_HEAP_MB`, checked | `trap 4 … out of memory`, rc=1, reportable. Verified at 4 and 16 MiB | +| table storage (slabs + heap values) | **none** | **SIGKILL, signal 9** (shell rc 137). Verified at 360 000 rows / 57 188 KiB under a 64 MiB cap | + +Three docs asserted that an allocation failure surfaces as a catchable +`WO_T_OOM`. For table storage it does not: `vm.overcommit_memory = 0` means +`malloc` succeeds and the kernel kills the process when it *touches* the pages, +so the checked-`malloc` code never runs. The trap path is real, but it is the +arena's. + +**Consequence, and the strongest available argument for databasev2 2's byte +budget:** a declared budget is the *only* way table storage can acquire a +checked ceiling, because `malloc` under default overcommit will never report a +problem. **Owner: databasev2 2.** + +### Swap: the ceiling that does not announce itself + +| Leg | 900 000 Int rows, 64 MiB cap | Wall | Final RSS | +| --- | --- | --- | --- | +| swap OFF (`MemorySwapMax=0`) | **SIGKILL at 360 000 rows** | — | 57 188 KiB | +| swap ON (256 MiB) | **completed, exit 0** | **148 s** | 62 264 KiB (rest paged out) | +| uncapped | completed, exit 0 | **150 s** | 169 416 KiB | + +**Swap cost ~1%.** A prior draft predicted "latency collapse"; the prediction had +the wrong sign. Inserting is append-mostly, so cold pages are written once and +never re-read — paging is sequential and off the critical path. The swap device +is a real disk file (`/swap.img`; no zram, zswap disabled), so this is genuine +disk paging. + +**Do not generalise this to "swap is fine".** It measures an append-mostly +workload. A random-read workload over a table larger than the cap is where the +collapse should appear, and it is **not yet measured** — which matters, because +that is exactly the access pattern databasev2 2's `resident: keys` creates. + +The operational consequence is that the RAM ceiling has two shapes and neither +reports itself: without swap the process vanishes on signal 9, with swap it +keeps returning 0 while serving from disk. A budget that fires at a *declared +threshold* is the only one that can speak before either happens. + +### Durability across the ceiling + +60 000 Int rows, 8 MiB cap, swap off, `WO_DATA` set — the process is OOM-killed +mid-insert, then replayed: + +| Claim | Result | +| --- | --- | +| the survivor is a contiguous prefix | ✅ ~40 000 rows, rows 1..M all present | +| every surviving row's payload is correct | ✅ every `v` matches `item_v(i)` | +| the truncated tail is not read as corruption | ✅ replay exits 0 | + +**Ack-after-fsync holds through an OOM kill** — the one shutdown path that skips +every cleanup handler. Gated as `db-bench`'s `ceiling` leg, which asserts the +*shape* of the survivor rather than its size: where the SIGKILL lands is the +scheduler's business, so `rows_recovered` carries ±100% tolerance. The leg +asserts the exit but never records it as a metric, so that when databasev2 2's +byte budget turns the kill into a checked refusal, the gate does not fail on the +improvement. diff --git a/docs/stories/00-status.md b/docs/stories/00-status.md index abbb9e0..1344dfc 100644 --- a/docs/stories/00-status.md +++ b/docs/stories/00-status.md @@ -67,6 +67,57 @@ behind this board; live Obsidian Dataview views: ## ▶ NEXT PLAN +### Landed 2026-08-27 — databasev2 1, the RAM ceiling measured + +**Implemented last time (2026-08-27):** databasev2 1 refined (three forks +settled) and implemented. A text-heavy `Wide` reference shape beside the +Int-only `Item`; `growth N int|text` in the db-bench sample, reading its OWN +`/proc/self/status` RSS at each decile because the driver's 250 ms poll misses +the value *at* a boundary; `growth-verify`, which asserts the survivor of a +crash is a contiguous intact prefix; and two harness legs — four footprint legs +under a rootless cgroup v2 cap, and a `ceiling` leg that deliberately dies at +the cap and then replays. 121 checks, 0 failures. + +**Key findings (measured, not asserted):** per-row footprint is **96.5–100 B** +Int-only and **320.6–324 B** text-heavy — **3.3×**, not the "order of magnitude" +three docs asserted. Read as the median of per-decile marginals, never a +two-point slope: index doublings make a two-point read swing 2× (96 vs 205 B/row +for one shape). **Two predictions in the iteration's own premise were wrong.** +The ceiling is not a catchable `WO_T_OOM` for table storage — it is **SIGKILL, +signal 9**, because `vm.overcommit_memory = 0` lets `malloc` succeed and the +kernel kills on page *touch*, so the checked path never runs (the VM arena is +the opposite: `WO_HEAP_MB` is checked and traps). And swap is not "latency +collapse": 900 000 rows inside a 64 MiB cap with swap finished in **148 s +against 150 s uncapped** — ~1%, on a real disk swap file with no zram. Also +measured: **ack-after-fsync holds through an OOM kill** — ~40 000 rows came back +as an intact prefix, no holes, not read as corruption. + +**Learned:** an append-mostly workload never re-touches its cold pages, so swap +costs it nothing — the collapse belongs to *random reads* over an oversized +table, which is precisely the pattern iteration 2's `resident: keys` creates and +is **still unmeasured**. The RAM ceiling therefore has two shapes and neither +announces itself: without swap the process vanishes on signal 9, with swap it +keeps returning 0 while serving from disk. That is the argument for a budget +that fires at a declared threshold instead of at exhaustion. + +**Dependencies unblocked — one, by *removing* it:** iteration 2's +resident-footprint budget default was to be derived from "swap onset". **There is +no onset.** Swap-off jumps straight from working to SIGKILL; swap-on shows no +degradation to detect. Iteration 2 must pick its budget on other grounds rather +than wait on a number this slice cannot produce. Iteration 3's replay baseline is +still NOT delivered — `bench/baseline.json` times no replay. + +**Next steps:** the read-heavy-over-cap leg is the single most valuable +follow-up, and it is what makes `p99_departure_decile` mean anything (the +footprint legs never approach their 512 MiB cap, so it is legitimately 0 today). +Then iteration 2's 5c/5d. + +**`.dev/reference` used:** none. Sources were the kernel's own interfaces — +cgroup v2 `memory.max`/`memory.swap.max`, `/proc/self/status`, `/proc/swaps` and +`vm.overcommit_memory`. + +--- + ### Landed 2026-08-25 — packaging + release pipeline (off-chain, no story) **Implemented last time (2026-08-25):** the toolchain became installable @@ -620,8 +671,11 @@ declares a budget. Rows live in `malloc`'d slabs whose addresses are stable forever; there is no eviction, spill or paging anywhere in `database/src/`; the WAL never checkpoints so boot replays all history; and durability is one process-global `WO_DATA`, so no table can say it matters more than another. An -allocation failure *is* a clean catchable `WO_T_OOM` — but swap thrash arrives -first and carries no error signal at all. +allocation failure is a clean catchable `WO_T_OOM` **only in the VM arena** — +table storage has no ceiling and is SIGKILLed instead (measured, databasev2 1). +Where swap exists the ceiling may never announce itself at all: an append-mostly +900k-row run finished *at uncapped speed* inside a 64 MiB cap (148 s vs 150 s), +serving from disk with no error signal. **The lever** is per-table storage modes, which is why this track has a grammar iteration. Six pending iterations moved here from the language track (their old @@ -630,7 +684,7 @@ the language arc as v1 history. | # | Iteration | State | | --- | --- | --- | -| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | ⬜ `readiness: refine` — its three forks are open, so despite being first it is NOT startable without a brainstorm — nobody here can say what happens at 90% RAM. Curve not cliff: swap onset, latency departure, the three exits (checked trap / swap thrash / OOM killer), and `kill -9` durability *at exhaustion*. Output is `perf-targets.md` + baseline rows, not prose | +| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | 🔄 **MEASURED 2026-08-27** — `readiness: ready`, `status: in-progress` (two criteria outstanding), forks settled, harness landed (121 checks). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Outstanding: the **random-read-over-cap** collapse (unmeasured, and it is `resident: keys`'s own access pattern) and iteration 3's replay baseline. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from | | 2 | [per-table storage: `durable` and `resident`](databasev2/02-table-storage-modes.md) | 🔄 **the language enrichment — the `durable` half is DONE and usable.** Two optional `@table` keys, `durable: true\|false` and `resident: all\|keys`, both defaulting to today's behaviour (all 28 existing declarations compile unchanged, no golden moved). Landed: the grammar, WO-E224 (a durable `ref` into a volatile table is refused), `.wob` v7 carrying both properties in spare `flags` bits, `durable: false` actually skipping the WAL (measured: 50 inserts → 1500 bytes durable, **0** volatile) with a mode-mismatch startup refusal, plus offset capture and read-a-row-from-an-offset. Outstanding: 5c/5d (the id→offset map and rewiring `wo_row_ptr`'s 11 call sites, slab scans and `@unique`/FK across the boundary — not yet written up), the two runtime refusals, and closeout. [spec](../superpowers/specs/2026-08-26-table-residency-design.md) · [plan](../superpowers/plans/2026-08-26-table-residency.md) | | 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ⬜ snapshot + truncate: disk reclaimed, replay bounded | | 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ⬜ **`readiness: ready` — the one startable iteration in the repo** (four forks confirmed settled 2026-08-20). Close the 66× gap iteration 22 measured (durable 4.5k vs ram 297k inserts/s) | diff --git a/docs/stories/databasev2/00-story.md b/docs/stories/databasev2/00-story.md index ba8731e..5d4957a 100644 --- a/docs/stories/databasev2/00-story.md +++ b/docs/stories/databasev2/00-story.md @@ -44,7 +44,14 @@ The bill comes due at the ceiling. Read from the engine as it stands: Worth being precise, because the failure mode determines the fix — and the good news is that the engine's own behaviour is clean: -**An allocation failure is a catchable trap, not a crash.** Every `malloc` in +**Corrected 2026-08-27 by measurement.** This section used to open "an +allocation failure is a catchable trap, not a crash", and that is true only of +the VM arena. Table storage has no ceiling, and with `vm.overcommit_memory = 0` +its `malloc` never fails — the process is **SIGKILLed** (rc=137, measured at +360 000 rows under a 64 MiB cap). The checked path below is real, but it is the +arena's, not the store's. See [iteration 1](01-ram-ceiling-measurement.md). + +Every `malloc` in the row encoder is checked and jumps to an `oom` label; `DB_ERR_OOM` maps to `WO_T_OOM`, which a program can `try`/`catch`. So a writeonce program that runs out of memory *refuses the insert* rather than corrupting or dying. That is a @@ -61,9 +68,22 @@ battery proves that much. So the honest problem statement is not "malloc fails". It is: **there is no declared budget, no back-pressure as the budget is approached, and no way to -distinguish data that must be resident from data that merely is.** Iteration -[1](01-ram-ceiling-measurement.md) exists to replace this paragraph with -numbers before anything is designed on top of it. +distinguish data that must be resident from data that merely is.** + +**Iteration [1](01-ram-ceiling-measurement.md) has now measured this +(2026-08-27), and it strengthened the statement rather than softening it.** A row +costs **96.5–100 B** Int-only and **320.6–324 B** text-heavy (3.3× apart, so no +single per-row number can bound RAM). At the ceiling the engine has exactly two +behaviours and **neither one tells anybody**: without swap the process is +**SIGKILLed on signal 9** — table storage has no checked ceiling, and under +`vm.overcommit_memory = 0` its `malloc` succeeds and the kernel kills on page +touch — and with swap it **keeps returning 0 while serving from disk**, finishing +900 000 rows in 148 s against 150 s uncapped. Durability is the one thing that +does hold: acked writes came back as an intact prefix across an OOM kill. + +That is why "back-pressure at exhaustion" is not a design option. Exhaustion +either kills without warning or never arrives. Only a **declared threshold** can +speak in time. ## The lever: per-table storage modes @@ -113,7 +133,7 @@ before its mechanism existed; the history is in | # | Iteration | Delivers | Needs | | --- | --- | --- | --- | -| 1 | [RAM ceiling: measure the breaking point](01-ram-ceiling-measurement.md) | what actually happens from 50% RAM to OOM — swap onset, latency cliff, trap behaviour, `kill -9` survival | nothing; extends iteration 22's harness | +| 1 | [RAM ceiling: measure the breaking point](01-ram-ceiling-measurement.md) | 🔄 **measured 2026-08-27**: footprint per shape (3.3× apart), the two silent exits (SIGKILL vs swap-serving-from-disk at ~uncapped speed), and ack-after-fsync surviving an OOM kill. Outstanding: the random-read-over-cap collapse, and a replay baseline | nothing; extends iteration 22's harness | | 2 | [per-table storage](02-table-storage-modes.md) | the grammar: `durable: true\|false` and `resident: all\|keys`, per table, replacing the global `WO_DATA` all-or-nothing. **In progress — the `durable` half is done** | 1 for the budget default | | 3 | [WAL checkpoint](03-wal-checkpoint.md) *(was language 32)* | snapshot + truncate: disk reclaimed, replay bounded | 4 composes | | 4 | [io_uring group commit](04-io-uring-commit.md) *(was language 23)* | close the 66× durable/RAM write gap (4.5k vs 297k inserts/s) | the arc (landed) | diff --git a/docs/stories/databasev2/01-ram-ceiling-measurement.md b/docs/stories/databasev2/01-ram-ceiling-measurement.md index 17b47cc..07e7110 100644 --- a/docs/stories/databasev2/01-ram-ceiling-measurement.md +++ b/docs/stories/databasev2/01-ram-ceiling-measurement.md @@ -1,149 +1,263 @@ --- track: databasev2 iteration: "1" -status: pending -readiness: refine +status: in-progress +readiness: ready --- -# databasev2 1 — the RAM ceiling: measure the breaking point before designing for it +# databasev2 1 — the RAM ceiling: measure the breaking point > Part of [Story — databasev2: the database beyond RAM](00-story.md). > +> **Refined 2026-08-27; the three forks are settled below and the decisions are +> locked.** No spec document: the deliverable is numbers plus a harness leg, and +> the design fits in this file — the same call +> [7](07-single-file-db.md) makes. +> > **First because the repo's own doctrine says so.** "Always inspect crashsites. -> Always measure. Never assume." Every later iteration in this track — the -> storage modes' defaults, the eviction policy, the tiering threshold — is a -> decision that should follow from a number. Right now nobody in this project -> can say what happens to a writeonce program at 90% of RAM, and designing -> tiering without that is guessing with extra steps. +> Always measure. Never assume." Two other iterations already cite numbers this +> one was supposed to produce: [2](02-table-storage-modes.md)'s resident-footprint +> budget defaults to a fraction of host memory whose value comes from here, and +> [3](03-wal-checkpoint.md)'s before/after replay criterion has no "before" +> because `bench/baseline.json` carries 75 metrics and **zero** for replay, +> restart, boot or recovery. Iteration 22 proved restart *correctness*; it never +> timed it. -## Goals +## The design, as settled -- **Find the curve, not the cliff.** Not "does it die" — it dies, everything - does. What matters is the shape on the way down: at what fraction of RAM does - p99 read latency leave its 1µs baseline, what does insert throughput do as - slabs stop coming from a warm allocator, and how much warning is there between - "fine" and "unusable". -- **Characterise all three exits.** The engine can leave the happy path three - ways and they are not equally survivable: a checked `malloc` failure - (`DB_ERR_OOM` → `WO_T_OOM`, a catchable trap — the clean one), swap thrash - (no trap, no error, just latency collapse — the dangerous one because nothing - reports it), and the external OOM killer (`SIGKILL`, skipping every shutdown - path). Establish which arrives first under realistic limits, because the - answer determines whether the fix is back-pressure or eviction. -- **Prove the durability floor holds at the ceiling.** Iteration 22's `kill -9` - battery proved acked writes survive under load. Re-run it *at memory - exhaustion*, which is a different and nastier state — an allocation failure - mid-commit is exactly where an ack-before-durable bug would hide. -- **Publish numbers others can build on.** The output is a section in - `perf-targets.md` and rows in `bench/baseline.json`, not a paragraph of - prose. A measurement that only printed once is not a measurement. +**Measure the curve, not the cliff.** Everything dies at the ceiling; what +matters is the shape on the way down — where p99 leaves its 1µs baseline, what +insert throughput does as slabs stop coming from a warm allocator, and how much +warning there is between "fine" and "unusable". -## Phases +### Fork 1 — the limit mechanism: rootless cgroup v2 via `systemd-run --user` -### Phase A — a workload that can actually reach the ceiling +`systemd-run --user --scope -p MemoryMax=N -p MemorySwapMax=M`. Verified on the +dev box: the `memory` controller is delegated to +`user.slice/user-.slice`, a scope's `memory.max` reads back exactly as set, +and no passwordless sudo is needed. Being cgroup-scoped also isolates the +measurement from whatever else the box is doing, which matters — the dev box was +at 22.9 of 31.7 GiB with 4.6 GiB of swap already in use when this was refined. -- Extend `docs/examples/db-bench` with a growth mode: insert until a target RSS - fraction, holding row shape and index count constant so the variable is size - alone. -- Run it under an explicit memory limit (a cgroup or `ulimit`) rather than on a - big box — "it survived on a 64 GB workstation" measures the workstation. -- Record RSS against row count so the per-row overhead is known: slab headroom, - the id hash, the secondary-index multimaps and the per-row engine-owned values - (`db_text`, `db_rec`, `db_multi`, `db_map` are each their own allocation). -- Verify: RSS growth is linear and its slope is written down; the run is - reproducible twice within the tolerance policy iteration 22 established. +`ulimit -v` is **rejected**: it bounds address space, not resident set, which is +the wrong quantity for an engine that `malloc`s slabs — and it is actively +broken under ASan, whose huge virtual reservations trip it long before any real +memory pressure. -### Phase B — the latency and throughput curve +If the mechanism is unavailable (no systemd, no delegation), the harness **skips +the growth legs loudly and names why**. It must never silently fall back to +measuring an uncapped box, because "it survived on a 32 GiB workstation" +measures the workstation. -- Sample read p50/p99, query p99 and insert throughput at fixed fractions of the - limit, so the result is a curve rather than two endpoints. -- Separate the two effects deliberately: allocator pressure (still resident) and - swap (no longer resident). They have different fixes and conflating them would - send iteration 6 after the wrong one. -- Include the DB-actor path, since a cross-shard statement's reply materialises - a copy — memory pressure and the actor RPC interact and nobody has looked. -- Verify: the curve is recorded per metric class with iteration 22's per-class - tolerances; the swap onset point is identified, not interpolated. +### Fork 2 — the reference shapes: both, reported separately -### Phase C — the three exits, deliberately triggered +Per-row footprint differs substantially between an Int-only row and a text-heavy +one, because a `Text` column is a separate `db_text` allocation per row on top of +the slab slot. **Measured 2026-08-27: 96.5 B/row Int-only vs 320.6 B/row with +three Text columns — 3.3×.** An earlier draft of this section said "an order of +magnitude"; that was an unmeasured guess and this iteration exists to replace +exactly that kind of claim. 3.3× is still more than enough to make a single +"bytes per row" number useless, which is the decision it was supporting. -- Drive a checked allocation failure and confirm `WO_T_OOM` is catchable, the - insert is refused whole, no partial row or index entry is left, and the - process continues serving. -- Drive swap thrash and record what a client sees. This is the case with no - error signal at all, and naming it is most of the value of this iteration. -- Drive the OOM killer under a cgroup limit and confirm what survives: replay - the WAL and check every acked write is present. -- Verify: the trap path leaves no torn state (row count and index agree after a - refused insert); replay after `SIGKILL` at exhaustion loses no acked write. +`db-bench` already supplies half of this: `items` (`k: Int`, `v: Int`, plus a +`bucket` ref) is the Int-only reference as it stands. The work is one text-heavy +shape beside it, with footprint reported per shape. -### Phase D — write it down where decisions get made +### Fork 3 — swap: in scope, as a controlled dimension -- A `perf-targets.md` section with the curve, the swap onset, the per-row - overhead and the exit characterisation. -- Baseline rows for the growth metrics so a regression is caught by the existing - gate rather than by a person remembering. -- A short statement of what the numbers *imply* for iterations 2, 5 and 6 — - which is the point of going first. -- Verify: `just db-bench` green against the extended baseline; the gate bites - when a growth metric is doctored. +Not a confound to wish away — `MemorySwapMax` is the knob that separates the two +exits this iteration exists to characterise. Both were measured, and **both +turned out differently than this iteration predicted.** + +| Leg | Predicted | Measured | +| --- | --- | --- | +| swap-off | catchable `WO_T_OOM` from checked `malloc` | **SIGKILL, signal 9** (shell rc 137). No trap, no message | +| swap-on | latency collapse | **no degradation at all**: 148 s vs 150 s uncapped | + +**Prediction 1 was wrong because of overcommit.** With `vm.overcommit_memory = 0` +`malloc` succeeds and the process dies when it *touches* the pages, so table +storage never gets the chance to report failure. The trap path is real but +belongs to a different allocator: + +| Allocator | Ceiling | Failure mode | +| --- | --- | --- | +| VM object arena | `WO_HEAP_MB`, checked | `trap 4` / `WO_T_OOM`, exit 1, reportable | +| table storage (slabs + heap values) | **none** | SIGKILL under overcommit | + +**This is the strongest argument available for [iteration 2](02-table-storage-modes.md)'s +byte budget:** a declared budget is the only way table storage can acquire a +checked ceiling, because `malloc` under default overcommit will never tell it +there is a problem. + +**Prediction 2 was wrong because of access pattern.** 900 000 Int rows under a +64 MiB cap with 256 MiB of swap finished in **148 s** with RSS pinned at 62 MiB; +the same workload uncapped took **150 s** at 165 MiB RSS. Swap cost +approximately nothing. The reason is that inserting is append-mostly: cold pages +are written out once and never read again, so paging is sequential and off the +critical path. The swap is a real disk file (`/swap.img`, no zram, zswap +disabled), so this is genuine disk paging, not compressed RAM. + +**The correct generalisation is narrower than "swap is fine".** This measures an +append-mostly workload. A workload that reads randomly across a table larger +than the cap is the one that collapses, and this iteration did *not* measure +that — see Outstanding. + +## Progress + +| Piece | State | +| --- | --- | +| `Wide` text-heavy reference shape (`db-bench/types.wo`) | ✅ | +| `growth N int\|text` — insert, per-decile RSS and read latency | ✅ | +| the sample reads its OWN RSS via `/proc/self/status` | ✅ — the driver polls every 250 ms and would miss the value *at* a decile boundary | +| `growth-verify` — the survivor is a contiguous intact prefix | ✅ | +| rootless cap wrapper, swap on/off legs | ✅ 4 footprint legs | +| footprint metric = **median of marginals**, doublings counted separately | ✅ | +| `ceiling` leg: dies at the cap, then replay must be intact | ✅ gated | +| baseline + tolerance policy | ✅ 121 checks; footprint at ±10%, kill-timing metrics at ±100% | +| `perf-targets.md` §5 | ✅ | +| **resident-footprint fraction for [iteration 2](02-table-storage-modes.md)** | ⬜ **not delivered — the premise it rested on is false**, see Outstanding | +| **replay/restart baseline for [iteration 3](03-wal-checkpoint.md)** | ⬜ not delivered | +| **the random-read-over-cap collapse** | ⬜ not measured | + +## Measured + +Footprint, reproducible inside 2% across runs: + +| What | Int-only (`Item`) | Text-heavy (`Wide`) | +| --- | --- | --- | +| steady-state footprint | **96.5–100 B/row** | **320.6–324 B/row** | +| doubling steps | 3 (at ~24k and ~48k rows) | 2 | +| base process RSS | ≈ 3.9 MiB, excluded from the per-row figure | same | + +Ratio **3.3×** — not the "order of magnitude" an earlier draft asserted. Enough +on its own to make a single "bytes per row" number useless, which is the decision +it was supporting ([2](02-table-storage-modes.md), fork 5). + +The ceiling, 60 000 Int rows under an 8 MiB cap with swap off and `WO_DATA` set: + +| Question | Answer | +| --- | --- | +| how does it die? | **SIGKILL, signal 9.** No refusal, no diagnostic | +| what survives? | **a contiguous intact prefix** — ~40 000 rows, every `v` correct, no holes, not reported as corruption | + +**Ack-after-fsync holds through an OOM kill.** That is the one shutdown path +which skips every cleanup handler, and the durable prefix came back whole. + +**The finding that matters most is the swap leg succeeding.** It did not fail, +did not warn, and returned 0. A deployment in that state looks healthy while +serving from disk. That is the exit with no error signal, and it is why +[iteration 5](05-bounded-tables-eviction.md)'s back-pressure must act at a +declared threshold rather than at exhaustion — exhaustion either kills without +warning or silently does not arrive. ## Acceptance Criteria -- **Given** the growth workload under a fixed memory limit, **when** it runs - twice, **then** RSS-per-row agrees within the tolerance policy and the slope - is recorded in `perf-targets.md`. -- **Given** the workload at rising RAM fractions, **when** latency is sampled, - **then** the fraction at which read p99 first leaves its baseline is - identified as a measured point, not an estimate. -- **Given** a deliberately induced allocation failure, **when** an insert is - attempted, **then** it traps `WO_T_OOM` catchably, the table's row count is - unchanged, every index agrees with the slab contents, and the process keeps - serving subsequent requests. -- **Given** swap thrash, **when** a client issues reads, **then** the observed - degradation is quantified and the fact that **no error is surfaced** is - recorded explicitly as a finding. -- **Given** a cgroup limit and a workload that exceeds it, **when** the OOM - killer fires, **then** replaying the WAL shows every acked write present — - ack-after-fsync holding in the one shutdown path that skips all cleanup. +Met: + +- **Given** the growth workload under a fixed cap, **when** it runs twice, + **then** RSS-per-row agrees inside tolerance and the slope is recorded per + shape. ✅ inside 2%; `perf-targets.md` §5. +- **Given** the swap-off leg, **when** the cap is exceeded, **then** the exit is + identified and recorded. ✅ **SIGKILL, signal 9** — not the catchable trap this + criterion originally expected, which is the whole point of measuring. The + "process keeps serving" half of the original wording is **void**: nothing + survives a SIGKILL. +- **Given** a cap exceeded with `WO_DATA` set, **when** the process is killed at + exhaustion, **then** replay shows the acked writes present. ✅ ~40 000 rows, + contiguous, no holes, no corruption report. Gated as the `ceiling` leg. +- **Given** the swap-on leg, **when** the same point is reached, **then** the + degradation is quantified **and the absence of any error signal recorded**. + ✅ degradation is **nil** for this workload (148 s vs 150 s uncapped) and the + silence is total. Both halves are findings; the first inverted the prediction. - **Given** the extended baseline, **when** a growth metric is doctored, **then** - `just db-bench` fails on exactly that metric. + the gate fails on exactly that metric. ✅ text footprint +20% → + `FAIL gate.growth.text.noswap.bytes_per_row -- 388 vs baseline 324`, 1 of 104. +- **Given** a host without the cap mechanism, **when** the harness runs, **then** + the legs are skipped with a named reason and the rest still passes. ✅ + `cap_wrapper` returns None unless the `memory` controller is delegated; there + is no uncapped fallback. + +Outstanding: + +- **The resident-footprint fraction for iteration 2's budget default. NOT + delivered, and the premise is false.** It was to be derived from the + swap-onset point — but there is no onset: swap-off jumps straight from + working to SIGKILL, and swap-on shows no degradation to detect an onset in. + **Iteration 2 must pick its budget on other grounds** (host RAM fraction, or + an explicit developer-declared figure) rather than waiting on a number this + iteration cannot produce. This is the most important thing this slice learned + and it removes a dependency rather than satisfying it. +- **The random-read-over-cap collapse.** Not measured. This is where the "latency + collapse" prediction may still be true, and it is the workload that matters + for [2](02-table-storage-modes.md)'s `resident: keys`, whose whole premise is + reading rows back from a log larger than RAM. Needs a read-heavy leg over a + table exceeding the cap. **The single most valuable follow-up.** +- **Given** rising fractions of the cap, **when** latency is sampled, **then** + the p99 departure point is recorded. Partially: the sampler and metric exist + and are gated, but the footprint legs never approach their 512 MiB cap, so + `p99_departure_decile` is legitimately 0 and proves nothing. It becomes + meaningful only with the read-heavy leg above. +- **A replay/restart baseline for iteration 3.** Not delivered; `growth` exercises + `WO_DATA` but nothing times replay. Cheap to add, still absent from + `bench/baseline.json`. ## Out Of Scope -- **Any fix.** This iteration measures. Eviction is - [5](05-bounded-tables-eviction.md), tiering is [6](06-cold-tiering.md), - declared budgets are [2](02-table-storage-modes.md). Shipping a fix inside the - measurement slice would remove the ability to tell whether it helped. -- **Changing the OOM behaviour.** The checked-`malloc`-to-catchable-trap path is - good and should not be touched; if the measurement finds a hole in it, that is - a bug fix, reported separately. -- **A memory profiler or allocator instrumentation.** Observability is language - iteration 30. RSS from the OS and the existing `time.ticks` are enough for a - curve. -- **Multi-machine or sharded-across-hosts scaling.** One binary owns its data; - cross-process is [9](09-cross-program-tables.md). -- **Comparing against SQLite at the ceiling.** `bench/compare/go-sqlite` exists - and the comparison would be interesting, but SQLite's whole architecture is - the paged design this project rejected — the numbers would not inform any - decision here. +- **Any fix.** This measures. Declared budgets are [2](02-table-storage-modes.md), + eviction is [5](05-bounded-tables-eviction.md), tiering is 2's `resident: keys`. +- **Changing the OOM behaviour.** The checked-`malloc` code is untouched. The + measurement showed it is largely unreachable for table storage under default + overcommit — a finding to hand to [2](02-table-storage-modes.md), not a bug to + fix here, and emphatically not a licence to start setting + `vm.overcommit_memory`. +- **A memory profiler or allocator instrumentation** — observability is language + iteration 30. RSS from `/proc` plus `time.ticks` is enough for a curve. +- **Comparing against SQLite at the ceiling.** `bench/compare/go-sqlite` exists, + but SQLite's paged architecture is the design this project rejected, so the + numbers would inform no decision here. +- **Multi-host scaling** — one binary owns its data. -## Info +## Info — the forks, settled -Forks the spec must settle: +1. **The limit mechanism is rootless cgroup v2** via + `systemd-run --user --scope -p MemoryMax -p MemorySwapMax`. `ulimit -v` was + rejected: it bounds address space, not resident set, and ASan's virtual + reservations trip it long before real pressure. No sudo needed; it also + isolates the run from the rest of the box, which mattered — the dev box sat + at 22.9 of 31.7 GiB throughout. +2. **Both reference shapes, reported separately.** 3.3× apart; one number would + be a fiction. +3. **Swap is a dimension, not a footnote** — settled by getting it wrong first. + An early run looked like the cap was unenforced because the process held + 400 MiB inside a 64 MiB limit; it was swapping, which is the phenomenon under + study. +4. **Footprint is read as the median of per-decile marginals**, not a two-point + slope, so a slab doubling does not smear into the per-row figure. Doublings + are counted as their own metric. +5. **Kill-timing metrics carry ±100% tolerance.** `rows_recovered` depends on + where the SIGKILL landed; gating it tightly would be gating the scheduler. + The invariant asserted instead is the *shape* of the survivor. +6. **The ceiling leg asserts `rc`, never records it.** When iteration 2's byte + budget lands, death should become a checked refusal — the gate must not fail + on that improvement. -1. **What is the limit mechanism for the harness?** A cgroup v2 `memory.max` is - the closest thing to how this would actually be deployed; `ulimit -v` is - simpler but bounds address space rather than resident set, which for an engine - that `malloc`s slabs is a materially different constraint. Leaning cgroup, and - the campaign already runs off the fast path so the setup cost is acceptable. -2. **Which table shape is the reference?** Per-row overhead depends heavily on - whether fields are scalars or heap values — a `Text` column is a separate - `db_text` allocation per row, so a text-heavy table and an Int-only table - will produce very different slopes. Probably both, reported separately, - because "bytes per row" is meaningless without saying which row. -3. **Is swap even in scope for the target deployment?** If the intended answer - is "run with swap off and let the OOM killer decide", the swap curve is - informational rather than load-bearing — but that stance should be stated in - the doctrine, not assumed. It also changes which exit iteration 5's - back-pressure is defending against. +## History — four corrections worth keeping + +**"An order of magnitude" was a guess.** The per-shape difference is 3.3×. An +iteration whose purpose is replacing unmeasured claims had one in its own +premise. + +**The clean-exit premise was wrong.** This file and the residency spec both +asserted the ceiling surfaces as a catchable `WO_T_OOM`. It is a SIGKILL. +Overcommit means the allocator never learns there is a problem. + +**The latency-collapse premise was wrong too.** Swap cost ~1% on an +append-mostly workload (148 s vs 150 s). The prediction was not merely +imprecise, it had the wrong sign. The narrower claim that survives is that a +*random-read* workload over an oversized table is the one at risk, and that +remains unmeasured. + +**A SIGKILL was once labelled a "checked refusal"** by the harness, because +`subprocess` reports signal death as a negative `returncode` (`-9`) while the +shell spells the same event `137`. The leg existed specifically to tell those +two apart. Fixed, and the distinction is now spelled out at the comparison. diff --git a/docs/stories/databasev2/02-table-storage-modes.md b/docs/stories/databasev2/02-table-storage-modes.md index 4c5f263..d5c6c7f 100644 --- a/docs/stories/databasev2/02-table-storage-modes.md +++ b/docs/stories/databasev2/02-table-storage-modes.md @@ -154,7 +154,7 @@ Outstanding: resident. Roughly doubles the resident index; stated at the declaration so the cost is visible. 5. **The budget is bytes, not rows** — a text-heavy row and an Int-only row - differ by an order of magnitude, so a row count cannot bound RAM. + differ by 3.3× (measured, databasev2 1), so a row count cannot bound RAM. ## History — two corrections worth keeping diff --git a/docs/superpowers/specs/2026-08-26-table-residency-design.md b/docs/superpowers/specs/2026-08-26-table-residency-design.md index 47121e6..2342168 100644 --- a/docs/superpowers/specs/2026-08-26-table-residency-design.md +++ b/docs/superpowers/specs/2026-08-26-table-residency-design.md @@ -19,7 +19,7 @@ | Which storage architecture | **One engine, log-structured.** The WAL already holds every row; keep an in-RAM id→offset map and read rows back with `pread`. No second engine. | | Row cache | **None in user space.** The kernel page cache is the hot copy — the repo's own stated position in `exploration/postgresql/buffer-and-checkpoint.md`: "`pread` against an fd that already has its page cached is a memcpy… the page cache is the one cache we want", and the reason the engine avoids `O_DIRECT`. | | `@unique` on a non-resident table | **Allowed; its index is unconditionally resident.** Settled here rather than deferred — see Constraints. | -| Budget unit | **Bytes** (estimated resident footprint). Rows is the meaningless unit: a text-heavy row and an Int-only row differ by an order of magnitude, so a row count cannot bound RAM. | +| Budget unit | **Bytes** (estimated resident footprint). Rows is the meaningless unit: a text-heavy row and an Int-only row differ by 3.3× (measured, databasev2 1), so a row count cannot bound RAM. | | Rejected architectures | `mmap` and a buffer pool stay out — see Alternatives rejected. `discarded.md`'s paged-engine rejection is amended to *partly revisited*, not reversed. | ## The problem, read off the engine @@ -37,9 +37,15 @@ Facts, each verified in source rather than assumed: `WO_DATA` is set. `db.c` guards every WAL append with a null check on `vm->rt.wal`, so with no `WO_DATA` **every table is silently volatile** — a program can declare nothing and lose everything. -- An allocation failure is clean: every `malloc` in the row encoder is checked - and `DB_ERR_OOM` maps to `WO_T_OOM`, a catchable trap. The dangerous exit is - the one *before* that — swap thrash, which carries no error signal at all. +- An allocation failure is clean *in principle*: every `malloc` in the row + encoder is checked and `DB_ERR_OOM` maps to `WO_T_OOM`. **Corrected 2026-08-27 + by measurement (databasev2 1): that path does not fire in practice.** With + `vm.overcommit_memory = 0`, `malloc` succeeds and the process is SIGKILLed + when it touches the pages — measured rc=137 at 360 000 rows under a 64 MiB + cgroup cap. The checked-trap path belongs to the VM arena (`WO_HEAP_MB`, + verified `trap 4 ... out of memory`), not to table storage, which has no + ceiling at all. This makes the byte budget below the ONLY mechanism by which + table storage can acquire one. The measurements that bound the design, from iteration 22: durable inserts ≈4.5k/s against RAM ≈297k/s (the 66× fsync gap); reads 1.3M ops/s at p50 1µs diff --git a/scripts/db-bench.py b/scripts/db-bench.py index 9d814cc..930f325 100755 --- a/scripts/db-bench.py +++ b/scripts/db-bench.py @@ -225,6 +225,14 @@ def tolerance_for(key): mix*: scheduling-dependent small counts. read/query + all .sN.*: machine jitter, and at post-index-µs scale a 1µs histogram step on a 7µs p50 is already 14%.""" + # databasev2 1: footprint is a STRUCTURAL number -- 96.5 vs 320.6 B/row + # reproduced to <2% across runs -- so it gets a tight tolerance and is the + # one growth metric worth gating. The doubling COUNT and the latency + # samples are allowed to move: doublings depend on where N lands relative + # to a pow2 rehash, and at 1us p50 a single histogram step is already 100%. + if ".bytes_per_row" in key: return 10 + if key.startswith("growth."): return 100 + if key.startswith("ceiling."): return 100 if ".mixread." in key or ".mixwrite." in key: return 50 if ".sN." in key: return 50 if ".read." in key or ".query." in key: return 50 @@ -248,6 +256,183 @@ def write_baseline(metrics): json.dump(base, open(BASELINE, "w"), indent=1, sort_keys=True) ok(f"baseline written ({len(base) - 1} metrics)") +# ---- databasev2 1: the RAM ceiling ----------------------------------------- + +GROWTH_N = 20000 if QUICK else 200000 +GROWTH_SHAPES = ("int", "text") + + +def cap_wrapper(mem_mb, swap_mb): + """systemd-run --user --scope argv prefix that caps memory rootlessly, or + None when the mechanism is unavailable. + + cgroup v2 with the `memory` controller delegated to the user slice is the + only mechanism used. `ulimit -v` is deliberately NOT a fallback: it bounds + address space, not resident set, which is the wrong quantity for an engine + that mallocs slabs, and ASan's huge virtual reservations trip it long + before real memory pressure. When the cap is unavailable the legs are + SKIPPED and say so -- never silently run uncapped, because "it survived on + a 32 GiB workstation" measures the workstation.""" + if not shutil.which("systemd-run"): + return None + try: + with open("/proc/self/cgroup") as f: + mine = f.readline().strip().split(":")[-1] + ctl = f"/sys/fs/cgroup{os.path.dirname(mine)}/cgroup.controllers" + if "memory" not in open(ctl).read().split(): + return None + except OSError: + return None + return ["systemd-run", "--user", "--scope", "--quiet", + "-p", f"MemoryMax={mem_mb}M", "-p", f"MemorySwapMax={swap_mb}M", "--"] + + +def parse_growth(lines): + """(rows, rss_kb) samples plus per-decile read p50/p99, from the sample's + own `growthrss` / `growthN` lines. RSS is read by the SAMPLE, not polled + here: the driver polls every 250 ms and would miss the value AT a decile + boundary, and per-row footprint is this iteration's headline number.""" + pts, lat = [], {} + for l in lines: + f = l.split() + if f and f[0] == "growthrss" and len(f) == 4: + pts.append((int(f[2]), int(f[3]))) + elif f and f[0].startswith("growth") and len(f) == 5 and f[0][6:].isdigit(): + lat[int(f[0][6:])] = (int(f[3]), int(f[4])) + return pts, lat + + +def bytes_per_row(pts): + """Steady-state marginal footprint = MEDIAN of the per-interval marginals. + + Not a two-point slope: the id hash and index buckets are open-addressing + pow2 and DOUBLE periodically, so a two-point slope lands arbitrarily on or + off a doubling and swings 2x (measured: 96 vs 205 B/row for the same shape). + The median rejects those steps; they are reported separately as `doublings` + because a transient RSS step is exactly what a resident-footprint budget + must leave headroom for.""" + marg = sorted((k1 - k0) * 1024.0 / (r1 - r0) + for (r0, k0), (r1, k1) in zip(pts, pts[1:]) if r1 > r0) + if not marg: + return None, 0 + med = marg[len(marg) // 2] + doublings = sum(1 for m in marg if m > med * 1.5) + return med, doublings + + +def growth(metrics): + """Per-shape footprint and the read-latency curve, under a rootless cap, + with swap ON and OFF. + + What this leg actually measures is FOOTPRINT. It does not reach the cap: + GROWTH_N rows need far less than the 512 MiB cap, so both swap legs are + identical by construction and p99_departure_decile is legitimately 0. + The ceiling itself is ceiling() below -- keep the two separate, because a + footprint regression and a ceiling-behaviour change are different faults. + + Two earlier claims in this docstring were measured FALSE and are recorded + in docs/stories/databasev2/01-ram-ceiling-measurement.md: swap-off is not + a "clean checked-malloc" path (it is SIGKILL, rc=137), and swap-on is not + "latency collapse" (900k rows finished in 148s capped-with-swap vs 150s + uncapped -- an append-mostly workload never re-touches its cold pages).""" + wrap = cap_wrapper(512, 0) + if wrap is None: + ok("growth: SKIPPED -- no rootless cgroup v2 memory cap on this host") + metrics["growth.available"] = 0 + return + metrics["growth.available"] = 1 + for shape in GROWTH_SHAPES: + for legname, swap_mb in (("noswap", 0), ("swap", 256)): + w = cap_wrapper(512, swap_mb) + env = dict(os.environ) + argv = w + [BIN, "growth", str(GROWTH_N), shape] + pr = subprocess.run(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, + text=True, env=env, timeout=900) + lines = pr.stdout.splitlines() + pts, lat = parse_growth(lines) + key = f"growth.{shape}.{legname}" + if not pts: + bad(f"{key}: produced no samples", (lines[-1] if lines else "no output")) + continue + bpr, doublings = bytes_per_row(pts) + metrics[f"{key}.bytes_per_row"] = int(round(bpr)) + metrics[f"{key}.doublings"] = doublings + metrics[f"{key}.rows"] = pts[-1][0] + metrics[f"{key}.rss_kb"] = pts[-1][1] + if lat: + last = max(lat) + metrics[f"{key}.read_p50us"] = lat[last][0] + metrics[f"{key}.read_p99us"] = lat[last][1] + # the curve's departure point: first decile whose p99 exceeds + # 4x the first decile's, as a MEASURED sample not an estimate + first = lat[min(lat)][1] + dep = next((d for d in sorted(lat) if lat[d][1] > max(first, 1) * 4), 0) + metrics[f"{key}.p99_departure_decile"] = dep + ok(f"{key}: {int(round(bpr))} B/row steady, {doublings} doubling step(s), " + f"{pts[-1][0]} rows in {pts[-1][1]} KiB") + + + +CEIL_N, CEIL_CAP_MB = 60000, 8 + +def ceiling(metrics): + """The ceiling itself, and the durability claim across it. + + Sized so the process CANNOT fit: 60k Int rows need ~9.7 MiB resident + (96.5 B/row measured, plus a ~3.9 MiB base) under an 8 MiB cap, swap off. + Two things are under test and the second is the one that matters: + + 1. HOW it dies. Measured: SIGKILL, rc=137 -- not a refusal. Table + storage has no checked ceiling, and under vm.overcommit_memory=0 + malloc succeeds and the process dies TOUCHING the pages, so it never + gets the chance to report failure. (The VM object arena is the + opposite: WO_HEAP_MB is checked and traps.) rc is asserted, not + recorded as a metric -- when databasev2 2's byte budget lands this + should become a checked refusal, and the gate must not fail on that + improvement. + + 2. WHAT SURVIVES. With WO_DATA set, replay must yield a contiguous + intact prefix: rows 1..M present with the right v, no holes, and not + reported as corruption. M is wherever the kill landed -- the SHAPE of + the survivor is the claim, not its size, so rows_recovered carries a + wide tolerance. This is ack-after-fsync holding in the one shutdown + path that skips every cleanup handler.""" + wrap = cap_wrapper(CEIL_CAP_MB, 0) + if wrap is None: + ok("ceiling: SKIPPED -- no rootless cgroup v2 memory cap on this host") + return + data = os.path.join(ROOT, "bench", f"tmp.{os.getpid()}.ceiling") + shutil.rmtree(data, ignore_errors=True); os.makedirs(data, exist_ok=True) + env = dict(os.environ); env["WO_DATA"] = data + pr = subprocess.run(wrap + [BIN, "growth", str(CEIL_N), "int"], + stdout=subprocess.PIPE, stderr=subprocess.STDOUT, + text=True, env=env, timeout=900) + if pr.returncode == 0: + bad("ceiling: process SURVIVED the cap", + f"{CEIL_N} rows fit under {CEIL_CAP_MB} MiB -- footprint changed, resize the leg") + shutil.rmtree(data, ignore_errors=True); return + # subprocess returncode is NEGATIVE for signal death (-9 = SIGKILL); 137 + # is the SHELL spelling of the same event (128+9). Getting this backwards + # once labelled a SIGKILL as a "checked refusal", which is the exact + # distinction this leg exists to report. + if pr.returncode < 0: + sig = -pr.returncode + how = f"killed by signal {sig}" + (" (SIGKILL -- no checked refusal)" if sig == 9 else "") + else: + how = f"exited {pr.returncode} (checked refusal)" + ok(f"ceiling: died at the cap, {how}") + vr = subprocess.run([BIN, "growth-verify"], stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, text=True, env=env, timeout=900) + m = re.search(r"^growthverify (\d+)$", vr.stdout, re.M) + if vr.returncode == 0 and m and int(m.group(1)) > 0: + metrics["ceiling.rows_recovered"] = int(m.group(1)) + ok(f"ceiling: durable prefix intact across the kill -- {m.group(1)} rows, no holes") + else: + bad("ceiling: durable prefix broken across the kill", + (vr.stdout.strip().splitlines() or ["no output"])[-1][:160]) + shutil.rmtree(data, ignore_errors=True) + + def main(): # --check : gate-only evaluation of a recorded run — the # gate-bites smoke doctors a copy and this mode must FAIL on it @@ -260,6 +445,8 @@ def main(): build() metrics = campaign() durability(metrics) + growth(metrics) + ceiling(metrics) os.makedirs(RESULTS_DIR, exist_ok=True) stamp = time.strftime("%Y%m%d-%H%M%S") out = os.path.join(RESULTS_DIR, f"run-{stamp}{'-quick' if QUICK else ''}.json")