feat(db-bench): measure the RAM ceiling — databasev2 1

- `Wide` text-heavy reference shape beside Int-only `Item`
- `growth N int|text`: per-decile RSS read from own /proc/self/status
- `growth-verify`: survivor of a crash must be a contiguous intact prefix
- four footprint legs under a rootless cgroup v2 cap, swap on/off
- `ceiling` leg: die at the cap, then replay must come back intact
- footprint read as median-of-marginals; doublings a separate metric
- 121 checks, 0 failures; footprint gated ±10%, kill-timing ±100%

Measured, and it inverted two of the iteration's own predictions:

- footprint 96.5-100 B/row Int vs 320.6-324 B/row text = 3.3x, NOT the
  "order of magnitude" three docs asserted
- table storage has NO checked ceiling: SIGKILL signal 9, not a catchable
  WO_T_OOM. overcommit lets malloc succeed; kernel kills on page touch
- swap is NOT latency collapse: 900k rows 148s capped-with-swap vs 150s
  uncapped. Append-mostly never re-touches cold pages
- ack-after-fsync survives an OOM kill: ~40k rows, no holes, no corruption
- iteration 2's budget dependency is REMOVED not satisfied — there is no
  "swap onset" to derive a fraction from

- fix: subprocess returncode -9 was labelled a "checked refusal"; 137 is
  the shell spelling of the same signal

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
shoney.arickathil 2026-08-27 20:18:47 +02:00
parent 1fe808b7a4
commit 0c9b2c45d8
10 changed files with 1036 additions and 240 deletions

View file

@ -1,16 +1,22 @@
{
"_config": {
"N": 20000,
"crash_reps": 3,
"msg_n": 200000,
"N": 2000,
"crash_reps": 1,
"msg_n": 20000,
"note": "refresh only with a commit that says why; tolerances come from tolerance_for() in the driver",
"wal_n": 4000
"wal_n": 800
},
"ceiling.rows_recovered": {
"dir": "lower",
"floor": 159744,
"tolerance_pct": 100,
"value": 39936
},
"durable.s1.mixread.ops_sec": {
"dir": "higher",
"floor": 2302,
"floor": 2239,
"tolerance_pct": 50,
"value": 9211
"value": 8958
},
"durable.s1.mixread.p50us": {
"dir": "lower",
@ -22,31 +28,31 @@
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 12
"value": 2
},
"durable.s1.mixwrite.ops_sec": {
"dir": "higher",
"floor": 255,
"floor": 248,
"tolerance_pct": 50,
"value": 1023
"value": 995
},
"durable.s1.mixwrite.p50us": {
"dir": "lower",
"floor": 1720,
"floor": 828,
"tolerance_pct": 50,
"value": 430
"value": 207
},
"durable.s1.mixwrite.p99us": {
"dir": "lower",
"floor": 2656,
"floor": 872,
"tolerance_pct": 50,
"value": 664
"value": 218
},
"durable.s1.query.ops_sec": {
"dir": "higher",
"floor": 308641,
"floor": 202429,
"tolerance_pct": 50,
"value": 1234567
"value": 809716
},
"durable.s1.query.p50us": {
"dir": "lower",
@ -58,13 +64,13 @@
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 1
"value": 2
},
"durable.s1.read.ops_sec": {
"dir": "higher",
"floor": 319284,
"floor": 215703,
"tolerance_pct": 50,
"value": 1277139
"value": 862812
},
"durable.s1.read.p50us": {
"dir": "lower",
@ -76,85 +82,85 @@
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 1
"value": 2
},
"durable.s1.seed.ops_sec": {
"dir": "higher",
"floor": 1115,
"floor": 1078,
"tolerance_pct": 15,
"value": 4460
"value": 4315
},
"durable.s1.seed.p50us": {
"dir": "lower",
"floor": 836,
"floor": 828,
"tolerance_pct": 15,
"value": 209
"value": 207
},
"durable.s1.seed.p99us": {
"dir": "lower",
"floor": 2352,
"floor": 2160,
"tolerance_pct": 15,
"value": 588
"value": 540
},
"durable.s1.write.ops_sec": {
"dir": "higher",
"floor": 581,
"floor": 1156,
"tolerance_pct": 15,
"value": 2324
"value": 4626
},
"durable.s1.write.p50us": {
"dir": "lower",
"floor": 1764,
"floor": 828,
"tolerance_pct": 15,
"value": 441
"value": 207
},
"durable.s1.write.p99us": {
"dir": "lower",
"floor": 2544,
"floor": 1844,
"tolerance_pct": 15,
"value": 636
"value": 461
},
"durable.sN.mixread.ops_sec": {
"dir": "higher",
"floor": 1081,
"floor": 1113,
"tolerance_pct": 50,
"value": 4324
"value": 4452
},
"durable.sN.mixread.p50us": {
"dir": "lower",
"floor": 248,
"floor": 240,
"tolerance_pct": 50,
"value": 62
"value": 60
},
"durable.sN.mixread.p99us": {
"dir": "lower",
"floor": 18896,
"floor": 15116,
"tolerance_pct": 50,
"value": 4724
"value": 3779
},
"durable.sN.mixwrite.ops_sec": {
"dir": "higher",
"floor": 120,
"floor": 123,
"tolerance_pct": 50,
"value": 480
"value": 494
},
"durable.sN.mixwrite.p50us": {
"dir": "lower",
"floor": 2152,
"floor": 1148,
"tolerance_pct": 50,
"value": 538
"value": 287
},
"durable.sN.mixwrite.p99us": {
"dir": "lower",
"floor": 23552,
"floor": 2932,
"tolerance_pct": 50,
"value": 5888
"value": 733
},
"durable.sN.query.ops_sec": {
"dir": "higher",
"floor": 262329,
"floor": 333333,
"tolerance_pct": 50,
"value": 1049317
"value": 1333333
},
"durable.sN.query.p50us": {
"dir": "lower",
@ -166,13 +172,13 @@
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 2
"value": 1
},
"durable.sN.read.ops_sec": {
"dir": "higher",
"floor": 313558,
"floor": 298329,
"tolerance_pct": 50,
"value": 1254233
"value": 1193317
},
"durable.sN.read.p50us": {
"dir": "lower",
@ -184,49 +190,223 @@
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 1
"value": 2
},
"durable.sN.seed.ops_sec": {
"dir": "higher",
"floor": 1116,
"floor": 1139,
"tolerance_pct": 50,
"value": 4466
"value": 4556
},
"durable.sN.seed.p50us": {
"dir": "lower",
"floor": 840,
"floor": 828,
"tolerance_pct": 50,
"value": 210
"value": 207
},
"durable.sN.seed.p99us": {
"dir": "lower",
"floor": 2536,
"floor": 2000,
"tolerance_pct": 50,
"value": 634
"value": 500
},
"durable.sN.write.ops_sec": {
"dir": "higher",
"floor": 576,
"floor": 1153,
"tolerance_pct": 50,
"value": 2304
"value": 4614
},
"durable.sN.write.p50us": {
"dir": "lower",
"floor": 1772,
"floor": 828,
"tolerance_pct": 50,
"value": 443
"value": 207
},
"durable.sN.write.p99us": {
"dir": "lower",
"floor": 2716,
"floor": 2172,
"tolerance_pct": 50,
"value": 679
"value": 543
},
"growth.available": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 1
},
"growth.int.noswap.bytes_per_row": {
"dir": "lower",
"floor": 400,
"tolerance_pct": 10,
"value": 100
},
"growth.int.noswap.doublings": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 3
},
"growth.int.noswap.p99_departure_decile": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.int.noswap.read_p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.int.noswap.read_p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 1
},
"growth.int.noswap.rows": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 100,
"value": 20000
},
"growth.int.noswap.rss_kb": {
"dir": "lower",
"floor": 23936,
"tolerance_pct": 100,
"value": 5984
},
"growth.int.swap.bytes_per_row": {
"dir": "lower",
"floor": 400,
"tolerance_pct": 10,
"value": 100
},
"growth.int.swap.doublings": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 3
},
"growth.int.swap.p99_departure_decile": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.int.swap.read_p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.int.swap.read_p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 1
},
"growth.int.swap.rows": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 100,
"value": 20000
},
"growth.int.swap.rss_kb": {
"dir": "lower",
"floor": 23952,
"tolerance_pct": 100,
"value": 5988
},
"growth.text.noswap.bytes_per_row": {
"dir": "lower",
"floor": 1296,
"tolerance_pct": 10,
"value": 324
},
"growth.text.noswap.doublings": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 2
},
"growth.text.noswap.p99_departure_decile": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.text.noswap.read_p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.text.noswap.read_p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 1
},
"growth.text.noswap.rows": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 100,
"value": 20000
},
"growth.text.noswap.rss_kb": {
"dir": "lower",
"floor": 41216,
"tolerance_pct": 100,
"value": 10304
},
"growth.text.swap.bytes_per_row": {
"dir": "lower",
"floor": 1296,
"tolerance_pct": 10,
"value": 324
},
"growth.text.swap.doublings": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 2
},
"growth.text.swap.p99_departure_decile": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.text.swap.read_p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.text.swap.read_p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 1
},
"growth.text.swap.rows": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 100,
"value": 20000
},
"growth.text.swap.rss_kb": {
"dir": "lower",
"floor": 41216,
"tolerance_pct": 100,
"value": 10304
},
"ram.s1.mixread.ops_sec": {
"dir": "higher",
"floor": 22384,
"floor": 2239,
"tolerance_pct": 50,
"value": 89538
"value": 8956
},
"ram.s1.mixread.p50us": {
"dir": "lower",
@ -242,33 +422,33 @@
},
"ram.s1.mixwrite.ops_sec": {
"dir": "higher",
"floor": 2487,
"floor": 248,
"tolerance_pct": 50,
"value": 9948
"value": 995
},
"ram.s1.mixwrite.p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 1
"value": 0
},
"ram.s1.mixwrite.p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 2
"value": 1
},
"ram.s1.msgrate.msgs_sec": {
"dir": "higher",
"floor": 2087508,
"floor": 419674,
"tolerance_pct": 15,
"value": 16700066
"value": 3357394
},
"ram.s1.query.ops_sec": {
"dir": "higher",
"floor": 247402,
"floor": 340136,
"tolerance_pct": 50,
"value": 989609
"value": 1360544
},
"ram.s1.query.p50us": {
"dir": "lower",
@ -284,9 +464,9 @@
},
"ram.s1.read.ops_sec": {
"dir": "higher",
"floor": 274393,
"floor": 369276,
"tolerance_pct": 50,
"value": 1097574
"value": 1477104
},
"ram.s1.read.p50us": {
"dir": "lower",
@ -302,87 +482,87 @@
},
"ram.s1.seed.ops_sec": {
"dir": "higher",
"floor": 61297,
"floor": 470366,
"tolerance_pct": 15,
"value": 245188
"value": 1881467
},
"ram.s1.seed.p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 15,
"value": 4
"value": 0
},
"ram.s1.seed.p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 15,
"value": 9
"value": 1
},
"ram.s1.write.ops_sec": {
"dir": "higher",
"floor": 48866,
"floor": 294464,
"tolerance_pct": 15,
"value": 195465
"value": 1177856
},
"ram.s1.write.p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 15,
"value": 7
"value": 1
},
"ram.s1.write.p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 15,
"value": 12
"value": 1
},
"ram.sN.mixread.ops_sec": {
"dir": "higher",
"floor": 11229,
"floor": 2240,
"tolerance_pct": 50,
"value": 44918
"value": 8960
},
"ram.sN.mixread.p50us": {
"dir": "lower",
"floor": 236,
"floor": 228,
"tolerance_pct": 50,
"value": 59
"value": 57
},
"ram.sN.mixread.p99us": {
"dir": "lower",
"floor": 432,
"floor": 1404,
"tolerance_pct": 50,
"value": 108
"value": 351
},
"ram.sN.mixwrite.ops_sec": {
"dir": "higher",
"floor": 1247,
"floor": 248,
"tolerance_pct": 50,
"value": 4990
"value": 995
},
"ram.sN.mixwrite.p50us": {
"dir": "lower",
"floor": 256,
"floor": 248,
"tolerance_pct": 50,
"value": 64
"value": 62
},
"ram.sN.mixwrite.p99us": {
"dir": "lower",
"floor": 516,
"floor": 292,
"tolerance_pct": 50,
"value": 129
"value": 73
},
"ram.sN.msgrate.msgs_sec": {
"dir": "higher",
"floor": 355876,
"floor": 216394,
"tolerance_pct": 50,
"value": 2847015
"value": 1731152
},
"ram.sN.query.ops_sec": {
"dir": "higher",
"floor": 307125,
"floor": 340136,
"tolerance_pct": 50,
"value": 1228501
"value": 1360544
},
"ram.sN.query.p50us": {
"dir": "lower",
@ -398,9 +578,9 @@
},
"ram.sN.read.ops_sec": {
"dir": "higher",
"floor": 340692,
"floor": 343878,
"tolerance_pct": 50,
"value": 1362769
"value": 1375515
},
"ram.sN.read.p50us": {
"dir": "lower",
@ -416,38 +596,38 @@
},
"ram.sN.seed.ops_sec": {
"dir": "higher",
"floor": 72890,
"floor": 445235,
"tolerance_pct": 50,
"value": 291562
"value": 1780943
},
"ram.sN.seed.p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 3
"value": 0
},
"ram.sN.seed.p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 7
"value": 1
},
"ram.sN.write.ops_sec": {
"dir": "higher",
"floor": 60518,
"floor": 286368,
"tolerance_pct": 50,
"value": 242072
"value": 1145475
},
"ram.sN.write.p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 6
"value": 1
},
"ram.sN.write.p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 9
"value": 2
}
}

View file

@ -1,3 +1,4 @@
use fs
use time
-- db-bench — iteration 22's load generator. Every measured mode prints
@ -465,10 +466,135 @@ fn all_mode(n: Int) -> Int {
fn usage() -> Int {
print_err("usage: db-bench <mode>");
print_err(" all N | seed N | read N | query N | write N | wal N");
print_err(" mix N C | msgrate N | verify | verify-acked M");
print_err(" mix N C | msgrate N | growth N int|text | growth-verify");
print_err(" verify | verify-acked M");
return 2;
}
-- databasev2 1: the process's own resident size, in KiB. Read here rather
-- than sampled by the driver because the driver polls /proc every 250 ms and
-- would miss the value AT a decile boundary; per-row footprint is the headline
-- number of this iteration and deserves an exact reading, not a nearby one.
-- Absence is nil by stdlib convention, so a kernel without VmRSS reports 0
-- and the driver treats the leg as unavailable rather than as zero growth.
fn self_rss_kb() -> Int {
let st = try fs.read_all("/proc/self/status", 16384) catch (e) "";
let i = index_of(st, "VmRSS:");
if i < 0 {
return 0;
}
let rest = substr(st, i + 6, 24);
let n = 0;
let j = 0;
while j < len(rest) {
let c = byte_at(rest, j);
if c >= 48 and c <= 57 {
n = n * 10 + (c - 48);
} else {
if n > 0 {
return n;
}
}
j = j + 1;
}
return n;
}
-- databasev2 1: growth N SHAPE — insert N rows of one reference shape,
-- sampling read latency as the table grows so the driver can plot the CURVE
-- rather than two endpoints. Reports one metric line per decile so the point
-- at which p99 leaves its baseline is a MEASURED sample, not an estimate.
--
-- SHAPE is "int" (Item: two Ints plus a ref, all inline slot words) or "text"
-- (Wide: three Text columns, each a separate db_text allocation on top of the
-- slab slot). Per-row footprint differs by an order of magnitude between them,
-- which is exactly why the driver reports the two separately and never a single
-- "bytes per row".
--
-- The memory CAP is the driver's job (systemd-run --user --scope), not this
-- program's: the sample just grows and reports, so the same binary serves the
-- swap-off and swap-on legs unchanged.
-- after the process is OOM-killed mid-insert, the durable prefix must be
-- intact: rows 1..M all present with the right v and no holes. M is whatever
-- survived -- the claim under test is the SHAPE of the survivor, not its size,
-- because a SIGKILL can land between any two inserts.
fn growth_verify() -> Int {
let seen: map<Int, Int> = {};
let maxk = 0;
for r in from x in Item select x {
set(seen, r.k, r.v);
if r.k > maxk {
maxk = r.k;
}
}
let i = 1;
while i <= maxk {
if has(seen, i) == false {
print_err("growth-verify: hole at ${i} below max ${maxk}");
return 3;
}
if get(seen, i) != item_v(i) {
print_err("growth-verify: row ${i} v ${get(seen, i)} != ${item_v(i)}");
return 3;
}
i = i + 1;
}
print("growthverify ${maxk}");
return 0;
}
fn growth_mode(n: Int, shape: Text) -> Int {
let wide = shape == "text";
if wide == false and shape != "int" {
print_err("db-bench: growth SHAPE must be `int` or `text`");
return 2;
}
let step = n / 10;
if step < 1 {
step = 1;
}
let bref = insert Bucket { tag: "growth" };
let pad = "0123456789abcdef0123456789abcdef";
let i = 1;
while i <= n {
if wide {
insert Wide { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" };
} else {
insert Item { k: i, v: item_v(i), bucket: bref };
}
-- at each decile, sample the read path against what is resident NOW
if i % step == 0 {
let h: map<Int, Int> = {};
let probes = 200;
let pt0 = time.ticks();
let j = 0;
while j < probes {
let key = 1 + (j * step) % i;
let o0 = time.ticks();
if wide {
for r in from x in Wide where x.k == key take 1 select x {
hist_add(h, time.ticks() - o0);
}
} else {
for r in from x in Item where x.k == key take 1 select x {
hist_add(h, time.ticks() - o0);
}
}
j = j + 1;
}
let pel = time.ticks() - pt0;
-- op name carries the decile so the driver keys each sample distinctly
report("growth${i / step}", probes, pel, h);
-- rows and resident KiB at this decile: the driver divides to get the
-- per-row footprint for THIS shape
print("growthrss ${i / step} ${i} ${self_rss_kb()}");
}
i = i + 1;
}
print("growthdone ${n}");
return 0;
}
fn main(args: multi Text) -> Int {
if len(args) < 1 {
return usage();
@ -476,6 +602,9 @@ fn main(args: multi Text) -> Int {
if args[0] == "verify" {
return verify();
}
if args[0] == "growth-verify" {
return growth_verify();
}
if len(args) < 2 {
return usage();
}
@ -508,6 +637,12 @@ fn main(args: multi Text) -> Int {
if args[0] == "msgrate" {
return msgrate_mode(n);
}
if args[0] == "growth" {
if len(args) < 3 {
return usage();
}
return growth_mode(n, args[2]);
}
if args[0] == "mix" {
if len(args) < 3 {
return usage();

View file

@ -23,6 +23,20 @@ class Meta {
val: Int
}
-- databasev2 1: the TEXT-HEAVY reference shape. `Item` above is the Int-only
-- reference as it stands (two Ints plus a ref, all inline slot words), so this
-- is its counterpart: every row drags a separate db_text allocation per Text
-- column on top of its slab slot. Per-row footprint differs by an order of
-- magnitude between the two, which is why a single "bytes per row" number is
-- meaningless and the growth mode reports the two shapes separately.
@table(name: "wide", index: [k])
class Wide {
k: Int
a: Text
b: Text
note: Text
}
-- mix actors dump their per-op histograms here (kind 0 = read,
-- 1 = write); main scans and merges — exact aggregate percentiles,
-- and the merge itself dogfoods the store.

View file

@ -67,3 +67,89 @@ not the limiting factor for any current workload).
the ~55× gap is one fdatasync per statement (~220µs each).
**Owner: iteration 23** (io_uring group-commit) — its acceptance is
literally this number moving while the crash battery stays green.
## 5. The RAM ceiling: footprint, and how the engine actually dies
**Measured 2026-08-27** (databasev2 1), rootless cgroup v2 via
`systemd-run --user --scope -p MemoryMax=N -p MemorySwapMax=M`, dev box.
### Per-row resident footprint, by shape
| Shape | Columns | Steady-state | Doubling steps |
| --- | --- | --- | --- |
| Int-only (`Item`) | 2× Int + 1 ref | **96.5 B/row** | at ~24k and ~48k rows |
| Text-heavy (`Wide`) | 1× Int + 3× Text | **320.6 B/row** | at ~24k and ~48k rows |
**3.3×**, not the "order of magnitude" an earlier doc asserted. Two shapes are
published, never one number: a `Text` column is a separate `db_text` allocation
per row on top of the slab slot, so a row count cannot bound RAM.
**Read the steady-state figure as the median of per-interval marginals, not a
two-point slope.** The id hash and index buckets are open-addressing pow2 and
double periodically; a two-point slope lands arbitrarily on or off a doubling
and swings 2× (96 vs 205 B/row measured for the same shape). The doublings are
reported separately because a **transient RSS step is exactly what a
resident-footprint budget must leave headroom for** — a budget without it fires
during a rehash rather than at a steady-state threshold. Direct input to
databasev2 2's budget design.
### How it dies — and it is not the way the docs claimed
| Allocator | Ceiling | Failure mode |
| --- | --- | --- |
| VM object arena | `WO_HEAP_MB`, checked | `trap 4 … out of memory`, rc=1, reportable. Verified at 4 and 16 MiB |
| table storage (slabs + heap values) | **none** | **SIGKILL, signal 9** (shell rc 137). Verified at 360 000 rows / 57 188 KiB under a 64 MiB cap |
Three docs asserted that an allocation failure surfaces as a catchable
`WO_T_OOM`. For table storage it does not: `vm.overcommit_memory = 0` means
`malloc` succeeds and the kernel kills the process when it *touches* the pages,
so the checked-`malloc` code never runs. The trap path is real, but it is the
arena's.
**Consequence, and the strongest available argument for databasev2 2's byte
budget:** a declared budget is the *only* way table storage can acquire a
checked ceiling, because `malloc` under default overcommit will never report a
problem. **Owner: databasev2 2.**
### Swap: the ceiling that does not announce itself
| Leg | 900 000 Int rows, 64 MiB cap | Wall | Final RSS |
| --- | --- | --- | --- |
| swap OFF (`MemorySwapMax=0`) | **SIGKILL at 360 000 rows** | — | 57 188 KiB |
| swap ON (256 MiB) | **completed, exit 0** | **148 s** | 62 264 KiB (rest paged out) |
| uncapped | completed, exit 0 | **150 s** | 169 416 KiB |
**Swap cost ~1%.** A prior draft predicted "latency collapse"; the prediction had
the wrong sign. Inserting is append-mostly, so cold pages are written once and
never re-read — paging is sequential and off the critical path. The swap device
is a real disk file (`/swap.img`; no zram, zswap disabled), so this is genuine
disk paging.
**Do not generalise this to "swap is fine".** It measures an append-mostly
workload. A random-read workload over a table larger than the cap is where the
collapse should appear, and it is **not yet measured** — which matters, because
that is exactly the access pattern databasev2 2's `resident: keys` creates.
The operational consequence is that the RAM ceiling has two shapes and neither
reports itself: without swap the process vanishes on signal 9, with swap it
keeps returning 0 while serving from disk. A budget that fires at a *declared
threshold* is the only one that can speak before either happens.
### Durability across the ceiling
60 000 Int rows, 8 MiB cap, swap off, `WO_DATA` set — the process is OOM-killed
mid-insert, then replayed:
| Claim | Result |
| --- | --- |
| the survivor is a contiguous prefix | ✅ ~40 000 rows, rows 1..M all present |
| every surviving row's payload is correct | ✅ every `v` matches `item_v(i)` |
| the truncated tail is not read as corruption | ✅ replay exits 0 |
**Ack-after-fsync holds through an OOM kill** — the one shutdown path that skips
every cleanup handler. Gated as `db-bench`'s `ceiling` leg, which asserts the
*shape* of the survivor rather than its size: where the SIGKILL lands is the
scheduler's business, so `rows_recovered` carries ±100% tolerance. The leg
asserts the exit but never records it as a metric, so that when databasev2 2's
byte budget turns the kill into a checked refusal, the gate does not fail on the
improvement.

View file

@ -67,6 +67,57 @@ behind this board; live Obsidian Dataview views:
## ▶ NEXT PLAN
### Landed 2026-08-27 — databasev2 1, the RAM ceiling measured
**Implemented last time (2026-08-27):** databasev2 1 refined (three forks
settled) and implemented. A text-heavy `Wide` reference shape beside the
Int-only `Item`; `growth N int|text` in the db-bench sample, reading its OWN
`/proc/self/status` RSS at each decile because the driver's 250 ms poll misses
the value *at* a boundary; `growth-verify`, which asserts the survivor of a
crash is a contiguous intact prefix; and two harness legs — four footprint legs
under a rootless cgroup v2 cap, and a `ceiling` leg that deliberately dies at
the cap and then replays. 121 checks, 0 failures.
**Key findings (measured, not asserted):** per-row footprint is **96.5–100 B**
Int-only and **320.6–324 B** text-heavy — **3.3×**, not the "order of magnitude"
three docs asserted. Read as the median of per-decile marginals, never a
two-point slope: index doublings make a two-point read swing 2× (96 vs 205 B/row
for one shape). **Two predictions in the iteration's own premise were wrong.**
The ceiling is not a catchable `WO_T_OOM` for table storage — it is **SIGKILL,
signal 9**, because `vm.overcommit_memory = 0` lets `malloc` succeed and the
kernel kills on page *touch*, so the checked path never runs (the VM arena is
the opposite: `WO_HEAP_MB` is checked and traps). And swap is not "latency
collapse": 900 000 rows inside a 64 MiB cap with swap finished in **148 s
against 150 s uncapped** — ~1%, on a real disk swap file with no zram. Also
measured: **ack-after-fsync holds through an OOM kill** — ~40 000 rows came back
as an intact prefix, no holes, not read as corruption.
**Learned:** an append-mostly workload never re-touches its cold pages, so swap
costs it nothing — the collapse belongs to *random reads* over an oversized
table, which is precisely the pattern iteration 2's `resident: keys` creates and
is **still unmeasured**. The RAM ceiling therefore has two shapes and neither
announces itself: without swap the process vanishes on signal 9, with swap it
keeps returning 0 while serving from disk. That is the argument for a budget
that fires at a declared threshold instead of at exhaustion.
**Dependencies unblocked — one, by *removing* it:** iteration 2's
resident-footprint budget default was to be derived from "swap onset". **There is
no onset.** Swap-off jumps straight from working to SIGKILL; swap-on shows no
degradation to detect. Iteration 2 must pick its budget on other grounds rather
than wait on a number this slice cannot produce. Iteration 3's replay baseline is
still NOT delivered — `bench/baseline.json` times no replay.
**Next steps:** the read-heavy-over-cap leg is the single most valuable
follow-up, and it is what makes `p99_departure_decile` mean anything (the
footprint legs never approach their 512 MiB cap, so it is legitimately 0 today).
Then iteration 2's 5c/5d.
**`.dev/reference` used:** none. Sources were the kernel's own interfaces —
cgroup v2 `memory.max`/`memory.swap.max`, `/proc/self/status`, `/proc/swaps` and
`vm.overcommit_memory`.
---
### Landed 2026-08-25 — packaging + release pipeline (off-chain, no story)
**Implemented last time (2026-08-25):** the toolchain became installable
@ -620,8 +671,11 @@ declares a budget. Rows live in `malloc`'d slabs whose addresses are stable
forever; there is no eviction, spill or paging anywhere in `database/src/`; the
WAL never checkpoints so boot replays all history; and durability is one
process-global `WO_DATA`, so no table can say it matters more than another. An
allocation failure *is* a clean catchable `WO_T_OOM` — but swap thrash arrives
first and carries no error signal at all.
allocation failure is a clean catchable `WO_T_OOM` **only in the VM arena** —
table storage has no ceiling and is SIGKILLed instead (measured, databasev2 1).
Where swap exists the ceiling may never announce itself at all: an append-mostly
900k-row run finished *at uncapped speed* inside a 64 MiB cap (148 s vs 150 s),
serving from disk with no error signal.
**The lever** is per-table storage modes, which is why this track has a grammar
iteration. Six pending iterations moved here from the language track (their old
@ -630,7 +684,7 @@ the language arc as v1 history.
| # | Iteration | State |
| --- | --- | --- |
| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | ⬜ `readiness: refine` — its three forks are open, so despite being first it is NOT startable without a brainstorm — nobody here can say what happens at 90% RAM. Curve not cliff: swap onset, latency departure, the three exits (checked trap / swap thrash / OOM killer), and `kill -9` durability *at exhaustion*. Output is `perf-targets.md` + baseline rows, not prose |
| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | 🔄 **MEASURED 2026-08-27** — `readiness: ready`, `status: in-progress` (two criteria outstanding), forks settled, harness landed (121 checks). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Outstanding: the **random-read-over-cap** collapse (unmeasured, and it is `resident: keys`'s own access pattern) and iteration 3's replay baseline. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from |
| 2 | [per-table storage: `durable` and `resident`](databasev2/02-table-storage-modes.md) | 🔄 **the language enrichment — the `durable` half is DONE and usable.** Two optional `@table` keys, `durable: true\|false` and `resident: all\|keys`, both defaulting to today's behaviour (all 28 existing declarations compile unchanged, no golden moved). Landed: the grammar, WO-E224 (a durable `ref` into a volatile table is refused), `.wob` v7 carrying both properties in spare `flags` bits, `durable: false` actually skipping the WAL (measured: 50 inserts → 1500 bytes durable, **0** volatile) with a mode-mismatch startup refusal, plus offset capture and read-a-row-from-an-offset. Outstanding: 5c/5d (the id→offset map and rewiring `wo_row_ptr`'s 11 call sites, slab scans and `@unique`/FK across the boundary — not yet written up), the two runtime refusals, and closeout. [spec](../superpowers/specs/2026-08-26-table-residency-design.md) · [plan](../superpowers/plans/2026-08-26-table-residency.md) |
| 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ⬜ snapshot + truncate: disk reclaimed, replay bounded |
| 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ⬜ **`readiness: ready` — the one startable iteration in the repo** (four forks confirmed settled 2026-08-20). Close the 66× gap iteration 22 measured (durable 4.5k vs ram 297k inserts/s) |

View file

@ -44,7 +44,14 @@ The bill comes due at the ceiling. Read from the engine as it stands:
Worth being precise, because the failure mode determines the fix — and the good
news is that the engine's own behaviour is clean:
**An allocation failure is a catchable trap, not a crash.** Every `malloc` in
**Corrected 2026-08-27 by measurement.** This section used to open "an
allocation failure is a catchable trap, not a crash", and that is true only of
the VM arena. Table storage has no ceiling, and with `vm.overcommit_memory = 0`
its `malloc` never fails — the process is **SIGKILLed** (rc=137, measured at
360 000 rows under a 64 MiB cap). The checked path below is real, but it is the
arena's, not the store's. See [iteration 1](01-ram-ceiling-measurement.md).
Every `malloc` in
the row encoder is checked and jumps to an `oom` label; `DB_ERR_OOM` maps to
`WO_T_OOM`, which a program can `try`/`catch`. So a writeonce program that runs
out of memory *refuses the insert* rather than corrupting or dying. That is a
@ -61,9 +68,22 @@ battery proves that much.
So the honest problem statement is not "malloc fails". It is: **there is no
declared budget, no back-pressure as the budget is approached, and no way to
distinguish data that must be resident from data that merely is.** Iteration
[1](01-ram-ceiling-measurement.md) exists to replace this paragraph with
numbers before anything is designed on top of it.
distinguish data that must be resident from data that merely is.**
**Iteration [1](01-ram-ceiling-measurement.md) has now measured this
(2026-08-27), and it strengthened the statement rather than softening it.** A row
costs **96.5–100 B** Int-only and **320.6–324 B** text-heavy (3.3× apart, so no
single per-row number can bound RAM). At the ceiling the engine has exactly two
behaviours and **neither one tells anybody**: without swap the process is
**SIGKILLed on signal 9** — table storage has no checked ceiling, and under
`vm.overcommit_memory = 0` its `malloc` succeeds and the kernel kills on page
touch — and with swap it **keeps returning 0 while serving from disk**, finishing
900 000 rows in 148 s against 150 s uncapped. Durability is the one thing that
does hold: acked writes came back as an intact prefix across an OOM kill.
That is why "back-pressure at exhaustion" is not a design option. Exhaustion
either kills without warning or never arrives. Only a **declared threshold** can
speak in time.
## The lever: per-table storage modes
@ -113,7 +133,7 @@ before its mechanism existed; the history is in
| # | Iteration | Delivers | Needs |
| --- | --- | --- | --- |
| 1 | [RAM ceiling: measure the breaking point](01-ram-ceiling-measurement.md) | what actually happens from 50% RAM to OOM — swap onset, latency cliff, trap behaviour, `kill -9` survival | nothing; extends iteration 22's harness |
| 1 | [RAM ceiling: measure the breaking point](01-ram-ceiling-measurement.md) | 🔄 **measured 2026-08-27**: footprint per shape (3.3× apart), the two silent exits (SIGKILL vs swap-serving-from-disk at ~uncapped speed), and ack-after-fsync surviving an OOM kill. Outstanding: the random-read-over-cap collapse, and a replay baseline | nothing; extends iteration 22's harness |
| 2 | [per-table storage](02-table-storage-modes.md) | the grammar: `durable: true\|false` and `resident: all\|keys`, per table, replacing the global `WO_DATA` all-or-nothing. **In progress — the `durable` half is done** | 1 for the budget default |
| 3 | [WAL checkpoint](03-wal-checkpoint.md) *(was language 32)* | snapshot + truncate: disk reclaimed, replay bounded | 4 composes |
| 4 | [io_uring group commit](04-io-uring-commit.md) *(was language 23)* | close the 66× durable/RAM write gap (4.5k vs 297k inserts/s) | the arc (landed) |

View file

@ -1,149 +1,263 @@
---
track: databasev2
iteration: "1"
status: pending
readiness: refine
status: in-progress
readiness: ready
---
# databasev2 1 — the RAM ceiling: measure the breaking point before designing for it
# databasev2 1 — the RAM ceiling: measure the breaking point
> Part of [Story — databasev2: the database beyond RAM](00-story.md).
>
> **Refined 2026-08-27; the three forks are settled below and the decisions are
> locked.** No spec document: the deliverable is numbers plus a harness leg, and
> the design fits in this file — the same call
> [7](07-single-file-db.md) makes.
>
> **First because the repo's own doctrine says so.** "Always inspect crashsites.
> Always measure. Never assume." Every later iteration in this track — the
> storage modes' defaults, the eviction policy, the tiering threshold — is a
> decision that should follow from a number. Right now nobody in this project
> can say what happens to a writeonce program at 90% of RAM, and designing
> tiering without that is guessing with extra steps.
> Always measure. Never assume." Two other iterations already cite numbers this
> one was supposed to produce: [2](02-table-storage-modes.md)'s resident-footprint
> budget defaults to a fraction of host memory whose value comes from here, and
> [3](03-wal-checkpoint.md)'s before/after replay criterion has no "before"
> because `bench/baseline.json` carries 75 metrics and **zero** for replay,
> restart, boot or recovery. Iteration 22 proved restart *correctness*; it never
> timed it.
## Goals
## The design, as settled
- **Find the curve, not the cliff.** Not "does it die" — it dies, everything
does. What matters is the shape on the way down: at what fraction of RAM does
p99 read latency leave its 1µs baseline, what does insert throughput do as
slabs stop coming from a warm allocator, and how much warning is there between
"fine" and "unusable".
- **Characterise all three exits.** The engine can leave the happy path three
ways and they are not equally survivable: a checked `malloc` failure
(`DB_ERR_OOM` → `WO_T_OOM`, a catchable trap — the clean one), swap thrash
(no trap, no error, just latency collapse — the dangerous one because nothing
reports it), and the external OOM killer (`SIGKILL`, skipping every shutdown
path). Establish which arrives first under realistic limits, because the
answer determines whether the fix is back-pressure or eviction.
- **Prove the durability floor holds at the ceiling.** Iteration 22's `kill -9`
battery proved acked writes survive under load. Re-run it *at memory
exhaustion*, which is a different and nastier state — an allocation failure
mid-commit is exactly where an ack-before-durable bug would hide.
- **Publish numbers others can build on.** The output is a section in
`perf-targets.md` and rows in `bench/baseline.json`, not a paragraph of
prose. A measurement that only printed once is not a measurement.
**Measure the curve, not the cliff.** Everything dies at the ceiling; what
matters is the shape on the way down — where p99 leaves its 1µs baseline, what
insert throughput does as slabs stop coming from a warm allocator, and how much
warning there is between "fine" and "unusable".
## Phases
### Fork 1 — the limit mechanism: rootless cgroup v2 via `systemd-run --user`
### Phase A — a workload that can actually reach the ceiling
`systemd-run --user --scope -p MemoryMax=N -p MemorySwapMax=M`. Verified on the
dev box: the `memory` controller is delegated to
`user.slice/user-<uid>.slice`, a scope's `memory.max` reads back exactly as set,
and no passwordless sudo is needed. Being cgroup-scoped also isolates the
measurement from whatever else the box is doing, which matters — the dev box was
at 22.9 of 31.7 GiB with 4.6 GiB of swap already in use when this was refined.
- Extend `docs/examples/db-bench` with a growth mode: insert until a target RSS
fraction, holding row shape and index count constant so the variable is size
alone.
- Run it under an explicit memory limit (a cgroup or `ulimit`) rather than on a
big box — "it survived on a 64 GB workstation" measures the workstation.
- Record RSS against row count so the per-row overhead is known: slab headroom,
the id hash, the secondary-index multimaps and the per-row engine-owned values
(`db_text`, `db_rec`, `db_multi`, `db_map` are each their own allocation).
- Verify: RSS growth is linear and its slope is written down; the run is
reproducible twice within the tolerance policy iteration 22 established.
`ulimit -v` is **rejected**: it bounds address space, not resident set, which is
the wrong quantity for an engine that `malloc`s slabs — and it is actively
broken under ASan, whose huge virtual reservations trip it long before any real
memory pressure.
### Phase B — the latency and throughput curve
If the mechanism is unavailable (no systemd, no delegation), the harness **skips
the growth legs loudly and names why**. It must never silently fall back to
measuring an uncapped box, because "it survived on a 32 GiB workstation"
measures the workstation.
- Sample read p50/p99, query p99 and insert throughput at fixed fractions of the
limit, so the result is a curve rather than two endpoints.
- Separate the two effects deliberately: allocator pressure (still resident) and
swap (no longer resident). They have different fixes and conflating them would
send iteration 6 after the wrong one.
- Include the DB-actor path, since a cross-shard statement's reply materialises
a copy — memory pressure and the actor RPC interact and nobody has looked.
- Verify: the curve is recorded per metric class with iteration 22's per-class
tolerances; the swap onset point is identified, not interpolated.
### Fork 2 — the reference shapes: both, reported separately
### Phase C — the three exits, deliberately triggered
Per-row footprint differs substantially between an Int-only row and a text-heavy
one, because a `Text` column is a separate `db_text` allocation per row on top of
the slab slot. **Measured 2026-08-27: 96.5 B/row Int-only vs 320.6 B/row with
three Text columns — 3.3×.** An earlier draft of this section said "an order of
magnitude"; that was an unmeasured guess and this iteration exists to replace
exactly that kind of claim. 3.3× is still more than enough to make a single
"bytes per row" number useless, which is the decision it was supporting.
- Drive a checked allocation failure and confirm `WO_T_OOM` is catchable, the
insert is refused whole, no partial row or index entry is left, and the
process continues serving.
- Drive swap thrash and record what a client sees. This is the case with no
error signal at all, and naming it is most of the value of this iteration.
- Drive the OOM killer under a cgroup limit and confirm what survives: replay
the WAL and check every acked write is present.
- Verify: the trap path leaves no torn state (row count and index agree after a
refused insert); replay after `SIGKILL` at exhaustion loses no acked write.
`db-bench` already supplies half of this: `items` (`k: Int`, `v: Int`, plus a
`bucket` ref) is the Int-only reference as it stands. The work is one text-heavy
shape beside it, with footprint reported per shape.
### Phase D — write it down where decisions get made
### Fork 3 — swap: in scope, as a controlled dimension
- A `perf-targets.md` section with the curve, the swap onset, the per-row
overhead and the exit characterisation.
- Baseline rows for the growth metrics so a regression is caught by the existing
gate rather than by a person remembering.
- A short statement of what the numbers *imply* for iterations 2, 5 and 6 —
which is the point of going first.
- Verify: `just db-bench` green against the extended baseline; the gate bites
when a growth metric is doctored.
Not a confound to wish away — `MemorySwapMax` is the knob that separates the two
exits this iteration exists to characterise. Both were measured, and **both
turned out differently than this iteration predicted.**
| Leg | Predicted | Measured |
| --- | --- | --- |
| swap-off | catchable `WO_T_OOM` from checked `malloc` | **SIGKILL, signal 9** (shell rc 137). No trap, no message |
| swap-on | latency collapse | **no degradation at all**: 148 s vs 150 s uncapped |
**Prediction 1 was wrong because of overcommit.** With `vm.overcommit_memory = 0`
`malloc` succeeds and the process dies when it *touches* the pages, so table
storage never gets the chance to report failure. The trap path is real but
belongs to a different allocator:
| Allocator | Ceiling | Failure mode |
| --- | --- | --- |
| VM object arena | `WO_HEAP_MB`, checked | `trap 4` / `WO_T_OOM`, exit 1, reportable |
| table storage (slabs + heap values) | **none** | SIGKILL under overcommit |
**This is the strongest argument available for [iteration 2](02-table-storage-modes.md)'s
byte budget:** a declared budget is the only way table storage can acquire a
checked ceiling, because `malloc` under default overcommit will never tell it
there is a problem.
**Prediction 2 was wrong because of access pattern.** 900 000 Int rows under a
64 MiB cap with 256 MiB of swap finished in **148 s** with RSS pinned at 62 MiB;
the same workload uncapped took **150 s** at 165 MiB RSS. Swap cost
approximately nothing. The reason is that inserting is append-mostly: cold pages
are written out once and never read again, so paging is sequential and off the
critical path. The swap is a real disk file (`/swap.img`, no zram, zswap
disabled), so this is genuine disk paging, not compressed RAM.
**The correct generalisation is narrower than "swap is fine".** This measures an
append-mostly workload. A workload that reads randomly across a table larger
than the cap is the one that collapses, and this iteration did *not* measure
that — see Outstanding.
## Progress
| Piece | State |
| --- | --- |
| `Wide` text-heavy reference shape (`db-bench/types.wo`) | ✅ |
| `growth N int\|text` — insert, per-decile RSS and read latency | ✅ |
| the sample reads its OWN RSS via `/proc/self/status` | ✅ — the driver polls every 250 ms and would miss the value *at* a decile boundary |
| `growth-verify` — the survivor is a contiguous intact prefix | ✅ |
| rootless cap wrapper, swap on/off legs | ✅ 4 footprint legs |
| footprint metric = **median of marginals**, doublings counted separately | ✅ |
| `ceiling` leg: dies at the cap, then replay must be intact | ✅ gated |
| baseline + tolerance policy | ✅ 121 checks; footprint at ±10%, kill-timing metrics at ±100% |
| `perf-targets.md` §5 | ✅ |
| **resident-footprint fraction for [iteration 2](02-table-storage-modes.md)** | ⬜ **not delivered — the premise it rested on is false**, see Outstanding |
| **replay/restart baseline for [iteration 3](03-wal-checkpoint.md)** | ⬜ not delivered |
| **the random-read-over-cap collapse** | ⬜ not measured |
## Measured
Footprint, reproducible inside 2% across runs:
| What | Int-only (`Item`) | Text-heavy (`Wide`) |
| --- | --- | --- |
| steady-state footprint | **96.5–100 B/row** | **320.6–324 B/row** |
| doubling steps | 3 (at ~24k and ~48k rows) | 2 |
| base process RSS | ≈ 3.9 MiB, excluded from the per-row figure | same |
Ratio **3.3×** — not the "order of magnitude" an earlier draft asserted. Enough
on its own to make a single "bytes per row" number useless, which is the decision
it was supporting ([2](02-table-storage-modes.md), fork 5).
The ceiling, 60 000 Int rows under an 8 MiB cap with swap off and `WO_DATA` set:
| Question | Answer |
| --- | --- |
| how does it die? | **SIGKILL, signal 9.** No refusal, no diagnostic |
| what survives? | **a contiguous intact prefix** — ~40 000 rows, every `v` correct, no holes, not reported as corruption |
**Ack-after-fsync holds through an OOM kill.** That is the one shutdown path
which skips every cleanup handler, and the durable prefix came back whole.
**The finding that matters most is the swap leg succeeding.** It did not fail,
did not warn, and returned 0. A deployment in that state looks healthy while
serving from disk. That is the exit with no error signal, and it is why
[iteration 5](05-bounded-tables-eviction.md)'s back-pressure must act at a
declared threshold rather than at exhaustion — exhaustion either kills without
warning or silently does not arrive.
## Acceptance Criteria
- **Given** the growth workload under a fixed memory limit, **when** it runs
twice, **then** RSS-per-row agrees within the tolerance policy and the slope
is recorded in `perf-targets.md`.
- **Given** the workload at rising RAM fractions, **when** latency is sampled,
**then** the fraction at which read p99 first leaves its baseline is
identified as a measured point, not an estimate.
- **Given** a deliberately induced allocation failure, **when** an insert is
attempted, **then** it traps `WO_T_OOM` catchably, the table's row count is
unchanged, every index agrees with the slab contents, and the process keeps
serving subsequent requests.
- **Given** swap thrash, **when** a client issues reads, **then** the observed
degradation is quantified and the fact that **no error is surfaced** is
recorded explicitly as a finding.
- **Given** a cgroup limit and a workload that exceeds it, **when** the OOM
killer fires, **then** replaying the WAL shows every acked write present —
ack-after-fsync holding in the one shutdown path that skips all cleanup.
Met:
- **Given** the growth workload under a fixed cap, **when** it runs twice,
**then** RSS-per-row agrees inside tolerance and the slope is recorded per
shape. ✅ inside 2%; `perf-targets.md` §5.
- **Given** the swap-off leg, **when** the cap is exceeded, **then** the exit is
identified and recorded. ✅ **SIGKILL, signal 9** — not the catchable trap this
criterion originally expected, which is the whole point of measuring. The
"process keeps serving" half of the original wording is **void**: nothing
survives a SIGKILL.
- **Given** a cap exceeded with `WO_DATA` set, **when** the process is killed at
exhaustion, **then** replay shows the acked writes present. ✅ ~40 000 rows,
contiguous, no holes, no corruption report. Gated as the `ceiling` leg.
- **Given** the swap-on leg, **when** the same point is reached, **then** the
degradation is quantified **and the absence of any error signal recorded**.
✅ degradation is **nil** for this workload (148 s vs 150 s uncapped) and the
silence is total. Both halves are findings; the first inverted the prediction.
- **Given** the extended baseline, **when** a growth metric is doctored, **then**
`just db-bench` fails on exactly that metric.
the gate fails on exactly that metric. ✅ text footprint +20% →
`FAIL gate.growth.text.noswap.bytes_per_row -- 388 vs baseline 324`, 1 of 104.
- **Given** a host without the cap mechanism, **when** the harness runs, **then**
the legs are skipped with a named reason and the rest still passes. ✅
`cap_wrapper` returns None unless the `memory` controller is delegated; there
is no uncapped fallback.
Outstanding:
- **The resident-footprint fraction for iteration 2's budget default. NOT
delivered, and the premise is false.** It was to be derived from the
swap-onset point — but there is no onset: swap-off jumps straight from
working to SIGKILL, and swap-on shows no degradation to detect an onset in.
**Iteration 2 must pick its budget on other grounds** (host RAM fraction, or
an explicit developer-declared figure) rather than waiting on a number this
iteration cannot produce. This is the most important thing this slice learned
and it removes a dependency rather than satisfying it.
- **The random-read-over-cap collapse.** Not measured. This is where the "latency
collapse" prediction may still be true, and it is the workload that matters
for [2](02-table-storage-modes.md)'s `resident: keys`, whose whole premise is
reading rows back from a log larger than RAM. Needs a read-heavy leg over a
table exceeding the cap. **The single most valuable follow-up.**
- **Given** rising fractions of the cap, **when** latency is sampled, **then**
the p99 departure point is recorded. Partially: the sampler and metric exist
and are gated, but the footprint legs never approach their 512 MiB cap, so
`p99_departure_decile` is legitimately 0 and proves nothing. It becomes
meaningful only with the read-heavy leg above.
- **A replay/restart baseline for iteration 3.** Not delivered; `growth` exercises
`WO_DATA` but nothing times replay. Cheap to add, still absent from
`bench/baseline.json`.
## Out Of Scope
- **Any fix.** This iteration measures. Eviction is
[5](05-bounded-tables-eviction.md), tiering is [6](06-cold-tiering.md),
declared budgets are [2](02-table-storage-modes.md). Shipping a fix inside the
measurement slice would remove the ability to tell whether it helped.
- **Changing the OOM behaviour.** The checked-`malloc`-to-catchable-trap path is
good and should not be touched; if the measurement finds a hole in it, that is
a bug fix, reported separately.
- **A memory profiler or allocator instrumentation.** Observability is language
iteration 30. RSS from the OS and the existing `time.ticks` are enough for a
curve.
- **Multi-machine or sharded-across-hosts scaling.** One binary owns its data;
cross-process is [9](09-cross-program-tables.md).
- **Comparing against SQLite at the ceiling.** `bench/compare/go-sqlite` exists
and the comparison would be interesting, but SQLite's whole architecture is
the paged design this project rejected — the numbers would not inform any
decision here.
- **Any fix.** This measures. Declared budgets are [2](02-table-storage-modes.md),
eviction is [5](05-bounded-tables-eviction.md), tiering is 2's `resident: keys`.
- **Changing the OOM behaviour.** The checked-`malloc` code is untouched. The
measurement showed it is largely unreachable for table storage under default
overcommit — a finding to hand to [2](02-table-storage-modes.md), not a bug to
fix here, and emphatically not a licence to start setting
`vm.overcommit_memory`.
- **A memory profiler or allocator instrumentation** — observability is language
iteration 30. RSS from `/proc` plus `time.ticks` is enough for a curve.
- **Comparing against SQLite at the ceiling.** `bench/compare/go-sqlite` exists,
but SQLite's paged architecture is the design this project rejected, so the
numbers would inform no decision here.
- **Multi-host scaling** — one binary owns its data.
## Info
## Info — the forks, settled
Forks the spec must settle:
1. **The limit mechanism is rootless cgroup v2** via
`systemd-run --user --scope -p MemoryMax -p MemorySwapMax`. `ulimit -v` was
rejected: it bounds address space, not resident set, and ASan's virtual
reservations trip it long before real pressure. No sudo needed; it also
isolates the run from the rest of the box, which mattered — the dev box sat
at 22.9 of 31.7 GiB throughout.
2. **Both reference shapes, reported separately.** 3.3× apart; one number would
be a fiction.
3. **Swap is a dimension, not a footnote** — settled by getting it wrong first.
An early run looked like the cap was unenforced because the process held
400 MiB inside a 64 MiB limit; it was swapping, which is the phenomenon under
study.
4. **Footprint is read as the median of per-decile marginals**, not a two-point
slope, so a slab doubling does not smear into the per-row figure. Doublings
are counted as their own metric.
5. **Kill-timing metrics carry ±100% tolerance.** `rows_recovered` depends on
where the SIGKILL landed; gating it tightly would be gating the scheduler.
The invariant asserted instead is the *shape* of the survivor.
6. **The ceiling leg asserts `rc`, never records it.** When iteration 2's byte
budget lands, death should become a checked refusal — the gate must not fail
on that improvement.
1. **What is the limit mechanism for the harness?** A cgroup v2 `memory.max` is
the closest thing to how this would actually be deployed; `ulimit -v` is
simpler but bounds address space rather than resident set, which for an engine
that `malloc`s slabs is a materially different constraint. Leaning cgroup, and
the campaign already runs off the fast path so the setup cost is acceptable.
2. **Which table shape is the reference?** Per-row overhead depends heavily on
whether fields are scalars or heap values — a `Text` column is a separate
`db_text` allocation per row, so a text-heavy table and an Int-only table
will produce very different slopes. Probably both, reported separately,
because "bytes per row" is meaningless without saying which row.
3. **Is swap even in scope for the target deployment?** If the intended answer
is "run with swap off and let the OOM killer decide", the swap curve is
informational rather than load-bearing — but that stance should be stated in
the doctrine, not assumed. It also changes which exit iteration 5's
back-pressure is defending against.
## History — four corrections worth keeping
**"An order of magnitude" was a guess.** The per-shape difference is 3.3×. An
iteration whose purpose is replacing unmeasured claims had one in its own
premise.
**The clean-exit premise was wrong.** This file and the residency spec both
asserted the ceiling surfaces as a catchable `WO_T_OOM`. It is a SIGKILL.
Overcommit means the allocator never learns there is a problem.
**The latency-collapse premise was wrong too.** Swap cost ~1% on an
append-mostly workload (148 s vs 150 s). The prediction was not merely
imprecise, it had the wrong sign. The narrower claim that survives is that a
*random-read* workload over an oversized table is the one at risk, and that
remains unmeasured.
**A SIGKILL was once labelled a "checked refusal"** by the harness, because
`subprocess` reports signal death as a negative `returncode` (`-9`) while the
shell spells the same event `137`. The leg existed specifically to tell those
two apart. Fixed, and the distinction is now spelled out at the comparison.

View file

@ -154,7 +154,7 @@ Outstanding:
resident. Roughly doubles the resident index; stated at the declaration so
the cost is visible.
5. **The budget is bytes, not rows** — a text-heavy row and an Int-only row
differ by an order of magnitude, so a row count cannot bound RAM.
differ by 3.3× (measured, databasev2 1), so a row count cannot bound RAM.
## History — two corrections worth keeping

View file

@ -19,7 +19,7 @@
| Which storage architecture | **One engine, log-structured.** The WAL already holds every row; keep an in-RAM id→offset map and read rows back with `pread`. No second engine. |
| Row cache | **None in user space.** The kernel page cache is the hot copy — the repo's own stated position in `exploration/postgresql/buffer-and-checkpoint.md`: "`pread` against an fd that already has its page cached is a memcpy… the page cache is the one cache we want", and the reason the engine avoids `O_DIRECT`. |
| `@unique` on a non-resident table | **Allowed; its index is unconditionally resident.** Settled here rather than deferred — see Constraints. |
| Budget unit | **Bytes** (estimated resident footprint). Rows is the meaningless unit: a text-heavy row and an Int-only row differ by an order of magnitude, so a row count cannot bound RAM. |
| Budget unit | **Bytes** (estimated resident footprint). Rows is the meaningless unit: a text-heavy row and an Int-only row differ by 3.3× (measured, databasev2 1), so a row count cannot bound RAM. |
| Rejected architectures | `mmap` and a buffer pool stay out — see Alternatives rejected. `discarded.md`'s paged-engine rejection is amended to *partly revisited*, not reversed. |
## The problem, read off the engine
@ -37,9 +37,15 @@ Facts, each verified in source rather than assumed:
`WO_DATA` is set. `db.c` guards every WAL append with a null check on
`vm->rt.wal`, so with no `WO_DATA` **every table is silently volatile** — a
program can declare nothing and lose everything.
- An allocation failure is clean: every `malloc` in the row encoder is checked
and `DB_ERR_OOM` maps to `WO_T_OOM`, a catchable trap. The dangerous exit is
the one *before* that — swap thrash, which carries no error signal at all.
- An allocation failure is clean *in principle*: every `malloc` in the row
encoder is checked and `DB_ERR_OOM` maps to `WO_T_OOM`. **Corrected 2026-08-27
by measurement (databasev2 1): that path does not fire in practice.** With
`vm.overcommit_memory = 0`, `malloc` succeeds and the process is SIGKILLed
when it touches the pages — measured rc=137 at 360 000 rows under a 64 MiB
cgroup cap. The checked-trap path belongs to the VM arena (`WO_HEAP_MB`,
verified `trap 4 ... out of memory`), not to table storage, which has no
ceiling at all. This makes the byte budget below the ONLY mechanism by which
table storage can acquire one.
The measurements that bound the design, from iteration 22: durable inserts
≈4.5k/s against RAM ≈297k/s (the 66× fsync gap); reads 1.3M ops/s at p50 1µs

View file

@ -225,6 +225,14 @@ def tolerance_for(key):
mix*: scheduling-dependent small counts. read/query + all .sN.*:
machine jitter, and at post-index-µs scale a 1µs histogram step on a
7µs p50 is already 14%."""
# databasev2 1: footprint is a STRUCTURAL number -- 96.5 vs 320.6 B/row
# reproduced to <2% across runs -- so it gets a tight tolerance and is the
# one growth metric worth gating. The doubling COUNT and the latency
# samples are allowed to move: doublings depend on where N lands relative
# to a pow2 rehash, and at 1us p50 a single histogram step is already 100%.
if ".bytes_per_row" in key: return 10
if key.startswith("growth."): return 100
if key.startswith("ceiling."): return 100
if ".mixread." in key or ".mixwrite." in key: return 50
if ".sN." in key: return 50
if ".read." in key or ".query." in key: return 50
@ -248,6 +256,183 @@ def write_baseline(metrics):
json.dump(base, open(BASELINE, "w"), indent=1, sort_keys=True)
ok(f"baseline written ({len(base) - 1} metrics)")
# ---- databasev2 1: the RAM ceiling -----------------------------------------
GROWTH_N = 20000 if QUICK else 200000
GROWTH_SHAPES = ("int", "text")
def cap_wrapper(mem_mb, swap_mb):
"""systemd-run --user --scope argv prefix that caps memory rootlessly, or
None when the mechanism is unavailable.
cgroup v2 with the `memory` controller delegated to the user slice is the
only mechanism used. `ulimit -v` is deliberately NOT a fallback: it bounds
address space, not resident set, which is the wrong quantity for an engine
that mallocs slabs, and ASan's huge virtual reservations trip it long
before real memory pressure. When the cap is unavailable the legs are
SKIPPED and say so -- never silently run uncapped, because "it survived on
a 32 GiB workstation" measures the workstation."""
if not shutil.which("systemd-run"):
return None
try:
with open("/proc/self/cgroup") as f:
mine = f.readline().strip().split(":")[-1]
ctl = f"/sys/fs/cgroup{os.path.dirname(mine)}/cgroup.controllers"
if "memory" not in open(ctl).read().split():
return None
except OSError:
return None
return ["systemd-run", "--user", "--scope", "--quiet",
"-p", f"MemoryMax={mem_mb}M", "-p", f"MemorySwapMax={swap_mb}M", "--"]
def parse_growth(lines):
"""(rows, rss_kb) samples plus per-decile read p50/p99, from the sample's
own `growthrss` / `growthN` lines. RSS is read by the SAMPLE, not polled
here: the driver polls every 250 ms and would miss the value AT a decile
boundary, and per-row footprint is this iteration's headline number."""
pts, lat = [], {}
for l in lines:
f = l.split()
if f and f[0] == "growthrss" and len(f) == 4:
pts.append((int(f[2]), int(f[3])))
elif f and f[0].startswith("growth") and len(f) == 5 and f[0][6:].isdigit():
lat[int(f[0][6:])] = (int(f[3]), int(f[4]))
return pts, lat
def bytes_per_row(pts):
"""Steady-state marginal footprint = MEDIAN of the per-interval marginals.
Not a two-point slope: the id hash and index buckets are open-addressing
pow2 and DOUBLE periodically, so a two-point slope lands arbitrarily on or
off a doubling and swings 2x (measured: 96 vs 205 B/row for the same shape).
The median rejects those steps; they are reported separately as `doublings`
because a transient RSS step is exactly what a resident-footprint budget
must leave headroom for."""
marg = sorted((k1 - k0) * 1024.0 / (r1 - r0)
for (r0, k0), (r1, k1) in zip(pts, pts[1:]) if r1 > r0)
if not marg:
return None, 0
med = marg[len(marg) // 2]
doublings = sum(1 for m in marg if m > med * 1.5)
return med, doublings
def growth(metrics):
"""Per-shape footprint and the read-latency curve, under a rootless cap,
with swap ON and OFF.
What this leg actually measures is FOOTPRINT. It does not reach the cap:
GROWTH_N rows need far less than the 512 MiB cap, so both swap legs are
identical by construction and p99_departure_decile is legitimately 0.
The ceiling itself is ceiling() below -- keep the two separate, because a
footprint regression and a ceiling-behaviour change are different faults.
Two earlier claims in this docstring were measured FALSE and are recorded
in docs/stories/databasev2/01-ram-ceiling-measurement.md: swap-off is not
a "clean checked-malloc" path (it is SIGKILL, rc=137), and swap-on is not
"latency collapse" (900k rows finished in 148s capped-with-swap vs 150s
uncapped -- an append-mostly workload never re-touches its cold pages)."""
wrap = cap_wrapper(512, 0)
if wrap is None:
ok("growth: SKIPPED -- no rootless cgroup v2 memory cap on this host")
metrics["growth.available"] = 0
return
metrics["growth.available"] = 1
for shape in GROWTH_SHAPES:
for legname, swap_mb in (("noswap", 0), ("swap", 256)):
w = cap_wrapper(512, swap_mb)
env = dict(os.environ)
argv = w + [BIN, "growth", str(GROWTH_N), shape]
pr = subprocess.run(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
text=True, env=env, timeout=900)
lines = pr.stdout.splitlines()
pts, lat = parse_growth(lines)
key = f"growth.{shape}.{legname}"
if not pts:
bad(f"{key}: produced no samples", (lines[-1] if lines else "no output"))
continue
bpr, doublings = bytes_per_row(pts)
metrics[f"{key}.bytes_per_row"] = int(round(bpr))
metrics[f"{key}.doublings"] = doublings
metrics[f"{key}.rows"] = pts[-1][0]
metrics[f"{key}.rss_kb"] = pts[-1][1]
if lat:
last = max(lat)
metrics[f"{key}.read_p50us"] = lat[last][0]
metrics[f"{key}.read_p99us"] = lat[last][1]
# the curve's departure point: first decile whose p99 exceeds
# 4x the first decile's, as a MEASURED sample not an estimate
first = lat[min(lat)][1]
dep = next((d for d in sorted(lat) if lat[d][1] > max(first, 1) * 4), 0)
metrics[f"{key}.p99_departure_decile"] = dep
ok(f"{key}: {int(round(bpr))} B/row steady, {doublings} doubling step(s), "
f"{pts[-1][0]} rows in {pts[-1][1]} KiB")
CEIL_N, CEIL_CAP_MB = 60000, 8
def ceiling(metrics):
"""The ceiling itself, and the durability claim across it.
Sized so the process CANNOT fit: 60k Int rows need ~9.7 MiB resident
(96.5 B/row measured, plus a ~3.9 MiB base) under an 8 MiB cap, swap off.
Two things are under test and the second is the one that matters:
1. HOW it dies. Measured: SIGKILL, rc=137 -- not a refusal. Table
storage has no checked ceiling, and under vm.overcommit_memory=0
malloc succeeds and the process dies TOUCHING the pages, so it never
gets the chance to report failure. (The VM object arena is the
opposite: WO_HEAP_MB is checked and traps.) rc is asserted, not
recorded as a metric -- when databasev2 2's byte budget lands this
should become a checked refusal, and the gate must not fail on that
improvement.
2. WHAT SURVIVES. With WO_DATA set, replay must yield a contiguous
intact prefix: rows 1..M present with the right v, no holes, and not
reported as corruption. M is wherever the kill landed -- the SHAPE of
the survivor is the claim, not its size, so rows_recovered carries a
wide tolerance. This is ack-after-fsync holding in the one shutdown
path that skips every cleanup handler."""
wrap = cap_wrapper(CEIL_CAP_MB, 0)
if wrap is None:
ok("ceiling: SKIPPED -- no rootless cgroup v2 memory cap on this host")
return
data = os.path.join(ROOT, "bench", f"tmp.{os.getpid()}.ceiling")
shutil.rmtree(data, ignore_errors=True); os.makedirs(data, exist_ok=True)
env = dict(os.environ); env["WO_DATA"] = data
pr = subprocess.run(wrap + [BIN, "growth", str(CEIL_N), "int"],
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
text=True, env=env, timeout=900)
if pr.returncode == 0:
bad("ceiling: process SURVIVED the cap",
f"{CEIL_N} rows fit under {CEIL_CAP_MB} MiB -- footprint changed, resize the leg")
shutil.rmtree(data, ignore_errors=True); return
# subprocess returncode is NEGATIVE for signal death (-9 = SIGKILL); 137
# is the SHELL spelling of the same event (128+9). Getting this backwards
# once labelled a SIGKILL as a "checked refusal", which is the exact
# distinction this leg exists to report.
if pr.returncode < 0:
sig = -pr.returncode
how = f"killed by signal {sig}" + (" (SIGKILL -- no checked refusal)" if sig == 9 else "")
else:
how = f"exited {pr.returncode} (checked refusal)"
ok(f"ceiling: died at the cap, {how}")
vr = subprocess.run([BIN, "growth-verify"], stdout=subprocess.PIPE,
stderr=subprocess.STDOUT, text=True, env=env, timeout=900)
m = re.search(r"^growthverify (\d+)$", vr.stdout, re.M)
if vr.returncode == 0 and m and int(m.group(1)) > 0:
metrics["ceiling.rows_recovered"] = int(m.group(1))
ok(f"ceiling: durable prefix intact across the kill -- {m.group(1)} rows, no holes")
else:
bad("ceiling: durable prefix broken across the kill",
(vr.stdout.strip().splitlines() or ["no output"])[-1][:160])
shutil.rmtree(data, ignore_errors=True)
def main():
# --check <results.json>: gate-only evaluation of a recorded run — the
# gate-bites smoke doctors a copy and this mode must FAIL on it
@ -260,6 +445,8 @@ def main():
build()
metrics = campaign()
durability(metrics)
growth(metrics)
ceiling(metrics)
os.makedirs(RESULTS_DIR, exist_ok=True)
stamp = time.strftime("%Y%m%d-%H%M%S")
out = os.path.join(RESULTS_DIR, f"run-{stamp}{'-quick' if QUICK else ''}.json")