feat(db-bench): measure the RAM ceiling — databasev2 1

- `Wide` text-heavy reference shape beside Int-only `Item`
- `growth N int|text`: per-decile RSS read from own /proc/self/status
- `growth-verify`: survivor of a crash must be a contiguous intact prefix
- four footprint legs under a rootless cgroup v2 cap, swap on/off
- `ceiling` leg: die at the cap, then replay must come back intact
- footprint read as median-of-marginals; doublings a separate metric
- 121 checks, 0 failures; footprint gated ±10%, kill-timing ±100%

Measured, and it inverted two of the iteration's own predictions:

- footprint 96.5-100 B/row Int vs 320.6-324 B/row text = 3.3x, NOT the
  "order of magnitude" three docs asserted
- table storage has NO checked ceiling: SIGKILL signal 9, not a catchable
  WO_T_OOM. overcommit lets malloc succeed; kernel kills on page touch
- swap is NOT latency collapse: 900k rows 148s capped-with-swap vs 150s
  uncapped. Append-mostly never re-touches cold pages
- ack-after-fsync survives an OOM kill: ~40k rows, no holes, no corruption
- iteration 2's budget dependency is REMOVED not satisfied — there is no
  "swap onset" to derive a fraction from

- fix: subprocess returncode -9 was labelled a "checked refusal"; 137 is
  the shell spelling of the same signal

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
shoney.arickathil 2026-08-27 20:18:47 +02:00
parent 1fe808b7a4
commit 0c9b2c45d8
10 changed files with 1036 additions and 240 deletions

View file

@ -1,16 +1,22 @@
{ {
"_config": { "_config": {
"N": 20000, "N": 2000,
"crash_reps": 3, "crash_reps": 1,
"msg_n": 200000, "msg_n": 20000,
"note": "refresh only with a commit that says why; tolerances come from tolerance_for() in the driver", "note": "refresh only with a commit that says why; tolerances come from tolerance_for() in the driver",
"wal_n": 4000 "wal_n": 800
},
"ceiling.rows_recovered": {
"dir": "lower",
"floor": 159744,
"tolerance_pct": 100,
"value": 39936
}, },
"durable.s1.mixread.ops_sec": { "durable.s1.mixread.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 2302, "floor": 2239,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 9211 "value": 8958
}, },
"durable.s1.mixread.p50us": { "durable.s1.mixread.p50us": {
"dir": "lower", "dir": "lower",
@ -22,31 +28,31 @@
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 12 "value": 2
}, },
"durable.s1.mixwrite.ops_sec": { "durable.s1.mixwrite.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 255, "floor": 248,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 1023 "value": 995
}, },
"durable.s1.mixwrite.p50us": { "durable.s1.mixwrite.p50us": {
"dir": "lower", "dir": "lower",
"floor": 1720, "floor": 828,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 430 "value": 207
}, },
"durable.s1.mixwrite.p99us": { "durable.s1.mixwrite.p99us": {
"dir": "lower", "dir": "lower",
"floor": 2656, "floor": 872,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 664 "value": 218
}, },
"durable.s1.query.ops_sec": { "durable.s1.query.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 308641, "floor": 202429,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 1234567 "value": 809716
}, },
"durable.s1.query.p50us": { "durable.s1.query.p50us": {
"dir": "lower", "dir": "lower",
@ -58,13 +64,13 @@
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 1 "value": 2
}, },
"durable.s1.read.ops_sec": { "durable.s1.read.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 319284, "floor": 215703,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 1277139 "value": 862812
}, },
"durable.s1.read.p50us": { "durable.s1.read.p50us": {
"dir": "lower", "dir": "lower",
@ -76,85 +82,85 @@
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 1 "value": 2
}, },
"durable.s1.seed.ops_sec": { "durable.s1.seed.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 1115, "floor": 1078,
"tolerance_pct": 15, "tolerance_pct": 15,
"value": 4460 "value": 4315
}, },
"durable.s1.seed.p50us": { "durable.s1.seed.p50us": {
"dir": "lower", "dir": "lower",
"floor": 836, "floor": 828,
"tolerance_pct": 15, "tolerance_pct": 15,
"value": 209 "value": 207
}, },
"durable.s1.seed.p99us": { "durable.s1.seed.p99us": {
"dir": "lower", "dir": "lower",
"floor": 2352, "floor": 2160,
"tolerance_pct": 15, "tolerance_pct": 15,
"value": 588 "value": 540
}, },
"durable.s1.write.ops_sec": { "durable.s1.write.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 581, "floor": 1156,
"tolerance_pct": 15, "tolerance_pct": 15,
"value": 2324 "value": 4626
}, },
"durable.s1.write.p50us": { "durable.s1.write.p50us": {
"dir": "lower", "dir": "lower",
"floor": 1764, "floor": 828,
"tolerance_pct": 15, "tolerance_pct": 15,
"value": 441 "value": 207
}, },
"durable.s1.write.p99us": { "durable.s1.write.p99us": {
"dir": "lower", "dir": "lower",
"floor": 2544, "floor": 1844,
"tolerance_pct": 15, "tolerance_pct": 15,
"value": 636 "value": 461
}, },
"durable.sN.mixread.ops_sec": { "durable.sN.mixread.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 1081, "floor": 1113,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 4324 "value": 4452
}, },
"durable.sN.mixread.p50us": { "durable.sN.mixread.p50us": {
"dir": "lower", "dir": "lower",
"floor": 248, "floor": 240,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 62 "value": 60
}, },
"durable.sN.mixread.p99us": { "durable.sN.mixread.p99us": {
"dir": "lower", "dir": "lower",
"floor": 18896, "floor": 15116,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 4724 "value": 3779
}, },
"durable.sN.mixwrite.ops_sec": { "durable.sN.mixwrite.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 120, "floor": 123,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 480 "value": 494
}, },
"durable.sN.mixwrite.p50us": { "durable.sN.mixwrite.p50us": {
"dir": "lower", "dir": "lower",
"floor": 2152, "floor": 1148,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 538 "value": 287
}, },
"durable.sN.mixwrite.p99us": { "durable.sN.mixwrite.p99us": {
"dir": "lower", "dir": "lower",
"floor": 23552, "floor": 2932,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 5888 "value": 733
}, },
"durable.sN.query.ops_sec": { "durable.sN.query.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 262329, "floor": 333333,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 1049317 "value": 1333333
}, },
"durable.sN.query.p50us": { "durable.sN.query.p50us": {
"dir": "lower", "dir": "lower",
@ -166,13 +172,13 @@
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 2 "value": 1
}, },
"durable.sN.read.ops_sec": { "durable.sN.read.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 313558, "floor": 298329,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 1254233 "value": 1193317
}, },
"durable.sN.read.p50us": { "durable.sN.read.p50us": {
"dir": "lower", "dir": "lower",
@ -184,49 +190,223 @@
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 1 "value": 2
}, },
"durable.sN.seed.ops_sec": { "durable.sN.seed.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 1116, "floor": 1139,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 4466 "value": 4556
}, },
"durable.sN.seed.p50us": { "durable.sN.seed.p50us": {
"dir": "lower", "dir": "lower",
"floor": 840, "floor": 828,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 210 "value": 207
}, },
"durable.sN.seed.p99us": { "durable.sN.seed.p99us": {
"dir": "lower", "dir": "lower",
"floor": 2536, "floor": 2000,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 634 "value": 500
}, },
"durable.sN.write.ops_sec": { "durable.sN.write.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 576, "floor": 1153,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 2304 "value": 4614
}, },
"durable.sN.write.p50us": { "durable.sN.write.p50us": {
"dir": "lower", "dir": "lower",
"floor": 1772, "floor": 828,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 443 "value": 207
}, },
"durable.sN.write.p99us": { "durable.sN.write.p99us": {
"dir": "lower", "dir": "lower",
"floor": 2716, "floor": 2172,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 679 "value": 543
},
"growth.available": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 1
},
"growth.int.noswap.bytes_per_row": {
"dir": "lower",
"floor": 400,
"tolerance_pct": 10,
"value": 100
},
"growth.int.noswap.doublings": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 3
},
"growth.int.noswap.p99_departure_decile": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.int.noswap.read_p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.int.noswap.read_p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 1
},
"growth.int.noswap.rows": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 100,
"value": 20000
},
"growth.int.noswap.rss_kb": {
"dir": "lower",
"floor": 23936,
"tolerance_pct": 100,
"value": 5984
},
"growth.int.swap.bytes_per_row": {
"dir": "lower",
"floor": 400,
"tolerance_pct": 10,
"value": 100
},
"growth.int.swap.doublings": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 3
},
"growth.int.swap.p99_departure_decile": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.int.swap.read_p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.int.swap.read_p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 1
},
"growth.int.swap.rows": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 100,
"value": 20000
},
"growth.int.swap.rss_kb": {
"dir": "lower",
"floor": 23952,
"tolerance_pct": 100,
"value": 5988
},
"growth.text.noswap.bytes_per_row": {
"dir": "lower",
"floor": 1296,
"tolerance_pct": 10,
"value": 324
},
"growth.text.noswap.doublings": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 2
},
"growth.text.noswap.p99_departure_decile": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.text.noswap.read_p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.text.noswap.read_p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 1
},
"growth.text.noswap.rows": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 100,
"value": 20000
},
"growth.text.noswap.rss_kb": {
"dir": "lower",
"floor": 41216,
"tolerance_pct": 100,
"value": 10304
},
"growth.text.swap.bytes_per_row": {
"dir": "lower",
"floor": 1296,
"tolerance_pct": 10,
"value": 324
},
"growth.text.swap.doublings": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 2
},
"growth.text.swap.p99_departure_decile": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.text.swap.read_p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"growth.text.swap.read_p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 1
},
"growth.text.swap.rows": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 100,
"value": 20000
},
"growth.text.swap.rss_kb": {
"dir": "lower",
"floor": 41216,
"tolerance_pct": 100,
"value": 10304
}, },
"ram.s1.mixread.ops_sec": { "ram.s1.mixread.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 22384, "floor": 2239,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 89538 "value": 8956
}, },
"ram.s1.mixread.p50us": { "ram.s1.mixread.p50us": {
"dir": "lower", "dir": "lower",
@ -242,33 +422,33 @@
}, },
"ram.s1.mixwrite.ops_sec": { "ram.s1.mixwrite.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 2487, "floor": 248,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 9948 "value": 995
}, },
"ram.s1.mixwrite.p50us": { "ram.s1.mixwrite.p50us": {
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 1 "value": 0
}, },
"ram.s1.mixwrite.p99us": { "ram.s1.mixwrite.p99us": {
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 2 "value": 1
}, },
"ram.s1.msgrate.msgs_sec": { "ram.s1.msgrate.msgs_sec": {
"dir": "higher", "dir": "higher",
"floor": 2087508, "floor": 419674,
"tolerance_pct": 15, "tolerance_pct": 15,
"value": 16700066 "value": 3357394
}, },
"ram.s1.query.ops_sec": { "ram.s1.query.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 247402, "floor": 340136,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 989609 "value": 1360544
}, },
"ram.s1.query.p50us": { "ram.s1.query.p50us": {
"dir": "lower", "dir": "lower",
@ -284,9 +464,9 @@
}, },
"ram.s1.read.ops_sec": { "ram.s1.read.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 274393, "floor": 369276,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 1097574 "value": 1477104
}, },
"ram.s1.read.p50us": { "ram.s1.read.p50us": {
"dir": "lower", "dir": "lower",
@ -302,87 +482,87 @@
}, },
"ram.s1.seed.ops_sec": { "ram.s1.seed.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 61297, "floor": 470366,
"tolerance_pct": 15, "tolerance_pct": 15,
"value": 245188 "value": 1881467
}, },
"ram.s1.seed.p50us": { "ram.s1.seed.p50us": {
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 15, "tolerance_pct": 15,
"value": 4 "value": 0
}, },
"ram.s1.seed.p99us": { "ram.s1.seed.p99us": {
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 15, "tolerance_pct": 15,
"value": 9 "value": 1
}, },
"ram.s1.write.ops_sec": { "ram.s1.write.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 48866, "floor": 294464,
"tolerance_pct": 15, "tolerance_pct": 15,
"value": 195465 "value": 1177856
}, },
"ram.s1.write.p50us": { "ram.s1.write.p50us": {
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 15, "tolerance_pct": 15,
"value": 7 "value": 1
}, },
"ram.s1.write.p99us": { "ram.s1.write.p99us": {
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 15, "tolerance_pct": 15,
"value": 12 "value": 1
}, },
"ram.sN.mixread.ops_sec": { "ram.sN.mixread.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 11229, "floor": 2240,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 44918 "value": 8960
}, },
"ram.sN.mixread.p50us": { "ram.sN.mixread.p50us": {
"dir": "lower", "dir": "lower",
"floor": 236, "floor": 228,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 59 "value": 57
}, },
"ram.sN.mixread.p99us": { "ram.sN.mixread.p99us": {
"dir": "lower", "dir": "lower",
"floor": 432, "floor": 1404,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 108 "value": 351
}, },
"ram.sN.mixwrite.ops_sec": { "ram.sN.mixwrite.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 1247, "floor": 248,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 4990 "value": 995
}, },
"ram.sN.mixwrite.p50us": { "ram.sN.mixwrite.p50us": {
"dir": "lower", "dir": "lower",
"floor": 256, "floor": 248,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 64 "value": 62
}, },
"ram.sN.mixwrite.p99us": { "ram.sN.mixwrite.p99us": {
"dir": "lower", "dir": "lower",
"floor": 516, "floor": 292,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 129 "value": 73
}, },
"ram.sN.msgrate.msgs_sec": { "ram.sN.msgrate.msgs_sec": {
"dir": "higher", "dir": "higher",
"floor": 355876, "floor": 216394,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 2847015 "value": 1731152
}, },
"ram.sN.query.ops_sec": { "ram.sN.query.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 307125, "floor": 340136,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 1228501 "value": 1360544
}, },
"ram.sN.query.p50us": { "ram.sN.query.p50us": {
"dir": "lower", "dir": "lower",
@ -398,9 +578,9 @@
}, },
"ram.sN.read.ops_sec": { "ram.sN.read.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 340692, "floor": 343878,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 1362769 "value": 1375515
}, },
"ram.sN.read.p50us": { "ram.sN.read.p50us": {
"dir": "lower", "dir": "lower",
@ -416,38 +596,38 @@
}, },
"ram.sN.seed.ops_sec": { "ram.sN.seed.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 72890, "floor": 445235,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 291562 "value": 1780943
}, },
"ram.sN.seed.p50us": { "ram.sN.seed.p50us": {
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 3 "value": 0
}, },
"ram.sN.seed.p99us": { "ram.sN.seed.p99us": {
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 7 "value": 1
}, },
"ram.sN.write.ops_sec": { "ram.sN.write.ops_sec": {
"dir": "higher", "dir": "higher",
"floor": 60518, "floor": 286368,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 242072 "value": 1145475
}, },
"ram.sN.write.p50us": { "ram.sN.write.p50us": {
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 6 "value": 1
}, },
"ram.sN.write.p99us": { "ram.sN.write.p99us": {
"dir": "lower", "dir": "lower",
"floor": 100, "floor": 100,
"tolerance_pct": 50, "tolerance_pct": 50,
"value": 9 "value": 2
} }
} }

View file

@ -1,3 +1,4 @@
use fs
use time use time
-- db-bench — iteration 22's load generator. Every measured mode prints -- db-bench — iteration 22's load generator. Every measured mode prints
@ -465,10 +466,135 @@ fn all_mode(n: Int) -> Int {
fn usage() -> Int { fn usage() -> Int {
print_err("usage: db-bench <mode>"); print_err("usage: db-bench <mode>");
print_err(" all N | seed N | read N | query N | write N | wal N"); print_err(" all N | seed N | read N | query N | write N | wal N");
print_err(" mix N C | msgrate N | verify | verify-acked M"); print_err(" mix N C | msgrate N | growth N int|text | growth-verify");
print_err(" verify | verify-acked M");
return 2; return 2;
} }
-- databasev2 1: the process's own resident size, in KiB. Read here rather
-- than sampled by the driver because the driver polls /proc every 250 ms and
-- would miss the value AT a decile boundary; per-row footprint is the headline
-- number of this iteration and deserves an exact reading, not a nearby one.
-- Absence is nil by stdlib convention, so a kernel without VmRSS reports 0
-- and the driver treats the leg as unavailable rather than as zero growth.
fn self_rss_kb() -> Int {
let st = try fs.read_all("/proc/self/status", 16384) catch (e) "";
let i = index_of(st, "VmRSS:");
if i < 0 {
return 0;
}
let rest = substr(st, i + 6, 24);
let n = 0;
let j = 0;
while j < len(rest) {
let c = byte_at(rest, j);
if c >= 48 and c <= 57 {
n = n * 10 + (c - 48);
} else {
if n > 0 {
return n;
}
}
j = j + 1;
}
return n;
}
-- databasev2 1: growth N SHAPE — insert N rows of one reference shape,
-- sampling read latency as the table grows so the driver can plot the CURVE
-- rather than two endpoints. Reports one metric line per decile so the point
-- at which p99 leaves its baseline is a MEASURED sample, not an estimate.
--
-- SHAPE is "int" (Item: two Ints plus a ref, all inline slot words) or "text"
-- (Wide: three Text columns, each a separate db_text allocation on top of the
-- slab slot). Per-row footprint differs by an order of magnitude between them,
-- which is exactly why the driver reports the two separately and never a single
-- "bytes per row".
--
-- The memory CAP is the driver's job (systemd-run --user --scope), not this
-- program's: the sample just grows and reports, so the same binary serves the
-- swap-off and swap-on legs unchanged.
-- after the process is OOM-killed mid-insert, the durable prefix must be
-- intact: rows 1..M all present with the right v and no holes. M is whatever
-- survived -- the claim under test is the SHAPE of the survivor, not its size,
-- because a SIGKILL can land between any two inserts.
fn growth_verify() -> Int {
let seen: map<Int, Int> = {};
let maxk = 0;
for r in from x in Item select x {
set(seen, r.k, r.v);
if r.k > maxk {
maxk = r.k;
}
}
let i = 1;
while i <= maxk {
if has(seen, i) == false {
print_err("growth-verify: hole at ${i} below max ${maxk}");
return 3;
}
if get(seen, i) != item_v(i) {
print_err("growth-verify: row ${i} v ${get(seen, i)} != ${item_v(i)}");
return 3;
}
i = i + 1;
}
print("growthverify ${maxk}");
return 0;
}
fn growth_mode(n: Int, shape: Text) -> Int {
let wide = shape == "text";
if wide == false and shape != "int" {
print_err("db-bench: growth SHAPE must be `int` or `text`");
return 2;
}
let step = n / 10;
if step < 1 {
step = 1;
}
let bref = insert Bucket { tag: "growth" };
let pad = "0123456789abcdef0123456789abcdef";
let i = 1;
while i <= n {
if wide {
insert Wide { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" };
} else {
insert Item { k: i, v: item_v(i), bucket: bref };
}
-- at each decile, sample the read path against what is resident NOW
if i % step == 0 {
let h: map<Int, Int> = {};
let probes = 200;
let pt0 = time.ticks();
let j = 0;
while j < probes {
let key = 1 + (j * step) % i;
let o0 = time.ticks();
if wide {
for r in from x in Wide where x.k == key take 1 select x {
hist_add(h, time.ticks() - o0);
}
} else {
for r in from x in Item where x.k == key take 1 select x {
hist_add(h, time.ticks() - o0);
}
}
j = j + 1;
}
let pel = time.ticks() - pt0;
-- op name carries the decile so the driver keys each sample distinctly
report("growth${i / step}", probes, pel, h);
-- rows and resident KiB at this decile: the driver divides to get the
-- per-row footprint for THIS shape
print("growthrss ${i / step} ${i} ${self_rss_kb()}");
}
i = i + 1;
}
print("growthdone ${n}");
return 0;
}
fn main(args: multi Text) -> Int { fn main(args: multi Text) -> Int {
if len(args) < 1 { if len(args) < 1 {
return usage(); return usage();
@ -476,6 +602,9 @@ fn main(args: multi Text) -> Int {
if args[0] == "verify" { if args[0] == "verify" {
return verify(); return verify();
} }
if args[0] == "growth-verify" {
return growth_verify();
}
if len(args) < 2 { if len(args) < 2 {
return usage(); return usage();
} }
@ -508,6 +637,12 @@ fn main(args: multi Text) -> Int {
if args[0] == "msgrate" { if args[0] == "msgrate" {
return msgrate_mode(n); return msgrate_mode(n);
} }
if args[0] == "growth" {
if len(args) < 3 {
return usage();
}
return growth_mode(n, args[2]);
}
if args[0] == "mix" { if args[0] == "mix" {
if len(args) < 3 { if len(args) < 3 {
return usage(); return usage();

View file

@ -23,6 +23,20 @@ class Meta {
val: Int val: Int
} }
-- databasev2 1: the TEXT-HEAVY reference shape. `Item` above is the Int-only
-- reference as it stands (two Ints plus a ref, all inline slot words), so this
-- is its counterpart: every row drags a separate db_text allocation per Text
-- column on top of its slab slot. Per-row footprint differs by an order of
-- magnitude between the two, which is why a single "bytes per row" number is
-- meaningless and the growth mode reports the two shapes separately.
@table(name: "wide", index: [k])
class Wide {
k: Int
a: Text
b: Text
note: Text
}
-- mix actors dump their per-op histograms here (kind 0 = read, -- mix actors dump their per-op histograms here (kind 0 = read,
-- 1 = write); main scans and merges — exact aggregate percentiles, -- 1 = write); main scans and merges — exact aggregate percentiles,
-- and the merge itself dogfoods the store. -- and the merge itself dogfoods the store.

View file

@ -67,3 +67,89 @@ not the limiting factor for any current workload).
the ~55× gap is one fdatasync per statement (~220µs each). the ~55× gap is one fdatasync per statement (~220µs each).
**Owner: iteration 23** (io_uring group-commit) — its acceptance is **Owner: iteration 23** (io_uring group-commit) — its acceptance is
literally this number moving while the crash battery stays green. literally this number moving while the crash battery stays green.
## 5. The RAM ceiling: footprint, and how the engine actually dies
**Measured 2026-08-27** (databasev2 1), rootless cgroup v2 via
`systemd-run --user --scope -p MemoryMax=N -p MemorySwapMax=M`, dev box.
### Per-row resident footprint, by shape
| Shape | Columns | Steady-state | Doubling steps |
| --- | --- | --- | --- |
| Int-only (`Item`) | 2× Int + 1 ref | **96.5 B/row** | at ~24k and ~48k rows |
| Text-heavy (`Wide`) | 1× Int + 3× Text | **320.6 B/row** | at ~24k and ~48k rows |
**3.3×**, not the "order of magnitude" an earlier doc asserted. Two shapes are
published, never one number: a `Text` column is a separate `db_text` allocation
per row on top of the slab slot, so a row count cannot bound RAM.
**Read the steady-state figure as the median of per-interval marginals, not a
two-point slope.** The id hash and index buckets are open-addressing pow2 and
double periodically; a two-point slope lands arbitrarily on or off a doubling
and swings 2× (96 vs 205 B/row measured for the same shape). The doublings are
reported separately because a **transient RSS step is exactly what a
resident-footprint budget must leave headroom for** — a budget without it fires
during a rehash rather than at a steady-state threshold. Direct input to
databasev2 2's budget design.
### How it dies — and it is not the way the docs claimed
| Allocator | Ceiling | Failure mode |
| --- | --- | --- |
| VM object arena | `WO_HEAP_MB`, checked | `trap 4 … out of memory`, rc=1, reportable. Verified at 4 and 16 MiB |
| table storage (slabs + heap values) | **none** | **SIGKILL, signal 9** (shell rc 137). Verified at 360 000 rows / 57 188 KiB under a 64 MiB cap |
Three docs asserted that an allocation failure surfaces as a catchable
`WO_T_OOM`. For table storage it does not: `vm.overcommit_memory = 0` means
`malloc` succeeds and the kernel kills the process when it *touches* the pages,
so the checked-`malloc` code never runs. The trap path is real, but it is the
arena's.
**Consequence, and the strongest available argument for databasev2 2's byte
budget:** a declared budget is the *only* way table storage can acquire a
checked ceiling, because `malloc` under default overcommit will never report a
problem. **Owner: databasev2 2.**
### Swap: the ceiling that does not announce itself
| Leg | 900 000 Int rows, 64 MiB cap | Wall | Final RSS |
| --- | --- | --- | --- |
| swap OFF (`MemorySwapMax=0`) | **SIGKILL at 360 000 rows** | — | 57 188 KiB |
| swap ON (256 MiB) | **completed, exit 0** | **148 s** | 62 264 KiB (rest paged out) |
| uncapped | completed, exit 0 | **150 s** | 169 416 KiB |
**Swap cost ~1%.** A prior draft predicted "latency collapse"; the prediction had
the wrong sign. Inserting is append-mostly, so cold pages are written once and
never re-read — paging is sequential and off the critical path. The swap device
is a real disk file (`/swap.img`; no zram, zswap disabled), so this is genuine
disk paging.
**Do not generalise this to "swap is fine".** It measures an append-mostly
workload. A random-read workload over a table larger than the cap is where the
collapse should appear, and it is **not yet measured** — which matters, because
that is exactly the access pattern databasev2 2's `resident: keys` creates.
The operational consequence is that the RAM ceiling has two shapes and neither
reports itself: without swap the process vanishes on signal 9, with swap it
keeps returning 0 while serving from disk. A budget that fires at a *declared
threshold* is the only one that can speak before either happens.
### Durability across the ceiling
60 000 Int rows, 8 MiB cap, swap off, `WO_DATA` set — the process is OOM-killed
mid-insert, then replayed:
| Claim | Result |
| --- | --- |
| the survivor is a contiguous prefix | ✅ ~40 000 rows, rows 1..M all present |
| every surviving row's payload is correct | ✅ every `v` matches `item_v(i)` |
| the truncated tail is not read as corruption | ✅ replay exits 0 |
**Ack-after-fsync holds through an OOM kill** — the one shutdown path that skips
every cleanup handler. Gated as `db-bench`'s `ceiling` leg, which asserts the
*shape* of the survivor rather than its size: where the SIGKILL lands is the
scheduler's business, so `rows_recovered` carries ±100% tolerance. The leg
asserts the exit but never records it as a metric, so that when databasev2 2's
byte budget turns the kill into a checked refusal, the gate does not fail on the
improvement.

View file

@ -67,6 +67,57 @@ behind this board; live Obsidian Dataview views:
## ▶ NEXT PLAN ## ▶ NEXT PLAN
### Landed 2026-08-27 — databasev2 1, the RAM ceiling measured
**Implemented last time (2026-08-27):** databasev2 1 refined (three forks
settled) and implemented. A text-heavy `Wide` reference shape beside the
Int-only `Item`; `growth N int|text` in the db-bench sample, reading its OWN
`/proc/self/status` RSS at each decile because the driver's 250 ms poll misses
the value *at* a boundary; `growth-verify`, which asserts the survivor of a
crash is a contiguous intact prefix; and two harness legs — four footprint legs
under a rootless cgroup v2 cap, and a `ceiling` leg that deliberately dies at
the cap and then replays. 121 checks, 0 failures.
**Key findings (measured, not asserted):** per-row footprint is **96.5–100 B**
Int-only and **320.6–324 B** text-heavy — **3.3×**, not the "order of magnitude"
three docs asserted. Read as the median of per-decile marginals, never a
two-point slope: index doublings make a two-point read swing 2× (96 vs 205 B/row
for one shape). **Two predictions in the iteration's own premise were wrong.**
The ceiling is not a catchable `WO_T_OOM` for table storage — it is **SIGKILL,
signal 9**, because `vm.overcommit_memory = 0` lets `malloc` succeed and the
kernel kills on page *touch*, so the checked path never runs (the VM arena is
the opposite: `WO_HEAP_MB` is checked and traps). And swap is not "latency
collapse": 900 000 rows inside a 64 MiB cap with swap finished in **148 s
against 150 s uncapped** — ~1%, on a real disk swap file with no zram. Also
measured: **ack-after-fsync holds through an OOM kill** — ~40 000 rows came back
as an intact prefix, no holes, not read as corruption.
**Learned:** an append-mostly workload never re-touches its cold pages, so swap
costs it nothing — the collapse belongs to *random reads* over an oversized
table, which is precisely the pattern iteration 2's `resident: keys` creates and
is **still unmeasured**. The RAM ceiling therefore has two shapes and neither
announces itself: without swap the process vanishes on signal 9, with swap it
keeps returning 0 while serving from disk. That is the argument for a budget
that fires at a declared threshold instead of at exhaustion.
**Dependencies unblocked — one, by *removing* it:** iteration 2's
resident-footprint budget default was to be derived from "swap onset". **There is
no onset.** Swap-off jumps straight from working to SIGKILL; swap-on shows no
degradation to detect. Iteration 2 must pick its budget on other grounds rather
than wait on a number this slice cannot produce. Iteration 3's replay baseline is
still NOT delivered — `bench/baseline.json` times no replay.
**Next steps:** the read-heavy-over-cap leg is the single most valuable
follow-up, and it is what makes `p99_departure_decile` mean anything (the
footprint legs never approach their 512 MiB cap, so it is legitimately 0 today).
Then iteration 2's 5c/5d.
**`.dev/reference` used:** none. Sources were the kernel's own interfaces —
cgroup v2 `memory.max`/`memory.swap.max`, `/proc/self/status`, `/proc/swaps` and
`vm.overcommit_memory`.
---
### Landed 2026-08-25 — packaging + release pipeline (off-chain, no story) ### Landed 2026-08-25 — packaging + release pipeline (off-chain, no story)
**Implemented last time (2026-08-25):** the toolchain became installable **Implemented last time (2026-08-25):** the toolchain became installable
@ -620,8 +671,11 @@ declares a budget. Rows live in `malloc`'d slabs whose addresses are stable
forever; there is no eviction, spill or paging anywhere in `database/src/`; the forever; there is no eviction, spill or paging anywhere in `database/src/`; the
WAL never checkpoints so boot replays all history; and durability is one WAL never checkpoints so boot replays all history; and durability is one
process-global `WO_DATA`, so no table can say it matters more than another. An process-global `WO_DATA`, so no table can say it matters more than another. An
allocation failure *is* a clean catchable `WO_T_OOM` — but swap thrash arrives allocation failure is a clean catchable `WO_T_OOM` **only in the VM arena** —
first and carries no error signal at all. table storage has no ceiling and is SIGKILLed instead (measured, databasev2 1).
Where swap exists the ceiling may never announce itself at all: an append-mostly
900k-row run finished *at uncapped speed* inside a 64 MiB cap (148 s vs 150 s),
serving from disk with no error signal.
**The lever** is per-table storage modes, which is why this track has a grammar **The lever** is per-table storage modes, which is why this track has a grammar
iteration. Six pending iterations moved here from the language track (their old iteration. Six pending iterations moved here from the language track (their old
@ -630,7 +684,7 @@ the language arc as v1 history.
| # | Iteration | State | | # | Iteration | State |
| --- | --- | --- | | --- | --- | --- |
| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | ⬜ `readiness: refine` — its three forks are open, so despite being first it is NOT startable without a brainstorm — nobody here can say what happens at 90% RAM. Curve not cliff: swap onset, latency departure, the three exits (checked trap / swap thrash / OOM killer), and `kill -9` durability *at exhaustion*. Output is `perf-targets.md` + baseline rows, not prose | | 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | 🔄 **MEASURED 2026-08-27** — `readiness: ready`, `status: in-progress` (two criteria outstanding), forks settled, harness landed (121 checks). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Outstanding: the **random-read-over-cap** collapse (unmeasured, and it is `resident: keys`'s own access pattern) and iteration 3's replay baseline. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from |
| 2 | [per-table storage: `durable` and `resident`](databasev2/02-table-storage-modes.md) | 🔄 **the language enrichment — the `durable` half is DONE and usable.** Two optional `@table` keys, `durable: true\|false` and `resident: all\|keys`, both defaulting to today's behaviour (all 28 existing declarations compile unchanged, no golden moved). Landed: the grammar, WO-E224 (a durable `ref` into a volatile table is refused), `.wob` v7 carrying both properties in spare `flags` bits, `durable: false` actually skipping the WAL (measured: 50 inserts → 1500 bytes durable, **0** volatile) with a mode-mismatch startup refusal, plus offset capture and read-a-row-from-an-offset. Outstanding: 5c/5d (the id→offset map and rewiring `wo_row_ptr`'s 11 call sites, slab scans and `@unique`/FK across the boundary — not yet written up), the two runtime refusals, and closeout. [spec](../superpowers/specs/2026-08-26-table-residency-design.md) · [plan](../superpowers/plans/2026-08-26-table-residency.md) | | 2 | [per-table storage: `durable` and `resident`](databasev2/02-table-storage-modes.md) | 🔄 **the language enrichment — the `durable` half is DONE and usable.** Two optional `@table` keys, `durable: true\|false` and `resident: all\|keys`, both defaulting to today's behaviour (all 28 existing declarations compile unchanged, no golden moved). Landed: the grammar, WO-E224 (a durable `ref` into a volatile table is refused), `.wob` v7 carrying both properties in spare `flags` bits, `durable: false` actually skipping the WAL (measured: 50 inserts → 1500 bytes durable, **0** volatile) with a mode-mismatch startup refusal, plus offset capture and read-a-row-from-an-offset. Outstanding: 5c/5d (the id→offset map and rewiring `wo_row_ptr`'s 11 call sites, slab scans and `@unique`/FK across the boundary — not yet written up), the two runtime refusals, and closeout. [spec](../superpowers/specs/2026-08-26-table-residency-design.md) · [plan](../superpowers/plans/2026-08-26-table-residency.md) |
| 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ⬜ snapshot + truncate: disk reclaimed, replay bounded | | 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ⬜ snapshot + truncate: disk reclaimed, replay bounded |
| 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ⬜ **`readiness: ready` — the one startable iteration in the repo** (four forks confirmed settled 2026-08-20). Close the 66× gap iteration 22 measured (durable 4.5k vs ram 297k inserts/s) | | 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ⬜ **`readiness: ready` — the one startable iteration in the repo** (four forks confirmed settled 2026-08-20). Close the 66× gap iteration 22 measured (durable 4.5k vs ram 297k inserts/s) |

View file

@ -44,7 +44,14 @@ The bill comes due at the ceiling. Read from the engine as it stands:
Worth being precise, because the failure mode determines the fix — and the good Worth being precise, because the failure mode determines the fix — and the good
news is that the engine's own behaviour is clean: news is that the engine's own behaviour is clean:
**An allocation failure is a catchable trap, not a crash.** Every `malloc` in **Corrected 2026-08-27 by measurement.** This section used to open "an
allocation failure is a catchable trap, not a crash", and that is true only of
the VM arena. Table storage has no ceiling, and with `vm.overcommit_memory = 0`
its `malloc` never fails — the process is **SIGKILLed** (rc=137, measured at
360 000 rows under a 64 MiB cap). The checked path below is real, but it is the
arena's, not the store's. See [iteration 1](01-ram-ceiling-measurement.md).
Every `malloc` in
the row encoder is checked and jumps to an `oom` label; `DB_ERR_OOM` maps to the row encoder is checked and jumps to an `oom` label; `DB_ERR_OOM` maps to
`WO_T_OOM`, which a program can `try`/`catch`. So a writeonce program that runs `WO_T_OOM`, which a program can `try`/`catch`. So a writeonce program that runs
out of memory *refuses the insert* rather than corrupting or dying. That is a out of memory *refuses the insert* rather than corrupting or dying. That is a
@ -61,9 +68,22 @@ battery proves that much.
So the honest problem statement is not "malloc fails". It is: **there is no So the honest problem statement is not "malloc fails". It is: **there is no
declared budget, no back-pressure as the budget is approached, and no way to declared budget, no back-pressure as the budget is approached, and no way to
distinguish data that must be resident from data that merely is.** Iteration distinguish data that must be resident from data that merely is.**
[1](01-ram-ceiling-measurement.md) exists to replace this paragraph with
numbers before anything is designed on top of it. **Iteration [1](01-ram-ceiling-measurement.md) has now measured this
(2026-08-27), and it strengthened the statement rather than softening it.** A row
costs **96.5–100 B** Int-only and **320.6–324 B** text-heavy (3.3× apart, so no
single per-row number can bound RAM). At the ceiling the engine has exactly two
behaviours and **neither one tells anybody**: without swap the process is
**SIGKILLed on signal 9** — table storage has no checked ceiling, and under
`vm.overcommit_memory = 0` its `malloc` succeeds and the kernel kills on page
touch — and with swap it **keeps returning 0 while serving from disk**, finishing
900 000 rows in 148 s against 150 s uncapped. Durability is the one thing that
does hold: acked writes came back as an intact prefix across an OOM kill.
That is why "back-pressure at exhaustion" is not a design option. Exhaustion
either kills without warning or never arrives. Only a **declared threshold** can
speak in time.
## The lever: per-table storage modes ## The lever: per-table storage modes
@ -113,7 +133,7 @@ before its mechanism existed; the history is in
| # | Iteration | Delivers | Needs | | # | Iteration | Delivers | Needs |
| --- | --- | --- | --- | | --- | --- | --- | --- |
| 1 | [RAM ceiling: measure the breaking point](01-ram-ceiling-measurement.md) | what actually happens from 50% RAM to OOM — swap onset, latency cliff, trap behaviour, `kill -9` survival | nothing; extends iteration 22's harness | | 1 | [RAM ceiling: measure the breaking point](01-ram-ceiling-measurement.md) | 🔄 **measured 2026-08-27**: footprint per shape (3.3× apart), the two silent exits (SIGKILL vs swap-serving-from-disk at ~uncapped speed), and ack-after-fsync surviving an OOM kill. Outstanding: the random-read-over-cap collapse, and a replay baseline | nothing; extends iteration 22's harness |
| 2 | [per-table storage](02-table-storage-modes.md) | the grammar: `durable: true\|false` and `resident: all\|keys`, per table, replacing the global `WO_DATA` all-or-nothing. **In progress — the `durable` half is done** | 1 for the budget default | | 2 | [per-table storage](02-table-storage-modes.md) | the grammar: `durable: true\|false` and `resident: all\|keys`, per table, replacing the global `WO_DATA` all-or-nothing. **In progress — the `durable` half is done** | 1 for the budget default |
| 3 | [WAL checkpoint](03-wal-checkpoint.md) *(was language 32)* | snapshot + truncate: disk reclaimed, replay bounded | 4 composes | | 3 | [WAL checkpoint](03-wal-checkpoint.md) *(was language 32)* | snapshot + truncate: disk reclaimed, replay bounded | 4 composes |
| 4 | [io_uring group commit](04-io-uring-commit.md) *(was language 23)* | close the 66× durable/RAM write gap (4.5k vs 297k inserts/s) | the arc (landed) | | 4 | [io_uring group commit](04-io-uring-commit.md) *(was language 23)* | close the 66× durable/RAM write gap (4.5k vs 297k inserts/s) | the arc (landed) |

View file

@ -1,149 +1,263 @@
--- ---
track: databasev2 track: databasev2
iteration: "1" iteration: "1"
status: pending status: in-progress
readiness: refine readiness: ready
--- ---
# databasev2 1 — the RAM ceiling: measure the breaking point before designing for it # databasev2 1 — the RAM ceiling: measure the breaking point
> Part of [Story — databasev2: the database beyond RAM](00-story.md). > Part of [Story — databasev2: the database beyond RAM](00-story.md).
> >
> **Refined 2026-08-27; the three forks are settled below and the decisions are
> locked.** No spec document: the deliverable is numbers plus a harness leg, and
> the design fits in this file — the same call
> [7](07-single-file-db.md) makes.
>
> **First because the repo's own doctrine says so.** "Always inspect crashsites. > **First because the repo's own doctrine says so.** "Always inspect crashsites.
> Always measure. Never assume." Every later iteration in this track — the > Always measure. Never assume." Two other iterations already cite numbers this
> storage modes' defaults, the eviction policy, the tiering threshold — is a > one was supposed to produce: [2](02-table-storage-modes.md)'s resident-footprint
> decision that should follow from a number. Right now nobody in this project > budget defaults to a fraction of host memory whose value comes from here, and
> can say what happens to a writeonce program at 90% of RAM, and designing > [3](03-wal-checkpoint.md)'s before/after replay criterion has no "before"
> tiering without that is guessing with extra steps. > because `bench/baseline.json` carries 75 metrics and **zero** for replay,
> restart, boot or recovery. Iteration 22 proved restart *correctness*; it never
> timed it.
## Goals ## The design, as settled
- **Find the curve, not the cliff.** Not "does it die" — it dies, everything **Measure the curve, not the cliff.** Everything dies at the ceiling; what
does. What matters is the shape on the way down: at what fraction of RAM does matters is the shape on the way down — where p99 leaves its 1µs baseline, what
p99 read latency leave its 1µs baseline, what does insert throughput do as insert throughput does as slabs stop coming from a warm allocator, and how much
slabs stop coming from a warm allocator, and how much warning is there between warning there is between "fine" and "unusable".
"fine" and "unusable".
- **Characterise all three exits.** The engine can leave the happy path three
ways and they are not equally survivable: a checked `malloc` failure
(`DB_ERR_OOM` → `WO_T_OOM`, a catchable trap — the clean one), swap thrash
(no trap, no error, just latency collapse — the dangerous one because nothing
reports it), and the external OOM killer (`SIGKILL`, skipping every shutdown
path). Establish which arrives first under realistic limits, because the
answer determines whether the fix is back-pressure or eviction.
- **Prove the durability floor holds at the ceiling.** Iteration 22's `kill -9`
battery proved acked writes survive under load. Re-run it *at memory
exhaustion*, which is a different and nastier state — an allocation failure
mid-commit is exactly where an ack-before-durable bug would hide.
- **Publish numbers others can build on.** The output is a section in
`perf-targets.md` and rows in `bench/baseline.json`, not a paragraph of
prose. A measurement that only printed once is not a measurement.
## Phases ### Fork 1 — the limit mechanism: rootless cgroup v2 via `systemd-run --user`
### Phase A — a workload that can actually reach the ceiling `systemd-run --user --scope -p MemoryMax=N -p MemorySwapMax=M`. Verified on the
dev box: the `memory` controller is delegated to
`user.slice/user-<uid>.slice`, a scope's `memory.max` reads back exactly as set,
and no passwordless sudo is needed. Being cgroup-scoped also isolates the
measurement from whatever else the box is doing, which matters — the dev box was
at 22.9 of 31.7 GiB with 4.6 GiB of swap already in use when this was refined.
- Extend `docs/examples/db-bench` with a growth mode: insert until a target RSS `ulimit -v` is **rejected**: it bounds address space, not resident set, which is
fraction, holding row shape and index count constant so the variable is size the wrong quantity for an engine that `malloc`s slabs — and it is actively
alone. broken under ASan, whose huge virtual reservations trip it long before any real
- Run it under an explicit memory limit (a cgroup or `ulimit`) rather than on a memory pressure.
big box — "it survived on a 64 GB workstation" measures the workstation.
- Record RSS against row count so the per-row overhead is known: slab headroom,
the id hash, the secondary-index multimaps and the per-row engine-owned values
(`db_text`, `db_rec`, `db_multi`, `db_map` are each their own allocation).
- Verify: RSS growth is linear and its slope is written down; the run is
reproducible twice within the tolerance policy iteration 22 established.
### Phase B — the latency and throughput curve If the mechanism is unavailable (no systemd, no delegation), the harness **skips
the growth legs loudly and names why**. It must never silently fall back to
measuring an uncapped box, because "it survived on a 32 GiB workstation"
measures the workstation.
- Sample read p50/p99, query p99 and insert throughput at fixed fractions of the ### Fork 2 — the reference shapes: both, reported separately
limit, so the result is a curve rather than two endpoints.
- Separate the two effects deliberately: allocator pressure (still resident) and
swap (no longer resident). They have different fixes and conflating them would
send iteration 6 after the wrong one.
- Include the DB-actor path, since a cross-shard statement's reply materialises
a copy — memory pressure and the actor RPC interact and nobody has looked.
- Verify: the curve is recorded per metric class with iteration 22's per-class
tolerances; the swap onset point is identified, not interpolated.
### Phase C — the three exits, deliberately triggered Per-row footprint differs substantially between an Int-only row and a text-heavy
one, because a `Text` column is a separate `db_text` allocation per row on top of
the slab slot. **Measured 2026-08-27: 96.5 B/row Int-only vs 320.6 B/row with
three Text columns — 3.3×.** An earlier draft of this section said "an order of
magnitude"; that was an unmeasured guess and this iteration exists to replace
exactly that kind of claim. 3.3× is still more than enough to make a single
"bytes per row" number useless, which is the decision it was supporting.
- Drive a checked allocation failure and confirm `WO_T_OOM` is catchable, the `db-bench` already supplies half of this: `items` (`k: Int`, `v: Int`, plus a
insert is refused whole, no partial row or index entry is left, and the `bucket` ref) is the Int-only reference as it stands. The work is one text-heavy
process continues serving. shape beside it, with footprint reported per shape.
- Drive swap thrash and record what a client sees. This is the case with no
error signal at all, and naming it is most of the value of this iteration.
- Drive the OOM killer under a cgroup limit and confirm what survives: replay
the WAL and check every acked write is present.
- Verify: the trap path leaves no torn state (row count and index agree after a
refused insert); replay after `SIGKILL` at exhaustion loses no acked write.
### Phase D — write it down where decisions get made ### Fork 3 — swap: in scope, as a controlled dimension
- A `perf-targets.md` section with the curve, the swap onset, the per-row Not a confound to wish away — `MemorySwapMax` is the knob that separates the two
overhead and the exit characterisation. exits this iteration exists to characterise. Both were measured, and **both
- Baseline rows for the growth metrics so a regression is caught by the existing turned out differently than this iteration predicted.**
gate rather than by a person remembering.
- A short statement of what the numbers *imply* for iterations 2, 5 and 6 — | Leg | Predicted | Measured |
which is the point of going first. | --- | --- | --- |
- Verify: `just db-bench` green against the extended baseline; the gate bites | swap-off | catchable `WO_T_OOM` from checked `malloc` | **SIGKILL, signal 9** (shell rc 137). No trap, no message |
when a growth metric is doctored. | swap-on | latency collapse | **no degradation at all**: 148 s vs 150 s uncapped |
**Prediction 1 was wrong because of overcommit.** With `vm.overcommit_memory = 0`
`malloc` succeeds and the process dies when it *touches* the pages, so table
storage never gets the chance to report failure. The trap path is real but
belongs to a different allocator:
| Allocator | Ceiling | Failure mode |
| --- | --- | --- |
| VM object arena | `WO_HEAP_MB`, checked | `trap 4` / `WO_T_OOM`, exit 1, reportable |
| table storage (slabs + heap values) | **none** | SIGKILL under overcommit |
**This is the strongest argument available for [iteration 2](02-table-storage-modes.md)'s
byte budget:** a declared budget is the only way table storage can acquire a
checked ceiling, because `malloc` under default overcommit will never tell it
there is a problem.
**Prediction 2 was wrong because of access pattern.** 900 000 Int rows under a
64 MiB cap with 256 MiB of swap finished in **148 s** with RSS pinned at 62 MiB;
the same workload uncapped took **150 s** at 165 MiB RSS. Swap cost
approximately nothing. The reason is that inserting is append-mostly: cold pages
are written out once and never read again, so paging is sequential and off the
critical path. The swap is a real disk file (`/swap.img`, no zram, zswap
disabled), so this is genuine disk paging, not compressed RAM.
**The correct generalisation is narrower than "swap is fine".** This measures an
append-mostly workload. A workload that reads randomly across a table larger
than the cap is the one that collapses, and this iteration did *not* measure
that — see Outstanding.
## Progress
| Piece | State |
| --- | --- |
| `Wide` text-heavy reference shape (`db-bench/types.wo`) | ✅ |
| `growth N int\|text` — insert, per-decile RSS and read latency | ✅ |
| the sample reads its OWN RSS via `/proc/self/status` | ✅ — the driver polls every 250 ms and would miss the value *at* a decile boundary |
| `growth-verify` — the survivor is a contiguous intact prefix | ✅ |
| rootless cap wrapper, swap on/off legs | ✅ 4 footprint legs |
| footprint metric = **median of marginals**, doublings counted separately | ✅ |
| `ceiling` leg: dies at the cap, then replay must be intact | ✅ gated |
| baseline + tolerance policy | ✅ 121 checks; footprint at ±10%, kill-timing metrics at ±100% |
| `perf-targets.md` §5 | ✅ |
| **resident-footprint fraction for [iteration 2](02-table-storage-modes.md)** | ⬜ **not delivered — the premise it rested on is false**, see Outstanding |
| **replay/restart baseline for [iteration 3](03-wal-checkpoint.md)** | ⬜ not delivered |
| **the random-read-over-cap collapse** | ⬜ not measured |
## Measured
Footprint, reproducible inside 2% across runs:
| What | Int-only (`Item`) | Text-heavy (`Wide`) |
| --- | --- | --- |
| steady-state footprint | **96.5–100 B/row** | **320.6–324 B/row** |
| doubling steps | 3 (at ~24k and ~48k rows) | 2 |
| base process RSS | ≈ 3.9 MiB, excluded from the per-row figure | same |
Ratio **3.3×** — not the "order of magnitude" an earlier draft asserted. Enough
on its own to make a single "bytes per row" number useless, which is the decision
it was supporting ([2](02-table-storage-modes.md), fork 5).
The ceiling, 60 000 Int rows under an 8 MiB cap with swap off and `WO_DATA` set:
| Question | Answer |
| --- | --- |
| how does it die? | **SIGKILL, signal 9.** No refusal, no diagnostic |
| what survives? | **a contiguous intact prefix** — ~40 000 rows, every `v` correct, no holes, not reported as corruption |
**Ack-after-fsync holds through an OOM kill.** That is the one shutdown path
which skips every cleanup handler, and the durable prefix came back whole.
**The finding that matters most is the swap leg succeeding.** It did not fail,
did not warn, and returned 0. A deployment in that state looks healthy while
serving from disk. That is the exit with no error signal, and it is why
[iteration 5](05-bounded-tables-eviction.md)'s back-pressure must act at a
declared threshold rather than at exhaustion — exhaustion either kills without
warning or silently does not arrive.
## Acceptance Criteria ## Acceptance Criteria
- **Given** the growth workload under a fixed memory limit, **when** it runs Met:
twice, **then** RSS-per-row agrees within the tolerance policy and the slope
is recorded in `perf-targets.md`. - **Given** the growth workload under a fixed cap, **when** it runs twice,
- **Given** the workload at rising RAM fractions, **when** latency is sampled, **then** RSS-per-row agrees inside tolerance and the slope is recorded per
**then** the fraction at which read p99 first leaves its baseline is shape. ✅ inside 2%; `perf-targets.md` §5.
identified as a measured point, not an estimate. - **Given** the swap-off leg, **when** the cap is exceeded, **then** the exit is
- **Given** a deliberately induced allocation failure, **when** an insert is identified and recorded. ✅ **SIGKILL, signal 9** — not the catchable trap this
attempted, **then** it traps `WO_T_OOM` catchably, the table's row count is criterion originally expected, which is the whole point of measuring. The
unchanged, every index agrees with the slab contents, and the process keeps "process keeps serving" half of the original wording is **void**: nothing
serving subsequent requests. survives a SIGKILL.
- **Given** swap thrash, **when** a client issues reads, **then** the observed - **Given** a cap exceeded with `WO_DATA` set, **when** the process is killed at
degradation is quantified and the fact that **no error is surfaced** is exhaustion, **then** replay shows the acked writes present. ✅ ~40 000 rows,
recorded explicitly as a finding. contiguous, no holes, no corruption report. Gated as the `ceiling` leg.
- **Given** a cgroup limit and a workload that exceeds it, **when** the OOM - **Given** the swap-on leg, **when** the same point is reached, **then** the
killer fires, **then** replaying the WAL shows every acked write present — degradation is quantified **and the absence of any error signal recorded**.
ack-after-fsync holding in the one shutdown path that skips all cleanup. ✅ degradation is **nil** for this workload (148 s vs 150 s uncapped) and the
silence is total. Both halves are findings; the first inverted the prediction.
- **Given** the extended baseline, **when** a growth metric is doctored, **then** - **Given** the extended baseline, **when** a growth metric is doctored, **then**
`just db-bench` fails on exactly that metric. the gate fails on exactly that metric. ✅ text footprint +20% →
`FAIL gate.growth.text.noswap.bytes_per_row -- 388 vs baseline 324`, 1 of 104.
- **Given** a host without the cap mechanism, **when** the harness runs, **then**
the legs are skipped with a named reason and the rest still passes. ✅
`cap_wrapper` returns None unless the `memory` controller is delegated; there
is no uncapped fallback.
Outstanding:
- **The resident-footprint fraction for iteration 2's budget default. NOT
delivered, and the premise is false.** It was to be derived from the
swap-onset point — but there is no onset: swap-off jumps straight from
working to SIGKILL, and swap-on shows no degradation to detect an onset in.
**Iteration 2 must pick its budget on other grounds** (host RAM fraction, or
an explicit developer-declared figure) rather than waiting on a number this
iteration cannot produce. This is the most important thing this slice learned
and it removes a dependency rather than satisfying it.
- **The random-read-over-cap collapse.** Not measured. This is where the "latency
collapse" prediction may still be true, and it is the workload that matters
for [2](02-table-storage-modes.md)'s `resident: keys`, whose whole premise is
reading rows back from a log larger than RAM. Needs a read-heavy leg over a
table exceeding the cap. **The single most valuable follow-up.**
- **Given** rising fractions of the cap, **when** latency is sampled, **then**
the p99 departure point is recorded. Partially: the sampler and metric exist
and are gated, but the footprint legs never approach their 512 MiB cap, so
`p99_departure_decile` is legitimately 0 and proves nothing. It becomes
meaningful only with the read-heavy leg above.
- **A replay/restart baseline for iteration 3.** Not delivered; `growth` exercises
`WO_DATA` but nothing times replay. Cheap to add, still absent from
`bench/baseline.json`.
## Out Of Scope ## Out Of Scope
- **Any fix.** This iteration measures. Eviction is - **Any fix.** This measures. Declared budgets are [2](02-table-storage-modes.md),
[5](05-bounded-tables-eviction.md), tiering is [6](06-cold-tiering.md), eviction is [5](05-bounded-tables-eviction.md), tiering is 2's `resident: keys`.
declared budgets are [2](02-table-storage-modes.md). Shipping a fix inside the - **Changing the OOM behaviour.** The checked-`malloc` code is untouched. The
measurement slice would remove the ability to tell whether it helped. measurement showed it is largely unreachable for table storage under default
- **Changing the OOM behaviour.** The checked-`malloc`-to-catchable-trap path is overcommit — a finding to hand to [2](02-table-storage-modes.md), not a bug to
good and should not be touched; if the measurement finds a hole in it, that is fix here, and emphatically not a licence to start setting
a bug fix, reported separately. `vm.overcommit_memory`.
- **A memory profiler or allocator instrumentation.** Observability is language - **A memory profiler or allocator instrumentation** — observability is language
iteration 30. RSS from the OS and the existing `time.ticks` are enough for a iteration 30. RSS from `/proc` plus `time.ticks` is enough for a curve.
curve. - **Comparing against SQLite at the ceiling.** `bench/compare/go-sqlite` exists,
- **Multi-machine or sharded-across-hosts scaling.** One binary owns its data; but SQLite's paged architecture is the design this project rejected, so the
cross-process is [9](09-cross-program-tables.md). numbers would inform no decision here.
- **Comparing against SQLite at the ceiling.** `bench/compare/go-sqlite` exists - **Multi-host scaling** — one binary owns its data.
and the comparison would be interesting, but SQLite's whole architecture is
the paged design this project rejected — the numbers would not inform any
decision here.
## Info ## Info — the forks, settled
Forks the spec must settle: 1. **The limit mechanism is rootless cgroup v2** via
`systemd-run --user --scope -p MemoryMax -p MemorySwapMax`. `ulimit -v` was
rejected: it bounds address space, not resident set, and ASan's virtual
reservations trip it long before real pressure. No sudo needed; it also
isolates the run from the rest of the box, which mattered — the dev box sat
at 22.9 of 31.7 GiB throughout.
2. **Both reference shapes, reported separately.** 3.3× apart; one number would
be a fiction.
3. **Swap is a dimension, not a footnote** — settled by getting it wrong first.
An early run looked like the cap was unenforced because the process held
400 MiB inside a 64 MiB limit; it was swapping, which is the phenomenon under
study.
4. **Footprint is read as the median of per-decile marginals**, not a two-point
slope, so a slab doubling does not smear into the per-row figure. Doublings
are counted as their own metric.
5. **Kill-timing metrics carry ±100% tolerance.** `rows_recovered` depends on
where the SIGKILL landed; gating it tightly would be gating the scheduler.
The invariant asserted instead is the *shape* of the survivor.
6. **The ceiling leg asserts `rc`, never records it.** When iteration 2's byte
budget lands, death should become a checked refusal — the gate must not fail
on that improvement.
1. **What is the limit mechanism for the harness?** A cgroup v2 `memory.max` is ## History — four corrections worth keeping
the closest thing to how this would actually be deployed; `ulimit -v` is
simpler but bounds address space rather than resident set, which for an engine **"An order of magnitude" was a guess.** The per-shape difference is 3.3×. An
that `malloc`s slabs is a materially different constraint. Leaning cgroup, and iteration whose purpose is replacing unmeasured claims had one in its own
the campaign already runs off the fast path so the setup cost is acceptable. premise.
2. **Which table shape is the reference?** Per-row overhead depends heavily on
whether fields are scalars or heap values — a `Text` column is a separate **The clean-exit premise was wrong.** This file and the residency spec both
`db_text` allocation per row, so a text-heavy table and an Int-only table asserted the ceiling surfaces as a catchable `WO_T_OOM`. It is a SIGKILL.
will produce very different slopes. Probably both, reported separately, Overcommit means the allocator never learns there is a problem.
because "bytes per row" is meaningless without saying which row.
3. **Is swap even in scope for the target deployment?** If the intended answer **The latency-collapse premise was wrong too.** Swap cost ~1% on an
is "run with swap off and let the OOM killer decide", the swap curve is append-mostly workload (148 s vs 150 s). The prediction was not merely
informational rather than load-bearing — but that stance should be stated in imprecise, it had the wrong sign. The narrower claim that survives is that a
the doctrine, not assumed. It also changes which exit iteration 5's *random-read* workload over an oversized table is the one at risk, and that
back-pressure is defending against. remains unmeasured.
**A SIGKILL was once labelled a "checked refusal"** by the harness, because
`subprocess` reports signal death as a negative `returncode` (`-9`) while the
shell spells the same event `137`. The leg existed specifically to tell those
two apart. Fixed, and the distinction is now spelled out at the comparison.

View file

@ -154,7 +154,7 @@ Outstanding:
resident. Roughly doubles the resident index; stated at the declaration so resident. Roughly doubles the resident index; stated at the declaration so
the cost is visible. the cost is visible.
5. **The budget is bytes, not rows** — a text-heavy row and an Int-only row 5. **The budget is bytes, not rows** — a text-heavy row and an Int-only row
differ by an order of magnitude, so a row count cannot bound RAM. differ by 3.3× (measured, databasev2 1), so a row count cannot bound RAM.
## History — two corrections worth keeping ## History — two corrections worth keeping

View file

@ -19,7 +19,7 @@
| Which storage architecture | **One engine, log-structured.** The WAL already holds every row; keep an in-RAM id→offset map and read rows back with `pread`. No second engine. | | Which storage architecture | **One engine, log-structured.** The WAL already holds every row; keep an in-RAM id→offset map and read rows back with `pread`. No second engine. |
| Row cache | **None in user space.** The kernel page cache is the hot copy — the repo's own stated position in `exploration/postgresql/buffer-and-checkpoint.md`: "`pread` against an fd that already has its page cached is a memcpy… the page cache is the one cache we want", and the reason the engine avoids `O_DIRECT`. | | Row cache | **None in user space.** The kernel page cache is the hot copy — the repo's own stated position in `exploration/postgresql/buffer-and-checkpoint.md`: "`pread` against an fd that already has its page cached is a memcpy… the page cache is the one cache we want", and the reason the engine avoids `O_DIRECT`. |
| `@unique` on a non-resident table | **Allowed; its index is unconditionally resident.** Settled here rather than deferred — see Constraints. | | `@unique` on a non-resident table | **Allowed; its index is unconditionally resident.** Settled here rather than deferred — see Constraints. |
| Budget unit | **Bytes** (estimated resident footprint). Rows is the meaningless unit: a text-heavy row and an Int-only row differ by an order of magnitude, so a row count cannot bound RAM. | | Budget unit | **Bytes** (estimated resident footprint). Rows is the meaningless unit: a text-heavy row and an Int-only row differ by 3.3× (measured, databasev2 1), so a row count cannot bound RAM. |
| Rejected architectures | `mmap` and a buffer pool stay out — see Alternatives rejected. `discarded.md`'s paged-engine rejection is amended to *partly revisited*, not reversed. | | Rejected architectures | `mmap` and a buffer pool stay out — see Alternatives rejected. `discarded.md`'s paged-engine rejection is amended to *partly revisited*, not reversed. |
## The problem, read off the engine ## The problem, read off the engine
@ -37,9 +37,15 @@ Facts, each verified in source rather than assumed:
`WO_DATA` is set. `db.c` guards every WAL append with a null check on `WO_DATA` is set. `db.c` guards every WAL append with a null check on
`vm->rt.wal`, so with no `WO_DATA` **every table is silently volatile** — a `vm->rt.wal`, so with no `WO_DATA` **every table is silently volatile** — a
program can declare nothing and lose everything. program can declare nothing and lose everything.
- An allocation failure is clean: every `malloc` in the row encoder is checked - An allocation failure is clean *in principle*: every `malloc` in the row
and `DB_ERR_OOM` maps to `WO_T_OOM`, a catchable trap. The dangerous exit is encoder is checked and `DB_ERR_OOM` maps to `WO_T_OOM`. **Corrected 2026-08-27
the one *before* that — swap thrash, which carries no error signal at all. by measurement (databasev2 1): that path does not fire in practice.** With
`vm.overcommit_memory = 0`, `malloc` succeeds and the process is SIGKILLed
when it touches the pages — measured rc=137 at 360 000 rows under a 64 MiB
cgroup cap. The checked-trap path belongs to the VM arena (`WO_HEAP_MB`,
verified `trap 4 ... out of memory`), not to table storage, which has no
ceiling at all. This makes the byte budget below the ONLY mechanism by which
table storage can acquire one.
The measurements that bound the design, from iteration 22: durable inserts The measurements that bound the design, from iteration 22: durable inserts
≈4.5k/s against RAM ≈297k/s (the 66× fsync gap); reads 1.3M ops/s at p50 1µs ≈4.5k/s against RAM ≈297k/s (the 66× fsync gap); reads 1.3M ops/s at p50 1µs

View file

@ -225,6 +225,14 @@ def tolerance_for(key):
mix*: scheduling-dependent small counts. read/query + all .sN.*: mix*: scheduling-dependent small counts. read/query + all .sN.*:
machine jitter, and at post-index-µs scale a 1µs histogram step on a machine jitter, and at post-index-µs scale a 1µs histogram step on a
7µs p50 is already 14%.""" 7µs p50 is already 14%."""
# databasev2 1: footprint is a STRUCTURAL number -- 96.5 vs 320.6 B/row
# reproduced to <2% across runs -- so it gets a tight tolerance and is the
# one growth metric worth gating. The doubling COUNT and the latency
# samples are allowed to move: doublings depend on where N lands relative
# to a pow2 rehash, and at 1us p50 a single histogram step is already 100%.
if ".bytes_per_row" in key: return 10
if key.startswith("growth."): return 100
if key.startswith("ceiling."): return 100
if ".mixread." in key or ".mixwrite." in key: return 50 if ".mixread." in key or ".mixwrite." in key: return 50
if ".sN." in key: return 50 if ".sN." in key: return 50
if ".read." in key or ".query." in key: return 50 if ".read." in key or ".query." in key: return 50
@ -248,6 +256,183 @@ def write_baseline(metrics):
json.dump(base, open(BASELINE, "w"), indent=1, sort_keys=True) json.dump(base, open(BASELINE, "w"), indent=1, sort_keys=True)
ok(f"baseline written ({len(base) - 1} metrics)") ok(f"baseline written ({len(base) - 1} metrics)")
# ---- databasev2 1: the RAM ceiling -----------------------------------------
GROWTH_N = 20000 if QUICK else 200000
GROWTH_SHAPES = ("int", "text")
def cap_wrapper(mem_mb, swap_mb):
"""systemd-run --user --scope argv prefix that caps memory rootlessly, or
None when the mechanism is unavailable.
cgroup v2 with the `memory` controller delegated to the user slice is the
only mechanism used. `ulimit -v` is deliberately NOT a fallback: it bounds
address space, not resident set, which is the wrong quantity for an engine
that mallocs slabs, and ASan's huge virtual reservations trip it long
before real memory pressure. When the cap is unavailable the legs are
SKIPPED and say so -- never silently run uncapped, because "it survived on
a 32 GiB workstation" measures the workstation."""
if not shutil.which("systemd-run"):
return None
try:
with open("/proc/self/cgroup") as f:
mine = f.readline().strip().split(":")[-1]
ctl = f"/sys/fs/cgroup{os.path.dirname(mine)}/cgroup.controllers"
if "memory" not in open(ctl).read().split():
return None
except OSError:
return None
return ["systemd-run", "--user", "--scope", "--quiet",
"-p", f"MemoryMax={mem_mb}M", "-p", f"MemorySwapMax={swap_mb}M", "--"]
def parse_growth(lines):
"""(rows, rss_kb) samples plus per-decile read p50/p99, from the sample's
own `growthrss` / `growthN` lines. RSS is read by the SAMPLE, not polled
here: the driver polls every 250 ms and would miss the value AT a decile
boundary, and per-row footprint is this iteration's headline number."""
pts, lat = [], {}
for l in lines:
f = l.split()
if f and f[0] == "growthrss" and len(f) == 4:
pts.append((int(f[2]), int(f[3])))
elif f and f[0].startswith("growth") and len(f) == 5 and f[0][6:].isdigit():
lat[int(f[0][6:])] = (int(f[3]), int(f[4]))
return pts, lat
def bytes_per_row(pts):
"""Steady-state marginal footprint = MEDIAN of the per-interval marginals.
Not a two-point slope: the id hash and index buckets are open-addressing
pow2 and DOUBLE periodically, so a two-point slope lands arbitrarily on or
off a doubling and swings 2x (measured: 96 vs 205 B/row for the same shape).
The median rejects those steps; they are reported separately as `doublings`
because a transient RSS step is exactly what a resident-footprint budget
must leave headroom for."""
marg = sorted((k1 - k0) * 1024.0 / (r1 - r0)
for (r0, k0), (r1, k1) in zip(pts, pts[1:]) if r1 > r0)
if not marg:
return None, 0
med = marg[len(marg) // 2]
doublings = sum(1 for m in marg if m > med * 1.5)
return med, doublings
def growth(metrics):
"""Per-shape footprint and the read-latency curve, under a rootless cap,
with swap ON and OFF.
What this leg actually measures is FOOTPRINT. It does not reach the cap:
GROWTH_N rows need far less than the 512 MiB cap, so both swap legs are
identical by construction and p99_departure_decile is legitimately 0.
The ceiling itself is ceiling() below -- keep the two separate, because a
footprint regression and a ceiling-behaviour change are different faults.
Two earlier claims in this docstring were measured FALSE and are recorded
in docs/stories/databasev2/01-ram-ceiling-measurement.md: swap-off is not
a "clean checked-malloc" path (it is SIGKILL, rc=137), and swap-on is not
"latency collapse" (900k rows finished in 148s capped-with-swap vs 150s
uncapped -- an append-mostly workload never re-touches its cold pages)."""
wrap = cap_wrapper(512, 0)
if wrap is None:
ok("growth: SKIPPED -- no rootless cgroup v2 memory cap on this host")
metrics["growth.available"] = 0
return
metrics["growth.available"] = 1
for shape in GROWTH_SHAPES:
for legname, swap_mb in (("noswap", 0), ("swap", 256)):
w = cap_wrapper(512, swap_mb)
env = dict(os.environ)
argv = w + [BIN, "growth", str(GROWTH_N), shape]
pr = subprocess.run(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
text=True, env=env, timeout=900)
lines = pr.stdout.splitlines()
pts, lat = parse_growth(lines)
key = f"growth.{shape}.{legname}"
if not pts:
bad(f"{key}: produced no samples", (lines[-1] if lines else "no output"))
continue
bpr, doublings = bytes_per_row(pts)
metrics[f"{key}.bytes_per_row"] = int(round(bpr))
metrics[f"{key}.doublings"] = doublings
metrics[f"{key}.rows"] = pts[-1][0]
metrics[f"{key}.rss_kb"] = pts[-1][1]
if lat:
last = max(lat)
metrics[f"{key}.read_p50us"] = lat[last][0]
metrics[f"{key}.read_p99us"] = lat[last][1]
# the curve's departure point: first decile whose p99 exceeds
# 4x the first decile's, as a MEASURED sample not an estimate
first = lat[min(lat)][1]
dep = next((d for d in sorted(lat) if lat[d][1] > max(first, 1) * 4), 0)
metrics[f"{key}.p99_departure_decile"] = dep
ok(f"{key}: {int(round(bpr))} B/row steady, {doublings} doubling step(s), "
f"{pts[-1][0]} rows in {pts[-1][1]} KiB")
CEIL_N, CEIL_CAP_MB = 60000, 8
def ceiling(metrics):
"""The ceiling itself, and the durability claim across it.
Sized so the process CANNOT fit: 60k Int rows need ~9.7 MiB resident
(96.5 B/row measured, plus a ~3.9 MiB base) under an 8 MiB cap, swap off.
Two things are under test and the second is the one that matters:
1. HOW it dies. Measured: SIGKILL, rc=137 -- not a refusal. Table
storage has no checked ceiling, and under vm.overcommit_memory=0
malloc succeeds and the process dies TOUCHING the pages, so it never
gets the chance to report failure. (The VM object arena is the
opposite: WO_HEAP_MB is checked and traps.) rc is asserted, not
recorded as a metric -- when databasev2 2's byte budget lands this
should become a checked refusal, and the gate must not fail on that
improvement.
2. WHAT SURVIVES. With WO_DATA set, replay must yield a contiguous
intact prefix: rows 1..M present with the right v, no holes, and not
reported as corruption. M is wherever the kill landed -- the SHAPE of
the survivor is the claim, not its size, so rows_recovered carries a
wide tolerance. This is ack-after-fsync holding in the one shutdown
path that skips every cleanup handler."""
wrap = cap_wrapper(CEIL_CAP_MB, 0)
if wrap is None:
ok("ceiling: SKIPPED -- no rootless cgroup v2 memory cap on this host")
return
data = os.path.join(ROOT, "bench", f"tmp.{os.getpid()}.ceiling")
shutil.rmtree(data, ignore_errors=True); os.makedirs(data, exist_ok=True)
env = dict(os.environ); env["WO_DATA"] = data
pr = subprocess.run(wrap + [BIN, "growth", str(CEIL_N), "int"],
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
text=True, env=env, timeout=900)
if pr.returncode == 0:
bad("ceiling: process SURVIVED the cap",
f"{CEIL_N} rows fit under {CEIL_CAP_MB} MiB -- footprint changed, resize the leg")
shutil.rmtree(data, ignore_errors=True); return
# subprocess returncode is NEGATIVE for signal death (-9 = SIGKILL); 137
# is the SHELL spelling of the same event (128+9). Getting this backwards
# once labelled a SIGKILL as a "checked refusal", which is the exact
# distinction this leg exists to report.
if pr.returncode < 0:
sig = -pr.returncode
how = f"killed by signal {sig}" + (" (SIGKILL -- no checked refusal)" if sig == 9 else "")
else:
how = f"exited {pr.returncode} (checked refusal)"
ok(f"ceiling: died at the cap, {how}")
vr = subprocess.run([BIN, "growth-verify"], stdout=subprocess.PIPE,
stderr=subprocess.STDOUT, text=True, env=env, timeout=900)
m = re.search(r"^growthverify (\d+)$", vr.stdout, re.M)
if vr.returncode == 0 and m and int(m.group(1)) > 0:
metrics["ceiling.rows_recovered"] = int(m.group(1))
ok(f"ceiling: durable prefix intact across the kill -- {m.group(1)} rows, no holes")
else:
bad("ceiling: durable prefix broken across the kill",
(vr.stdout.strip().splitlines() or ["no output"])[-1][:160])
shutil.rmtree(data, ignore_errors=True)
def main(): def main():
# --check <results.json>: gate-only evaluation of a recorded run — the # --check <results.json>: gate-only evaluation of a recorded run — the
# gate-bites smoke doctors a copy and this mode must FAIL on it # gate-bites smoke doctors a copy and this mode must FAIL on it
@ -260,6 +445,8 @@ def main():
build() build()
metrics = campaign() metrics = campaign()
durability(metrics) durability(metrics)
growth(metrics)
ceiling(metrics)
os.makedirs(RESULTS_DIR, exist_ok=True) os.makedirs(RESULTS_DIR, exist_ok=True)
stamp = time.strftime("%Y%m%d-%H%M%S") stamp = time.strftime("%Y%m%d-%H%M%S")
out = os.path.join(RESULTS_DIR, f"run-{stamp}{'-quick' if QUICK else ''}.json") out = os.path.join(RESULTS_DIR, f"run-{stamp}{'-quick' if QUICK else ''}.json")