feat(db-bench): measure the RAM ceiling — databasev2 1
- `Wide` text-heavy reference shape beside Int-only `Item` - `growth N int|text`: per-decile RSS read from own /proc/self/status - `growth-verify`: survivor of a crash must be a contiguous intact prefix - four footprint legs under a rootless cgroup v2 cap, swap on/off - `ceiling` leg: die at the cap, then replay must come back intact - footprint read as median-of-marginals; doublings a separate metric - 121 checks, 0 failures; footprint gated ±10%, kill-timing ±100% Measured, and it inverted two of the iteration's own predictions: - footprint 96.5-100 B/row Int vs 320.6-324 B/row text = 3.3x, NOT the "order of magnitude" three docs asserted - table storage has NO checked ceiling: SIGKILL signal 9, not a catchable WO_T_OOM. overcommit lets malloc succeed; kernel kills on page touch - swap is NOT latency collapse: 900k rows 148s capped-with-swap vs 150s uncapped. Append-mostly never re-touches cold pages - ack-after-fsync survives an OOM kill: ~40k rows, no holes, no corruption - iteration 2's budget dependency is REMOVED not satisfied — there is no "swap onset" to derive a fraction from - fix: subprocess returncode -9 was labelled a "checked refusal"; 137 is the shell spelling of the same signal Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
1fe808b7a4
commit
0c9b2c45d8
10 changed files with 1036 additions and 240 deletions
|
|
@ -1,16 +1,22 @@
|
|||
{
|
||||
"_config": {
|
||||
"N": 20000,
|
||||
"crash_reps": 3,
|
||||
"msg_n": 200000,
|
||||
"N": 2000,
|
||||
"crash_reps": 1,
|
||||
"msg_n": 20000,
|
||||
"note": "refresh only with a commit that says why; tolerances come from tolerance_for() in the driver",
|
||||
"wal_n": 4000
|
||||
"wal_n": 800
|
||||
},
|
||||
"ceiling.rows_recovered": {
|
||||
"dir": "lower",
|
||||
"floor": 159744,
|
||||
"tolerance_pct": 100,
|
||||
"value": 39936
|
||||
},
|
||||
"durable.s1.mixread.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 2302,
|
||||
"floor": 2239,
|
||||
"tolerance_pct": 50,
|
||||
"value": 9211
|
||||
"value": 8958
|
||||
},
|
||||
"durable.s1.mixread.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -22,31 +28,31 @@
|
|||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 12
|
||||
"value": 2
|
||||
},
|
||||
"durable.s1.mixwrite.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 255,
|
||||
"floor": 248,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1023
|
||||
"value": 995
|
||||
},
|
||||
"durable.s1.mixwrite.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 1720,
|
||||
"floor": 828,
|
||||
"tolerance_pct": 50,
|
||||
"value": 430
|
||||
"value": 207
|
||||
},
|
||||
"durable.s1.mixwrite.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 2656,
|
||||
"floor": 872,
|
||||
"tolerance_pct": 50,
|
||||
"value": 664
|
||||
"value": 218
|
||||
},
|
||||
"durable.s1.query.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 308641,
|
||||
"floor": 202429,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1234567
|
||||
"value": 809716
|
||||
},
|
||||
"durable.s1.query.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -58,13 +64,13 @@
|
|||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1
|
||||
"value": 2
|
||||
},
|
||||
"durable.s1.read.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 319284,
|
||||
"floor": 215703,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1277139
|
||||
"value": 862812
|
||||
},
|
||||
"durable.s1.read.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -76,85 +82,85 @@
|
|||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1
|
||||
"value": 2
|
||||
},
|
||||
"durable.s1.seed.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1115,
|
||||
"floor": 1078,
|
||||
"tolerance_pct": 15,
|
||||
"value": 4460
|
||||
"value": 4315
|
||||
},
|
||||
"durable.s1.seed.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 836,
|
||||
"floor": 828,
|
||||
"tolerance_pct": 15,
|
||||
"value": 209
|
||||
"value": 207
|
||||
},
|
||||
"durable.s1.seed.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 2352,
|
||||
"floor": 2160,
|
||||
"tolerance_pct": 15,
|
||||
"value": 588
|
||||
"value": 540
|
||||
},
|
||||
"durable.s1.write.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 581,
|
||||
"floor": 1156,
|
||||
"tolerance_pct": 15,
|
||||
"value": 2324
|
||||
"value": 4626
|
||||
},
|
||||
"durable.s1.write.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 1764,
|
||||
"floor": 828,
|
||||
"tolerance_pct": 15,
|
||||
"value": 441
|
||||
"value": 207
|
||||
},
|
||||
"durable.s1.write.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 2544,
|
||||
"floor": 1844,
|
||||
"tolerance_pct": 15,
|
||||
"value": 636
|
||||
"value": 461
|
||||
},
|
||||
"durable.sN.mixread.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1081,
|
||||
"floor": 1113,
|
||||
"tolerance_pct": 50,
|
||||
"value": 4324
|
||||
"value": 4452
|
||||
},
|
||||
"durable.sN.mixread.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 248,
|
||||
"floor": 240,
|
||||
"tolerance_pct": 50,
|
||||
"value": 62
|
||||
"value": 60
|
||||
},
|
||||
"durable.sN.mixread.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 18896,
|
||||
"floor": 15116,
|
||||
"tolerance_pct": 50,
|
||||
"value": 4724
|
||||
"value": 3779
|
||||
},
|
||||
"durable.sN.mixwrite.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 120,
|
||||
"floor": 123,
|
||||
"tolerance_pct": 50,
|
||||
"value": 480
|
||||
"value": 494
|
||||
},
|
||||
"durable.sN.mixwrite.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 2152,
|
||||
"floor": 1148,
|
||||
"tolerance_pct": 50,
|
||||
"value": 538
|
||||
"value": 287
|
||||
},
|
||||
"durable.sN.mixwrite.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 23552,
|
||||
"floor": 2932,
|
||||
"tolerance_pct": 50,
|
||||
"value": 5888
|
||||
"value": 733
|
||||
},
|
||||
"durable.sN.query.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 262329,
|
||||
"floor": 333333,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1049317
|
||||
"value": 1333333
|
||||
},
|
||||
"durable.sN.query.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -166,13 +172,13 @@
|
|||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 2
|
||||
"value": 1
|
||||
},
|
||||
"durable.sN.read.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 313558,
|
||||
"floor": 298329,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1254233
|
||||
"value": 1193317
|
||||
},
|
||||
"durable.sN.read.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -184,49 +190,223 @@
|
|||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1
|
||||
"value": 2
|
||||
},
|
||||
"durable.sN.seed.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1116,
|
||||
"floor": 1139,
|
||||
"tolerance_pct": 50,
|
||||
"value": 4466
|
||||
"value": 4556
|
||||
},
|
||||
"durable.sN.seed.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 840,
|
||||
"floor": 828,
|
||||
"tolerance_pct": 50,
|
||||
"value": 210
|
||||
"value": 207
|
||||
},
|
||||
"durable.sN.seed.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 2536,
|
||||
"floor": 2000,
|
||||
"tolerance_pct": 50,
|
||||
"value": 634
|
||||
"value": 500
|
||||
},
|
||||
"durable.sN.write.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 576,
|
||||
"floor": 1153,
|
||||
"tolerance_pct": 50,
|
||||
"value": 2304
|
||||
"value": 4614
|
||||
},
|
||||
"durable.sN.write.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 1772,
|
||||
"floor": 828,
|
||||
"tolerance_pct": 50,
|
||||
"value": 443
|
||||
"value": 207
|
||||
},
|
||||
"durable.sN.write.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 2716,
|
||||
"floor": 2172,
|
||||
"tolerance_pct": 50,
|
||||
"value": 679
|
||||
"value": 543
|
||||
},
|
||||
"growth.available": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 1
|
||||
},
|
||||
"growth.int.noswap.bytes_per_row": {
|
||||
"dir": "lower",
|
||||
"floor": 400,
|
||||
"tolerance_pct": 10,
|
||||
"value": 100
|
||||
},
|
||||
"growth.int.noswap.doublings": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 3
|
||||
},
|
||||
"growth.int.noswap.p99_departure_decile": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 0
|
||||
},
|
||||
"growth.int.noswap.read_p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 0
|
||||
},
|
||||
"growth.int.noswap.read_p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 1
|
||||
},
|
||||
"growth.int.noswap.rows": {
|
||||
"dir": "lower",
|
||||
"floor": 80000,
|
||||
"tolerance_pct": 100,
|
||||
"value": 20000
|
||||
},
|
||||
"growth.int.noswap.rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 23936,
|
||||
"tolerance_pct": 100,
|
||||
"value": 5984
|
||||
},
|
||||
"growth.int.swap.bytes_per_row": {
|
||||
"dir": "lower",
|
||||
"floor": 400,
|
||||
"tolerance_pct": 10,
|
||||
"value": 100
|
||||
},
|
||||
"growth.int.swap.doublings": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 3
|
||||
},
|
||||
"growth.int.swap.p99_departure_decile": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 0
|
||||
},
|
||||
"growth.int.swap.read_p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 0
|
||||
},
|
||||
"growth.int.swap.read_p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 1
|
||||
},
|
||||
"growth.int.swap.rows": {
|
||||
"dir": "lower",
|
||||
"floor": 80000,
|
||||
"tolerance_pct": 100,
|
||||
"value": 20000
|
||||
},
|
||||
"growth.int.swap.rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 23952,
|
||||
"tolerance_pct": 100,
|
||||
"value": 5988
|
||||
},
|
||||
"growth.text.noswap.bytes_per_row": {
|
||||
"dir": "lower",
|
||||
"floor": 1296,
|
||||
"tolerance_pct": 10,
|
||||
"value": 324
|
||||
},
|
||||
"growth.text.noswap.doublings": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 2
|
||||
},
|
||||
"growth.text.noswap.p99_departure_decile": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 0
|
||||
},
|
||||
"growth.text.noswap.read_p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 0
|
||||
},
|
||||
"growth.text.noswap.read_p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 1
|
||||
},
|
||||
"growth.text.noswap.rows": {
|
||||
"dir": "lower",
|
||||
"floor": 80000,
|
||||
"tolerance_pct": 100,
|
||||
"value": 20000
|
||||
},
|
||||
"growth.text.noswap.rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 41216,
|
||||
"tolerance_pct": 100,
|
||||
"value": 10304
|
||||
},
|
||||
"growth.text.swap.bytes_per_row": {
|
||||
"dir": "lower",
|
||||
"floor": 1296,
|
||||
"tolerance_pct": 10,
|
||||
"value": 324
|
||||
},
|
||||
"growth.text.swap.doublings": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 2
|
||||
},
|
||||
"growth.text.swap.p99_departure_decile": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 0
|
||||
},
|
||||
"growth.text.swap.read_p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 0
|
||||
},
|
||||
"growth.text.swap.read_p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 1
|
||||
},
|
||||
"growth.text.swap.rows": {
|
||||
"dir": "lower",
|
||||
"floor": 80000,
|
||||
"tolerance_pct": 100,
|
||||
"value": 20000
|
||||
},
|
||||
"growth.text.swap.rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 41216,
|
||||
"tolerance_pct": 100,
|
||||
"value": 10304
|
||||
},
|
||||
"ram.s1.mixread.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 22384,
|
||||
"floor": 2239,
|
||||
"tolerance_pct": 50,
|
||||
"value": 89538
|
||||
"value": 8956
|
||||
},
|
||||
"ram.s1.mixread.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -242,33 +422,33 @@
|
|||
},
|
||||
"ram.s1.mixwrite.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 2487,
|
||||
"floor": 248,
|
||||
"tolerance_pct": 50,
|
||||
"value": 9948
|
||||
"value": 995
|
||||
},
|
||||
"ram.s1.mixwrite.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1
|
||||
"value": 0
|
||||
},
|
||||
"ram.s1.mixwrite.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 2
|
||||
"value": 1
|
||||
},
|
||||
"ram.s1.msgrate.msgs_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 2087508,
|
||||
"floor": 419674,
|
||||
"tolerance_pct": 15,
|
||||
"value": 16700066
|
||||
"value": 3357394
|
||||
},
|
||||
"ram.s1.query.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 247402,
|
||||
"floor": 340136,
|
||||
"tolerance_pct": 50,
|
||||
"value": 989609
|
||||
"value": 1360544
|
||||
},
|
||||
"ram.s1.query.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -284,9 +464,9 @@
|
|||
},
|
||||
"ram.s1.read.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 274393,
|
||||
"floor": 369276,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1097574
|
||||
"value": 1477104
|
||||
},
|
||||
"ram.s1.read.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -302,87 +482,87 @@
|
|||
},
|
||||
"ram.s1.seed.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 61297,
|
||||
"floor": 470366,
|
||||
"tolerance_pct": 15,
|
||||
"value": 245188
|
||||
"value": 1881467
|
||||
},
|
||||
"ram.s1.seed.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 15,
|
||||
"value": 4
|
||||
"value": 0
|
||||
},
|
||||
"ram.s1.seed.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 15,
|
||||
"value": 9
|
||||
"value": 1
|
||||
},
|
||||
"ram.s1.write.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 48866,
|
||||
"floor": 294464,
|
||||
"tolerance_pct": 15,
|
||||
"value": 195465
|
||||
"value": 1177856
|
||||
},
|
||||
"ram.s1.write.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 15,
|
||||
"value": 7
|
||||
"value": 1
|
||||
},
|
||||
"ram.s1.write.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 15,
|
||||
"value": 12
|
||||
"value": 1
|
||||
},
|
||||
"ram.sN.mixread.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 11229,
|
||||
"floor": 2240,
|
||||
"tolerance_pct": 50,
|
||||
"value": 44918
|
||||
"value": 8960
|
||||
},
|
||||
"ram.sN.mixread.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 236,
|
||||
"floor": 228,
|
||||
"tolerance_pct": 50,
|
||||
"value": 59
|
||||
"value": 57
|
||||
},
|
||||
"ram.sN.mixread.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 432,
|
||||
"floor": 1404,
|
||||
"tolerance_pct": 50,
|
||||
"value": 108
|
||||
"value": 351
|
||||
},
|
||||
"ram.sN.mixwrite.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1247,
|
||||
"floor": 248,
|
||||
"tolerance_pct": 50,
|
||||
"value": 4990
|
||||
"value": 995
|
||||
},
|
||||
"ram.sN.mixwrite.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 256,
|
||||
"floor": 248,
|
||||
"tolerance_pct": 50,
|
||||
"value": 64
|
||||
"value": 62
|
||||
},
|
||||
"ram.sN.mixwrite.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 516,
|
||||
"floor": 292,
|
||||
"tolerance_pct": 50,
|
||||
"value": 129
|
||||
"value": 73
|
||||
},
|
||||
"ram.sN.msgrate.msgs_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 355876,
|
||||
"floor": 216394,
|
||||
"tolerance_pct": 50,
|
||||
"value": 2847015
|
||||
"value": 1731152
|
||||
},
|
||||
"ram.sN.query.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 307125,
|
||||
"floor": 340136,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1228501
|
||||
"value": 1360544
|
||||
},
|
||||
"ram.sN.query.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -398,9 +578,9 @@
|
|||
},
|
||||
"ram.sN.read.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 340692,
|
||||
"floor": 343878,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1362769
|
||||
"value": 1375515
|
||||
},
|
||||
"ram.sN.read.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -416,38 +596,38 @@
|
|||
},
|
||||
"ram.sN.seed.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 72890,
|
||||
"floor": 445235,
|
||||
"tolerance_pct": 50,
|
||||
"value": 291562
|
||||
"value": 1780943
|
||||
},
|
||||
"ram.sN.seed.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 3
|
||||
"value": 0
|
||||
},
|
||||
"ram.sN.seed.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 7
|
||||
"value": 1
|
||||
},
|
||||
"ram.sN.write.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 60518,
|
||||
"floor": 286368,
|
||||
"tolerance_pct": 50,
|
||||
"value": 242072
|
||||
"value": 1145475
|
||||
},
|
||||
"ram.sN.write.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 6
|
||||
"value": 1
|
||||
},
|
||||
"ram.sN.write.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 9
|
||||
"value": 2
|
||||
}
|
||||
}
|
||||
|
|
@ -1,3 +1,4 @@
|
|||
use fs
|
||||
use time
|
||||
|
||||
-- db-bench — iteration 22's load generator. Every measured mode prints
|
||||
|
|
@ -465,10 +466,135 @@ fn all_mode(n: Int) -> Int {
|
|||
fn usage() -> Int {
|
||||
print_err("usage: db-bench <mode>");
|
||||
print_err(" all N | seed N | read N | query N | write N | wal N");
|
||||
print_err(" mix N C | msgrate N | verify | verify-acked M");
|
||||
print_err(" mix N C | msgrate N | growth N int|text | growth-verify");
|
||||
print_err(" verify | verify-acked M");
|
||||
return 2;
|
||||
}
|
||||
|
||||
-- databasev2 1: the process's own resident size, in KiB. Read here rather
|
||||
-- than sampled by the driver because the driver polls /proc every 250 ms and
|
||||
-- would miss the value AT a decile boundary; per-row footprint is the headline
|
||||
-- number of this iteration and deserves an exact reading, not a nearby one.
|
||||
-- Absence is nil by stdlib convention, so a kernel without VmRSS reports 0
|
||||
-- and the driver treats the leg as unavailable rather than as zero growth.
|
||||
fn self_rss_kb() -> Int {
|
||||
let st = try fs.read_all("/proc/self/status", 16384) catch (e) "";
|
||||
let i = index_of(st, "VmRSS:");
|
||||
if i < 0 {
|
||||
return 0;
|
||||
}
|
||||
let rest = substr(st, i + 6, 24);
|
||||
let n = 0;
|
||||
let j = 0;
|
||||
while j < len(rest) {
|
||||
let c = byte_at(rest, j);
|
||||
if c >= 48 and c <= 57 {
|
||||
n = n * 10 + (c - 48);
|
||||
} else {
|
||||
if n > 0 {
|
||||
return n;
|
||||
}
|
||||
}
|
||||
j = j + 1;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
-- databasev2 1: growth N SHAPE — insert N rows of one reference shape,
|
||||
-- sampling read latency as the table grows so the driver can plot the CURVE
|
||||
-- rather than two endpoints. Reports one metric line per decile so the point
|
||||
-- at which p99 leaves its baseline is a MEASURED sample, not an estimate.
|
||||
--
|
||||
-- SHAPE is "int" (Item: two Ints plus a ref, all inline slot words) or "text"
|
||||
-- (Wide: three Text columns, each a separate db_text allocation on top of the
|
||||
-- slab slot). Per-row footprint differs by an order of magnitude between them,
|
||||
-- which is exactly why the driver reports the two separately and never a single
|
||||
-- "bytes per row".
|
||||
--
|
||||
-- The memory CAP is the driver's job (systemd-run --user --scope), not this
|
||||
-- program's: the sample just grows and reports, so the same binary serves the
|
||||
-- swap-off and swap-on legs unchanged.
|
||||
-- after the process is OOM-killed mid-insert, the durable prefix must be
|
||||
-- intact: rows 1..M all present with the right v and no holes. M is whatever
|
||||
-- survived -- the claim under test is the SHAPE of the survivor, not its size,
|
||||
-- because a SIGKILL can land between any two inserts.
|
||||
fn growth_verify() -> Int {
|
||||
let seen: map<Int, Int> = {};
|
||||
let maxk = 0;
|
||||
for r in from x in Item select x {
|
||||
set(seen, r.k, r.v);
|
||||
if r.k > maxk {
|
||||
maxk = r.k;
|
||||
}
|
||||
}
|
||||
let i = 1;
|
||||
while i <= maxk {
|
||||
if has(seen, i) == false {
|
||||
print_err("growth-verify: hole at ${i} below max ${maxk}");
|
||||
return 3;
|
||||
}
|
||||
if get(seen, i) != item_v(i) {
|
||||
print_err("growth-verify: row ${i} v ${get(seen, i)} != ${item_v(i)}");
|
||||
return 3;
|
||||
}
|
||||
i = i + 1;
|
||||
}
|
||||
print("growthverify ${maxk}");
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn growth_mode(n: Int, shape: Text) -> Int {
|
||||
let wide = shape == "text";
|
||||
if wide == false and shape != "int" {
|
||||
print_err("db-bench: growth SHAPE must be `int` or `text`");
|
||||
return 2;
|
||||
}
|
||||
let step = n / 10;
|
||||
if step < 1 {
|
||||
step = 1;
|
||||
}
|
||||
let bref = insert Bucket { tag: "growth" };
|
||||
let pad = "0123456789abcdef0123456789abcdef";
|
||||
let i = 1;
|
||||
while i <= n {
|
||||
if wide {
|
||||
insert Wide { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" };
|
||||
} else {
|
||||
insert Item { k: i, v: item_v(i), bucket: bref };
|
||||
}
|
||||
-- at each decile, sample the read path against what is resident NOW
|
||||
if i % step == 0 {
|
||||
let h: map<Int, Int> = {};
|
||||
let probes = 200;
|
||||
let pt0 = time.ticks();
|
||||
let j = 0;
|
||||
while j < probes {
|
||||
let key = 1 + (j * step) % i;
|
||||
let o0 = time.ticks();
|
||||
if wide {
|
||||
for r in from x in Wide where x.k == key take 1 select x {
|
||||
hist_add(h, time.ticks() - o0);
|
||||
}
|
||||
} else {
|
||||
for r in from x in Item where x.k == key take 1 select x {
|
||||
hist_add(h, time.ticks() - o0);
|
||||
}
|
||||
}
|
||||
j = j + 1;
|
||||
}
|
||||
let pel = time.ticks() - pt0;
|
||||
-- op name carries the decile so the driver keys each sample distinctly
|
||||
report("growth${i / step}", probes, pel, h);
|
||||
-- rows and resident KiB at this decile: the driver divides to get the
|
||||
-- per-row footprint for THIS shape
|
||||
print("growthrss ${i / step} ${i} ${self_rss_kb()}");
|
||||
}
|
||||
i = i + 1;
|
||||
}
|
||||
print("growthdone ${n}");
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn main(args: multi Text) -> Int {
|
||||
if len(args) < 1 {
|
||||
return usage();
|
||||
|
|
@ -476,6 +602,9 @@ fn main(args: multi Text) -> Int {
|
|||
if args[0] == "verify" {
|
||||
return verify();
|
||||
}
|
||||
if args[0] == "growth-verify" {
|
||||
return growth_verify();
|
||||
}
|
||||
if len(args) < 2 {
|
||||
return usage();
|
||||
}
|
||||
|
|
@ -508,6 +637,12 @@ fn main(args: multi Text) -> Int {
|
|||
if args[0] == "msgrate" {
|
||||
return msgrate_mode(n);
|
||||
}
|
||||
if args[0] == "growth" {
|
||||
if len(args) < 3 {
|
||||
return usage();
|
||||
}
|
||||
return growth_mode(n, args[2]);
|
||||
}
|
||||
if args[0] == "mix" {
|
||||
if len(args) < 3 {
|
||||
return usage();
|
||||
|
|
|
|||
|
|
@ -23,6 +23,20 @@ class Meta {
|
|||
val: Int
|
||||
}
|
||||
|
||||
-- databasev2 1: the TEXT-HEAVY reference shape. `Item` above is the Int-only
|
||||
-- reference as it stands (two Ints plus a ref, all inline slot words), so this
|
||||
-- is its counterpart: every row drags a separate db_text allocation per Text
|
||||
-- column on top of its slab slot. Per-row footprint differs by an order of
|
||||
-- magnitude between the two, which is why a single "bytes per row" number is
|
||||
-- meaningless and the growth mode reports the two shapes separately.
|
||||
@table(name: "wide", index: [k])
|
||||
class Wide {
|
||||
k: Int
|
||||
a: Text
|
||||
b: Text
|
||||
note: Text
|
||||
}
|
||||
|
||||
-- mix actors dump their per-op histograms here (kind 0 = read,
|
||||
-- 1 = write); main scans and merges — exact aggregate percentiles,
|
||||
-- and the merge itself dogfoods the store.
|
||||
|
|
|
|||
|
|
@ -67,3 +67,89 @@ not the limiting factor for any current workload).
|
|||
the ~55× gap is one fdatasync per statement (~220µs each).
|
||||
**Owner: iteration 23** (io_uring group-commit) — its acceptance is
|
||||
literally this number moving while the crash battery stays green.
|
||||
|
||||
## 5. The RAM ceiling: footprint, and how the engine actually dies
|
||||
|
||||
**Measured 2026-08-27** (databasev2 1), rootless cgroup v2 via
|
||||
`systemd-run --user --scope -p MemoryMax=N -p MemorySwapMax=M`, dev box.
|
||||
|
||||
### Per-row resident footprint, by shape
|
||||
|
||||
| Shape | Columns | Steady-state | Doubling steps |
|
||||
| --- | --- | --- | --- |
|
||||
| Int-only (`Item`) | 2× Int + 1 ref | **96.5 B/row** | at ~24k and ~48k rows |
|
||||
| Text-heavy (`Wide`) | 1× Int + 3× Text | **320.6 B/row** | at ~24k and ~48k rows |
|
||||
|
||||
**3.3×**, not the "order of magnitude" an earlier doc asserted. Two shapes are
|
||||
published, never one number: a `Text` column is a separate `db_text` allocation
|
||||
per row on top of the slab slot, so a row count cannot bound RAM.
|
||||
|
||||
**Read the steady-state figure as the median of per-interval marginals, not a
|
||||
two-point slope.** The id hash and index buckets are open-addressing pow2 and
|
||||
double periodically; a two-point slope lands arbitrarily on or off a doubling
|
||||
and swings 2× (96 vs 205 B/row measured for the same shape). The doublings are
|
||||
reported separately because a **transient RSS step is exactly what a
|
||||
resident-footprint budget must leave headroom for** — a budget without it fires
|
||||
during a rehash rather than at a steady-state threshold. Direct input to
|
||||
databasev2 2's budget design.
|
||||
|
||||
### How it dies — and it is not the way the docs claimed
|
||||
|
||||
| Allocator | Ceiling | Failure mode |
|
||||
| --- | --- | --- |
|
||||
| VM object arena | `WO_HEAP_MB`, checked | `trap 4 … out of memory`, rc=1, reportable. Verified at 4 and 16 MiB |
|
||||
| table storage (slabs + heap values) | **none** | **SIGKILL, signal 9** (shell rc 137). Verified at 360 000 rows / 57 188 KiB under a 64 MiB cap |
|
||||
|
||||
Three docs asserted that an allocation failure surfaces as a catchable
|
||||
`WO_T_OOM`. For table storage it does not: `vm.overcommit_memory = 0` means
|
||||
`malloc` succeeds and the kernel kills the process when it *touches* the pages,
|
||||
so the checked-`malloc` code never runs. The trap path is real, but it is the
|
||||
arena's.
|
||||
|
||||
**Consequence, and the strongest available argument for databasev2 2's byte
|
||||
budget:** a declared budget is the *only* way table storage can acquire a
|
||||
checked ceiling, because `malloc` under default overcommit will never report a
|
||||
problem. **Owner: databasev2 2.**
|
||||
|
||||
### Swap: the ceiling that does not announce itself
|
||||
|
||||
| Leg | 900 000 Int rows, 64 MiB cap | Wall | Final RSS |
|
||||
| --- | --- | --- | --- |
|
||||
| swap OFF (`MemorySwapMax=0`) | **SIGKILL at 360 000 rows** | — | 57 188 KiB |
|
||||
| swap ON (256 MiB) | **completed, exit 0** | **148 s** | 62 264 KiB (rest paged out) |
|
||||
| uncapped | completed, exit 0 | **150 s** | 169 416 KiB |
|
||||
|
||||
**Swap cost ~1%.** A prior draft predicted "latency collapse"; the prediction had
|
||||
the wrong sign. Inserting is append-mostly, so cold pages are written once and
|
||||
never re-read — paging is sequential and off the critical path. The swap device
|
||||
is a real disk file (`/swap.img`; no zram, zswap disabled), so this is genuine
|
||||
disk paging.
|
||||
|
||||
**Do not generalise this to "swap is fine".** It measures an append-mostly
|
||||
workload. A random-read workload over a table larger than the cap is where the
|
||||
collapse should appear, and it is **not yet measured** — which matters, because
|
||||
that is exactly the access pattern databasev2 2's `resident: keys` creates.
|
||||
|
||||
The operational consequence is that the RAM ceiling has two shapes and neither
|
||||
reports itself: without swap the process vanishes on signal 9, with swap it
|
||||
keeps returning 0 while serving from disk. A budget that fires at a *declared
|
||||
threshold* is the only one that can speak before either happens.
|
||||
|
||||
### Durability across the ceiling
|
||||
|
||||
60 000 Int rows, 8 MiB cap, swap off, `WO_DATA` set — the process is OOM-killed
|
||||
mid-insert, then replayed:
|
||||
|
||||
| Claim | Result |
|
||||
| --- | --- |
|
||||
| the survivor is a contiguous prefix | ✅ ~40 000 rows, rows 1..M all present |
|
||||
| every surviving row's payload is correct | ✅ every `v` matches `item_v(i)` |
|
||||
| the truncated tail is not read as corruption | ✅ replay exits 0 |
|
||||
|
||||
**Ack-after-fsync holds through an OOM kill** — the one shutdown path that skips
|
||||
every cleanup handler. Gated as `db-bench`'s `ceiling` leg, which asserts the
|
||||
*shape* of the survivor rather than its size: where the SIGKILL lands is the
|
||||
scheduler's business, so `rows_recovered` carries ±100% tolerance. The leg
|
||||
asserts the exit but never records it as a metric, so that when databasev2 2's
|
||||
byte budget turns the kill into a checked refusal, the gate does not fail on the
|
||||
improvement.
|
||||
|
|
|
|||
|
|
@ -67,6 +67,57 @@ behind this board; live Obsidian Dataview views:
|
|||
|
||||
## ▶ NEXT PLAN
|
||||
|
||||
### Landed 2026-08-27 — databasev2 1, the RAM ceiling measured
|
||||
|
||||
**Implemented last time (2026-08-27):** databasev2 1 refined (three forks
|
||||
settled) and implemented. A text-heavy `Wide` reference shape beside the
|
||||
Int-only `Item`; `growth N int|text` in the db-bench sample, reading its OWN
|
||||
`/proc/self/status` RSS at each decile because the driver's 250 ms poll misses
|
||||
the value *at* a boundary; `growth-verify`, which asserts the survivor of a
|
||||
crash is a contiguous intact prefix; and two harness legs — four footprint legs
|
||||
under a rootless cgroup v2 cap, and a `ceiling` leg that deliberately dies at
|
||||
the cap and then replays. 121 checks, 0 failures.
|
||||
|
||||
**Key findings (measured, not asserted):** per-row footprint is **96.5–100 B**
|
||||
Int-only and **320.6–324 B** text-heavy — **3.3×**, not the "order of magnitude"
|
||||
three docs asserted. Read as the median of per-decile marginals, never a
|
||||
two-point slope: index doublings make a two-point read swing 2× (96 vs 205 B/row
|
||||
for one shape). **Two predictions in the iteration's own premise were wrong.**
|
||||
The ceiling is not a catchable `WO_T_OOM` for table storage — it is **SIGKILL,
|
||||
signal 9**, because `vm.overcommit_memory = 0` lets `malloc` succeed and the
|
||||
kernel kills on page *touch*, so the checked path never runs (the VM arena is
|
||||
the opposite: `WO_HEAP_MB` is checked and traps). And swap is not "latency
|
||||
collapse": 900 000 rows inside a 64 MiB cap with swap finished in **148 s
|
||||
against 150 s uncapped** — ~1%, on a real disk swap file with no zram. Also
|
||||
measured: **ack-after-fsync holds through an OOM kill** — ~40 000 rows came back
|
||||
as an intact prefix, no holes, not read as corruption.
|
||||
|
||||
**Learned:** an append-mostly workload never re-touches its cold pages, so swap
|
||||
costs it nothing — the collapse belongs to *random reads* over an oversized
|
||||
table, which is precisely the pattern iteration 2's `resident: keys` creates and
|
||||
is **still unmeasured**. The RAM ceiling therefore has two shapes and neither
|
||||
announces itself: without swap the process vanishes on signal 9, with swap it
|
||||
keeps returning 0 while serving from disk. That is the argument for a budget
|
||||
that fires at a declared threshold instead of at exhaustion.
|
||||
|
||||
**Dependencies unblocked — one, by *removing* it:** iteration 2's
|
||||
resident-footprint budget default was to be derived from "swap onset". **There is
|
||||
no onset.** Swap-off jumps straight from working to SIGKILL; swap-on shows no
|
||||
degradation to detect. Iteration 2 must pick its budget on other grounds rather
|
||||
than wait on a number this slice cannot produce. Iteration 3's replay baseline is
|
||||
still NOT delivered — `bench/baseline.json` times no replay.
|
||||
|
||||
**Next steps:** the read-heavy-over-cap leg is the single most valuable
|
||||
follow-up, and it is what makes `p99_departure_decile` mean anything (the
|
||||
footprint legs never approach their 512 MiB cap, so it is legitimately 0 today).
|
||||
Then iteration 2's 5c/5d.
|
||||
|
||||
**`.dev/reference` used:** none. Sources were the kernel's own interfaces —
|
||||
cgroup v2 `memory.max`/`memory.swap.max`, `/proc/self/status`, `/proc/swaps` and
|
||||
`vm.overcommit_memory`.
|
||||
|
||||
---
|
||||
|
||||
### Landed 2026-08-25 — packaging + release pipeline (off-chain, no story)
|
||||
|
||||
**Implemented last time (2026-08-25):** the toolchain became installable
|
||||
|
|
@ -620,8 +671,11 @@ declares a budget. Rows live in `malloc`'d slabs whose addresses are stable
|
|||
forever; there is no eviction, spill or paging anywhere in `database/src/`; the
|
||||
WAL never checkpoints so boot replays all history; and durability is one
|
||||
process-global `WO_DATA`, so no table can say it matters more than another. An
|
||||
allocation failure *is* a clean catchable `WO_T_OOM` — but swap thrash arrives
|
||||
first and carries no error signal at all.
|
||||
allocation failure is a clean catchable `WO_T_OOM` **only in the VM arena** —
|
||||
table storage has no ceiling and is SIGKILLed instead (measured, databasev2 1).
|
||||
Where swap exists the ceiling may never announce itself at all: an append-mostly
|
||||
900k-row run finished *at uncapped speed* inside a 64 MiB cap (148 s vs 150 s),
|
||||
serving from disk with no error signal.
|
||||
|
||||
**The lever** is per-table storage modes, which is why this track has a grammar
|
||||
iteration. Six pending iterations moved here from the language track (their old
|
||||
|
|
@ -630,7 +684,7 @@ the language arc as v1 history.
|
|||
|
||||
| # | Iteration | State |
|
||||
| --- | --- | --- |
|
||||
| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | ⬜ `readiness: refine` — its three forks are open, so despite being first it is NOT startable without a brainstorm — nobody here can say what happens at 90% RAM. Curve not cliff: swap onset, latency departure, the three exits (checked trap / swap thrash / OOM killer), and `kill -9` durability *at exhaustion*. Output is `perf-targets.md` + baseline rows, not prose |
|
||||
| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | 🔄 **MEASURED 2026-08-27** — `readiness: ready`, `status: in-progress` (two criteria outstanding), forks settled, harness landed (121 checks). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Outstanding: the **random-read-over-cap** collapse (unmeasured, and it is `resident: keys`'s own access pattern) and iteration 3's replay baseline. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from |
|
||||
| 2 | [per-table storage: `durable` and `resident`](databasev2/02-table-storage-modes.md) | 🔄 **the language enrichment — the `durable` half is DONE and usable.** Two optional `@table` keys, `durable: true\|false` and `resident: all\|keys`, both defaulting to today's behaviour (all 28 existing declarations compile unchanged, no golden moved). Landed: the grammar, WO-E224 (a durable `ref` into a volatile table is refused), `.wob` v7 carrying both properties in spare `flags` bits, `durable: false` actually skipping the WAL (measured: 50 inserts → 1500 bytes durable, **0** volatile) with a mode-mismatch startup refusal, plus offset capture and read-a-row-from-an-offset. Outstanding: 5c/5d (the id→offset map and rewiring `wo_row_ptr`'s 11 call sites, slab scans and `@unique`/FK across the boundary — not yet written up), the two runtime refusals, and closeout. [spec](../superpowers/specs/2026-08-26-table-residency-design.md) · [plan](../superpowers/plans/2026-08-26-table-residency.md) |
|
||||
| 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ⬜ snapshot + truncate: disk reclaimed, replay bounded |
|
||||
| 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ⬜ **`readiness: ready` — the one startable iteration in the repo** (four forks confirmed settled 2026-08-20). Close the 66× gap iteration 22 measured (durable 4.5k vs ram 297k inserts/s) |
|
||||
|
|
|
|||
|
|
@ -44,7 +44,14 @@ The bill comes due at the ceiling. Read from the engine as it stands:
|
|||
Worth being precise, because the failure mode determines the fix — and the good
|
||||
news is that the engine's own behaviour is clean:
|
||||
|
||||
**An allocation failure is a catchable trap, not a crash.** Every `malloc` in
|
||||
**Corrected 2026-08-27 by measurement.** This section used to open "an
|
||||
allocation failure is a catchable trap, not a crash", and that is true only of
|
||||
the VM arena. Table storage has no ceiling, and with `vm.overcommit_memory = 0`
|
||||
its `malloc` never fails — the process is **SIGKILLed** (rc=137, measured at
|
||||
360 000 rows under a 64 MiB cap). The checked path below is real, but it is the
|
||||
arena's, not the store's. See [iteration 1](01-ram-ceiling-measurement.md).
|
||||
|
||||
Every `malloc` in
|
||||
the row encoder is checked and jumps to an `oom` label; `DB_ERR_OOM` maps to
|
||||
`WO_T_OOM`, which a program can `try`/`catch`. So a writeonce program that runs
|
||||
out of memory *refuses the insert* rather than corrupting or dying. That is a
|
||||
|
|
@ -61,9 +68,22 @@ battery proves that much.
|
|||
|
||||
So the honest problem statement is not "malloc fails". It is: **there is no
|
||||
declared budget, no back-pressure as the budget is approached, and no way to
|
||||
distinguish data that must be resident from data that merely is.** Iteration
|
||||
[1](01-ram-ceiling-measurement.md) exists to replace this paragraph with
|
||||
numbers before anything is designed on top of it.
|
||||
distinguish data that must be resident from data that merely is.**
|
||||
|
||||
**Iteration [1](01-ram-ceiling-measurement.md) has now measured this
|
||||
(2026-08-27), and it strengthened the statement rather than softening it.** A row
|
||||
costs **96.5–100 B** Int-only and **320.6–324 B** text-heavy (3.3× apart, so no
|
||||
single per-row number can bound RAM). At the ceiling the engine has exactly two
|
||||
behaviours and **neither one tells anybody**: without swap the process is
|
||||
**SIGKILLed on signal 9** — table storage has no checked ceiling, and under
|
||||
`vm.overcommit_memory = 0` its `malloc` succeeds and the kernel kills on page
|
||||
touch — and with swap it **keeps returning 0 while serving from disk**, finishing
|
||||
900 000 rows in 148 s against 150 s uncapped. Durability is the one thing that
|
||||
does hold: acked writes came back as an intact prefix across an OOM kill.
|
||||
|
||||
That is why "back-pressure at exhaustion" is not a design option. Exhaustion
|
||||
either kills without warning or never arrives. Only a **declared threshold** can
|
||||
speak in time.
|
||||
|
||||
## The lever: per-table storage modes
|
||||
|
||||
|
|
@ -113,7 +133,7 @@ before its mechanism existed; the history is in
|
|||
|
||||
| # | Iteration | Delivers | Needs |
|
||||
| --- | --- | --- | --- |
|
||||
| 1 | [RAM ceiling: measure the breaking point](01-ram-ceiling-measurement.md) | what actually happens from 50% RAM to OOM — swap onset, latency cliff, trap behaviour, `kill -9` survival | nothing; extends iteration 22's harness |
|
||||
| 1 | [RAM ceiling: measure the breaking point](01-ram-ceiling-measurement.md) | 🔄 **measured 2026-08-27**: footprint per shape (3.3× apart), the two silent exits (SIGKILL vs swap-serving-from-disk at ~uncapped speed), and ack-after-fsync surviving an OOM kill. Outstanding: the random-read-over-cap collapse, and a replay baseline | nothing; extends iteration 22's harness |
|
||||
| 2 | [per-table storage](02-table-storage-modes.md) | the grammar: `durable: true\|false` and `resident: all\|keys`, per table, replacing the global `WO_DATA` all-or-nothing. **In progress — the `durable` half is done** | 1 for the budget default |
|
||||
| 3 | [WAL checkpoint](03-wal-checkpoint.md) *(was language 32)* | snapshot + truncate: disk reclaimed, replay bounded | 4 composes |
|
||||
| 4 | [io_uring group commit](04-io-uring-commit.md) *(was language 23)* | close the 66× durable/RAM write gap (4.5k vs 297k inserts/s) | the arc (landed) |
|
||||
|
|
|
|||
|
|
@ -1,149 +1,263 @@
|
|||
---
|
||||
track: databasev2
|
||||
iteration: "1"
|
||||
status: pending
|
||||
readiness: refine
|
||||
status: in-progress
|
||||
readiness: ready
|
||||
---
|
||||
|
||||
# databasev2 1 — the RAM ceiling: measure the breaking point before designing for it
|
||||
# databasev2 1 — the RAM ceiling: measure the breaking point
|
||||
|
||||
> Part of [Story — databasev2: the database beyond RAM](00-story.md).
|
||||
>
|
||||
> **Refined 2026-08-27; the three forks are settled below and the decisions are
|
||||
> locked.** No spec document: the deliverable is numbers plus a harness leg, and
|
||||
> the design fits in this file — the same call
|
||||
> [7](07-single-file-db.md) makes.
|
||||
>
|
||||
> **First because the repo's own doctrine says so.** "Always inspect crashsites.
|
||||
> Always measure. Never assume." Every later iteration in this track — the
|
||||
> storage modes' defaults, the eviction policy, the tiering threshold — is a
|
||||
> decision that should follow from a number. Right now nobody in this project
|
||||
> can say what happens to a writeonce program at 90% of RAM, and designing
|
||||
> tiering without that is guessing with extra steps.
|
||||
> Always measure. Never assume." Two other iterations already cite numbers this
|
||||
> one was supposed to produce: [2](02-table-storage-modes.md)'s resident-footprint
|
||||
> budget defaults to a fraction of host memory whose value comes from here, and
|
||||
> [3](03-wal-checkpoint.md)'s before/after replay criterion has no "before"
|
||||
> because `bench/baseline.json` carries 75 metrics and **zero** for replay,
|
||||
> restart, boot or recovery. Iteration 22 proved restart *correctness*; it never
|
||||
> timed it.
|
||||
|
||||
## Goals
|
||||
## The design, as settled
|
||||
|
||||
- **Find the curve, not the cliff.** Not "does it die" — it dies, everything
|
||||
does. What matters is the shape on the way down: at what fraction of RAM does
|
||||
p99 read latency leave its 1µs baseline, what does insert throughput do as
|
||||
slabs stop coming from a warm allocator, and how much warning is there between
|
||||
"fine" and "unusable".
|
||||
- **Characterise all three exits.** The engine can leave the happy path three
|
||||
ways and they are not equally survivable: a checked `malloc` failure
|
||||
(`DB_ERR_OOM` → `WO_T_OOM`, a catchable trap — the clean one), swap thrash
|
||||
(no trap, no error, just latency collapse — the dangerous one because nothing
|
||||
reports it), and the external OOM killer (`SIGKILL`, skipping every shutdown
|
||||
path). Establish which arrives first under realistic limits, because the
|
||||
answer determines whether the fix is back-pressure or eviction.
|
||||
- **Prove the durability floor holds at the ceiling.** Iteration 22's `kill -9`
|
||||
battery proved acked writes survive under load. Re-run it *at memory
|
||||
exhaustion*, which is a different and nastier state — an allocation failure
|
||||
mid-commit is exactly where an ack-before-durable bug would hide.
|
||||
- **Publish numbers others can build on.** The output is a section in
|
||||
`perf-targets.md` and rows in `bench/baseline.json`, not a paragraph of
|
||||
prose. A measurement that only printed once is not a measurement.
|
||||
**Measure the curve, not the cliff.** Everything dies at the ceiling; what
|
||||
matters is the shape on the way down — where p99 leaves its 1µs baseline, what
|
||||
insert throughput does as slabs stop coming from a warm allocator, and how much
|
||||
warning there is between "fine" and "unusable".
|
||||
|
||||
## Phases
|
||||
### Fork 1 — the limit mechanism: rootless cgroup v2 via `systemd-run --user`
|
||||
|
||||
### Phase A — a workload that can actually reach the ceiling
|
||||
`systemd-run --user --scope -p MemoryMax=N -p MemorySwapMax=M`. Verified on the
|
||||
dev box: the `memory` controller is delegated to
|
||||
`user.slice/user-<uid>.slice`, a scope's `memory.max` reads back exactly as set,
|
||||
and no passwordless sudo is needed. Being cgroup-scoped also isolates the
|
||||
measurement from whatever else the box is doing, which matters — the dev box was
|
||||
at 22.9 of 31.7 GiB with 4.6 GiB of swap already in use when this was refined.
|
||||
|
||||
- Extend `docs/examples/db-bench` with a growth mode: insert until a target RSS
|
||||
fraction, holding row shape and index count constant so the variable is size
|
||||
alone.
|
||||
- Run it under an explicit memory limit (a cgroup or `ulimit`) rather than on a
|
||||
big box — "it survived on a 64 GB workstation" measures the workstation.
|
||||
- Record RSS against row count so the per-row overhead is known: slab headroom,
|
||||
the id hash, the secondary-index multimaps and the per-row engine-owned values
|
||||
(`db_text`, `db_rec`, `db_multi`, `db_map` are each their own allocation).
|
||||
- Verify: RSS growth is linear and its slope is written down; the run is
|
||||
reproducible twice within the tolerance policy iteration 22 established.
|
||||
`ulimit -v` is **rejected**: it bounds address space, not resident set, which is
|
||||
the wrong quantity for an engine that `malloc`s slabs — and it is actively
|
||||
broken under ASan, whose huge virtual reservations trip it long before any real
|
||||
memory pressure.
|
||||
|
||||
### Phase B — the latency and throughput curve
|
||||
If the mechanism is unavailable (no systemd, no delegation), the harness **skips
|
||||
the growth legs loudly and names why**. It must never silently fall back to
|
||||
measuring an uncapped box, because "it survived on a 32 GiB workstation"
|
||||
measures the workstation.
|
||||
|
||||
- Sample read p50/p99, query p99 and insert throughput at fixed fractions of the
|
||||
limit, so the result is a curve rather than two endpoints.
|
||||
- Separate the two effects deliberately: allocator pressure (still resident) and
|
||||
swap (no longer resident). They have different fixes and conflating them would
|
||||
send iteration 6 after the wrong one.
|
||||
- Include the DB-actor path, since a cross-shard statement's reply materialises
|
||||
a copy — memory pressure and the actor RPC interact and nobody has looked.
|
||||
- Verify: the curve is recorded per metric class with iteration 22's per-class
|
||||
tolerances; the swap onset point is identified, not interpolated.
|
||||
### Fork 2 — the reference shapes: both, reported separately
|
||||
|
||||
### Phase C — the three exits, deliberately triggered
|
||||
Per-row footprint differs substantially between an Int-only row and a text-heavy
|
||||
one, because a `Text` column is a separate `db_text` allocation per row on top of
|
||||
the slab slot. **Measured 2026-08-27: 96.5 B/row Int-only vs 320.6 B/row with
|
||||
three Text columns — 3.3×.** An earlier draft of this section said "an order of
|
||||
magnitude"; that was an unmeasured guess and this iteration exists to replace
|
||||
exactly that kind of claim. 3.3× is still more than enough to make a single
|
||||
"bytes per row" number useless, which is the decision it was supporting.
|
||||
|
||||
- Drive a checked allocation failure and confirm `WO_T_OOM` is catchable, the
|
||||
insert is refused whole, no partial row or index entry is left, and the
|
||||
process continues serving.
|
||||
- Drive swap thrash and record what a client sees. This is the case with no
|
||||
error signal at all, and naming it is most of the value of this iteration.
|
||||
- Drive the OOM killer under a cgroup limit and confirm what survives: replay
|
||||
the WAL and check every acked write is present.
|
||||
- Verify: the trap path leaves no torn state (row count and index agree after a
|
||||
refused insert); replay after `SIGKILL` at exhaustion loses no acked write.
|
||||
`db-bench` already supplies half of this: `items` (`k: Int`, `v: Int`, plus a
|
||||
`bucket` ref) is the Int-only reference as it stands. The work is one text-heavy
|
||||
shape beside it, with footprint reported per shape.
|
||||
|
||||
### Phase D — write it down where decisions get made
|
||||
### Fork 3 — swap: in scope, as a controlled dimension
|
||||
|
||||
- A `perf-targets.md` section with the curve, the swap onset, the per-row
|
||||
overhead and the exit characterisation.
|
||||
- Baseline rows for the growth metrics so a regression is caught by the existing
|
||||
gate rather than by a person remembering.
|
||||
- A short statement of what the numbers *imply* for iterations 2, 5 and 6 —
|
||||
which is the point of going first.
|
||||
- Verify: `just db-bench` green against the extended baseline; the gate bites
|
||||
when a growth metric is doctored.
|
||||
Not a confound to wish away — `MemorySwapMax` is the knob that separates the two
|
||||
exits this iteration exists to characterise. Both were measured, and **both
|
||||
turned out differently than this iteration predicted.**
|
||||
|
||||
| Leg | Predicted | Measured |
|
||||
| --- | --- | --- |
|
||||
| swap-off | catchable `WO_T_OOM` from checked `malloc` | **SIGKILL, signal 9** (shell rc 137). No trap, no message |
|
||||
| swap-on | latency collapse | **no degradation at all**: 148 s vs 150 s uncapped |
|
||||
|
||||
**Prediction 1 was wrong because of overcommit.** With `vm.overcommit_memory = 0`
|
||||
`malloc` succeeds and the process dies when it *touches* the pages, so table
|
||||
storage never gets the chance to report failure. The trap path is real but
|
||||
belongs to a different allocator:
|
||||
|
||||
| Allocator | Ceiling | Failure mode |
|
||||
| --- | --- | --- |
|
||||
| VM object arena | `WO_HEAP_MB`, checked | `trap 4` / `WO_T_OOM`, exit 1, reportable |
|
||||
| table storage (slabs + heap values) | **none** | SIGKILL under overcommit |
|
||||
|
||||
**This is the strongest argument available for [iteration 2](02-table-storage-modes.md)'s
|
||||
byte budget:** a declared budget is the only way table storage can acquire a
|
||||
checked ceiling, because `malloc` under default overcommit will never tell it
|
||||
there is a problem.
|
||||
|
||||
**Prediction 2 was wrong because of access pattern.** 900 000 Int rows under a
|
||||
64 MiB cap with 256 MiB of swap finished in **148 s** with RSS pinned at 62 MiB;
|
||||
the same workload uncapped took **150 s** at 165 MiB RSS. Swap cost
|
||||
approximately nothing. The reason is that inserting is append-mostly: cold pages
|
||||
are written out once and never read again, so paging is sequential and off the
|
||||
critical path. The swap is a real disk file (`/swap.img`, no zram, zswap
|
||||
disabled), so this is genuine disk paging, not compressed RAM.
|
||||
|
||||
**The correct generalisation is narrower than "swap is fine".** This measures an
|
||||
append-mostly workload. A workload that reads randomly across a table larger
|
||||
than the cap is the one that collapses, and this iteration did *not* measure
|
||||
that — see Outstanding.
|
||||
|
||||
## Progress
|
||||
|
||||
| Piece | State |
|
||||
| --- | --- |
|
||||
| `Wide` text-heavy reference shape (`db-bench/types.wo`) | ✅ |
|
||||
| `growth N int\|text` — insert, per-decile RSS and read latency | ✅ |
|
||||
| the sample reads its OWN RSS via `/proc/self/status` | ✅ — the driver polls every 250 ms and would miss the value *at* a decile boundary |
|
||||
| `growth-verify` — the survivor is a contiguous intact prefix | ✅ |
|
||||
| rootless cap wrapper, swap on/off legs | ✅ 4 footprint legs |
|
||||
| footprint metric = **median of marginals**, doublings counted separately | ✅ |
|
||||
| `ceiling` leg: dies at the cap, then replay must be intact | ✅ gated |
|
||||
| baseline + tolerance policy | ✅ 121 checks; footprint at ±10%, kill-timing metrics at ±100% |
|
||||
| `perf-targets.md` §5 | ✅ |
|
||||
| **resident-footprint fraction for [iteration 2](02-table-storage-modes.md)** | ⬜ **not delivered — the premise it rested on is false**, see Outstanding |
|
||||
| **replay/restart baseline for [iteration 3](03-wal-checkpoint.md)** | ⬜ not delivered |
|
||||
| **the random-read-over-cap collapse** | ⬜ not measured |
|
||||
|
||||
## Measured
|
||||
|
||||
Footprint, reproducible inside 2% across runs:
|
||||
|
||||
| What | Int-only (`Item`) | Text-heavy (`Wide`) |
|
||||
| --- | --- | --- |
|
||||
| steady-state footprint | **96.5–100 B/row** | **320.6–324 B/row** |
|
||||
| doubling steps | 3 (at ~24k and ~48k rows) | 2 |
|
||||
| base process RSS | ≈ 3.9 MiB, excluded from the per-row figure | same |
|
||||
|
||||
Ratio **3.3×** — not the "order of magnitude" an earlier draft asserted. Enough
|
||||
on its own to make a single "bytes per row" number useless, which is the decision
|
||||
it was supporting ([2](02-table-storage-modes.md), fork 5).
|
||||
|
||||
The ceiling, 60 000 Int rows under an 8 MiB cap with swap off and `WO_DATA` set:
|
||||
|
||||
| Question | Answer |
|
||||
| --- | --- |
|
||||
| how does it die? | **SIGKILL, signal 9.** No refusal, no diagnostic |
|
||||
| what survives? | **a contiguous intact prefix** — ~40 000 rows, every `v` correct, no holes, not reported as corruption |
|
||||
|
||||
**Ack-after-fsync holds through an OOM kill.** That is the one shutdown path
|
||||
which skips every cleanup handler, and the durable prefix came back whole.
|
||||
|
||||
**The finding that matters most is the swap leg succeeding.** It did not fail,
|
||||
did not warn, and returned 0. A deployment in that state looks healthy while
|
||||
serving from disk. That is the exit with no error signal, and it is why
|
||||
[iteration 5](05-bounded-tables-eviction.md)'s back-pressure must act at a
|
||||
declared threshold rather than at exhaustion — exhaustion either kills without
|
||||
warning or silently does not arrive.
|
||||
|
||||
## Acceptance Criteria
|
||||
|
||||
- **Given** the growth workload under a fixed memory limit, **when** it runs
|
||||
twice, **then** RSS-per-row agrees within the tolerance policy and the slope
|
||||
is recorded in `perf-targets.md`.
|
||||
- **Given** the workload at rising RAM fractions, **when** latency is sampled,
|
||||
**then** the fraction at which read p99 first leaves its baseline is
|
||||
identified as a measured point, not an estimate.
|
||||
- **Given** a deliberately induced allocation failure, **when** an insert is
|
||||
attempted, **then** it traps `WO_T_OOM` catchably, the table's row count is
|
||||
unchanged, every index agrees with the slab contents, and the process keeps
|
||||
serving subsequent requests.
|
||||
- **Given** swap thrash, **when** a client issues reads, **then** the observed
|
||||
degradation is quantified and the fact that **no error is surfaced** is
|
||||
recorded explicitly as a finding.
|
||||
- **Given** a cgroup limit and a workload that exceeds it, **when** the OOM
|
||||
killer fires, **then** replaying the WAL shows every acked write present —
|
||||
ack-after-fsync holding in the one shutdown path that skips all cleanup.
|
||||
Met:
|
||||
|
||||
- **Given** the growth workload under a fixed cap, **when** it runs twice,
|
||||
**then** RSS-per-row agrees inside tolerance and the slope is recorded per
|
||||
shape. ✅ inside 2%; `perf-targets.md` §5.
|
||||
- **Given** the swap-off leg, **when** the cap is exceeded, **then** the exit is
|
||||
identified and recorded. ✅ **SIGKILL, signal 9** — not the catchable trap this
|
||||
criterion originally expected, which is the whole point of measuring. The
|
||||
"process keeps serving" half of the original wording is **void**: nothing
|
||||
survives a SIGKILL.
|
||||
- **Given** a cap exceeded with `WO_DATA` set, **when** the process is killed at
|
||||
exhaustion, **then** replay shows the acked writes present. ✅ ~40 000 rows,
|
||||
contiguous, no holes, no corruption report. Gated as the `ceiling` leg.
|
||||
- **Given** the swap-on leg, **when** the same point is reached, **then** the
|
||||
degradation is quantified **and the absence of any error signal recorded**.
|
||||
✅ degradation is **nil** for this workload (148 s vs 150 s uncapped) and the
|
||||
silence is total. Both halves are findings; the first inverted the prediction.
|
||||
- **Given** the extended baseline, **when** a growth metric is doctored, **then**
|
||||
`just db-bench` fails on exactly that metric.
|
||||
the gate fails on exactly that metric. ✅ text footprint +20% →
|
||||
`FAIL gate.growth.text.noswap.bytes_per_row -- 388 vs baseline 324`, 1 of 104.
|
||||
- **Given** a host without the cap mechanism, **when** the harness runs, **then**
|
||||
the legs are skipped with a named reason and the rest still passes. ✅
|
||||
`cap_wrapper` returns None unless the `memory` controller is delegated; there
|
||||
is no uncapped fallback.
|
||||
|
||||
Outstanding:
|
||||
|
||||
- **The resident-footprint fraction for iteration 2's budget default. NOT
|
||||
delivered, and the premise is false.** It was to be derived from the
|
||||
swap-onset point — but there is no onset: swap-off jumps straight from
|
||||
working to SIGKILL, and swap-on shows no degradation to detect an onset in.
|
||||
**Iteration 2 must pick its budget on other grounds** (host RAM fraction, or
|
||||
an explicit developer-declared figure) rather than waiting on a number this
|
||||
iteration cannot produce. This is the most important thing this slice learned
|
||||
and it removes a dependency rather than satisfying it.
|
||||
- **The random-read-over-cap collapse.** Not measured. This is where the "latency
|
||||
collapse" prediction may still be true, and it is the workload that matters
|
||||
for [2](02-table-storage-modes.md)'s `resident: keys`, whose whole premise is
|
||||
reading rows back from a log larger than RAM. Needs a read-heavy leg over a
|
||||
table exceeding the cap. **The single most valuable follow-up.**
|
||||
- **Given** rising fractions of the cap, **when** latency is sampled, **then**
|
||||
the p99 departure point is recorded. Partially: the sampler and metric exist
|
||||
and are gated, but the footprint legs never approach their 512 MiB cap, so
|
||||
`p99_departure_decile` is legitimately 0 and proves nothing. It becomes
|
||||
meaningful only with the read-heavy leg above.
|
||||
- **A replay/restart baseline for iteration 3.** Not delivered; `growth` exercises
|
||||
`WO_DATA` but nothing times replay. Cheap to add, still absent from
|
||||
`bench/baseline.json`.
|
||||
|
||||
## Out Of Scope
|
||||
|
||||
- **Any fix.** This iteration measures. Eviction is
|
||||
[5](05-bounded-tables-eviction.md), tiering is [6](06-cold-tiering.md),
|
||||
declared budgets are [2](02-table-storage-modes.md). Shipping a fix inside the
|
||||
measurement slice would remove the ability to tell whether it helped.
|
||||
- **Changing the OOM behaviour.** The checked-`malloc`-to-catchable-trap path is
|
||||
good and should not be touched; if the measurement finds a hole in it, that is
|
||||
a bug fix, reported separately.
|
||||
- **A memory profiler or allocator instrumentation.** Observability is language
|
||||
iteration 30. RSS from the OS and the existing `time.ticks` are enough for a
|
||||
curve.
|
||||
- **Multi-machine or sharded-across-hosts scaling.** One binary owns its data;
|
||||
cross-process is [9](09-cross-program-tables.md).
|
||||
- **Comparing against SQLite at the ceiling.** `bench/compare/go-sqlite` exists
|
||||
and the comparison would be interesting, but SQLite's whole architecture is
|
||||
the paged design this project rejected — the numbers would not inform any
|
||||
decision here.
|
||||
- **Any fix.** This measures. Declared budgets are [2](02-table-storage-modes.md),
|
||||
eviction is [5](05-bounded-tables-eviction.md), tiering is 2's `resident: keys`.
|
||||
- **Changing the OOM behaviour.** The checked-`malloc` code is untouched. The
|
||||
measurement showed it is largely unreachable for table storage under default
|
||||
overcommit — a finding to hand to [2](02-table-storage-modes.md), not a bug to
|
||||
fix here, and emphatically not a licence to start setting
|
||||
`vm.overcommit_memory`.
|
||||
- **A memory profiler or allocator instrumentation** — observability is language
|
||||
iteration 30. RSS from `/proc` plus `time.ticks` is enough for a curve.
|
||||
- **Comparing against SQLite at the ceiling.** `bench/compare/go-sqlite` exists,
|
||||
but SQLite's paged architecture is the design this project rejected, so the
|
||||
numbers would inform no decision here.
|
||||
- **Multi-host scaling** — one binary owns its data.
|
||||
|
||||
## Info
|
||||
## Info — the forks, settled
|
||||
|
||||
Forks the spec must settle:
|
||||
1. **The limit mechanism is rootless cgroup v2** via
|
||||
`systemd-run --user --scope -p MemoryMax -p MemorySwapMax`. `ulimit -v` was
|
||||
rejected: it bounds address space, not resident set, and ASan's virtual
|
||||
reservations trip it long before real pressure. No sudo needed; it also
|
||||
isolates the run from the rest of the box, which mattered — the dev box sat
|
||||
at 22.9 of 31.7 GiB throughout.
|
||||
2. **Both reference shapes, reported separately.** 3.3× apart; one number would
|
||||
be a fiction.
|
||||
3. **Swap is a dimension, not a footnote** — settled by getting it wrong first.
|
||||
An early run looked like the cap was unenforced because the process held
|
||||
400 MiB inside a 64 MiB limit; it was swapping, which is the phenomenon under
|
||||
study.
|
||||
4. **Footprint is read as the median of per-decile marginals**, not a two-point
|
||||
slope, so a slab doubling does not smear into the per-row figure. Doublings
|
||||
are counted as their own metric.
|
||||
5. **Kill-timing metrics carry ±100% tolerance.** `rows_recovered` depends on
|
||||
where the SIGKILL landed; gating it tightly would be gating the scheduler.
|
||||
The invariant asserted instead is the *shape* of the survivor.
|
||||
6. **The ceiling leg asserts `rc`, never records it.** When iteration 2's byte
|
||||
budget lands, death should become a checked refusal — the gate must not fail
|
||||
on that improvement.
|
||||
|
||||
1. **What is the limit mechanism for the harness?** A cgroup v2 `memory.max` is
|
||||
the closest thing to how this would actually be deployed; `ulimit -v` is
|
||||
simpler but bounds address space rather than resident set, which for an engine
|
||||
that `malloc`s slabs is a materially different constraint. Leaning cgroup, and
|
||||
the campaign already runs off the fast path so the setup cost is acceptable.
|
||||
2. **Which table shape is the reference?** Per-row overhead depends heavily on
|
||||
whether fields are scalars or heap values — a `Text` column is a separate
|
||||
`db_text` allocation per row, so a text-heavy table and an Int-only table
|
||||
will produce very different slopes. Probably both, reported separately,
|
||||
because "bytes per row" is meaningless without saying which row.
|
||||
3. **Is swap even in scope for the target deployment?** If the intended answer
|
||||
is "run with swap off and let the OOM killer decide", the swap curve is
|
||||
informational rather than load-bearing — but that stance should be stated in
|
||||
the doctrine, not assumed. It also changes which exit iteration 5's
|
||||
back-pressure is defending against.
|
||||
## History — four corrections worth keeping
|
||||
|
||||
**"An order of magnitude" was a guess.** The per-shape difference is 3.3×. An
|
||||
iteration whose purpose is replacing unmeasured claims had one in its own
|
||||
premise.
|
||||
|
||||
**The clean-exit premise was wrong.** This file and the residency spec both
|
||||
asserted the ceiling surfaces as a catchable `WO_T_OOM`. It is a SIGKILL.
|
||||
Overcommit means the allocator never learns there is a problem.
|
||||
|
||||
**The latency-collapse premise was wrong too.** Swap cost ~1% on an
|
||||
append-mostly workload (148 s vs 150 s). The prediction was not merely
|
||||
imprecise, it had the wrong sign. The narrower claim that survives is that a
|
||||
*random-read* workload over an oversized table is the one at risk, and that
|
||||
remains unmeasured.
|
||||
|
||||
**A SIGKILL was once labelled a "checked refusal"** by the harness, because
|
||||
`subprocess` reports signal death as a negative `returncode` (`-9`) while the
|
||||
shell spells the same event `137`. The leg existed specifically to tell those
|
||||
two apart. Fixed, and the distinction is now spelled out at the comparison.
|
||||
|
|
|
|||
|
|
@ -154,7 +154,7 @@ Outstanding:
|
|||
resident. Roughly doubles the resident index; stated at the declaration so
|
||||
the cost is visible.
|
||||
5. **The budget is bytes, not rows** — a text-heavy row and an Int-only row
|
||||
differ by an order of magnitude, so a row count cannot bound RAM.
|
||||
differ by 3.3× (measured, databasev2 1), so a row count cannot bound RAM.
|
||||
|
||||
## History — two corrections worth keeping
|
||||
|
||||
|
|
|
|||
|
|
@ -19,7 +19,7 @@
|
|||
| Which storage architecture | **One engine, log-structured.** The WAL already holds every row; keep an in-RAM id→offset map and read rows back with `pread`. No second engine. |
|
||||
| Row cache | **None in user space.** The kernel page cache is the hot copy — the repo's own stated position in `exploration/postgresql/buffer-and-checkpoint.md`: "`pread` against an fd that already has its page cached is a memcpy… the page cache is the one cache we want", and the reason the engine avoids `O_DIRECT`. |
|
||||
| `@unique` on a non-resident table | **Allowed; its index is unconditionally resident.** Settled here rather than deferred — see Constraints. |
|
||||
| Budget unit | **Bytes** (estimated resident footprint). Rows is the meaningless unit: a text-heavy row and an Int-only row differ by an order of magnitude, so a row count cannot bound RAM. |
|
||||
| Budget unit | **Bytes** (estimated resident footprint). Rows is the meaningless unit: a text-heavy row and an Int-only row differ by 3.3× (measured, databasev2 1), so a row count cannot bound RAM. |
|
||||
| Rejected architectures | `mmap` and a buffer pool stay out — see Alternatives rejected. `discarded.md`'s paged-engine rejection is amended to *partly revisited*, not reversed. |
|
||||
|
||||
## The problem, read off the engine
|
||||
|
|
@ -37,9 +37,15 @@ Facts, each verified in source rather than assumed:
|
|||
`WO_DATA` is set. `db.c` guards every WAL append with a null check on
|
||||
`vm->rt.wal`, so with no `WO_DATA` **every table is silently volatile** — a
|
||||
program can declare nothing and lose everything.
|
||||
- An allocation failure is clean: every `malloc` in the row encoder is checked
|
||||
and `DB_ERR_OOM` maps to `WO_T_OOM`, a catchable trap. The dangerous exit is
|
||||
the one *before* that — swap thrash, which carries no error signal at all.
|
||||
- An allocation failure is clean *in principle*: every `malloc` in the row
|
||||
encoder is checked and `DB_ERR_OOM` maps to `WO_T_OOM`. **Corrected 2026-08-27
|
||||
by measurement (databasev2 1): that path does not fire in practice.** With
|
||||
`vm.overcommit_memory = 0`, `malloc` succeeds and the process is SIGKILLed
|
||||
when it touches the pages — measured rc=137 at 360 000 rows under a 64 MiB
|
||||
cgroup cap. The checked-trap path belongs to the VM arena (`WO_HEAP_MB`,
|
||||
verified `trap 4 ... out of memory`), not to table storage, which has no
|
||||
ceiling at all. This makes the byte budget below the ONLY mechanism by which
|
||||
table storage can acquire one.
|
||||
|
||||
The measurements that bound the design, from iteration 22: durable inserts
|
||||
≈4.5k/s against RAM ≈297k/s (the 66× fsync gap); reads 1.3M ops/s at p50 1µs
|
||||
|
|
|
|||
|
|
@ -225,6 +225,14 @@ def tolerance_for(key):
|
|||
mix*: scheduling-dependent small counts. read/query + all .sN.*:
|
||||
machine jitter, and at post-index-µs scale a 1µs histogram step on a
|
||||
7µs p50 is already 14%."""
|
||||
# databasev2 1: footprint is a STRUCTURAL number -- 96.5 vs 320.6 B/row
|
||||
# reproduced to <2% across runs -- so it gets a tight tolerance and is the
|
||||
# one growth metric worth gating. The doubling COUNT and the latency
|
||||
# samples are allowed to move: doublings depend on where N lands relative
|
||||
# to a pow2 rehash, and at 1us p50 a single histogram step is already 100%.
|
||||
if ".bytes_per_row" in key: return 10
|
||||
if key.startswith("growth."): return 100
|
||||
if key.startswith("ceiling."): return 100
|
||||
if ".mixread." in key or ".mixwrite." in key: return 50
|
||||
if ".sN." in key: return 50
|
||||
if ".read." in key or ".query." in key: return 50
|
||||
|
|
@ -248,6 +256,183 @@ def write_baseline(metrics):
|
|||
json.dump(base, open(BASELINE, "w"), indent=1, sort_keys=True)
|
||||
ok(f"baseline written ({len(base) - 1} metrics)")
|
||||
|
||||
# ---- databasev2 1: the RAM ceiling -----------------------------------------
|
||||
|
||||
GROWTH_N = 20000 if QUICK else 200000
|
||||
GROWTH_SHAPES = ("int", "text")
|
||||
|
||||
|
||||
def cap_wrapper(mem_mb, swap_mb):
|
||||
"""systemd-run --user --scope argv prefix that caps memory rootlessly, or
|
||||
None when the mechanism is unavailable.
|
||||
|
||||
cgroup v2 with the `memory` controller delegated to the user slice is the
|
||||
only mechanism used. `ulimit -v` is deliberately NOT a fallback: it bounds
|
||||
address space, not resident set, which is the wrong quantity for an engine
|
||||
that mallocs slabs, and ASan's huge virtual reservations trip it long
|
||||
before real memory pressure. When the cap is unavailable the legs are
|
||||
SKIPPED and say so -- never silently run uncapped, because "it survived on
|
||||
a 32 GiB workstation" measures the workstation."""
|
||||
if not shutil.which("systemd-run"):
|
||||
return None
|
||||
try:
|
||||
with open("/proc/self/cgroup") as f:
|
||||
mine = f.readline().strip().split(":")[-1]
|
||||
ctl = f"/sys/fs/cgroup{os.path.dirname(mine)}/cgroup.controllers"
|
||||
if "memory" not in open(ctl).read().split():
|
||||
return None
|
||||
except OSError:
|
||||
return None
|
||||
return ["systemd-run", "--user", "--scope", "--quiet",
|
||||
"-p", f"MemoryMax={mem_mb}M", "-p", f"MemorySwapMax={swap_mb}M", "--"]
|
||||
|
||||
|
||||
def parse_growth(lines):
|
||||
"""(rows, rss_kb) samples plus per-decile read p50/p99, from the sample's
|
||||
own `growthrss` / `growthN` lines. RSS is read by the SAMPLE, not polled
|
||||
here: the driver polls every 250 ms and would miss the value AT a decile
|
||||
boundary, and per-row footprint is this iteration's headline number."""
|
||||
pts, lat = [], {}
|
||||
for l in lines:
|
||||
f = l.split()
|
||||
if f and f[0] == "growthrss" and len(f) == 4:
|
||||
pts.append((int(f[2]), int(f[3])))
|
||||
elif f and f[0].startswith("growth") and len(f) == 5 and f[0][6:].isdigit():
|
||||
lat[int(f[0][6:])] = (int(f[3]), int(f[4]))
|
||||
return pts, lat
|
||||
|
||||
|
||||
def bytes_per_row(pts):
|
||||
"""Steady-state marginal footprint = MEDIAN of the per-interval marginals.
|
||||
|
||||
Not a two-point slope: the id hash and index buckets are open-addressing
|
||||
pow2 and DOUBLE periodically, so a two-point slope lands arbitrarily on or
|
||||
off a doubling and swings 2x (measured: 96 vs 205 B/row for the same shape).
|
||||
The median rejects those steps; they are reported separately as `doublings`
|
||||
because a transient RSS step is exactly what a resident-footprint budget
|
||||
must leave headroom for."""
|
||||
marg = sorted((k1 - k0) * 1024.0 / (r1 - r0)
|
||||
for (r0, k0), (r1, k1) in zip(pts, pts[1:]) if r1 > r0)
|
||||
if not marg:
|
||||
return None, 0
|
||||
med = marg[len(marg) // 2]
|
||||
doublings = sum(1 for m in marg if m > med * 1.5)
|
||||
return med, doublings
|
||||
|
||||
|
||||
def growth(metrics):
|
||||
"""Per-shape footprint and the read-latency curve, under a rootless cap,
|
||||
with swap ON and OFF.
|
||||
|
||||
What this leg actually measures is FOOTPRINT. It does not reach the cap:
|
||||
GROWTH_N rows need far less than the 512 MiB cap, so both swap legs are
|
||||
identical by construction and p99_departure_decile is legitimately 0.
|
||||
The ceiling itself is ceiling() below -- keep the two separate, because a
|
||||
footprint regression and a ceiling-behaviour change are different faults.
|
||||
|
||||
Two earlier claims in this docstring were measured FALSE and are recorded
|
||||
in docs/stories/databasev2/01-ram-ceiling-measurement.md: swap-off is not
|
||||
a "clean checked-malloc" path (it is SIGKILL, rc=137), and swap-on is not
|
||||
"latency collapse" (900k rows finished in 148s capped-with-swap vs 150s
|
||||
uncapped -- an append-mostly workload never re-touches its cold pages)."""
|
||||
wrap = cap_wrapper(512, 0)
|
||||
if wrap is None:
|
||||
ok("growth: SKIPPED -- no rootless cgroup v2 memory cap on this host")
|
||||
metrics["growth.available"] = 0
|
||||
return
|
||||
metrics["growth.available"] = 1
|
||||
for shape in GROWTH_SHAPES:
|
||||
for legname, swap_mb in (("noswap", 0), ("swap", 256)):
|
||||
w = cap_wrapper(512, swap_mb)
|
||||
env = dict(os.environ)
|
||||
argv = w + [BIN, "growth", str(GROWTH_N), shape]
|
||||
pr = subprocess.run(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
|
||||
text=True, env=env, timeout=900)
|
||||
lines = pr.stdout.splitlines()
|
||||
pts, lat = parse_growth(lines)
|
||||
key = f"growth.{shape}.{legname}"
|
||||
if not pts:
|
||||
bad(f"{key}: produced no samples", (lines[-1] if lines else "no output"))
|
||||
continue
|
||||
bpr, doublings = bytes_per_row(pts)
|
||||
metrics[f"{key}.bytes_per_row"] = int(round(bpr))
|
||||
metrics[f"{key}.doublings"] = doublings
|
||||
metrics[f"{key}.rows"] = pts[-1][0]
|
||||
metrics[f"{key}.rss_kb"] = pts[-1][1]
|
||||
if lat:
|
||||
last = max(lat)
|
||||
metrics[f"{key}.read_p50us"] = lat[last][0]
|
||||
metrics[f"{key}.read_p99us"] = lat[last][1]
|
||||
# the curve's departure point: first decile whose p99 exceeds
|
||||
# 4x the first decile's, as a MEASURED sample not an estimate
|
||||
first = lat[min(lat)][1]
|
||||
dep = next((d for d in sorted(lat) if lat[d][1] > max(first, 1) * 4), 0)
|
||||
metrics[f"{key}.p99_departure_decile"] = dep
|
||||
ok(f"{key}: {int(round(bpr))} B/row steady, {doublings} doubling step(s), "
|
||||
f"{pts[-1][0]} rows in {pts[-1][1]} KiB")
|
||||
|
||||
|
||||
|
||||
CEIL_N, CEIL_CAP_MB = 60000, 8
|
||||
|
||||
def ceiling(metrics):
|
||||
"""The ceiling itself, and the durability claim across it.
|
||||
|
||||
Sized so the process CANNOT fit: 60k Int rows need ~9.7 MiB resident
|
||||
(96.5 B/row measured, plus a ~3.9 MiB base) under an 8 MiB cap, swap off.
|
||||
Two things are under test and the second is the one that matters:
|
||||
|
||||
1. HOW it dies. Measured: SIGKILL, rc=137 -- not a refusal. Table
|
||||
storage has no checked ceiling, and under vm.overcommit_memory=0
|
||||
malloc succeeds and the process dies TOUCHING the pages, so it never
|
||||
gets the chance to report failure. (The VM object arena is the
|
||||
opposite: WO_HEAP_MB is checked and traps.) rc is asserted, not
|
||||
recorded as a metric -- when databasev2 2's byte budget lands this
|
||||
should become a checked refusal, and the gate must not fail on that
|
||||
improvement.
|
||||
|
||||
2. WHAT SURVIVES. With WO_DATA set, replay must yield a contiguous
|
||||
intact prefix: rows 1..M present with the right v, no holes, and not
|
||||
reported as corruption. M is wherever the kill landed -- the SHAPE of
|
||||
the survivor is the claim, not its size, so rows_recovered carries a
|
||||
wide tolerance. This is ack-after-fsync holding in the one shutdown
|
||||
path that skips every cleanup handler."""
|
||||
wrap = cap_wrapper(CEIL_CAP_MB, 0)
|
||||
if wrap is None:
|
||||
ok("ceiling: SKIPPED -- no rootless cgroup v2 memory cap on this host")
|
||||
return
|
||||
data = os.path.join(ROOT, "bench", f"tmp.{os.getpid()}.ceiling")
|
||||
shutil.rmtree(data, ignore_errors=True); os.makedirs(data, exist_ok=True)
|
||||
env = dict(os.environ); env["WO_DATA"] = data
|
||||
pr = subprocess.run(wrap + [BIN, "growth", str(CEIL_N), "int"],
|
||||
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
|
||||
text=True, env=env, timeout=900)
|
||||
if pr.returncode == 0:
|
||||
bad("ceiling: process SURVIVED the cap",
|
||||
f"{CEIL_N} rows fit under {CEIL_CAP_MB} MiB -- footprint changed, resize the leg")
|
||||
shutil.rmtree(data, ignore_errors=True); return
|
||||
# subprocess returncode is NEGATIVE for signal death (-9 = SIGKILL); 137
|
||||
# is the SHELL spelling of the same event (128+9). Getting this backwards
|
||||
# once labelled a SIGKILL as a "checked refusal", which is the exact
|
||||
# distinction this leg exists to report.
|
||||
if pr.returncode < 0:
|
||||
sig = -pr.returncode
|
||||
how = f"killed by signal {sig}" + (" (SIGKILL -- no checked refusal)" if sig == 9 else "")
|
||||
else:
|
||||
how = f"exited {pr.returncode} (checked refusal)"
|
||||
ok(f"ceiling: died at the cap, {how}")
|
||||
vr = subprocess.run([BIN, "growth-verify"], stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT, text=True, env=env, timeout=900)
|
||||
m = re.search(r"^growthverify (\d+)$", vr.stdout, re.M)
|
||||
if vr.returncode == 0 and m and int(m.group(1)) > 0:
|
||||
metrics["ceiling.rows_recovered"] = int(m.group(1))
|
||||
ok(f"ceiling: durable prefix intact across the kill -- {m.group(1)} rows, no holes")
|
||||
else:
|
||||
bad("ceiling: durable prefix broken across the kill",
|
||||
(vr.stdout.strip().splitlines() or ["no output"])[-1][:160])
|
||||
shutil.rmtree(data, ignore_errors=True)
|
||||
|
||||
|
||||
def main():
|
||||
# --check <results.json>: gate-only evaluation of a recorded run — the
|
||||
# gate-bites smoke doctors a copy and this mode must FAIL on it
|
||||
|
|
@ -260,6 +445,8 @@ def main():
|
|||
build()
|
||||
metrics = campaign()
|
||||
durability(metrics)
|
||||
growth(metrics)
|
||||
ceiling(metrics)
|
||||
os.makedirs(RESULTS_DIR, exist_ok=True)
|
||||
stamp = time.strftime("%Y%m%d-%H%M%S")
|
||||
out = os.path.join(RESULTS_DIR, f"run-{stamp}{'-quick' if QUICK else ''}.json")
|
||||
|
|
|
|||
Loading…
Reference in a new issue