feat(db-bench): random-read-over-cap leg — the 273x collapse
- `randread N R` in the sample: fill N rows, read R across the WHOLE range - Weyl order `i*2654435761 mod n` — no RNG in the language, none needed; both legs read the SAME key order so residency is the only variable - `randread` driver leg: control (256 MiB, does not bind) vs over-cap (6 MiB + swap), sizes kept modest — quick resolves it in ~5s - gates the RATIO, not the absolutes: over-cap reads/sec belongs to the box's swap device, the factor between two runs belongs to the engine - reads must all resolve (hits == R) or the leg fails; a collapse measured over unresolved reads is noise - 133 checks, 0 failures; gate bites on a doctored collapse_x Measured — this closes the gap the swap leg left: - resident 1 851 166 reads/sec, p50 0us p99 1us - over-cap 6 771 reads/sec, p50 128us p99 487us - 273x throughput, ~480x p99, all 20 000 reads resolving in both - so the two access patterns sit ~270x apart under identical pressure: append-mostly insert ~1%, random read 273x - departure is a STEP not a curve (1us -> 487us, nothing between), which is why p99_departure_decile finds no knee — there is none - caveat recorded, NOT inherited: this is demand-paged anonymous memory through swap (4 KiB/fault, no readahead). `resident: keys` preads via the page cache — should be better, but databasev2 2 task 7 must measure its own read path. New criterion added there Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
0c9b2c45d8
commit
a873cf7331
9 changed files with 437 additions and 137 deletions
|
|
@ -8,15 +8,15 @@
|
|||
},
|
||||
"ceiling.rows_recovered": {
|
||||
"dir": "lower",
|
||||
"floor": 159744,
|
||||
"floor": 159828,
|
||||
"tolerance_pct": 100,
|
||||
"value": 39936
|
||||
"value": 39957
|
||||
},
|
||||
"durable.s1.mixread.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 2239,
|
||||
"tolerance_pct": 50,
|
||||
"value": 8958
|
||||
"value": 8957
|
||||
},
|
||||
"durable.s1.mixread.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -44,15 +44,15 @@
|
|||
},
|
||||
"durable.s1.mixwrite.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 872,
|
||||
"floor": 868,
|
||||
"tolerance_pct": 50,
|
||||
"value": 218
|
||||
"value": 217
|
||||
},
|
||||
"durable.s1.query.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 202429,
|
||||
"floor": 215517,
|
||||
"tolerance_pct": 50,
|
||||
"value": 809716
|
||||
"value": 862068
|
||||
},
|
||||
"durable.s1.query.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -64,13 +64,13 @@
|
|||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 2
|
||||
"value": 1
|
||||
},
|
||||
"durable.s1.read.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 215703,
|
||||
"floor": 221827,
|
||||
"tolerance_pct": 50,
|
||||
"value": 862812
|
||||
"value": 887311
|
||||
},
|
||||
"durable.s1.read.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -86,45 +86,45 @@
|
|||
},
|
||||
"durable.s1.seed.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1078,
|
||||
"floor": 1128,
|
||||
"tolerance_pct": 15,
|
||||
"value": 4315
|
||||
"value": 4515
|
||||
},
|
||||
"durable.s1.seed.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 828,
|
||||
"floor": 832,
|
||||
"tolerance_pct": 15,
|
||||
"value": 207
|
||||
"value": 208
|
||||
},
|
||||
"durable.s1.seed.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 2160,
|
||||
"floor": 2180,
|
||||
"tolerance_pct": 15,
|
||||
"value": 540
|
||||
"value": 545
|
||||
},
|
||||
"durable.s1.write.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1156,
|
||||
"floor": 1141,
|
||||
"tolerance_pct": 15,
|
||||
"value": 4626
|
||||
"value": 4565
|
||||
},
|
||||
"durable.s1.write.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 828,
|
||||
"floor": 832,
|
||||
"tolerance_pct": 15,
|
||||
"value": 207
|
||||
"value": 208
|
||||
},
|
||||
"durable.s1.write.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 1844,
|
||||
"floor": 2256,
|
||||
"tolerance_pct": 15,
|
||||
"value": 461
|
||||
"value": 564
|
||||
},
|
||||
"durable.sN.mixread.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1113,
|
||||
"floor": 1117,
|
||||
"tolerance_pct": 50,
|
||||
"value": 4452
|
||||
"value": 4468
|
||||
},
|
||||
"durable.sN.mixread.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -134,33 +134,33 @@
|
|||
},
|
||||
"durable.sN.mixread.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 15116,
|
||||
"floor": 19900,
|
||||
"tolerance_pct": 50,
|
||||
"value": 3779
|
||||
"value": 4975
|
||||
},
|
||||
"durable.sN.mixwrite.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 123,
|
||||
"floor": 124,
|
||||
"tolerance_pct": 50,
|
||||
"value": 494
|
||||
"value": 496
|
||||
},
|
||||
"durable.sN.mixwrite.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 1148,
|
||||
"floor": 1104,
|
||||
"tolerance_pct": 50,
|
||||
"value": 287
|
||||
"value": 276
|
||||
},
|
||||
"durable.sN.mixwrite.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 2932,
|
||||
"floor": 1928,
|
||||
"tolerance_pct": 50,
|
||||
"value": 733
|
||||
"value": 482
|
||||
},
|
||||
"durable.sN.query.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 333333,
|
||||
"floor": 331125,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1333333
|
||||
"value": 1324503
|
||||
},
|
||||
"durable.sN.query.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -176,9 +176,9 @@
|
|||
},
|
||||
"durable.sN.read.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 298329,
|
||||
"floor": 341064,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1193317
|
||||
"value": 1364256
|
||||
},
|
||||
"durable.sN.read.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -190,13 +190,13 @@
|
|||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 2
|
||||
"value": 1
|
||||
},
|
||||
"durable.sN.seed.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1139,
|
||||
"floor": 1125,
|
||||
"tolerance_pct": 50,
|
||||
"value": 4556
|
||||
"value": 4502
|
||||
},
|
||||
"durable.sN.seed.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -206,27 +206,27 @@
|
|||
},
|
||||
"durable.sN.seed.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 2000,
|
||||
"floor": 2624,
|
||||
"tolerance_pct": 50,
|
||||
"value": 500
|
||||
"value": 656
|
||||
},
|
||||
"durable.sN.write.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1153,
|
||||
"floor": 1150,
|
||||
"tolerance_pct": 50,
|
||||
"value": 4614
|
||||
"value": 4600
|
||||
},
|
||||
"durable.sN.write.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 828,
|
||||
"floor": 832,
|
||||
"tolerance_pct": 50,
|
||||
"value": 207
|
||||
"value": 208
|
||||
},
|
||||
"durable.sN.write.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 2172,
|
||||
"floor": 2348,
|
||||
"tolerance_pct": 50,
|
||||
"value": 543
|
||||
"value": 587
|
||||
},
|
||||
"growth.available": {
|
||||
"dir": "lower",
|
||||
|
|
@ -236,9 +236,9 @@
|
|||
},
|
||||
"growth.int.noswap.bytes_per_row": {
|
||||
"dir": "lower",
|
||||
"floor": 400,
|
||||
"floor": 392,
|
||||
"tolerance_pct": 10,
|
||||
"value": 100
|
||||
"value": 98
|
||||
},
|
||||
"growth.int.noswap.doublings": {
|
||||
"dir": "lower",
|
||||
|
|
@ -272,15 +272,15 @@
|
|||
},
|
||||
"growth.int.noswap.rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 23936,
|
||||
"floor": 23920,
|
||||
"tolerance_pct": 100,
|
||||
"value": 5984
|
||||
"value": 5980
|
||||
},
|
||||
"growth.int.swap.bytes_per_row": {
|
||||
"dir": "lower",
|
||||
"floor": 400,
|
||||
"floor": 392,
|
||||
"tolerance_pct": 10,
|
||||
"value": 100
|
||||
"value": 98
|
||||
},
|
||||
"growth.int.swap.doublings": {
|
||||
"dir": "lower",
|
||||
|
|
@ -314,15 +314,15 @@
|
|||
},
|
||||
"growth.int.swap.rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 23952,
|
||||
"floor": 23920,
|
||||
"tolerance_pct": 100,
|
||||
"value": 5988
|
||||
"value": 5980
|
||||
},
|
||||
"growth.text.noswap.bytes_per_row": {
|
||||
"dir": "lower",
|
||||
"floor": 1296,
|
||||
"floor": 1288,
|
||||
"tolerance_pct": 10,
|
||||
"value": 324
|
||||
"value": 322
|
||||
},
|
||||
"growth.text.noswap.doublings": {
|
||||
"dir": "lower",
|
||||
|
|
@ -356,15 +356,15 @@
|
|||
},
|
||||
"growth.text.noswap.rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 41216,
|
||||
"floor": 41184,
|
||||
"tolerance_pct": 100,
|
||||
"value": 10304
|
||||
"value": 10296
|
||||
},
|
||||
"growth.text.swap.bytes_per_row": {
|
||||
"dir": "lower",
|
||||
"floor": 1296,
|
||||
"floor": 1288,
|
||||
"tolerance_pct": 10,
|
||||
"value": 324
|
||||
"value": 322
|
||||
},
|
||||
"growth.text.swap.doublings": {
|
||||
"dir": "lower",
|
||||
|
|
@ -398,15 +398,15 @@
|
|||
},
|
||||
"growth.text.swap.rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 41216,
|
||||
"floor": 41200,
|
||||
"tolerance_pct": 100,
|
||||
"value": 10304
|
||||
"value": 10300
|
||||
},
|
||||
"ram.s1.mixread.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 2239,
|
||||
"floor": 2236,
|
||||
"tolerance_pct": 50,
|
||||
"value": 8956
|
||||
"value": 8945
|
||||
},
|
||||
"ram.s1.mixread.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -424,13 +424,13 @@
|
|||
"dir": "higher",
|
||||
"floor": 248,
|
||||
"tolerance_pct": 50,
|
||||
"value": 995
|
||||
"value": 993
|
||||
},
|
||||
"ram.s1.mixwrite.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 0
|
||||
"value": 1
|
||||
},
|
||||
"ram.s1.mixwrite.p99us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -440,15 +440,15 @@
|
|||
},
|
||||
"ram.s1.msgrate.msgs_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 419674,
|
||||
"floor": 392834,
|
||||
"tolerance_pct": 15,
|
||||
"value": 3357394
|
||||
"value": 3142677
|
||||
},
|
||||
"ram.s1.query.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 340136,
|
||||
"floor": 214592,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1360544
|
||||
"value": 858369
|
||||
},
|
||||
"ram.s1.query.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -460,13 +460,13 @@
|
|||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1
|
||||
"value": 2
|
||||
},
|
||||
"ram.s1.read.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 369276,
|
||||
"floor": 226860,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1477104
|
||||
"value": 907441
|
||||
},
|
||||
"ram.s1.read.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -478,31 +478,31 @@
|
|||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1
|
||||
"value": 2
|
||||
},
|
||||
"ram.s1.seed.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 470366,
|
||||
"floor": 273373,
|
||||
"tolerance_pct": 15,
|
||||
"value": 1881467
|
||||
"value": 1093493
|
||||
},
|
||||
"ram.s1.seed.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 15,
|
||||
"value": 0
|
||||
"value": 1
|
||||
},
|
||||
"ram.s1.seed.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 15,
|
||||
"value": 1
|
||||
"value": 2
|
||||
},
|
||||
"ram.s1.write.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 294464,
|
||||
"floor": 159134,
|
||||
"tolerance_pct": 15,
|
||||
"value": 1177856
|
||||
"value": 636537
|
||||
},
|
||||
"ram.s1.write.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -514,25 +514,25 @@
|
|||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 15,
|
||||
"value": 1
|
||||
"value": 3
|
||||
},
|
||||
"ram.sN.mixread.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 2240,
|
||||
"floor": 2239,
|
||||
"tolerance_pct": 50,
|
||||
"value": 8960
|
||||
"value": 8958
|
||||
},
|
||||
"ram.sN.mixread.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 228,
|
||||
"floor": 232,
|
||||
"tolerance_pct": 50,
|
||||
"value": 57
|
||||
"value": 58
|
||||
},
|
||||
"ram.sN.mixread.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 1404,
|
||||
"floor": 1032,
|
||||
"tolerance_pct": 50,
|
||||
"value": 351
|
||||
"value": 258
|
||||
},
|
||||
"ram.sN.mixwrite.ops_sec": {
|
||||
"dir": "higher",
|
||||
|
|
@ -542,27 +542,27 @@
|
|||
},
|
||||
"ram.sN.mixwrite.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 248,
|
||||
"floor": 256,
|
||||
"tolerance_pct": 50,
|
||||
"value": 62
|
||||
"value": 64
|
||||
},
|
||||
"ram.sN.mixwrite.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 292,
|
||||
"floor": 328,
|
||||
"tolerance_pct": 50,
|
||||
"value": 73
|
||||
"value": 82
|
||||
},
|
||||
"ram.sN.msgrate.msgs_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 216394,
|
||||
"floor": 229885,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1731152
|
||||
"value": 1839080
|
||||
},
|
||||
"ram.sN.query.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 340136,
|
||||
"floor": 337837,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1360544
|
||||
"value": 1351351
|
||||
},
|
||||
"ram.sN.query.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -578,9 +578,9 @@
|
|||
},
|
||||
"ram.sN.read.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 343878,
|
||||
"floor": 342935,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1375515
|
||||
"value": 1371742
|
||||
},
|
||||
"ram.sN.read.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -596,27 +596,27 @@
|
|||
},
|
||||
"ram.sN.seed.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 445235,
|
||||
"floor": 295159,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1780943
|
||||
"value": 1180637
|
||||
},
|
||||
"ram.sN.seed.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 0
|
||||
"value": 1
|
||||
},
|
||||
"ram.sN.seed.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1
|
||||
"value": 2
|
||||
},
|
||||
"ram.sN.write.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 286368,
|
||||
"floor": 289017,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1145475
|
||||
"value": 1156069
|
||||
},
|
||||
"ram.sN.write.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -629,5 +629,59 @@
|
|||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 2
|
||||
},
|
||||
"randread.collapse_x": {
|
||||
"dir": "lower",
|
||||
"floor": 1136,
|
||||
"tolerance_pct": 100,
|
||||
"value": 284
|
||||
},
|
||||
"randread.overcap.filled_rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 25024,
|
||||
"tolerance_pct": 100,
|
||||
"value": 6256
|
||||
},
|
||||
"randread.overcap.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1669,
|
||||
"tolerance_pct": 100,
|
||||
"value": 6676
|
||||
},
|
||||
"randread.overcap.read_p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 548,
|
||||
"tolerance_pct": 100,
|
||||
"value": 137
|
||||
},
|
||||
"randread.overcap.read_p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 1920,
|
||||
"tolerance_pct": 100,
|
||||
"value": 480
|
||||
},
|
||||
"randread.resident.filled_rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 54000,
|
||||
"tolerance_pct": 100,
|
||||
"value": 13500
|
||||
},
|
||||
"randread.resident.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 475556,
|
||||
"tolerance_pct": 100,
|
||||
"value": 1902225
|
||||
},
|
||||
"randread.resident.read_p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 0
|
||||
},
|
||||
"randread.resident.read_p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 1
|
||||
}
|
||||
}
|
||||
|
|
@ -467,6 +467,7 @@ fn usage() -> Int {
|
|||
print_err("usage: db-bench <mode>");
|
||||
print_err(" all N | seed N | read N | query N | write N | wal N");
|
||||
print_err(" mix N C | msgrate N | growth N int|text | growth-verify");
|
||||
print_err(" randread N R");
|
||||
print_err(" verify | verify-acked M");
|
||||
return 2;
|
||||
}
|
||||
|
|
@ -518,6 +519,48 @@ fn self_rss_kb() -> Int {
|
|||
-- intact: rows 1..M all present with the right v and no holes. M is whatever
|
||||
-- survived -- the claim under test is the SHAPE of the survivor, not its size,
|
||||
-- because a SIGKILL can land between any two inserts.
|
||||
-- databasev2 1: randread N R -- fill N rows, then read R of them by key in a
|
||||
-- Weyl-sequence order that spreads across the WHOLE range. Under a cap smaller
|
||||
-- than the table most of those reads must fault a page back in.
|
||||
--
|
||||
-- This is the leg the swap measurement was MISSING. `growth` inserts, and
|
||||
-- inserting is append-mostly: cold pages are written once and never re-read, so
|
||||
-- swap cost it ~1% (148s vs 150s uncapped). Random reads over an oversized
|
||||
-- table are the opposite access pattern -- and they are exactly what
|
||||
-- databasev2 2's `resident: keys` creates, since it reads rows back from a log
|
||||
-- larger than RAM. No RNG in the language and none needed: i*2654435761 mod n
|
||||
-- is a Weyl sequence, deterministic and spread, so the two legs read the SAME
|
||||
-- key order and only residency differs.
|
||||
fn randread_mode(n: Int, r: Int) -> Int {
|
||||
let bref = insert Bucket { tag: "randread" };
|
||||
let i = 1;
|
||||
while i <= n {
|
||||
insert Item { k: i, v: item_v(i), bucket: bref };
|
||||
i = i + 1;
|
||||
}
|
||||
print("randreadfilled ${n} ${self_rss_kb()}");
|
||||
let h: map<Int, Int> = {};
|
||||
let hits = 0;
|
||||
let t0 = time.ticks();
|
||||
let j = 0;
|
||||
while j < r {
|
||||
let key = 1 + (j * 2654435761) % n;
|
||||
let o0 = time.ticks();
|
||||
for row in from x in Item where x.k == key take 1 select x {
|
||||
if row.v == item_v(key) {
|
||||
hits = hits + 1;
|
||||
}
|
||||
}
|
||||
hist_add(h, time.ticks() - o0);
|
||||
j = j + 1;
|
||||
}
|
||||
let el = time.ticks() - t0;
|
||||
report("randread", r, el, h);
|
||||
-- hits proves the reads RESOLVED; a collapse measured over misses is noise
|
||||
print("randreadrss ${self_rss_kb()} ${hits}");
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn growth_verify() -> Int {
|
||||
let seen: map<Int, Int> = {};
|
||||
let maxk = 0;
|
||||
|
|
@ -643,6 +686,17 @@ fn main(args: multi Text) -> Int {
|
|||
}
|
||||
return growth_mode(n, args[2]);
|
||||
}
|
||||
if args[0] == "randread" {
|
||||
if len(args) < 3 {
|
||||
return usage();
|
||||
}
|
||||
let rr = parse_int(args[2]);
|
||||
if rr == nil or rr < 1 {
|
||||
print_err("db-bench: <r> must be a positive number");
|
||||
return 2;
|
||||
}
|
||||
return randread_mode(n, rr);
|
||||
}
|
||||
if args[0] == "mix" {
|
||||
if len(args) < 3 {
|
||||
return usage();
|
||||
|
|
|
|||
|
|
@ -126,9 +126,7 @@ is a real disk file (`/swap.img`; no zram, zswap disabled), so this is genuine
|
|||
disk paging.
|
||||
|
||||
**Do not generalise this to "swap is fine".** It measures an append-mostly
|
||||
workload. A random-read workload over a table larger than the cap is where the
|
||||
collapse should appear, and it is **not yet measured** — which matters, because
|
||||
that is exactly the access pattern databasev2 2's `resident: keys` creates.
|
||||
workload — and the opposite pattern was then measured too, below.
|
||||
|
||||
The operational consequence is that the RAM ceiling has two shapes and neither
|
||||
reports itself: without swap the process vanishes on signal 9, with swap it
|
||||
|
|
@ -153,3 +151,41 @@ scheduler's business, so `rows_recovered` carries ±100% tolerance. The leg
|
|||
asserts the exit but never records it as a metric, so that when databasev2 2's
|
||||
byte budget turns the kill into a checked refusal, the gate does not fail on the
|
||||
improvement.
|
||||
|
||||
### Random reads over an oversized table: 273×
|
||||
|
||||
60 000 Int rows, both legs reading the **same** Weyl key order
|
||||
(`i*2654435761 mod n`), differing only in the cap:
|
||||
|
||||
| Leg | Cap | Throughput | p50 | p99 |
|
||||
| --- | --- | --- | --- | --- |
|
||||
| all resident | 256 MiB | **1 851 166 reads/s** | 0 µs | **1 µs** |
|
||||
| over-cap, swap on | 6 MiB | **6 771 reads/s** | 128 µs | **487 µs** |
|
||||
|
||||
All 20 000 reads resolved in both legs, so this is the cost of faulting pages
|
||||
back, not of failed lookups. Swap-off is not an option in this configuration —
|
||||
it is SIGKILLed.
|
||||
|
||||
**The two access patterns are ~270× apart under identical memory pressure:**
|
||||
|
||||
| Pattern | Cost of exceeding RAM |
|
||||
| --- | --- |
|
||||
| append-mostly insert | **~1%** (cold pages written once, never re-read) |
|
||||
| random read across the table | **273×** |
|
||||
|
||||
**Departure is a step, not a curve.** 1 µs to 487 µs with nothing in between —
|
||||
`p99_departure_decile` looks for a gentle knee that does not exist. Residency is
|
||||
close to binary, which is why a budget must fire at a *declared* threshold: there
|
||||
is no early warning in the latency signal to react to.
|
||||
|
||||
**Mechanism caveat, and it is a design input for databasev2 2.** This is
|
||||
demand-paging of *anonymous slab memory* through swap — 4 KiB per fault, no
|
||||
readahead. `resident: keys` instead `pread`s rows from the WAL, through the
|
||||
**page cache**: same physical constraint, different mechanism, plausibly a better
|
||||
constant because file reads get readahead and a shared cache. **That is a
|
||||
hypothesis.** 273× bounds what *swapping* costs; iteration 2 must measure its own
|
||||
read path rather than inherit this figure.
|
||||
|
||||
Gated as `db-bench`'s `randread` leg, which gates the **ratio** — the absolute
|
||||
reads/sec of the over-cap half is the box's swap device, while the factor between
|
||||
two runs differing only in their cap is the engine's.
|
||||
|
|
|
|||
|
|
@ -75,8 +75,9 @@ Int-only `Item`; `growth N int|text` in the db-bench sample, reading its OWN
|
|||
`/proc/self/status` RSS at each decile because the driver's 250 ms poll misses
|
||||
the value *at* a boundary; `growth-verify`, which asserts the survivor of a
|
||||
crash is a contiguous intact prefix; and two harness legs — four footprint legs
|
||||
under a rootless cgroup v2 cap, and a `ceiling` leg that deliberately dies at
|
||||
the cap and then replays. 121 checks, 0 failures.
|
||||
under a rootless cgroup v2 cap, a `ceiling` leg that deliberately dies at the cap
|
||||
and then replays, and a `randread` leg that reads an oversized table randomly.
|
||||
133 checks, 0 failures.
|
||||
|
||||
**Key findings (measured, not asserted):** per-row footprint is **96.5–100 B**
|
||||
Int-only and **320.6–324 B** text-heavy — **3.3×**, not the "order of magnitude"
|
||||
|
|
@ -90,12 +91,17 @@ the opposite: `WO_HEAP_MB` is checked and traps). And swap is not "latency
|
|||
collapse": 900 000 rows inside a 64 MiB cap with swap finished in **148 s
|
||||
against 150 s uncapped** — ~1%, on a real disk swap file with no zram. Also
|
||||
measured: **ack-after-fsync holds through an OOM kill** — ~40 000 rows came back
|
||||
as an intact prefix, no holes, not read as corruption.
|
||||
as an intact prefix, no holes, not read as corruption. And the pattern the swap
|
||||
leg was missing: **random reads over an oversized table collapse 273×.**
|
||||
|
||||
**Learned:** an append-mostly workload never re-touches its cold pages, so swap
|
||||
costs it nothing — the collapse belongs to *random reads* over an oversized
|
||||
table, which is precisely the pattern iteration 2's `resident: keys` creates and
|
||||
is **still unmeasured**. The RAM ceiling therefore has two shapes and neither
|
||||
costs it nothing — and the opposite pattern was then measured on the same day.
|
||||
The `randread` leg reads randomly across a table larger than the cap, both legs
|
||||
walking the SAME Weyl key order so residency is the only variable: **273×
|
||||
throughput collapse** (1 851 166 → 6 771 reads/s), p99 **1 µs → 487 µs**, all
|
||||
20 000 reads resolving in both. So the two access patterns sit ~270× apart under
|
||||
identical memory pressure, and **departure is a step, not a curve** — which is
|
||||
why `p99_departure_decile` finds nothing: there is no knee to find. The RAM ceiling therefore has two shapes and neither
|
||||
announces itself: without swap the process vanishes on signal 9, with swap it
|
||||
keeps returning 0 while serving from disk. That is the argument for a budget
|
||||
that fires at a declared threshold instead of at exhaustion.
|
||||
|
|
@ -107,10 +113,13 @@ degradation to detect. Iteration 2 must pick its budget on other grounds rather
|
|||
than wait on a number this slice cannot produce. Iteration 3's replay baseline is
|
||||
still NOT delivered — `bench/baseline.json` times no replay.
|
||||
|
||||
**Next steps:** the read-heavy-over-cap leg is the single most valuable
|
||||
follow-up, and it is what makes `p99_departure_decile` mean anything (the
|
||||
footprint legs never approach their 512 MiB cap, so it is legitimately 0 today).
|
||||
Then iteration 2's 5c/5d.
|
||||
**Next steps:** iteration 2's 5c/5d. Its task 7 gained a criterion from this:
|
||||
`resident: keys` must measure its OWN read path rather than inherit 273×. That
|
||||
number bounds demand-paged anonymous memory through swap (4 KiB per fault, no
|
||||
readahead); `pread` through the page cache should beat it, and **the entire value
|
||||
of `resident: keys` rests on how much** — if it is not materially better than
|
||||
swapping, the design buys nothing the kernel was not already doing. Still absent:
|
||||
a replay baseline for iteration 3.
|
||||
|
||||
**`.dev/reference` used:** none. Sources were the kernel's own interfaces —
|
||||
cgroup v2 `memory.max`/`memory.swap.max`, `/proc/self/status`, `/proc/swaps` and
|
||||
|
|
@ -684,7 +693,7 @@ the language arc as v1 history.
|
|||
|
||||
| # | Iteration | State |
|
||||
| --- | --- | --- |
|
||||
| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | 🔄 **MEASURED 2026-08-27** — `readiness: ready`, `status: in-progress` (two criteria outstanding), forks settled, harness landed (121 checks). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Outstanding: the **random-read-over-cap** collapse (unmeasured, and it is `resident: keys`'s own access pattern) and iteration 3's replay baseline. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from |
|
||||
| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | 🔄 **MEASURED 2026-08-27** — `readiness: ready`, `status: in-progress` (iteration 3 replay baseline still undelivered), forks settled, harness landed (**133 checks**). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Also measured: **random reads over an oversized table collapse 273×** (1.85M vs 6 771 reads/s, p99 1 µs vs 487 µs) — so the two access patterns sit ~270× apart under the same pressure, and departure is a **step, not a curve**. Outstanding: iteration 3's replay baseline. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from |
|
||||
| 2 | [per-table storage: `durable` and `resident`](databasev2/02-table-storage-modes.md) | 🔄 **the language enrichment — the `durable` half is DONE and usable.** Two optional `@table` keys, `durable: true\|false` and `resident: all\|keys`, both defaulting to today's behaviour (all 28 existing declarations compile unchanged, no golden moved). Landed: the grammar, WO-E224 (a durable `ref` into a volatile table is refused), `.wob` v7 carrying both properties in spare `flags` bits, `durable: false` actually skipping the WAL (measured: 50 inserts → 1500 bytes durable, **0** volatile) with a mode-mismatch startup refusal, plus offset capture and read-a-row-from-an-offset. Outstanding: 5c/5d (the id→offset map and rewiring `wo_row_ptr`'s 11 call sites, slab scans and `@unique`/FK across the boundary — not yet written up), the two runtime refusals, and closeout. [spec](../superpowers/specs/2026-08-26-table-residency-design.md) · [plan](../superpowers/plans/2026-08-26-table-residency.md) |
|
||||
| 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ⬜ snapshot + truncate: disk reclaimed, replay bounded |
|
||||
| 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ⬜ **`readiness: ready` — the one startable iteration in the repo** (four forks confirmed settled 2026-08-20). Close the 66× gap iteration 22 measured (durable 4.5k vs ram 297k inserts/s) |
|
||||
|
|
|
|||
|
|
@ -81,9 +81,16 @@ touch — and with swap it **keeps returning 0 while serving from disk**, finish
|
|||
900 000 rows in 148 s against 150 s uncapped. Durability is the one thing that
|
||||
does hold: acked writes came back as an intact prefix across an OOM kill.
|
||||
|
||||
And when swap does absorb it, the price depends entirely on access pattern:
|
||||
inserting pays **~1%**, while reading randomly across the table pays **273×**
|
||||
(1 851 166 reads/s resident against 6 771 over-cap, p99 1 µs against 487 µs).
|
||||
That second number is the one this track must respect, because it is the access
|
||||
pattern [2](02-table-storage-modes.md)'s `resident: keys` creates by design.
|
||||
|
||||
That is why "back-pressure at exhaustion" is not a design option. Exhaustion
|
||||
either kills without warning or never arrives. Only a **declared threshold** can
|
||||
speak in time.
|
||||
either kills without warning or never arrives — and the latency signal offers no
|
||||
early warning either, since departure is a **step** (1 µs to 487 µs, nothing in
|
||||
between) rather than a curve. Only a **declared threshold** can speak in time.
|
||||
|
||||
## The lever: per-table storage modes
|
||||
|
||||
|
|
@ -133,7 +140,7 @@ before its mechanism existed; the history is in
|
|||
|
||||
| # | Iteration | Delivers | Needs |
|
||||
| --- | --- | --- | --- |
|
||||
| 1 | [RAM ceiling: measure the breaking point](01-ram-ceiling-measurement.md) | 🔄 **measured 2026-08-27**: footprint per shape (3.3× apart), the two silent exits (SIGKILL vs swap-serving-from-disk at ~uncapped speed), and ack-after-fsync surviving an OOM kill. Outstanding: the random-read-over-cap collapse, and a replay baseline | nothing; extends iteration 22's harness |
|
||||
| 1 | [RAM ceiling: measure the breaking point](01-ram-ceiling-measurement.md) | 🔄 **measured 2026-08-27**: footprint per shape (3.3× apart), the two silent exits (SIGKILL vs swap-serving-from-disk at ~uncapped speed), and ack-after-fsync surviving an OOM kill. Also measured: the **273× random-read collapse** over an oversized table. Outstanding: a replay baseline | nothing; extends iteration 22's harness |
|
||||
| 2 | [per-table storage](02-table-storage-modes.md) | the grammar: `durable: true\|false` and `resident: all\|keys`, per table, replacing the global `WO_DATA` all-or-nothing. **In progress — the `durable` half is done** | 1 for the budget default |
|
||||
| 3 | [WAL checkpoint](03-wal-checkpoint.md) *(was language 32)* | snapshot + truncate: disk reclaimed, replay bounded | 4 composes |
|
||||
| 4 | [io_uring group commit](04-io-uring-commit.md) *(was language 23)* | close the 66× durable/RAM write gap (4.5k vs 297k inserts/s) | the arc (landed) |
|
||||
|
|
|
|||
|
|
@ -97,10 +97,10 @@ are written out once and never read again, so paging is sequential and off the
|
|||
critical path. The swap is a real disk file (`/swap.img`, no zram, zswap
|
||||
disabled), so this is genuine disk paging, not compressed RAM.
|
||||
|
||||
**The correct generalisation is narrower than "swap is fine".** This measures an
|
||||
append-mostly workload. A workload that reads randomly across a table larger
|
||||
than the cap is the one that collapses, and this iteration did *not* measure
|
||||
that — see Outstanding.
|
||||
**The correct generalisation is narrower than "swap is fine", and the narrow
|
||||
claim was then measured too.** The 1% figure belongs to an append-mostly
|
||||
workload. Reading *randomly* across a table larger than the cap collapses
|
||||
**273×** — see below. Same cap, same swap, opposite access pattern.
|
||||
|
||||
## Progress
|
||||
|
||||
|
|
@ -113,11 +113,12 @@ that — see Outstanding.
|
|||
| rootless cap wrapper, swap on/off legs | ✅ 4 footprint legs |
|
||||
| footprint metric = **median of marginals**, doublings counted separately | ✅ |
|
||||
| `ceiling` leg: dies at the cap, then replay must be intact | ✅ gated |
|
||||
| baseline + tolerance policy | ✅ 121 checks; footprint at ±10%, kill-timing metrics at ±100% |
|
||||
| `randread` leg: control vs over-cap, same key order | ✅ gated |
|
||||
| baseline + tolerance policy | ✅ 133 checks; footprint at ±10%, kill-timing metrics at ±100% |
|
||||
| `perf-targets.md` §5 | ✅ |
|
||||
| **resident-footprint fraction for [iteration 2](02-table-storage-modes.md)** | ⬜ **not delivered — the premise it rested on is false**, see Outstanding |
|
||||
| **replay/restart baseline for [iteration 3](03-wal-checkpoint.md)** | ⬜ not delivered |
|
||||
| **the random-read-over-cap collapse** | ⬜ not measured |
|
||||
| `randread N R` + the `randread` leg — random reads over an oversized table | ✅ **273x collapse measured** |
|
||||
|
||||
## Measured
|
||||
|
||||
|
|
@ -143,6 +144,36 @@ The ceiling, 60 000 Int rows under an 8 MiB cap with swap off and `WO_DATA` set:
|
|||
**Ack-after-fsync holds through an OOM kill.** That is the one shutdown path
|
||||
which skips every cleanup handler, and the durable prefix came back whole.
|
||||
|
||||
Random reads over an oversized table — 60 000 rows, same Weyl key order in both
|
||||
legs, only the cap differs:
|
||||
|
||||
| Leg | Cap | Throughput | p50 | p99 | RSS after fill |
|
||||
| --- | --- | --- | --- | --- | --- |
|
||||
| control, all resident | 256 MiB | **1 851 166 reads/s** | 0 µs | **1 µs** | 13 508 KiB |
|
||||
| over-cap, swap on | 6 MiB | **6 771 reads/s** | 128 µs | **487 µs** | 6 980 KiB |
|
||||
|
||||
**273× throughput collapse, ~480× on p99.** All 20 000 reads resolved correctly
|
||||
in both legs, so this is the cost of faulting pages back in, not of failing
|
||||
lookups. Swap off is not an alternative here: that configuration is simply
|
||||
SIGKILLed.
|
||||
|
||||
So the two access patterns sit ~270× apart under identical memory pressure:
|
||||
|
||||
| Access pattern | Cost of exceeding RAM |
|
||||
| --- | --- |
|
||||
| append-mostly insert | **~1%** — cold pages written once, never re-read |
|
||||
| random read across the table | **273×** — almost every read faults |
|
||||
|
||||
**The mechanism caveat matters for [iteration 2](02-table-storage-modes.md).**
|
||||
This measures demand-paging of *anonymous slab memory* through swap: 4 KiB at a
|
||||
time, on fault, with no readahead. `resident: keys` will instead `pread` rows
|
||||
from the WAL, which goes through the **page cache** — the same physical
|
||||
constraint (data larger than RAM means disk I/O) but a different mechanism, and
|
||||
plausibly a better constant, because file reads get readahead and a shared cache
|
||||
while swap-in does not. **That is a hypothesis, not a result.** The honest
|
||||
reading is that 273× bounds what *swapping* costs, and iteration 2 must measure
|
||||
its own read path rather than inherit this number.
|
||||
|
||||
**The finding that matters most is the swap leg succeeding.** It did not fail,
|
||||
did not warn, and returned 0. A deployment in that state looks healthy while
|
||||
serving from disk. That is the exit with no error signal, and it is why
|
||||
|
|
@ -176,6 +207,10 @@ Met:
|
|||
the legs are skipped with a named reason and the rest still passes. ✅
|
||||
`cap_wrapper` returns None unless the `memory` controller is delegated; there
|
||||
is no uncapped fallback.
|
||||
- **Given** a table larger than the cap, **when** it is read randomly, **then**
|
||||
the degradation is quantified. ✅ **273× throughput, ~480× p99**, both legs
|
||||
reading the same key order with all reads resolving. This closes the gap the
|
||||
swap leg left, and it is the pattern `resident: keys` creates.
|
||||
|
||||
Outstanding:
|
||||
|
||||
|
|
@ -187,16 +222,12 @@ Outstanding:
|
|||
an explicit developer-declared figure) rather than waiting on a number this
|
||||
iteration cannot produce. This is the most important thing this slice learned
|
||||
and it removes a dependency rather than satisfying it.
|
||||
- **The random-read-over-cap collapse.** Not measured. This is where the "latency
|
||||
collapse" prediction may still be true, and it is the workload that matters
|
||||
for [2](02-table-storage-modes.md)'s `resident: keys`, whose whole premise is
|
||||
reading rows back from a log larger than RAM. Needs a read-heavy leg over a
|
||||
table exceeding the cap. **The single most valuable follow-up.**
|
||||
- **Given** rising fractions of the cap, **when** latency is sampled, **then**
|
||||
the p99 departure point is recorded. Partially: the sampler and metric exist
|
||||
and are gated, but the footprint legs never approach their 512 MiB cap, so
|
||||
`p99_departure_decile` is legitimately 0 and proves nothing. It becomes
|
||||
meaningful only with the read-heavy leg above.
|
||||
the p99 departure point is recorded. Partially, and now with a real answer
|
||||
elsewhere: `p99_departure_decile` stays 0 because the footprint legs never
|
||||
approach their 512 MiB cap, but the departure itself is measured by the
|
||||
`randread` leg as a **step, not a curve** — 1 µs resident, 487 µs over-cap.
|
||||
There is no gentle departure to find; residency is close to binary.
|
||||
- **A replay/restart baseline for iteration 3.** Not delivered; `growth` exercises
|
||||
`WO_DATA` but nothing times replay. Cheap to add, still absent from
|
||||
`bench/baseline.json`.
|
||||
|
|
@ -237,7 +268,12 @@ Outstanding:
|
|||
5. **Kill-timing metrics carry ±100% tolerance.** `rows_recovered` depends on
|
||||
where the SIGKILL landed; gating it tightly would be gating the scheduler.
|
||||
The invariant asserted instead is the *shape* of the survivor.
|
||||
6. **The ceiling leg asserts `rc`, never records it.** When iteration 2's byte
|
||||
6. **`randread` gates the RATIO, not the absolutes.** The over-cap half is swap
|
||||
I/O, so its reads/sec belongs to the box; the collapse factor between two
|
||||
runs that differ only in their cap belongs to the engine. Both legs read the
|
||||
same Weyl key order (`i*2654435761 mod n` — no RNG in the language, and none
|
||||
needed) so residency is the only variable.
|
||||
7. **The ceiling leg asserts `rc`, never records it.** When iteration 2's byte
|
||||
budget lands, death should become a checked refusal — the gate must not fail
|
||||
on that improvement.
|
||||
|
||||
|
|
|
|||
|
|
@ -118,6 +118,14 @@ Outstanding:
|
|||
**then** a refusal naming the table and the annotation. *(task 6)*
|
||||
- **Given** the `resident: all` read baseline, **when** re-measured, **then**
|
||||
inside tolerance — no cost for a feature not used. *(task 7)*
|
||||
- **Given** a `resident: keys` table larger than RAM, **when** read randomly,
|
||||
**then** its read cost is **measured against the resident baseline on its own
|
||||
read path**, not inherited from databasev2 1's swap figure. *(task 7)* — that
|
||||
figure is **273×** for demand-paged anonymous memory
|
||||
([1](01-ram-ceiling-measurement.md)); `pread` through the page cache should do
|
||||
better, and the whole value of `resident: keys` rests on how much better. If it
|
||||
is not materially better than swapping, the design buys nothing that the
|
||||
kernel was not already doing.
|
||||
|
||||
## Out Of Scope
|
||||
|
||||
|
|
|
|||
|
|
@ -61,6 +61,21 @@ stay resident, rows do not.** It buys roughly two orders of magnitude of table
|
|||
size, not infinity, and the spec says so plainly because a design sold as
|
||||
unlimited gets deployed as if it were.
|
||||
|
||||
**What the trade costs, bounded by measurement (databasev2 1, 2026-08-27).**
|
||||
Buying table size with disk reads is not free, and the price is large: random
|
||||
reads across a table larger than RAM measured **273× slower** than resident ones
|
||||
(1 851 166 reads/s against 6 771; p99 1 µs against 487 µs), and the transition is
|
||||
a **step, not a curve** — there is no gentle region to operate in. That figure is
|
||||
an *upper bound on the mechanism this spec does not use*: it is demand-paging of
|
||||
anonymous memory through swap, 4 KiB per fault with no readahead, whereas
|
||||
`resident: keys` `pread`s from the WAL through the page cache, which gets
|
||||
readahead and a shared cache. The constant should therefore be better — **but
|
||||
that is a hypothesis and task 7 must measure it, not inherit it.** Two things
|
||||
follow regardless: `resident: keys` must stay opt-in per table (it is), and the
|
||||
hot-set question is not deferrable decoration — it is
|
||||
[iteration 5](../../stories/databasev2/05-bounded-tables-eviction.md) and it
|
||||
decides whether this design is usable for anything read-heavy.
|
||||
|
||||
## The design
|
||||
|
||||
### Grammar
|
||||
|
|
|
|||
|
|
@ -233,6 +233,7 @@ def tolerance_for(key):
|
|||
if ".bytes_per_row" in key: return 10
|
||||
if key.startswith("growth."): return 100
|
||||
if key.startswith("ceiling."): return 100
|
||||
if key.startswith("randread."): return 100
|
||||
if ".mixread." in key or ".mixwrite." in key: return 50
|
||||
if ".sN." in key: return 50
|
||||
if ".read." in key or ".query." in key: return 50
|
||||
|
|
@ -374,6 +375,10 @@ def growth(metrics):
|
|||
|
||||
|
||||
CEIL_N, CEIL_CAP_MB = 60000, 8
|
||||
RAND_N = 60000 if QUICK else 200000
|
||||
RAND_R = 20000 if QUICK else 40000
|
||||
RAND_CAP_MB = 6 if QUICK else 14 # over-cap: holds roughly a third of the rows
|
||||
RAND_FIT_MB = 256 # control: same mechanism, cap simply does not bind
|
||||
|
||||
def ceiling(metrics):
|
||||
"""The ceiling itself, and the durability claim across it.
|
||||
|
|
@ -433,6 +438,81 @@ def ceiling(metrics):
|
|||
shutil.rmtree(data, ignore_errors=True)
|
||||
|
||||
|
||||
|
||||
def parse_randread(lines):
|
||||
"""ops/sec, p50, p99, resolved-read count and post-fill RSS from the
|
||||
sample's own randread lines. `randreadfilled` also starts with "randread",
|
||||
so match f[0] exactly, not by prefix."""
|
||||
ops = p50 = p99 = hits = filled = None
|
||||
for l in lines:
|
||||
f = l.split()
|
||||
if not f:
|
||||
continue
|
||||
if f[0] == "randread" and len(f) == 5:
|
||||
ops, p50, p99 = int(f[2]), int(f[3]), int(f[4])
|
||||
elif f[0] == "randreadrss" and len(f) == 3:
|
||||
hits = int(f[2])
|
||||
elif f[0] == "randreadfilled" and len(f) == 3:
|
||||
filled = int(f[2])
|
||||
return ops, p50, p99, hits, filled
|
||||
|
||||
|
||||
def randread(metrics):
|
||||
"""Random reads over a table LARGER than the memory cap -- the access
|
||||
pattern the swap measurement was missing.
|
||||
|
||||
growth() only inserts, and inserting is append-mostly: cold pages are
|
||||
written once and never re-read, so swap cost it ~1% (148s vs 150s
|
||||
uncapped). That result is real but does NOT generalise to "swap is fine".
|
||||
This leg reads back across the whole range in a Weyl-sequence order, so
|
||||
most reads must fault a page in.
|
||||
|
||||
It matters because it is databasev2 2's `resident: keys` access pattern:
|
||||
that design reads rows back from a log larger than RAM by construction.
|
||||
|
||||
Two runs, identical except for the cap, reading the SAME key order:
|
||||
- control (RAND_FIT_MB): cap does not bind, everything resident
|
||||
- over-cap (RAND_CAP_MB): ~a third of the rows fit; swap ON, because
|
||||
with swap off this configuration is simply SIGKILLed (see ceiling())
|
||||
The headline is collapse_x, the throughput ratio between them. Tolerances
|
||||
are wide: the over-cap half is swap I/O, so its absolute numbers are the
|
||||
box's, while the RATIO is the property of the engine."""
|
||||
if cap_wrapper(RAND_FIT_MB, 0) is None:
|
||||
ok("randread: SKIPPED -- no rootless cgroup v2 memory cap on this host")
|
||||
return
|
||||
res = {}
|
||||
for legname, cap_mb, swap_mb in (("resident", RAND_FIT_MB, 0),
|
||||
("overcap", RAND_CAP_MB, 256)):
|
||||
w = cap_wrapper(cap_mb, swap_mb)
|
||||
pr = subprocess.run(w + [BIN, "randread", str(RAND_N), str(RAND_R)],
|
||||
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
|
||||
text=True, env=dict(os.environ), timeout=900)
|
||||
lines = pr.stdout.splitlines()
|
||||
ops, p50, p99, hits, filled = parse_randread(lines)
|
||||
key = f"randread.{legname}"
|
||||
if pr.returncode != 0 or ops is None:
|
||||
bad(f"{key}: run failed", f"rc={pr.returncode} {(lines[-1:] or ['no output'])[0][:120]}")
|
||||
return
|
||||
if hits != RAND_R:
|
||||
# a collapse measured over reads that did not resolve is noise
|
||||
bad(f"{key}: only {hits}/{RAND_R} reads resolved", "keys must all exist")
|
||||
return
|
||||
metrics[f"{key}.ops_sec"] = ops
|
||||
metrics[f"{key}.read_p50us"] = p50
|
||||
metrics[f"{key}.read_p99us"] = p99
|
||||
metrics[f"{key}.filled_rss_kb"] = filled
|
||||
res[legname] = ops
|
||||
ok(f"{key}: {ops} reads/sec, p50 {p50}us p99 {p99}us, {filled} KiB after fill")
|
||||
collapse = res["resident"] // max(res["overcap"], 1)
|
||||
metrics["randread.collapse_x"] = collapse
|
||||
if collapse < 2:
|
||||
bad("randread: NO collapse -- the cap did not bind",
|
||||
f"{RAND_N} rows fit under {RAND_CAP_MB} MiB, resize the leg")
|
||||
else:
|
||||
ok(f"randread: random reads over an oversized table collapse {collapse}x "
|
||||
f"({res['resident']} -> {res['overcap']} reads/sec)")
|
||||
|
||||
|
||||
def main():
|
||||
# --check <results.json>: gate-only evaluation of a recorded run — the
|
||||
# gate-bites smoke doctors a copy and this mode must FAIL on it
|
||||
|
|
@ -447,6 +527,7 @@ def main():
|
|||
durability(metrics)
|
||||
growth(metrics)
|
||||
ceiling(metrics)
|
||||
randread(metrics)
|
||||
os.makedirs(RESULTS_DIR, exist_ok=True)
|
||||
stamp = time.strftime("%Y%m%d-%H%M%S")
|
||||
out = os.path.join(RESULTS_DIR, f"run-{stamp}{'-quick' if QUICK else ''}.json")
|
||||
|
|
|
|||
Loading…
Reference in a new issue