feat(db-bench): random-read-over-cap leg — the 273x collapse

- `randread N R` in the sample: fill N rows, read R across the WHOLE range
- Weyl order `i*2654435761 mod n` — no RNG in the language, none needed;
  both legs read the SAME key order so residency is the only variable
- `randread` driver leg: control (256 MiB, does not bind) vs over-cap
  (6 MiB + swap), sizes kept modest — quick resolves it in ~5s
- gates the RATIO, not the absolutes: over-cap reads/sec belongs to the
  box's swap device, the factor between two runs belongs to the engine
- reads must all resolve (hits == R) or the leg fails; a collapse measured
  over unresolved reads is noise
- 133 checks, 0 failures; gate bites on a doctored collapse_x

Measured — this closes the gap the swap leg left:

- resident 1 851 166 reads/sec, p50 0us p99 1us
- over-cap    6 771 reads/sec, p50 128us p99 487us
- 273x throughput, ~480x p99, all 20 000 reads resolving in both
- so the two access patterns sit ~270x apart under identical pressure:
  append-mostly insert ~1%, random read 273x
- departure is a STEP not a curve (1us -> 487us, nothing between), which
  is why p99_departure_decile finds no knee — there is none

- caveat recorded, NOT inherited: this is demand-paged anonymous memory
  through swap (4 KiB/fault, no readahead). `resident: keys` preads via
  the page cache — should be better, but databasev2 2 task 7 must measure
  its own read path. New criterion added there

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
shoney.arickathil 2026-08-27 20:56:19 +02:00
parent 0c9b2c45d8
commit a873cf7331
9 changed files with 437 additions and 137 deletions

View file

@ -8,15 +8,15 @@
},
"ceiling.rows_recovered": {
"dir": "lower",
"floor": 159744,
"floor": 159828,
"tolerance_pct": 100,
"value": 39936
"value": 39957
},
"durable.s1.mixread.ops_sec": {
"dir": "higher",
"floor": 2239,
"tolerance_pct": 50,
"value": 8958
"value": 8957
},
"durable.s1.mixread.p50us": {
"dir": "lower",
@ -44,15 +44,15 @@
},
"durable.s1.mixwrite.p99us": {
"dir": "lower",
"floor": 872,
"floor": 868,
"tolerance_pct": 50,
"value": 218
"value": 217
},
"durable.s1.query.ops_sec": {
"dir": "higher",
"floor": 202429,
"floor": 215517,
"tolerance_pct": 50,
"value": 809716
"value": 862068
},
"durable.s1.query.p50us": {
"dir": "lower",
@ -64,13 +64,13 @@
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 2
"value": 1
},
"durable.s1.read.ops_sec": {
"dir": "higher",
"floor": 215703,
"floor": 221827,
"tolerance_pct": 50,
"value": 862812
"value": 887311
},
"durable.s1.read.p50us": {
"dir": "lower",
@ -86,45 +86,45 @@
},
"durable.s1.seed.ops_sec": {
"dir": "higher",
"floor": 1078,
"floor": 1128,
"tolerance_pct": 15,
"value": 4315
"value": 4515
},
"durable.s1.seed.p50us": {
"dir": "lower",
"floor": 828,
"floor": 832,
"tolerance_pct": 15,
"value": 207
"value": 208
},
"durable.s1.seed.p99us": {
"dir": "lower",
"floor": 2160,
"floor": 2180,
"tolerance_pct": 15,
"value": 540
"value": 545
},
"durable.s1.write.ops_sec": {
"dir": "higher",
"floor": 1156,
"floor": 1141,
"tolerance_pct": 15,
"value": 4626
"value": 4565
},
"durable.s1.write.p50us": {
"dir": "lower",
"floor": 828,
"floor": 832,
"tolerance_pct": 15,
"value": 207
"value": 208
},
"durable.s1.write.p99us": {
"dir": "lower",
"floor": 1844,
"floor": 2256,
"tolerance_pct": 15,
"value": 461
"value": 564
},
"durable.sN.mixread.ops_sec": {
"dir": "higher",
"floor": 1113,
"floor": 1117,
"tolerance_pct": 50,
"value": 4452
"value": 4468
},
"durable.sN.mixread.p50us": {
"dir": "lower",
@ -134,33 +134,33 @@
},
"durable.sN.mixread.p99us": {
"dir": "lower",
"floor": 15116,
"floor": 19900,
"tolerance_pct": 50,
"value": 3779
"value": 4975
},
"durable.sN.mixwrite.ops_sec": {
"dir": "higher",
"floor": 123,
"floor": 124,
"tolerance_pct": 50,
"value": 494
"value": 496
},
"durable.sN.mixwrite.p50us": {
"dir": "lower",
"floor": 1148,
"floor": 1104,
"tolerance_pct": 50,
"value": 287
"value": 276
},
"durable.sN.mixwrite.p99us": {
"dir": "lower",
"floor": 2932,
"floor": 1928,
"tolerance_pct": 50,
"value": 733
"value": 482
},
"durable.sN.query.ops_sec": {
"dir": "higher",
"floor": 333333,
"floor": 331125,
"tolerance_pct": 50,
"value": 1333333
"value": 1324503
},
"durable.sN.query.p50us": {
"dir": "lower",
@ -176,9 +176,9 @@
},
"durable.sN.read.ops_sec": {
"dir": "higher",
"floor": 298329,
"floor": 341064,
"tolerance_pct": 50,
"value": 1193317
"value": 1364256
},
"durable.sN.read.p50us": {
"dir": "lower",
@ -190,13 +190,13 @@
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 2
"value": 1
},
"durable.sN.seed.ops_sec": {
"dir": "higher",
"floor": 1139,
"floor": 1125,
"tolerance_pct": 50,
"value": 4556
"value": 4502
},
"durable.sN.seed.p50us": {
"dir": "lower",
@ -206,27 +206,27 @@
},
"durable.sN.seed.p99us": {
"dir": "lower",
"floor": 2000,
"floor": 2624,
"tolerance_pct": 50,
"value": 500
"value": 656
},
"durable.sN.write.ops_sec": {
"dir": "higher",
"floor": 1153,
"floor": 1150,
"tolerance_pct": 50,
"value": 4614
"value": 4600
},
"durable.sN.write.p50us": {
"dir": "lower",
"floor": 828,
"floor": 832,
"tolerance_pct": 50,
"value": 207
"value": 208
},
"durable.sN.write.p99us": {
"dir": "lower",
"floor": 2172,
"floor": 2348,
"tolerance_pct": 50,
"value": 543
"value": 587
},
"growth.available": {
"dir": "lower",
@ -236,9 +236,9 @@
},
"growth.int.noswap.bytes_per_row": {
"dir": "lower",
"floor": 400,
"floor": 392,
"tolerance_pct": 10,
"value": 100
"value": 98
},
"growth.int.noswap.doublings": {
"dir": "lower",
@ -272,15 +272,15 @@
},
"growth.int.noswap.rss_kb": {
"dir": "lower",
"floor": 23936,
"floor": 23920,
"tolerance_pct": 100,
"value": 5984
"value": 5980
},
"growth.int.swap.bytes_per_row": {
"dir": "lower",
"floor": 400,
"floor": 392,
"tolerance_pct": 10,
"value": 100
"value": 98
},
"growth.int.swap.doublings": {
"dir": "lower",
@ -314,15 +314,15 @@
},
"growth.int.swap.rss_kb": {
"dir": "lower",
"floor": 23952,
"floor": 23920,
"tolerance_pct": 100,
"value": 5988
"value": 5980
},
"growth.text.noswap.bytes_per_row": {
"dir": "lower",
"floor": 1296,
"floor": 1288,
"tolerance_pct": 10,
"value": 324
"value": 322
},
"growth.text.noswap.doublings": {
"dir": "lower",
@ -356,15 +356,15 @@
},
"growth.text.noswap.rss_kb": {
"dir": "lower",
"floor": 41216,
"floor": 41184,
"tolerance_pct": 100,
"value": 10304
"value": 10296
},
"growth.text.swap.bytes_per_row": {
"dir": "lower",
"floor": 1296,
"floor": 1288,
"tolerance_pct": 10,
"value": 324
"value": 322
},
"growth.text.swap.doublings": {
"dir": "lower",
@ -398,15 +398,15 @@
},
"growth.text.swap.rss_kb": {
"dir": "lower",
"floor": 41216,
"floor": 41200,
"tolerance_pct": 100,
"value": 10304
"value": 10300
},
"ram.s1.mixread.ops_sec": {
"dir": "higher",
"floor": 2239,
"floor": 2236,
"tolerance_pct": 50,
"value": 8956
"value": 8945
},
"ram.s1.mixread.p50us": {
"dir": "lower",
@ -424,13 +424,13 @@
"dir": "higher",
"floor": 248,
"tolerance_pct": 50,
"value": 995
"value": 993
},
"ram.s1.mixwrite.p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 0
"value": 1
},
"ram.s1.mixwrite.p99us": {
"dir": "lower",
@ -440,15 +440,15 @@
},
"ram.s1.msgrate.msgs_sec": {
"dir": "higher",
"floor": 419674,
"floor": 392834,
"tolerance_pct": 15,
"value": 3357394
"value": 3142677
},
"ram.s1.query.ops_sec": {
"dir": "higher",
"floor": 340136,
"floor": 214592,
"tolerance_pct": 50,
"value": 1360544
"value": 858369
},
"ram.s1.query.p50us": {
"dir": "lower",
@ -460,13 +460,13 @@
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 1
"value": 2
},
"ram.s1.read.ops_sec": {
"dir": "higher",
"floor": 369276,
"floor": 226860,
"tolerance_pct": 50,
"value": 1477104
"value": 907441
},
"ram.s1.read.p50us": {
"dir": "lower",
@ -478,31 +478,31 @@
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 1
"value": 2
},
"ram.s1.seed.ops_sec": {
"dir": "higher",
"floor": 470366,
"floor": 273373,
"tolerance_pct": 15,
"value": 1881467
"value": 1093493
},
"ram.s1.seed.p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 15,
"value": 0
"value": 1
},
"ram.s1.seed.p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 15,
"value": 1
"value": 2
},
"ram.s1.write.ops_sec": {
"dir": "higher",
"floor": 294464,
"floor": 159134,
"tolerance_pct": 15,
"value": 1177856
"value": 636537
},
"ram.s1.write.p50us": {
"dir": "lower",
@ -514,25 +514,25 @@
"dir": "lower",
"floor": 100,
"tolerance_pct": 15,
"value": 1
"value": 3
},
"ram.sN.mixread.ops_sec": {
"dir": "higher",
"floor": 2240,
"floor": 2239,
"tolerance_pct": 50,
"value": 8960
"value": 8958
},
"ram.sN.mixread.p50us": {
"dir": "lower",
"floor": 228,
"floor": 232,
"tolerance_pct": 50,
"value": 57
"value": 58
},
"ram.sN.mixread.p99us": {
"dir": "lower",
"floor": 1404,
"floor": 1032,
"tolerance_pct": 50,
"value": 351
"value": 258
},
"ram.sN.mixwrite.ops_sec": {
"dir": "higher",
@ -542,27 +542,27 @@
},
"ram.sN.mixwrite.p50us": {
"dir": "lower",
"floor": 248,
"floor": 256,
"tolerance_pct": 50,
"value": 62
"value": 64
},
"ram.sN.mixwrite.p99us": {
"dir": "lower",
"floor": 292,
"floor": 328,
"tolerance_pct": 50,
"value": 73
"value": 82
},
"ram.sN.msgrate.msgs_sec": {
"dir": "higher",
"floor": 216394,
"floor": 229885,
"tolerance_pct": 50,
"value": 1731152
"value": 1839080
},
"ram.sN.query.ops_sec": {
"dir": "higher",
"floor": 340136,
"floor": 337837,
"tolerance_pct": 50,
"value": 1360544
"value": 1351351
},
"ram.sN.query.p50us": {
"dir": "lower",
@ -578,9 +578,9 @@
},
"ram.sN.read.ops_sec": {
"dir": "higher",
"floor": 343878,
"floor": 342935,
"tolerance_pct": 50,
"value": 1375515
"value": 1371742
},
"ram.sN.read.p50us": {
"dir": "lower",
@ -596,27 +596,27 @@
},
"ram.sN.seed.ops_sec": {
"dir": "higher",
"floor": 445235,
"floor": 295159,
"tolerance_pct": 50,
"value": 1780943
"value": 1180637
},
"ram.sN.seed.p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 0
"value": 1
},
"ram.sN.seed.p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 1
"value": 2
},
"ram.sN.write.ops_sec": {
"dir": "higher",
"floor": 286368,
"floor": 289017,
"tolerance_pct": 50,
"value": 1145475
"value": 1156069
},
"ram.sN.write.p50us": {
"dir": "lower",
@ -629,5 +629,59 @@
"floor": 100,
"tolerance_pct": 50,
"value": 2
},
"randread.collapse_x": {
"dir": "lower",
"floor": 1136,
"tolerance_pct": 100,
"value": 284
},
"randread.overcap.filled_rss_kb": {
"dir": "lower",
"floor": 25024,
"tolerance_pct": 100,
"value": 6256
},
"randread.overcap.ops_sec": {
"dir": "higher",
"floor": 1669,
"tolerance_pct": 100,
"value": 6676
},
"randread.overcap.read_p50us": {
"dir": "lower",
"floor": 548,
"tolerance_pct": 100,
"value": 137
},
"randread.overcap.read_p99us": {
"dir": "lower",
"floor": 1920,
"tolerance_pct": 100,
"value": 480
},
"randread.resident.filled_rss_kb": {
"dir": "lower",
"floor": 54000,
"tolerance_pct": 100,
"value": 13500
},
"randread.resident.ops_sec": {
"dir": "higher",
"floor": 475556,
"tolerance_pct": 100,
"value": 1902225
},
"randread.resident.read_p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
},
"randread.resident.read_p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 1
}
}

View file

@ -467,6 +467,7 @@ fn usage() -> Int {
print_err("usage: db-bench <mode>");
print_err(" all N | seed N | read N | query N | write N | wal N");
print_err(" mix N C | msgrate N | growth N int|text | growth-verify");
print_err(" randread N R");
print_err(" verify | verify-acked M");
return 2;
}
@ -518,6 +519,48 @@ fn self_rss_kb() -> Int {
-- intact: rows 1..M all present with the right v and no holes. M is whatever
-- survived -- the claim under test is the SHAPE of the survivor, not its size,
-- because a SIGKILL can land between any two inserts.
-- databasev2 1: randread N R -- fill N rows, then read R of them by key in a
-- Weyl-sequence order that spreads across the WHOLE range. Under a cap smaller
-- than the table most of those reads must fault a page back in.
--
-- This is the leg the swap measurement was MISSING. `growth` inserts, and
-- inserting is append-mostly: cold pages are written once and never re-read, so
-- swap cost it ~1% (148s vs 150s uncapped). Random reads over an oversized
-- table are the opposite access pattern -- and they are exactly what
-- databasev2 2's `resident: keys` creates, since it reads rows back from a log
-- larger than RAM. No RNG in the language and none needed: i*2654435761 mod n
-- is a Weyl sequence, deterministic and spread, so the two legs read the SAME
-- key order and only residency differs.
fn randread_mode(n: Int, r: Int) -> Int {
let bref = insert Bucket { tag: "randread" };
let i = 1;
while i <= n {
insert Item { k: i, v: item_v(i), bucket: bref };
i = i + 1;
}
print("randreadfilled ${n} ${self_rss_kb()}");
let h: map<Int, Int> = {};
let hits = 0;
let t0 = time.ticks();
let j = 0;
while j < r {
let key = 1 + (j * 2654435761) % n;
let o0 = time.ticks();
for row in from x in Item where x.k == key take 1 select x {
if row.v == item_v(key) {
hits = hits + 1;
}
}
hist_add(h, time.ticks() - o0);
j = j + 1;
}
let el = time.ticks() - t0;
report("randread", r, el, h);
-- hits proves the reads RESOLVED; a collapse measured over misses is noise
print("randreadrss ${self_rss_kb()} ${hits}");
return 0;
}
fn growth_verify() -> Int {
let seen: map<Int, Int> = {};
let maxk = 0;
@ -643,6 +686,17 @@ fn main(args: multi Text) -> Int {
}
return growth_mode(n, args[2]);
}
if args[0] == "randread" {
if len(args) < 3 {
return usage();
}
let rr = parse_int(args[2]);
if rr == nil or rr < 1 {
print_err("db-bench: <r> must be a positive number");
return 2;
}
return randread_mode(n, rr);
}
if args[0] == "mix" {
if len(args) < 3 {
return usage();

View file

@ -126,9 +126,7 @@ is a real disk file (`/swap.img`; no zram, zswap disabled), so this is genuine
disk paging.
**Do not generalise this to "swap is fine".** It measures an append-mostly
workload. A random-read workload over a table larger than the cap is where the
collapse should appear, and it is **not yet measured** — which matters, because
that is exactly the access pattern databasev2 2's `resident: keys` creates.
workload — and the opposite pattern was then measured too, below.
The operational consequence is that the RAM ceiling has two shapes and neither
reports itself: without swap the process vanishes on signal 9, with swap it
@ -153,3 +151,41 @@ scheduler's business, so `rows_recovered` carries ±100% tolerance. The leg
asserts the exit but never records it as a metric, so that when databasev2 2's
byte budget turns the kill into a checked refusal, the gate does not fail on the
improvement.
### Random reads over an oversized table: 273×
60 000 Int rows, both legs reading the **same** Weyl key order
(`i*2654435761 mod n`), differing only in the cap:
| Leg | Cap | Throughput | p50 | p99 |
| --- | --- | --- | --- | --- |
| all resident | 256 MiB | **1 851 166 reads/s** | 0 µs | **1 µs** |
| over-cap, swap on | 6 MiB | **6 771 reads/s** | 128 µs | **487 µs** |
All 20 000 reads resolved in both legs, so this is the cost of faulting pages
back, not of failed lookups. Swap-off is not an option in this configuration —
it is SIGKILLed.
**The two access patterns are ~270× apart under identical memory pressure:**
| Pattern | Cost of exceeding RAM |
| --- | --- |
| append-mostly insert | **~1%** (cold pages written once, never re-read) |
| random read across the table | **273×** |
**Departure is a step, not a curve.** 1 µs to 487 µs with nothing in between —
`p99_departure_decile` looks for a gentle knee that does not exist. Residency is
close to binary, which is why a budget must fire at a *declared* threshold: there
is no early warning in the latency signal to react to.
**Mechanism caveat, and it is a design input for databasev2 2.** This is
demand-paging of *anonymous slab memory* through swap — 4 KiB per fault, no
readahead. `resident: keys` instead `pread`s rows from the WAL, through the
**page cache**: same physical constraint, different mechanism, plausibly a better
constant because file reads get readahead and a shared cache. **That is a
hypothesis.** 273× bounds what *swapping* costs; iteration 2 must measure its own
read path rather than inherit this figure.
Gated as `db-bench`'s `randread` leg, which gates the **ratio** — the absolute
reads/sec of the over-cap half is the box's swap device, while the factor between
two runs differing only in their cap is the engine's.

View file

@ -75,8 +75,9 @@ Int-only `Item`; `growth N int|text` in the db-bench sample, reading its OWN
`/proc/self/status` RSS at each decile because the driver's 250 ms poll misses
the value *at* a boundary; `growth-verify`, which asserts the survivor of a
crash is a contiguous intact prefix; and two harness legs — four footprint legs
under a rootless cgroup v2 cap, and a `ceiling` leg that deliberately dies at
the cap and then replays. 121 checks, 0 failures.
under a rootless cgroup v2 cap, a `ceiling` leg that deliberately dies at the cap
and then replays, and a `randread` leg that reads an oversized table randomly.
133 checks, 0 failures.
**Key findings (measured, not asserted):** per-row footprint is **96.5–100 B**
Int-only and **320.6–324 B** text-heavy — **3.3×**, not the "order of magnitude"
@ -90,12 +91,17 @@ the opposite: `WO_HEAP_MB` is checked and traps). And swap is not "latency
collapse": 900 000 rows inside a 64 MiB cap with swap finished in **148 s
against 150 s uncapped** — ~1%, on a real disk swap file with no zram. Also
measured: **ack-after-fsync holds through an OOM kill** — ~40 000 rows came back
as an intact prefix, no holes, not read as corruption.
as an intact prefix, no holes, not read as corruption. And the pattern the swap
leg was missing: **random reads over an oversized table collapse 273×.**
**Learned:** an append-mostly workload never re-touches its cold pages, so swap
costs it nothing — the collapse belongs to *random reads* over an oversized
table, which is precisely the pattern iteration 2's `resident: keys` creates and
is **still unmeasured**. The RAM ceiling therefore has two shapes and neither
costs it nothing — and the opposite pattern was then measured on the same day.
The `randread` leg reads randomly across a table larger than the cap, both legs
walking the SAME Weyl key order so residency is the only variable: **273×
throughput collapse** (1 851 166 → 6 771 reads/s), p99 **1 µs → 487 µs**, all
20 000 reads resolving in both. So the two access patterns sit ~270× apart under
identical memory pressure, and **departure is a step, not a curve** — which is
why `p99_departure_decile` finds nothing: there is no knee to find. The RAM ceiling therefore has two shapes and neither
announces itself: without swap the process vanishes on signal 9, with swap it
keeps returning 0 while serving from disk. That is the argument for a budget
that fires at a declared threshold instead of at exhaustion.
@ -107,10 +113,13 @@ degradation to detect. Iteration 2 must pick its budget on other grounds rather
than wait on a number this slice cannot produce. Iteration 3's replay baseline is
still NOT delivered — `bench/baseline.json` times no replay.
**Next steps:** the read-heavy-over-cap leg is the single most valuable
follow-up, and it is what makes `p99_departure_decile` mean anything (the
footprint legs never approach their 512 MiB cap, so it is legitimately 0 today).
Then iteration 2's 5c/5d.
**Next steps:** iteration 2's 5c/5d. Its task 7 gained a criterion from this:
`resident: keys` must measure its OWN read path rather than inherit 273×. That
number bounds demand-paged anonymous memory through swap (4 KiB per fault, no
readahead); `pread` through the page cache should beat it, and **the entire value
of `resident: keys` rests on how much** — if it is not materially better than
swapping, the design buys nothing the kernel was not already doing. Still absent:
a replay baseline for iteration 3.
**`.dev/reference` used:** none. Sources were the kernel's own interfaces —
cgroup v2 `memory.max`/`memory.swap.max`, `/proc/self/status`, `/proc/swaps` and
@ -684,7 +693,7 @@ the language arc as v1 history.
| # | Iteration | State |
| --- | --- | --- |
| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | 🔄 **MEASURED 2026-08-27** — `readiness: ready`, `status: in-progress` (two criteria outstanding), forks settled, harness landed (121 checks). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Outstanding: the **random-read-over-cap** collapse (unmeasured, and it is `resident: keys`'s own access pattern) and iteration 3's replay baseline. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from |
| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | 🔄 **MEASURED 2026-08-27** — `readiness: ready`, `status: in-progress` (iteration 3 replay baseline still undelivered), forks settled, harness landed (**133 checks**). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Also measured: **random reads over an oversized table collapse 273×** (1.85M vs 6 771 reads/s, p99 1 µs vs 487 µs) — so the two access patterns sit ~270× apart under the same pressure, and departure is a **step, not a curve**. Outstanding: iteration 3's replay baseline. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from |
| 2 | [per-table storage: `durable` and `resident`](databasev2/02-table-storage-modes.md) | 🔄 **the language enrichment — the `durable` half is DONE and usable.** Two optional `@table` keys, `durable: true\|false` and `resident: all\|keys`, both defaulting to today's behaviour (all 28 existing declarations compile unchanged, no golden moved). Landed: the grammar, WO-E224 (a durable `ref` into a volatile table is refused), `.wob` v7 carrying both properties in spare `flags` bits, `durable: false` actually skipping the WAL (measured: 50 inserts → 1500 bytes durable, **0** volatile) with a mode-mismatch startup refusal, plus offset capture and read-a-row-from-an-offset. Outstanding: 5c/5d (the id→offset map and rewiring `wo_row_ptr`'s 11 call sites, slab scans and `@unique`/FK across the boundary — not yet written up), the two runtime refusals, and closeout. [spec](../superpowers/specs/2026-08-26-table-residency-design.md) · [plan](../superpowers/plans/2026-08-26-table-residency.md) |
| 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ⬜ snapshot + truncate: disk reclaimed, replay bounded |
| 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ⬜ **`readiness: ready` — the one startable iteration in the repo** (four forks confirmed settled 2026-08-20). Close the 66× gap iteration 22 measured (durable 4.5k vs ram 297k inserts/s) |

View file

@ -81,9 +81,16 @@ touch — and with swap it **keeps returning 0 while serving from disk**, finish
900 000 rows in 148 s against 150 s uncapped. Durability is the one thing that
does hold: acked writes came back as an intact prefix across an OOM kill.
And when swap does absorb it, the price depends entirely on access pattern:
inserting pays **~1%**, while reading randomly across the table pays **273×**
(1 851 166 reads/s resident against 6 771 over-cap, p99 1 µs against 487 µs).
That second number is the one this track must respect, because it is the access
pattern [2](02-table-storage-modes.md)'s `resident: keys` creates by design.
That is why "back-pressure at exhaustion" is not a design option. Exhaustion
either kills without warning or never arrives. Only a **declared threshold** can
speak in time.
either kills without warning or never arrives — and the latency signal offers no
early warning either, since departure is a **step** (1 µs to 487 µs, nothing in
between) rather than a curve. Only a **declared threshold** can speak in time.
## The lever: per-table storage modes
@ -133,7 +140,7 @@ before its mechanism existed; the history is in
| # | Iteration | Delivers | Needs |
| --- | --- | --- | --- |
| 1 | [RAM ceiling: measure the breaking point](01-ram-ceiling-measurement.md) | 🔄 **measured 2026-08-27**: footprint per shape (3.3× apart), the two silent exits (SIGKILL vs swap-serving-from-disk at ~uncapped speed), and ack-after-fsync surviving an OOM kill. Outstanding: the random-read-over-cap collapse, and a replay baseline | nothing; extends iteration 22's harness |
| 1 | [RAM ceiling: measure the breaking point](01-ram-ceiling-measurement.md) | 🔄 **measured 2026-08-27**: footprint per shape (3.3× apart), the two silent exits (SIGKILL vs swap-serving-from-disk at ~uncapped speed), and ack-after-fsync surviving an OOM kill. Also measured: the **273× random-read collapse** over an oversized table. Outstanding: a replay baseline | nothing; extends iteration 22's harness |
| 2 | [per-table storage](02-table-storage-modes.md) | the grammar: `durable: true\|false` and `resident: all\|keys`, per table, replacing the global `WO_DATA` all-or-nothing. **In progress — the `durable` half is done** | 1 for the budget default |
| 3 | [WAL checkpoint](03-wal-checkpoint.md) *(was language 32)* | snapshot + truncate: disk reclaimed, replay bounded | 4 composes |
| 4 | [io_uring group commit](04-io-uring-commit.md) *(was language 23)* | close the 66× durable/RAM write gap (4.5k vs 297k inserts/s) | the arc (landed) |

View file

@ -97,10 +97,10 @@ are written out once and never read again, so paging is sequential and off the
critical path. The swap is a real disk file (`/swap.img`, no zram, zswap
disabled), so this is genuine disk paging, not compressed RAM.
**The correct generalisation is narrower than "swap is fine".** This measures an
append-mostly workload. A workload that reads randomly across a table larger
than the cap is the one that collapses, and this iteration did *not* measure
that — see Outstanding.
**The correct generalisation is narrower than "swap is fine", and the narrow
claim was then measured too.** The 1% figure belongs to an append-mostly
workload. Reading *randomly* across a table larger than the cap collapses
**273×** — see below. Same cap, same swap, opposite access pattern.
## Progress
@ -113,11 +113,12 @@ that — see Outstanding.
| rootless cap wrapper, swap on/off legs | ✅ 4 footprint legs |
| footprint metric = **median of marginals**, doublings counted separately | ✅ |
| `ceiling` leg: dies at the cap, then replay must be intact | ✅ gated |
| baseline + tolerance policy | ✅ 121 checks; footprint at ±10%, kill-timing metrics at ±100% |
| `randread` leg: control vs over-cap, same key order | ✅ gated |
| baseline + tolerance policy | ✅ 133 checks; footprint at ±10%, kill-timing metrics at ±100% |
| `perf-targets.md` §5 | ✅ |
| **resident-footprint fraction for [iteration 2](02-table-storage-modes.md)** | ⬜ **not delivered — the premise it rested on is false**, see Outstanding |
| **replay/restart baseline for [iteration 3](03-wal-checkpoint.md)** | ⬜ not delivered |
| **the random-read-over-cap collapse** | ⬜ not measured |
| `randread N R` + the `randread` leg — random reads over an oversized table | ✅ **273x collapse measured** |
## Measured
@ -143,6 +144,36 @@ The ceiling, 60 000 Int rows under an 8 MiB cap with swap off and `WO_DATA` set:
**Ack-after-fsync holds through an OOM kill.** That is the one shutdown path
which skips every cleanup handler, and the durable prefix came back whole.
Random reads over an oversized table — 60 000 rows, same Weyl key order in both
legs, only the cap differs:
| Leg | Cap | Throughput | p50 | p99 | RSS after fill |
| --- | --- | --- | --- | --- | --- |
| control, all resident | 256 MiB | **1 851 166 reads/s** | 0 µs | **1 µs** | 13 508 KiB |
| over-cap, swap on | 6 MiB | **6 771 reads/s** | 128 µs | **487 µs** | 6 980 KiB |
**273× throughput collapse, ~480× on p99.** All 20 000 reads resolved correctly
in both legs, so this is the cost of faulting pages back in, not of failing
lookups. Swap off is not an alternative here: that configuration is simply
SIGKILLed.
So the two access patterns sit ~270× apart under identical memory pressure:
| Access pattern | Cost of exceeding RAM |
| --- | --- |
| append-mostly insert | **~1%** — cold pages written once, never re-read |
| random read across the table | **273×** — almost every read faults |
**The mechanism caveat matters for [iteration 2](02-table-storage-modes.md).**
This measures demand-paging of *anonymous slab memory* through swap: 4 KiB at a
time, on fault, with no readahead. `resident: keys` will instead `pread` rows
from the WAL, which goes through the **page cache** — the same physical
constraint (data larger than RAM means disk I/O) but a different mechanism, and
plausibly a better constant, because file reads get readahead and a shared cache
while swap-in does not. **That is a hypothesis, not a result.** The honest
reading is that 273× bounds what *swapping* costs, and iteration 2 must measure
its own read path rather than inherit this number.
**The finding that matters most is the swap leg succeeding.** It did not fail,
did not warn, and returned 0. A deployment in that state looks healthy while
serving from disk. That is the exit with no error signal, and it is why
@ -176,6 +207,10 @@ Met:
the legs are skipped with a named reason and the rest still passes. ✅
`cap_wrapper` returns None unless the `memory` controller is delegated; there
is no uncapped fallback.
- **Given** a table larger than the cap, **when** it is read randomly, **then**
the degradation is quantified. ✅ **273× throughput, ~480× p99**, both legs
reading the same key order with all reads resolving. This closes the gap the
swap leg left, and it is the pattern `resident: keys` creates.
Outstanding:
@ -187,16 +222,12 @@ Outstanding:
an explicit developer-declared figure) rather than waiting on a number this
iteration cannot produce. This is the most important thing this slice learned
and it removes a dependency rather than satisfying it.
- **The random-read-over-cap collapse.** Not measured. This is where the "latency
collapse" prediction may still be true, and it is the workload that matters
for [2](02-table-storage-modes.md)'s `resident: keys`, whose whole premise is
reading rows back from a log larger than RAM. Needs a read-heavy leg over a
table exceeding the cap. **The single most valuable follow-up.**
- **Given** rising fractions of the cap, **when** latency is sampled, **then**
the p99 departure point is recorded. Partially: the sampler and metric exist
and are gated, but the footprint legs never approach their 512 MiB cap, so
`p99_departure_decile` is legitimately 0 and proves nothing. It becomes
meaningful only with the read-heavy leg above.
the p99 departure point is recorded. Partially, and now with a real answer
elsewhere: `p99_departure_decile` stays 0 because the footprint legs never
approach their 512 MiB cap, but the departure itself is measured by the
`randread` leg as a **step, not a curve** — 1 µs resident, 487 µs over-cap.
There is no gentle departure to find; residency is close to binary.
- **A replay/restart baseline for iteration 3.** Not delivered; `growth` exercises
`WO_DATA` but nothing times replay. Cheap to add, still absent from
`bench/baseline.json`.
@ -237,7 +268,12 @@ Outstanding:
5. **Kill-timing metrics carry ±100% tolerance.** `rows_recovered` depends on
where the SIGKILL landed; gating it tightly would be gating the scheduler.
The invariant asserted instead is the *shape* of the survivor.
6. **The ceiling leg asserts `rc`, never records it.** When iteration 2's byte
6. **`randread` gates the RATIO, not the absolutes.** The over-cap half is swap
I/O, so its reads/sec belongs to the box; the collapse factor between two
runs that differ only in their cap belongs to the engine. Both legs read the
same Weyl key order (`i*2654435761 mod n` — no RNG in the language, and none
needed) so residency is the only variable.
7. **The ceiling leg asserts `rc`, never records it.** When iteration 2's byte
budget lands, death should become a checked refusal — the gate must not fail
on that improvement.

View file

@ -118,6 +118,14 @@ Outstanding:
**then** a refusal naming the table and the annotation. *(task 6)*
- **Given** the `resident: all` read baseline, **when** re-measured, **then**
inside tolerance — no cost for a feature not used. *(task 7)*
- **Given** a `resident: keys` table larger than RAM, **when** read randomly,
**then** its read cost is **measured against the resident baseline on its own
read path**, not inherited from databasev2 1's swap figure. *(task 7)* — that
figure is **273×** for demand-paged anonymous memory
([1](01-ram-ceiling-measurement.md)); `pread` through the page cache should do
better, and the whole value of `resident: keys` rests on how much better. If it
is not materially better than swapping, the design buys nothing that the
kernel was not already doing.
## Out Of Scope

View file

@ -61,6 +61,21 @@ stay resident, rows do not.** It buys roughly two orders of magnitude of table
size, not infinity, and the spec says so plainly because a design sold as
unlimited gets deployed as if it were.
**What the trade costs, bounded by measurement (databasev2 1, 2026-08-27).**
Buying table size with disk reads is not free, and the price is large: random
reads across a table larger than RAM measured **273× slower** than resident ones
(1 851 166 reads/s against 6 771; p99 1 µs against 487 µs), and the transition is
a **step, not a curve** — there is no gentle region to operate in. That figure is
an *upper bound on the mechanism this spec does not use*: it is demand-paging of
anonymous memory through swap, 4 KiB per fault with no readahead, whereas
`resident: keys` `pread`s from the WAL through the page cache, which gets
readahead and a shared cache. The constant should therefore be better — **but
that is a hypothesis and task 7 must measure it, not inherit it.** Two things
follow regardless: `resident: keys` must stay opt-in per table (it is), and the
hot-set question is not deferrable decoration — it is
[iteration 5](../../stories/databasev2/05-bounded-tables-eviction.md) and it
decides whether this design is usable for anything read-heavy.
## The design
### Grammar

View file

@ -233,6 +233,7 @@ def tolerance_for(key):
if ".bytes_per_row" in key: return 10
if key.startswith("growth."): return 100
if key.startswith("ceiling."): return 100
if key.startswith("randread."): return 100
if ".mixread." in key or ".mixwrite." in key: return 50
if ".sN." in key: return 50
if ".read." in key or ".query." in key: return 50
@ -374,6 +375,10 @@ def growth(metrics):
CEIL_N, CEIL_CAP_MB = 60000, 8
RAND_N = 60000 if QUICK else 200000
RAND_R = 20000 if QUICK else 40000
RAND_CAP_MB = 6 if QUICK else 14 # over-cap: holds roughly a third of the rows
RAND_FIT_MB = 256 # control: same mechanism, cap simply does not bind
def ceiling(metrics):
"""The ceiling itself, and the durability claim across it.
@ -433,6 +438,81 @@ def ceiling(metrics):
shutil.rmtree(data, ignore_errors=True)
def parse_randread(lines):
"""ops/sec, p50, p99, resolved-read count and post-fill RSS from the
sample's own randread lines. `randreadfilled` also starts with "randread",
so match f[0] exactly, not by prefix."""
ops = p50 = p99 = hits = filled = None
for l in lines:
f = l.split()
if not f:
continue
if f[0] == "randread" and len(f) == 5:
ops, p50, p99 = int(f[2]), int(f[3]), int(f[4])
elif f[0] == "randreadrss" and len(f) == 3:
hits = int(f[2])
elif f[0] == "randreadfilled" and len(f) == 3:
filled = int(f[2])
return ops, p50, p99, hits, filled
def randread(metrics):
"""Random reads over a table LARGER than the memory cap -- the access
pattern the swap measurement was missing.
growth() only inserts, and inserting is append-mostly: cold pages are
written once and never re-read, so swap cost it ~1% (148s vs 150s
uncapped). That result is real but does NOT generalise to "swap is fine".
This leg reads back across the whole range in a Weyl-sequence order, so
most reads must fault a page in.
It matters because it is databasev2 2's `resident: keys` access pattern:
that design reads rows back from a log larger than RAM by construction.
Two runs, identical except for the cap, reading the SAME key order:
- control (RAND_FIT_MB): cap does not bind, everything resident
- over-cap (RAND_CAP_MB): ~a third of the rows fit; swap ON, because
with swap off this configuration is simply SIGKILLed (see ceiling())
The headline is collapse_x, the throughput ratio between them. Tolerances
are wide: the over-cap half is swap I/O, so its absolute numbers are the
box's, while the RATIO is the property of the engine."""
if cap_wrapper(RAND_FIT_MB, 0) is None:
ok("randread: SKIPPED -- no rootless cgroup v2 memory cap on this host")
return
res = {}
for legname, cap_mb, swap_mb in (("resident", RAND_FIT_MB, 0),
("overcap", RAND_CAP_MB, 256)):
w = cap_wrapper(cap_mb, swap_mb)
pr = subprocess.run(w + [BIN, "randread", str(RAND_N), str(RAND_R)],
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
text=True, env=dict(os.environ), timeout=900)
lines = pr.stdout.splitlines()
ops, p50, p99, hits, filled = parse_randread(lines)
key = f"randread.{legname}"
if pr.returncode != 0 or ops is None:
bad(f"{key}: run failed", f"rc={pr.returncode} {(lines[-1:] or ['no output'])[0][:120]}")
return
if hits != RAND_R:
# a collapse measured over reads that did not resolve is noise
bad(f"{key}: only {hits}/{RAND_R} reads resolved", "keys must all exist")
return
metrics[f"{key}.ops_sec"] = ops
metrics[f"{key}.read_p50us"] = p50
metrics[f"{key}.read_p99us"] = p99
metrics[f"{key}.filled_rss_kb"] = filled
res[legname] = ops
ok(f"{key}: {ops} reads/sec, p50 {p50}us p99 {p99}us, {filled} KiB after fill")
collapse = res["resident"] // max(res["overcap"], 1)
metrics["randread.collapse_x"] = collapse
if collapse < 2:
bad("randread: NO collapse -- the cap did not bind",
f"{RAND_N} rows fit under {RAND_CAP_MB} MiB, resize the leg")
else:
ok(f"randread: random reads over an oversized table collapse {collapse}x "
f"({res['resident']} -> {res['overcap']} reads/sec)")
def main():
# --check <results.json>: gate-only evaluation of a recorded run — the
# gate-bites smoke doctors a copy and this mode must FAIL on it
@ -447,6 +527,7 @@ def main():
durability(metrics)
growth(metrics)
ceiling(metrics)
randread(metrics)
os.makedirs(RESULTS_DIR, exist_ok=True)
stamp = time.strftime("%Y%m%d-%H%M%S")
out = os.path.join(RESULTS_DIR, f"run-{stamp}{'-quick' if QUICK else ''}.json")