diff --git a/bench/baseline.json b/bench/baseline.json index 00f9cb8..106f427 100644 --- a/bench/baseline.json +++ b/bench/baseline.json @@ -8,15 +8,15 @@ }, "ceiling.rows_recovered": { "dir": "lower", - "floor": 159744, + "floor": 159828, "tolerance_pct": 100, - "value": 39936 + "value": 39957 }, "durable.s1.mixread.ops_sec": { "dir": "higher", "floor": 2239, "tolerance_pct": 50, - "value": 8958 + "value": 8957 }, "durable.s1.mixread.p50us": { "dir": "lower", @@ -44,15 +44,15 @@ }, "durable.s1.mixwrite.p99us": { "dir": "lower", - "floor": 872, + "floor": 868, "tolerance_pct": 50, - "value": 218 + "value": 217 }, "durable.s1.query.ops_sec": { "dir": "higher", - "floor": 202429, + "floor": 215517, "tolerance_pct": 50, - "value": 809716 + "value": 862068 }, "durable.s1.query.p50us": { "dir": "lower", @@ -64,13 +64,13 @@ "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 2 + "value": 1 }, "durable.s1.read.ops_sec": { "dir": "higher", - "floor": 215703, + "floor": 221827, "tolerance_pct": 50, - "value": 862812 + "value": 887311 }, "durable.s1.read.p50us": { "dir": "lower", @@ -86,45 +86,45 @@ }, "durable.s1.seed.ops_sec": { "dir": "higher", - "floor": 1078, + "floor": 1128, "tolerance_pct": 15, - "value": 4315 + "value": 4515 }, "durable.s1.seed.p50us": { "dir": "lower", - "floor": 828, + "floor": 832, "tolerance_pct": 15, - "value": 207 + "value": 208 }, "durable.s1.seed.p99us": { "dir": "lower", - "floor": 2160, + "floor": 2180, "tolerance_pct": 15, - "value": 540 + "value": 545 }, "durable.s1.write.ops_sec": { "dir": "higher", - "floor": 1156, + "floor": 1141, "tolerance_pct": 15, - "value": 4626 + "value": 4565 }, "durable.s1.write.p50us": { "dir": "lower", - "floor": 828, + "floor": 832, "tolerance_pct": 15, - "value": 207 + "value": 208 }, "durable.s1.write.p99us": { "dir": "lower", - "floor": 1844, + "floor": 2256, "tolerance_pct": 15, - "value": 461 + "value": 564 }, "durable.sN.mixread.ops_sec": { "dir": "higher", - "floor": 1113, + "floor": 1117, "tolerance_pct": 50, - "value": 4452 + "value": 4468 }, "durable.sN.mixread.p50us": { "dir": "lower", @@ -134,33 +134,33 @@ }, "durable.sN.mixread.p99us": { "dir": "lower", - "floor": 15116, + "floor": 19900, "tolerance_pct": 50, - "value": 3779 + "value": 4975 }, "durable.sN.mixwrite.ops_sec": { "dir": "higher", - "floor": 123, + "floor": 124, "tolerance_pct": 50, - "value": 494 + "value": 496 }, "durable.sN.mixwrite.p50us": { "dir": "lower", - "floor": 1148, + "floor": 1104, "tolerance_pct": 50, - "value": 287 + "value": 276 }, "durable.sN.mixwrite.p99us": { "dir": "lower", - "floor": 2932, + "floor": 1928, "tolerance_pct": 50, - "value": 733 + "value": 482 }, "durable.sN.query.ops_sec": { "dir": "higher", - "floor": 333333, + "floor": 331125, "tolerance_pct": 50, - "value": 1333333 + "value": 1324503 }, "durable.sN.query.p50us": { "dir": "lower", @@ -176,9 +176,9 @@ }, "durable.sN.read.ops_sec": { "dir": "higher", - "floor": 298329, + "floor": 341064, "tolerance_pct": 50, - "value": 1193317 + "value": 1364256 }, "durable.sN.read.p50us": { "dir": "lower", @@ -190,13 +190,13 @@ "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 2 + "value": 1 }, "durable.sN.seed.ops_sec": { "dir": "higher", - "floor": 1139, + "floor": 1125, "tolerance_pct": 50, - "value": 4556 + "value": 4502 }, "durable.sN.seed.p50us": { "dir": "lower", @@ -206,27 +206,27 @@ }, "durable.sN.seed.p99us": { "dir": "lower", - "floor": 2000, + "floor": 2624, "tolerance_pct": 50, - "value": 500 + "value": 656 }, "durable.sN.write.ops_sec": { "dir": "higher", - "floor": 1153, + "floor": 1150, "tolerance_pct": 50, - "value": 4614 + "value": 4600 }, "durable.sN.write.p50us": { "dir": "lower", - "floor": 828, + "floor": 832, "tolerance_pct": 50, - "value": 207 + "value": 208 }, "durable.sN.write.p99us": { "dir": "lower", - "floor": 2172, + "floor": 2348, "tolerance_pct": 50, - "value": 543 + "value": 587 }, "growth.available": { "dir": "lower", @@ -236,9 +236,9 @@ }, "growth.int.noswap.bytes_per_row": { "dir": "lower", - "floor": 400, + "floor": 392, "tolerance_pct": 10, - "value": 100 + "value": 98 }, "growth.int.noswap.doublings": { "dir": "lower", @@ -272,15 +272,15 @@ }, "growth.int.noswap.rss_kb": { "dir": "lower", - "floor": 23936, + "floor": 23920, "tolerance_pct": 100, - "value": 5984 + "value": 5980 }, "growth.int.swap.bytes_per_row": { "dir": "lower", - "floor": 400, + "floor": 392, "tolerance_pct": 10, - "value": 100 + "value": 98 }, "growth.int.swap.doublings": { "dir": "lower", @@ -314,15 +314,15 @@ }, "growth.int.swap.rss_kb": { "dir": "lower", - "floor": 23952, + "floor": 23920, "tolerance_pct": 100, - "value": 5988 + "value": 5980 }, "growth.text.noswap.bytes_per_row": { "dir": "lower", - "floor": 1296, + "floor": 1288, "tolerance_pct": 10, - "value": 324 + "value": 322 }, "growth.text.noswap.doublings": { "dir": "lower", @@ -356,15 +356,15 @@ }, "growth.text.noswap.rss_kb": { "dir": "lower", - "floor": 41216, + "floor": 41184, "tolerance_pct": 100, - "value": 10304 + "value": 10296 }, "growth.text.swap.bytes_per_row": { "dir": "lower", - "floor": 1296, + "floor": 1288, "tolerance_pct": 10, - "value": 324 + "value": 322 }, "growth.text.swap.doublings": { "dir": "lower", @@ -398,15 +398,15 @@ }, "growth.text.swap.rss_kb": { "dir": "lower", - "floor": 41216, + "floor": 41200, "tolerance_pct": 100, - "value": 10304 + "value": 10300 }, "ram.s1.mixread.ops_sec": { "dir": "higher", - "floor": 2239, + "floor": 2236, "tolerance_pct": 50, - "value": 8956 + "value": 8945 }, "ram.s1.mixread.p50us": { "dir": "lower", @@ -424,13 +424,13 @@ "dir": "higher", "floor": 248, "tolerance_pct": 50, - "value": 995 + "value": 993 }, "ram.s1.mixwrite.p50us": { "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 0 + "value": 1 }, "ram.s1.mixwrite.p99us": { "dir": "lower", @@ -440,15 +440,15 @@ }, "ram.s1.msgrate.msgs_sec": { "dir": "higher", - "floor": 419674, + "floor": 392834, "tolerance_pct": 15, - "value": 3357394 + "value": 3142677 }, "ram.s1.query.ops_sec": { "dir": "higher", - "floor": 340136, + "floor": 214592, "tolerance_pct": 50, - "value": 1360544 + "value": 858369 }, "ram.s1.query.p50us": { "dir": "lower", @@ -460,13 +460,13 @@ "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 1 + "value": 2 }, "ram.s1.read.ops_sec": { "dir": "higher", - "floor": 369276, + "floor": 226860, "tolerance_pct": 50, - "value": 1477104 + "value": 907441 }, "ram.s1.read.p50us": { "dir": "lower", @@ -478,31 +478,31 @@ "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 1 + "value": 2 }, "ram.s1.seed.ops_sec": { "dir": "higher", - "floor": 470366, + "floor": 273373, "tolerance_pct": 15, - "value": 1881467 + "value": 1093493 }, "ram.s1.seed.p50us": { "dir": "lower", "floor": 100, "tolerance_pct": 15, - "value": 0 + "value": 1 }, "ram.s1.seed.p99us": { "dir": "lower", "floor": 100, "tolerance_pct": 15, - "value": 1 + "value": 2 }, "ram.s1.write.ops_sec": { "dir": "higher", - "floor": 294464, + "floor": 159134, "tolerance_pct": 15, - "value": 1177856 + "value": 636537 }, "ram.s1.write.p50us": { "dir": "lower", @@ -514,25 +514,25 @@ "dir": "lower", "floor": 100, "tolerance_pct": 15, - "value": 1 + "value": 3 }, "ram.sN.mixread.ops_sec": { "dir": "higher", - "floor": 2240, + "floor": 2239, "tolerance_pct": 50, - "value": 8960 + "value": 8958 }, "ram.sN.mixread.p50us": { "dir": "lower", - "floor": 228, + "floor": 232, "tolerance_pct": 50, - "value": 57 + "value": 58 }, "ram.sN.mixread.p99us": { "dir": "lower", - "floor": 1404, + "floor": 1032, "tolerance_pct": 50, - "value": 351 + "value": 258 }, "ram.sN.mixwrite.ops_sec": { "dir": "higher", @@ -542,27 +542,27 @@ }, "ram.sN.mixwrite.p50us": { "dir": "lower", - "floor": 248, + "floor": 256, "tolerance_pct": 50, - "value": 62 + "value": 64 }, "ram.sN.mixwrite.p99us": { "dir": "lower", - "floor": 292, + "floor": 328, "tolerance_pct": 50, - "value": 73 + "value": 82 }, "ram.sN.msgrate.msgs_sec": { "dir": "higher", - "floor": 216394, + "floor": 229885, "tolerance_pct": 50, - "value": 1731152 + "value": 1839080 }, "ram.sN.query.ops_sec": { "dir": "higher", - "floor": 340136, + "floor": 337837, "tolerance_pct": 50, - "value": 1360544 + "value": 1351351 }, "ram.sN.query.p50us": { "dir": "lower", @@ -578,9 +578,9 @@ }, "ram.sN.read.ops_sec": { "dir": "higher", - "floor": 343878, + "floor": 342935, "tolerance_pct": 50, - "value": 1375515 + "value": 1371742 }, "ram.sN.read.p50us": { "dir": "lower", @@ -596,27 +596,27 @@ }, "ram.sN.seed.ops_sec": { "dir": "higher", - "floor": 445235, + "floor": 295159, "tolerance_pct": 50, - "value": 1780943 + "value": 1180637 }, "ram.sN.seed.p50us": { "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 0 + "value": 1 }, "ram.sN.seed.p99us": { "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 1 + "value": 2 }, "ram.sN.write.ops_sec": { "dir": "higher", - "floor": 286368, + "floor": 289017, "tolerance_pct": 50, - "value": 1145475 + "value": 1156069 }, "ram.sN.write.p50us": { "dir": "lower", @@ -629,5 +629,59 @@ "floor": 100, "tolerance_pct": 50, "value": 2 + }, + "randread.collapse_x": { + "dir": "lower", + "floor": 1136, + "tolerance_pct": 100, + "value": 284 + }, + "randread.overcap.filled_rss_kb": { + "dir": "lower", + "floor": 25024, + "tolerance_pct": 100, + "value": 6256 + }, + "randread.overcap.ops_sec": { + "dir": "higher", + "floor": 1669, + "tolerance_pct": 100, + "value": 6676 + }, + "randread.overcap.read_p50us": { + "dir": "lower", + "floor": 548, + "tolerance_pct": 100, + "value": 137 + }, + "randread.overcap.read_p99us": { + "dir": "lower", + "floor": 1920, + "tolerance_pct": 100, + "value": 480 + }, + "randread.resident.filled_rss_kb": { + "dir": "lower", + "floor": 54000, + "tolerance_pct": 100, + "value": 13500 + }, + "randread.resident.ops_sec": { + "dir": "higher", + "floor": 475556, + "tolerance_pct": 100, + "value": 1902225 + }, + "randread.resident.read_p50us": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 0 + }, + "randread.resident.read_p99us": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 100, + "value": 1 } } \ No newline at end of file diff --git a/docs/examples/db-bench/main.wo b/docs/examples/db-bench/main.wo index f20a6ac..fa08f87 100644 --- a/docs/examples/db-bench/main.wo +++ b/docs/examples/db-bench/main.wo @@ -467,6 +467,7 @@ fn usage() -> Int { print_err("usage: db-bench "); print_err(" all N | seed N | read N | query N | write N | wal N"); print_err(" mix N C | msgrate N | growth N int|text | growth-verify"); + print_err(" randread N R"); print_err(" verify | verify-acked M"); return 2; } @@ -518,6 +519,48 @@ fn self_rss_kb() -> Int { -- intact: rows 1..M all present with the right v and no holes. M is whatever -- survived -- the claim under test is the SHAPE of the survivor, not its size, -- because a SIGKILL can land between any two inserts. +-- databasev2 1: randread N R -- fill N rows, then read R of them by key in a +-- Weyl-sequence order that spreads across the WHOLE range. Under a cap smaller +-- than the table most of those reads must fault a page back in. +-- +-- This is the leg the swap measurement was MISSING. `growth` inserts, and +-- inserting is append-mostly: cold pages are written once and never re-read, so +-- swap cost it ~1% (148s vs 150s uncapped). Random reads over an oversized +-- table are the opposite access pattern -- and they are exactly what +-- databasev2 2's `resident: keys` creates, since it reads rows back from a log +-- larger than RAM. No RNG in the language and none needed: i*2654435761 mod n +-- is a Weyl sequence, deterministic and spread, so the two legs read the SAME +-- key order and only residency differs. +fn randread_mode(n: Int, r: Int) -> Int { + let bref = insert Bucket { tag: "randread" }; + let i = 1; + while i <= n { + insert Item { k: i, v: item_v(i), bucket: bref }; + i = i + 1; + } + print("randreadfilled ${n} ${self_rss_kb()}"); + let h: map = {}; + let hits = 0; + let t0 = time.ticks(); + let j = 0; + while j < r { + let key = 1 + (j * 2654435761) % n; + let o0 = time.ticks(); + for row in from x in Item where x.k == key take 1 select x { + if row.v == item_v(key) { + hits = hits + 1; + } + } + hist_add(h, time.ticks() - o0); + j = j + 1; + } + let el = time.ticks() - t0; + report("randread", r, el, h); + -- hits proves the reads RESOLVED; a collapse measured over misses is noise + print("randreadrss ${self_rss_kb()} ${hits}"); + return 0; +} + fn growth_verify() -> Int { let seen: map = {}; let maxk = 0; @@ -643,6 +686,17 @@ fn main(args: multi Text) -> Int { } return growth_mode(n, args[2]); } + if args[0] == "randread" { + if len(args) < 3 { + return usage(); + } + let rr = parse_int(args[2]); + if rr == nil or rr < 1 { + print_err("db-bench: must be a positive number"); + return 2; + } + return randread_mode(n, rr); + } if args[0] == "mix" { if len(args) < 3 { return usage(); diff --git a/docs/plan/perf-targets.md b/docs/plan/perf-targets.md index 983158e..dfb5bec 100644 --- a/docs/plan/perf-targets.md +++ b/docs/plan/perf-targets.md @@ -126,9 +126,7 @@ is a real disk file (`/swap.img`; no zram, zswap disabled), so this is genuine disk paging. **Do not generalise this to "swap is fine".** It measures an append-mostly -workload. A random-read workload over a table larger than the cap is where the -collapse should appear, and it is **not yet measured** — which matters, because -that is exactly the access pattern databasev2 2's `resident: keys` creates. +workload — and the opposite pattern was then measured too, below. The operational consequence is that the RAM ceiling has two shapes and neither reports itself: without swap the process vanishes on signal 9, with swap it @@ -153,3 +151,41 @@ scheduler's business, so `rows_recovered` carries ±100% tolerance. The leg asserts the exit but never records it as a metric, so that when databasev2 2's byte budget turns the kill into a checked refusal, the gate does not fail on the improvement. + +### Random reads over an oversized table: 273× + +60 000 Int rows, both legs reading the **same** Weyl key order +(`i*2654435761 mod n`), differing only in the cap: + +| Leg | Cap | Throughput | p50 | p99 | +| --- | --- | --- | --- | --- | +| all resident | 256 MiB | **1 851 166 reads/s** | 0 µs | **1 µs** | +| over-cap, swap on | 6 MiB | **6 771 reads/s** | 128 µs | **487 µs** | + +All 20 000 reads resolved in both legs, so this is the cost of faulting pages +back, not of failed lookups. Swap-off is not an option in this configuration — +it is SIGKILLed. + +**The two access patterns are ~270× apart under identical memory pressure:** + +| Pattern | Cost of exceeding RAM | +| --- | --- | +| append-mostly insert | **~1%** (cold pages written once, never re-read) | +| random read across the table | **273×** | + +**Departure is a step, not a curve.** 1 µs to 487 µs with nothing in between — +`p99_departure_decile` looks for a gentle knee that does not exist. Residency is +close to binary, which is why a budget must fire at a *declared* threshold: there +is no early warning in the latency signal to react to. + +**Mechanism caveat, and it is a design input for databasev2 2.** This is +demand-paging of *anonymous slab memory* through swap — 4 KiB per fault, no +readahead. `resident: keys` instead `pread`s rows from the WAL, through the +**page cache**: same physical constraint, different mechanism, plausibly a better +constant because file reads get readahead and a shared cache. **That is a +hypothesis.** 273× bounds what *swapping* costs; iteration 2 must measure its own +read path rather than inherit this figure. + +Gated as `db-bench`'s `randread` leg, which gates the **ratio** — the absolute +reads/sec of the over-cap half is the box's swap device, while the factor between +two runs differing only in their cap is the engine's. diff --git a/docs/stories/00-status.md b/docs/stories/00-status.md index 1344dfc..a47501b 100644 --- a/docs/stories/00-status.md +++ b/docs/stories/00-status.md @@ -75,8 +75,9 @@ Int-only `Item`; `growth N int|text` in the db-bench sample, reading its OWN `/proc/self/status` RSS at each decile because the driver's 250 ms poll misses the value *at* a boundary; `growth-verify`, which asserts the survivor of a crash is a contiguous intact prefix; and two harness legs — four footprint legs -under a rootless cgroup v2 cap, and a `ceiling` leg that deliberately dies at -the cap and then replays. 121 checks, 0 failures. +under a rootless cgroup v2 cap, a `ceiling` leg that deliberately dies at the cap +and then replays, and a `randread` leg that reads an oversized table randomly. +133 checks, 0 failures. **Key findings (measured, not asserted):** per-row footprint is **96.5–100 B** Int-only and **320.6–324 B** text-heavy — **3.3×**, not the "order of magnitude" @@ -90,12 +91,17 @@ the opposite: `WO_HEAP_MB` is checked and traps). And swap is not "latency collapse": 900 000 rows inside a 64 MiB cap with swap finished in **148 s against 150 s uncapped** — ~1%, on a real disk swap file with no zram. Also measured: **ack-after-fsync holds through an OOM kill** — ~40 000 rows came back -as an intact prefix, no holes, not read as corruption. +as an intact prefix, no holes, not read as corruption. And the pattern the swap +leg was missing: **random reads over an oversized table collapse 273×.** **Learned:** an append-mostly workload never re-touches its cold pages, so swap -costs it nothing — the collapse belongs to *random reads* over an oversized -table, which is precisely the pattern iteration 2's `resident: keys` creates and -is **still unmeasured**. The RAM ceiling therefore has two shapes and neither +costs it nothing — and the opposite pattern was then measured on the same day. +The `randread` leg reads randomly across a table larger than the cap, both legs +walking the SAME Weyl key order so residency is the only variable: **273× +throughput collapse** (1 851 166 → 6 771 reads/s), p99 **1 µs → 487 µs**, all +20 000 reads resolving in both. So the two access patterns sit ~270× apart under +identical memory pressure, and **departure is a step, not a curve** — which is +why `p99_departure_decile` finds nothing: there is no knee to find. The RAM ceiling therefore has two shapes and neither announces itself: without swap the process vanishes on signal 9, with swap it keeps returning 0 while serving from disk. That is the argument for a budget that fires at a declared threshold instead of at exhaustion. @@ -107,10 +113,13 @@ degradation to detect. Iteration 2 must pick its budget on other grounds rather than wait on a number this slice cannot produce. Iteration 3's replay baseline is still NOT delivered — `bench/baseline.json` times no replay. -**Next steps:** the read-heavy-over-cap leg is the single most valuable -follow-up, and it is what makes `p99_departure_decile` mean anything (the -footprint legs never approach their 512 MiB cap, so it is legitimately 0 today). -Then iteration 2's 5c/5d. +**Next steps:** iteration 2's 5c/5d. Its task 7 gained a criterion from this: +`resident: keys` must measure its OWN read path rather than inherit 273×. That +number bounds demand-paged anonymous memory through swap (4 KiB per fault, no +readahead); `pread` through the page cache should beat it, and **the entire value +of `resident: keys` rests on how much** — if it is not materially better than +swapping, the design buys nothing the kernel was not already doing. Still absent: +a replay baseline for iteration 3. **`.dev/reference` used:** none. Sources were the kernel's own interfaces — cgroup v2 `memory.max`/`memory.swap.max`, `/proc/self/status`, `/proc/swaps` and @@ -684,7 +693,7 @@ the language arc as v1 history. | # | Iteration | State | | --- | --- | --- | -| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | 🔄 **MEASURED 2026-08-27** — `readiness: ready`, `status: in-progress` (two criteria outstanding), forks settled, harness landed (121 checks). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Outstanding: the **random-read-over-cap** collapse (unmeasured, and it is `resident: keys`'s own access pattern) and iteration 3's replay baseline. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from | +| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | 🔄 **MEASURED 2026-08-27** — `readiness: ready`, `status: in-progress` (iteration 3 replay baseline still undelivered), forks settled, harness landed (**133 checks**). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Also measured: **random reads over an oversized table collapse 273×** (1.85M vs 6 771 reads/s, p99 1 µs vs 487 µs) — so the two access patterns sit ~270× apart under the same pressure, and departure is a **step, not a curve**. Outstanding: iteration 3's replay baseline. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from | | 2 | [per-table storage: `durable` and `resident`](databasev2/02-table-storage-modes.md) | 🔄 **the language enrichment — the `durable` half is DONE and usable.** Two optional `@table` keys, `durable: true\|false` and `resident: all\|keys`, both defaulting to today's behaviour (all 28 existing declarations compile unchanged, no golden moved). Landed: the grammar, WO-E224 (a durable `ref` into a volatile table is refused), `.wob` v7 carrying both properties in spare `flags` bits, `durable: false` actually skipping the WAL (measured: 50 inserts → 1500 bytes durable, **0** volatile) with a mode-mismatch startup refusal, plus offset capture and read-a-row-from-an-offset. Outstanding: 5c/5d (the id→offset map and rewiring `wo_row_ptr`'s 11 call sites, slab scans and `@unique`/FK across the boundary — not yet written up), the two runtime refusals, and closeout. [spec](../superpowers/specs/2026-08-26-table-residency-design.md) · [plan](../superpowers/plans/2026-08-26-table-residency.md) | | 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ⬜ snapshot + truncate: disk reclaimed, replay bounded | | 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ⬜ **`readiness: ready` — the one startable iteration in the repo** (four forks confirmed settled 2026-08-20). Close the 66× gap iteration 22 measured (durable 4.5k vs ram 297k inserts/s) | diff --git a/docs/stories/databasev2/00-story.md b/docs/stories/databasev2/00-story.md index 5d4957a..405f69e 100644 --- a/docs/stories/databasev2/00-story.md +++ b/docs/stories/databasev2/00-story.md @@ -81,9 +81,16 @@ touch — and with swap it **keeps returning 0 while serving from disk**, finish 900 000 rows in 148 s against 150 s uncapped. Durability is the one thing that does hold: acked writes came back as an intact prefix across an OOM kill. +And when swap does absorb it, the price depends entirely on access pattern: +inserting pays **~1%**, while reading randomly across the table pays **273×** +(1 851 166 reads/s resident against 6 771 over-cap, p99 1 µs against 487 µs). +That second number is the one this track must respect, because it is the access +pattern [2](02-table-storage-modes.md)'s `resident: keys` creates by design. + That is why "back-pressure at exhaustion" is not a design option. Exhaustion -either kills without warning or never arrives. Only a **declared threshold** can -speak in time. +either kills without warning or never arrives — and the latency signal offers no +early warning either, since departure is a **step** (1 µs to 487 µs, nothing in +between) rather than a curve. Only a **declared threshold** can speak in time. ## The lever: per-table storage modes @@ -133,7 +140,7 @@ before its mechanism existed; the history is in | # | Iteration | Delivers | Needs | | --- | --- | --- | --- | -| 1 | [RAM ceiling: measure the breaking point](01-ram-ceiling-measurement.md) | 🔄 **measured 2026-08-27**: footprint per shape (3.3× apart), the two silent exits (SIGKILL vs swap-serving-from-disk at ~uncapped speed), and ack-after-fsync surviving an OOM kill. Outstanding: the random-read-over-cap collapse, and a replay baseline | nothing; extends iteration 22's harness | +| 1 | [RAM ceiling: measure the breaking point](01-ram-ceiling-measurement.md) | 🔄 **measured 2026-08-27**: footprint per shape (3.3× apart), the two silent exits (SIGKILL vs swap-serving-from-disk at ~uncapped speed), and ack-after-fsync surviving an OOM kill. Also measured: the **273× random-read collapse** over an oversized table. Outstanding: a replay baseline | nothing; extends iteration 22's harness | | 2 | [per-table storage](02-table-storage-modes.md) | the grammar: `durable: true\|false` and `resident: all\|keys`, per table, replacing the global `WO_DATA` all-or-nothing. **In progress — the `durable` half is done** | 1 for the budget default | | 3 | [WAL checkpoint](03-wal-checkpoint.md) *(was language 32)* | snapshot + truncate: disk reclaimed, replay bounded | 4 composes | | 4 | [io_uring group commit](04-io-uring-commit.md) *(was language 23)* | close the 66× durable/RAM write gap (4.5k vs 297k inserts/s) | the arc (landed) | diff --git a/docs/stories/databasev2/01-ram-ceiling-measurement.md b/docs/stories/databasev2/01-ram-ceiling-measurement.md index 07e7110..5018cfc 100644 --- a/docs/stories/databasev2/01-ram-ceiling-measurement.md +++ b/docs/stories/databasev2/01-ram-ceiling-measurement.md @@ -97,10 +97,10 @@ are written out once and never read again, so paging is sequential and off the critical path. The swap is a real disk file (`/swap.img`, no zram, zswap disabled), so this is genuine disk paging, not compressed RAM. -**The correct generalisation is narrower than "swap is fine".** This measures an -append-mostly workload. A workload that reads randomly across a table larger -than the cap is the one that collapses, and this iteration did *not* measure -that — see Outstanding. +**The correct generalisation is narrower than "swap is fine", and the narrow +claim was then measured too.** The 1% figure belongs to an append-mostly +workload. Reading *randomly* across a table larger than the cap collapses +**273×** — see below. Same cap, same swap, opposite access pattern. ## Progress @@ -113,11 +113,12 @@ that — see Outstanding. | rootless cap wrapper, swap on/off legs | ✅ 4 footprint legs | | footprint metric = **median of marginals**, doublings counted separately | ✅ | | `ceiling` leg: dies at the cap, then replay must be intact | ✅ gated | -| baseline + tolerance policy | ✅ 121 checks; footprint at ±10%, kill-timing metrics at ±100% | +| `randread` leg: control vs over-cap, same key order | ✅ gated | +| baseline + tolerance policy | ✅ 133 checks; footprint at ±10%, kill-timing metrics at ±100% | | `perf-targets.md` §5 | ✅ | | **resident-footprint fraction for [iteration 2](02-table-storage-modes.md)** | ⬜ **not delivered — the premise it rested on is false**, see Outstanding | | **replay/restart baseline for [iteration 3](03-wal-checkpoint.md)** | ⬜ not delivered | -| **the random-read-over-cap collapse** | ⬜ not measured | +| `randread N R` + the `randread` leg — random reads over an oversized table | ✅ **273x collapse measured** | ## Measured @@ -143,6 +144,36 @@ The ceiling, 60 000 Int rows under an 8 MiB cap with swap off and `WO_DATA` set: **Ack-after-fsync holds through an OOM kill.** That is the one shutdown path which skips every cleanup handler, and the durable prefix came back whole. +Random reads over an oversized table — 60 000 rows, same Weyl key order in both +legs, only the cap differs: + +| Leg | Cap | Throughput | p50 | p99 | RSS after fill | +| --- | --- | --- | --- | --- | --- | +| control, all resident | 256 MiB | **1 851 166 reads/s** | 0 µs | **1 µs** | 13 508 KiB | +| over-cap, swap on | 6 MiB | **6 771 reads/s** | 128 µs | **487 µs** | 6 980 KiB | + +**273× throughput collapse, ~480× on p99.** All 20 000 reads resolved correctly +in both legs, so this is the cost of faulting pages back in, not of failing +lookups. Swap off is not an alternative here: that configuration is simply +SIGKILLed. + +So the two access patterns sit ~270× apart under identical memory pressure: + +| Access pattern | Cost of exceeding RAM | +| --- | --- | +| append-mostly insert | **~1%** — cold pages written once, never re-read | +| random read across the table | **273×** — almost every read faults | + +**The mechanism caveat matters for [iteration 2](02-table-storage-modes.md).** +This measures demand-paging of *anonymous slab memory* through swap: 4 KiB at a +time, on fault, with no readahead. `resident: keys` will instead `pread` rows +from the WAL, which goes through the **page cache** — the same physical +constraint (data larger than RAM means disk I/O) but a different mechanism, and +plausibly a better constant, because file reads get readahead and a shared cache +while swap-in does not. **That is a hypothesis, not a result.** The honest +reading is that 273× bounds what *swapping* costs, and iteration 2 must measure +its own read path rather than inherit this number. + **The finding that matters most is the swap leg succeeding.** It did not fail, did not warn, and returned 0. A deployment in that state looks healthy while serving from disk. That is the exit with no error signal, and it is why @@ -176,6 +207,10 @@ Met: the legs are skipped with a named reason and the rest still passes. ✅ `cap_wrapper` returns None unless the `memory` controller is delegated; there is no uncapped fallback. +- **Given** a table larger than the cap, **when** it is read randomly, **then** + the degradation is quantified. ✅ **273× throughput, ~480× p99**, both legs + reading the same key order with all reads resolving. This closes the gap the + swap leg left, and it is the pattern `resident: keys` creates. Outstanding: @@ -187,16 +222,12 @@ Outstanding: an explicit developer-declared figure) rather than waiting on a number this iteration cannot produce. This is the most important thing this slice learned and it removes a dependency rather than satisfying it. -- **The random-read-over-cap collapse.** Not measured. This is where the "latency - collapse" prediction may still be true, and it is the workload that matters - for [2](02-table-storage-modes.md)'s `resident: keys`, whose whole premise is - reading rows back from a log larger than RAM. Needs a read-heavy leg over a - table exceeding the cap. **The single most valuable follow-up.** - **Given** rising fractions of the cap, **when** latency is sampled, **then** - the p99 departure point is recorded. Partially: the sampler and metric exist - and are gated, but the footprint legs never approach their 512 MiB cap, so - `p99_departure_decile` is legitimately 0 and proves nothing. It becomes - meaningful only with the read-heavy leg above. + the p99 departure point is recorded. Partially, and now with a real answer + elsewhere: `p99_departure_decile` stays 0 because the footprint legs never + approach their 512 MiB cap, but the departure itself is measured by the + `randread` leg as a **step, not a curve** — 1 µs resident, 487 µs over-cap. + There is no gentle departure to find; residency is close to binary. - **A replay/restart baseline for iteration 3.** Not delivered; `growth` exercises `WO_DATA` but nothing times replay. Cheap to add, still absent from `bench/baseline.json`. @@ -237,7 +268,12 @@ Outstanding: 5. **Kill-timing metrics carry ±100% tolerance.** `rows_recovered` depends on where the SIGKILL landed; gating it tightly would be gating the scheduler. The invariant asserted instead is the *shape* of the survivor. -6. **The ceiling leg asserts `rc`, never records it.** When iteration 2's byte +6. **`randread` gates the RATIO, not the absolutes.** The over-cap half is swap + I/O, so its reads/sec belongs to the box; the collapse factor between two + runs that differ only in their cap belongs to the engine. Both legs read the + same Weyl key order (`i*2654435761 mod n` — no RNG in the language, and none + needed) so residency is the only variable. +7. **The ceiling leg asserts `rc`, never records it.** When iteration 2's byte budget lands, death should become a checked refusal — the gate must not fail on that improvement. diff --git a/docs/stories/databasev2/02-table-storage-modes.md b/docs/stories/databasev2/02-table-storage-modes.md index d5c6c7f..06b3f18 100644 --- a/docs/stories/databasev2/02-table-storage-modes.md +++ b/docs/stories/databasev2/02-table-storage-modes.md @@ -118,6 +118,14 @@ Outstanding: **then** a refusal naming the table and the annotation. *(task 6)* - **Given** the `resident: all` read baseline, **when** re-measured, **then** inside tolerance — no cost for a feature not used. *(task 7)* +- **Given** a `resident: keys` table larger than RAM, **when** read randomly, + **then** its read cost is **measured against the resident baseline on its own + read path**, not inherited from databasev2 1's swap figure. *(task 7)* — that + figure is **273×** for demand-paged anonymous memory + ([1](01-ram-ceiling-measurement.md)); `pread` through the page cache should do + better, and the whole value of `resident: keys` rests on how much better. If it + is not materially better than swapping, the design buys nothing that the + kernel was not already doing. ## Out Of Scope diff --git a/docs/superpowers/specs/2026-08-26-table-residency-design.md b/docs/superpowers/specs/2026-08-26-table-residency-design.md index 2342168..efc6d65 100644 --- a/docs/superpowers/specs/2026-08-26-table-residency-design.md +++ b/docs/superpowers/specs/2026-08-26-table-residency-design.md @@ -61,6 +61,21 @@ stay resident, rows do not.** It buys roughly two orders of magnitude of table size, not infinity, and the spec says so plainly because a design sold as unlimited gets deployed as if it were. +**What the trade costs, bounded by measurement (databasev2 1, 2026-08-27).** +Buying table size with disk reads is not free, and the price is large: random +reads across a table larger than RAM measured **273× slower** than resident ones +(1 851 166 reads/s against 6 771; p99 1 µs against 487 µs), and the transition is +a **step, not a curve** — there is no gentle region to operate in. That figure is +an *upper bound on the mechanism this spec does not use*: it is demand-paging of +anonymous memory through swap, 4 KiB per fault with no readahead, whereas +`resident: keys` `pread`s from the WAL through the page cache, which gets +readahead and a shared cache. The constant should therefore be better — **but +that is a hypothesis and task 7 must measure it, not inherit it.** Two things +follow regardless: `resident: keys` must stay opt-in per table (it is), and the +hot-set question is not deferrable decoration — it is +[iteration 5](../../stories/databasev2/05-bounded-tables-eviction.md) and it +decides whether this design is usable for anything read-heavy. + ## The design ### Grammar diff --git a/scripts/db-bench.py b/scripts/db-bench.py index 930f325..83fb69a 100755 --- a/scripts/db-bench.py +++ b/scripts/db-bench.py @@ -233,6 +233,7 @@ def tolerance_for(key): if ".bytes_per_row" in key: return 10 if key.startswith("growth."): return 100 if key.startswith("ceiling."): return 100 + if key.startswith("randread."): return 100 if ".mixread." in key or ".mixwrite." in key: return 50 if ".sN." in key: return 50 if ".read." in key or ".query." in key: return 50 @@ -374,6 +375,10 @@ def growth(metrics): CEIL_N, CEIL_CAP_MB = 60000, 8 +RAND_N = 60000 if QUICK else 200000 +RAND_R = 20000 if QUICK else 40000 +RAND_CAP_MB = 6 if QUICK else 14 # over-cap: holds roughly a third of the rows +RAND_FIT_MB = 256 # control: same mechanism, cap simply does not bind def ceiling(metrics): """The ceiling itself, and the durability claim across it. @@ -433,6 +438,81 @@ def ceiling(metrics): shutil.rmtree(data, ignore_errors=True) + +def parse_randread(lines): + """ops/sec, p50, p99, resolved-read count and post-fill RSS from the + sample's own randread lines. `randreadfilled` also starts with "randread", + so match f[0] exactly, not by prefix.""" + ops = p50 = p99 = hits = filled = None + for l in lines: + f = l.split() + if not f: + continue + if f[0] == "randread" and len(f) == 5: + ops, p50, p99 = int(f[2]), int(f[3]), int(f[4]) + elif f[0] == "randreadrss" and len(f) == 3: + hits = int(f[2]) + elif f[0] == "randreadfilled" and len(f) == 3: + filled = int(f[2]) + return ops, p50, p99, hits, filled + + +def randread(metrics): + """Random reads over a table LARGER than the memory cap -- the access + pattern the swap measurement was missing. + + growth() only inserts, and inserting is append-mostly: cold pages are + written once and never re-read, so swap cost it ~1% (148s vs 150s + uncapped). That result is real but does NOT generalise to "swap is fine". + This leg reads back across the whole range in a Weyl-sequence order, so + most reads must fault a page in. + + It matters because it is databasev2 2's `resident: keys` access pattern: + that design reads rows back from a log larger than RAM by construction. + + Two runs, identical except for the cap, reading the SAME key order: + - control (RAND_FIT_MB): cap does not bind, everything resident + - over-cap (RAND_CAP_MB): ~a third of the rows fit; swap ON, because + with swap off this configuration is simply SIGKILLed (see ceiling()) + The headline is collapse_x, the throughput ratio between them. Tolerances + are wide: the over-cap half is swap I/O, so its absolute numbers are the + box's, while the RATIO is the property of the engine.""" + if cap_wrapper(RAND_FIT_MB, 0) is None: + ok("randread: SKIPPED -- no rootless cgroup v2 memory cap on this host") + return + res = {} + for legname, cap_mb, swap_mb in (("resident", RAND_FIT_MB, 0), + ("overcap", RAND_CAP_MB, 256)): + w = cap_wrapper(cap_mb, swap_mb) + pr = subprocess.run(w + [BIN, "randread", str(RAND_N), str(RAND_R)], + stdout=subprocess.PIPE, stderr=subprocess.STDOUT, + text=True, env=dict(os.environ), timeout=900) + lines = pr.stdout.splitlines() + ops, p50, p99, hits, filled = parse_randread(lines) + key = f"randread.{legname}" + if pr.returncode != 0 or ops is None: + bad(f"{key}: run failed", f"rc={pr.returncode} {(lines[-1:] or ['no output'])[0][:120]}") + return + if hits != RAND_R: + # a collapse measured over reads that did not resolve is noise + bad(f"{key}: only {hits}/{RAND_R} reads resolved", "keys must all exist") + return + metrics[f"{key}.ops_sec"] = ops + metrics[f"{key}.read_p50us"] = p50 + metrics[f"{key}.read_p99us"] = p99 + metrics[f"{key}.filled_rss_kb"] = filled + res[legname] = ops + ok(f"{key}: {ops} reads/sec, p50 {p50}us p99 {p99}us, {filled} KiB after fill") + collapse = res["resident"] // max(res["overcap"], 1) + metrics["randread.collapse_x"] = collapse + if collapse < 2: + bad("randread: NO collapse -- the cap did not bind", + f"{RAND_N} rows fit under {RAND_CAP_MB} MiB, resize the leg") + else: + ok(f"randread: random reads over an oversized table collapse {collapse}x " + f"({res['resident']} -> {res['overcap']} reads/sec)") + + def main(): # --check : gate-only evaluation of a recorded run — the # gate-bites smoke doctors a copy and this mode must FAIL on it @@ -447,6 +527,7 @@ def main(): durability(metrics) growth(metrics) ceiling(metrics) + randread(metrics) os.makedirs(RESULTS_DIR, exist_ok=True) stamp = time.strftime("%Y%m%d-%H%M%S") out = os.path.join(RESULTS_DIR, f"run-{stamp}{'-quick' if QUICK else ''}.json")