From 841f6c8f3fd9a054ec97c93ad8e7c7c0624a734f Mon Sep 17 00:00:00 2001 From: "shoney.arickathil" Date: Sun, 30 Aug 2026 15:50:34 +0200 Subject: [PATCH] feat(db2-keys): gate the residency measurement, close out task 7 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - new `residency` leg in db-bench.py driving docs/examples/residency-bench: two tables identical except the annotation, control cap + binding cap - ITS OWN PROGRAM, not a db-bench mode: declaring a resident: keys table is a WHOLE-PROGRAM constraint, so the no-WO_DATA refusal fires for every mode in the module. Putting those classes in db-bench's shared types made growth/ceiling/randread — which run without WO_DATA — refuse to start. Caught by running the leg, not by reading it - gates the RATIOS, waives the absolutes: ops/sec under a cap is swap and disk I/O and belongs to the box. Same split randread makes - rss_ratio 2.55 floor 2.0 tol 10% (structural, like bytes_per_row); overcap_vs_swap_x 1.53 floor 1.0; in_ram_cost_x 4.23 ceiling 8.0; all_collapse_x 105.4 floor 2.0 - all_collapse_x exists because the leg's FIRST run silently measured nothing: at QUICK's 40k rows a 48 MiB cap binds neither mode, so the "over-cap" half was not over cap. The cap now scales with N and the leg asserts it binds - verified the gate bites: rss_ratio 1.4, overcap_vs_swap_x 0.6 and in_ram_cost_x 12.0 are all rejected - task 7 closed: both criteria moved to Met with how each was verified Co-Authored-By: Claude Opus 5 (1M context) (cherry picked from commit a310496664982c372f51b43113465eb8ad9e9fb5) --- bench/baseline.json | 24 ++ docs/examples/db-bench/main.wo | 211 ------------------ docs/examples/db-bench/types.wo | 44 ---- docs/examples/residency-bench/main.wo | 152 +++++++++++++ docs/examples/residency-bench/types.wo | 34 +++ docs/examples/residency-bench/wo.toml | 6 + .../databasev2/02-table-storage-modes.md | 59 +++-- scripts/db-bench.py | 157 ++++++++++++- 8 files changed, 416 insertions(+), 271 deletions(-) create mode 100644 docs/examples/residency-bench/main.wo create mode 100644 docs/examples/residency-bench/types.wo create mode 100644 docs/examples/residency-bench/wo.toml diff --git a/bench/baseline.json b/bench/baseline.json index cc50d72..4c70033 100644 --- a/bench/baseline.json +++ b/bench/baseline.json @@ -863,5 +863,29 @@ "floor": 100, "tolerance_pct": 100, "value": 3 + }, + "residency.all_collapse_x": { + "dir": "higher", + "floor": 2.0, + "tolerance_pct": 100, + "value": 105.4 + }, + "residency.in_ram_cost_x": { + "dir": "lower", + "floor": 8.0, + "tolerance_pct": 50, + "value": 4.23 + }, + "residency.overcap_vs_swap_x": { + "dir": "higher", + "floor": 1.0, + "tolerance_pct": 100, + "value": 1.53 + }, + "residency.rss_ratio": { + "dir": "higher", + "floor": 2.0, + "tolerance_pct": 10, + "value": 2.55 } } \ No newline at end of file diff --git a/docs/examples/db-bench/main.wo b/docs/examples/db-bench/main.wo index dc87dc3..548da08 100644 --- a/docs/examples/db-bench/main.wo +++ b/docs/examples/db-bench/main.wo @@ -651,190 +651,6 @@ fn replayseed_mode(n: Int, m: Int) -> Int { return 0; } --- databasev2 2 task 7: the residency A/B, one mode per table so the two runs --- differ ONLY in the annotation. Same fill, same Weyl key order, same read --- count as randread_mode above — the comparison is against that leg's own --- 273x swap figure, measured on the same box under the same cap. --- The WIDE half of the residency A/B. Same structure as kread_all/kread_keys --- but with Text columns, which is the only shape where dropping a payload --- frees anything: an Int's value is its inline slot word, a Text's is a --- separate allocation. --- databasev2 2 task 7, the GB-scale leg. Same A/B as wread_*, but each row --- carries ~2 KB of Text so a realistic data volume is reachable in a few --- hundred thousand inserts rather than millions — the insert path is --- fsync-bound at roughly 2 000 rows/s, so row COUNT is the expensive axis and --- row SIZE is nearly free. --- --- This is the shape the mode actually claims: data far larger than the cap, --- with only the id map, the indexes and the (never-released) slabs resident. -fn heavy_pad() -> Text { - let p = "0123456789abcdef0123456789abcdef"; - let out = ""; - let i = 0; - while i < 20 { out = out .. p; i = i + 1; } - return out; -} - -fn hread_all(n: Int, r: Int) -> Int { - let pad = heavy_pad(); - let i = 1; - while i <= n { - insert WideA { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}" }; - i = i + 1; - } - print("hreadfilled ${n} ${self_rss_kb()}"); - let h: map = {}; - let hits = 0; - let t0 = time.ticks(); - let j = 0; - while j < r { - let key = 1 + (j * 2654435761) % n; - let o0 = time.ticks(); - for row in from x in WideA where x.k == key take 1 select x { - if len(row.a) > 0 { hits = hits + 1; } - } - hist_add(h, time.ticks() - o0); - j = j + 1; - } - let el = time.ticks() - t0; - report("hreadall", r, el, h); - print("hreadallrss ${self_rss_kb()} ${hits}"); - return 0; -} - -fn hread_keys(n: Int, r: Int) -> Int { - let pad = heavy_pad(); - let i = 1; - while i <= n { - insert WideK { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}" }; - i = i + 1; - } - print("hreadfilled ${n} ${self_rss_kb()}"); - let h: map = {}; - let hits = 0; - let t0 = time.ticks(); - let j = 0; - while j < r { - let key = 1 + (j * 2654435761) % n; - let o0 = time.ticks(); - for row in from x in WideK where x.k == key take 1 select x { - if len(row.a) > 0 { hits = hits + 1; } - } - hist_add(h, time.ticks() - o0); - j = j + 1; - } - let el = time.ticks() - t0; - report("hreadkeys", r, el, h); - print("hreadkeysrss ${self_rss_kb()} ${hits}"); - return 0; -} - -fn wread_all(n: Int, r: Int) -> Int { - let pad = "0123456789abcdef0123456789abcdef"; - let i = 1; - while i <= n { - insert WideA { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" }; - i = i + 1; - } - print("wreadfilled ${n} ${self_rss_kb()}"); - let h: map = {}; - let hits = 0; - let t0 = time.ticks(); - let j = 0; - while j < r { - let key = 1 + (j * 2654435761) % n; - let o0 = time.ticks(); - for row in from x in WideA where x.k == key take 1 select x { - if len(row.a) > 0 { hits = hits + 1; } - } - hist_add(h, time.ticks() - o0); - j = j + 1; - } - let el = time.ticks() - t0; - report("wreadall", r, el, h); - print("wreadallrss ${self_rss_kb()} ${hits}"); - return 0; -} - -fn wread_keys(n: Int, r: Int) -> Int { - let pad = "0123456789abcdef0123456789abcdef"; - let i = 1; - while i <= n { - insert WideK { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" }; - i = i + 1; - } - print("wreadfilled ${n} ${self_rss_kb()}"); - let h: map = {}; - let hits = 0; - let t0 = time.ticks(); - let j = 0; - while j < r { - let key = 1 + (j * 2654435761) % n; - let o0 = time.ticks(); - for row in from x in WideK where x.k == key take 1 select x { - if len(row.a) > 0 { hits = hits + 1; } - } - hist_add(h, time.ticks() - o0); - j = j + 1; - } - let el = time.ticks() - t0; - report("wreadkeys", r, el, h); - print("wreadkeysrss ${self_rss_kb()} ${hits}"); - return 0; -} - -fn kread_all(n: Int, r: Int) -> Int { - let i = 1; - while i <= n { - insert RowA { k: i, v: item_v(i) }; - i = i + 1; - } - print("kreadfilled ${n} ${self_rss_kb()}"); - let h: map = {}; - let hits = 0; - let t0 = time.ticks(); - let j = 0; - while j < r { - let key = 1 + (j * 2654435761) % n; - let o0 = time.ticks(); - for row in from x in RowA where x.k == key take 1 select x { - if row.v == item_v(key) { hits = hits + 1; } - } - hist_add(h, time.ticks() - o0); - j = j + 1; - } - let el = time.ticks() - t0; - report("kreadall", r, el, h); - print("kreadallrss ${self_rss_kb()} ${hits}"); - return 0; -} - -fn kread_keys(n: Int, r: Int) -> Int { - let i = 1; - while i <= n { - insert RowK { k: i, v: item_v(i) }; - i = i + 1; - } - print("kreadfilled ${n} ${self_rss_kb()}"); - let h: map = {}; - let hits = 0; - let t0 = time.ticks(); - let j = 0; - while j < r { - let key = 1 + (j * 2654435761) % n; - let o0 = time.ticks(); - for row in from x in RowK where x.k == key take 1 select x { - if row.v == item_v(key) { hits = hits + 1; } - } - hist_add(h, time.ticks() - o0); - j = j + 1; - } - let el = time.ticks() - t0; - report("kreadkeys", r, el, h); - print("kreadkeysrss ${self_rss_kb()} ${hits}"); - return 0; -} - fn randread_mode(n: Int, r: Int) -> Int { let bref = insert Bucket { tag: "randread" }; let i = 1; @@ -1008,33 +824,6 @@ fn main(args: multi Text) -> Int { } return replayseed_mode(n, mm); } - if args[0] == "hreadall" or args[0] == "hreadkeys" { - if len(args) < 3 { print_err("usage: hreadall|hreadkeys N R"); return 2; } - let hn = parse_int(args[1]); - let hr = parse_int(args[2]); - if hn == nil or hr == nil { print_err("db-bench: N and R must be positive"); return 2; } - if args[0] == "hreadall" { return hread_all(hn, hr); } - return hread_keys(hn, hr); - } - if args[0] == "wreadall" or args[0] == "wreadkeys" { - if len(args) < 3 { print_err("usage: wreadall|wreadkeys N R"); return 2; } - let wn = parse_int(args[1]); - let wr = parse_int(args[2]); - if wn == nil or wr == nil { print_err("db-bench: N and R must be positive"); return 2; } - if args[0] == "wreadall" { return wread_all(wn, wr); } - return wread_keys(wn, wr); - } - if args[0] == "kreadall" or args[0] == "kreadkeys" { - if len(args) < 3 { print_err("usage: kreadall|kreadkeys N R"); return 2; } - let n = parse_int(args[1]); - let rr = parse_int(args[2]); - if n == nil or rr == nil { - print_err("db-bench: N and R must be positive numbers"); - return 2; - } - if args[0] == "kreadall" { return kread_all(n, rr); } - return kread_keys(n, rr); - } if args[0] == "randread" { if len(args) < 3 { return usage(); diff --git a/docs/examples/db-bench/types.wo b/docs/examples/db-bench/types.wo index 9f924e9..25639c2 100644 --- a/docs/examples/db-bench/types.wo +++ b/docs/examples/db-bench/types.wo @@ -57,47 +57,3 @@ class MixJob { class Flood { n: Int } - --- databasev2 2 task 7: the residency A/B. These two are IDENTICAL except for --- the `resident` annotation, so a difference between them is the storage --- mode's doing and nothing else's. Item above carries a `ref` and a second --- index, which would confound the comparison. --- --- The question they exist to answer: iteration 1 measured a 273x collapse for --- random reads over a table larger than RAM, on demand-paged anonymous memory. --- `resident: keys` reads rows back with pread through the page cache instead. --- If that is not materially better than 273x, the mode buys nothing the kernel --- was not already doing. -@table(name: "rowsall", index: [k], durable: true, resident: all) -class RowA { - k: Int - v: Int -} - -@table(name: "rowskeys", index: [k], durable: true, resident: keys) -class RowK { - k: Int - v: Int -} - --- databasev2 2 task 7, second A/B: the WIDE shape. The Int-only pair above --- cannot show what keys-residency is for — dropping a payload frees each --- field's VALUE, and an Int's value IS its inline slot word, so nothing is --- freed and the slab stays allocated either way. A row with Text columns --- drags separate db_text allocations that dropping genuinely releases, which --- is the only shape where the mode can pay for itself. -@table(name: "wideall", index: [k], durable: true, resident: all) -class WideA { - k: Int - a: Text - b: Text - note: Text -} - -@table(name: "widekeys", index: [k], durable: true, resident: keys) -class WideK { - k: Int - a: Text - b: Text - note: Text -} diff --git a/docs/examples/residency-bench/main.wo b/docs/examples/residency-bench/main.wo new file mode 100644 index 0000000..6d40f32 --- /dev/null +++ b/docs/examples/residency-bench/main.wo @@ -0,0 +1,152 @@ +-- residency-bench — databasev2 2 task 7. +-- +-- Two tables identical except the `resident` annotation (see types.wo), the +-- same fill, the same Weyl key order, the same read count. Run under a cgroup +-- memory cap by scripts/db-bench.py's `residency` leg: +-- +-- control : a cap that binds neither mode -> what the mode COSTS +-- over-cap: a cap between the two resident sets -> what the mode BUYS +-- +-- Prints the same shape db-bench's own legs do, so the harness parses it with +-- the same helpers: ` ` plus an rss/hits line. +use fs +use time + +fn hist_add(mut h: map, us: Int) { + let b = us; + if b < 0 { + b = 0; + } + if b > 20000 { + b = 20000; + } + if has(h, b) { + set(h, b, get(h, b) + 1); + } else { + set(h, b, 1); + } +} + +fn hist_pct(h: map, total: Int, pct: Int) -> Int { + let target = total * pct / 100; + if target < 1 { + target = 1; + } + let seen = 0; + let b = 0; + while b <= 20000 { + if has(h, b) { + seen = seen + get(h, b); + if seen >= target { + return b; + } + } + b = b + 1; + } + return 20000; +} + +fn report(op: Text, n: Int, total_us: Int, h: map) { + let us = total_us; + if us < 1 { + us = 1; + } + let rate = n * 1000000 / us; + print("${op} ${n} ${rate} ${hist_pct(h, n, 50)} ${hist_pct(h, n, 99)}"); +} + +fn self_rss_kb() -> Int { + let st = try fs.read_all("/proc/self/status", 16384) catch (e) ""; + let i = index_of(st, "VmRSS:"); + if i < 0 { + return 0; + } + let rest = substr(st, i + 6, 24); + let n = 0; + let j = 0; + while j < len(rest) { + let c = byte_at(rest, j); + if c >= 48 and c <= 57 { + n = n * 10 + (c - 48); + } else { + if n > 0 { + return n; + } + } + j = j + 1; + } + return n; +} + +fn wide_pad() -> Text { + return "0123456789abcdef0123456789abcdef"; +} + +fn wread_all(n: Int, r: Int) -> Int { + let pad = wide_pad(); + let i = 1; + while i <= n { + insert WideA { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" }; + i = i + 1; + } + print("wreadfilled ${n} ${self_rss_kb()}"); + let h: map = {}; + let hits = 0; + let t0 = time.ticks(); + let j = 0; + while j < r { + let key = 1 + (j * 2654435761) % n; + let o0 = time.ticks(); + for row in from x in WideA where x.k == key take 1 select x { + if len(row.a) > 0 { hits = hits + 1; } + } + hist_add(h, time.ticks() - o0); + j = j + 1; + } + report("wreadall", r, time.ticks() - t0, h); + print("wreadallrss ${self_rss_kb()} ${hits}"); + return 0; +} + +fn wread_keys(n: Int, r: Int) -> Int { + let pad = wide_pad(); + let i = 1; + while i <= n { + insert WideK { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" }; + i = i + 1; + } + print("wreadfilled ${n} ${self_rss_kb()}"); + let h: map = {}; + let hits = 0; + let t0 = time.ticks(); + let j = 0; + while j < r { + let key = 1 + (j * 2654435761) % n; + let o0 = time.ticks(); + for row in from x in WideK where x.k == key take 1 select x { + if len(row.a) > 0 { hits = hits + 1; } + } + hist_add(h, time.ticks() - o0); + j = j + 1; + } + report("wreadkeys", r, time.ticks() - t0, h); + print("wreadkeysrss ${self_rss_kb()} ${hits}"); + return 0; +} + +fn main(args: multi Text) -> Int { + if len(args) < 3 { + print_err("usage: residency-bench wreadall|wreadkeys N R"); + return 2; + } + let n = parse_int(args[1]); + let r = parse_int(args[2]); + if n == nil or r == nil { + print_err("residency-bench: N and R must be positive numbers"); + return 2; + } + if args[0] == "wreadall" { return wread_all(n, r); } + if args[0] == "wreadkeys" { return wread_keys(n, r); } + print_err("residency-bench: unknown mode ${args[0]}"); + return 2; +} diff --git a/docs/examples/residency-bench/types.wo b/docs/examples/residency-bench/types.wo new file mode 100644 index 0000000..3befcd7 --- /dev/null +++ b/docs/examples/residency-bench/types.wo @@ -0,0 +1,34 @@ +-- databasev2 2 task 7: the residency A/B. +-- +-- These two tables are IDENTICAL except for the `resident` annotation, so any +-- difference between them is the storage mode's doing and nothing else's. +-- +-- WHY THIS IS ITS OWN PROGRAM rather than a mode inside db-bench: declaring a +-- `resident: keys` table is a WHOLE-PROGRAM constraint. Its rows live only in +-- the write-ahead log, so the runtime refuses to start without WO_DATA — and +-- that refusal fires for every mode in the module, including ones that never +-- touch the table. Putting these classes in db-bench's shared module made its +-- `growth`, `ceiling` and `randread` legs, which deliberately run WITHOUT +-- WO_DATA, refuse to start. +-- +-- The shape is WIDE on purpose. An Int-only pair shows the two modes as +-- indistinguishable, and that is structural: dropping a payload frees each +-- field's VALUE, and an Int's value IS its inline slot word, so nothing is +-- freed and the slab stays allocated either way. Only rows with Text columns — +-- separate allocations that dropping genuinely releases — can show what the +-- mode is for. +@table(name: "wideall", index: [k], durable: true, resident: all) +class WideA { + k: Int + a: Text + b: Text + note: Text +} + +@table(name: "widekeys", index: [k], durable: true, resident: keys) +class WideK { + k: Int + a: Text + b: Text + note: Text +} diff --git a/docs/examples/residency-bench/wo.toml b/docs/examples/residency-bench/wo.toml new file mode 100644 index 0000000..d26e30e --- /dev/null +++ b/docs/examples/residency-bench/wo.toml @@ -0,0 +1,6 @@ +name = "residency-bench" +version = "0.1.0" +description = "databasev2 2 task 7: resident: all vs resident: keys, identical tables, one annotation apart" + +[runtime] +wo = ">= 0.1" diff --git a/docs/stories/databasev2/02-table-storage-modes.md b/docs/stories/databasev2/02-table-storage-modes.md index 15e9374..d26d96d 100644 --- a/docs/stories/databasev2/02-table-storage-modes.md +++ b/docs/stories/databasev2/02-table-storage-modes.md @@ -66,7 +66,7 @@ declared per-table policy. Durability is untouched and unconditional. | 5c | shared borrow/release accessor, then id→offset storage | ✅ `2e347de` (accessor, pure refactor, `db-bench --quick` 85/0), `18ce4d5` (offset storage), `f9c36ef` (insert + boot wiring) | | 5d | rewire the readers: remaining `wo_row_ptr` sites, slab scans, FK restrict, `@unique` across the boundary | ✅ `11a92df` (db.c), + this commit (table.c, wal.c, compaction). Updates **refused**, not rewired — see below | | 6 | the two runtime refusals (no-`WO_DATA`, the byte budget) | ⬜ | -| 7 | measure, gate, document, close out | 🔄 measured 2026-08-30 (below); gate + closeout outstanding | +| 7 | measure, gate, document, close out | ✅ measured, gated and documented 2026-08-30 | **The `durable` half is complete and usable.** A volatile table is a full table in-process — same indexes, same `@unique`, same FK restrict, same query surface @@ -274,9 +274,49 @@ over-capacity table fast — under a hard memory cap it is within 1.5× of simpl letting the kernel swap. The honest guidance is "use it to fit more, not to go faster", and the docs should say so. -**Still outstanding for task 7:** wire these legs into `scripts/db-bench.py` -with tolerances and a baseline entry, and re-measure the `resident: all` read -baseline to confirm no cost for a feature not used. +**Gated 2026-08-30.** `scripts/db-bench.py` grew a `residency` leg driving +`docs/examples/residency-bench` — its own program, because declaring a +`resident: keys` table is a WHOLE-PROGRAM constraint: the runtime refuses to +start without `WO_DATA`, for every mode in the module. Putting those classes in +db-bench's shared types made `growth`, `ceiling` and `randread` — which +deliberately run without `WO_DATA` — refuse to start. That regression was caught +by running the leg, not by reading it. + +What is gated, and what deliberately is not, follows `randread`'s existing +split: the absolute ops/sec under a cap is swap and disk I/O and belongs to the +box, so it is recorded and waived; the RATIOS are the engine's property. + +| metric | baseline | floor | tolerance | +| --- | --- | --- | --- | +| `residency.rss_ratio` | 2.55 | 2.0 | 10% | +| `residency.overcap_vs_swap_x` | 1.53 | 1.0 | 100% | +| `residency.in_ram_cost_x` | 4.23 | 8.0 (ceiling) | 50% | +| `residency.all_collapse_x` | 105.4 | 2.0 | 100% | + +`rss_ratio` carries the tight tolerance because footprint is structural — the +same class of number as `bytes_per_row`. The two throughput ratios are guarded +by their FLOORS rather than their bands, which is this harness's established +answer to a metric whose absolute value belongs to the disk. `all_collapse_x` +exists only to assert the cap actually binds; a leg whose "over-cap" half is not +over cap silently measures nothing, which is exactly what the first run of this +leg did. + +Verified by feeding the gate a breaching run: `rss_ratio` 1.4, +`overcap_vs_swap_x` 0.6 and `in_ram_cost_x` 12.0 are all rejected. +- **Given** the `resident: all` read baseline, **when** re-measured, **then** + inside tolerance — no cost for a feature not used. ✅ Verified as a + by-product of the residency leg: `resident: all` is unchanged at 1 354 554 + reads/sec uncapped, and every other db-bench leg still runs, which the + keys-resident classes had briefly broken by forcing `WO_DATA` module-wide. +- **Given** a `resident: keys` table larger than RAM, **when** read randomly, + **then** its read cost is measured against the resident baseline on its own + read path. ✅ Measured 2026-08-30 and the answer is qualified: **1.53×** + faster than letting the kernel swap under a cap that binds one and not the + other — real, but nowhere near iteration 1's 273× swap figure would suggest, + because cgroup limits charge the page cache, so moving rows to a file does + not escape a container memory limit. The unambiguous win is footprint: + **2.55×** smaller resident set. Full method, numbers and the failed first + attempt above; gated by `residency.*`. Outstanding: @@ -285,17 +325,6 @@ Outstanding: every write)* - **Given** the resident footprint crossing the budget, **when** it does, **then** a refusal naming the table and the annotation. *(task 6)* -- **Given** the `resident: all` read baseline, **when** re-measured, **then** - inside tolerance — no cost for a feature not used. *(task 7)* -- **Given** a `resident: keys` table larger than RAM, **when** read randomly, - **then** its read cost is **measured against the resident baseline on its own - read path**, not inherited from databasev2 1's swap figure. *(task 7)* — that - figure is **273×** for demand-paged anonymous memory - ([1](01-ram-ceiling-measurement.md)); `pread` through the page cache should do - better, and the whole value of `resident: keys` rests on how much better. If it - is not materially better than swapping, the design buys nothing that the - kernel was not already doing. - ## Out Of Scope - **Checkpoint and compaction** — [3](03-wal-checkpoint.md). Boot rebuilds the diff --git a/scripts/db-bench.py b/scripts/db-bench.py index 9fdd902..366fff7 100755 --- a/scripts/db-bench.py +++ b/scripts/db-bench.py @@ -13,10 +13,16 @@ Exit 0 = campaign green; 1 = gate breach or a durability leg failed. Plan deviation, disclosed: one python driver instead of bash+python — the live stdout sampling and JSON assembly are the whole job. """ -import json, os, re, subprocess, sys, time, random, shutil +import json, os, re, subprocess, sys, tempfile, time, random, shutil ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) BIN = os.path.join(ROOT, "docs/examples/db-bench/target/db-bench") +# databasev2 2 task 7. Its own program, not a mode in db-bench: declaring a +# `resident: keys` table is a WHOLE-PROGRAM constraint — the runtime refuses to +# start without WO_DATA, for every mode in the module. Putting those classes in +# db-bench's shared types made growth/ceiling/randread, which deliberately run +# WITHOUT WO_DATA, refuse to start. +RESID_BIN = os.path.join(ROOT, "docs/examples/residency-bench/target/residency-bench") WOC = os.path.join(ROOT, "compiler/_build/default/bin/woc") WOVM = os.path.join(ROOT, "runtime/wovm") BASELINE = os.path.join(ROOT, "bench/baseline.json") @@ -60,6 +66,10 @@ def build(): "-o", BIN, "--runtime", WOVM], capture_output=True, text=True) if r.returncode != 0: bad("build", r.stderr.strip()[:200]); sys.exit(1) + r = subprocess.run([WOC, "build", os.path.join(ROOT, "docs/examples/residency-bench"), + "-o", RESID_BIN, "--runtime", WOVM], capture_output=True, text=True) + if r.returncode != 0: + bad("build residency-bench", r.stderr.strip()[:200]); sys.exit(1) ok("builds") def run(args, env_extra, timeout, sample_after=None): @@ -303,6 +313,26 @@ def tolerance_for(key): if key.startswith("growth."): return 100 if key.startswith("ceiling."): return 100 if key.startswith("randread."): return 100 + # databasev2 2 task 7. Same split randread makes, for the same reason: the + # absolute ops/sec under a cap is swap and disk I/O and belongs to the box, + # so it is recorded and waived. The RATIOS are the engine's property. + # + # rss_ratio is the headline claim — keys must hold the same rows in a + # materially smaller resident set or the mode has no purpose — and it is + # STRUCTURAL: 87.5 MiB against 34.4 MiB, reproducible, the same class of + # number as bytes_per_row. It gets the same tight tolerance, and residency() + # additionally hard-fails below 2.0x regardless of drift. + if key == "residency.rss_ratio": return 10 + # in_ram_cost is a throughput ratio between two cached runs: stable in + # shape (a pread and a fold against a pointer dereference) but it moves + # with page-cache weather, so it is gated loosely rather than waived. + if key == "residency.in_ram_cost_x": return 50 + # overcap_vs_swap compares two I/O-bound runs, so BOTH halves are the box's. + # The ratio is still worth recording — it is the answer to the question the + # iteration was written to ask — but residency() guards the direction of it + # (keys must not be slower than swapping) rather than its magnitude. + if key == "residency.overcap_vs_swap_x": return 100 + if key.startswith("residency."): return 100 if key.startswith("replay."): return 100 # databasev2 4: batch SHAPE follows arrival timing, so gating it tightly # would gate the scheduler — what must hold is that the mean exceeds one @@ -558,6 +588,130 @@ def ceiling(metrics): +# databasev2 2 task 7: does `resident: keys` beat letting the kernel swap? +RESID_N = 40000 if QUICK else 200000 +RESID_R = 10000 if QUICK else 40000 +RESID_FIT_MB = 256 # control: neither mode is under pressure +# Between the two resident sets, so `all` pages and `keys` does not. This MUST +# scale with N: at QUICK's 40k rows `all` needs only ~17 MiB, so a 48 MiB cap +# binds neither mode and the comparison silently becomes "keys is slower when +# nothing is under pressure" — which is true, and not what this leg asks. +RESID_CAP_MB = 10 if QUICK else 48 + + +def parse_resid(lines, op): + """` `, plus the mode's own rss/hits line.""" + ops = p50 = p99 = rss = hits = filled = None + for l in lines: + f = l.split() + if not f: + continue + if f[0] == op and len(f) == 5: + ops, p50, p99 = int(f[2]), int(f[3]), int(f[4]) + elif f[0] == op + "rss" and len(f) == 3: + rss, hits = int(f[1]), int(f[2]) + elif f[0] == "wreadfilled" and len(f) == 3: + filled = int(f[2]) + return ops, p50, p99, rss, hits, filled + + +def residency(metrics): + """`resident: keys` against `resident: all`, on tables IDENTICAL except the + annotation, so a difference is the storage mode's doing and nothing else's. + + Measured 2026-08-30 on the WIDE shape deliberately. An Int-only pair shows + the two modes as indistinguishable, and that is structural rather than + surprising: dropping a payload frees each field's VALUE, and an Int's value + IS its inline slot word, so nothing is freed and the slab stays allocated + either way. A benchmark built on that shape would condemn the feature for a + reason that has nothing to do with the feature. + + WHAT IS GATED, and what deliberately is not. The RATIOS are the engine's + property and get real tolerances; the absolute ops/sec under a cap is swap + and disk I/O, so it belongs to the box and is recorded but waived. This is + the same split randread() already makes for the same reason. + + The headline claim is rss_ratio: keys must hold the same rows in a + materially smaller resident set, or the mode has no purpose. The measured + figure was 2.55x (34.4 MiB against 87.5 MiB).""" + if cap_wrapper(RESID_FIT_MB, 0) is None: + ok("residency: SKIPPED -- no rootless cgroup v2 memory cap on this host") + return + res = {} + for mode, op in (("all", "wreadall"), ("keys", "wreadkeys")): + for legname, cap_mb, swap_mb in (("fit", RESID_FIT_MB, 0), + ("cap", RESID_CAP_MB, 512)): + w = cap_wrapper(cap_mb, swap_mb) + data = tempfile.mkdtemp(prefix="resid-") + env = dict(os.environ) + env["WO_DATA"] = data # a keys-resident table cannot run without one + env["WO_SHARDS"] = "1" + pr = subprocess.run(w + [RESID_BIN, op, str(RESID_N), str(RESID_R)], + stdout=subprocess.PIPE, stderr=subprocess.STDOUT, + text=True, env=env, timeout=1800) + shutil.rmtree(data, ignore_errors=True) + lines = pr.stdout.splitlines() + ops, p50, p99, rss, hits, filled = parse_resid(lines, op) + key = f"residency.{mode}.{legname}" + if pr.returncode != 0 or ops is None: + bad(f"{key}: run failed", + f"rc={pr.returncode} {(lines[-1:] or ['no output'])[0][:120]}") + return + if hits != RESID_R: + # a comparison over reads that did not resolve measures nothing + bad(f"{key}: only {hits}/{RESID_R} reads resolved", "keys must all exist") + return + metrics[f"{key}.ops_sec"] = ops + metrics[f"{key}.read_p50us"] = p50 + metrics[f"{key}.read_p99us"] = p99 + metrics[f"{key}.filled_rss_kb"] = filled + res[f"{mode}.{legname}"] = (ops, filled) + ok(f"{key}: {ops} reads/sec, p50 {p50}us p99 {p99}us, {filled} KiB after fill") + + all_rss = res["all.fit"][1] + keys_rss = res["keys.fit"][1] + rss_ratio = round(all_rss / max(keys_rss, 1), 2) + metrics["residency.rss_ratio"] = rss_ratio + + # what the mode costs when memory is NOT tight: a pread and a fold per row + # against a pointer dereference + in_ram_cost = round(res["all.fit"][0] / max(res["keys.fit"][0], 1), 2) + metrics["residency.in_ram_cost_x"] = in_ram_cost + + # the question the iteration was written to answer: under a cap that binds + # `all` and not `keys`, is keys actually better than swapping? + vs_swap = round(res["keys.cap"][0] / max(res["all.cap"][0], 1), 2) + metrics["residency.overcap_vs_swap_x"] = vs_swap + + if rss_ratio < 2.0: + bad("residency: keys saves less than 2x RSS", + f"{all_rss} KiB vs {keys_rss} KiB = {rss_ratio}x -- the mode's whole purpose") + else: + ok(f"residency: keys holds the same rows in {rss_ratio}x less RSS " + f"({all_rss} -> {keys_rss} KiB)") + + # The cap must actually BIND the resident half, or the comparison is + # meaningless. randread learned this the same way; assert it rather than + # trusting the constants to stay right as N changes. + all_collapse = res["all.fit"][0] / max(res["all.cap"][0], 1) + metrics["residency.all_collapse_x"] = round(all_collapse, 2) + if all_collapse < 2.0: + bad("residency: the cap did not bind `resident: all`", + f"{res['all.fit'][0]} -> {res['all.cap'][0]} reads/sec is only " + f"{all_collapse:.2f}x; {RESID_N} rows fit under {RESID_CAP_MB} MiB, resize the leg") + return + + if vs_swap < 1.0: + bad("residency: keys is SLOWER than letting the kernel swap", + f"{res['keys.cap'][0]} vs {res['all.cap'][0]} reads/sec under a " + f"{RESID_CAP_MB} MiB cap -- the mode buys nothing here") + else: + ok(f"residency: under a {RESID_CAP_MB} MiB cap keys is {vs_swap}x swapping " + f"({res['keys.cap'][0]} vs {res['all.cap'][0]} reads/sec)") + + ok(f"residency: costs {in_ram_cost}x read throughput when memory is not tight") + + def parse_randread(lines): """ops/sec, p50, p99, resolved-read count and post-fill RSS from the sample's own randread lines. `randreadfilled` also starts with "randread", @@ -847,6 +1001,7 @@ def main(): growth(metrics) ceiling(metrics) randread(metrics) + residency(metrics) replay(metrics) checkpoint_leg(metrics) os.makedirs(RESULTS_DIR, exist_ok=True)