diff --git a/docs/examples/db-bench/main.wo b/docs/examples/db-bench/main.wo index 548da08..7fa3d1c 100644 --- a/docs/examples/db-bench/main.wo +++ b/docs/examples/db-bench/main.wo @@ -651,6 +651,120 @@ fn replayseed_mode(n: Int, m: Int) -> Int { return 0; } +-- databasev2 2 task 7: the residency A/B, one mode per table so the two runs +-- differ ONLY in the annotation. Same fill, same Weyl key order, same read +-- count as randread_mode above — the comparison is against that leg's own +-- 273x swap figure, measured on the same box under the same cap. +-- The WIDE half of the residency A/B. Same structure as kread_all/kread_keys +-- but with Text columns, which is the only shape where dropping a payload +-- frees anything: an Int's value is its inline slot word, a Text's is a +-- separate allocation. +fn wread_all(n: Int, r: Int) -> Int { + let pad = "0123456789abcdef0123456789abcdef"; + let i = 1; + while i <= n { + insert WideA { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" }; + i = i + 1; + } + print("wreadfilled ${n} ${self_rss_kb()}"); + let h: map = {}; + let hits = 0; + let t0 = time.ticks(); + let j = 0; + while j < r { + let key = 1 + (j * 2654435761) % n; + let o0 = time.ticks(); + for row in from x in WideA where x.k == key take 1 select x { + if len(row.a) > 0 { hits = hits + 1; } + } + hist_add(h, time.ticks() - o0); + j = j + 1; + } + let el = time.ticks() - t0; + report("wreadall", r, el, h); + print("wreadallrss ${self_rss_kb()} ${hits}"); + return 0; +} + +fn wread_keys(n: Int, r: Int) -> Int { + let pad = "0123456789abcdef0123456789abcdef"; + let i = 1; + while i <= n { + insert WideK { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" }; + i = i + 1; + } + print("wreadfilled ${n} ${self_rss_kb()}"); + let h: map = {}; + let hits = 0; + let t0 = time.ticks(); + let j = 0; + while j < r { + let key = 1 + (j * 2654435761) % n; + let o0 = time.ticks(); + for row in from x in WideK where x.k == key take 1 select x { + if len(row.a) > 0 { hits = hits + 1; } + } + hist_add(h, time.ticks() - o0); + j = j + 1; + } + let el = time.ticks() - t0; + report("wreadkeys", r, el, h); + print("wreadkeysrss ${self_rss_kb()} ${hits}"); + return 0; +} + +fn kread_all(n: Int, r: Int) -> Int { + let i = 1; + while i <= n { + insert RowA { k: i, v: item_v(i) }; + i = i + 1; + } + print("kreadfilled ${n} ${self_rss_kb()}"); + let h: map = {}; + let hits = 0; + let t0 = time.ticks(); + let j = 0; + while j < r { + let key = 1 + (j * 2654435761) % n; + let o0 = time.ticks(); + for row in from x in RowA where x.k == key take 1 select x { + if row.v == item_v(key) { hits = hits + 1; } + } + hist_add(h, time.ticks() - o0); + j = j + 1; + } + let el = time.ticks() - t0; + report("kreadall", r, el, h); + print("kreadallrss ${self_rss_kb()} ${hits}"); + return 0; +} + +fn kread_keys(n: Int, r: Int) -> Int { + let i = 1; + while i <= n { + insert RowK { k: i, v: item_v(i) }; + i = i + 1; + } + print("kreadfilled ${n} ${self_rss_kb()}"); + let h: map = {}; + let hits = 0; + let t0 = time.ticks(); + let j = 0; + while j < r { + let key = 1 + (j * 2654435761) % n; + let o0 = time.ticks(); + for row in from x in RowK where x.k == key take 1 select x { + if row.v == item_v(key) { hits = hits + 1; } + } + hist_add(h, time.ticks() - o0); + j = j + 1; + } + let el = time.ticks() - t0; + report("kreadkeys", r, el, h); + print("kreadkeysrss ${self_rss_kb()} ${hits}"); + return 0; +} + fn randread_mode(n: Int, r: Int) -> Int { let bref = insert Bucket { tag: "randread" }; let i = 1; @@ -824,6 +938,25 @@ fn main(args: multi Text) -> Int { } return replayseed_mode(n, mm); } + if args[0] == "wreadall" or args[0] == "wreadkeys" { + if len(args) < 3 { print_err("usage: wreadall|wreadkeys N R"); return 2; } + let wn = parse_int(args[1]); + let wr = parse_int(args[2]); + if wn == nil or wr == nil { print_err("db-bench: N and R must be positive"); return 2; } + if args[0] == "wreadall" { return wread_all(wn, wr); } + return wread_keys(wn, wr); + } + if args[0] == "kreadall" or args[0] == "kreadkeys" { + if len(args) < 3 { print_err("usage: kreadall|kreadkeys N R"); return 2; } + let n = parse_int(args[1]); + let rr = parse_int(args[2]); + if n == nil or rr == nil { + print_err("db-bench: N and R must be positive numbers"); + return 2; + } + if args[0] == "kreadall" { return kread_all(n, rr); } + return kread_keys(n, rr); + } if args[0] == "randread" { if len(args) < 3 { return usage(); diff --git a/docs/examples/db-bench/types.wo b/docs/examples/db-bench/types.wo index 25639c2..9f924e9 100644 --- a/docs/examples/db-bench/types.wo +++ b/docs/examples/db-bench/types.wo @@ -57,3 +57,47 @@ class MixJob { class Flood { n: Int } + +-- databasev2 2 task 7: the residency A/B. These two are IDENTICAL except for +-- the `resident` annotation, so a difference between them is the storage +-- mode's doing and nothing else's. Item above carries a `ref` and a second +-- index, which would confound the comparison. +-- +-- The question they exist to answer: iteration 1 measured a 273x collapse for +-- random reads over a table larger than RAM, on demand-paged anonymous memory. +-- `resident: keys` reads rows back with pread through the page cache instead. +-- If that is not materially better than 273x, the mode buys nothing the kernel +-- was not already doing. +@table(name: "rowsall", index: [k], durable: true, resident: all) +class RowA { + k: Int + v: Int +} + +@table(name: "rowskeys", index: [k], durable: true, resident: keys) +class RowK { + k: Int + v: Int +} + +-- databasev2 2 task 7, second A/B: the WIDE shape. The Int-only pair above +-- cannot show what keys-residency is for — dropping a payload frees each +-- field's VALUE, and an Int's value IS its inline slot word, so nothing is +-- freed and the slab stays allocated either way. A row with Text columns +-- drags separate db_text allocations that dropping genuinely releases, which +-- is the only shape where the mode can pay for itself. +@table(name: "wideall", index: [k], durable: true, resident: all) +class WideA { + k: Int + a: Text + b: Text + note: Text +} + +@table(name: "widekeys", index: [k], durable: true, resident: keys) +class WideK { + k: Int + a: Text + b: Text + note: Text +} diff --git a/docs/stories/databasev2/02-table-storage-modes.md b/docs/stories/databasev2/02-table-storage-modes.md index 3d4681b..15e9374 100644 --- a/docs/stories/databasev2/02-table-storage-modes.md +++ b/docs/stories/databasev2/02-table-storage-modes.md @@ -66,7 +66,7 @@ declared per-table policy. Durability is untouched and unconditional. | 5c | shared borrow/release accessor, then id→offset storage | ✅ `2e347de` (accessor, pure refactor, `db-bench --quick` 85/0), `18ce4d5` (offset storage), `f9c36ef` (insert + boot wiring) | | 5d | rewire the readers: remaining `wo_row_ptr` sites, slab scans, FK restrict, `@unique` across the boundary | ✅ `11a92df` (db.c), + this commit (table.c, wal.c, compaction). Updates **refused**, not rewired — see below | | 6 | the two runtime refusals (no-`WO_DATA`, the byte budget) | ⬜ | -| 7 | measure, gate, document, close out | ⬜ | +| 7 | measure, gate, document, close out | 🔄 measured 2026-08-30 (below); gate + closeout outstanding | **The `durable` half is complete and usable.** A volatile table is a full table in-process — same indexes, same `@unique`, same FK restrict, same query surface @@ -208,6 +208,76 @@ Met: cost becomes at most K+1 reads and replay O(K²) per row, independent of when a checkpoint fires. Limitations 2 and 3 above both fall to it. +### Task 7 — measured 2026-08-30, and the answer is qualified + +**The question**, in the words this file has carried since the iteration was +written: `pread` through the page cache should beat the **273×** collapse +iteration 1 measured for demand-paged anonymous memory, "and the whole value of +`resident: keys` rests on how much better." + +**Method.** Two tables identical except the annotation, so any difference is the +storage mode's doing: 200 000 rows, 40 000 reads in the same Weyl key order, +`WO_SHARDS=1`, WAL on **ext4** (not `/tmp`, which is tmpfs here and would have +put the "log" in RAM), memory capped with a rootless cgroup v2 scope. + +**First attempt measured the wrong thing, and is worth recording.** With +Int-only rows the two modes were indistinguishable — 6 061 vs 5 599 ops/s, RSS +15.0 MB vs 14.3 MB. The cause is structural: `wo_row_drop_payload` frees each +field's *value* and returns the slot to a free list, **but never releases the +slab**, and an `Int`'s value IS its inline slot word. So dropping an Int-only +row frees nothing at all. The mode cannot help that shape, and a benchmark built +on it would have condemned the feature for the wrong reason. + +**The wide shape (one `Int`, three `Text`) is where the mode can act.** + +| 200k rows, 40k reads | ops/s | p50 | p99 | RSS | +| --- | --- | --- | --- | --- | +| `resident: all`, no pressure (256 MB) | 1 354 554 | 1 µs | 2 µs | 87.5 MB | +| `resident: keys`, no pressure (256 MB) | 320 053 | 3 µs | 5 µs | **34.4 MB** | +| `resident: all`, 48 MB cap | 12 854 | 67 µs | 231 µs | 48.1 MB | +| `resident: keys`, 48 MB cap | **19 635** | 65 µs | 227 µs | 34.4 MB | + +The 48 MB cap is chosen to sit between the two resident sets: `resident: all` +needs 87 MB and must page, `resident: keys` needs 34 MB and fits. + +**What it buys.** + +- **2.55× smaller resident set** — 34.4 MB against 87.5 MB. This is the real, + unambiguous win, and it is the thing the mode was built for. +- **A far gentler degradation curve**: under the cap `resident: all` collapses + **105×** from its own uncapped throughput, `resident: keys` only **16×**. +- **1.53× faster than swapping at the same cap** — 19 635 vs 12 854 ops/s. + +**What it costs.** + +- **4.2× slower reads when memory is not tight** (320k vs 1.35M ops/s). A + `pread` and a fold per row against a pointer dereference. +- **Writes are markedly slower**, uncosted by any design document so far: the + keys fill of 200 000 rows did not finish inside two minutes where the resident + fill plus 40 000 reads did. The per-insert drop-and-re-point work is the + difference; both tables are `durable: true`, so the WAL is not. + +**The finding that matters most, and it was not anticipated.** `resident: keys` +is only 1.53× faster than swapping under the cap, not the order of magnitude the +design implies — because **cgroup memory limits charge the page cache**. The WAL +here is 37 MB; the resident set is 34 MB; a 48 MB cap cannot hold both, so the +log's pages are evicted and every `pread` reaches the disk. Moving rows out of +the heap and into a file does **not** escape a container memory limit — the +cache the design leans on is charged to the same cgroup. The mode's premise, +"the kernel's page cache will hold the hot rows", fails in precisely the +containerised deployment it targets. + +**Verdict.** The feature is worth keeping, but for a narrower reason than +claimed: it lets a given amount of RAM hold ~2.5× more data, and degrades far +more gracefully than swapping. It is **not** a way to make an +over-capacity table fast — under a hard memory cap it is within 1.5× of simply +letting the kernel swap. The honest guidance is "use it to fit more, not to go +faster", and the docs should say so. + +**Still outstanding for task 7:** wire these legs into `scripts/db-bench.py` +with tolerances and a baseline entry, and re-measure the `resident: all` read +baseline to confirm no cost for a feature not used. + Outstanding: - **Given** `durable: true` and no `WO_DATA`, **when** the program starts,