feat(db2-keys): gate the residency measurement, close out task 7

- new `residency` leg in db-bench.py driving docs/examples/residency-bench:
  two tables identical except the annotation, control cap + binding cap
- ITS OWN PROGRAM, not a db-bench mode: declaring a resident: keys table
  is a WHOLE-PROGRAM constraint, so the no-WO_DATA refusal fires for
  every mode in the module. Putting those classes in db-bench's shared
  types made growth/ceiling/randread — which run without WO_DATA —
  refuse to start. Caught by running the leg, not by reading it
- gates the RATIOS, waives the absolutes: ops/sec under a cap is swap
  and disk I/O and belongs to the box. Same split randread makes
- rss_ratio 2.55 floor 2.0 tol 10% (structural, like bytes_per_row);
  overcap_vs_swap_x 1.53 floor 1.0; in_ram_cost_x 4.23 ceiling 8.0;
  all_collapse_x 105.4 floor 2.0
- all_collapse_x exists because the leg's FIRST run silently measured
  nothing: at QUICK's 40k rows a 48 MiB cap binds neither mode, so the
  "over-cap" half was not over cap. The cap now scales with N and the
  leg asserts it binds
- verified the gate bites: rss_ratio 1.4, overcap_vs_swap_x 0.6 and
  in_ram_cost_x 12.0 are all rejected
- task 7 closed: both criteria moved to Met with how each was verified

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
(cherry picked from commit a310496664982c372f51b43113465eb8ad9e9fb5)
This commit is contained in:
shoney.arickathil 2026-08-30 15:50:34 +02:00
parent 6fc5b4b7d4
commit 841f6c8f3f
8 changed files with 416 additions and 271 deletions

View file

@ -863,5 +863,29 @@
"floor": 100,
"tolerance_pct": 100,
"value": 3
},
"residency.all_collapse_x": {
"dir": "higher",
"floor": 2.0,
"tolerance_pct": 100,
"value": 105.4
},
"residency.in_ram_cost_x": {
"dir": "lower",
"floor": 8.0,
"tolerance_pct": 50,
"value": 4.23
},
"residency.overcap_vs_swap_x": {
"dir": "higher",
"floor": 1.0,
"tolerance_pct": 100,
"value": 1.53
},
"residency.rss_ratio": {
"dir": "higher",
"floor": 2.0,
"tolerance_pct": 10,
"value": 2.55
}
}

View file

@ -651,190 +651,6 @@ fn replayseed_mode(n: Int, m: Int) -> Int {
return 0;
}
-- databasev2 2 task 7: the residency A/B, one mode per table so the two runs
-- differ ONLY in the annotation. Same fill, same Weyl key order, same read
-- count as randread_mode above — the comparison is against that leg's own
-- 273x swap figure, measured on the same box under the same cap.
-- The WIDE half of the residency A/B. Same structure as kread_all/kread_keys
-- but with Text columns, which is the only shape where dropping a payload
-- frees anything: an Int's value is its inline slot word, a Text's is a
-- separate allocation.
-- databasev2 2 task 7, the GB-scale leg. Same A/B as wread_*, but each row
-- carries ~2 KB of Text so a realistic data volume is reachable in a few
-- hundred thousand inserts rather than millions — the insert path is
-- fsync-bound at roughly 2 000 rows/s, so row COUNT is the expensive axis and
-- row SIZE is nearly free.
--
-- This is the shape the mode actually claims: data far larger than the cap,
-- with only the id map, the indexes and the (never-released) slabs resident.
fn heavy_pad() -> Text {
let p = "0123456789abcdef0123456789abcdef";
let out = "";
let i = 0;
while i < 20 { out = out .. p; i = i + 1; }
return out;
}
fn hread_all(n: Int, r: Int) -> Int {
let pad = heavy_pad();
let i = 1;
while i <= n {
insert WideA { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}" };
i = i + 1;
}
print("hreadfilled ${n} ${self_rss_kb()}");
let h: map<Int, Int> = {};
let hits = 0;
let t0 = time.ticks();
let j = 0;
while j < r {
let key = 1 + (j * 2654435761) % n;
let o0 = time.ticks();
for row in from x in WideA where x.k == key take 1 select x {
if len(row.a) > 0 { hits = hits + 1; }
}
hist_add(h, time.ticks() - o0);
j = j + 1;
}
let el = time.ticks() - t0;
report("hreadall", r, el, h);
print("hreadallrss ${self_rss_kb()} ${hits}");
return 0;
}
fn hread_keys(n: Int, r: Int) -> Int {
let pad = heavy_pad();
let i = 1;
while i <= n {
insert WideK { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}" };
i = i + 1;
}
print("hreadfilled ${n} ${self_rss_kb()}");
let h: map<Int, Int> = {};
let hits = 0;
let t0 = time.ticks();
let j = 0;
while j < r {
let key = 1 + (j * 2654435761) % n;
let o0 = time.ticks();
for row in from x in WideK where x.k == key take 1 select x {
if len(row.a) > 0 { hits = hits + 1; }
}
hist_add(h, time.ticks() - o0);
j = j + 1;
}
let el = time.ticks() - t0;
report("hreadkeys", r, el, h);
print("hreadkeysrss ${self_rss_kb()} ${hits}");
return 0;
}
fn wread_all(n: Int, r: Int) -> Int {
let pad = "0123456789abcdef0123456789abcdef";
let i = 1;
while i <= n {
insert WideA { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" };
i = i + 1;
}
print("wreadfilled ${n} ${self_rss_kb()}");
let h: map<Int, Int> = {};
let hits = 0;
let t0 = time.ticks();
let j = 0;
while j < r {
let key = 1 + (j * 2654435761) % n;
let o0 = time.ticks();
for row in from x in WideA where x.k == key take 1 select x {
if len(row.a) > 0 { hits = hits + 1; }
}
hist_add(h, time.ticks() - o0);
j = j + 1;
}
let el = time.ticks() - t0;
report("wreadall", r, el, h);
print("wreadallrss ${self_rss_kb()} ${hits}");
return 0;
}
fn wread_keys(n: Int, r: Int) -> Int {
let pad = "0123456789abcdef0123456789abcdef";
let i = 1;
while i <= n {
insert WideK { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" };
i = i + 1;
}
print("wreadfilled ${n} ${self_rss_kb()}");
let h: map<Int, Int> = {};
let hits = 0;
let t0 = time.ticks();
let j = 0;
while j < r {
let key = 1 + (j * 2654435761) % n;
let o0 = time.ticks();
for row in from x in WideK where x.k == key take 1 select x {
if len(row.a) > 0 { hits = hits + 1; }
}
hist_add(h, time.ticks() - o0);
j = j + 1;
}
let el = time.ticks() - t0;
report("wreadkeys", r, el, h);
print("wreadkeysrss ${self_rss_kb()} ${hits}");
return 0;
}
fn kread_all(n: Int, r: Int) -> Int {
let i = 1;
while i <= n {
insert RowA { k: i, v: item_v(i) };
i = i + 1;
}
print("kreadfilled ${n} ${self_rss_kb()}");
let h: map<Int, Int> = {};
let hits = 0;
let t0 = time.ticks();
let j = 0;
while j < r {
let key = 1 + (j * 2654435761) % n;
let o0 = time.ticks();
for row in from x in RowA where x.k == key take 1 select x {
if row.v == item_v(key) { hits = hits + 1; }
}
hist_add(h, time.ticks() - o0);
j = j + 1;
}
let el = time.ticks() - t0;
report("kreadall", r, el, h);
print("kreadallrss ${self_rss_kb()} ${hits}");
return 0;
}
fn kread_keys(n: Int, r: Int) -> Int {
let i = 1;
while i <= n {
insert RowK { k: i, v: item_v(i) };
i = i + 1;
}
print("kreadfilled ${n} ${self_rss_kb()}");
let h: map<Int, Int> = {};
let hits = 0;
let t0 = time.ticks();
let j = 0;
while j < r {
let key = 1 + (j * 2654435761) % n;
let o0 = time.ticks();
for row in from x in RowK where x.k == key take 1 select x {
if row.v == item_v(key) { hits = hits + 1; }
}
hist_add(h, time.ticks() - o0);
j = j + 1;
}
let el = time.ticks() - t0;
report("kreadkeys", r, el, h);
print("kreadkeysrss ${self_rss_kb()} ${hits}");
return 0;
}
fn randread_mode(n: Int, r: Int) -> Int {
let bref = insert Bucket { tag: "randread" };
let i = 1;
@ -1008,33 +824,6 @@ fn main(args: multi Text) -> Int {
}
return replayseed_mode(n, mm);
}
if args[0] == "hreadall" or args[0] == "hreadkeys" {
if len(args) < 3 { print_err("usage: hreadall|hreadkeys N R"); return 2; }
let hn = parse_int(args[1]);
let hr = parse_int(args[2]);
if hn == nil or hr == nil { print_err("db-bench: N and R must be positive"); return 2; }
if args[0] == "hreadall" { return hread_all(hn, hr); }
return hread_keys(hn, hr);
}
if args[0] == "wreadall" or args[0] == "wreadkeys" {
if len(args) < 3 { print_err("usage: wreadall|wreadkeys N R"); return 2; }
let wn = parse_int(args[1]);
let wr = parse_int(args[2]);
if wn == nil or wr == nil { print_err("db-bench: N and R must be positive"); return 2; }
if args[0] == "wreadall" { return wread_all(wn, wr); }
return wread_keys(wn, wr);
}
if args[0] == "kreadall" or args[0] == "kreadkeys" {
if len(args) < 3 { print_err("usage: kreadall|kreadkeys N R"); return 2; }
let n = parse_int(args[1]);
let rr = parse_int(args[2]);
if n == nil or rr == nil {
print_err("db-bench: N and R must be positive numbers");
return 2;
}
if args[0] == "kreadall" { return kread_all(n, rr); }
return kread_keys(n, rr);
}
if args[0] == "randread" {
if len(args) < 3 {
return usage();

View file

@ -57,47 +57,3 @@ class MixJob {
class Flood {
n: Int
}
-- databasev2 2 task 7: the residency A/B. These two are IDENTICAL except for
-- the `resident` annotation, so a difference between them is the storage
-- mode's doing and nothing else's. Item above carries a `ref` and a second
-- index, which would confound the comparison.
--
-- The question they exist to answer: iteration 1 measured a 273x collapse for
-- random reads over a table larger than RAM, on demand-paged anonymous memory.
-- `resident: keys` reads rows back with pread through the page cache instead.
-- If that is not materially better than 273x, the mode buys nothing the kernel
-- was not already doing.
@table(name: "rowsall", index: [k], durable: true, resident: all)
class RowA {
k: Int
v: Int
}
@table(name: "rowskeys", index: [k], durable: true, resident: keys)
class RowK {
k: Int
v: Int
}
-- databasev2 2 task 7, second A/B: the WIDE shape. The Int-only pair above
-- cannot show what keys-residency is for — dropping a payload frees each
-- field's VALUE, and an Int's value IS its inline slot word, so nothing is
-- freed and the slab stays allocated either way. A row with Text columns
-- drags separate db_text allocations that dropping genuinely releases, which
-- is the only shape where the mode can pay for itself.
@table(name: "wideall", index: [k], durable: true, resident: all)
class WideA {
k: Int
a: Text
b: Text
note: Text
}
@table(name: "widekeys", index: [k], durable: true, resident: keys)
class WideK {
k: Int
a: Text
b: Text
note: Text
}

View file

@ -0,0 +1,152 @@
-- residency-bench — databasev2 2 task 7.
--
-- Two tables identical except the `resident` annotation (see types.wo), the
-- same fill, the same Weyl key order, the same read count. Run under a cgroup
-- memory cap by scripts/db-bench.py's `residency` leg:
--
-- control : a cap that binds neither mode -> what the mode COSTS
-- over-cap: a cap between the two resident sets -> what the mode BUYS
--
-- Prints the same shape db-bench's own legs do, so the harness parses it with
-- the same helpers: `<op> <n> <ops/sec> <p50> <p99>` plus an rss/hits line.
use fs
use time
fn hist_add(mut h: map<Int, Int>, us: Int) {
let b = us;
if b < 0 {
b = 0;
}
if b > 20000 {
b = 20000;
}
if has(h, b) {
set(h, b, get(h, b) + 1);
} else {
set(h, b, 1);
}
}
fn hist_pct(h: map<Int, Int>, total: Int, pct: Int) -> Int {
let target = total * pct / 100;
if target < 1 {
target = 1;
}
let seen = 0;
let b = 0;
while b <= 20000 {
if has(h, b) {
seen = seen + get(h, b);
if seen >= target {
return b;
}
}
b = b + 1;
}
return 20000;
}
fn report(op: Text, n: Int, total_us: Int, h: map<Int, Int>) {
let us = total_us;
if us < 1 {
us = 1;
}
let rate = n * 1000000 / us;
print("${op} ${n} ${rate} ${hist_pct(h, n, 50)} ${hist_pct(h, n, 99)}");
}
fn self_rss_kb() -> Int {
let st = try fs.read_all("/proc/self/status", 16384) catch (e) "";
let i = index_of(st, "VmRSS:");
if i < 0 {
return 0;
}
let rest = substr(st, i + 6, 24);
let n = 0;
let j = 0;
while j < len(rest) {
let c = byte_at(rest, j);
if c >= 48 and c <= 57 {
n = n * 10 + (c - 48);
} else {
if n > 0 {
return n;
}
}
j = j + 1;
}
return n;
}
fn wide_pad() -> Text {
return "0123456789abcdef0123456789abcdef";
}
fn wread_all(n: Int, r: Int) -> Int {
let pad = wide_pad();
let i = 1;
while i <= n {
insert WideA { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" };
i = i + 1;
}
print("wreadfilled ${n} ${self_rss_kb()}");
let h: map<Int, Int> = {};
let hits = 0;
let t0 = time.ticks();
let j = 0;
while j < r {
let key = 1 + (j * 2654435761) % n;
let o0 = time.ticks();
for row in from x in WideA where x.k == key take 1 select x {
if len(row.a) > 0 { hits = hits + 1; }
}
hist_add(h, time.ticks() - o0);
j = j + 1;
}
report("wreadall", r, time.ticks() - t0, h);
print("wreadallrss ${self_rss_kb()} ${hits}");
return 0;
}
fn wread_keys(n: Int, r: Int) -> Int {
let pad = wide_pad();
let i = 1;
while i <= n {
insert WideK { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" };
i = i + 1;
}
print("wreadfilled ${n} ${self_rss_kb()}");
let h: map<Int, Int> = {};
let hits = 0;
let t0 = time.ticks();
let j = 0;
while j < r {
let key = 1 + (j * 2654435761) % n;
let o0 = time.ticks();
for row in from x in WideK where x.k == key take 1 select x {
if len(row.a) > 0 { hits = hits + 1; }
}
hist_add(h, time.ticks() - o0);
j = j + 1;
}
report("wreadkeys", r, time.ticks() - t0, h);
print("wreadkeysrss ${self_rss_kb()} ${hits}");
return 0;
}
fn main(args: multi Text) -> Int {
if len(args) < 3 {
print_err("usage: residency-bench wreadall|wreadkeys N R");
return 2;
}
let n = parse_int(args[1]);
let r = parse_int(args[2]);
if n == nil or r == nil {
print_err("residency-bench: N and R must be positive numbers");
return 2;
}
if args[0] == "wreadall" { return wread_all(n, r); }
if args[0] == "wreadkeys" { return wread_keys(n, r); }
print_err("residency-bench: unknown mode ${args[0]}");
return 2;
}

View file

@ -0,0 +1,34 @@
-- databasev2 2 task 7: the residency A/B.
--
-- These two tables are IDENTICAL except for the `resident` annotation, so any
-- difference between them is the storage mode's doing and nothing else's.
--
-- WHY THIS IS ITS OWN PROGRAM rather than a mode inside db-bench: declaring a
-- `resident: keys` table is a WHOLE-PROGRAM constraint. Its rows live only in
-- the write-ahead log, so the runtime refuses to start without WO_DATA — and
-- that refusal fires for every mode in the module, including ones that never
-- touch the table. Putting these classes in db-bench's shared module made its
-- `growth`, `ceiling` and `randread` legs, which deliberately run WITHOUT
-- WO_DATA, refuse to start.
--
-- The shape is WIDE on purpose. An Int-only pair shows the two modes as
-- indistinguishable, and that is structural: dropping a payload frees each
-- field's VALUE, and an Int's value IS its inline slot word, so nothing is
-- freed and the slab stays allocated either way. Only rows with Text columns —
-- separate allocations that dropping genuinely releases — can show what the
-- mode is for.
@table(name: "wideall", index: [k], durable: true, resident: all)
class WideA {
k: Int
a: Text
b: Text
note: Text
}
@table(name: "widekeys", index: [k], durable: true, resident: keys)
class WideK {
k: Int
a: Text
b: Text
note: Text
}

View file

@ -0,0 +1,6 @@
name = "residency-bench"
version = "0.1.0"
description = "databasev2 2 task 7: resident: all vs resident: keys, identical tables, one annotation apart"
[runtime]
wo = ">= 0.1"

View file

@ -66,7 +66,7 @@ declared per-table policy. Durability is untouched and unconditional.
| 5c | shared borrow/release accessor, then id→offset storage | ✅ `2e347de` (accessor, pure refactor, `db-bench --quick` 85/0), `18ce4d5` (offset storage), `f9c36ef` (insert + boot wiring) |
| 5d | rewire the readers: remaining `wo_row_ptr` sites, slab scans, FK restrict, `@unique` across the boundary | ✅ `11a92df` (db.c), + this commit (table.c, wal.c, compaction). Updates **refused**, not rewired — see below |
| 6 | the two runtime refusals (no-`WO_DATA`, the byte budget) | ⬜ |
| 7 | measure, gate, document, close out | 🔄 measured 2026-08-30 (below); gate + closeout outstanding |
| 7 | measure, gate, document, close out | ✅ measured, gated and documented 2026-08-30 |
**The `durable` half is complete and usable.** A volatile table is a full table
in-process — same indexes, same `@unique`, same FK restrict, same query surface
@ -274,9 +274,49 @@ over-capacity table fast — under a hard memory cap it is within 1.5× of simpl
letting the kernel swap. The honest guidance is "use it to fit more, not to go
faster", and the docs should say so.
**Still outstanding for task 7:** wire these legs into `scripts/db-bench.py`
with tolerances and a baseline entry, and re-measure the `resident: all` read
baseline to confirm no cost for a feature not used.
**Gated 2026-08-30.** `scripts/db-bench.py` grew a `residency` leg driving
`docs/examples/residency-bench` — its own program, because declaring a
`resident: keys` table is a WHOLE-PROGRAM constraint: the runtime refuses to
start without `WO_DATA`, for every mode in the module. Putting those classes in
db-bench's shared types made `growth`, `ceiling` and `randread` — which
deliberately run without `WO_DATA` — refuse to start. That regression was caught
by running the leg, not by reading it.
What is gated, and what deliberately is not, follows `randread`'s existing
split: the absolute ops/sec under a cap is swap and disk I/O and belongs to the
box, so it is recorded and waived; the RATIOS are the engine's property.
| metric | baseline | floor | tolerance |
| --- | --- | --- | --- |
| `residency.rss_ratio` | 2.55 | 2.0 | 10% |
| `residency.overcap_vs_swap_x` | 1.53 | 1.0 | 100% |
| `residency.in_ram_cost_x` | 4.23 | 8.0 (ceiling) | 50% |
| `residency.all_collapse_x` | 105.4 | 2.0 | 100% |
`rss_ratio` carries the tight tolerance because footprint is structural — the
same class of number as `bytes_per_row`. The two throughput ratios are guarded
by their FLOORS rather than their bands, which is this harness's established
answer to a metric whose absolute value belongs to the disk. `all_collapse_x`
exists only to assert the cap actually binds; a leg whose "over-cap" half is not
over cap silently measures nothing, which is exactly what the first run of this
leg did.
Verified by feeding the gate a breaching run: `rss_ratio` 1.4,
`overcap_vs_swap_x` 0.6 and `in_ram_cost_x` 12.0 are all rejected.
- **Given** the `resident: all` read baseline, **when** re-measured, **then**
inside tolerance — no cost for a feature not used. ✅ Verified as a
by-product of the residency leg: `resident: all` is unchanged at 1 354 554
reads/sec uncapped, and every other db-bench leg still runs, which the
keys-resident classes had briefly broken by forcing `WO_DATA` module-wide.
- **Given** a `resident: keys` table larger than RAM, **when** read randomly,
**then** its read cost is measured against the resident baseline on its own
read path. ✅ Measured 2026-08-30 and the answer is qualified: **1.53×**
faster than letting the kernel swap under a cap that binds one and not the
other — real, but nowhere near iteration 1's 273× swap figure would suggest,
because cgroup limits charge the page cache, so moving rows to a file does
not escape a container memory limit. The unambiguous win is footprint:
**2.55×** smaller resident set. Full method, numbers and the failed first
attempt above; gated by `residency.*`.
Outstanding:
@ -285,17 +325,6 @@ Outstanding:
every write)*
- **Given** the resident footprint crossing the budget, **when** it does,
**then** a refusal naming the table and the annotation. *(task 6)*
- **Given** the `resident: all` read baseline, **when** re-measured, **then**
inside tolerance — no cost for a feature not used. *(task 7)*
- **Given** a `resident: keys` table larger than RAM, **when** read randomly,
**then** its read cost is **measured against the resident baseline on its own
read path**, not inherited from databasev2 1's swap figure. *(task 7)* — that
figure is **273×** for demand-paged anonymous memory
([1](01-ram-ceiling-measurement.md)); `pread` through the page cache should do
better, and the whole value of `resident: keys` rests on how much better. If it
is not materially better than swapping, the design buys nothing that the
kernel was not already doing.
## Out Of Scope
- **Checkpoint and compaction** — [3](03-wal-checkpoint.md). Boot rebuilds the

View file

@ -13,10 +13,16 @@ Exit 0 = campaign green; 1 = gate breach or a durability leg failed.
Plan deviation, disclosed: one python driver instead of bash+python —
the live stdout sampling and JSON assembly are the whole job.
"""
import json, os, re, subprocess, sys, time, random, shutil
import json, os, re, subprocess, sys, tempfile, time, random, shutil
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
BIN = os.path.join(ROOT, "docs/examples/db-bench/target/db-bench")
# databasev2 2 task 7. Its own program, not a mode in db-bench: declaring a
# `resident: keys` table is a WHOLE-PROGRAM constraint — the runtime refuses to
# start without WO_DATA, for every mode in the module. Putting those classes in
# db-bench's shared types made growth/ceiling/randread, which deliberately run
# WITHOUT WO_DATA, refuse to start.
RESID_BIN = os.path.join(ROOT, "docs/examples/residency-bench/target/residency-bench")
WOC = os.path.join(ROOT, "compiler/_build/default/bin/woc")
WOVM = os.path.join(ROOT, "runtime/wovm")
BASELINE = os.path.join(ROOT, "bench/baseline.json")
@ -60,6 +66,10 @@ def build():
"-o", BIN, "--runtime", WOVM], capture_output=True, text=True)
if r.returncode != 0:
bad("build", r.stderr.strip()[:200]); sys.exit(1)
r = subprocess.run([WOC, "build", os.path.join(ROOT, "docs/examples/residency-bench"),
"-o", RESID_BIN, "--runtime", WOVM], capture_output=True, text=True)
if r.returncode != 0:
bad("build residency-bench", r.stderr.strip()[:200]); sys.exit(1)
ok("builds")
def run(args, env_extra, timeout, sample_after=None):
@ -303,6 +313,26 @@ def tolerance_for(key):
if key.startswith("growth."): return 100
if key.startswith("ceiling."): return 100
if key.startswith("randread."): return 100
# databasev2 2 task 7. Same split randread makes, for the same reason: the
# absolute ops/sec under a cap is swap and disk I/O and belongs to the box,
# so it is recorded and waived. The RATIOS are the engine's property.
#
# rss_ratio is the headline claim — keys must hold the same rows in a
# materially smaller resident set or the mode has no purpose — and it is
# STRUCTURAL: 87.5 MiB against 34.4 MiB, reproducible, the same class of
# number as bytes_per_row. It gets the same tight tolerance, and residency()
# additionally hard-fails below 2.0x regardless of drift.
if key == "residency.rss_ratio": return 10
# in_ram_cost is a throughput ratio between two cached runs: stable in
# shape (a pread and a fold against a pointer dereference) but it moves
# with page-cache weather, so it is gated loosely rather than waived.
if key == "residency.in_ram_cost_x": return 50
# overcap_vs_swap compares two I/O-bound runs, so BOTH halves are the box's.
# The ratio is still worth recording — it is the answer to the question the
# iteration was written to ask — but residency() guards the direction of it
# (keys must not be slower than swapping) rather than its magnitude.
if key == "residency.overcap_vs_swap_x": return 100
if key.startswith("residency."): return 100
if key.startswith("replay."): return 100
# databasev2 4: batch SHAPE follows arrival timing, so gating it tightly
# would gate the scheduler — what must hold is that the mean exceeds one
@ -558,6 +588,130 @@ def ceiling(metrics):
# databasev2 2 task 7: does `resident: keys` beat letting the kernel swap?
RESID_N = 40000 if QUICK else 200000
RESID_R = 10000 if QUICK else 40000
RESID_FIT_MB = 256 # control: neither mode is under pressure
# Between the two resident sets, so `all` pages and `keys` does not. This MUST
# scale with N: at QUICK's 40k rows `all` needs only ~17 MiB, so a 48 MiB cap
# binds neither mode and the comparison silently becomes "keys is slower when
# nothing is under pressure" — which is true, and not what this leg asks.
RESID_CAP_MB = 10 if QUICK else 48
def parse_resid(lines, op):
"""`<op> <n> <ops/sec> <p50> <p99>`, plus the mode's own rss/hits line."""
ops = p50 = p99 = rss = hits = filled = None
for l in lines:
f = l.split()
if not f:
continue
if f[0] == op and len(f) == 5:
ops, p50, p99 = int(f[2]), int(f[3]), int(f[4])
elif f[0] == op + "rss" and len(f) == 3:
rss, hits = int(f[1]), int(f[2])
elif f[0] == "wreadfilled" and len(f) == 3:
filled = int(f[2])
return ops, p50, p99, rss, hits, filled
def residency(metrics):
"""`resident: keys` against `resident: all`, on tables IDENTICAL except the
annotation, so a difference is the storage mode's doing and nothing else's.
Measured 2026-08-30 on the WIDE shape deliberately. An Int-only pair shows
the two modes as indistinguishable, and that is structural rather than
surprising: dropping a payload frees each field's VALUE, and an Int's value
IS its inline slot word, so nothing is freed and the slab stays allocated
either way. A benchmark built on that shape would condemn the feature for a
reason that has nothing to do with the feature.
WHAT IS GATED, and what deliberately is not. The RATIOS are the engine's
property and get real tolerances; the absolute ops/sec under a cap is swap
and disk I/O, so it belongs to the box and is recorded but waived. This is
the same split randread() already makes for the same reason.
The headline claim is rss_ratio: keys must hold the same rows in a
materially smaller resident set, or the mode has no purpose. The measured
figure was 2.55x (34.4 MiB against 87.5 MiB)."""
if cap_wrapper(RESID_FIT_MB, 0) is None:
ok("residency: SKIPPED -- no rootless cgroup v2 memory cap on this host")
return
res = {}
for mode, op in (("all", "wreadall"), ("keys", "wreadkeys")):
for legname, cap_mb, swap_mb in (("fit", RESID_FIT_MB, 0),
("cap", RESID_CAP_MB, 512)):
w = cap_wrapper(cap_mb, swap_mb)
data = tempfile.mkdtemp(prefix="resid-")
env = dict(os.environ)
env["WO_DATA"] = data # a keys-resident table cannot run without one
env["WO_SHARDS"] = "1"
pr = subprocess.run(w + [RESID_BIN, op, str(RESID_N), str(RESID_R)],
stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
text=True, env=env, timeout=1800)
shutil.rmtree(data, ignore_errors=True)
lines = pr.stdout.splitlines()
ops, p50, p99, rss, hits, filled = parse_resid(lines, op)
key = f"residency.{mode}.{legname}"
if pr.returncode != 0 or ops is None:
bad(f"{key}: run failed",
f"rc={pr.returncode} {(lines[-1:] or ['no output'])[0][:120]}")
return
if hits != RESID_R:
# a comparison over reads that did not resolve measures nothing
bad(f"{key}: only {hits}/{RESID_R} reads resolved", "keys must all exist")
return
metrics[f"{key}.ops_sec"] = ops
metrics[f"{key}.read_p50us"] = p50
metrics[f"{key}.read_p99us"] = p99
metrics[f"{key}.filled_rss_kb"] = filled
res[f"{mode}.{legname}"] = (ops, filled)
ok(f"{key}: {ops} reads/sec, p50 {p50}us p99 {p99}us, {filled} KiB after fill")
all_rss = res["all.fit"][1]
keys_rss = res["keys.fit"][1]
rss_ratio = round(all_rss / max(keys_rss, 1), 2)
metrics["residency.rss_ratio"] = rss_ratio
# what the mode costs when memory is NOT tight: a pread and a fold per row
# against a pointer dereference
in_ram_cost = round(res["all.fit"][0] / max(res["keys.fit"][0], 1), 2)
metrics["residency.in_ram_cost_x"] = in_ram_cost
# the question the iteration was written to answer: under a cap that binds
# `all` and not `keys`, is keys actually better than swapping?
vs_swap = round(res["keys.cap"][0] / max(res["all.cap"][0], 1), 2)
metrics["residency.overcap_vs_swap_x"] = vs_swap
if rss_ratio < 2.0:
bad("residency: keys saves less than 2x RSS",
f"{all_rss} KiB vs {keys_rss} KiB = {rss_ratio}x -- the mode's whole purpose")
else:
ok(f"residency: keys holds the same rows in {rss_ratio}x less RSS "
f"({all_rss} -> {keys_rss} KiB)")
# The cap must actually BIND the resident half, or the comparison is
# meaningless. randread learned this the same way; assert it rather than
# trusting the constants to stay right as N changes.
all_collapse = res["all.fit"][0] / max(res["all.cap"][0], 1)
metrics["residency.all_collapse_x"] = round(all_collapse, 2)
if all_collapse < 2.0:
bad("residency: the cap did not bind `resident: all`",
f"{res['all.fit'][0]} -> {res['all.cap'][0]} reads/sec is only "
f"{all_collapse:.2f}x; {RESID_N} rows fit under {RESID_CAP_MB} MiB, resize the leg")
return
if vs_swap < 1.0:
bad("residency: keys is SLOWER than letting the kernel swap",
f"{res['keys.cap'][0]} vs {res['all.cap'][0]} reads/sec under a "
f"{RESID_CAP_MB} MiB cap -- the mode buys nothing here")
else:
ok(f"residency: under a {RESID_CAP_MB} MiB cap keys is {vs_swap}x swapping "
f"({res['keys.cap'][0]} vs {res['all.cap'][0]} reads/sec)")
ok(f"residency: costs {in_ram_cost}x read throughput when memory is not tight")
def parse_randread(lines):
"""ops/sec, p50, p99, resolved-read count and post-fill RSS from the
sample's own randread lines. `randreadfilled` also starts with "randread",
@ -847,6 +1001,7 @@ def main():
growth(metrics)
ceiling(metrics)
randread(metrics)
residency(metrics)
replay(metrics)
checkpoint_leg(metrics)
os.makedirs(RESULTS_DIR, exist_ok=True)