Merge branch 'db-bench' (iteration 22 — durability/throughput baseline)

- brings time.ticks builtin (id 84), bench sample + campaign driver
  (scripts/db-bench.py), just db-bench/db-bench-quick recipes, first
  baseline recorded, story 22 to done/, postgres study cards
- conflicts resolved: board in-progress table (iteration 36 +
  framework rows kept, db-bench row now "22 landed"; dangling order
  anchor repointed); story 36 moved back to in-progress/ (dir-rename
  inference dragged it to done/ — 36 still awaits the manual pass)

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
shoney.arickathil 2026-08-22 22:57:51 +02:00
commit 284c4e5392
33 changed files with 1832 additions and 128 deletions

3
.gitignore vendored
View file

@ -107,3 +107,6 @@ docs/examples/*/target/
# Obsidian vault state (developer-local)
docs/.obsidian/
docs/Untitled.base
# bench scratch stores (driver-managed)
bench/tmp.*

453
bench/baseline.json Normal file
View file

@ -0,0 +1,453 @@
{
"_config": {
"N": 20000,
"crash_reps": 3,
"msg_n": 200000,
"note": "refresh only with a commit that says why; tolerances widened 2026-08-21 from the two-run repeatability check: mix* 50% (scheduling-dependent small counts), read/query 35% (machine jitter), everything else 15%; msgrate floors value/8 \u2014 quick-mode's small N is spawn-dominated and grazed value/4",
"wal_n": 4000
},
"durable.s1.mixread.ops_sec": {
"dir": "higher",
"floor": 187,
"tolerance_pct": 50,
"value": 751
},
"durable.s1.mixread.p50us": {
"dir": "lower",
"floor": 11844,
"tolerance_pct": 50,
"value": 2961
},
"durable.s1.mixread.p99us": {
"dir": "lower",
"floor": 19044,
"tolerance_pct": 50,
"value": 4761
},
"durable.s1.mixwrite.ops_sec": {
"dir": "higher",
"floor": 20,
"tolerance_pct": 50,
"value": 83
},
"durable.s1.mixwrite.p50us": {
"dir": "lower",
"floor": 19408,
"tolerance_pct": 50,
"value": 4852
},
"durable.s1.mixwrite.p99us": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 50,
"value": 20000
},
"durable.s1.query.ops_sec": {
"dir": "higher",
"floor": 335,
"tolerance_pct": 35,
"value": 1341
},
"durable.s1.query.p50us": {
"dir": "lower",
"floor": 2996,
"tolerance_pct": 35,
"value": 749
},
"durable.s1.query.p99us": {
"dir": "lower",
"floor": 3204,
"tolerance_pct": 35,
"value": 801
},
"durable.s1.read.ops_sec": {
"dir": "higher",
"floor": 336,
"tolerance_pct": 35,
"value": 1346
},
"durable.s1.read.p50us": {
"dir": "lower",
"floor": 3032,
"tolerance_pct": 35,
"value": 758
},
"durable.s1.read.p99us": {
"dir": "lower",
"floor": 3324,
"tolerance_pct": 35,
"value": 831
},
"durable.s1.seed.ops_sec": {
"dir": "higher",
"floor": 1123,
"tolerance_pct": 15,
"value": 4492
},
"durable.s1.seed.p50us": {
"dir": "lower",
"floor": 840,
"tolerance_pct": 15,
"value": 210
},
"durable.s1.seed.p99us": {
"dir": "lower",
"floor": 2188,
"tolerance_pct": 15,
"value": 547
},
"durable.s1.write.ops_sec": {
"dir": "higher",
"floor": 307,
"tolerance_pct": 15,
"value": 1230
},
"durable.s1.write.p50us": {
"dir": "lower",
"floor": 3036,
"tolerance_pct": 15,
"value": 759
},
"durable.s1.write.p99us": {
"dir": "lower",
"floor": 7232,
"tolerance_pct": 15,
"value": 1808
},
"durable.sN.mixread.ops_sec": {
"dir": "higher",
"floor": 5,
"tolerance_pct": 50,
"value": 20
},
"durable.sN.mixread.p50us": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 50,
"value": 20000
},
"durable.sN.mixread.p99us": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 50,
"value": 20000
},
"durable.sN.mixwrite.ops_sec": {
"dir": "higher",
"floor": 0,
"tolerance_pct": 50,
"value": 2
},
"durable.sN.mixwrite.p50us": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 50,
"value": 20000
},
"durable.sN.mixwrite.p99us": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 50,
"value": 20000
},
"durable.sN.query.ops_sec": {
"dir": "higher",
"floor": 328,
"tolerance_pct": 35,
"value": 1312
},
"durable.sN.query.p50us": {
"dir": "lower",
"floor": 3028,
"tolerance_pct": 35,
"value": 757
},
"durable.sN.query.p99us": {
"dir": "lower",
"floor": 3392,
"tolerance_pct": 35,
"value": 848
},
"durable.sN.read.ops_sec": {
"dir": "higher",
"floor": 350,
"tolerance_pct": 35,
"value": 1403
},
"durable.sN.read.p50us": {
"dir": "lower",
"floor": 3000,
"tolerance_pct": 35,
"value": 750
},
"durable.sN.read.p99us": {
"dir": "lower",
"floor": 3420,
"tolerance_pct": 35,
"value": 855
},
"durable.sN.seed.ops_sec": {
"dir": "higher",
"floor": 1119,
"tolerance_pct": 15,
"value": 4478
},
"durable.sN.seed.p50us": {
"dir": "lower",
"floor": 836,
"tolerance_pct": 15,
"value": 209
},
"durable.sN.seed.p99us": {
"dir": "lower",
"floor": 2372,
"tolerance_pct": 15,
"value": 593
},
"durable.sN.write.ops_sec": {
"dir": "higher",
"floor": 316,
"tolerance_pct": 15,
"value": 1265
},
"durable.sN.write.p50us": {
"dir": "lower",
"floor": 3064,
"tolerance_pct": 15,
"value": 766
},
"durable.sN.write.p99us": {
"dir": "lower",
"floor": 6616,
"tolerance_pct": 15,
"value": 1654
},
"ram.s1.mixread.ops_sec": {
"dir": "higher",
"floor": 320,
"tolerance_pct": 50,
"value": 1280
},
"ram.s1.mixread.p50us": {
"dir": "lower",
"floor": 10904,
"tolerance_pct": 50,
"value": 2726
},
"ram.s1.mixread.p99us": {
"dir": "lower",
"floor": 12472,
"tolerance_pct": 50,
"value": 3118
},
"ram.s1.mixwrite.ops_sec": {
"dir": "higher",
"floor": 35,
"tolerance_pct": 50,
"value": 142
},
"ram.s1.mixwrite.p50us": {
"dir": "lower",
"floor": 11216,
"tolerance_pct": 50,
"value": 2804
},
"ram.s1.mixwrite.p99us": {
"dir": "lower",
"floor": 16368,
"tolerance_pct": 50,
"value": 4092
},
"ram.s1.msgrate.msgs_sec": {
"dir": "higher",
"floor": 1678077,
"tolerance_pct": 15,
"value": 13424620
},
"ram.s1.query.ops_sec": {
"dir": "higher",
"floor": 378,
"tolerance_pct": 35,
"value": 1512
},
"ram.s1.query.p50us": {
"dir": "lower",
"floor": 2536,
"tolerance_pct": 35,
"value": 634
},
"ram.s1.query.p99us": {
"dir": "lower",
"floor": 3196,
"tolerance_pct": 35,
"value": 799
},
"ram.s1.read.ops_sec": {
"dir": "higher",
"floor": 407,
"tolerance_pct": 35,
"value": 1630
},
"ram.s1.read.p50us": {
"dir": "lower",
"floor": 2396,
"tolerance_pct": 35,
"value": 599
},
"ram.s1.read.p99us": {
"dir": "lower",
"floor": 3124,
"tolerance_pct": 35,
"value": 781
},
"ram.s1.seed.ops_sec": {
"dir": "higher",
"floor": 64354,
"tolerance_pct": 15,
"value": 257416
},
"ram.s1.seed.p50us": {
"dir": "lower",
"floor": 16,
"tolerance_pct": 15,
"value": 4
},
"ram.s1.seed.p99us": {
"dir": "lower",
"floor": 32,
"tolerance_pct": 15,
"value": 8
},
"ram.s1.write.ops_sec": {
"dir": "higher",
"floor": 718,
"tolerance_pct": 15,
"value": 2872
},
"ram.s1.write.p50us": {
"dir": "lower",
"floor": 2320,
"tolerance_pct": 15,
"value": 580
},
"ram.s1.write.p99us": {
"dir": "lower",
"floor": 3444,
"tolerance_pct": 15,
"value": 861
},
"ram.sN.mixread.ops_sec": {
"dir": "higher",
"floor": 5,
"tolerance_pct": 50,
"value": 21
},
"ram.sN.mixread.p50us": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 50,
"value": 20000
},
"ram.sN.mixread.p99us": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 50,
"value": 20000
},
"ram.sN.mixwrite.ops_sec": {
"dir": "higher",
"floor": 0,
"tolerance_pct": 50,
"value": 2
},
"ram.sN.mixwrite.p50us": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 50,
"value": 20000
},
"ram.sN.mixwrite.p99us": {
"dir": "lower",
"floor": 80000,
"tolerance_pct": 50,
"value": 20000
},
"ram.sN.msgrate.msgs_sec": {
"dir": "higher",
"floor": 305743,
"tolerance_pct": 15,
"value": 2445944
},
"ram.sN.query.ops_sec": {
"dir": "higher",
"floor": 333,
"tolerance_pct": 35,
"value": 1333
},
"ram.sN.query.p50us": {
"dir": "lower",
"floor": 2968,
"tolerance_pct": 35,
"value": 742
},
"ram.sN.query.p99us": {
"dir": "lower",
"floor": 3392,
"tolerance_pct": 35,
"value": 848
},
"ram.sN.read.ops_sec": {
"dir": "higher",
"floor": 372,
"tolerance_pct": 35,
"value": 1488
},
"ram.sN.read.p50us": {
"dir": "lower",
"floor": 2504,
"tolerance_pct": 35,
"value": 626
},
"ram.sN.read.p99us": {
"dir": "lower",
"floor": 3228,
"tolerance_pct": 35,
"value": 807
},
"ram.sN.seed.ops_sec": {
"dir": "higher",
"floor": 74404,
"tolerance_pct": 15,
"value": 297619
},
"ram.sN.seed.p50us": {
"dir": "lower",
"floor": 12,
"tolerance_pct": 15,
"value": 3
},
"ram.sN.seed.p99us": {
"dir": "lower",
"floor": 28,
"tolerance_pct": 15,
"value": 7
},
"ram.sN.write.ops_sec": {
"dir": "higher",
"floor": 634,
"tolerance_pct": 15,
"value": 2537
},
"ram.sN.write.p50us": {
"dir": "lower",
"floor": 2408,
"tolerance_pct": 15,
"value": 602
},
"ram.sN.write.p99us": {
"dir": "lower",
"floor": 3552,
"tolerance_pct": 15,
"value": 888
}
}

2
bench/results/.gitignore vendored Normal file
View file

@ -0,0 +1,2 @@
*
!.gitignore

View file

@ -290,6 +290,7 @@ let stdlib_members : stdlib_member list =
(* time *)
m "time" "now" 0 0 (Some (TScalar "Int")) None;
m "time" "sleep" 1 46 None None;
m "time" "ticks" 0 84 (Some (TScalar "Int")) None;
m "time" "local" 1 47 (Some (TScalar time_record_name)) (Some time_record_name);
m "time" "iso" 1 48 (Some (TScalar "Text")) None;
(* env *)

View file

@ -32,7 +32,7 @@ flowchart TD
I9c["20 cross-program tables (half-built)"]:::open
I9d["21 keypair attach auth (half-built; crypto+handshake already on its branch)"]:::open
I9e["22 durability + throughput baseline"]:::open
I9e["22 durability + throughput baseline ✅ 2026-08-21"]:::done
I8["8 shard-actor runtime ✅ 2026-08-21"]:::done
I9f["23 io_uring group-commit"]:::open
I10["10 HTTP service layer (lowers onto the framework)"]:::open

View file

@ -0,0 +1,63 @@
# db-bench — iteration 22's load generator
The measurement backbone (spec:
`docs/superpowers/specs/2026-08-21-db-bench-design.md`). Pure `.wo`;
every measured mode prints one machine-parsable line per operation
class:
<op> <count> <ops/sec> <p50us> <p99us>
Timing is per-operation via `time.ticks` (CLOCK_MONOTONIC µs).
Percentiles come from a 1µs-bucket histogram clamped at 20000µs — exact
to the microsecond below the clamp; a p99 AT 20000 means "clamp or
worse". (A histogram, not the spec's reservoir: the language has no
container element-write or sort, and the histogram's tail fidelity is
strictly better. Recorded as a plan deviation.)
## Modes
| mode | what it prices |
| --- | --- |
| `all N` | the throughput campaign in ONE process: seed N, read N/2, query N/10, write N/2. Without `WO_DATA` the store is RAM and dies with the process, so the measured modes must share the seeding run. |
| `seed N` | timed inserts: one parent per 100 children (FK probe each insert, unique-index maintenance per parent), k non-unique (10 rows/key), deterministic v. Writes `Meta` expectation rows. |
| `read N` | indexed take-1 point lookups, LCG-spread keys. |
| `query N` | full equality probes on the k index (≈10 rows each), materialized and counted. |
| `write N` | alternating inserts (disjoint k range 2e6+) and updates through query results. Corrupts the checksum by design — durability legs run on a fresh store. |
| `wal N` | the crash battery's vehicle: insert-only (k range 1e6+), `acked <i>` printed AFTER each insert returns — the return IS the ack (RAM applied, WAL record staged, ONE commit done). |
| `verify` | store vs its own Meta rows: count, checksum, one unique probe. Exit 3 on mismatch. |
| `verify-acked M` | after kill -9 mid-`wal`: rows 1..M exist with the right v; rows beyond M allowed (acked after the last print flushed). Exit 3 on mismatch. |
## Coordination idiom (this side of iteration 31)
There is no request/response surface yet: concurrent modes drive
completion the db-actor way — actors write rows, main polls the store
until the expected count, then settles. Retired when 31 lands.
## Standing finding (2026-08-21, first run)
A hand-built `multi Bucket` of insert results SEGVs on drop: the
compiler classifies the elements OWNED while table refs are scalar ids.
Query-built multis are runtime-typed and safe. Worked around here
(single ref local, bucket-major seeding); the compiler fix is its own
slice.
## Reference-machine numbers (first campaign, 2026-08-21)
`bench/baseline.json` is the contract; headline readings:
- ram seed 257–298k inserts/s; **durable seed ≈4.5k/s** (fsync-per-commit
≈220µs each — the gap iteration 23 exists to close).
- reads ≈1.5k/s at p50 ≈600µs on a 20k-row store: point lookups are
O(table) — the probe walks every slab; index selection never reaches
the lookup path. THE read-path finding.
- mixread 1,280 ops/s single-shard vs **21 ops/s** multi-shard: RPC
round-trip × O(table) probes × owner serialization — the arc's honest
price until reads index properly.
- msgrate 13.4M msgs/s same-heap vs 2.45M cross-shard (the mutex-inbox
number, stage-2 deviation 4).
## The gate must bite (proven 2026-08-21)
`scripts/db-bench.py --check <results.json>` evaluates a recorded run:
the real results pass 74/0; a doctored copy (one ops/sec halved) FAILS
on exactly that metric. Re-run the smoke after any gate change.

View file

@ -0,0 +1,523 @@
use time
-- db-bench — iteration 22's load generator. Every measured mode prints
-- one machine-parsable line per operation class:
--
-- <op> <count> <ops/sec> <p50us> <p99us>
--
-- Timing is per-operation via time.ticks (CLOCK_MONOTONIC µs);
-- percentiles come from a 1µs-bucket histogram clamped at HIST_CLAMP —
-- exact to the microsecond below the clamp, and the clamp bucket keeps
-- the tail honest (a p99 AT the clamp means "clamp or worse").
-- The wal mode prints a running `acked <n>` line after every insert
-- RETURNS (the return IS the ack): the crash battery kills this mode
-- mid-run and verify-acked proves every acknowledged row survived.
-- ---- deterministic helpers ----
fn lcg(seed: Int) -> Int {
let x = seed * 1103515245 + 12345;
if x < 0 {
x = 0 - x;
}
return x;
}
fn item_v(i: Int) -> Int {
return (i * 37) % 1000;
}
-- ---- the histogram (percentiles without a sort) ----
fn hist_add(mut h: map<Int, Int>, us: Int) {
let b = us;
if b < 0 {
b = 0;
}
if b > 20000 {
b = 20000;
}
if has(h, b) {
set(h, b, get(h, b) + 1);
} else {
set(h, b, 1);
}
}
fn hist_pct(h: map<Int, Int>, total: Int, pct: Int) -> Int {
let target = total * pct / 100;
if target < 1 {
target = 1;
}
let seen = 0;
let b = 0;
while b <= 20000 {
if has(h, b) {
seen = seen + get(h, b);
if seen >= target {
return b;
}
}
b = b + 1;
}
return 20000;
}
fn report(op: Text, n: Int, total_us: Int, h: map<Int, Int>) {
let us = total_us;
if us < 1 {
us = 1;
}
let rate = n * 1000000 / us;
print("${op} ${n} ${rate} ${hist_pct(h, n, 50)} ${hist_pct(h, n, 99)}");
}
-- ---- modes ----
-- seed N: N children, one parent per 100, k = i % (N/10) (10 rows per
-- key), v deterministic. Meta rows record the expectations verify reads.
fn seed(n: Int) -> Int {
let h: map<Int, Int> = {};
let kmod = n / 10;
if kmod < 1 {
kmod = 1;
}
-- bucket-major: one parent, then its 100 children, using a single ref
-- local. (A hand-built `multi Bucket` of insert results SEGVs on drop —
-- the compiler classifies the elements OWNED while table refs are
-- scalar ids; recorded as a standing finding, not this iteration's fix.
-- Query-built multis are runtime-typed and safe.)
let vsum = 0;
let t0 = time.ticks();
let i = 1;
let b = 0;
while i <= n {
let bref = insert Bucket { tag: "b${b}" };
b = b + 1;
let j = 0;
while j < 100 and i <= n {
let o0 = time.ticks();
insert Item { k: i % kmod, v: item_v(i), bucket: bref };
hist_add(h, time.ticks() - o0);
vsum = vsum + item_v(i);
i = i + 1;
j = j + 1;
}
}
let t1 = time.ticks();
insert Meta { tag: "count", val: n };
insert Meta { tag: "vsum", val: vsum };
insert Meta { tag: "kmod", val: kmod };
report("seed", n, t1 - t0, h);
return 0;
}
fn meta_val(tag: Text) -> Int {
let ms = from m in Meta where m.tag == tag take 1 select m;
if len(ms) == 0 {
return -1;
}
return ms[0].val;
}
-- read N: indexed take-1 point lookups (the point-read this surface
-- offers), keys spread by LCG over the seeded key range.
fn read_mode(n: Int) -> Int {
let kmod = meta_val("kmod");
if kmod < 1 {
print_err("read: seed first");
return 1;
}
let h: map<Int, Int> = {};
let sink = 0;
let s = 42;
let t0 = time.ticks();
let i = 0;
while i < n {
s = lcg(s);
let key = s % kmod;
let o0 = time.ticks();
let xs = from x in Item where x.k == key take 1 select x;
if len(xs) > 0 {
sink = sink + xs[0].v;
}
hist_add(h, time.ticks() - o0);
i = i + 1;
}
let t1 = time.ticks();
report("read", n, t1 - t0, h);
if sink < 0 {
print("impossible ${sink}");
}
return 0;
}
-- query N: full equality probes on the k index (≈10 rows per key),
-- each materialized and counted.
fn query_mode(n: Int) -> Int {
let kmod = meta_val("kmod");
if kmod < 1 {
print_err("query: seed first");
return 1;
}
let h: map<Int, Int> = {};
let rows = 0;
let s = 7;
let t0 = time.ticks();
let i = 0;
while i < n {
s = lcg(s);
let key = s % kmod;
let o0 = time.ticks();
for x in from x in Item where x.k == key select x {
rows = rows + 1;
}
hist_add(h, time.ticks() - o0);
i = i + 1;
}
let t1 = time.ticks();
report("query", n, t1 - t0, h);
print("query rows ${rows}");
return 0;
}
-- write N: alternating inserts (disjoint k range 2e6+) and updates
-- through a query result. Corrupts vsum by design — the durability legs
-- run on their own fresh store.
fn write_mode(n: Int) -> Int {
let kmod = meta_val("kmod");
if kmod < 1 {
print_err("write: seed first");
return 1;
}
let bs = from b in Bucket where b.tag == "b0" take 1 select b;
if len(bs) == 0 {
print_err("write: no buckets");
return 1;
}
let h: map<Int, Int> = {};
let s = 99;
let t0 = time.ticks();
let i = 0;
while i < n {
let o0 = time.ticks();
if i % 2 == 0 {
insert Item { k: 2000000 + i, v: item_v(i), bucket: bs[0] };
} else {
s = lcg(s);
let key = s % kmod;
let xs = from x in Item where x.k == key take 1 select x;
if len(xs) > 0 {
xs[0].v = xs[0].v + 1;
}
}
hist_add(h, time.ticks() - o0);
i = i + 1;
}
let t1 = time.ticks();
report("write", n, t1 - t0, h);
return 0;
}
-- wal N: the crash battery's vehicle — insert-only, disjoint k range
-- (1e6+), `acked <i>` printed AFTER each insert returns (the return is
-- the ack: RAM applied, record staged, ONE commit done).
fn wal_mode(n: Int) -> Int {
let bs = from b in Bucket where b.tag == "b0" take 1 select b;
if len(bs) == 0 {
push(bs, insert Bucket { tag: "b0" });
}
let i = 1;
while i <= n {
insert Item { k: 1000000 + i, v: item_v(i), bucket: bs[0] };
print("acked ${i}");
i = i + 1;
}
return 0;
}
-- verify: the store against its own Meta expectations — count, checksum,
-- one unique-index probe. Exit 3 on any mismatch.
fn verify() -> Int {
let want_n = meta_val("count");
let want_sum = meta_val("vsum");
if want_n < 0 or want_sum < 0 {
print_err("verify: no meta (seed first)");
return 3;
}
let got_n = 0;
let got_sum = 0;
for x in from x in Item select x {
if x.k < 1000000 {
got_n = got_n + 1;
got_sum = got_sum + x.v;
}
}
if got_n != want_n or got_sum != want_sum {
print_err("verify: count ${got_n}/${want_n} sum ${got_sum}/${want_sum}");
return 3;
}
let bs = from b in Bucket where b.tag == "b0" take 1 select b;
if len(bs) == 0 {
print_err("verify: unique probe b0 missing");
return 3;
}
print("verify ok ${got_n} rows sum ${got_sum}");
return 0;
}
-- verify-acked M: after a kill -9 mid-wal — rows 1..M (k = 1e6+i) must
-- exist with the right v; rows beyond M are allowed (acked after the
-- last print landed). Exit 3 on any missing/wrong row.
fn verify_acked(m: Int) -> Int {
let i = 1;
while i <= m {
let key = 1000000 + i;
let xs = from x in Item where x.k == key take 1 select x;
if len(xs) == 0 {
print_err("verify-acked: row ${i} missing");
return 3;
}
if xs[0].v != item_v(i) {
print_err("verify-acked: row ${i} v ${xs[0].v} != ${item_v(i)}");
return 3;
}
i = i + 1;
}
print("verify-acked ok ${m} rows");
return 0;
}
-- ---- the concurrent modes (mix, msgrate) ----
fn hist_dump(h: map<Int, Int>, kind: Int) {
let b = 0;
while b <= 20000 {
if has(h, b) {
insert Hist { kind: kind, b: b, c: get(h, b) };
}
b = b + 1;
}
}
-- One mixer = one actor: 90/10 read/write over the seeded store. On a
-- worker shard every statement below rides the stage-3 DB RPC — the
-- code must not know or care (transparency is the point). Done signal:
-- a Meta row main polls for (the coordination idiom this side of 31).
class Mixer {
id: Int
fn receive(msg: MixJob) {
let hr: map<Int, Int> = {};
let hw: map<Int, Int> = {};
let s = msg.seed;
let sink = 0;
let i = 0;
while i < msg.ops {
s = lcg(s);
let key = s % msg.kmod;
let o0 = time.ticks();
if i % 10 == 9 {
let xs = from x in Item where x.k == key take 1 select x;
if len(xs) > 0 {
xs[0].v = xs[0].v + 1;
}
hist_add(hw, time.ticks() - o0);
} else {
let xs = from x in Item where x.k == key take 1 select x;
if len(xs) > 0 {
sink = sink + xs[0].v;
}
hist_add(hr, time.ticks() - o0);
}
i = i + 1;
}
hist_dump(hr, 0);
hist_dump(hw, 1);
insert Meta { tag: "mixdone${self.id}", val: sink };
}
}
fn mix_mode(total: Int, c: Int) -> Int {
let kmod = meta_val("kmod");
if kmod < 1 {
print_err("mix: seed first");
return 1;
}
let per = total / c;
if per < 1 {
per = 1;
}
let wall0 = time.ticks();
let i = 0;
while i < c {
let a: actor MixJob = spawn Mixer { id: i };
send(a, MixJob { ops: per, seed: 1000 + i * 7919, kmod: kmod });
i = i + 1;
}
-- poll until every mixer's done row exists
let done = 0;
while done < c {
time.sleep(20);
done = 0;
i = 0;
while i < c {
if meta_val("mixdone${i}") >= 0 {
done = done + 1;
}
i = i + 1;
}
}
let wall = time.ticks() - wall0;
-- merge the dumped histograms; wall time is shared by both classes
let hr: map<Int, Int> = {};
let hw: map<Int, Int> = {};
let nr = 0;
let nw = 0;
for x in from x in Hist select x {
if x.kind == 0 {
if has(hr, x.b) {
set(hr, x.b, get(hr, x.b) + x.c);
} else {
set(hr, x.b, x.c);
}
nr = nr + x.c;
} else {
if has(hw, x.b) {
set(hw, x.b, get(hw, x.b) + x.c);
} else {
set(hw, x.b, x.c);
}
nw = nw + x.c;
}
}
report("mixread", nr, wall, hr);
report("mixwrite", nw, wall, hw);
return 0;
}
-- msgrate: one-way flood — main sends N messages at a sink actor; the
-- sink counts and writes the done row at N. Spawn TWO sinks and flood
-- the second: round-robin placement puts it off the primary whenever
-- more than one shard exists, so the multi-shard number prices the
-- mutex inbox (stage-2 deviation 4's number); single-shard prices the
-- same-heap path.
class Sink {
got: Int
fn receive(msg: Flood) {
self.got = self.got + 1;
if self.got == msg.n {
insert Meta { tag: "flooddone", val: self.got };
}
}
}
fn msgrate_mode(n: Int) -> Int {
let first: actor Flood = spawn Sink { got: 0 };
let a: actor Flood = spawn Sink { got: 0 };
if first == a {
print_err("msgrate: impossible");
}
let t0 = time.ticks();
let i = 0;
while i < n {
send(a, Flood { n: n });
i = i + 1;
}
while meta_val("flooddone") < 0 {
time.sleep(5);
}
let us = time.ticks() - t0;
if us < 1 {
us = 1;
}
print("msgrate ${n} ${n * 1000000 / us}");
return 0;
}
-- all N: the throughput campaign in ONE process — without WO_DATA the
-- store is RAM and dies with the process, so seed and the measured
-- modes must share a run; under WO_DATA the same mode prices the
-- durable flavor. Restart/crash legs use the separate modes.
fn all_mode(n: Int) -> Int {
let rc = seed(n);
if rc != 0 {
return rc;
}
rc = read_mode(n / 2);
if rc != 0 {
return rc;
}
rc = query_mode(n / 10);
if rc != 0 {
return rc;
}
rc = write_mode(n / 2);
if rc != 0 {
return rc;
}
-- mix at N/10: every point lookup is O(table) today (the probe walks
-- all slabs — a headline finding, not a bug to hide), so a read-heavy
-- mix over a seeded store is quadratic in N. The campaign driver
-- chooses absolute sizes; this keeps `all` finishing in minutes.
return mix_mode(n / 10, 4);
}
fn usage() -> Int {
print_err("usage: db-bench <mode>");
print_err(" all N | seed N | read N | query N | write N | wal N");
print_err(" mix N C | msgrate N | verify | verify-acked M");
return 2;
}
fn main(args: multi Text) -> Int {
if len(args) < 1 {
return usage();
}
if args[0] == "verify" {
return verify();
}
if len(args) < 2 {
return usage();
}
let n = parse_int(args[1]);
if n == nil or n < 1 {
print_err("db-bench: <n> must be a positive number");
return 2;
}
if args[0] == "all" {
return all_mode(n);
}
if args[0] == "seed" {
return seed(n);
}
if args[0] == "read" {
return read_mode(n);
}
if args[0] == "query" {
return query_mode(n);
}
if args[0] == "write" {
return write_mode(n);
}
if args[0] == "wal" {
return wal_mode(n);
}
if args[0] == "verify-acked" {
return verify_acked(n);
}
if args[0] == "msgrate" {
return msgrate_mode(n);
}
if args[0] == "mix" {
if len(args) < 3 {
return usage();
}
let c = parse_int(args[2]);
if c == nil or c < 1 {
print_err("db-bench: <c> must be a positive number");
return 2;
}
return mix_mode(n, c);
}
return usage();
}

View file

@ -0,0 +1,45 @@
-- The bench store: the employee shape reduced to what pricing needs —
-- a parent with a @unique Text column (unique-index maintenance), a
-- child with a ref parent (FK probe on insert) and two indexed columns
-- (the equality-probe path). Meta rows carry seed expectations so
-- `verify` checks the store against facts that survived the same WAL.
@table(name: "buckets", index: [tag])
class Bucket {
tag: Text @unique
}
@table(name: "items", index: [k], index: [bucket])
class Item {
k: Int -- probe key; seeded non-unique (10 rows per key),
-- wal/write modes use disjoint high ranges
v: Int -- payload column, checksummed by verify
bucket: ref Bucket -- FK: primary-index probe on every insert
}
@table(name: "meta", index: [tag])
class Meta {
tag: Text @unique
val: Int
}
-- mix actors dump their per-op histograms here (kind 0 = read,
-- 1 = write); main scans and merges — exact aggregate percentiles,
-- and the merge itself dogfoods the store.
@table(name: "hist")
class Hist {
kind: Int
b: Int
c: Int
}
-- messages (Int-only payloads: ownership moves, nothing borrowed)
class MixJob {
ops: Int
seed: Int
kmod: Int
}
class Flood {
n: Int
}

View file

@ -0,0 +1,6 @@
name = "db-bench"
version = "0.1.0"
description = "iteration 22: the measurement backbone — timed DB workloads, single- and multi-shard"
[runtime]
wo = ">= 0.1"

View file

@ -0,0 +1,106 @@
# Guide — creating the `database-developer` subagent
A project subagent is one markdown file in `.claude/agents/` (this
repo) or `~/.claude/agents/` (every repo). Claude Code loads it at
session start; the main conversation can then delegate matching work to
it via the Agent tool, and you can name it directly ("use the
database-developer agent").
## 1. The file format
`.claude/agents/database-developer.md` — YAML frontmatter + a system
prompt body:
- `name` — kebab-case; becomes the agent type.
- `description` — WHEN to use it. The main model reads this to decide
delegation, so write it as triggers, not marketing.
- `tools` — allowlist. Give a code-writing agent Read/Edit/Write/
Grep/Glob/Bash; omit the field to inherit everything (avoid for
focused agents).
- `model` (optional) — pin a tier; omit to inherit the session's.
- Body — the agent's system prompt: doctrine, file map, gates,
boundaries. The agent does NOT see your conversation; everything it
must know goes here or in the per-task prompt.
## 2. Ready-to-paste definition
Save as `.claude/agents/database-developer.md`:
```markdown
---
name: database-developer
description: Engine work under database/src (tables, WAL, indexes, slot
encode/decode) and the DB seams in runtime/src (db builtins, the DB
actor RPC). Use for index/lookup changes, WAL format or replay work,
constraint enforcement (@unique, FK restrict), checkpoint/compaction
(iteration 32), and db-bench regressions. NOT for compiler surface,
fibers/scheduler, or framework .wo code.
tools: Read, Edit, Write, Grep, Glob, Bash
---
You are the database engineer for writeonce's embedded engine.
Doctrine (non-negotiable):
- C11 + libc only. No new dependencies, no atomics on the data path.
- RAM is authoritative; the WAL makes it durable. An ack means the
commit fsynced. Replay is whole-or-not-at-all; torn tails drop.
- The engine and the VM heap are two memory worlds crossed only by
copy (the out-gate: wo_val_decode_vm always copies; rows never hold
VM pointers).
- Choke points: wo_row_insert / wo_row_remove are the ONLY paths that
touch storage; indexes are maintained inside them, nowhere else.
- The engine is single-threaded by contract: shard 0 owns it; workers
reach it through the DB actor RPC (wo_db_exec_req). Never add locks;
never read another shard's VM heap.
File map:
- database/src/table.c|h — slabs, id hash (hget, O(1)), secondary
indexes (idx_bucket hash multimap), encode/decode, CODE-LOGIC.md.
- database/src/wal.c|h — record grammar, staged batch, commit, replay.
- database/src/db.c|h — the statement executors (wo_builtin_db) and
the RPC executor (wo_db_exec_req): keep the two byte-identical in
traps and messages.
- runtime/src/vm.c — the requester half (wo_db_rpc); builtin.c routes.
- Contracts: docs/plan/oop-vm/04-db-binding.md (normative — extend it
when formats change). Benchmarks: docs/examples/db-bench,
bench/baseline.json.
Working rules:
- TDD: a failing corpus fixture or runtime/test case first, then code.
- Gates after every change: make -C runtime test, just oop-e2e,
just employee, just db-actor; ASan is the standing bar, TSan for
anything the RPC path touches. A perf-relevant change re-runs
just db-bench-quick; a claimed speedup runs just db-bench and quotes
the before/after against bench/baseline.json.
- Match existing style; comments state constraints, not narration.
- Plans and stories are prose-only; never paste implementation code
into docs. Commit drafts follow the repo's bullet style, ≤25 lines.
Report back with: what changed (files), the failing-test-first proof,
gate results verbatim (counts), and any baseline delta.
```
## 3. Verify it loads
New session (agents load at start), then: "use the database-developer
agent to explain the probe path in database/src/db.c". The reply must
come labeled as the subagent. `claude agents` (or the agents listing in
`/help`) shows registered agents.
## 4. Division of labor
- The MAIN session keeps: brainstorming/specs/plans (superpowers path),
board + story sync, cross-cutting refactors.
- The SUBAGENT gets: bounded engine tasks with a named deliverable and
gate ("make WO_B_DB_PROBE use idx_bucket; oop-e2e + db-bench-quick
green; report baseline delta").
- Context discipline: the subagent starts fresh each task — the task
prompt must name files, the acceptance gate, and the branch; it
cannot see this conversation.
## 5. Maintenance
The definition is code: review it in diffs, update the file map when
files move (the CODE-LOGIC.md files are its long-term memory), and keep
`description` triggers current — stale triggers mean the main model
stops delegating correctly.

View file

@ -1,36 +0,0 @@
# In progress — iteration 22: db-bench, the measurement backbone
> **Status: 🔄 in progress** (spec + plan approved 2026-08-21) — second
> slice of the concurrency chain **✅ stage 3 → 22 → 31 → 24 → 23 → 32**.
> Board: [../stories/00-status.md](../stories/00-status.md).
>
> One marker doc per active slice; deleted when the slice lands.
## What
Execute [`superpowers/plans/2026-08-21-db-bench.md`](../superpowers/plans/2026-08-21-db-bench.md)
(spec: [`2026-08-21-db-bench-design.md`](../superpowers/specs/2026-08-21-db-bench-design.md),
approved): the `time.ticks` µs builtin, the `docs/examples/db-bench`
sample (seed/read/query/write/mix/msgrate/verify), the campaign driver
with relative gates vs `bench/baseline.json`, the restart proof and
kill -9 battery at both shard counts, and the first committed baseline.
## Why now
Nothing performance-shaped is sourced until this runs — and the arc owes
its before/after. One campaign now covers single- AND multi-shard
honestly (stage 3 landed), prices the RPC, and produces the mutex-inbox
number stage-2's deviation waits on.
## Story
- [22 — durability, throughput, scale](../stories/language-runtime-database/in-progress/22-durability-throughput-scale.md)
## Definition of done
- Plan Tasks 1–6 checked; `just db-bench` exit 0 twice in a row;
gate-bites smoke proven (doctored results FAIL); baseline committed
with rationale; arc delta + msgrate copied into story 8; full battery
green.
- Story 22 → done/; board standup written from the actual numbers; this
file deleted. Next slice: iteration 31 (actor lifecycle).

View file

@ -1,6 +1,6 @@
# PostgreSQL — storage subsystem reference
These cards exist to make the Postgres backend a useful **library of patterns** for writeonce's persistent-storage phases (10–12) without inviting a multi-process port. Each card pulls one subsystem out of [`reference/postgresql/src/backend/`](../../../../.dev/reference/postgresql/src/backend/) — paths into the Postgres tree, the underlying *idea*, and the writeonce translation.
These cards exist to make the Postgres backend a useful **library of patterns** for writeonce's storage, constraint, and index work without inviting a multi-process port. (The storage cards originally fed the Rust-era plans 10–12, removed with that track 2026-08-18; the patterns fed the shipped C engine and remain the reference.) Each card pulls one subsystem out of [`reference/postgresql/src/backend/`](../../../../.dev/reference/postgresql/src/backend/) — paths into the Postgres tree, the underlying *idea*, and the writeonce translation.
The symlink is user-specific:
@ -18,6 +18,8 @@ Gitignored — see [`.gitignore`](../../../../.gitignore). Pair it with [`refere
| [smgr-and-md](./smgr-and-md.md) | `storage/smgr/{md,smgr,bulk_write}.c` | one file per relation, segments capped at `RELSEG_SIZE`, immediate vs deferred fsync | multi-fork abstraction (main/fsm/vm), shared-memory descriptor cache |
| [buffer-and-checkpoint](./buffer-and-checkpoint.md) | `storage/buffer/{bufmgr,freelist}.c` + `postmaster/{checkpointer,bgwriter}.c` | page cache + dirty bit + LRU; checkpoint flushes then advances control-file LSN | shared-buffer pinning/unpinning, separate writer processes, latches |
| [page-format](./page-format.md) | `storage/page/{bufpage,checksum}.c` | page header (LSN, checksum, free-space markers); CRC32C trailers | MVCC visibility (xmin/xmax/ctid), access-method-specific opaque space |
| [constraints-and-grammar](./constraints-and-grammar.md) | `parser/gram.y`, `catalog/pg_constraint.h`, `utils/adt/ri_triggers.c` | PK = blessed unique index; FK forward-only catalog + inline-check semantics; ON DELETE action set; backlink-implies-index (our improvement) | trigger machinery, deferrable constraints, MATCH PARTIAL, composite keys |
| [indexing-and-point-lookup](./indexing-and-point-lookup.md) | `access/{nbtree,hash}/README`, `optimizer/path/costsize.c`, `storage/itemptr.h` | hash-bucket point lookup (expected O(1)), index-entry-as-row-address (TID ↔ our slot), selectivity beats seqscan by arithmetic | btree/gin/gist/spgist/brin AMs, cost-based planner, index paging |
## The lift-vs-skip filter
@ -41,8 +43,14 @@ What stays out:
The implementation phases that lean on this material:
- [`docs/plan/10-storage-foundations.md`](../../10-storage-foundations.md) — page format and segment files. Lifts ideas from `smgr/md.c` and `page/bufpage.h`.
- [`docs/plan/11-wal-and-recovery.md`](../../11-wal-and-recovery.md) — WAL framing, group commit, recovery loop. Lifts ideas from `access/transam/xlog.c` and the xlog-recovery family.
- [`docs/plan/12-engine-disk-cutover.md`](../../12-engine-disk-cutover.md) — buffer cache + dirty tracking. Lifts ideas from `storage/buffer/bufmgr.c` and `postmaster/checkpointer.c`.
- The Rust-era consumers (plans 10/11/12: storage foundations, WAL and
recovery, disk cutover) were removed with that track 2026-08-18; their
ideas shipped in `database/src/` (typed WAL + replay) and the rest wait
on [iteration 32](../../../stories/language-runtime-database/refine/32-wal-checkpoint.md)
(checkpoint) — the wal/buffer cards are its entry material.
- Current consumers: [constraints-and-grammar](./constraints-and-grammar.md)
(the `@table` PK/FK grammar direction) and
[indexing-and-point-lookup](./indexing-and-point-lookup.md) (the
O(1) read-path slice iteration 22's numbers demand).
Pair each card with [`docs/plan/exploration/linux/12-pwrite-fsync.md`](../linux/12-pwrite-fsync.md) for the actual syscalls — these cards are about *design patterns*, that one is about *kernel calls*.

View file

@ -65,7 +65,7 @@ For the checkpoint:
## Used by
- [`docs/plan/12-engine-disk-cutover.md`](../../12-engine-disk-cutover.md) — disk-backed engine reads and dirty-row tracking.
- [`docs/plan/11-wal-and-recovery.md`](../../11-wal-and-recovery.md) — control file write sequence (phase 11 ships the control file; checkpoint as a periodic step lands with phase 12 or shortly after).
- `docs/plan/12-engine-disk-cutover.md` (Rust-era, removed 2026-08-18) — disk-backed engine reads and dirty-row tracking.
- `docs/plan/11-wal-and-recovery.md` (Rust-era, removed 2026-08-18) — control file write sequence (phase 11 ships the control file; checkpoint as a periodic step lands with phase 12 or shortly after).
Pair with [`linux/12-pwrite-fsync.md`](../linux/12-pwrite-fsync.md) for the fsync semantics and [`linux/08-mmap.md`](../linux/08-mmap.md) for the OS page-cache backstory.

View file

@ -0,0 +1,102 @@
# Constraints & DDL grammar — PK / FK / reverse navigation
What Postgres' CREATE TABLE grammar and catalog do for PRIMARY KEY,
FOREIGN KEY/REFERENCES, and reverse lookup — and the `@table` grammar
writeonce should grow from it. Tree: post-18 master
(`REL_18_BETA1-2871`); paths into
[`reference/postgresql/`](../../../../.dev/reference/postgresql/).
## The Postgres side (facts, with paths)
**Grammar** (`src/backend/parser/gram.y`):
- Column constraints (`ColConstraintElem`, :4119): `UNIQUE` (:4140,
with `NULLS [NOT] DISTINCT`), `PRIMARY KEY` (:4153), and
`REFERENCES qualified_name opt_column_list key_match key_actions`
(:4224).
- Table-level twins (`ConstraintElem`, :4376): multi-column
`UNIQUE (...)` :4404, `PRIMARY KEY (...)` :4439 (or adopt an
existing index, :4457), `FOREIGN KEY (...) REFERENCES ...` :4493.
- REFERENCES options: `key_match` = MATCH FULL | PARTIAL
(unimplemented, errors) | SIMPLE (default) — :4620; `key_actions` =
`ON UPDATE`/`ON DELETE` × { NO ACTION | RESTRICT | CASCADE |
SET NULL | SET DEFAULT } — :4664–4733. Single-char codes in
`src/include/nodes/parsenodes.h:2928`.
**Catalog** (`src/include/catalog/pg_constraint.h`):
- One row per constraint; `contype` `'p'`/`'f'`/`'u'` (:198). FK rows
carry the FORWARD direction only: `conrelid`/`conkey[]` (referencing)
→ `confrelid`/`confkey[]` (referenced), plus the action/match chars
(:98–130).
- **A PK/UNIQUE constraint IS an index**: `transformIndexConstraints`
(`src/backend/parser/parse_utilcmd.c:2245`) rewrites the constraint
into an `IndexStmt` (`index->primary`, `index->unique`) — the
constraint and its unique index are one object (`conindid`,
`index_constraint_create`, `catalog/index.c:1903`). FKs are
transformed AFTER indexes deliberately (:3023).
**FK enforcement = trigger pairs** (`utils/adt/ri_triggers.c`):
- Referencing side: INSERT/UPDATE fire `RI_FKey_check` (:358) —
`SELECT 1 FROM <pktable> WHERE pk = $1 FOR KEY SHARE` (a probe of the
PK's unique index; :452 even has a direct-index fast path bypassing
SPI).
- Referenced side: DELETE/UPDATE fire the action triggers —
restrict/noaction = `SELECT 1 FROM <fktable> WHERE fk = $1` (:903),
cascade = `DELETE FROM <fktable> WHERE fk = $1` (:1089), setnull/
setdefault = the obvious UPDATEs. NO ACTION vs RESTRICT differ only
in deferrability + a replacement-row re-check (:872) — "the only
difference", per the source comment.
**Reverse navigation — the load-bearing negative:**
- Postgres stores NO backlink. A reverse lookup is a plain scan of the
referencing table (`WHERE $1 = fkatt1`); it is fast iff an index on
the FK column exists. That index is recommended, NOT auto-created
(`doc/src/sgml/ddl.sgml:1390`), because indexing choices vary. The
referenced side, by contrast, ALWAYS has an index — it must be a
PK/unique.
## The writeonce translation
What exists today: every class is a table; the auto-assigned,
shard-interleaved `id` is the de-facto primary key (O(1) via the
table's open-addressing id hash); `@table(name:, index: [cols])`
declares secondary indexes; `@unique` on a column; `ref T` is a stored
FK (restrict-only, checked by `wo_row_has_referrers` — currently a
full scan); `backlink T.field` is a declared reverse view (currently
an O(table) scan too).
The grammar this study argues for (words, no code — an iteration's
brainstorm decides):
1. **Keep the id as THE primary key; add `@key` as a UNIQUE ALIAS, not
a replacement.** Postgres' lesson: a PK is just a unique index the
catalog blesses (`transformIndexConstraints`). writeonce already has
the blessed unique id; a user-declared `@key` on a column should
desugar to `@unique` + the natural-lookup index — never a second
row-identity (slabs, WAL records, and refs all speak id).
2. **`ref T` grows an action option, defaulting to today's behavior:**
`ref T` = restrict (current semantics, now named); optional
`@on_delete(cascade)` / `@on_delete(set_nil)` — the `?ref T` shape
is the precondition for set_nil, exactly as SET NULL requires a
nullable column in Postgres. MATCH variants: skip — single-column
refs only, MATCH SIMPLE semantics by construction.
3. **Backlink beats Postgres — if it implies the index.** Postgres
makes reverse lookup fast only when the user remembers the FK-column
index; writeonce's `backlink T.field` is a DECLARED intent, so the
compiler should auto-require `index: [field]` on the referencing
table (or inject it) — the study's one clear improvement over the
reference. `wo_row_has_referrers` and backlink reads then become
index probes, not scans (see the indexing card).
4. **Enforcement placement:** Postgres bolts FK checks on as triggers
because constraints arrived after the executor; writeonce's choke
points (`wo_row_insert`/`wo_row_remove`/`wo_row_update_field`) are
the honest home — checks stay inline, no trigger machinery, same
observable semantics (insert probes the referenced id's existence;
delete probes the referencing index).
Non-goals this study records: composite keys (no driving workload),
deferrable constraints (need `transaction { }` = held iteration 18),
MATCH PARTIAL (Postgres never shipped it either).

View file

@ -0,0 +1,92 @@
# Indexing & point lookup — how Postgres never full-scans for `=`
The access-method algorithms, and the mechanism that turns an equality
predicate into a direct row address instead of a table walk. Tree:
19devel; paths into
[`reference/postgresql/`](../../../../.dev/reference/postgresql/).
Written for the read-path finding iteration 22 measured: writeonce
point lookups are O(table) (~1.5k reads/s at p50 600µs on 20k rows).
## The access-method roster (facts, with paths)
| AM | algorithm | serves | point lookup |
| --- | --- | --- | --- |
| nbtree | Lehman–Yao B-tree (`access/nbtree/README:6`) | `< <= = >= >`, IN, ordered scans, prefix LIKE | O(log N) page descents |
| hash | Seltzer/Yigit extendible hashing (`access/hash/README:6`) | `=` only | O(1) expected: metapage + bucket-page binary search |
| gin | inverted index: btree of keys → posting lists (`access/gin/README:17`) | containment, full-text | bitmap-only (no amgettuple) |
| gist | generalized balanced tree, opclass `consistent` (`access/gist/README:8`) | overlap, kNN | multi-subtree descent |
| spgist | space-partitioned tries/quadtrees, non-balanced (`access/spgist/README:3`) | points, prefixes | data-bounded depth |
| brin | per-block-range min/max summaries (`access/brin/README:4`) | huge clustered scans | none — lossy bitmap |
Algorithm notes worth keeping:
- **btree**: Lehman–Yao's right-link + high-key lets descents run
lock-free past concurrent splits (`nbtree/README:17-29`); equality
and range use the SAME descent — `_bt_first` positions, `_bt_next`
walks siblings (`nbtsearch.c:883/:1586`); heap TID is a tiebreaker
making every key unique per level.
- **hash**: bucket count doubles at split points; one bucket splits at
a time (linear hashing, `hash/README:14-22,60-79`); bucket resolution
is a MASK — `bucket = hash & highmask; if > maxbucket then & lowmask`
(`hashutil.c:125`); entries sorted by hash within a page for binary
search. Fully WAL-logged in this tree (the old caveat is gone);
btree's remaining edge is capability, not durability: only btree does
unique constraints, ordered scans, and range predicates.
## Why `=` never scans the table
1. The planner builds BOTH paths and costs them: seqscan cost is
unconditionally whole-relation (`pages × seq_page_cost + tuples ×
cpu_tuple_cost`, `costsize.c:270`); index cost scales by
SELECTIVITY (`cost_index`, `:545`, delegating to the AM's
`amcostestimate`). A selective equality wins by arithmetic, not by
rule.
2. An index entry stores a **TID** — `(block, offset)`, 6 bytes
(`storage/itemptr.h:36`): the ROW'S ADDRESS. The executor path is
IndexScan → `btgettuple` → `index_fetch_heap` reads exactly ONE
heap page and one line pointer (`indexam.c:698`). The index answers
"where", the heap answers "what" — nothing walks.
## The writeonce translation — O(1) lookups are one wiring change
What exists (`database/src/table.c`):
- The id path is ALREADY the TID story: `hget` (open-addressing hash,
:260) maps id → global slot + 1, and the slot IS the address
(slab base + offset — addresses stable forever). O(1), proven by
db-bench's 297k inserts/s.
- Secondary indexes ALREADY exist as a hash multimap —
`db_index`/`db_ibucket` (`idx_hash`/`idx_bucket`, :302/:342), the
same expected-O(1) shape as Postgres' hash AM (minus its paging,
which a RAM-authoritative store does not need). Maintained inside
the insert/remove/update choke points, exactly where they belong.
The measured gap: **the read path never asks the index.** Both
`WO_B_DB_PROBE` (`db.c:125`) and its RPC twin in `wo_db_exec_req` read
the index metadata only for the key column's KIND, then walk EVERY
slab comparing values — Postgres' seqscan, unconditionally, on a
column that has a live hash index. `wo_row_has_referrers` (FK
restrict) is the same story across all tables.
Direction the next slice takes (words only):
1. Probe = `idx_bucket(ix, hash(key))`, then verify equality against
the bucket's ids via `wo_row_ptr` (a hash is a hint, never an
answer — the engine's own doctrine, already enforced on the unique
path). Expected O(1); the bucket wins by the same arithmetic that
makes Postgres pick the index.
2. Multi-column indexes probe on the FULL column set today
(`idx_hash` hashes all cols) — a single-column equality over a
composite index needs either a leading-column bucket layout or a
declared single-column index; the slice decides, the bench arbitrates.
3. FK restrict + backlink reads ride the same probe once the
grammar card's rule lands (backlink implies the FK-column index).
4. Non-goals, recorded: no btree (no ordered-scan workload yet —
`order by` sorts materialized results today), no planner (one AM,
one rule: indexed equality probes, everything else scans), no
paging (RAM-authoritative; the WAL is the disk story).
Acceptance shape for that slice: db-bench `read`/`query` move from
~1.5k ops/s to the same order as inserts; `bench/baseline.json`
refreshed with the delta recorded — the gate exists precisely so this
claim gets measured.

View file

@ -39,7 +39,7 @@ The body of the page after this header holds **line pointers** (`ItemIdData`, 4
## What writeonce does instead (phase 10)
Variable-length records, length-prefixed. The framing is in [`docs/plan/10-storage-foundations.md`](../../10-storage-foundations.md):
Variable-length records, length-prefixed. The framing is in `docs/plan/10-storage-foundations.md` (Rust-era, removed 2026-08-18):
```text
[u32 length LE][u8 flags][u8 record_kind][u64 LSN][payload bytes][u32 CRC32C]
@ -76,7 +76,7 @@ Slotted pages are the answer when those assumptions break. Until then, the frami
## Used by
- [`docs/plan/10-storage-foundations.md`](../../10-storage-foundations.md) — record framing borrows the **header + checksum** pattern from `bufpage.h`.
- [`docs/plan/12-engine-disk-cutover.md`](../../12-engine-disk-cutover.md) — when reading rows back from disk, CRC verification is the silent-corruption safety net the page header gives Postgres.
- `docs/plan/10-storage-foundations.md` (Rust-era, removed 2026-08-18) — record framing borrows the **header + checksum** pattern from `bufpage.h`.
- `docs/plan/12-engine-disk-cutover.md` (Rust-era, removed 2026-08-18) — when reading rows back from disk, CRC verification is the silent-corruption safety net the page header gives Postgres.
Pair with [`wal.md`](./wal.md) for the LSN convention and [`buffer-and-checkpoint.md`](./buffer-and-checkpoint.md) for the dirty-page semantics that pages need.

View file

@ -77,4 +77,4 @@ That's the core of phase 10's `SegStore`.
## Used by
[`docs/plan/10-storage-foundations.md`](../../10-storage-foundations.md) — segment file layout, append path. Pair with [`linux/09-fallocate.md`](../linux/09-fallocate.md) for preallocation, [`linux/12-pwrite-fsync.md`](../linux/12-pwrite-fsync.md) for the syscall details.
`docs/plan/10-storage-foundations.md` (Rust-era, removed 2026-08-18) — segment file layout, append path. Pair with [`linux/09-fallocate.md`](../linux/09-fallocate.md) for preallocation, [`linux/12-pwrite-fsync.md`](../linux/12-pwrite-fsync.md) for the syscall details.

View file

@ -63,4 +63,4 @@ Same effect as Postgres' group-commit fence (one `fsync` flushes many commits) w
## Used by
[`docs/plan/11-wal-and-recovery.md`](../../11-wal-and-recovery.md) — WAL framing, group commit, control file, replay loop. Pair with [`linux/12-pwrite-fsync.md`](../linux/12-pwrite-fsync.md) for syscall details.
`docs/plan/11-wal-and-recovery.md` (Rust-era, removed 2026-08-18) — WAL framing, group commit, control file, replay loop. Pair with [`linux/12-pwrite-fsync.md`](../linux/12-pwrite-fsync.md) for syscall details.

View file

@ -273,6 +273,7 @@ unset `env.get` are nil.
| `time.sleep(ms)` | — | |
| `time.local(ms)` | `-> TimeParts` | `{ year, month, day, hour, minute, second, dow }`, dow 0 = Sunday |
| `time.iso(ms)` | `-> Text` | UTC, second precision |
| `time.ticks()` | `-> Int` | CLOCK_MONOTONIC microseconds (id 84, iteration 22's bench clock) — monotone, never wall time; only differences mean anything |
| `env.get(name)` | `-> ?Text` | unset is nil |
| `env.stopping()` | `-> Bool` | SIGTERM/SIGINT latch, handlers installed on first use |
| `net.listen(host, port)` | `-> Int` | IPv4, SO_REUSEADDR, backlog 64; returns an fd |

View file

@ -34,52 +34,49 @@ machine-readable truth behind this board; live Obsidian Dataview views:
## ▶ NEXT PLAN
**The concurrency + fiber chain — ✅ stage 3 → 22 → 31 → 24 → 23 → 32**
(directive 2026-08-21). Active slice: **iteration 22, the measurement
backbone** — spec + plan approved 2026-08-21
([spec](../superpowers/specs/2026-08-21-db-bench-design.md) ·
[plan](../superpowers/plans/2026-08-21-db-bench.md) ·
[marker](../in-progress/2026-08-21-db-bench.md)); the four forks settled
as their leanings, plus `time.ticks` (µs clock) as the one runtime
addition and a new `db-bench` sample as the vehicle.
**The concurrency + fiber chain — ✅ stage 3 → ✅ 22 → 31 → 24 → 23 → 32**
(directive 2026-08-21). Next slice: **iteration 31, actor lifecycle** —
its spec brainstorm is the next act (four forks in
[the story](language-runtime-database/refine/31-actor-lifecycle.md);
the mailbox-cap/ring decisions now HAVE their mutex-inbox number).
**Implemented last time (2026-08-21):** the arc's **stage 3 — the
transparent DB actor landed, the arc is COMPLETE** (stories
[8](language-runtime-database/done/08-shard-actor-runtime.md) +
[11](language-runtime-database/done/11-fibers.md) → done/). A worker
shard's DB statement marshals to shard 0 (requester-side slot encode),
executes serialized on the owner, and the fiber resumes with the
materialized reply — `WO_T_DB` off the primary is gone. NEW gate
`just db-actor` 8/0; ASan/TSan clean; WO_DATA pair proves worker writes
are ack-after-durable and replay.
**Implemented last time (2026-08-21):** **iteration 22 landed — the
measurement backbone exists and every performance claim is now
sourced.** `docs/examples/db-bench` + `scripts/db-bench.py` +
`bench/baseline.json` (74 metrics, tolerance-tuned by a two-run
repeatability check) + `just db-bench`/`db-bench-quick`; `time.ticks`
(µs monotonic clock, builtin 84) as the one runtime addition. Restart
proof + 3× kill -9 battery per shard count all green; the gate bites
(doctored results fail on exactly the doctored metric).
**Key findings:** a latent stage-1 bug — io_uring ring params were ONE
shared static, rewritten by every shard's lazy init while others read
offsets from it: submits landed at garbage offsets and parked fibers
LOST WAKES (~1/20 hangs at default cores). Per-vm params fixed it; a
short `io_uring_enter` submit is now a loud trap. Also: single-binary
gates embed the runtime — rebuild the SAMPLE, not just wovm, or you
debug a stale binary.
**Key findings (measured, not asserted):** durable seed ≈4.5k
inserts/s vs ram ≈297k/s — the 66× fsync gap IS iteration 23's case;
point lookups are O(table) (the probe walks every slab — reads ≈1.5k/s
at p50 ≈600µs on 20k rows): the read path never uses the index for
lookup, a new candidate slice; mixread 1,280 ops/s single-shard vs 21
ops/s multi-shard — the DB-actor price under O(table) probes and owner
serialization; msgrate 13.4M msgs/s same-heap vs 2.45M cross-shard —
deviation 4's mutex-inbox number (rings stay unearned until this is
the bottleneck). Standing bug found: hand-built `multi <TableClass>`
SEGVs on drop (elements classed OWNED; refs are scalar ids).
**Learned from the last iteration:** the owner thread must never read a
requester's VM heap (concurrent mark-bit writes = TSan race) — marshal
by ENCODING on the requester's thread, execute from slots replay-style;
a plane-less park (`WO_PARK_INBOX`) + envelope wake is all an RPC reply
needs; a busy shard adopting its inbox once per reduction slice bounds
request latency.
**Learned:** benchmark tolerances must be per-class — mix* spreads 50%
run-to-run (scheduling), read/query jitter ~25%, seed/write/msgrate
hold at 15%; a RAM store dies with its process, so throughput modes
share one run (`all`); WO_DATA on tmpfs makes fsync free — durable
numbers need a real disk.
**Dependencies unblocked:** 22's multi-shard campaign (the store is
correct under shards now); 24's serving model (fiber-per-connection has
a database it can touch from any shard); the framework ledger rows the
arc gates stay ⏸ until their own slices.
**Dependencies unblocked:** 23 (has its fsync baseline to beat), 31
(has the mutex-inbox number), 32 (has the restart/replay timing
machinery), and every future optimization (the gate that catches
regressions is live).
**Next steps:** 22 (baselines single- AND multi-shard + the mutex-inbox
number; precursor recorded in story 8: remote insert ≈8µs/op RAM-only)
→ 31 (lifecycle) → 24 (chat) → 23 (io_uring group-commit) → 32 (WAL
checkpoint). Held tail resumes on its own precedence notes.
**Next steps:** 31 (lifecycle spec brainstorm) → 24 (chat) → 23
(io_uring group-commit — target: close the 4.5k→297k durable gap) →
32 (WAL checkpoint). Held tail resumes on its own precedence notes.
**`.dev/reference` used:** `linux` (io_uring uapi struct layouts and the
"single event loop" card — both load-bearing in the ring-params fix).
**`.dev/reference` used:** none this slice (the LW_SOAK discipline and
linkcheck.py precedent came from in-repo scripts).
---
@ -233,7 +230,7 @@ that sequences its tasks. Read one, approve, then the next starts.
| 9b | [`@table`, relations, query](language-runtime-database/done/09b-table-relations-query.md) | 🔄 query surface + relations + FK done (branch query-surface); group-by parked |
| 19 | [Float + Bytes](language-runtime-database/done/19-missing-scalar-types.md) | ✅ **landed 2026-08-20** — `.wob` v5: Float constant tag, field kinds 6/7, opcodes 34-41 (IEEE-quiet f64), builtins 70-83. Full stack: literals, arithmetic, `@table` column, WAL bit-exact replay, json fractions in / shortest-round-trip out, `?Float` reserved-NaN nil, total-order index (NaN last, `-0.0` == `+0.0`), Bytes + base64. No implicit Int/Float mixing (WO-E201); `float`/`trunc` are the only bridges. Proof: web-app price is a real Float (`{"price":9.99}`), `just web-app` 23/0; corpus 103/0 |
| 11 | [Fibers](language-runtime-database/done/11-fibers.md) | ✅ **landed 2026-08-21** with the arc (`just fibers` 10/0); fs-park re-scoped out of v1, disclosed in the story |
| 22 | [Durability, throughput, scale](language-runtime-database/in-progress/22-durability-throughput-scale.md) | 🔄 **spec + plan approved 2026-08-21, executing** — db-bench sample + campaign gates; forks settled |
| 22 | [Durability, throughput, scale](language-runtime-database/done/22-durability-throughput-scale.md) | ✅ **landed 2026-08-21** — db-bench + baseline.json (74 metrics) + restart/kill -9 proofs both shard counts; durable 4.5k vs ram 297k inserts/s, reads O(table), msgrate 13.4M/2.45M |
| 31 | [Actor lifecycle](language-runtime-database/refine/31-actor-lifecycle.md) | ⬜ needs a spec first — third in chain (story written 2026-08-21) |
| 24 | [chat: WebSocket workload](language-runtime-database/refine/24-chat-websocket-workload.md) | ⬜ fourth in chain — the arc's acceptance; after 31 |
| 23 | [io_uring group-commit](language-runtime-database/refine/23-io-uring-commit.md) | ⬜ fifth in chain, after stage 3 + 22 |
@ -257,8 +254,8 @@ that sequences its tasks. Read one, approve, then the next starts.
| Track | Item | Where |
| -------- | --------------------------------------------------------------------------- | ---------------------------------------------------------- |
| Language | 🔄 [iteration 36 — operator parity](language-runtime-database/in-progress/36-operator-parity.md): `not`, bitwise `& \| ^ << >>`, hex/binary/`_` literals, compound assigns — CODE LANDED 2026-08-22 (branch operator-parity, `.wob` v6, all gates green; reference project `.dev/reference/go` drove the design). Awaiting the developer's MANUAL pass on `docs/examples/operators/` (no test fixtures by directive); unblocks story 34's pure-`.wo` HMAC question | [plan](../superpowers/plans/2026-08-22-operator-parity.md) |
| Language | the framework v1-polish slice landed 2026-08-20 (branch framework-v1, awaiting merge); next per the order: brainstorm 20/21's forks | [order](#implementation-order-re-sequenced-2026-08-20--code-review-pass) |
| Runtime | **iteration 22: db-bench** — spec + plan approved 2026-08-21; executing | [marker](../in-progress/2026-08-21-db-bench.md) · [plan](../superpowers/plans/2026-08-21-db-bench.md) |
| Language | the framework v1-polish slice landed 2026-08-20 (branch framework-v1, awaiting merge); next per the order: brainstorm 20/21's forks | [order](#implementation-order-re-sequenced-2026-08-21--concurrency-chain) |
| Runtime | nothing active — 22 landed 2026-08-21; next per the chain: iteration 31's spec brainstorm | [order](#implementation-order-re-sequenced-2026-08-21--concurrency-chain) |
The active slice's marker doc lives in [`in-progress/`](../in-progress/) —
one file, deleted when the slice lands. Everything else pending is the
@ -446,8 +443,10 @@ precedence notes for resumption.
database is broken today. Plan of record:
[`2026-08-20-shard-fiber-arc.md`](../superpowers/plans/2026-08-20-shard-fiber-arc.md)
(stages 1+2 landed 2026-08-20, branch `concurrency-arc`).
2. 🔄 **22** — IN PROGRESS (spec + plan approved 2026-08-21) — the
measurement backbone: restart-persistence proof + baseline
2. ✅ **22** — LANDED 2026-08-21 (`just db-bench`, baseline committed,
gate bites; headline: durable 4.5k vs ram 297k inserts/s, reads
O(table), mixread 21 ops/s multi-shard, msgrate 2.45M cross-shard).
Was: the measurement backbone: restart-persistence proof + baseline
benchmark (durable + RAM-only), single- AND multi-shard in one
campaign, plus the stage-2 mutex-inbox number (rings only if the mutex
costs). It has never run — no `bench/baseline.json`, no `just db-bench`;

View file

@ -100,7 +100,7 @@ still pending IS the runtime-concurrency chain; order:
| 11 | 15 | [deps: `wo.toml [deps]`](done/15-deps-package-manager.md) | exact-rev git deps + `wo.lock` + `.wo-deps`; flat-only, offline once locked |
| 12 | 16 | [web framework](done/16-web-framework.md) | the `.wo` framework v1 (router, middleware, auth, all three body hooks) consumed via `[deps]` |
| 13 | 8+11 | [Shard-actor runtime](done/08-shard-actor-runtime.md) · [Fibers](done/11-fibers.md) | ✅ **THE ARC LANDED 2026-08-21** — stages 1+2 (fibers/budget/actors/io_uring plane; pinned shards, envelope sends, home-routed frees, WO-E222) + stage 3's transparent DB actor: worker statements marshal to shard 0, ack-after-owner-fsync, materialized replies (`just db-actor` 8/0, ASan/TSan, WAL replay pair). fs-park re-scoped out (disclosed in story 11). |
| 14 | 22 | [Durability, throughput, scale](in-progress/22-durability-throughput-scale.md) | 🔄 **spec + plan approved 2026-08-21, executing** — db-bench sample (`time.ticks` µs clock, seed/read/query/write/mix/msgrate/verify), campaign gates vs `bench/baseline.json`, restart + kill -9 proofs at both shard counts; the arc's delta and the mutex-inbox number come out of the first run. *(was 9e)* |
| 14 | 22 | [Durability, throughput, scale](done/22-durability-throughput-scale.md) | ✅ **LANDED 2026-08-21** — db-bench sample + `time.ticks` (builtin 84) + campaign driver + `bench/baseline.json` (74 metrics, tolerance-tuned); restart + 3× kill -9 proofs at both shard counts, gate bites. Measured: durable 4.5k vs ram 297k inserts/s (23's case); reads O(table) ≈1.5k/s (new candidate slice); mixread 1,280→21 ops/s single→multi (the arc's price); msgrate 13.4M/2.45M (deviation 4's number). *(was 9e)* |
| 15 | 30 | Observability, CI, fuzz *(no story file yet)* | **NEW** — runtime counters + a profiler hook, 22's harness wired to run per change instead of by hand, and a fuzz target on the parser and `.wob` loader. The whole proof-maturity gap had no iteration to point at. |
| 16 | 19 | [Float + Bytes](done/19-missing-scalar-types.md) | **LANDED 2026-08-20** — `.wob` v5; the full stack: IEEE-quiet f64 through literals/VM/@table/WAL/json + Bytes as the binary carrier, no implicit mixing, total-order indexes. Unblocks 24 (WS frames) and the crypto fork (digests). *(was 20)* |
| 17 | 31 | [Actor lifecycle](refine/31-actor-lifecycle.md) | request/response (today `send` is one-way and callers `sleep` to await), bounded mailboxes with backpressure (today the FIFO just grows), actor death/supervision, and timers beyond `time.sleep`. 24 cannot be written honestly without these. *(story written 2026-08-21)* |

View file

@ -97,7 +97,7 @@ status: done
- **Gated by the benchmark (2026-08-15):** this is the "implement garbage
collection" lever of the performance arc — tri-color mark-sweep replacing
RC changes the write path's tail latency, so landing it means re-running
iteration [22](../in-progress/22-durability-throughput-scale.md) and recording the
iteration [22](../done/22-durability-throughput-scale.md) and recording the
delta (does tracing help or hurt p99 under write load?).
- **Constraint added by the database track (2026-08-15):** a GC-managed value
in a `@table` field is a compile error (the engine/heap bulkhead — 9b

View file

@ -141,6 +141,19 @@ chain: 1
## Info
**The arc's measured delta** (iteration 22's first campaign,
2026-08-21, `bench/baseline.json` — N=20000, 4 mix actors, real-disk
WO_DATA; single-shard column = the local path, multi-shard = the
stage-3 RPC):
| metric | WO_SHARDS=1 | default cores | reading |
| --- | --- | --- | --- |
| seed inserts/s (ram) | 257,416 | 297,619 | local either way (primary seeds); parity ✓ |
| seed inserts/s (durable) | 4,492 | 4,478 | fsync-per-commit ≈220µs dominates — the 57× ram gap is iteration 23's case |
| read ops/s (ram, primary) | 1,630 | 1,488 | point lookups are O(table): the probe walks every slab — the read-path finding |
| mixread ops/s (actors) | 1,280 | **21** | the RPC price × O(table) probes × owner serialization — the arc's honest cost until reads index properly |
| msgrate msgs/s | 13,424,620 | 2,445,944 | same-heap vs mutex-inbox: 5.5× — deviation 4's number; rings stay unearned until this is the bottleneck |
**Guarantee contract** (stage-3 refinement 2026-08-21; moved here from
the slice's marker doc when it landed):
@ -159,7 +172,7 @@ the slice's marker doc when it landed):
- The VM's object header has carried a shard id since iteration 2 — no
relayout.
- **Gated by the benchmark:** landing the arc means re-running
[22](../in-progress/22-durability-throughput-scale.md) at the concurrency
[22](../done/22-durability-throughput-scale.md) at the concurrency
scale it unlocks and recording the before/after delta; it is also
where [23](../refine/23-io-uring-commit.md) gets a thread to overlap
durability against.

View file

@ -1,6 +1,6 @@
---
iteration: "22"
status: in-progress
status: done
chain: 2
---
@ -16,7 +16,26 @@ chain: 2
> performance work, because each of those must be gated by re-running THIS
> iteration's benchmark and showing the number moved the right way.
>
> ~~**No spec exists yet.** The forks in *Info* are genuine decisions.~~
> **✅ LANDED 2026-08-21** — the measurement backbone exists and has
> run: `docs/examples/db-bench` (seed/read/query/write/mix/msgrate/
> wal/verify, per-op `time.ticks` µs timing, 1µs-histogram percentiles),
> `scripts/db-bench.py` (campaign driver + gates), `bench/baseline.json`
> (74 metrics, the first contract — tolerances tuned by a two-run
> repeatability check: mix* 50%, read/query 35%, rest 15%), `just
> db-bench` / `db-bench-quick`. Proofs: restart-persistence + 3× kill -9
> battery at BOTH shard counts, all acked rows present every time; the
> gate BITES (doctored results fail on exactly the doctored metric).
> Headline findings: durable seed ≈4.5k/s vs ram ≈297k/s (iteration
> 23's case, measured); point lookups are O(table) — reads ≈1.5k/s at
> p50 ≈600µs on 20k rows (the probe walks every slab); mixread 1,280
> ops/s single- vs 21 ops/s multi-shard (the arc's honest price);
> msgrate 13.4M same-heap vs 2.45M cross-shard (deviation 4's
> mutex-inbox number). The arc's delta table lives in story 8.
> Deviations, disclosed in the plan: histogram not reservoir; `all` +
> `wal` modes added (RAM store dies with the process; clean acked-line
> crash vehicle); one python driver; `time.ticks` routed via an explicit
> dispatch arm. Standing finding: hand-built `multi <TableClass>` SEGVs
> on drop (elements classed OWNED, refs are scalar ids) — own slice.
>
> **SPEC APPROVED 2026-08-21** — the four forks below are SETTLED as
> their recorded leanings (developer confirmation), plus two new

View file

@ -1,6 +1,7 @@
# Iteration 22 — db-bench: durability proof, throughput, scale (implementation plan)
> **Status: ready to execute (2026-08-21).** Board:
> **Status: ✅ LANDED 2026-08-21** — all six tasks executed inline,
> deviations disclosed per task. Board:
> [docs/00-status.md](../../stories/00-status.md).
> **For agentic workers:** REQUIRED SUB-SKILL: Use
@ -29,7 +30,7 @@ stdlib-only (one stdlib-table row).
**Spec:** [`../specs/2026-08-21-db-bench-design.md`](../specs/2026-08-21-db-bench-design.md)
(approved 2026-08-21, normative — the four forks + decisions 5/6 live
there). Story:
[`22-durability-throughput-scale.md`](../../stories/language-runtime-database/in-progress/22-durability-throughput-scale.md).
[`22-durability-throughput-scale.md`](../../stories/language-runtime-database/done/22-durability-throughput-scale.md).
## Global Constraints
@ -67,16 +68,16 @@ there). Story:
the difference of two calls is a duration). Later tasks time every
operation with it.
- [ ] Corpus fixture first: two `time.ticks()` calls around a spin loop;
assert the difference is non-negative and the second call is >= the
first (exact values are machine noise — the fixture asserts ordering
and that the builtin exists). Verify it FAILS today (unknown builtin).
- [ ] Wire the id (84), the sysio case (clock_gettime MONOTONIC,
seconds*1e6 + nsec/1e3, as Int), the compiler table rows, the two
contract-doc rows (name, arity 0, return Int µs, monotonic-not-wall
wording).
- [ ] Verify: fixture green under `just oop-e2e`; full battery green.
Commit.
- [x] Corpus fixture first (`run/time-ticks`): RED as WO-E406, then
green — asserts monotonicity + nonzero, values stay machine noise.
- [x] Wired: wob.h id 84 + WO_B_MAX bump; the sysio case; ONE compiler
table row (types.ml is the single stdlib table — the plan's guess of
sibling tables in owner/emit was wrong, no mirror needed); the sys
dispatch range needed an explicit `C == WO_B_TIME_TICKS` arm (the
range gate stops at PROC_RUN). DEVIATION: `07-systems-stdlib.md`
does not exist (a planned doc never written) — the row went into
`08-builtin-surface.md` alone.
- [x] Verified: oop-e2e 104/0 (was 103); full battery green. Commit.
## Task 2 — db-bench sample: tables, serial modes, per-op stats
@ -183,15 +184,22 @@ there). Story:
(the arc's recorded delta: single- vs multi-shard columns),
`docs/examples/db-bench/README.md` (the reference-machine numbers).
- [ ] Full campaign on this machine; inspect the tail; commit the
baseline with a message that says it IS the first contract.
- [ ] Gate-bites smoke (spec acceptance): doctor a copy of the results
(halve one ops/sec) and run the gate against it — must FAIL; then the
real results — must PASS. Record the procedure in the README.
- [ ] Copy the arc delta + msgrate numbers into story 8's record and
the mutex-inbox note (arc plan deviation 4 references it).
- [ ] Verify: `just db-bench` exit 0 twice in a row (repeatability);
battery. Commit.
- [x] Full campaign run twice (20 checks/0 fail each incl. restart
proofs and 3x kill -9 batteries per shard count); baseline committed
as the first contract. DEVIATION: the two-run repeatability check
found ~25% jitter on read/query latencies and scheduling-dependent
spread on mix* — per-metric tolerances tuned (mix* 50%, read/query
35%, rest 15%, rationale in the baseline's _config note); both runs
pass the tuned contract 74/0.
- [x] Gate-bites proven: `--check` on a doctored copy (one ops/sec
halved) FAILS on exactly that metric; both real runs PASS. Procedure
in the README (a `--check` gate-only mode was added for this).
- [x] Arc delta + msgrate recorded: story 8 carries the measured table
(durable seed 4.5k/s vs ram 297k/s = 23's case; mixread 1280 vs 21
ops/s = the RPC x O(table)-probe price; msgrate 13.4M vs 2.45M =
deviation 4's mutex-inbox number). Headline findings in the sample
README: point lookups are O(table) — the read-path finding.
- [x] Verified: two full campaigns green; battery green. Commit.
## Task 6 — closeout
@ -203,10 +211,10 @@ there). Story:
`00-dependency-graph.md` node, the spec's status banner (APPROVED →
landed), `docs/in-progress/` marker deleted.
- [ ] Docs synced (the standup answers written from the actual numbers);
links verified (the session's link-check loop); frontmatter matches
folders.
- [ ] Full battery + `just db-bench-quick` once more after doc edits.
- [x] Docs synced (standup from the measured numbers; story/spec/plan
banners; graph node; marker deleted); links verified; frontmatter
matches folders.
- [x] Full battery + `just db-bench-quick` once more after doc edits.
Commit.
## Self-review notes

View file

@ -1,13 +1,14 @@
# Iteration 22 — durability proof, throughput, and scale under load (design)
**Date:** 2026-08-21
**Status:** ✅ APPROVED 2026-08-21 (developer review) — plan ready:
[`2026-08-21-db-bench.md`](../plans/2026-08-21-db-bench.md).
**Status:** ✅ LANDED 2026-08-21 — implemented in full (plan:
[`2026-08-21-db-bench.md`](../plans/2026-08-21-db-bench.md), all six
tasks, deviations disclosed there); `bench/baseline.json` is live.
Board: [docs/00-status.md](../../stories/00-status.md)
**Scope:** the measurement backbone — a benchmark workload in `.wo`, a
campaign driver script, a tracked baseline contract, and the durability
proofs (restart persistence + crash battery), single- AND multi-shard.
**Relates to:** [story 22](../../stories/language-runtime-database/in-progress/22-durability-throughput-scale.md)
**Relates to:** [story 22](../../stories/language-runtime-database/done/22-durability-throughput-scale.md)
(the four forks settled below), the landed arc
([story 8's guarantee contract](../../stories/language-runtime-database/done/08-shard-actor-runtime.md)
— the stage-3 delta this iteration records), iteration 23 (the durable

View file

@ -60,6 +60,15 @@ fibers:
db-actor:
./scripts/db-actor-accept.sh
# db-bench: iteration 22's campaign (docs/examples/db-bench) — OFF the
# fast path, minutes long: ram+durable x 1/N shards, durability legs,
# gates vs bench/baseline.json. quick = seconds, floors only.
db-bench:
./scripts/db-bench.py
db-bench-quick:
./scripts/db-bench.py --quick
# install-accept: extract the dist tarball to a temp prefix, PATH it, and prove
# `woc version` + a from-scratch project build+run (self-located wovm) + the
# wo-constraint refusal all work — the "tarball install actually works" gate.

View file

@ -170,7 +170,8 @@ int wo_builtin(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
both ranges (WO_B_MAP_GET_OPT and anything added after it) stay here */
if (C == WO_B_JSON_ENCODE || C == WO_B_JSON_DECODE)
return wo_builtin_json(vm, R, ins, msg);
if (C >= WO_B_SYS_FIRST && C <= WO_B_PROC_RUN) return wo_builtin_sys(vm, R, ins, msg);
if ((C >= WO_B_SYS_FIRST && C <= WO_B_PROC_RUN) || C == WO_B_TIME_TICKS)
return wo_builtin_sys(vm, R, ins, msg);
if (C >= WO_B_DB_INSERT && C <= WO_B_DB_PROBE) {
/* arc stage 3: the database is an actor on shard 0. A worker shard
* has no engine by design — its statement marshals, parks, resumes

View file

@ -284,6 +284,14 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
vm->cur->park_done = 1;
return WO_SYS_PARKED;
}
case WO_B_TIME_TICKS: { /* the bench clock: CLOCK_MONOTONIC µs as Int.
* Never wall time — only differences mean
* anything (iteration 22's honest percentiles) */
struct timespec mts;
clock_gettime(CLOCK_MONOTONIC, &mts);
R[A] = (uint64_t)((int64_t)mts.tv_sec * 1000000 + mts.tv_nsec / 1000);
return 0;
}
case WO_B_TIME_LOCAL: { /* Parts: 0 year, 1 month (1..12), 2 day, 3 hour,
* 4 minute, 5 second, 6 dow (0 = Sunday) */
time_t secs = (time_t)((int64_t)R[B] / 1000);

View file

@ -447,9 +447,13 @@ enum {
WO_B_TEXT_OF_BYTES = 83, /* (bytes) -> fresh Text, verbatim. The caller
* asserts the bytes are text; no validation,
* because Unicode is explicitly out of scope */
WO_B_TIME_TICKS = 84, /* () -> Int, CLOCK_MONOTONIC microseconds —
* the bench clock (iteration 22). Monotone,
* never wall time: immune to NTP steps; only
* differences mean anything. */
};
#define WO_B_MAX 83u
#define WO_B_MAX 84u
/* ids at or above this one live in sysio.c, not builtin.c */
#define WO_B_SYS_FIRST WO_B_FS_EXISTS

243
scripts/db-bench.py Executable file
View file

@ -0,0 +1,243 @@
#!/usr/bin/env python3
"""db-bench campaign driver (iteration 22).
Runs the docs/examples/db-bench sample across the campaign matrix
(ram/durable x 1/default shards), collects the sample's metric lines
into one results JSON, runs the durability legs (restart proof + kill -9
battery), samples RSS/fd during the mix phase (LW_SOAK discipline), and
evaluates every metric against bench/baseline.json.
scripts/db-bench.py [--quick] [--write-baseline]
Exit 0 = campaign green; 1 = gate breach or a durability leg failed.
Plan deviation, disclosed: one python driver instead of bash+python —
the live stdout sampling and JSON assembly are the whole job.
"""
import json, os, re, subprocess, sys, time, random, shutil
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
BIN = os.path.join(ROOT, "docs/examples/db-bench/target/db-bench")
WOC = os.path.join(ROOT, "compiler/_build/default/bin/woc")
WOVM = os.path.join(ROOT, "runtime/wovm")
BASELINE = os.path.join(ROOT, "bench/baseline.json")
RESULTS_DIR = os.path.join(ROOT, "bench/results")
QUICK = "--quick" in sys.argv
WRITE_BASELINE = "--write-baseline" in sys.argv
N = 2000 if QUICK else 20000
MSG_N = 20000 if QUICK else 200000
WAL_N = 800 if QUICK else 4000
CRASH_REPS = 1 if QUICK else 3
LINE = re.compile(r"^(\w+) (\d+) (\d+) (\d+) (\d+)$")
MSGLINE = re.compile(r"^msgrate (\d+) (\d+)$")
passed, failed = [], []
def ok(name): passed.append(name); print(f"ok {name}")
def bad(name, why): failed.append(name); print(f"FAIL {name} -- {why}")
def build():
os.makedirs(os.path.dirname(BIN), exist_ok=True)
r = subprocess.run([WOC, "build", os.path.join(ROOT, "docs/examples/db-bench"),
"-o", BIN, "--runtime", WOVM], capture_output=True, text=True)
if r.returncode != 0:
bad("build", r.stderr.strip()[:200]); sys.exit(1)
ok("builds")
def run(args, env_extra, timeout, sample_after=None):
"""Run the sample; return (rc, lines, rss_growth_kb, fd_growth).
sample_after: stdout prefix that starts the RSS/fd baseline (the mix
phase begins after the 'write' report line lands)."""
env = dict(os.environ); env.update(env_extra)
p = subprocess.Popen([BIN] + args, stdout=subprocess.PIPE, stderr=subprocess.STDOUT,
text=True, env=env)
lines, base_rss, base_fd, peak_rss, peak_fd = [], None, None, None, None
def rss_fd():
try:
with open(f"/proc/{p.pid}/status") as f:
rss = next((int(l.split()[1]) for l in f if l.startswith("VmRSS:")), None)
fd = len(os.listdir(f"/proc/{p.pid}/fd"))
return rss, fd
except OSError:
return None, None
deadline = time.time() + timeout
import threading
def reader():
assert p.stdout is not None
for line in p.stdout:
lines.append(line.rstrip("\n"))
t = threading.Thread(target=reader); t.start()
while p.poll() is None and time.time() < deadline:
if sample_after is not None and any(l.startswith(sample_after) for l in lines):
r, f = rss_fd()
if r is not None:
if base_rss is None: base_rss, base_fd = r, f
peak_rss = r if peak_rss is None else max(peak_rss, r)
if f is not None:
peak_fd = f if peak_fd is None else max(peak_fd, f)
time.sleep(0.25)
if p.poll() is None:
p.kill(); t.join(); return 124, lines, None, None
t.join()
growth = (peak_rss - base_rss) if base_rss is not None and peak_rss is not None else None
fdg = (peak_fd - base_fd) if base_fd is not None and peak_fd is not None else None
return p.returncode, lines, growth, fdg
def parse_metrics(lines, into, prefix):
for l in lines:
m = LINE.match(l)
if m:
op, _cnt, rate, p50, p99 = m.group(1), *(int(x) for x in m.groups()[1:])
into[f"{prefix}.{op}.ops_sec"] = rate
into[f"{prefix}.{op}.p50us"] = p50
into[f"{prefix}.{op}.p99us"] = p99
m = MSGLINE.match(l)
if m:
into[f"{prefix}.msgrate.msgs_sec"] = int(m.group(2))
def campaign():
metrics = {}
ncores = os.cpu_count() or 1
for flavor in ("ram", "durable"):
for shards in (1, ncores):
tag = f"{flavor}.s{'1' if shards == 1 else 'N'}"
env = {"WO_SHARDS": str(shards)}
data = None
if flavor == "durable":
data = os.path.join(ROOT, "bench", f"tmp.{os.getpid()}.{tag}")
os.makedirs(data, exist_ok=True)
env["WO_DATA"] = data
rc, lines, rssg, fdg = run(["all", str(N)], env, 1800, sample_after="write ")
if rc != 0:
bad(f"{tag}.all", f"rc={rc} tail={lines[-2:]}")
else:
parse_metrics(lines, metrics, tag)
ok(f"{tag}.all")
if rssg is not None:
metrics[f"{tag}.mix.rss_growth_kb"] = rssg
metrics[f"{tag}.mix.fd_growth"] = fdg
# LW_SOAK discipline: mix inserts ~N/100 rows (slab
# growth is legitimate); the budget catches MB-class
# leaks, fd growth is zero-tolerance
if fdg and fdg > 0:
bad(f"{tag}.mix.fds", f"grew {fdg}")
else:
ok(f"{tag}.mix.fds flat")
if data: shutil.rmtree(data, ignore_errors=True)
# msgrate once per shard count, RAM only (no store dependency)
for shards in (1, ncores):
tag = f"msg.s{'1' if shards == 1 else 'N'}"
rc, lines, _, _ = run(["msgrate", str(MSG_N)], {"WO_SHARDS": str(shards)}, 300)
if rc != 0:
bad(tag, f"rc={rc}")
else:
parse_metrics(lines, metrics, tag.replace("msg.", "ram."))
ok(tag)
return metrics
def durability(metrics):
ncores = os.cpu_count() or 1
for shards in (1, ncores):
s = "s1" if shards == 1 else "sN"
env = {"WO_SHARDS": str(shards)}
# restart proof
data = os.path.join(ROOT, "bench", f"tmp.{os.getpid()}.restart.{s}")
os.makedirs(data, exist_ok=True)
env["WO_DATA"] = data
rc1, _, _, _ = run(["seed", "3000"], env, 300)
rc2, lines, _, _ = run(["verify"], env, 300)
if rc1 == 0 and rc2 == 0:
ok(f"restart.{s}: seeded store replays byte-true")
else:
bad(f"restart.{s}", f"seed rc={rc1} verify rc={rc2} {lines[-1:]}")
shutil.rmtree(data, ignore_errors=True)
# crash battery: kill -9 mid-wal, verify every acked row
for rep in range(CRASH_REPS):
data = os.path.join(ROOT, "bench", f"tmp.{os.getpid()}.crash.{s}.{rep}")
os.makedirs(data, exist_ok=True)
env["WO_DATA"] = data
e = dict(os.environ); e.update(env)
p = subprocess.Popen([BIN, "wal", str(WAL_N)], stdout=subprocess.PIPE,
stderr=subprocess.DEVNULL, text=True, env=e)
time.sleep(random.uniform(0.05, 0.5))
p.kill() # SIGKILL: no unwind, no flush — the honest crash
out, _ = p.communicate()
acked = 0
for l in out.splitlines():
if l.startswith("acked "):
acked = int(l.split()[1])
rc, lines, _, _ = run(["verify-acked", str(max(acked, 1))], env, 300)
if acked > 0 and rc == 0:
ok(f"crash.{s}.{rep}: {acked} acked rows all present after kill -9")
elif acked == 0:
ok(f"crash.{s}.{rep}: killed before first ack (nothing owed)")
else:
bad(f"crash.{s}.{rep}", f"acked={acked} verify rc={rc} {lines[-1:]}")
shutil.rmtree(data, ignore_errors=True)
def gate(metrics):
if not os.path.exists(BASELINE):
if WRITE_BASELINE:
write_baseline(metrics); return
bad("gate", "no bench/baseline.json (run --write-baseline once)"); return
base = json.load(open(BASELINE))
for key, spec in sorted(base.items()):
if key.startswith("_"): continue
got = metrics.get(key)
if got is None:
bad(f"gate.{key}", "metric missing from this run"); continue
val, tol, floor = spec["value"], spec.get("tolerance_pct", 15), spec.get("floor")
higher_is_better = spec.get("dir", "higher") == "higher"
if QUICK:
# quick mode: floors only — counts are too small for stable deltas
breach = floor is not None and ((got < floor) if higher_is_better else (got > floor))
(ok if not breach else lambda n: bad(n, f"{got} vs floor {floor}"))(f"gate.{key} (floor)")
continue
if higher_is_better:
rel_bad = got < val * (100 - tol) / 100
floor_bad = floor is not None and got < floor
else:
rel_bad = got > val * (100 + tol) / 100
floor_bad = floor is not None and got > floor
if rel_bad or floor_bad:
bad(f"gate.{key}", f"{got} vs baseline {val} (tol {tol}%, floor {floor})")
else:
ok(f"gate.{key} {got} (baseline {val})")
if WRITE_BASELINE:
write_baseline(metrics)
def write_baseline(metrics):
base = {"_config": {"N": N, "msg_n": MSG_N, "wal_n": WAL_N, "crash_reps": CRASH_REPS,
"note": "refresh only with a commit that says why"}}
for k, v in sorted(metrics.items()):
if k.endswith(("rss_growth_kb", "fd_growth")): continue
higher = k.endswith(("ops_sec", "msgs_sec"))
base[k] = {"value": v, "tolerance_pct": 15,
"floor": (v // 4 if higher else v * 4), "dir": "higher" if higher else "lower"}
os.makedirs(os.path.dirname(BASELINE), exist_ok=True)
json.dump(base, open(BASELINE, "w"), indent=1, sort_keys=True)
ok(f"baseline written ({len(base) - 1} metrics)")
def main():
# --check <results.json>: gate-only evaluation of a recorded run — the
# gate-bites smoke doctors a copy and this mode must FAIL on it
if "--check" in sys.argv:
f = sys.argv[sys.argv.index("--check") + 1]
gate(json.load(open(f)))
print()
print(f"db-bench --check: {len(passed) + len(failed)} checks, {len(failed)} failures")
sys.exit(1 if failed else 0)
build()
metrics = campaign()
durability(metrics)
os.makedirs(RESULTS_DIR, exist_ok=True)
stamp = time.strftime("%Y%m%d-%H%M%S")
out = os.path.join(RESULTS_DIR, f"run-{stamp}{'-quick' if QUICK else ''}.json")
json.dump(metrics, open(out, "w"), indent=1, sort_keys=True)
gate(metrics)
print()
print(f"db-bench: {len(passed) + len(failed)} checks, {len(failed)} failures "
f"(results: {os.path.relpath(out, ROOT)})")
sys.exit(1 if failed else 0)
main()

View file

@ -0,0 +1,2 @@
ticks: monotone
ticks: nonzero

View file

@ -0,0 +1,28 @@
use time
-- time.ticks: CLOCK_MONOTONIC microseconds as Int. Values are machine
-- noise; the fixture asserts existence and ordering — a later reading
-- never precedes an earlier one, and a spin's elapsed is non-negative.
fn spin(rounds: Int) {
let i = 0;
while i < rounds {
i = i + 1;
}
}
fn main() -> Int {
let t0 = time.ticks();
spin(100000);
let t1 = time.ticks();
if t1 >= t0 {
print("ticks: monotone");
} else {
print("ticks: BACKWARDS ${t1 - t0}");
}
if t0 > 0 {
print("ticks: nonzero");
} else {
print("ticks: zero start");
}
return 0;
}