Merge master into db-residency-doctrine — and close the two half-exposed features

The branch was 17 ahead / 25 behind with 11 conflicting files, and drifting
further: db.c had been rewritten twice on master since (group commit, then
compaction). Resolved rather than rebased so both histories stay legible.

Conflicts, and how each was settled:

- db.c: BOTH semantics kept. Master's fatal path and compaction check now sit
  behind the branch's `table_is_durable` predicate, in all three inline arms —
  a volatile table reaches neither the barrier nor the compaction check
- db-bench sample: every mode from both sides (growth, growth-verify, randread,
  replayseed, wmix) and ONE `boot` mode, which both sides had added
  independently
- db-bench.py: all six legs kept. Both sides had also grown the same
  WAL-size helper under different names; collapsed into one
- perf-targets: the branch's §5 (RAM ceiling) then master's §6/§7 — master's
  numbering had already assumed a §5 it did not have
- story frontmatter: master's `status` (the landing truth) plus the branch's
  `readiness` axis. 03 would have read `done` + `refine`, which is a
  contradiction — it was brainstormed and landed on master, so `ready`
- board: both standup blocks newest-first; master's chain rows (a superset);
  the branch's databasev2 1-2 rows with master's 3-4. Fixed a stray `|` in
  master's row 3
- baseline: master's, then REGENERATED from a full campaign — 143 metrics,
  132 checks, 0 failures with both sides' legs present

TWO HALF-EXPOSED FEATURES FIXED, because the merge rule is that master gets
no feature that is honoured in name only:

- `resident: keys` PARSED, set a .wob flag, and did nothing: rows stayed fully
  resident. A developer could declare a 120 GB table keys-resident, watch it
  compile, and be OOM-killed. The loader now REFUSES it with a message naming
  what to write instead, until tasks 5c/5d land. The compiler still parses it
  and its AST golden still passes, so the grammar work stays tested
- `durable: false` was honoured ONLY on the inline path. wo_db_exec_req had no
  guard at all, so a volatile table written from an actor on a worker shard
  would still be logged — precisely porch's session-table case, and precisely
  what iteration 2 exists to provide. All three request-path arms now carry the
  same predicate. Found by reading the merged code, not by a test: the obvious
  probe runs main() on the primary and therefore only exercises the inline path

Verified on the merged tree: wovm-test 0, woc-test 0, oop-e2e 122/0,
residency-accept 8/0, db-bench 132/0, linkcheck clean.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
shoney.arickathil 2026-08-29 10:14:25 +02:00
commit 02b4b13a52
56 changed files with 5200 additions and 352 deletions

View file

@ -1,22 +1,70 @@
{
"_config": {
"N": 2000,
"crash_reps": 1,
"msg_n": 20000,
"N": 20000,
"crash_reps": 3,
"msg_n": 200000,
"note": "refresh only with a commit that says why; tolerances come from tolerance_for() in the driver",
"wal_n": 800
"wal_n": 4000
},
"ceiling.rows_recovered": {
"dir": "lower",
"floor": 159744,
"floor": 159492,
"tolerance_pct": 100,
"value": 39936
"value": 39873
},
"ckpt.boot_off_ms": {
"dir": "lower",
"floor": 456,
"tolerance_pct": 400,
"value": 114
},
"ckpt.boot_on_ms": {
"dir": "lower",
"floor": 256,
"tolerance_pct": 400,
"value": 64
},
"ckpt.bytes_off": {
"dir": "lower",
"floor": 7876676,
"tolerance_pct": 400,
"value": 1969169
},
"ckpt.bytes_on": {
"dir": "lower",
"floor": 3696192,
"tolerance_pct": 400,
"value": 924048
},
"ckpt.compactions": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 400,
"value": 6
},
"ckpt.pause_us_max": {
"dir": "lower",
"floor": 33912,
"tolerance_pct": 400,
"value": 8478
},
"ckpt.pause_us_per_mb": {
"dir": "lower",
"floor": 65848,
"tolerance_pct": 100,
"value": 16462
},
"ckpt.reclaim_x": {
"dir": "higher",
"floor": 0.0,
"tolerance_pct": 15,
"value": 2.13
},
"durable.s1.mixread.ops_sec": {
"dir": "higher",
"floor": 2237,
"floor": 2452,
"tolerance_pct": 50,
"value": 8949
"value": 9809
},
"durable.s1.mixread.p50us": {
"dir": "lower",
@ -28,31 +76,31 @@
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 2
"value": 12
},
"durable.s1.mixwrite.ops_sec": {
"dir": "higher",
"floor": 248,
"floor": 272,
"tolerance_pct": 50,
"value": 994
"value": 1089
},
"durable.s1.mixwrite.p50us": {
"dir": "lower",
"floor": 820,
"floor": 1704,
"tolerance_pct": 50,
"value": 205
"value": 426
},
"durable.s1.mixwrite.p99us": {
"dir": "lower",
"floor": 872,
"floor": 1984,
"tolerance_pct": 50,
"value": 218
"value": 496
},
"durable.s1.query.ops_sec": {
"dir": "higher",
"floor": 335570,
"floor": 306372,
"tolerance_pct": 50,
"value": 1342281
"value": 1225490
},
"durable.s1.query.p50us": {
"dir": "lower",
@ -68,9 +116,9 @@
},
"durable.s1.read.ops_sec": {
"dir": "higher",
"floor": 347705,
"floor": 307389,
"tolerance_pct": 50,
"value": 1390820
"value": 1229558
},
"durable.s1.read.p50us": {
"dir": "lower",
@ -82,85 +130,121 @@
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 2
"value": 1
},
"durable.s1.seed.ops_sec": {
"dir": "higher",
"floor": 1103,
"floor": 1095,
"tolerance_pct": 15,
"value": 4415
"value": 4381
},
"durable.s1.seed.p50us": {
"dir": "lower",
"floor": 828,
"floor": 848,
"tolerance_pct": 15,
"value": 207
"value": 212
},
"durable.s1.seed.p99us": {
"dir": "lower",
"floor": 2092,
"floor": 2432,
"tolerance_pct": 15,
"value": 523
"value": 608
},
"durable.s1.wmix.mean_batch": {
"dir": "higher",
"floor": 0.0,
"tolerance_pct": 100,
"value": 1.0
},
"durable.s1.wmix.ops_sec": {
"dir": "higher",
"floor": 402,
"tolerance_pct": 15,
"value": 1611
},
"durable.s1.wmix.p50us": {
"dir": "lower",
"floor": 1764,
"tolerance_pct": 15,
"value": 441
},
"durable.s1.wmix.p99us": {
"dir": "lower",
"floor": 2684,
"tolerance_pct": 15,
"value": 671
},
"durable.s1.wmix.peak_batch": {
"dir": "higher",
"floor": 0,
"tolerance_pct": 100,
"value": 1
},
"durable.s1.wmix.peak_staged": {
"dir": "lower",
"floor": 196,
"tolerance_pct": 100,
"value": 49
},
"durable.s1.write.ops_sec": {
"dir": "higher",
"floor": 1155,
"floor": 573,
"tolerance_pct": 15,
"value": 4620
"value": 2294
},
"durable.s1.write.p50us": {
"dir": "lower",
"floor": 832,
"floor": 1760,
"tolerance_pct": 15,
"value": 208
"value": 440
},
"durable.s1.write.p99us": {
"dir": "lower",
"floor": 1948,
"floor": 2716,
"tolerance_pct": 15,
"value": 487
"value": 679
},
"durable.sN.mixread.ops_sec": {
"dir": "higher",
"floor": 1112,
"floor": 1183,
"tolerance_pct": 50,
"value": 4450
"value": 4733
},
"durable.sN.mixread.p50us": {
"dir": "lower",
"floor": 236,
"floor": 244,
"tolerance_pct": 50,
"value": 59
"value": 61
},
"durable.sN.mixread.p99us": {
"dir": "lower",
"floor": 13100,
"tolerance_pct": 50,
"value": 3275
"floor": 16200,
"tolerance_pct": 300,
"value": 4050
},
"durable.sN.mixwrite.ops_sec": {
"dir": "higher",
"floor": 123,
"floor": 131,
"tolerance_pct": 50,
"value": 494
"value": 525
},
"durable.sN.mixwrite.p50us": {
"dir": "lower",
"floor": 1160,
"floor": 2172,
"tolerance_pct": 50,
"value": 290
"value": 543
},
"durable.sN.mixwrite.p99us": {
"dir": "lower",
"floor": 2944,
"tolerance_pct": 50,
"value": 736
"floor": 16440,
"tolerance_pct": 300,
"value": 4110
},
"durable.sN.query.ops_sec": {
"dir": "higher",
"floor": 287356,
"floor": 308451,
"tolerance_pct": 50,
"value": 1149425
"value": 1233806
},
"durable.sN.query.p50us": {
"dir": "lower",
@ -171,14 +255,14 @@
"durable.sN.query.p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"tolerance_pct": 300,
"value": 1
},
"durable.sN.read.ops_sec": {
"dir": "higher",
"floor": 192752,
"floor": 248188,
"tolerance_pct": 50,
"value": 771010
"value": 992752
},
"durable.sN.read.p50us": {
"dir": "lower",
@ -189,44 +273,80 @@
"durable.sN.read.p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"tolerance_pct": 300,
"value": 2
},
"durable.sN.seed.ops_sec": {
"dir": "higher",
"floor": 1142,
"floor": 1104,
"tolerance_pct": 50,
"value": 4571
"value": 4418
},
"durable.sN.seed.p50us": {
"dir": "lower",
"floor": 836,
"floor": 844,
"tolerance_pct": 50,
"value": 209
"value": 211
},
"durable.sN.seed.p99us": {
"dir": "lower",
"floor": 1916,
"floor": 2188,
"tolerance_pct": 300,
"value": 547
},
"durable.sN.wmix.mean_batch": {
"dir": "higher",
"floor": 1.0,
"tolerance_pct": 100,
"value": 6.22
},
"durable.sN.wmix.ops_sec": {
"dir": "higher",
"floor": 1504,
"tolerance_pct": 50,
"value": 479
"value": 6017
},
"durable.sN.wmix.p50us": {
"dir": "lower",
"floor": 27184,
"tolerance_pct": 50,
"value": 6796
},
"durable.sN.wmix.p99us": {
"dir": "lower",
"floor": 37484,
"tolerance_pct": 300,
"value": 9371
},
"durable.sN.wmix.peak_batch": {
"dir": "higher",
"floor": 15,
"tolerance_pct": 100,
"value": 60
},
"durable.sN.wmix.peak_staged": {
"dir": "lower",
"floor": 11760,
"tolerance_pct": 100,
"value": 2940
},
"durable.sN.write.ops_sec": {
"dir": "higher",
"floor": 1010,
"floor": 580,
"tolerance_pct": 50,
"value": 4040
"value": 2320
},
"durable.sN.write.p50us": {
"dir": "lower",
"floor": 832,
"floor": 1760,
"tolerance_pct": 50,
"value": 208
"value": 440
},
"durable.sN.write.p99us": {
"dir": "lower",
"floor": 1996,
"tolerance_pct": 50,
"value": 499
"floor": 2688,
"tolerance_pct": 300,
"value": 672
},
"growth.available": {
"dir": "lower",
@ -236,9 +356,9 @@
},
"growth.int.noswap.bytes_per_row": {
"dir": "lower",
"floor": 392,
"floor": 440,
"tolerance_pct": 10,
"value": 98
"value": 110
},
"growth.int.noswap.doublings": {
"dir": "lower",
@ -266,21 +386,21 @@
},
"growth.int.noswap.rows": {
"dir": "lower",
"floor": 80000,
"floor": 800000,
"tolerance_pct": 100,
"value": 20000
"value": 200000
},
"growth.int.noswap.rss_kb": {
"dir": "lower",
"floor": 23968,
"floor": 168528,
"tolerance_pct": 100,
"value": 5992
"value": 42132
},
"growth.int.swap.bytes_per_row": {
"dir": "lower",
"floor": 392,
"floor": 440,
"tolerance_pct": 10,
"value": 98
"value": 110
},
"growth.int.swap.doublings": {
"dir": "lower",
@ -308,21 +428,21 @@
},
"growth.int.swap.rows": {
"dir": "lower",
"floor": 80000,
"floor": 800000,
"tolerance_pct": 100,
"value": 20000
"value": 200000
},
"growth.int.swap.rss_kb": {
"dir": "lower",
"floor": 23984,
"floor": 168576,
"tolerance_pct": 100,
"value": 5996
"value": 42144
},
"growth.text.noswap.bytes_per_row": {
"dir": "lower",
"floor": 1288,
"floor": 1284,
"tolerance_pct": 10,
"value": 322
"value": 321
},
"growth.text.noswap.doublings": {
"dir": "lower",
@ -350,21 +470,21 @@
},
"growth.text.noswap.rows": {
"dir": "lower",
"floor": 80000,
"floor": 800000,
"tolerance_pct": 100,
"value": 20000
"value": 200000
},
"growth.text.noswap.rss_kb": {
"dir": "lower",
"floor": 41216,
"floor": 343312,
"tolerance_pct": 100,
"value": 10304
"value": 85828
},
"growth.text.swap.bytes_per_row": {
"dir": "lower",
"floor": 1288,
"floor": 1284,
"tolerance_pct": 10,
"value": 322
"value": 321
},
"growth.text.swap.doublings": {
"dir": "lower",
@ -392,21 +512,21 @@
},
"growth.text.swap.rows": {
"dir": "lower",
"floor": 80000,
"floor": 800000,
"tolerance_pct": 100,
"value": 20000
"value": 200000
},
"growth.text.swap.rss_kb": {
"dir": "lower",
"floor": 41216,
"floor": 343328,
"tolerance_pct": 100,
"value": 10304
"value": 85832
},
"ram.s1.mixread.ops_sec": {
"dir": "higher",
"floor": 2236,
"floor": 22286,
"tolerance_pct": 50,
"value": 8947
"value": 89144
},
"ram.s1.mixread.p50us": {
"dir": "lower",
@ -422,9 +542,9 @@
},
"ram.s1.mixwrite.ops_sec": {
"dir": "higher",
"floor": 248,
"floor": 2476,
"tolerance_pct": 50,
"value": 994
"value": 9904
},
"ram.s1.mixwrite.p50us": {
"dir": "lower",
@ -440,15 +560,15 @@
},
"ram.s1.msgrate.msgs_sec": {
"dir": "higher",
"floor": 419322,
"tolerance_pct": 15,
"value": 3354579
"floor": 1336469,
"tolerance_pct": 70,
"value": 10691756
},
"ram.s1.query.ops_sec": {
"dir": "higher",
"floor": 324675,
"floor": 244857,
"tolerance_pct": 50,
"value": 1298701
"value": 979431
},
"ram.s1.query.p50us": {
"dir": "lower",
@ -464,9 +584,9 @@
},
"ram.s1.read.ops_sec": {
"dir": "higher",
"floor": 332889,
"floor": 252270,
"tolerance_pct": 50,
"value": 1331557
"value": 1009081
},
"ram.s1.read.p50us": {
"dir": "lower",
@ -482,87 +602,87 @@
},
"ram.s1.seed.ops_sec": {
"dir": "higher",
"floor": 375939,
"floor": 62904,
"tolerance_pct": 15,
"value": 1503759
"value": 251616
},
"ram.s1.seed.p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 15,
"value": 0
"value": 4
},
"ram.s1.seed.p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 15,
"value": 2
"value": 9
},
"ram.s1.write.ops_sec": {
"dir": "higher",
"floor": 272628,
"floor": 47770,
"tolerance_pct": 15,
"value": 1090512
"value": 191080
},
"ram.s1.write.p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 15,
"value": 1
"value": 8
},
"ram.s1.write.p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 15,
"value": 2
"value": 10
},
"ram.sN.mixread.ops_sec": {
"dir": "higher",
"floor": 2241,
"floor": 11218,
"tolerance_pct": 50,
"value": 8964
"value": 44874
},
"ram.sN.mixread.p50us": {
"dir": "lower",
"floor": 228,
"floor": 240,
"tolerance_pct": 50,
"value": 57
"value": 60
},
"ram.sN.mixread.p99us": {
"dir": "lower",
"floor": 1412,
"floor": 324,
"tolerance_pct": 50,
"value": 353
"value": 81
},
"ram.sN.mixwrite.ops_sec": {
"dir": "higher",
"floor": 249,
"floor": 1246,
"tolerance_pct": 50,
"value": 996
"value": 4986
},
"ram.sN.mixwrite.p50us": {
"dir": "lower",
"floor": 252,
"floor": 260,
"tolerance_pct": 50,
"value": 63
"value": 65
},
"ram.sN.mixwrite.p99us": {
"dir": "lower",
"floor": 280,
"floor": 356,
"tolerance_pct": 50,
"value": 70
"value": 89
},
"ram.sN.msgrate.msgs_sec": {
"dir": "higher",
"floor": 214795,
"tolerance_pct": 50,
"value": 1718360
"floor": 317323,
"tolerance_pct": 70,
"value": 2538586
},
"ram.sN.query.ops_sec": {
"dir": "higher",
"floor": 331125,
"floor": 291545,
"tolerance_pct": 50,
"value": 1324503
"value": 1166180
},
"ram.sN.query.p50us": {
"dir": "lower",
@ -578,9 +698,9 @@
},
"ram.sN.read.ops_sec": {
"dir": "higher",
"floor": 340599,
"floor": 317823,
"tolerance_pct": 50,
"value": 1362397
"value": 1271294
},
"ram.sN.read.p50us": {
"dir": "lower",
@ -596,87 +716,87 @@
},
"ram.sN.seed.ops_sec": {
"dir": "higher",
"floor": 353606,
"floor": 73305,
"tolerance_pct": 50,
"value": 1414427
"value": 293220
},
"ram.sN.seed.p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 1
"value": 3
},
"ram.sN.seed.p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 2
"value": 7
},
"ram.sN.write.ops_sec": {
"dir": "higher",
"floor": 290697,
"floor": 56810,
"tolerance_pct": 50,
"value": 1162790
"value": 227241
},
"ram.sN.write.p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 1
"value": 6
},
"ram.sN.write.p99us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 50,
"value": 2
"value": 10
},
"randread.collapse_x": {
"dir": "lower",
"floor": 1172,
"floor": 1084,
"tolerance_pct": 100,
"value": 293
"value": 271
},
"randread.overcap.filled_rss_kb": {
"dir": "lower",
"floor": 25680,
"floor": 58144,
"tolerance_pct": 100,
"value": 6420
"value": 14536
},
"randread.overcap.ops_sec": {
"dir": "higher",
"floor": 1665,
"floor": 1427,
"tolerance_pct": 100,
"value": 6661
"value": 5711
},
"randread.overcap.read_p50us": {
"dir": "lower",
"floor": 556,
"floor": 624,
"tolerance_pct": 100,
"value": 139
"value": 156
},
"randread.overcap.read_p99us": {
"dir": "lower",
"floor": 1920,
"floor": 1628,
"tolerance_pct": 100,
"value": 480
"value": 407
},
"randread.resident.filled_rss_kb": {
"dir": "lower",
"floor": 54080,
"floor": 168288,
"tolerance_pct": 100,
"value": 13520
"value": 42072
},
"randread.resident.ops_sec": {
"dir": "higher",
"floor": 488424,
"floor": 387281,
"tolerance_pct": 100,
"value": 1953697
"value": 1549126
},
"randread.resident.read_p50us": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 0
"value": 1
},
"randread.resident.read_p99us": {
"dir": "lower",
@ -686,57 +806,57 @@
},
"replay.history.ms": {
"dir": "lower",
"floor": 844,
"floor": 12876,
"tolerance_pct": 100,
"value": 211
"value": 3219
},
"replay.history.ns_per_record": {
"dir": "lower",
"floor": 21068,
"floor": 64384,
"tolerance_pct": 100,
"value": 5267
"value": 16096
},
"replay.history.records": {
"dir": "lower",
"floor": 160000,
"floor": 800000,
"tolerance_pct": 100,
"value": 40000
"value": 200000
},
"replay.history.wal_bytes": {
"dir": "lower",
"floor": 7840140,
"floor": 39200140,
"tolerance_pct": 100,
"value": 1960035
"value": 9800035
},
"replay.history_penalty_x": {
"dir": "lower",
"floor": 100,
"tolerance_pct": 100,
"value": 1.9
"value": 1.6
},
"replay.inserts.ms": {
"dir": "lower",
"floor": 444,
"floor": 8064,
"tolerance_pct": 100,
"value": 111
"value": 2016
},
"replay.inserts.ns_per_record": {
"dir": "lower",
"floor": 22120,
"floor": 80656,
"tolerance_pct": 100,
"value": 5530
"value": 20164
},
"replay.inserts.records": {
"dir": "lower",
"floor": 80000,
"floor": 400000,
"tolerance_pct": 100,
"value": 20000
"value": 100000
},
"replay.inserts.wal_bytes": {
"dir": "lower",
"floor": 3920140,
"floor": 19600140,
"tolerance_pct": 100,
"value": 980035
"value": 4900035
},
"replay.startup_ms": {
"dir": "lower",

View file

@ -295,6 +295,7 @@ let b_sha1 = 85
let b_sha256 = 86
let b_hmac_sha256 = 87
let b_call = 88
let b_monitor = 89
let b_split = 28
let b_split_ws = 29
let b_join = 30
@ -1110,7 +1111,7 @@ let is_builtin_name (n : string) =
"substr"; "trim"; "to_lower"; "char_of"; "parse_int"; "split"; "split_ws"; "join"; "slice";
"pop"; "shift"; "sort"; "reverse"; "remove"; "key_at"; "val_at";
(* the concurrency arc *)
"send"; "call";
"send"; "call"; "monitor";
(* iteration 19: Float bridges and Bytes surface *)
"float"; "trunc"; "parse_float"; "float_to_text"; "float_cmp"; "bytes_len"; "bytes_at";
"bytes_slice"; "bytes_eq"; "bytes_concat"; "base64_encode"; "base64_decode";
@ -3441,11 +3442,18 @@ and emit_call (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e : As
put f (ins_abc op_builtin dst base sm.Types.sm_builtin);
(* every stdlib member only READS its arguments, so one that was
freshly built here (`net.write(c, head .. resp.body)`) has no
other owner and dies with the call *)
other owner and dies with the call. The ONE exception:
`time.after`'s message (arg 2) MOVES to the runtime — the
timer owns it until delivery (iteration 24 T5). *)
let moves i =
alias = "time" && mname = "after" && i = 2
in
List.iteri
(fun i (a : Ast.expr) ->
drop_fresh_owned ~keep:dst p f (base + i) a;
drop_fresh_text ~keep:dst p f (base + i) a)
if not (moves i) then begin
drop_fresh_owned ~keep:dst p f (base + i) a;
drop_fresh_text ~keep:dst p f (base + i) a
end)
args
end)
| Some u -> (
@ -3722,7 +3730,7 @@ and emit_builtin (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e :
dangle the value just read) and the stores, which either copy (Text,
handled by copied_container_call) or take ownership (OWNED/GCREF). *)
let reader = List.mem name [ "get"; "latest"; "key_at"; "val_at" ] in
(if not (List.mem name [ "push"; "set"; "send"; "call" ]) then
(if not (List.mem name [ "push"; "set"; "send"; "call"; "monitor" ]) then
List.iteri
(fun i (a : Ast.expr) ->
(* a reader's result points into arg0 (the container) — dropping
@ -3754,6 +3762,7 @@ and emit_builtin (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e :
match name with
| "send" -> fixed b_send (* arc: msg (arg1) moved to the runtime — never dropped here *)
| "call" -> fixed b_call (* iteration 24: same move; the SCALAR reply lands in dst *)
| "monitor" -> fixed b_monitor (* T4: notice msg (arg2) moves to the runtime *)
| "now" -> fixed b_now
| "print" -> fixed b_print
| "print_int" -> fixed b_print_int

View file

@ -1348,6 +1348,10 @@ and analyze_call (ctx : ctx) (call_e : Ast.expr) (callee : Ast.expr) (args : Ast
iteration 24: call(addr, msg) moves its message identically. *)
| Ident "send" -> i = 1 && Types.StringMap.find_opt "send" ctx.syms.Types.free_fns = None
| Ident "call" -> i = 1 && Types.StringMap.find_opt "call" ctx.syms.Types.free_fns = None
(* T4/T5: the notice / timer message moves to the runtime too *)
| Ident "monitor" ->
i = 2 && Types.StringMap.find_opt "monitor" ctx.syms.Types.free_fns = None
| Field ({ kind = Ident "time"; _ }, "after") -> i = 2
| _ -> false
in
List.iteri
@ -1365,7 +1369,8 @@ and analyze_call (ctx : ctx) (call_e : Ast.expr) (callee : Ast.expr) (args : Ast
transfer ctx p
~what:
(match callee.kind with
| Ident "send" | Ident "call" ->
| Ident "send" | Ident "call" | Ident "monitor"
| Field ({ kind = Ident "time"; _ }, "after") ->
"cannot be sent — a message moves to the receiver"
| _ -> "cannot be stored in a container")
then record_move ctx p (MvArg "element"))

View file

@ -309,6 +309,8 @@ let stdlib_members : stdlib_member list =
m "net" "write_dl" 3 93 (Some (TScalar "Bool")) None;
m "net" "listen_unix" 1 94 (Some (TScalar "Int")) None;
m "net" "peer" 1 95 (Some (TScalar "Text")) None;
(* iteration 24 T5: one-shot timer — the msg MOVES to the runtime *)
m "time" "after" 3 90 None None;
(* proc *)
m "proc" "run" 2 56 (Some (TNullable (TScalar proc_record_name))) (Some proc_record_name);
(* json — both members are lowered specially (emit.ml): encode needs its
@ -1917,6 +1919,48 @@ let typecheck_program ~file ~(module_of : string -> string)
~message:"`call`'s first argument must be an `actor M` address" ())
| None -> ())
| _ -> ())
| None when name = "monitor" ->
(* iteration 24 T4: monitor(watched, observer, msg) — the
notice msg is typed against the OBSERVER's mailbox
(three-argument form: the caller may be main, which has
no mailbox). msg moves like send's. *)
(if List.length args <> 3 then
Diag.Collector.add collector
(Diag.error ~code:bad_arity_code ~file ~line:e.pos.line ~col:e.pos.col
~message:
(Printf.sprintf
"`monitor` takes 3 arguments (watched, observer, notice), given %d"
(List.length args))
())
else
match args with
| [ w; o; m ] -> (
(match confident_typ cenv w with
| Some (TActor _) | None -> ()
| Some _ ->
Diag.Collector.add collector
(Diag.error ~code:type_mismatch_code ~file ~line:w.pos.line
~col:w.pos.col
~message:"`monitor`'s first argument must be an `actor M` address" ()));
match confident_typ cenv o with
| Some (TActor want) -> (
match confident_typ cenv m with
| Some (TScalar got) when got <> want ->
Diag.Collector.add collector
(Diag.error ~code:type_mismatch_code ~file ~line:m.pos.line
~col:m.pos.col
~message:
(Printf.sprintf
"the observer receives `%s` — the notice is a `%s`" want got)
())
| _ -> ())
| Some _ ->
Diag.Collector.add collector
(Diag.error ~code:type_mismatch_code ~file ~line:o.pos.line
~col:o.pos.col
~message:"`monitor`'s second argument must be an `actor M` address" ())
| None -> ())
| _ -> ())
| None ->
let confident_types = List.map (confident_typ cenv) args in
check_builtin_call ~file collector name e.pos args confident_types)

View file

@ -119,3 +119,135 @@ rather than acknowledging what disk never got.
columns excluded (engine raw-eq is narrower than VM float-eq, and a
probe miss cannot be resurrected by a recheck). Pinned by
`tests/corpus/run/query-index-probe`.
## Group commit: one barrier per drain (databasev2 4 part A, 2026-08-28)
**What changed:** the engine used to commit per *statement*. `db.c` called
`wo_wal_commit` immediately after every append, at all six sites, so each row
change bought its own `pwrite` and its own `fdatasync`. Now the barrier belongs
to the drain, not to the statement.
**Where the barrier runs, and why there.** A statement on a worker shard has no
WAL to write — the runtime asserts workers hold neither `db` nor `wal` — so it
marshals to shard 0 and parks. Shard 0 executes those requests in its envelope
drain (`wo_vm_adopt`), and the drain now **holds each reply** instead of pushing
it as the statement finishes. When the queue empties it issues one barrier, then
releases every held reply.
Holding the reply is the whole mechanism. Pushing it early would unpark the
requester before its record was durable; holding it means each writer is
acknowledged after the barrier that carried *its own* record. That was always
the intended contract — it was simply true by accident before, because every
batch had exactly one member.
**Why the queue is the boundary.** Not a tick, and not a timer. A queue of one
gives a batch of one, so a lone writer pays exactly what it paid before; the
batch grows only when writes genuinely contend. A tick boundary would have
added latency even with nothing to batch against, which is taxing an idle
system to serve a busy one. There is nothing to tune, which is the point.
**Why the inline path is asymmetric.** A statement already on shard 0 stages and
commits before returning, batch size one. It cannot hold a reply because there
is nobody to reply to — it returns into its own fiber. Batching it would mean
parking that fiber on the barrier, which is part B's machinery. Two consequences
worth keeping in mind: single-shard configurations get no batching at all, by
design; and the inline commit is only safe because the drain commits
*unconditionally* whenever anything is staged, so the buffer is empty when an
inline statement runs. If that ever stops holding, the inline path would make
another statement's record durable early and acknowledge it to the wrong writer.
**One rule for failure: once a statement has mutated RAM, the outcomes are
durable or process death.** It replaced three behaviours that disagreed —
`insert` un-applied itself, while `update` and `delete` returned a catchable
trap and left RAM ahead of disk, which their own comments said out loud.
Batching would have multiplied that from one row to a whole batch. So a failed
stage or a failed barrier now prints one diagnostic (operation, log path,
`errno`, record count) and exits 3; `WO_T_IO` is unreachable from a write.
Retrying is not offered because it is unsound: on Linux a failed `fsync` may
already have discarded the dirty pages, so a second call can report success
having written nothing. Replay is the recovery that works.
**Measuring it.** `WO_WAL_STATS=1` makes the runtime print one line at exit —
batches, records, peak batch, peak staged bytes. Opt-in, because it would
otherwise pollute every durable program's output. The counters live in `wo_wal`
rather than behind a builtin: they are diagnostic, not part of the language.
`db-bench`'s `wmix N C` leg exists to exercise this at all — `mix` writes on one
op in ten with C=4, which produced a measured mean batch of 1.01, so it could
never have shown whether batching worked.
**If you are looking at this because writes got slower**, check the mean batch
first. Mean 1.0 means the mechanism is not engaging, which is expected for a
serial writer or a single-shard configuration and a bug anywhere else.
## Checkpoint: compaction by rewrite + rename (databasev2 3, 2026-08-29)
**The problem:** nothing ever removed superseded records, so the log grew
forever and boot replayed all history. Measured before this: 20 000 rows seeded
gave a 986 KB log; updating those same rows 20 000 times took it to 2.6 MB with
**the same live data**.
**Why one file and not a snapshot plus a tail.** Postgres does the opposite —
its WAL is a redo tail and the data lives in heap files, so a checkpoint flushes
pages and then recycles log segments; it never compacts. It cannot: its records
are page deltas, so a compacted redo log is not a store. **Ours are full row
images** — `apply_record` implements UPDATE as remove-then-recreate — so a log
of one record per live row *is* a complete store. That single difference deletes
the control file, the redo pointer, the second recovery source and the separate
process from this design. Recovery is not merely compatible with compaction; it
is completely unaware of it.
**Why `rename` is the whole crash-safety story.** The dump goes to a temp file,
which is fsynced, renamed over the live log, and then the parent directory is
fsynced (the rename is atomic in-kernel, but the directory entry is not durable
until the parent is — Postgres does the same for the same reason). Before the
rename the live log is intact and the temp is not authoritative; after it the new
log is complete. There is no instant at which a reader sees a mixture, so this
needs no recovery logic of its own. What Postgres achieves with a redo pointer
computed at checkpoint start and a control file written at the end, one syscall
achieves here — because we can swap the entire data set atomically and Postgres
cannot.
A crash mid-rewrite leaves a temp file. The next open **removes it**, and it is
deleted rather than ignored because a file full of well-formed records sitting
beside the log is exactly what a later reader mistakes for data.
**Why the dump flushes periodically, and why it does NOT fsync when it does.**
`stage()` grows the staging buffer by doubling and never shrinks it, so pushing a
whole store through one buffer would hold the entire store in RAM on top of the
store — the unbounded growth databasev2 1 measured as how this engine dies. So
the dump flushes every 256 records. It flushes with a plain write, **not** a
commit: intermediate durability is worthless because the temp is not
authoritative until the rename and is fsynced once immediately before it. Using
the committing path cost one barrier per 256 records and made the pause 8×
larger — measured 107 649 µs against 13 212 µs for a 2 MB live set, ~22 MB/s
against ~181 MB/s.
**Why the replacement is preallocated like the original.** The WAL is
preallocated so that appends never extend the file, which is what lets
`fdatasync` alone serve as the ack barrier. A replacement opened without it
would silently change that property, and the zero-padded tail the open-time scan
relies on.
**When it runs.** Only where the staging buffer is empty — right after a
barrier. Both write paths check: the drain (`vm.c`, after its commit and after
releasing held replies, since those records are already durable and should not
wait out a rewrite) and the inline path (`db.c`). Wiring only the drain left
`WO_SHARDS=1` never compacting, with its log growing forever: measured 536 KB
where the multi-shard run held 446 KB.
**The trigger** compares the log against what the *last* compaction actually
wrote, with an absolute floor. The denominator is measured rather than
estimated, because estimating the live size means estimating Text and the
compactor already knows the true number. There is deliberately **no timer**:
Postgres needs one because its dirty buffers are not durable until flushed, and
ours are durable at commit — an idle log does not grow.
**A failed compaction is a missed optimisation, not a durability event.** It
leaves the original log intact and returns an error the callers ignore. It must
never take `wo_wal_commit_fatal`'s path, which exists for a different problem.
**If you are here because a checkpoint misbehaved:** `WO_WAL_STATS=1` reports
compaction count, the stop-the-world pause (max and total) and the last
compaction's size. `WO_CHECKPOINT_BYTES` and `WO_CHECKPOINT_RATIO` move the
policy; setting a tiny floor forces compaction in a few writes, which is how the
gate tests it at all.

View file

@ -18,6 +18,23 @@ static int table_is_durable(const wo_db *db, uint32_t cid) {
return (db->classes[cid].flags & WO_CLASSF_VOLATILE) == 0u;
}
/* databasev2 3: the inline path's compaction check.
*
* The drain has its own (vm.c, after the barrier). This one exists because a
* statement running ON the owner shard never enters that drain, so without it
* a single-shard durable program's log grows FOREVER — measured: WO_SHARDS=1
* reached 536 KB where the multi-shard run held 446 KB, because the check was
* only wired into the drain.
*
* Safe here for the same reason it is safe there: the commit above just
* emptied the staging buffer. The result is ignored because a failed
* compaction is a missed optimisation, not a durability event. */
static void maybe_compact(wo_db *db, wo_wal *w) {
if (wo_wal_should_compact(w->off, w->compacted_bytes, wo_wal_ckpt_floor,
wo_wal_ckpt_ratio))
(void)wo_wal_compact(w, db);
}
int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
uint32_t A = wo_ins_a(ins), B = wo_ins_b(ins), C = wo_ins_c(ins);
wo_db *db = (wo_db *)vm->rt.db;
@ -36,15 +53,26 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
: WO_T_DB;
wo_wal *w = (wo_wal *)vm->rt.wal;
if (w && table_is_durable(db, cid)) {
/* RAM applied, record staged, ONE commit before the ack (the
* builtin's return). A failed commit is a failed write: the
* row is removed again so RAM never claims what disk never
* acknowledged, and the statement traps. */
if (wo_wal_append_insert(w, db, cid, id) != 0 || wo_wal_commit(w) != 0) {
wo_row_remove(db, cid, id);
*msg = "wal commit failed";
return WO_T_IO;
}
/* THE INLINE PATH KEEPS ITS OWN BARRIER, AND THAT ASYMMETRY IS
* DELIBERATE (databasev2 4 part A). The request path batches:
* wo_vm_adopt holds each reply and commits once per drain. This
* path cannot, because it has no reply to hold — it returns into
* its OWN fiber rather than unparking a requester. Do not "fix"
* this by dropping the commit: without it an inline statement
* would never be durable at all.
*
* Committing here is safe because the drain commits
* unconditionally whenever anything is staged, so the buffer is
* empty when this runs.
*
* The `table_is_durable` guard is databasev2 2's: a
* `@table(durable: false)` class is never staged, so it reaches
* neither this barrier nor the compaction check below.
*
* Failure is fatal, not a trap: the row is already in RAM. */
if (wo_wal_append_insert(w, db, cid, id) != 0) wo_wal_stage_fatal(w);
wo_wal_commit_fatal(w, 1);
maybe_compact(db, w);
}
R[A] = id;
return 0;
@ -58,10 +86,11 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
return ek == DB_ERR_UNIQUE ? WO_T_UNIQUE : ek == DB_ERR_OOM ? WO_T_OOM : WO_T_DB;
wo_wal *w = (wo_wal *)vm->rt.wal;
if (w && table_is_durable(db, cid)) {
if (wo_wal_append_update(w, db, cid, id) != 0 || wo_wal_commit(w) != 0) {
*msg = "wal commit failed"; /* RAM ahead of disk: trap, do not ack */
return WO_T_IO;
}
/* was: trap and leave RAM ahead of disk, which the old comment
* admitted. Now fatal — see the insert arm. */
if (wo_wal_append_update(w, db, cid, id) != 0) wo_wal_stage_fatal(w);
wo_wal_commit_fatal(w, 1);
maybe_compact(db, w);
}
R[A] = 0;
return 0;
@ -81,10 +110,9 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
}
wo_wal *w = (wo_wal *)vm->rt.wal;
if (w && table_is_durable(db, cid)) {
if (wo_wal_append_remove(w, cid, id) != 0 || wo_wal_commit(w) != 0) {
*msg = "wal commit failed";
return WO_T_IO;
}
if (wo_wal_append_remove(w, cid, id) != 0) wo_wal_stage_fatal(w);
wo_wal_commit_fatal(w, 1);
maybe_compact(db, w);
}
R[A] = 0;
return 0;
@ -226,13 +254,12 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) {
q->msg = m;
break;
}
if (w) {
if (wo_wal_append_insert(w, db, q->cid, id) != 0 || wo_wal_commit(w) != 0) {
wo_row_remove(db, q->cid, id);
q->status = WO_T_IO;
q->msg = "wal commit failed";
break;
}
if (w && table_is_durable(db, q->cid)) {
/* databasev2 4: staging failure is FATAL, not a trap. The row is
* already in RAM; of the three verbs only insert could undo
* itself, so continuing means RAM ahead of disk. One rule: once a
* statement has mutated RAM, the outcomes are durable or death. */
if (wo_wal_append_insert(w, db, q->cid, id) != 0) wo_wal_stage_fatal(w);
}
q->result = id;
break;
@ -244,12 +271,8 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) {
q->msg = m;
break;
}
if (w) {
if (wo_wal_append_update(w, db, q->cid, q->id) != 0 || wo_wal_commit(w) != 0) {
q->status = WO_T_IO;
q->msg = "wal commit failed";
break;
}
if (w && table_is_durable(db, q->cid)) {
if (wo_wal_append_update(w, db, q->cid, q->id) != 0) wo_wal_stage_fatal(w);
}
break;
}
@ -264,12 +287,8 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) {
q->msg = "no such row";
break;
}
if (w) {
if (wo_wal_append_remove(w, q->cid, q->id) != 0 || wo_wal_commit(w) != 0) {
q->status = WO_T_IO;
q->msg = "wal commit failed";
break;
}
if (w && table_is_durable(db, q->cid)) {
if (wo_wal_append_remove(w, q->cid, q->id) != 0) wo_wal_stage_fatal(w);
}
break;
}

View file

@ -4,7 +4,9 @@
#include "wal.h"
#include <errno.h>
#include <time.h>
#include <fcntl.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <unistd.h>
@ -302,6 +304,19 @@ int wo_wal_open(wo_wal *w, const char *path, uint64_t prealloc) {
memset(w, 0, sizeof(*w));
w->fd = open(path, O_RDWR | O_CREAT, 0644);
if (w->fd < 0) return -1;
w->path = strdup(path); /* NULL is tolerated: the diagnostic degrades */
/* databasev2 3: remove a stale compaction temp before doing anything else.
* The only way one exists is a crash before the rename, which means its
* records were never authoritative — the live log below is the truth. It is
* deleted rather than ignored because a file full of well-formed records
* sitting beside the log is exactly the thing a future reader mistakes for
* data. */
{
char tmp[4096];
if ((size_t)snprintf(tmp, sizeof tmp, "%s%s", path, WO_WAL_TMP_SUFFIX) < sizeof tmp)
(void)unlink(tmp);
}
w->prealloc = prealloc;
if (prealloc) {
/* best-effort: a filesystem without fallocate still works */
(void)posix_fallocate(w->fd, 0, (off_t)prealloc);
@ -317,6 +332,7 @@ int wo_wal_open(wo_wal *w, const char *path, uint64_t prealloc) {
void wo_wal_close(wo_wal *w) {
if (w->fd >= 0) close(w->fd);
free(w->path);
free(w->buf);
memset(w, 0, sizeof(*w));
w->fd = -1;
@ -390,7 +406,103 @@ int wo_wal_append_remove(wo_wal *w, uint32_t class_id, uint64_t id) {
}
int wo_wal_commit(wo_wal *w) {
if (!w->len) return 0;
if (!w->len) return 0; /* empty commits are not batches; do not count them */
if (w->len > w->stat_peak_staged) w->stat_peak_staged = w->len;
size_t at = 0;
while (at < w->len) {
ssize_t n = pwrite(w->fd, w->buf + at, w->len - at, (off_t)(w->off + at));
if (n < 0) {
if (errno == EINTR) continue;
return WO_WAL_ERR_WRITE;
}
at += (size_t)n;
}
if (fdatasync(w->fd) != 0) return WO_WAL_ERR_SYNC;
w->off += w->len;
w->len = 0; /* acked: the batch is durable */
return 0;
}
/* Nothing at either fatal point is recoverable: RAM holds changes the log
* does not, and this process can no longer serve reads that would survive a
* restart. Name what failed precisely enough to act on, then stop. */
static void wal_die(const wo_wal *w, const char *op, uint32_t nrec) {
fprintf(stderr,
"writeonce: DURABILITY FAILURE — %s failed on %s: %s\n"
" %u record(s) were NOT made durable and are not acknowledged.\n"
" The process is stopping: replay restores the last durable state.\n",
op, w->path ? w->path : "(the write-ahead log)", strerror(errno),
nrec);
exit(WO_EXIT_DURABILITY);
}
void wo_wal_stage_fatal(const wo_wal *w) { wal_die(w, "staging a record", 1); }
void wo_wal_commit_fatal(wo_wal *w, uint32_t nrec) {
int staged = w->len != 0;
int rc = wo_wal_commit(w);
if (rc == 0) {
if (staged) { /* count the barrier that actually happened */
w->stat_batches++;
w->stat_records += nrec;
if (nrec > w->stat_peak_batch) w->stat_peak_batch = nrec;
}
return;
}
wal_die(w, rc == WO_WAL_ERR_SYNC ? "fdatasync" : "pwrite", nrec);
}
uint64_t wo_wal_ckpt_floor = 4u << 20; /* 4 MiB: below this there is nothing worth reclaiming */
uint32_t wo_wal_ckpt_ratio = 3u; /* 3x the live-set's own size is enough history */
int wo_wal_should_compact(uint64_t used, uint64_t last, uint64_t floor, uint32_t ratio) {
if (used < floor) return 0; /* a small log has nothing to reclaim */
if (last == 0) return 1; /* past the floor and never compacted: do it once
* to establish the denominator */
if (ratio == 0) return 0; /* a zero ratio disables the policy rather than
* dividing by nothing */
return used > last * (uint64_t)ratio;
}
/* databasev2 3: how many records the dump stages before flushing.
*
* NOT unbounded: stage() grows the staging buffer by doubling and never
* shrinks it, so appending a whole store through one buffer would hold the
* entire store in RAM on top of the store itself — the unbounded growth
* databasev2 1 identified as how this engine dies. 256 records is a few tens
* of KiB per flush, which is large enough that the syscall cost is amortised
* and small enough that the buffer never matters. */
#define WO_WAL_COMPACT_FLUSH 256u
/* rename(2)'s atomicity is in-kernel: the new directory ENTRY is not durable
* until the parent directory is synced. Postgres does the same thing for the
* same reason. Best-effort — a filesystem that refuses to sync a directory
* still leaves a correct log, just one whose swap might not survive a power
* cut. */
static void sync_parent_dir(const char *path) {
char dir[4096];
size_t n = strlen(path);
if (n >= sizeof dir) return;
memcpy(dir, path, n + 1);
char *slash = strrchr(dir, '/');
if (slash == dir) dir[1] = '\0';
else if (slash) *slash = '\0';
else memcpy(dir, ".", 2);
int fd = open(dir, O_RDONLY);
if (fd < 0) return;
(void)fsync(fd);
close(fd);
}
/* databasev2 3: write the staged bytes WITHOUT a durability barrier.
*
* Only compaction's dump uses this. Intermediate durability there is worthless:
* the temp file is not authoritative until the rename, and it is fsynced once
* immediately before that. Using wo_wal_commit for the dump instead cost one
* fdatasync per 256 records — measured, that was most of the stop-the-world
* pause (~22 MB/s, where the fixed cost plus ~150 redundant syncs dominated a
* 2 MB dump). */
static int wal_write_nosync(wo_wal *w) {
size_t at = 0;
while (at < w->len) {
ssize_t n = pwrite(w->fd, w->buf + at, w->len - at, (off_t)(w->off + at));
@ -400,12 +512,110 @@ int wo_wal_commit(wo_wal *w) {
}
at += (size_t)n;
}
if (fdatasync(w->fd) != 0) return -1;
w->off += w->len;
w->len = 0; /* acked: the batch is durable */
w->len = 0;
return 0;
}
static uint64_t mono_us(void) {
struct timespec ts;
if (clock_gettime(CLOCK_MONOTONIC, &ts) != 0) return 0;
return (uint64_t)ts.tv_sec * 1000000ull + (uint64_t)ts.tv_nsec / 1000ull;
}
/* ============================================================================
* OBLIGATION FOR WHOEVER IMPLEMENTS `resident: keys` (databasev2 2, tasks
* 5c/5d) — READ THIS BEFORE STORING WAL OFFSETS.
*
* Compaction rewrites the log and MOVES EVERY RECORD. Any WAL byte offset
* captured from the old file is meaningless afterwards — not stale-but-
* readable, but pointing at an arbitrary byte of a different file.
*
* `resident: keys` stores exactly such an offset per row and reads rows back
* through it. So the loop below, which knows each record's NEW position as it
* writes it, MUST also rebuild that map. It is the cheap direction and the only
* one that keeps both features usable together; the alternative is forbidding
* compaction whenever such a table is live, which would mean the feature for
* huge tables is incompatible with the feature that stops their log growing.
*
* Nothing fails today because that storage half does not exist yet. It will
* fail later, and it will look like data corruption rather than a design gap.
* ==========================================================================*/
int wo_wal_compact(wo_wal *w, wo_db *db) {
/* staged records would be written into a file about to be replaced */
if (!w->path || w->len != 0) return -1;
uint64_t t0 = mono_us();
char tmp[4096];
if ((size_t)snprintf(tmp, sizeof tmp, "%s%s", w->path, WO_WAL_TMP_SUFFIX) >= sizeof tmp)
return -1;
(void)unlink(tmp); /* a stale one would otherwise be appended to */
wo_wal nw;
/* THE REPLACEMENT MUST BE PREALLOCATED LIKE THE ORIGINAL. The WAL is
* preallocated so appends never extend the file, which is precisely what
* makes fdatasync sufficient as the ack barrier — no file-size metadata
* has to reach disk for an acked record to be readable. Opening the
* replacement with prealloc 0 silently removed that property, and the
* crash battery caught it: records acked shortly before a kill went
* missing, with the log otherwise intact and self-consistent. */
if (wo_wal_open(&nw, tmp, w->prealloc) != 0) return -1;
/* one INSERT per live row, in the existing grammar, through the existing
* append path — so replay needs no second decoder and ids are preserved
* exactly (wo_wal_append_insert takes the id and reads the row) */
uint32_t pending = 0;
for (uint32_t cid = 0; cid < db->class_cnt; cid++) {
db_table *t = &db->tables[cid];
if (!t->slabs) continue; /* tables are created lazily */
uint32_t total = t->slab_cnt * DB_SLAB_ROWS;
for (uint32_t g = 0; g < total; g++) {
if (!(t->bitmap[g >> 6] & (1ull << (g & 63)))) continue;
db_row *r = (db_row *)(t->slabs[g / DB_SLAB_ROWS] +
(size_t)(g % DB_SLAB_ROWS) * t->row_size);
if (wo_wal_append_insert(&nw, db, cid, r->id) != 0) goto fail;
if (++pending >= WO_WAL_COMPACT_FLUSH) {
if (wal_write_nosync(&nw) != 0) goto fail;
pending = 0;
}
}
}
if (wal_write_nosync(&nw) != 0) goto fail; /* the tail batch */
/* THE dump's one and only barrier: everything above is just bytes in the
* page cache until this, and nothing reads the temp before the rename. */
if (fsync(nw.fd) != 0) goto fail;
uint64_t new_bytes = nw.off;
wo_wal_close(&nw);
/* THE SWITCH. Every crash point either side of this is safe. */
if (rename(tmp, w->path) != 0) {
(void)unlink(tmp);
return -1;
}
sync_parent_dir(w->path);
/* the old descriptor now refers to an unlinked inode */
if (w->fd >= 0) close(w->fd);
w->fd = open(w->path, O_RDWR);
if (w->fd < 0) return -1; /* the log is correct on disk; this process cannot go on */
w->off = new_bytes;
w->len = 0;
w->compacted_bytes = new_bytes;
{ /* the stop-the-world pause: nothing was served while this ran */
uint64_t el = mono_us() - t0;
w->stat_compactions++;
w->stat_compact_us_total += el;
if (el > w->stat_compact_us_max) w->stat_compact_us_max = el;
}
return 0;
fail:
wo_wal_close(&nw);
(void)unlink(tmp);
return -1; /* the live log is untouched and still usable */
}
static int apply_record(wo_db *db, const uint8_t *payload, uint32_t len) {
rbuf r = {payload, payload + len, 0};
uint8_t kind = rd_u8(&r);

View file

@ -47,10 +47,38 @@ enum { WO_WAL_INSERT = 1, WO_WAL_REMOVE = 2, WO_WAL_UPDATE = 3 };
typedef struct wo_wal {
int fd;
/* databasev2 4: where this WAL lives, so a durability failure can name
* the file it could not write. An abort diagnostic without the path
* sends an operator hunting. Owned here, freed by wo_wal_close. */
char *path;
uint64_t off; /* next write offset (the intact tail) */
/* staged batch: appended by wal_append_*, flushed by wal_commit */
uint8_t *buf;
size_t len, cap;
/* databasev2 4: group-commit diagnostics. Batching is worthless if
* batches are always one, and a throughput change would then have come
* from somewhere else — so the mechanism is measured, not assumed.
* peak_staged also settles whether the batch needs a cap with a number
* instead of a guess. Reported at exit under WO_WAL_STATS. */
uint64_t stat_batches; /* non-empty commits */
uint64_t stat_records; /* records those commits carried */
uint64_t stat_peak_batch; /* most records in one barrier */
uint64_t stat_peak_staged; /* most bytes staged behind one barrier */
/* databasev2 3: bytes the last compaction wrote. The trigger compares the
* log against THIS rather than an estimate of the live set — estimating
* would mean estimating Text, and the compactor knows the true number. */
uint64_t compacted_bytes;
/* databasev2 3: the preallocation this log was opened with. Compaction
* MUST give the replacement the same one: the WAL is preallocated so that
* appends never extend the file, which is what lets fdatasync alone be the
* ack barrier. A replacement without it silently weakens durability. */
uint64_t prealloc;
/* databasev2 3: what compaction actually did, reported under WO_WAL_STATS.
* The PAUSE is the number the spec refused to assume — compaction is
* stop-the-world, so its duration is the cost being weighed. */
uint64_t stat_compactions;
uint64_t stat_compact_us_max;
uint64_t stat_compact_us_total;
} wo_wal;
/* databasev2 2: the file offset the NEXT staged record will occupy.
@ -90,10 +118,94 @@ int wo_wal_append_remove(wo_wal *w, uint32_t class_id, uint64_t id);
* later optimization, recorded). Call AFTER the RAM update. */
int wo_wal_append_update(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id);
/* databasev2 4: which half of the barrier failed. A pwrite failure and an
* fdatasync failure are different operational problems (a short write vs a
* device refusing the flush), so the diagnostic must name the right one. */
#define WO_WAL_ERR_WRITE (-1)
#define WO_WAL_ERR_SYNC (-2)
/* The process exit status for a durability failure.
*
* 74 is sysexits' EX_IOERR, chosen deliberately over a small number: 1 is a
* trap and 2 is a loader refusal, but 3 and 4 are already used by SAMPLES for
* their own meanings — db-bench's own `verify` exits 3 on a checksum mismatch,
* and it is the gate that exercises durability, so a durability abort exiting 3
* would have been indistinguishable from the mismatch it is supposed to help
* diagnose. The low range belongs to programs; the runtime takes a high one. */
#define WO_EXIT_DURABILITY 74
/* Write the staged batch and fdatasync — the ack line. Empty batch = ok,
* no syscall. 0 ok, -1 write/sync failure (the batch stays staged). */
* no syscall. 0 ok, WO_WAL_ERR_WRITE / WO_WAL_ERR_SYNC on failure (the
* batch stays staged: a failed commit consumes nothing). */
int wo_wal_commit(wo_wal *w);
/* databasev2 3: the checkpoint trigger, as a PURE decision so it can be tested
* without a store — which is the only way a policy like this gets tested at all.
*
* [used] the log's used bytes; [last] what the LAST compaction wrote (0 if it
* has never run); [floor] the size below which compacting is not worth it;
* [ratio] the multiple of [last] that counts as too much history.
*
* The denominator is the last compaction's MEASURED output rather than an
* estimate of the live set: estimating would mean estimating Text, and the
* compactor already knows the true number.
*
* There is deliberately NO TIME component. Postgres' CheckPointTimeout exists
* to bound data loss from unflushed buffers; our records are durable at commit,
* so a checkpoint only reclaims space and shortens boot. An idle log does not
* grow, so a timer would fire with nothing to do.
*
* 1 = compact now, 0 = leave it. */
int wo_wal_should_compact(uint64_t used, uint64_t last, uint64_t floor, uint32_t ratio);
/* Defaults, overridable at boot by WO_CHECKPOINT_BYTES / WO_CHECKPOINT_RATIO.
* The knobs are what make the policy testable: a test sets a tiny floor and
* forces compaction in a few writes instead of waiting for megabytes. */
extern uint64_t wo_wal_ckpt_floor;
extern uint32_t wo_wal_ckpt_ratio;
/* databasev2 3: the temporary file compaction writes before the swap. Named
* next to the log so it lands on the same filesystem — rename(2) is only
* atomic within one. Boot removes a stale one (a crash before the rename). */
#define WO_WAL_TMP_SUFFIX ".compact"
/* databasev2 3: rewrite the log as one INSERT record per LIVE row, then swap
* it in with rename(2).
*
* Recovery is deliberately untouched: the result is an ordinary log in the
* ordinary grammar, replayed from byte 0. Crash safety comes from rename being
* atomic — before it the live log is intact and the temp file is not
* authoritative; after it the new log is complete. There is no window in which
* a reader sees a mixture, so this needs no recovery logic of its own.
*
* REFUSES if anything is staged (returns -1 without touching the log): those
* records would be written into a file about to be replaced. Callers must
* invoke this only where the staging buffer is empty — right after a barrier.
*
* A failure is a MISSED OPTIMISATION, not a durability event: the original log
* is left usable and the process keeps running. It must not take the fatal
* path wo_wal_commit_fatal takes.
*
* 0 ok, -1 on any failure. */
int wo_wal_compact(wo_wal *w, wo_db *db);
/* databasev2 4: a record could not even be STAGED (the row is already in
* RAM, so this is the same unrecoverable position as a failed barrier — see
* wo_wal_commit_fatal). Never returns. */
void wo_wal_stage_fatal(const wo_wal *w);
/* databasev2 4: commit, or END THE PROCESS.
*
* The one rule this iteration introduces: once a statement has mutated RAM,
* the only outcomes are durable or process death. Retrying is not an
* alternative — on Linux a failed fsync may already have discarded the dirty
* pages, so a second call can report success having written nothing. The
* recovery that works is replay, which returns the last durable state.
*
* [nrec] is the number of records in the batch, for the diagnostic only.
* Returns on success; never returns on failure. */
void wo_wal_commit_fatal(wo_wal *w, uint32_t nrec);
/* Boot replay: apply every intact record to [db] in order. Ids re-enter
* exactly as logged; each table's next_id advances past the replayed ids
* that belong to this shard. Returns the number of records applied, or -1

View file

@ -128,6 +128,7 @@ flowchart TD
classDef rt fill:#8250df,color:#fff,stroke:none
classDef gated fill:#eac54f,color:#000,stroke:none
classDef v2 fill:#0969da,color:#fff,stroke:none
classDef done fill:#1a7f37,color:#fff,stroke:none
I7b2["7b per-shard collector (done — the precondition 8 waited on)"]:::rt
I8x["8 shard-actor runtime: thread-per-core, ownership-move messages"]:::rt
@ -140,7 +141,7 @@ flowchart TD
STREAM2["request body streaming + backpressure"]:::gated
SRESP2["streaming responses + explicit commit point"]:::gated
CANCEL2["per-request cancellation propagation"]:::gated
PUBSUB2["pub/sub + WebSockets (rejected until here)"]:::gated
PUBSUB2["DONE 2026-08-27 — pub/sub + WebSockets (iteration 24: ws_accept + wsframe + room actors)"]:::done
ASYNC9C["20 async attach statements (rejected-for-now alternative)"]:::gated
TIMEOUTS2["idle timeouts become schedulable (net seam still needed)"]:::gated

View file

@ -0,0 +1,93 @@
# Iteration 24 T9 — the drain bug the gate was hiding
**Found 2026-08-27** while finishing T8/T9 on branch `chat-ws-lifecycle`.
Not fixed: the fix is an engine-level decision, recorded here so it is not
rediscovered.
## The symptom
`just chat`'s drain leg asserts both connected clients receive a WebSocket
close frame on `SIGTERM`. Against a **fresh** server it is flaky:
| Sample | Result |
| --- | --- |
| 5 fresh servers, 2 clients each | 4 × `close\|close`, 1 × `eof\|close` |
| 12 fresh servers | 3 failures, one of them `eof\|eof` |
| 16 fresh servers | 5 failures |
A failing client's socket reaches EOF with **no close frame and no
diagnostic** — the process exits and the kernel closes the fd.
## Why the gate never caught it
The drain leg did not start its own server. It inherited `$SRV` from the soak
leg — a server the soak had already pushed 1000 clients through, so every
shard was warm and every actor already scheduled. Draining a warm server hides
the cold-start race. Fixed in this change: **every leg now starts its own
server**, which is what exposed the bug.
## Root cause, traced
Instrumented the sample's actors (diagnostics not committed) and correlated
against failing runs:
1. `DIAG registry-shutdown rooms=1` — main's `send(reg, kind: 2)` **is**
delivered and the Registry runs.
2. `DIAG room-shutdown` — **never printed on a failing run.** The Room never
processes the `kind: 4` shutdown the Registry sends it.
3. The Writer's close branch never runs for the affected client, so no close
frame is written and the fd is never closed by the Writer. Its
`try net.write_dl(...)` is **not** failing — a diagnostic on that path
printed zero times.
4. A client that *does* get a close frame is usually saved by its own
**Reader** noticing `env.stopping()` and running its tail
(`DIAG reader-tail bob r2=1`), not by the room broadcast.
So the drain chain is main → Registry → Room → Writer, three hops across
shards, and **the Room's shard does not reliably adopt its inbox before the
engine stops.**
## What was ruled out
- **Not the spin budget.** Replacing `spin < 20000000` with a wall-clock
deadline of 1 s (`time.ticks()`) still failed 2 of 12. More time does not
help, which is the strongest evidence the room's shard is not being
scheduled at all rather than being scheduled late. That change was reverted:
it fixed nothing and cost a fixed 1 s on every shutdown.
- **Not `dummy_writer()` spawning during shutdown.** Hoisting it to a
Registry field spawned once at startup left 5 of 16 failing.
- **Not a write failure.** See point 3.
## The decision this needs
`main` cannot park after the stop flag (a park unwinds), so it spins — and
spinning is not a barrier. Either:
- **the engine drains pending inboxes before stopping**, so a `send` issued
before the stop flag is guaranteed delivered; or
- **the sample gets a real barrier** — the drain is acknowledged back to main,
which requires main to observe a reply without parking.
The first is the honest fix and belongs to the actor lifecycle (iteration 31,
absorbed into 24). It is a semantic guarantee — "a send before shutdown is
delivered" — not a tuning parameter, and it should be stated in the runtime's
lifecycle docs and pinned by a corpus fixture, not left to a spin count.
## Gate defects fixed alongside (all committed)
1. **fd check was core-count dependent.** `fds_before + 8` read lazy per-shard
init as a leak: shards initialise on first fiber, each taking one
`io_uring` + one `eventfd`, capped at `nproc`. On a 20-core box the first
wave legitimately adds 18. Measured 26 → 44 after 20 clients, then **still
44 after 40 more**. Replaced with the invariant the check is actually for:
a second wave must not raise the count. Core-count independent, and it
catches a slow leak that any fixed slack would hide.
2. **A failed leg orphaned its server.** The drain leg's python died on
`int("")` when `$SRV` was empty, so the soak server was never killed and
its listener broke the *next* run's soak on the same port. `cleanup` now
kills every server a run started, matched on the run's unique temp dir.
3. **Two legs the plan requires were missing** — `WO_SHARDS=1` (the
single-shard control that says a failure is placement's fault) and
`WO_MAILBOX=8` (the drop-slow-member backpressure path). Both added, both
green. The mailbox leg manufactures a genuinely slow member by shrinking
its `SO_RCVBUF`, so it needs no sleeps.

View file

@ -1,60 +0,0 @@
---
slice: "24" # the story that owns the status; see stories/24-chat-websocket-workload.md
status: in-progress
---
# Active slice — chat + actor lifecycle (iteration 24, absorbing 31 + 34)
Branch `chat-ws-lifecycle`. Spec:
[`superpowers/specs/2026-08-23-chat-websocket-actor-lifecycle-design.md`](superpowers/specs/2026-08-23-chat-websocket-actor-lifecycle-design.md)
· plan:
[`superpowers/plans/2026-08-23-chat-ws-lifecycle.md`](superpowers/plans/2026-08-23-chat-ws-lifecycle.md)
· board: [`stories/00-status.md`](stories/00-status.md).
## Progress (2026-08-23)
- ✅ **T1 crypto** (`d14fa9f`): sha1/sha256/hmac_sha256, ids 85–87, RFC
vectors 18/0, corpus pin. Story 34's C-builtin resolution delivered.
- ✅ **T2 bounded mailboxes** (`92754a8`): cap 1024 + `WO_MAILBOX`,
sender-side atomic reserve, WO_T_ACTOR (trap 13) catchable. Plus a
pre-existing compiler fix: try-arm Text places (bare `e.msg`) now
copy before the arm's scope dies (was ASan use-after-free + SEGV).
- ✅ **T6 WS upgrade** (`79cfa01`): `ws_accept` + accept-key + the
101 hijack sentinel; plain HTTP byte-identical (web-app 26/26).
- ✅ **T7 frame codec** (`7ad2ced`): pure-`.wo` RFC 6455 parse/serialize,
probe-verified against the RFC's own bytes.
- ✅ **T3 call/reply** (`ed69841`): `call` parks + typed scalar reply
(WO-E226 through actor-M erasure); actor DEATH landed with it —
callers never hang (mid-call + to-dead both trap catchably). Fixed
TRAPF's fiber-death leak/dangle en route.
Every landed task: full battery 12/12, fresh-built.
## Pending
- ⬜ **T4 monitor(watched, observer, msg)** — id 89. Most of the death
machinery exists (`actor_die`); T4 adds the per-actor monitor list,
the death walk delivering the observer's own M-typed notice,
monitor-of-already-dead firing immediately, full-observer notice =
disclosed stderr drop. Three-argument form (spec deviation, disclosed
in the plan: the caller may be `main`, which has no mailbox).
- ⬜ **T5 time.after(ms, addr, msg)** — id 90, one-shot, no cancel;
rides the T4 deadline plumbing; delivery = runtime send (full = drop
+ stderr line, dead = silent). Corpus: timer-delivery,
timer-generation (the cancel idiom). Both WO_IO backends.
- ⬜ **T8 chat sample** — docs/examples/chat: registry (`call`'s first
consumer), room actors (cap-trap drops slow members, `monitor` reaps
dead writers), reader/writer actor pair per connection over
ws_accept/wsframe; SIGTERM close choreography.
- ⬜ **T9 chat gate** — scripts/chat-accept.sh + raw-RFC6455 python
client; the spec's five checks (functional cross-shard — also the
deferred cross-shard `call` proof — handshake vector, 1k soak with a
`WO_MAILBOX=8` sub-run, drain under both backends + ASan, battery).
- ⬜ **T10 closeout** — stories 24/31/34 → done/ with banners (note the
scalar-reply v1 narrowing + three-argument monitor deviations), board
standup entry, graph nodes, framework README ledger rows, runtime +
chat CODE-LOGIC sections, delete this marker. Final battery.
This file is deleted when the slice lands (board convention). It lives flat in
`docs/` rather than a status folder — since 2026-08-26 no directory in this repo
encodes state; `status:` above is the only place it is recorded.

View file

@ -0,0 +1,82 @@
# `docs/examples/chat` — how the sample is put together
Iteration 24's acceptance workload: rooms, presence and broadcast over
WebSocket, actors on fibers across shards, one binary, no broker. It exists to
*drive* the actor work, so nearly every shape here is chosen to exercise
something the runtime claims.
Gate: `just chat` (`scripts/chat-accept.sh`), which logs to `/tmp/chat.log` —
`tail -F` it while the gate runs.
## The actors
| Actor | Owns | Answers |
| --- | --- | --- |
| `Registry` | name → room map, a fallback room | a `call` returning the room's address; spawns rooms on demand |
| `Room` | its member list (writer address + name) | join, leave, a text line, shutdown |
| `Reader` | the read half of one connection | nothing — it loops on the fd and sends onward |
| `Writer` | the **fd**, and the write half | text, pong, close |
| `ConnWorker` | one accepted connection | runs the HTTP layer over that fd |
`Registry` is the first honest consumer of `call`: the handler runs on the
connection worker's shard, the registry lives wherever placement put it, and
the reply is a scalar — the room's address. That is the cross-shard `call`
proof the gate asserts, not a contrivance added for it.
## Two actors per connection, not one
One fd, two directions, and they block independently. A single actor would have
to be inside `read` to notice the client, and inside `write` to deliver a
broadcast — it cannot be in both, so a broadcast would stall behind a quiet
client's read. Splitting them buys three things:
1. **The `Writer` is the sole writer of that fd.** Frames can never interleave,
which for a framed protocol is a correctness property and not a nicety.
2. **The `Reader` may block as long as it likes.** It sits in `read_dl` with a
30 s idle deadline and nothing else is waiting on it.
3. **The `Writer`'s mailbox becomes the backpressure point.** A slow client
stops draining its socket, its `Writer` blocks in `write_dl`, its mailbox
fills, and the room's next broadcast to it raises a catchable `WO_T_ACTOR`.
The room catches that and drops the member. **This is the whole reason the
mailbox cap is fail-fast** — the room survives its slowest member, and the
gate's `WO_MAILBOX=8` leg proves the path fires rather than assuming it.
`Room.say` is written around that: it shifts every member, tries the send, and
keeps only the members whose send succeeded — a failed one is sent a close and
dropped. So fan-out and eviction are the same pass.
## Who owns the fd
The `Writer`. It closes it, in every branch: a failed write sets `dead` and
closes; a close message writes the close frame and closes. The `Reader` closes
the fd itself in exactly one case — when its `send_close` to the writer traps,
meaning the writer is unreachable and nobody else will. Without that the fd
would leak on a dead-writer path.
`Writer.dead` guards against a second close, which matters because two
independent paths can decide a connection is finished (the reader seeing EOF,
and the room broadcasting shutdown).
## Shutdown choreography
On `env.stopping()` the accept loop stops and `main` sends one message to the
`Registry`, which fans out to every room; each room shifts its members and
sends each `Writer` a close; each writer writes the close frame and closes the
fd. `main` then spins — it may **not** park, because a park after the stop flag
unwinds — and returns, which is what stops the engine.
Independently, every `Reader` notices `env.stopping()` at its loop head and
runs its tail: leave the room, close the writer.
Both paths exist and that is deliberate: the reader path covers a connection
whose room is already gone, the room path covers a reader parked in a read that
has not come back yet.
**This is where iteration 40 came from.** The room path used to be unreliable:
a `Room` whose shard was idle at `SIGTERM` never adopted the shutdown message,
because an idle worker abandoned its inbox on stop. Clients that still got a
close frame were being saved by the reader path alone — which is why the
failure looked random and why a warmed-up server hid it. The engine now
guarantees that a send issued before the stop flag is delivered, so both paths
work as written. Nothing in this file changed to fix it, and that is the point:
the sample was right and the runtime was not.

335
docs/examples/chat/main.wo Normal file
View file

@ -0,0 +1,335 @@
-- chat — iteration 24's acceptance workload. Rooms, presence and
-- broadcast over WebSocket: every connection is a reader actor (sole fd
-- reader) plus a writer actor (sole fd writer); rooms and the registry
-- are actors; delivery between them is ownership-moving sends, across
-- shards when placement lands them there. One binary, no broker.
--
-- CHAT_TOKEN is not needed — chat is open; the framework serves it
-- through [deps] exactly like web-app:
-- woc . && ./target/chat 8080
-- ws://127.0.0.1:8080/ws?room=lobby&name=alice
--
-- The actor split exists because an actor takes ONE message at a time:
-- a single per-connection actor blocked in net read could never hear a
-- broadcast. The reader owns the socket's inbound half and the carry
-- buffer; the writer owns the outbound half so frames never interleave.
use env
use net
use time
use porch
use porch/http
use porch/router
-- ---- message types (one per actor) --------------------------------------
-- To a writer: 1 = text frame, 2 = close (frame + fd close), 3 = pong.
class WriterMsg {
kind: Int
text: Text
}
-- To a room: 1 = join, 2 = leave, 3 = text, 4 = shutdown (drain).
class RoomMsg {
kind: Int
name: Text
text: Text
writer: actor WriterMsg
}
-- To the registry: 1 = lookup (a `call` — the reply is the room's
-- address), 2 = shutdown every room (a `send` on SIGTERM).
class Lookup {
kind: Int
room: Text
}
-- To a reader: everything the connection's inbound loop needs.
class ReaderMsg {
fd: net.Conn
room: actor RoomMsg
writer: actor WriterMsg
name: Text
}
-- One connection accepted, one worker: builds its own App and runs the
-- framework's keep-alive loop (the serving-slice pattern).
class Conn {
fd: net.Conn
}
-- ---- the writer: sole owner of the outbound half -------------------------
class Writer {
fd: net.Conn
dead: Int
fn receive(msg: WriterMsg) {
if self.dead == 1 { return; }
if msg.kind == 1 {
let ok = try net.write_dl(self.fd, ws_text(msg.text), 2000) catch (e) false;
if ok == false {
-- a stalled or gone client: tear the fd; the reader will see EOF
-- and route the leave through the room
self.dead = 1;
net.close(self.fd);
}
return;
}
if msg.kind == 3 {
let ok2 = try net.write_dl(self.fd, ws_pong(msg.text), 2000) catch (e) false;
if ok2 == false {
self.dead = 1;
net.close(self.fd);
}
return;
}
-- close: the drain path (room shutdown or reader-detected close)
self.dead = 1;
let ig = try net.write_dl(self.fd, ws_close(), 1000) catch (e) false;
net.close(self.fd);
}
}
-- ---- the room: members, presence, fan-out --------------------------------
class Mem {
w: actor WriterMsg
name: Text
}
class Room {
members: multi Mem
fn receive(msg: RoomMsg) {
if msg.kind == 1 {
push(self.members, Mem { w: msg.writer, name: "${msg.name}" });
self.say("* ${msg.name} joined");
return;
}
if msg.kind == 2 {
let keep: multi Mem = [];
while len(self.members) > 0 {
let m = shift(self.members);
if m.name != msg.name { push(keep, m); }
}
self.members = keep;
self.say("* ${msg.name} left");
return;
}
if msg.kind == 3 {
self.say("${msg.name}: ${msg.text}");
return;
}
-- shutdown: every member gets a close frame; the list empties
while len(self.members) > 0 {
let m = shift(self.members);
let r = try send_close(m.w) catch (e) 0;
}
}
-- fan-out one line; a member whose mailbox is FULL is a slow client —
-- the fail-fast cap turns it into a drop-from-the-room (the backpressure
-- policy earning its keep)
fn say(line: Text) {
let keep: multi Mem = [];
while len(self.members) > 0 {
let m = shift(self.members);
let ok = try send_text(m.w, "${line}") catch (e) 0;
if ok == 1 {
push(keep, m);
} else {
let r = try send_close(m.w) catch (e) 0;
}
}
self.members = keep;
}
}
-- send wrappers: `try` is an expression, so give it Int results
fn send_text(w: actor WriterMsg, line: Text) -> Int {
send(w, WriterMsg { kind: 1, text: line });
return 1;
}
fn send_close(w: actor WriterMsg) -> Int {
send(w, WriterMsg { kind: 2, text: "" });
return 1;
}
-- ---- the registry: name -> room, spawn on demand --------------------------
class RoomRef {
r: actor RoomMsg
}
class Registry {
rooms: map<Text, RoomRef>
fallback: actor RoomMsg
fn receive(msg: Lookup) -> actor RoomMsg {
if msg.kind == 2 {
for k, v in self.rooms {
send(v.r, RoomMsg { kind: 4, name: "", text: "", writer: dummy_writer() });
}
return self.fallback;
}
if has(self.rooms, msg.room) == 1 {
let have = self.rooms[msg.room];
if have != nil {
return have.r;
}
}
let room: actor RoomMsg = spawn Room { members: [] };
self.rooms[msg.room] = RoomRef { r: room };
return room;
}
}
-- RoomMsg requires a writer field on every construction; the shutdown
-- message has no meaningful one, so a throwaway satisfies the shape (it
-- never receives anything — kind 4 reads no fields).
fn dummy_writer() -> actor WriterMsg {
let w: actor WriterMsg = spawn Writer { fd: 0 - 1, dead: 1 };
return w;
}
-- ---- the reader: sole owner of the inbound half ---------------------------
class Reader {
pad: Int
fn receive(msg: ReaderMsg) {
let carry = "";
let alive = true;
while alive {
if env.stopping() { alive = false; continue; }
let got = try net.read_dl(msg.fd, 4096, 30000) catch (e) nil;
if got == nil {
-- idle deadline or I/O trap: this client is done
alive = false;
continue;
}
let bytes = "${got}";
if len(bytes) == 0 {
alive = false;
continue;
}
carry = carry .. bytes;
let more = true;
while more {
let f = ws_parse(carry);
if f.kind == 0 {
more = false;
continue;
}
carry = f.rest;
if f.kind == 1 {
send(msg.room, RoomMsg { kind: 3, name: "${msg.name}", text: f.payload, writer: msg.writer });
continue;
}
if f.kind == 9 {
send(msg.writer, WriterMsg { kind: 3, text: f.payload });
continue;
}
if f.kind == 10 or f.kind == 2 {
continue; -- pongs ignored; binary tolerated (echo is not chat)
}
-- close frame or protocol error: stop reading
alive = false;
more = false;
}
}
-- the tail sends must survive full mailboxes (a leave storm after a
-- mass close): a trap here would kill the reader and orphan the fd
let r1 = try send_leave(msg.room, "${msg.name}", msg.writer) catch (e) 0;
let r2 = try send_close(msg.writer) catch (e) 0;
if r2 == 0 {
-- the writer is unreachable (full/dead): close the fd ourselves
net.close(msg.fd);
}
}
}
fn send_leave(room: actor RoomMsg, name: Text, w: actor WriterMsg) -> Int {
send(room, RoomMsg { kind: 2, name: name, text: "", writer: w });
return 1;
}
-- ---- HTTP: the upgrade route + usage --------------------------------------
class WsRoute {
reg: actor Lookup
fn handle(req: Req) -> Resp {
if ws_upgrade_valid(req) == false {
return bad_request("expected a websocket upgrade");
}
let rname = req.query["room"];
if rname == nil { return bad_request("expected ?room=<name>&name=<who>"); }
let who = req.query["name"];
if who == nil { return bad_request("expected ?room=<name>&name=<who>"); }
-- the cross-shard call: this handler runs on the connection worker's
-- shard, the registry lives wherever placement put it
let room = call(self.reg, Lookup { kind: 1, room: "${rname}" });
let fd = ws_accept(req);
let w: actor WriterMsg = spawn Writer { fd: fd, dead: 0 };
let rd: actor ReaderMsg = spawn Reader { pad: 0 };
send(room, RoomMsg { kind: 1, name: "${who}", text: "", writer: w });
send(rd, ReaderMsg { fd: fd, room: room, writer: w, name: "${who}" });
return hijacked();
}
}
class Usage {
pad: Int
fn handle(req: Req) -> Resp {
return ok_json("{\"ws\":\"/ws?room=<name>&name=<who>\"}");
}
}
fn build_app(reg: actor Lookup) -> App {
let app = App { middleware: [], routes: [] };
app.get("/", Usage { pad: 0 });
app.get("/ws", WsRoute { reg: reg });
return app;
}
class ConnWorker {
reg: actor Lookup
fn receive(msg: Conn) {
let app = build_app(self.reg);
app.handle_conn(msg.fd, 10000, 10000);
}
}
fn main(args: multi Text) -> Int {
if len(args) < 1 {
print_err("usage: chat <port>");
return 2;
}
let port = parse_int(args[0]);
if port == nil {
print_err("chat: <port> must be a number");
return 2;
}
let fb: actor RoomMsg = spawn Room { members: [] };
let reg: actor Lookup = spawn Registry { rooms: {}, fallback: fb };
let srv = net.listen("127.0.0.1", port);
print("listening on 127.0.0.1:${port}");
while true {
if env.stopping() {
-- the drain: every room broadcasts a close frame and writers flush.
-- main must NOT park here (a park after the stop flag unwinds), so
-- it SPINS — each loop back-edge pays a reduction, and the budget
-- hands the shard to the draining actors between slices; worker
-- shards keep adopting their inboxes until the engine stops.
send(reg, Lookup { kind: 2, room: "" });
let spin = 0;
while spin < 20000000 {
spin = spin + 1;
}
net.close(srv);
return 0;
}
let c = net.accept_dl(srv, 250);
if c != nil {
let w: actor Conn = spawn ConnWorker { reg: reg };
send(w, Conn { fd: c });
}
}
}

View file

@ -0,0 +1,9 @@
name = "chat"
version = "0.1.0"
description = "Iteration 24's acceptance workload: rooms + presence + broadcast over WebSocket — actors on fibers across shards, one binary, no broker"
[runtime]
wo = ">= 0.1"
[deps]
porch = { git = "https://github.com/shoneyj/porch", rev = "v0.1.0" }

View file

@ -56,7 +56,8 @@ place it runs.
feature.
- **Why `main` waits.** `main` is not an actor and has no mailbox, so it sleeps
rather than awaiting — the gap iteration 31's `call` closes for actors and
[24's marker](../../active-slice-2026-08-23-chat-ws-lifecycle.md) tracks.
[iteration 24](../../stories/language-runtime-database/24-chat-websocket-workload.md)
landed 2026-08-27.
Reasoning under the engine side: [`database/src/CODE-LOGIC.md`](../../../database/src/CODE-LOGIC.md).
Contract: [`plan/oop-vm/04-db-binding.md`](../../plan/oop-vm/04-db-binding.md).

View file

@ -24,9 +24,26 @@ strictly better. Recorded as a plan deviation.)
| `query N` | full equality probes on the k index (≈10 rows each), materialized and counted. |
| `write N` | alternating inserts (disjoint k range 2e6+) and updates through query results. Corrupts the checksum by design — durability legs run on a fresh store. |
| `wal N` | the crash battery's vehicle: insert-only (k range 1e6+), `acked <i>` printed AFTER each insert returns — the return IS the ack (RAM applied, WAL record staged, ONE commit done). |
| `wmix N C` | **databasev2 4:** every op a durable write (update through a query result), C at once. Exists because `mix` writes on one op in ten with C=4 — 20 writes in a quick run, measured mean batch **1.01** — so no existing leg could show whether group commit engages. Histogram kind 2, because a replayed store still holds the seeding run's kind-0/1 `Hist` rows. Seed first. |
| `boot` | **databasev2 3:** does NOTHING. With `WO_DATA` set the runtime replays the whole log before `main` runs, so a mode with no work of its own is the only honest way to price boot |
| `verify` | store vs its own Meta rows: count, checksum, one unique probe. Exit 3 on mismatch. |
| `verify-acked M` | after kill -9 mid-`wal`: rows 1..M exist with the right v; rows beyond M allowed (acked after the last print flushed). Exit 3 on mismatch. |
## Env knobs
| var | effect |
| --- | --- |
| `WO_DATA=<dir>` | durability on: replay `<dir>/shard-0.wal` at boot, log every write. Without it the store is RAM-only |
| `WO_SHARDS=<n>` | shard count. **`1` means every statement runs inline on shard 0 and group commit cannot engage** — batches form only where writes queue from other shards |
| `WO_CHECKPOINT_BYTES` / `WO_CHECKPOINT_RATIO` | **databasev2 3:** the checkpoint trigger — the log must exceed the floor AND exceed the ratio times the last compaction's own size. A tiny floor forces compaction in a few writes, which is how the gate tests the policy at all; an enormous one disables it, which is how the checkpoint leg measures the same workload with and without |
| `WO_WAL_STATS=1` | **databasev2 4:** print one line at exit — `walstats batches=… records=… peak_batch=… peak_staged=… compactions=… compact_us_max=… compact_us_total=… compacted_bytes=…`. Opt-in so it does not pollute every durable program's output. Mean batch is `records/batches`; **mean 1.0 means group commit is not engaging**, which is expected for a serial writer or `WO_SHARDS=1` and a bug anywhere else |
**Do not put `WO_DATA` on `/tmp`.** It is `tmpfs` on the reference machine,
where `fdatasync` is free: the same `wmix` run measured **195 000 ops/s at p50
1 µs** there against **2200 ops/s at p50 7200 µs** on ext4. There is no
durability barrier to price on a memory filesystem. The driver keeps its stores
under `bench/` for exactly this reason.
## Coordination idiom (this side of iteration 31)
There is no request/response surface yet: concurrent modes drive

View file

@ -338,6 +338,91 @@ class Mixer {
}
}
-- databasev2 4 part A: every op a durable write, C at once.
--
-- Why this leg exists. `mix` writes on one op in ten with C=4, so at most a
-- handful of writes are ever in flight and group commit has almost nothing to
-- batch: measured mean batch 1.01 over 3112 barriers, peak 3. That is a
-- property of the WORKLOAD, not of the mechanism, and without a write-
-- concurrent leg the iteration's payoff cannot be evaluated either way.
--
-- Updates rather than inserts: comparable to what `mixwrite` measures, and the
-- row count stays flat so a long run does not turn into a growth test.
-- Histogram kind 2, because a replayed store still holds the seeding run's
-- kind-0/1 Hist rows and merging those would report someone else's latencies.
class WJob {
ops: Int
seed: Int
kmod: Int
}
class WMixer {
id: Int
fn receive(msg: WJob) {
let hw: map<Int, Int> = {};
let s = msg.seed;
let i = 0;
while i < msg.ops {
s = lcg(s);
let key = s % msg.kmod;
let o0 = time.ticks();
for r in from x in Item where x.k == key take 1 select x {
r.v = r.v + 1;
}
hist_add(hw, time.ticks() - o0);
i = i + 1;
}
hist_dump(hw, 2);
insert Meta { tag: "wmixdone${self.id}", val: msg.ops };
}
}
fn wmix_mode(total: Int, c: Int) -> Int {
let kmod = meta_val("kmod");
if kmod < 1 {
print_err("wmix: seed first");
return 1;
}
let per = total / c;
if per < 1 {
per = 1;
}
let wall0 = time.ticks();
let i = 0;
while i < c {
let a: actor WJob = spawn WMixer { id: i };
send(a, WJob { ops: per, seed: 4242 + i * 7919, kmod: kmod });
i = i + 1;
}
let done = 0;
while done < c {
time.sleep(20);
done = 0;
i = 0;
while i < c {
if meta_val("wmixdone${i}") >= 0 {
done = done + 1;
}
i = i + 1;
}
}
let wall = time.ticks() - wall0;
let hw: map<Int, Int> = {};
let nw = 0;
for x in from x in Hist select x {
if x.kind == 2 {
if has(hw, x.b) {
set(hw, x.b, get(hw, x.b) + x.c);
} else {
set(hw, x.b, x.c);
}
nw = nw + x.c;
}
}
report("wmix", nw, wall, hw);
return 0;
}
fn mix_mode(total: Int, c: Int) -> Int {
let kmod = meta_val("kmod");
if kmod < 1 {
@ -466,7 +551,7 @@ fn all_mode(n: Int) -> Int {
fn usage() -> Int {
print_err("usage: db-bench <mode>");
print_err(" all N | seed N | read N | query N | write N | wal N");
print_err(" mix N C | msgrate N | growth N int|text | growth-verify");
print_err(" mix N C | wmix N C | msgrate N | growth N int|text | growth-verify");
print_err(" randread N R | replayseed N M | boot");
print_err(" verify | verify-acked M");
return 2;
@ -683,6 +768,10 @@ fn main(args: multi Text) -> Int {
if args[0] == "growth-verify" {
return growth_verify();
}
-- Does NOTHING. With WO_DATA set the runtime replays the whole log before
-- main runs, so a mode with no work of its own measures replay plus a fixed
-- process start — which is what "boot time" has to mean. Both databasev2 1
-- (replay baseline) and databasev2 3 (checkpoint boot) price boot with it.
if args[0] == "boot" {
return boot_mode();
}
@ -746,6 +835,17 @@ fn main(args: multi Text) -> Int {
}
return randread_mode(n, rr);
}
if args[0] == "wmix" {
if len(args) < 3 {
return usage();
}
let wc = parse_int(args[2]);
if wc == nil or wc < 1 {
print_err("db-bench: <c> must be a positive number");
return 2;
}
return wmix_mode(n, wc);
}
if args[0] == "mix" {
if len(args) < 3 {
return usage();

View file

@ -75,7 +75,11 @@ porch = { git = "https://github.com/shoneyj/porch", rev = "v0.1.0" }
- **TLS: none, anywhere.** Deploy behind nginx/caddy; the proxy terminates
TLS+ALPN and gives browsers HTTP/2 while this backend speaks HTTP/1.1
keep-alive. See the web-app sample's README for the nginx sketch.
- `Content-Length` bodies only (no chunked encoding), no WebSockets/SSE,
- `Content-Length` bodies only (no chunked encoding); **WebSockets ARE
supported since 2026-08-27** — `ws_accept` (`http/ws.wo`) performs the RFC
6455 handshake and hands back the hijacked `net.Conn`, and `http/wsframe.wo`
is a pure-`.wo` frame codec; `docs/examples/chat` is the worked example and
`just chat` its gate. **SSE is still absent**, and so is chunked encoding.
JSON-first (no templates). Form-encoded bodies parse through
`form_values(req)` (`+` and `%XX` decoded, nil on any other
content-type); multipart/form-data through `multipart_parts(req)`
@ -130,6 +134,7 @@ first (pure `.wo` cannot express it yet).
| Content negotiation | ✅ `media_type(req)` request-side; `accepts(req, mtype)` response-side (exact, type/*, */*; q-values stripped not ranked — ranking waits for an app serving alternates) — slice 2 |
| Trusted-proxy client IP | 🔶 `client_ip(req)` parses X-Forwarded-For; `net.peer(fd)` (iteration 35) exposes the peer — the verify middleware is now a pure-`.wo` candidate slice |
| Status/header setting · redirects | ✅ builders + `set_header` |
| WebSockets · pub/sub | ✅ **2026-08-27 (iteration 24)** — `ws_accept` does the RFC 6455 handshake and hands back the hijacked `net.Conn`; `http/wsframe.wo` is a pure-`.wo` frame codec. Rooms/presence/broadcast are actors in `docs/examples/chat`, gated by `just chat` (11 checks, 1000-client soak, both `WO_IO` backends, ASan clean). No SSE |
| Lazy body streaming + backpressure · streaming responses · explicit commit point | ⏸ UNBLOCKED by the arc (8/11 landed 2026-08-21) — stays parked until its own slice |
| ETag + conditional requests | ✅ `etag_for` (quoted base64 SHA-256) + `with_etag` (If-None-Match → 304) over iteration 34's digest builtins — slice 2 |
@ -140,7 +145,7 @@ first (pure `.wo` cannot express it yet).
| Ordered middleware chain | ✅ registration order, `?Resp` short-circuits |
| Request-scoped context | ✅ `req.ctx` map (slice 2): middleware writes, handlers read; identity stays in `principal` |
| Guaranteed teardown | 🔶 every fd closes on every path (gate-proven); no user teardown hooks yet |
| Cancellation into pending storage ops | ⏸ UNBLOCKED by the arc (8/11 landed 2026-08-21) — stays parked until its own slice |
| Cancellation into pending storage ops | ⏸ **unblocked, not built.** The arc landed 2026-08-21 and iteration 24 (2026-08-27) added the lifecycle a cancellation would ride — `call` with a catchable trap when the callee dies, bounded mailboxes, `monitor`, and `time.after` for a deadline. Nothing here consumes them yet; it stays parked until its own slice |
| Panic recovery | 🔶 trap = 500 and the server survives ✅; "rolls back the transaction" is framework v2 (needs `transaction { }`, iteration 18) |
### Storage integration (the differentiator — framework v2 territory)

View file

@ -29,6 +29,31 @@ Writeonce's phase 12 `Engine` keeps an `HashMap<(TypeName, SegmentOffset), Cache
The kernel page cache does most of the work. `pread` against an fd that already has its page cached is a memcpy. `pwrite` populates the page cache without going to disk until pressure or `fsync`. This is why writeonce explicitly does NOT use `O_DIRECT` (see [`linux/12-pwrite-fsync.md`](../linux/12-pwrite-fsync.md)) — the page cache is the one cache we want.
## Checkpoint — the writeonce shape
> **⚠ TWO CORRECTIONS, 2026-08-28** (found while brainstorming
> [databasev2 3](../../../stories/databasev2/03-wal-checkpoint.md); spec:
> [`2026-08-28-wal-checkpoint-design.md`](../../../superpowers/specs/2026-08-28-wal-checkpoint-design.md)).
>
> 1. **Postgres does NOT update its control file by rename.** The claim below
> that "Postgres does the same in `BasicOpenFile` + `fsync_parent_path`" is
> wrong: `update_controlfile` (`src/common/controldata_utils.c`) opens the
> existing file `O_WRONLY`, writes a zero-padded **full block in place**, and
> relies on **CRC32C** over the struct to detect a torn write. The
> `fsync(parent_dir)` reasoning below is still correct *for renames* — it is
> just not what Postgres does here.
> 2. **The checkpoint sketch below assumes writeonce has segment files.** It
> says records before the LSN are "*known* to be in the segment files". There
> are none: the WAL is writeonce's only durable form, replayed into RAM, and
> [databasev2 2](../../../stories/databasev2/02-table-storage-modes.md)
> deliberately rejected adding a paged store. This document predates the
> databasev2 direction, so read the loop below as a design for an
> architecture that was not chosen.
>
> What survived the comparison is the **ordering discipline**, not the
> architecture: publish the new "recovery starts here" atomically and last, so a
> crash falls back. writeonce gets that from one `rename` of the whole log —
> possible only because its records are full row images, where Postgres' are
> page deltas.
Postgres' checkpoint runs in a separate process and signals the postmaster when done. Writeonce's runs as a periodic loop step:

View file

@ -65,7 +65,7 @@ The metadata exists for exactly one reason: `json.encode`/`json.decode` are runt
- **the OS half** — fs.exists/list/stat/read_all/read_at/append, time.sleep/local/iso, env.get/stopping, net.listen/accept/read/write/close, proc.run. Ids 40–56; `runtime/src/sysio.c`. A member that returns a record takes that record's **class id as its last argument**, so the VM allocates what it fills without knowing any source type name.
- **json** — encode (value + the value's static kind), decode (text + the class id to build). Ids 57–58; `runtime/src/json.c`. Decode yields the zero word on malformed input rather than trapping, which is what makes `json.decode(t) as T` a checked decode.
- **59 `map_get_opt`** (`m[k]`'s optional read), **60 `text_copy`** (Text's ownership-boundary copy — Task 1 of the executable plan).
- **database** — **61 `db_insert`** (iteration 9, Task 3): window is R[B] = class id, R[B+1..] = one slot per **declared** field in declaration order; result R[A] = the new row's id. The loader validates the class-id slot statically (variable window: the field slots are validated at runtime by the engine against the class table). Engine failure traps `WO_T_DB`; a failed WAL commit traps `WO_T_IO` after un-applying the row. `database/src/db.c`.
- **database** — **61 `db_insert`** (iteration 9, Task 3): window is R[B] = class id, R[B+1..] = one slot per **declared** field in declaration order; result R[A] = the new row's id. The loader validates the class-id slot statically (variable window: the field slots are validated at runtime by the engine against the class table). Engine failure traps `WO_T_DB`. **A failed WAL commit no longer traps (databasev2 4, 2026-08-28): it ENDS THE PROCESS** with exit status 74 and a diagnostic naming the failing operation, the log path, `errno` and the batch size. `WO_T_IO` is unreachable from any DB write. The reason is that only `insert` could ever un-apply itself — `update` and `delete` never could, and their own comments admitted they left RAM ahead of disk — so continuing after a durability failure meant serving state that would not survive a restart. Retrying is not offered either: on Linux a failed `fsync` may already have discarded the dirty pages, so a second call can report success having written nothing. Replay is the recovery that works. `database/src/db.c`, `database/src/wal.c`.
**`?T` and nil.** A heap-shaped optional (`?Text`, `?Rec`, `?multi`, `?map`, `?@gc`) stores what `T` stores and spells nil as the **zero word** — every per-kind drop plan already ignores a zero slot, so `?T`'s field kind is `T`'s. A **nullable scalar** (`?Int`, `?Bool`, `?Timestamp`, `?Id`) cannot: `0` is a perfectly good `Int`, and real programs store it in a `?Int`. Its nil is therefore `WO_NIL_SCALAR` = −2^62 (not `INT64_MIN`: the compiler's own integers are 63-bit, so that value is not expressible on the emitting side). Such a field is marked `WOB_FIELD_NIL_SCALAR` in `field_class[i]`, which is how the runtime knows to write that word where it must produce absence itself — today only `json.decode` leaving a key absent, and `parse_int` on unparseable input.

View file

@ -118,11 +118,49 @@ R[B+1..] = one slot per declared field in declaration order (the literal's
order is irrelevant — slots are the class table's).
Execution: `wo_row_insert` (RAM, engine copies every value), then — when
durability is on — stage + **commit before the builtin returns**: the
builtin's return IS the acknowledgment, so ack-after-fsync holds at
statement granularity until iteration 8 brings tick-scoped group commit. A
failed commit un-applies the row and traps `WO_T_IO`; engine failures trap
`WO_T_DB`. Durability is opt-in: `WO_DATA=<dir>` makes the CLI replay
durability is on — stage, then a barrier before the acknowledgment. **Updated
2026-08-28 (databasev2 4 part A): group commit landed, and the barrier's
location now depends on which path the statement takes.**
A statement arriving from a worker shard marshals to shard 0 and parks; shard 0
stages every such request, issues **one** barrier when its queue empties, and
only then releases the held replies — so each writer is acknowledged after the
barrier that carried *its* record. A statement already running on shard 0 takes
the inline path and still commits before the builtin returns, because it has no
reply to hold: it returns into its own fiber, and batching it would require
parking that fiber on the barrier (deferred to part B). The boundary is the
queue draining, **not** the tick this document previously anticipated — a tick
would add latency to a lone writer, taxing an idle system to serve a busy one.
Measured: ~2.9× durable write throughput and ~2.1× lower p50 on a
write-concurrent workload; unchanged for a serial writer, which has nothing to
batch with.
**Compaction (databasev2 3, 2026-08-29) may run only where NOTHING IS STAGED.**
That is a correctness requirement, not a scheduling preference: the staging
buffer holds records destined for a file that compaction is about to replace, so
compacting with a non-empty buffer would either write them into a file about to
be discarded or lose them with it. In practice the safe points are immediately
after a barrier — the drain's, and the inline path's — and both are wired.
`wo_wal_compact` refuses a non-empty buffer as a backstop rather than trusting
its callers.
**Recovery is unchanged by compaction.** The result is an ordinary log in the
ordinary record grammar, replayed from byte 0; there is no snapshot, no second
source, no cutoff offset and no control file. Crash safety comes from `rename`
being atomic: before it the live log is intact and the temp file is not
authoritative, after it the new log is complete, and no reader can observe a
mixture. A crash mid-rewrite leaves a temp file, which the next open removes.
A failed compaction is a **missed optimisation, not a durability event** — the
original log is left usable and the process continues. It must not take the
fatal path below.
A failed commit **no longer traps — it ends the process** (exit 74, with a
diagnostic naming the operation, log path, `errno` and batch size). So does a
failed staging. `WO_T_IO` is unreachable from a DB write. One rule: once a
statement has mutated RAM, the outcomes are durable or death. Engine failures
still trap `WO_T_DB`. Durability is opt-in: `WO_DATA=<dir>` makes the CLI replay
`<dir>/shard-0.wal` before the entry runs and commit every insert; without
it the engine is RAM-only (every corpus fixture runs that way).

View file

@ -279,6 +279,8 @@ unset `env.get` are nil.
| `sha1(bytes)` | `-> Bytes` | 20-byte digest (id 85, iteration 34) — exists because RFC 6455's Sec-WebSocket-Accept demands SHA-1 |
| `sha256(bytes)` | `-> Bytes` | 32-byte digest (id 86, iteration 34) |
| `hmac_sha256(key, msg)` | `-> Bytes` | RFC 2104 over SHA-256, both args Bytes (id 87, iteration 34); key > 64 bytes hashed first |
| `monitor(watched, observer, msg)` | — | iteration 24 (id 89): the observer's own M-typed msg (MOVED) is delivered when watched dies (trap-death); already-dead delivers now; a full observer's notice is dropped with a stderr line — no fiber to trap |
| `time.after(ms, addr, msg)` | — | iteration 24 (id 90): one-shot timer — msg (MOVED) arrives as an ordinary send after ms on the arming shard; ms <= 0 delivers now; NO cancel — the generation-counter idiom (run/timer-generation) is the answer |
| `call(addr, msg)` | `-> R` | send that WAITS (id 88, iteration 24): the message moves like `send`'s, the caller's fiber parks until the receive's return value arrives. R = the receive's declared return type — every `receive(msg: M)` program-wide must agree on it and it must be a copyable scalar in v1 (WO-E226 otherwise). A dead callee traps WO_T_ACTOR, immediately or mid-call — a `call` never hangs |
| `env.get(name)` | `-> ?Text` | unset is nil |
| `env.stopping()` | `-> Bool` | SIGTERM/SIGINT latch, handlers installed on first use |

View file

@ -216,3 +216,187 @@ which is too coarse for the one number a checkpoint is meant to improve.
WAL bytes are measured as the file's **non-zero prefix**, never its size: shard
WALs are `fallocate`'d to 1 MiB, so an empty store reports 1048576.
## 6. WAL group commit: one barrier per drain (databasev2 4 part A)
**Measured 2026-08-28.** Before this, the engine committed per *statement*:
`db.c` called `wo_wal_commit` immediately after every append, so each row
change bought its own `pwrite` + `fdatasync`. Now shard 0 stages every queued
write request, issues one barrier, and only then releases the held replies.
### The controlled before/after
Same machine, same workload (`wmix 4000 32` — every op a durable update, 32
concurrent), same build except `db.c` and `vm.c`, two runs each, interleaved:
| | ops/sec | p50 | p99 |
| --- | --- | --- | --- |
| per-statement barrier | 2213 · 2177 | 7183 · 7251 µs | **20000 · 20000 µs** |
| group commit | **6216 · 6525** | **3458 · 3444 µs** | 11139 · 5971 µs |
**≈2.9× throughput, ≈2.1× lower p50.**
**The p99 "before" figure is at the histogram ceiling, not a measurement.**
`hist_add` clamps at 20000 µs, and both before-runs pinned there — so the true
before p99 is ≥20 ms and unknown. The improvement is *at least* 2.3×; the
honest statement is that the old p99 was off the end of the instrument.
### Confirmation from the committed baseline
The full campaign gives the same answer a second way. `s1` takes the inline
path, which commits per statement **by design**, so within one build the two
shard configurations are batching-off against batching-on:
| Leg | ops/sec | p50 | p99 | mean batch | peak batch |
| --- | --- | --- | --- | --- | --- |
| `durable.s1.wmix` (inline, unbatched) | 1467 | 455 µs | 721 µs | **1.0** | 1 |
| `durable.sN.wmix` (batched) | **5117** | 8208 µs | 12169 µs | **5.43** | 57 |
3.5× throughput, agreeing with the 2.9× above. Note `sN` latency is *higher*
while throughput is 3.5× better: 64 writers queueing behind one owner shard
trade per-op latency for barrier amortisation, which is what group commit is.
Batching scales with write concurrency exactly as designed — mean batch at
C = 4 / 16 / 64 was **1.13 / 1.76 / 5.35**, peak **3 / 10 / 39**.
### What did NOT improve, and why that was predicted
`durable.sN.mixwrite` went **480 → 492 ops/s** — unchanged. That is the metric
the spec *originally* named as the payoff, and correcting it was part of the
brainstorm: `mix` writes on one op in ten with C=4, so a quick run performs
**20 writes** and mean batch measured **1.01** over 3112 barriers. A workload
that never has two writes in flight cannot be helped by batching them.
`durable.*.seed` is likewise unchanged: a serial single writer has nothing to
batch with under any scheme.
**So the payoff is real but conditional: it appears exactly where concurrent
durable writes fan into the owner shard, and nowhere else.**
### Two traps worth recording
**Do not benchmark durability on `/tmp`.** It is `tmpfs` here, where
`fdatasync` is free — the same `wmix` run reported **195 000 ops/s at p50 1 µs**
there against **2200 ops/s at p50 7200 µs** on ext4. There is no barrier to
amortise on a memory filesystem, so a group-commit measurement taken there
measures nothing. `db-bench` gets this right by keeping its stores under
`bench/`.
**The record count is not the update count.** `wmix` staged 7755 records for
4000 updates because the histogram dump and the done-marker are themselves
durable inserts. They arrive as an end-of-run burst, which is batch-friendly,
so `mean_batch` is not purely update-driven. Peak staged bytes stayed small
(2793 B at C=64), which is what settled the decision to ship **no batch cap**:
the request queue's existing upstream bound is sufficient.
### The cost side: tail latency on the owner shard
Group commit is a trade, and the full battery made the other side of it visible.
**A bug first, caught by `durable.sN.mixread.p99`.** The drain initially held
*every* DB reply until the barrier — including **reads**, which stage nothing and
have no stake in durability. That parked readers behind an fsync for no reason
and pushed read p99 from ~1043 µs to **4057 µs**. Reads are now released
immediately; only a statement that actually staged a record has its reply held.
**What remains is inherent, not a bug.** A barrier now blocks the owner shard
**longer** (more records per fsync) even though it blocks **less often**, so
anything arriving during a barrier — reads included — waits behind it. Measured
across three full runs of the same build, `durable.sN.mixread.p99` came in at
**1043 / 2318 / 4147 µs** and `wmix.p99` at **8758 / 20000 µs**, a 2–4× spread
with the box near idle.
So the honest summary of part A on a single-threaded owner shard: **~3× write
throughput, at the price of a longer and noisier tail for everything queued
behind a barrier.** That is precisely what part B (async submission — submit the
barrier and keep serving) would undo, and it is a better argument for part B than
the "close the 66× gap" framing part B was originally given.
**Gating consequence.** `durable.sN.*.p99us` now carries a 100% tolerance,
because a 2–4×-variable tail gated at 50% gates the disk rather than the engine.
The **floor** is the real guard there, and it is not slack: `mixread`'s floor
(4172 µs) came within 25 µs of tripping on the worst observed run.
## 7. WAL checkpoint: compaction (databasev2 3)
**Measured 2026-08-29.** Before this the log grew forever: nothing ever removed
superseded records, so boot replayed all history and the file only ever got
bigger. Compaction rewrites it as one record per live row and swaps it in with
`rename`.
### Space and boot — the same workload, twice
Identical work, differing only in whether checkpointing may fire (an enormous
floor disables it). Full campaign:
| | checkpointing off | checkpointing on |
| --- | --- | --- |
| WAL used | 1 962 358 B | **907 094 B** |
| boot (median of 3, `boot` mode) | 114 ms | **64 ms** |
| compactions | 0 | 6 |
**2.16× space reclaimed, 1.78× faster boot.** Boot is measured with a mode that
does nothing at all: with `WO_DATA` set the runtime replays the whole log before
`main` runs, so a mode with no work of its own is the only honest way to price
replay. It is *not* measured through the driver's `run()` helper, which samples
RSS on a 250 ms poll — timings taken that way reported "251 ms" both with and
without checkpointing, which is the harness's clock rather than the engine's.
### The stop-the-world pause, and why it stopped being 8× worse
Compaction blocks the owner shard for its duration. The spec refused to assume
that was acceptable, so it is measured and gated against a stated **50 ms**
budget: a stall a serving process can absorb without a client seeing a timeout.
Measured **2 651 µs** on the full campaign — comfortably inside it.
It was not always. The first implementation flushed the dump through
`wo_wal_commit`, which `fdatasync`s, so a dump paid one barrier per 256 records:
| live set | pause, per-flush fsync | pause, one final fsync |
| --- | --- | --- |
| ~107 KB | 23 948 µs | **2 903 µs** |
| ~500 KB | 36 361 µs | **7 526 µs** |
| ~1.98 MB | 107 649 µs | **13 212 µs** |
Marginal rate went from **~22 MB/s to ~181 MB/s** — from sync-bound to
bandwidth-bound. Intermediate durability during a dump is worthless: the temp
file is not authoritative until the rename and is fsynced once immediately
before it, so those barriers bought nothing and cost 8×.
**The pause is O(live rows), and that is the number that eventually forces an
incremental design.** At ~181 MB/s a 1 GB live set implies roughly 5.5 s — well
past any interactive budget. The spec deliberately did not buy incremental
copying in advance; this is the measurement it is to be bought against.
### Gating
`ckpt.reclaim_x` is the feature's central claim and is gated tightly (15%).
Everything else in the leg — boot times, the pause, the byte counts — is
wall-clock or workload-shaped on a shared box and carries a wide tolerance,
because waiving them *all* would have left the leg ungated. The leg also
asserts two things directly rather than trusting a metric: that some compaction
actually ran (otherwise it proves nothing), and that the log really is smaller
with checkpointing on.
One direction bug worth recording: `reclaim_x` was first recorded as
lower-is-better by the default detector, which would have **passed "reclaimed
nothing" and failed an improvement** — the central claim gated backwards.
**Gate-tolerance corrections made while closing this iteration**, both recorded
because a widened tolerance that is not justified is indistinguishable from a
silenced regression:
- **`ckpt.pause_us_max` is no longer gated against a baseline.** The raw pause
scales with the live set, and this workload's live set is not fixed —
`wmix`'s `hist_dump` inserts a row per latency bucket, so a noisier box makes
more buckets, more rows, and a longer pause. What belongs to the engine is the
**rate**, so `ckpt.pause_us_per_mb` carries the real tolerance and the raw
pause keeps the absolute 50 ms budget as its guard.
- **`ram.*.msgrate.msgs_sec` moved from 15% to 70%, and this one is
pre-existing.** Across the ten full runs recorded on 2026-08-28/29 — several
predating the checkpoint work — it ranged **10.7M to 17.9M msgs/sec, a 1.67×
spread**. A 15% gate on a scheduling-bound throughput metric fails
intermittently whatever the engine does.
- **`durable.sN.*.p99us` moved from 100% to 300%**, with more evidence than the
first widening had: mixread p99 measured 1043 / 2318 / 4147 µs and mixwrite
1623 / 4446 µs across runs of the same build. The floors remain the real
guard, and they are not slack — mixread's came within 25 µs of tripping.

View file

@ -67,6 +67,161 @@ behind this board; live Obsidian Dataview views:
## ▶ NEXT PLAN
### Landed 2026-08-29 — databasev2 3, WAL checkpoint (the chain's last link)
**Implemented last time (2026-08-29):** compaction. The log used to grow forever
— nothing removed superseded records, so boot replayed all history. It is now
rewritten as one record per live row into a temp file and swapped in with
`rename`. Six tasks, brainstormed and spec'd first
([spec](../superpowers/specs/2026-08-28-wal-checkpoint-design.md) ·
[plan](../superpowers/plans/2026-08-28-wal-checkpoint.md)).
**Key findings (measured, not asserted):** **2.16× space reclaimed**
(1 962 358 → 907 094 B), **boot 114 → 64 ms**, stop-the-world pause **2 651 µs**
against a stated 50 ms budget. Reading `.dev/reference/postgresql` was what made
the design defensible rather than lazy: **Postgres never compacts its WAL**,
because its records are page deltas and a compacted redo log is not a store —
hence heap files, a control file, a redo pointer, a second recovery source and a
separate checkpointer process. Ours are **full row images**, so a compacted log
*is* a complete store, and all of that machinery disappears. What was worth
porting is the ordering discipline — publish the switch atomically and last — and
one `rename` provides it.
**Learned — two bugs of mine that measurement found, not review:** wiring the
trigger only into the drain left **`WO_SHARDS=1` never compacting**, its log
growing forever (536 KB where multi-shard held 446 KB), because a statement on
the owner shard never enters that drain. And the dump was **8× slower than
necessary**, flushing through the committing path and paying one `fdatasync` per
256 records for durability that is worthless before the rename — one final
barrier took a 2 MB dump from 107 649 µs to 13 212 µs, ~22 MB/s to ~181 MB/s.
Separately, the crash battery's *first* version failed on correct code ~1 run in
3: it acked deletes after committing them, so a kill in between made it demand a
row the engine was right to remove. Deletes now announce intent first.
**Dependencies unblocked:** every link in the concurrency + fiber chain has now
landed its planned work — stage 3 → 22 → 24 (absorbing 31 + 34) → 40 →
databasev2 4 part A → databasev2 3. **Not "complete", precisely:** chain 5 stays
`in-progress` because databasev2 4's part B was never done, and its premise was
invalidated by part A rather than satisfied. Nothing in the chain is blocked on
anything else in it.
**Next steps:** the honest queue is (1) databasev2 2's outstanding 5c/5d, whose
`resident: keys` half is unimplemented and now carries a recorded obligation —
compaction invalidates every WAL offset it stores, so the compactor must rebuild
that map; (2) databasev2 4 **part B**, whose premise was invalidated by part A
and which needs re-brainstorming rather than starting; (3) the O(live rows)
pause, ~5.5 s at a 1 GB live set, which is the number an incremental checkpoint
must be bought against.
**`.dev/reference` used:** `postgresql` — `xlog.c` (`CreateCheckPoint`, segment
recycling), `checkpointer.c` (the time-or-volume trigger), and
`controldata_utils.c`, which also corrected a prior exploration doc: Postgres
updates its control file **in place with a CRC**, not by rename.
---
### Landed 2026-08-28 — databasev2 4 part A, WAL group commit
**Implemented last time (2026-08-28):** one durability barrier per drain
instead of one per statement. Shard 0 stages every queued write request, holds
each reply, commits once when its queue empties, then releases all — so a writer
is acknowledged after the barrier that carried *its* record, which was the
intended contract all along and was true before only because every batch had one
member. Six tasks, brainstormed and spec'd first
([spec](../superpowers/specs/2026-08-28-wal-group-commit-design.md) ·
[plan](../superpowers/plans/2026-08-28-wal-group-commit.md)).
**Key findings (measured, not asserted):** **≈2.9× durable write throughput,
≈2.1× lower p50** on a write-concurrent workload, confirmed a second way by the
`s1`-vs-`sN` split within one build (1467 → 5117 ops/s, mean batch 1.0 → 5.43,
peak 57) — 2.9× and 3.5× agreeing. Batching scales with contention: mean batch
1.13 / 1.76 / 5.35 at C = 4 / 16 / 64. **The story's premise was wrong**: it
said "fsync-per-commit" and the engine was fsync-per-**statement**, committing
after every append at all six sites — so part A was closer to deleting calls
than adding a mechanism.
**Learned — three things the measurement corrected, not the code:**
(1) **`/tmp` is tmpfs here, where `fdatasync` is free.** The same run reported
195 000 ops/s at p50 1 µs there against 2200 at 7200 µs on ext4. A group-commit
measurement taken on a memory filesystem measures nothing; `db-bench` is right
to keep its stores under `bench/`. (2) **No existing leg could exercise the
feature** — `mix` writes on one op in ten with C=4, giving 20 writes and mean
batch 1.01, so a `wmix` write-concurrent leg had to be added or the payoff was
unevaluable either way. (3) **The before-p99 was off the instrument** —
`hist_add` clamps at 20000 µs and both before-runs pinned there, so the gain is
*at least* 2.3× and the true old p99 is unknown.
**Dependencies unblocked — and one dependency invalidated.** `WO_T_IO` is
unreachable from a DB write: a failed stage or barrier now ends the process
(exit 74, diagnosed), replacing three behaviours that disagreed — `insert`
un-applied itself while `update` and `delete` returned a catchable trap and
admitted in their own comments that they left RAM ahead of disk. **Part B's
premise is invalidated**: it was justified by "close the 66× durable gap", but
that gap is two problems. Concurrent fan-in was a batching problem and is now
~3× better; a **serial** writer waiting on one barrier is a latency problem that
batching cannot touch and io_uring does not obviously help either. Part B should
be re-brainstormed, not started.
**Next steps:** either re-brainstorm part B against its corrected premise, or
take chain 6 ([databasev2 3](databasev2/03-wal-checkpoint.md), WAL checkpoint),
which now has the replay "before" it lacked. **(Superseded 2026-08-29: it
landed.)** Two debts named rather than hidden:
the abort path is not exercised (forcing a real `fdatasync` failure needs mount
privileges), and single-shard concurrent batching needs the inline-path park —
the same machinery part B would need.
**`.dev/reference` used:** none this slice. The sources were the engine's own
code and the Linux `fsync`-failure semantics that make retrying unsound.
---
### Landed 2026-08-27 — iteration 24, chat + actor lifecycle (absorbing 31 + 34)
**Implemented last time (2026-08-27):** the slice closed and merged to master
(`ed5334d`, fast-forward). T4 `monitor` + T5 `time.after` (ids 89/90) had
landed on the branch; this session merged master in (adopting the `porch`
rename), finished T8/T9, fixed the gate, found and fixed a runtime bug, and did
T10. Iterations 31 and 34 land inside it.
**Key findings (measured, not asserted):** finishing the gate mattered more than
finishing the sample. Making **every leg start its own server** — instead of the
drain leg inheriting the soak's warmed one — exposed that **5 of 16**
fresh-server SIGTERM drains left a client at EOF with no close frame and no
diagnostic. Traced to `shard_main`: `NEXT_RUNNABLE()` already stated the
contract ("a WORKER on stop keeps DRAINING … close frames!") but the **idle**
branch reaped and broke, abandoning its inbox. An actor between messages is
exactly that idle case. Split out as
[40](language-runtime-database/40-shutdown-drain-guarantee.md); **20 of 20
clean** after. Also measured: the fd check had been core-count dependent — lazy
per-shard init takes one `io_uring` + one `eventfd` per shard, capped at
`nproc`, so 26 → 44 on a 20-core box read as a leak. **1000 connections left it
at 44**, which settled it.
**Learned:** three of the four gate failures were **stale build artifacts**, not
code. A branch switch leaves `compiler/_build/` and `runtime/build/` holding the
other branch's binaries, and a `woc` emitting `.wob` v7 against a v6 runtime
surfaces only as "no listener" — rebuild both before believing a gate failure.
And a gate that reuses another leg's server is not merely untidy: it hid a real
bug, and when its own leg failed it orphaned a listener that broke the *next*
run. Example apps now log to `/tmp/<app>.log` so a developer can `tail -F` them.
**Dependencies unblocked:** PUBSUB2 (WebSockets + pub/sub, rejected until this
point) is done; the porch ledger's WebSocket rows are ✅ and its cancellation row
is unblocked-not-built. Chain position 4 is complete, so **the chain's next link
is [databasev2 4](databasev2/04-io-uring-commit.md)** (io_uring group-commit).
Still blocked: CSRF and sessions — iteration 34 shipped HMAC but **there is
still no RNG**, and HMAC authenticates a token without being able to mint one,
which is [39](language-runtime-database/39-web-framework-parity.md)'s leading
item.
**Next steps:** databasev2 4, or databasev2 2's outstanding 5c/5d. One debt is
named rather than hidden: iteration 40's guarantee is proven only by the chat
gate — nothing in `runtime/test/` drives `wo_engine_start`/`wo_engine_stop` and
no corpus fixture can trigger a stop, so pinning it lower needs new
multithreaded test infrastructure.
**`.dev/reference` used:** none this slice. The sources were RFC 6455, RFC
3174/4231 for the digest vectors, and the kernel's own interfaces for the drain.
### Landed 2026-08-27 — databasev2 1, the RAM ceiling measured
**Implemented last time (2026-08-27):** databasev2 1 refined (three forks
@ -226,10 +381,16 @@ no reference project was consulted for the implementation).
**The concurrency + fiber chain — ✅ stage 3 → ✅ 22 → 🔄 24 (absorbing
31 + 34) → 23 → 32.** The chain's original order put 31 before 24; the
2026-08-23 directive absorbed 31 INTO 24, and 34 resolved with it, so
those three are one slice. **The live slice is iteration 24** — spec and
plan approved 2026-08-23, executing on branch `chat-ws-lifecycle`, five
of ten tasks landed. Its running state is the marker doc
([`2026-08-23-chat-ws-lifecycle.md`](../active-slice-2026-08-23-chat-ws-lifecycle.md)),
those three are one slice. **Iteration 24 is nine of ten tasks landed and MERGED TO MASTER
on 2026-08-27** (fast-forward, `ed5334d`): T1 crypto, T2 bounded mailboxes,
T3 call/reply, T4 `monitor` + T5 `time.after` (ids 89/90 — the reserved holes
are now filled), T6 ws upgrade, T7 frame codec, T8 chat sample, T9 the chat
gate. Verified on master: chat 11 checks 0 failures at the full 1000-client
soak, runtime battery 36 suites 0 fail, compiler 556 checks 0 fail, corpus
119 checks 0 fail. Only **T10 closeout** remains — which is what still holds
stories 24/31/34 open. Finishing T9 exposed and fixed a real runtime bug,
split out as [40](language-runtime-database/40-shutdown-drain-guarantee.md). Its running state is the marker doc
(the marker doc, deleted at closeout per the convention),
which is the file to read for what is done and what is next; stories
[31](language-runtime-database/31-actor-lifecycle.md) and
[34](language-runtime-database/34-crypto-builtins.md) keep
@ -279,6 +440,9 @@ both still literal holes in `wob.h`'s builtin enum; then T8 the chat
sample, T9 its gate, T10 closeout setting 24/31/34 to `status: done`) → 23
(io_uring group-commit — target: close the 4.5k→297k durable gap) →
32 (WAL checkpoint). Held tail resumes on its own precedence notes.
> (**Superseded 2026-08-28:** 24 landed, and 23's part A landed with it —
> "close the 4.5k→297k durable gap" turned out to be the wrong target; see
> the databasev2 4 row.)
**`.dev/reference` used:** none this slice (the LW_SOAK discipline and
linkcheck.py precedent came from in-repo scripts).
@ -436,11 +600,15 @@ that sequences its tasks. Read one, approve, then the next starts.
| 19 | [Float + Bytes](language-runtime-database/19-missing-scalar-types.md) | ✅ **landed 2026-08-20** — `.wob` v5: Float constant tag, field kinds 6/7, opcodes 34-41 (IEEE-quiet f64), builtins 70-83. Full stack: literals, arithmetic, `@table` column, WAL bit-exact replay, json fractions in / shortest-round-trip out, `?Float` reserved-NaN nil, total-order index (NaN last, `-0.0` == `+0.0`), Bytes + base64. No implicit Int/Float mixing (WO-E201); `float`/`trunc` are the only bridges. Proof: web-app price is a real Float (`{"price":9.99}`), `just web-app` 23/0; corpus 103/0 |
| 11 | [Fibers](language-runtime-database/11-fibers.md) | ✅ **landed 2026-08-21** with the arc (`just fibers` 10/0); fs-park re-scoped out of v1, disclosed in the story |
| 22 | [Durability, throughput, scale](language-runtime-database/22-durability-throughput-scale.md) | ✅ **landed 2026-08-21** — db-bench + baseline.json (74 metrics) + restart/kill -9 proofs both shard counts; durable 4.5k vs ram 297k inserts/s, reads O(table), msgrate 13.4M/2.45M |
| 31 | [Actor lifecycle](language-runtime-database/31-actor-lifecycle.md) | 🔄 **absorbed into 24** (directive 2026-08-23) and half landed there: `call` request/response with a typed scalar reply (`WO_B_CALL = 88`, WO-E226), bounded mailboxes (`WO_MAILBOX`, cap 1024, catchable `WO_T_ACTOR`), and actor death that traps callers instead of hanging them. Still open: `monitor` and `time.after` — ids **89 and 90 are reserved holes** in `wob.h`, which is the machine-checkable proof of what is left. Supervision trees stay out of v1 |
| 24 | [chat: WebSocket workload](language-runtime-database/24-chat-websocket-workload.md) | 🔄 **the live slice** (absorbing 31 + 34, directive 2026-08-23) — branch `chat-ws-lifecycle`, 5/10 tasks landed: crypto, bounded mailboxes, WS upgrade, frame codec, `call`/reply + actor death. Pending: `monitor`, `time.after`, the chat sample, its gate, closeout. State lives in [the marker](../active-slice-2026-08-23-chat-ws-lifecycle.md) |
| 31 | [Actor lifecycle](language-runtime-database/31-actor-lifecycle.md) | ✅ **LANDED 2026-08-27 inside 24** (directive 2026-08-23). All four mechanisms: `call`/reply with a typed scalar reply (`WO_B_CALL = 88`, WO-E226), bounded mailboxes (`WO_MAILBOX`, cap 1024, catchable `WO_T_ACTOR`), actor death that traps callers instead of hanging them, **`monitor` (89)** and **`time.after` (90)** — the reserved holes in `wob.h` are filled. A fifth mechanism it did not anticipate came out of proving the gate: the shutdown drain guarantee, [40](language-runtime-database/40-shutdown-drain-guarantee.md). Supervision trees stay out of v1 |
| 24 | [chat: WebSocket workload](language-runtime-database/24-chat-websocket-workload.md) | ✅ **LANDED 2026-08-27** (absorbing 31 + 34) — all ten tasks; merged to master `ed5334d`. `just chat` **11 checks, 0 failures** at the full 1000-client soak: handshake, functional matrix on both `WO_IO` backends and on one shard, the soak, the fd invariant, the SIGTERM drain, `WO_MAILBOX=8` backpressure, ASan clean. Finishing its gate found a real runtime bug, split out as [40](language-runtime-database/40-shutdown-drain-guarantee.md) |
| 23 | [io_uring group-commit](databasev2/04-io-uring-commit.md) | ✅ **part A LANDED 2026-08-28 — group commit**, one barrier per drain instead of one per statement (the engine was fsync-per-STATEMENT, not per commit; the story's premise was wrong). Shard 0 holds each reply, commits once when its queue empties, releases all — so a writer is acked after the barrier carrying ITS record. **≈2.9× durable write throughput, ≈2.1× lower p50**, two measurement methods agreeing (2.9× controlled, 3.5× s1-vs-sN); mean batch 5.43, peak 57. A durability failure is now **fatal (exit 74), not a catchable `WO_T_IO`** — replacing three behaviours that disagreed, two of which admitted leaving RAM ahead of disk. **What it did NOT do:** `durable.sN.mixwrite` 480→492 (unchanged — that workload does 20 writes at C=4, mean batch 1.01) and `seed` unchanged (serial writers have nothing to batch with). **This row used to say "close the 66× gap"; that target was mis-stated** — the gap is two problems and part A fixes only the concurrent one. ⬜ part B (io_uring) **needs re-brainstorming**, not starting on the old premise |
| 32 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) | ✅ **LANDED 2026-08-29 — the chain's last link.** Compaction rewrites the log as one record per live row and swaps it in with `rename`, so **recovery is completely unchanged** and crash safety comes from the filesystem rather than from code. **2.16× space reclaimed** (1 962 358 → 907 094 B), **boot 114 → 64 ms**, stop-the-world pause **2 651 µs** against a stated 50 ms budget. Read `.dev/reference/postgresql` for it: PG *never* compacts its WAL — its records are page deltas, so it needs heap files, a control file, a redo pointer and a separate process. Ours are full row images, so a compacted log IS a store, which deletes all of that. `kill -9` during compaction: 40 rounds/run, 10 clean runs, and **mutation-proven** — against in-place rewrite instead of `rename` the battery fails every time. Outstanding: the **`resident: keys` offset map** (compaction moves every record; the obligation is recorded at the compactor) and the O(live rows) pause, ~5.5 s at 1 GB, which is what an incremental design must be bought against |
| 33 | [Single-file store](databasev2/07-single-file-db.md) | ⬜ off-chain, small — `WO_DATA=<path>.db` file form; driver-only (story written 2026-08-22) |
| 34 | [Crypto builtins](language-runtime-database/34-crypto-builtins.md) | 🔄 **code landed** as 24's T1 (`d14fa9f`): `sha1`/`sha256`/`hmac_sha256`, ids 85–87 in `wob.h`, `runtime/src/crypto.c`, RFC/FIPS vectors 18/0, corpus pin. The 24 gate that once needed it is cleared. Frontmatter keeps `status: refine` only until 24's T10 closeout sets it to `done` |
| 38 | [Content platform capabilities](language-runtime-database/38-content-platform-capabilities.md) | ⬜ off-chain, needs a spec — the two capability families no iteration owns, confirmed against `runtime/src/wob.h`: `fs` mutation verbs (six fs builtins, ids 40–45; `append` creates-if-absent, so nothing is ever replaced, truncated, deleted or renamed) and `net.connect` (ids 51–55 + 91–95, no connect, and no `connect()` anywhere in `runtime/src/` — so no OIDC/SMTP/object-store/webhook/federation). Driven by a `docs/examples/vault` content-collaboration workload, in 28's mould. New builtins from 96 (89/90 reserved for 31); no `.wob` bump (`WOB_VERSION 6u`, last moved by 36). Story written 2026-08-26 from the "can it build a Nextcloud?" ask |
| 39 | [Web framework parity](language-runtime-database/39-web-framework-parity.md) | ⬜ off-chain, needs a spec — from [the Fiber v3.5.0 study](../plan/exploration/fiber/00-fiber-parity.md) (all 32 of its middleware read against `porch`; **nine already have a counterpart**). Leads with a **random-bytes builtin**: the framework ledger claimed CSRF/sessions were unblocked by iteration 34's HMAC, but HMAC authenticates a token and cannot mint one — there is no RNG anywhere in the runtime. Then cookies (absent both ways; `Resp.headers` being a map cannot carry two `Set-Cookie` lines), then limiter/idempotency (cheapest wins — `@table` + `time.ticks`, nothing new), sessions, CSRF, and the routing/response sugar. Streaming/SSE/compression, `@derive` binding, TTL cache, `proxy` and metrics all excluded with owners named |
| 40 | [Shutdown drain guarantee](language-runtime-database/40-shutdown-drain-guarantee.md) | ✅ **LANDED 2026-08-27 — chain 3, with 31; split out of 24.** One rule: **a message sent before the stop flag is observed must be delivered and run before the engine stops.** Found by measurement, not review: making the chat gate's drain leg start its OWN (cold) server exposed that **5 of 16** fresh-server SIGTERM drains left a WebSocket client at EOF with no close frame and no diagnostic. Traced to `shard_main` — `NEXT_RUNNABLE()` already stated the contract ("a WORKER on stop keeps DRAINING … close frames!") but the IDLE branch reaped and broke, abandoning its inbox for teardown to free. An actor between messages is exactly that idle case, which is why a WARM soak server hid it for so long. Fix is one branch honouring the primary's drain window, yielding on an empty poll. **20 of 20 clean after**; `just chat` 11 checks 0 failures at the full 1000-client soak (which also settled the fd question: 1000 connections left the count at 44); runtime battery 36 suites 0 fail, compiler 556 checks 0 fail. Ruled out: a bigger spin (a 1 s wall-clock deadline still failed 2 of 12) and spawn-during-shutdown. Outstanding: a pin below the gate — nothing in `runtime/test/` drives the engine start/stop and no corpus fixture can trigger a stop |
| 37 | [wo-html components](language-runtime-database/37-wo-html-components.md) | ✅ off-chain — LANDED 2026-08-25. Raw text literal (backtick, margin stripped at lex time, `{{ }}` auto-escapes) + the component layer: `Component`/`render_all`/`Layout` in wo-html, `ok_html` moved into the framework, site and shop both migrated |
| 35 | [net runtime seams](language-runtime-database/35-net-runtime-seams.md) | ⬜ off-chain — fd deadlines on the park plane, Unix sockets, peer address; owns the ledger's three 🔧 rows (story written 2026-08-22) |
| 25 | [HTTP service layer](../superpowers/plans/2026-08-01-http-service-layer.md) | ⏸ hold (2026-08-21) — story file removed; the plan doc remains |
@ -461,13 +629,16 @@ that sequences its tasks. Read one, approve, then the next starts.
| Language | 🔄 [iteration 36 — operator parity](language-runtime-database/36-operator-parity.md): `not`, bitwise `& \| ^ << >>`, hex/binary/`_` literals, compound assigns — CODE LANDED 2026-08-22 (branch operator-parity, `.wob` v6, all gates green; reference project `.dev/reference/go` drove the design). Awaiting the developer's MANUAL pass on `docs/examples/operators/` (no test fixtures by directive); unblocks story 34's pure-`.wo` HMAC question | [plan](../superpowers/plans/2026-08-22-operator-parity.md) |
| Language | the framework v1-polish slice landed 2026-08-20 (branch framework-v1, awaiting merge); next per the order: brainstorm 20/21's forks | [order](#implementation-order-re-sequenced-2026-08-21--concurrency-chain) |
| Runtime | ✅ **iteration 35 landed 2026-08-23** (branch `framework-v1b`, with framework v1 slice 2 + the serving slice): net deadlines/unix/peer (ids 91–95), fiber pooling, serve_conn + web-app fiber-per-connection — web-app gate 41/0, both WO_IO backends | [design](../superpowers/specs/2026-08-23-net-seams-park-design.md) |
| Runtime | 🔄 **iteration 24 (absorbing 31 + 34): chat + actor lifecycle** — spec + plan approved 2026-08-23 (24 absorbs 31 by directive; 34 resolved C-builtins); executing on branch `chat-ws-lifecycle` | [marker](../active-slice-2026-08-23-chat-ws-lifecycle.md) · [plan](../superpowers/plans/2026-08-23-chat-ws-lifecycle.md) |
The active slice's marker doc is
[`docs/active-slice-2026-08-23-chat-ws-lifecycle.md`](../active-slice-2026-08-23-chat-ws-lifecycle.md)
— one file, deleted when the slice lands. Everything else pending is the
concurrency chain (see *Pending* below); the held tail is every story
whose frontmatter reads `status: hold`.
**No slice is active.** Iteration 24 landed 2026-08-27 and its marker doc was
deleted per the convention. Everything pending is the concurrency chain (see
*Pending* below) — **the chain's next link is
[databasev2 4](databasev2/04-io-uring-commit.md)** (chain 5, the io_uring
group-commit write path, `was_language_iteration: 23`), which now has iteration
22's fsync-per-commit numbers in hand, plus databasev2 1's finding that the
write path is *not* where memory pressure bites (appending under a cap costs
~1%, random reads 273×). The held tail is every story whose frontmatter reads
`status: hold`.
### Landed 2026-08-14 — the compile-and-run milestone
@ -705,8 +876,8 @@ the language arc as v1 history.
| --- | --- | --- |
| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | ✅ **MEASURED 2026-08-27** — `readiness: ready`, `status: done`; forks settled, harness landed (**148 checks**). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Also measured: **random reads over an oversized table collapse 273×** (1.85M vs 6 771 reads/s, p99 1 µs vs 487 µs) — so the two access patterns sit ~270× apart under the same pressure, and departure is a **step, not a curve**. Replay measured too: **≈5.5 µs/record, 1.9× history penalty** (10M records ≈ 55 s of boot) — iteration 3's missing "before", now gated. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from |
| 2 | [per-table storage: `durable` and `resident`](databasev2/02-table-storage-modes.md) | 🔄 **the language enrichment — the `durable` half is DONE and usable.** Two optional `@table` keys, `durable: true\|false` and `resident: all\|keys`, both defaulting to today's behaviour (all 28 existing declarations compile unchanged, no golden moved). Landed: the grammar, WO-E224 (a durable `ref` into a volatile table is refused), `.wob` v7 carrying both properties in spare `flags` bits, `durable: false` actually skipping the WAL (measured: 50 inserts → 1500 bytes durable, **0** volatile) with a mode-mismatch startup refusal, plus offset capture and read-a-row-from-an-offset. Outstanding: 5c/5d (the id→offset map and rewiring `wo_row_ptr`'s 11 call sites, slab scans and `@unique`/FK across the boundary — not yet written up), the two runtime refusals, and closeout. [spec](../superpowers/specs/2026-08-26-table-residency-design.md) · [plan](../superpowers/plans/2026-08-26-table-residency.md) |
| 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ⬜ snapshot + truncate: disk reclaimed, replay bounded |
| 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ⬜ **`readiness: ready` — the one startable iteration in the repo** (four forks confirmed settled 2026-08-20). Close the 66× gap iteration 22 measured (durable 4.5k vs ram 297k inserts/s) |
| 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ✅ **LANDED 2026-08-29 — the chain's last link.** Compaction rewrites the log as one record per live row and swaps it in with `rename`, so **recovery is completely unchanged** and crash safety comes from the filesystem rather than from code. **2.16× space reclaimed** (1 962 358 → 907 094 B), **boot 114 → 64 ms**, stop-the-world pause **2 651 µs** against a stated 50 ms budget. Read `.dev/reference/postgresql` for it: PG *never* compacts its WAL — its records are page deltas, so it needs heap files, a control file, a redo pointer and a separate process. Ours are full row images, so a compacted log IS a store, which deletes all of that. `kill -9` during compaction: 40 rounds/run, 10 clean runs, and **mutation-proven** — against in-place rewrite instead of `rename` the battery fails every time. Outstanding: the **`resident: keys` offset map** (compaction moves every record; the obligation is recorded at the compactor) and the O(live rows) pause, ~5.5 s at 1 GB, which is what an incremental design must be bought against |
| 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ✅ **part A LANDED 2026-08-28 — group commit**, one barrier per drain instead of one per statement (the engine was fsync-per-STATEMENT, not per commit; the story's premise was wrong). Shard 0 holds each reply, commits once when its queue empties, releases all — so a writer is acked after the barrier carrying ITS record. **≈2.9× durable write throughput, ≈2.1× lower p50**, two measurement methods agreeing (2.9× controlled, 3.5× s1-vs-sN); mean batch 5.43, peak 57. A durability failure is now **fatal (exit 74), not a catchable `WO_T_IO`** — replacing three behaviours that disagreed, two of which admitted leaving RAM ahead of disk. **What it did NOT do:** `durable.sN.mixwrite` 480→492 (unchanged — that workload does 20 writes at C=4, mean batch 1.01) and `seed` unchanged (serial writers have nothing to batch with). **This row used to say "close the 66× gap"; that target was mis-stated** — the gap is two problems and part A fixes only the concurrent one. ⬜ part B (io_uring) **needs re-brainstorming**, not starting on the old premise |
| 5 | [Bounded tables and eviction](databasev2/05-bounded-tables-eviction.md) | ⬜ a declared capacity + refuse/evict/back-pressure, and a process-level pressure signal that sheds **before** the allocator or OS gets involved — turning the invisible failure into a managed one |
| 6 | [Cold tiering](databasev2/06-cold-tiering.md) | ⚠ **largely superseded by 2** — `resident: keys` took the ceiling-raising role; its user-space-working-set premise was rejected for the kernel page cache. Mostly forks: which shape, whether the index itself fits, whether the *language* surfaces the fault cost, and whether `@unique` on a cold table is refused outright. A paged B-tree stays rejected — if tiering needs one, reject tiering |
| 7 | [Single-file store](databasev2/07-single-file-db.md) *(was 33)* | ⬜ `WO_DATA=<path>.db`; driver-only, independent |

View file

@ -2,8 +2,8 @@
track: databasev2
iteration: "3"
was_language_iteration: "32"
status: pending
readiness: refine
status: done
readiness: ready
chain: 6
---
@ -29,6 +29,113 @@ chain: 6
> replay/restart numbers to justify its policy and must compose with
> 23's group-commit write path.
> **BRAINSTORMED 2026-08-28.** Spec:
> [`2026-08-28-wal-checkpoint-design.md`](../../superpowers/specs/2026-08-28-wal-checkpoint-design.md)
> · plan: [`2026-08-28-wal-checkpoint.md`](../../superpowers/plans/2026-08-28-wal-checkpoint.md)
> (6 tasks).
> Read `.dev/reference/postgresql` for this — and the conclusion was that
> Postgres' design is *unavailable* to us, which is what makes the simpler one
> legitimate.
>
> **The design in one sentence:** compact the log by rewriting it as one record
> per live row into a temp file, then `rename` it over the live WAL. Recovery is
> **completely unchanged** — boot still opens one file and replays it — and the
> crash criterion is satisfied by the filesystem rather than by code we must get
> right.
>
> **Why one file works here and not in Postgres.** Postgres never compacts its
> WAL: its records are page deltas, so a compacted redo log is not a store, and
> it must keep heap files, a control file, a redo pointer and a second recovery
> source. Ours are **full row images** — `apply_record` implements UPDATE as
> remove-then-recreate — so a compacted log *is* a complete store. That one
> difference deletes the control file, the redo pointer, the cutoff offset and
> the separate process from the design.
>
> **Forks settled:** no snapshot format (the compacted log is the snapshot); one
> source, not two; **volume-only trigger** as a ratio against the last
> compaction's own measured output, with an absolute floor — **no timer**,
> because Postgres' timer exists to bound loss from unflushed buffers and we have
> none; stop-the-world, with the pause measured against a stated budget rather
> than assumed acceptable.
>
> **The coupling that would otherwise be found late:** compaction moves every
> record, so it **invalidates every WAL offset**
> [iteration 2](02-table-storage-modes.md)'s `resident: keys` stores. The
> compactor rebuilds the offset map as it writes. Recorded now because iteration
> 2's storage half is unimplemented, so nothing breaks today — it would break
> later, looking like corruption rather than a design gap.
>
> **Measured on master 2026-08-28, grounding the whole iteration:** `seed 20000`
> leaves a 986 614-byte log; 20 000 updates take it to **2 590 262 bytes with the
> same live rows** (2.6× history for no data), and boot+verify on that store is
> **155 ms**.
## Progress — landed 2026-08-29
| # | Task | State |
| --- | --- | --- |
| 1 | `wo_wal_compact` — rewrite, fsync, rename, fsync parent, reopen | ✅ `8ea510d` |
| 2 | a stale compaction temp is removed at open | ✅ `8bfbd4b` |
| 3 | the trigger (pure decision + env knobs) and the ordering guard | ✅ `6dbcb9a` |
| 4 | `kill -9` DURING compaction — 40 rounds, mutation-proven | ✅ `9b283d5` |
| 5 | measure space, boot and the stop-the-world pause | ✅ `d87f65a` |
| 6 | closeout | ✅ this change |
### Measured
| | checkpointing off | checkpointing on |
| --- | --- | --- |
| WAL used | 1 962 358 B | **907 094 B** |
| boot | 114 ms | **64 ms** |
**2.16× space reclaimed, 1.78× faster boot**, stop-the-world pause **2 651 µs**
against a stated 50 ms budget. Full details, including the pause's scaling, are
in [`perf-targets.md`](../../plan/perf-targets.md) §7.
### Two bugs the work found, both mine
**Wiring only the drain left `WO_SHARDS=1` never compacting** — its log grew
forever (536 KB where the multi-shard run held 446 KB), because a statement on
the owner shard never enters that drain. Both write paths now check.
**The dump was 8× slower than it needed to be**, flushing through the
committing path and so paying one `fdatasync` per 256 records for durability
that is worthless before the rename. One final barrier took the pause from
107 649 µs to 13 212 µs on a 2 MB live set — ~22 MB/s to ~181 MB/s.
## Acceptance Criteria
Met:
- **Given** an aged store, **when** it is compacted, **then** disk is reclaimed.
✅ 2.16× on the full campaign, asserted rather than merely recorded — the leg
fails if the log is not smaller with checkpointing on.
- **Given** the same store, **when** it boots, **then** replay is bounded by the
live set rather than by history. ✅ 114 → 64 ms.
- **Given** `kill -9` at ANY instant during a checkpoint, **when** the process
restarts, **then** recovery produces the same consistent store as if the
checkpoint had never started, with no acknowledged write lost. ✅ 40 rounds
per run, 10 consecutive clean runs, and **proven to have teeth**: against the
design's rejected alternative (in-place rewrite instead of `rename`) the
battery fails every run with the log destroyed.
- **Given** the iteration-22 replay numbers, **then** a before/after delta is
recorded. ✅ `perf-targets.md` §7.
- **Given** writes arriving while a checkpoint runs, **then** the ack contract
holds. ✅ compaction runs only where nothing is staged, asserted by a test
that stages and requires refusal; `wo_wal_compact` also refuses as a backstop.
Outstanding:
- **The `resident: keys` offset map.** Compaction moves every record, so it
invalidates every WAL offset [iteration 2](02-table-storage-modes.md) stores.
The compactor must rebuild that map as it writes. **Nothing fails today**
because iteration 2's storage half is unimplemented — which is exactly why the
obligation is written at the compactor in `wal.c`, where the next implementer
hits it, rather than only in a spec they may not read.
- **The pause is O(live rows).** At ~181 MB/s a 1 GB live set implies ~5.5 s,
past any interactive budget. Incremental or forked copying was deliberately
not bought in advance; this is the number to buy it against.
## Goals
- **Disk space is reclaimed.** A checkpoint writes the live store as a

View file

@ -2,7 +2,7 @@
track: databasev2
iteration: "4"
was_language_iteration: "23"
status: pending
status: in-progress
readiness: ready
chain: 5
---
@ -59,6 +59,111 @@ chain: 5
> batch — under io_uring it becomes exactly one submission, so the two
> features compose without either knowing the other.
> **BRAINSTORMED 2026-08-28 — and SPLIT IN TWO.** Spec for part A:
> [`2026-08-28-wal-group-commit-design.md`](../../superpowers/specs/2026-08-28-wal-group-commit-design.md)
> · plan: [`2026-08-28-wal-group-commit.md`](../../superpowers/plans/2026-08-28-wal-group-commit.md)
> (6 tasks).
>
> **The payoff metric is `durable.sN.mixwrite`, not the s1 numbers.** Worker
> shards hold no WAL — the runtime asserts it — so every statement on a worker
> marshals to shard 0 and parks, while a statement already on shard 0 runs
> inline. Batches form only where there is a queue, so concurrent multi-shard
> writes batch and a single-shard or serial workload does not. The baseline
> shows why that is the right target anyway: **multi-shard concurrent writes are
> 480 ops/s at p99 5888 µs against single-shard's 1023 at p99 664 — adding
> shards makes durable writing WORSE today**, because every marshaled statement
> still buys its own barrier on the owner.
>
> **The premise below needed correcting.** This story says "replace
> fsync-per-commit with io_uring group-commit", but the engine does not commit
> per commit — it commits per **statement**: `db.c` calls `wo_wal_commit`
> immediately after every append, at all six sites, so every row change is one
> `pwrite` plus one `fdatasync`. That splits the goal into two independent
> wins, and only the second needs io_uring:
>
> - **Part A — batching.** Let many statements share one barrier. The staging
> buffer already holds any number of records; today it never holds more than
> one because the caller commits immediately. Mostly a deletion of calls.
> - **Part B — async submission.** The shard submits and keeps working instead
> of blocking in `fdatasync`. Deferred until A's measurement says whether the
> blocking boundary is still the bottleneck.
>
> **A is where most of the number lives.** Iteration 22 measured durable writes
> at 4460 ops/s and mixed writes at 1023 ops/s (p99 664 µs) against 1.28M ops/s
> for durable reads — ~290× apart, essentially all of it the per-statement
> barrier.
>
> **Forks settled in the brainstorm:** batch boundary is **queue-drain** (not
> the tick this story recorded — a tick taxes an idle system to serve a busy
> one); a failure between "RAM mutated" and "record durable" is a **fatal,
> diagnosed abort**, replacing today's uneven rollback where `insert` undoes
> itself and `update`/`delete` admit in a comment that they leave RAM ahead of
> disk. **That removes `WO_T_IO` from the write path** — a language-visible
> change, recorded here deliberately.
>
> `status: in-progress` because the brainstorm is done and the spec is
> approved; the plan is next. (The `readiness` axis that would say this
> precisely lives on the unmerged `db-residency-doctrine`.)
## Progress — part A landed 2026-08-28
| # | Task | State |
| --- | --- | --- |
| 1 | a failed barrier is detected, and fatal | ✅ `d3ff03e` |
| 2 | one barrier per drain; replies held | ✅ `b9b8a45` |
| 3 | the inline path takes the fatal rule, asymmetry documented | ✅ `a6ccdbe` |
| 4 | prove batches form — the `wmix` write-concurrent leg | ✅ `40d029c` |
| 5 | measure the payoff, gate it, record it | ✅ `d52ea8a` |
| 6 | closeout | ✅ this change |
| — | **part B — io_uring submission** | ⬜ **not started; its premise changed, see below** |
### The payoff, measured two ways
| Measurement | Before | After |
| --- | --- | --- |
| controlled (same build, only `db.c`/`vm.c` swapped; `wmix 4000 32`) | 2213 · 2177 ops/s, p50 7183 · 7251 µs | **6216 · 6525 ops/s, p50 3458 · 3444 µs** |
| committed baseline: `s1` inline vs `sN` batched | 1467 ops/s, mean batch 1.0 | **5117 ops/s, mean batch 5.43, peak 57** |
**≈2.9× throughput, ≈2.1× lower p50**, and the two methods agree (2.9× and
3.5×). Batching scales with contention: mean batch **1.13 / 1.76 / 5.35** at
C = 4 / 16 / 64.
### The cost side, and a bug the battery caught
**Reads were being held behind the barrier.** The drain first held *every* DB
reply until the commit — including reads, which stage nothing. `mixread` p99 rose
from ~1043 µs to **4057 µs** until only staging statements had their replies
held. Caught by the gate, not by review.
**What remains is inherent:** a barrier blocks the owner shard longer (more
records per fsync) though less often, so anything queued behind one waits. Three
full runs of the same build gave `durable.sN.mixread.p99` of **1043 / 2318 /
4147 µs** — a 2–4× spread near idle. So part A buys ~3× write throughput at the
cost of a longer, noisier tail on the owner shard. `durable.sN.*.p99us` was
re-baselined at 100% tolerance for that reason, with the floor as the real guard
(`mixread`'s came within 25 µs of tripping).
**This is the strongest argument for part B** — submitting the barrier and
continuing to serve is exactly what removes this cost.
### What did NOT improve — and it was predicted
- **`durable.sN.mixwrite`: 480 → 492 ops/s, i.e. unchanged.** This was the
spec's *original* payoff metric, and correcting it was part of the brainstorm:
`mix` writes on one op in ten with C=4, so a quick run performs **20 writes**
and measured mean batch **1.01**. A workload that never has two writes in
flight cannot be helped by batching them.
- **`durable.*.seed`: unchanged.** A serial single writer has nothing to batch
with, under any scheme.
- **This board's stated target was mis-stated.** It read "close the 66× gap
iteration 22 measured (durable 4.5k vs ram 297k inserts/s)". Part A does not
close that gap and structurally cannot: `seed` is serial, and one writer
waiting on one barrier is a **latency** problem, not a batching one. Recorded
rather than quietly renumbered.
- **The before-p99 is not a measurement.** `hist_add` clamps at 20000 µs and
both before-runs pinned exactly there, so the true value is ≥20 ms and
unknown. The gain is *at least* 2.3×.
## Goals
- **Replace fsync-per-commit with io_uring group-commit** on the WAL write
@ -77,26 +182,50 @@ chain: 5
## Acceptance Criteria
- What to achieve?
- **Given** the io_uring write path under the iteration-22 crash battery
(concurrent writers, kill -9 mid-stream, reboot, replay),
- **when** it runs,
- **then** every acknowledged write is present after replay and no
unacknowledged partial write is ever visible — the exact result the
fsync path gives, so durability is provably unchanged.
- What to achieve?
- **Given** the iteration-22 durable write benchmark,
- **when** it is run on the fsync-per-commit path and then the io_uring
group-commit path on the same machine,
- **then** the io_uring path's write throughput is materially higher and
its p99 commit latency lower, with the before/after numbers recorded —
the payoff, measured, not asserted.
- What to achieve?
- **Given** a kernel without io_uring (old, or restricted by seccomp),
- **when** the runtime starts,
- **then** it falls back to the pwrite + fdatasync path automatically and
correctly — io_uring is an accelerator, never a hard dependency, and a
binary that runs everywhere is the whole project's premise.
Met:
- **Given** the io_uring write path under iteration 22's crash battery, **when**
it runs, **then** every acknowledged write is present after replay. ✅ — the
criterion applies unchanged to part A's batching. `crash.sN` (the batched
path) recovered every acked row after `kill -9`, `crash.s1` likewise, and both
restart legs replay byte-true. This was the one thing batching could break.
- **Given** the durable write benchmark before and after, **then** throughput is
materially higher and p99 lower, recorded. ✅ ~2.9× and ~2.1× (p50); see
`perf-targets.md` §6. **Scoped honestly:** on a write-concurrent workload
only, and p99's "before" is at the histogram ceiling.
- **Given** batching, **when** it runs, **then** it is proven to engage rather
than assumed. ✅ mean batch 5.43, peak 57 on the gated leg, and the live
assertion fails the suite if the mean drops to 1.
- **Given** a durability failure, **when** it happens, **then** the engine does
not continue with RAM ahead of disk. ✅ fatal, diagnosed, exit 74 — replacing
three behaviours that disagreed.
Outstanding:
- **Given** a kernel without io_uring, **when** the runtime starts, **then** it
falls back automatically. *(part B — part A adds no syscall interface, so
nothing to fall back from yet.)*
- **Single-shard concurrent batching.** A statement on shard 0 commits inline
and cannot batch; doing so needs the inline path to park its fiber on the
barrier — the same machinery part B needs. So `WO_SHARDS=1` gets no batching
at all, by design and measured (mean batch 1.0).
- **The abort path is not exercised.** Forcing a real `fdatasync` failure needs a
full or read-only filesystem, which the gate cannot arrange without mount
privileges. The unit test proves the error is *detected*; the exit three lines
later is covered by inspection. Disclosed rather than papered over — iteration
40 was exactly a fatal path nothing exercised.
## Part B — its premise changed
Part B was justified by "close the 66× durable gap". Part A shows that framing
was wrong: the gap is **two** problems. Concurrent write fan-in was a batching
problem and is now ~3× better. What remains is a **serial** writer waiting on a
single barrier, which no amount of batching can help — and io_uring does not
obviously help it either, since one writer still needs one durable barrier
before its ack. Part B's real candidates are overlapping the barrier with other
work on the shard, and the inline-path park that single-shard batching also
needs. **It should be re-brainstormed against that, not started on the old
premise.**
## Out Of Scope

View file

@ -1,6 +1,6 @@
---
iteration: "24"
status: in-progress
status: done
readiness: ready
chain: 4
---
@ -20,6 +20,34 @@ chain: 4
> bounded mailboxes, actor death, timers). Iteration 19 LANDED
> 2026-08-20, so Bytes is available for frame parse/serialize.
> **✅ LANDED 2026-08-27** (branch `chat-ws-lifecycle`, merged to master
> `ed5334d`). Ten tasks: crypto (T1), bounded mailboxes (T2), `call`/reply and
> actor death (T3), `monitor` (T4), `time.after` (T5), the WS upgrade seam
> (T6), the pure-`.wo` frame codec (T7), the chat sample (T8), the gate (T9),
> this closeout (T10). It absorbed [31](31-actor-lifecycle.md) and
> [34](34-crypto-builtins.md), which land with it.
>
> **Gate — `just chat`, 11 checks, 0 failures** at the full 1000-client soak:
> handshake with an independently recomputed accept-key, the functional matrix
> (presence, broadcast, room isolation, leave) on **both** `WO_IO` backends and
> on a single shard, the 1k hot-room soak, the fd invariant, the SIGTERM drain,
> `WO_MAILBOX=8` backpressure, and an ASan run with zero leaks. Battery
> alongside: runtime 36 suites 0 fail, compiler 556 checks, corpus 119 checks.
> The sample logs to `/tmp/chat.log`.
>
> **Two disclosed deviations from the spec.** `monitor` takes **three**
> arguments (`watched, observer, msg`) rather than two, because the caller may
> be `main`, which has no mailbox and cannot be an implicit observer. And a
> `call` reply is a **typed scalar** in v1 — which is what let the agreement be
> checked at compile time (WO-E226) instead of carried as a tagged value.
>
> **What finishing the gate found.** Making every leg start its own server
> exposed a real runtime bug the warmed soak server had been hiding: on a fresh
> server, 5 of 16 SIGTERM drains left a client at EOF with no close frame. It
> was not this sample's fault — the fix is an engine guarantee, split out as
> [40](40-shutdown-drain-guarantee.md). Design notes:
> [`docs/examples/chat/CODE-LOGIC.md`](../../examples/chat/CODE-LOGIC.md).
## Why this iteration exists
Everything the framework ledger parks behind concurrency — WebSockets,

View file

@ -1,6 +1,6 @@
---
iteration: "31"
status: in-progress
status: done
readiness: ready
chain: 3
---
@ -17,6 +17,22 @@ chain: 3
> ([iteration 24](24-chat-websocket-workload.md)) cannot be written
> honestly without these four mechanisms.
> **✅ LANDED 2026-08-27 — INSIDE [24](24-chat-websocket-workload.md)**, per
> the 2026-08-23 directive that absorbed it. All four mechanisms shipped:
> `call`/reply with a typed scalar reply (id 88, WO-E226), **bounded mailboxes**
> (`WO_MAILBOX`, default 1024, fail-fast with a catchable `WO_T_ACTOR`),
> **actor death** that traps callers instead of hanging them, `monitor`
> (id 89) and `time.after` (id 90). Ids 89 and 90 were reserved holes in
> `wob.h`; they are filled.
>
> **A fifth mechanism was added that this story did not anticipate**: the
> shutdown drain guarantee, [40](40-shutdown-drain-guarantee.md). It is
> lifecycle semantics — this story gave actors a death notice, 40 gives the
> program a shutdown that does not lose mail — and it was found by measurement
> while proving 24's gate, not by review.
>
> How each piece works: `runtime/src/CODE-LOGIC.md`, "Actor lifecycle".
## Why this iteration exists
The arc's stages 1+2 shipped `spawn`/`send` mechanism without lifecycle:

View file

@ -1,6 +1,6 @@
---
iteration: "34"
status: in-progress
status: done
readiness: ready
---
@ -19,6 +19,18 @@ readiness: ready
> Off the concurrency chain but **gates chain position 4**: iteration
> 24's WebSocket handshake needs SHA-1 before chat can land.
> **✅ LANDED 2026-08-27 — inside [24](24-chat-websocket-workload.md)** as its
> task 1. The fork resolved to **C builtins**: `sha1` (85), `sha256` (86),
> `hmac_sha256` (87), each over one buffer returning a fresh `Bytes`. Pinned to
> the published vectors — RFC 3174, the SHA-256 vectors, RFC 4231 — in
> `runtime/test/test_crypto.c`, 18 checks, plus a corpus fixture hashing "abc"
> from `.wo`. This unblocked chain position 4: the WebSocket handshake needs
> SHA-1, and `just chat` verifies the accept-key independently.
>
> **The gap it did NOT close:** there is still no RNG in the runtime. HMAC
> authenticates a token and cannot mint one, so CSRF and sessions stay blocked
> — which is why [39](39-web-framework-parity.md) leads with a random-bytes
> builtin rather than treating them as unblocked.
## Why this iteration exists
Four consumers already wait on it, none able to proceed:

View file

@ -0,0 +1,158 @@
---
iteration: "40"
status: done
chain: 3
---
# iteration 40 — the shutdown drain guarantee: a send before the stop flag is delivered
> Part of [Story — one language, one runtime, one database, one binary](00-story.md).
>
> **Split out of [24](24-chat-websocket-workload.md) on 2026-08-27** because it
> is a runtime *semantic*, not a task in a sample's gate. It belongs to the
> actor lifecycle ([31](31-actor-lifecycle.md), absorbed into 24) and it is
> the half of "lifecycle" that nothing had stated: 31 gave actors a death
> notice, this gives the program a shutdown that does not lose mail.
>
> **Found by measurement, not review.** The chat gate's drain leg had been
> passing only because it drained a server the 1k soak had already warmed.
> Making every leg start its own server exposed it:
> [`2026-08-27-chat-drain-finding.md`](../../2026-08-27-chat-drain-finding.md).
## The rule
**A message sent before the stop flag is observed must be delivered and run
before the engine stops.** One sentence, and it is the whole iteration. It is a
guarantee, not a tuning parameter — which is why a spin count could never
express it.
What it does *not* promise: that a message sent *after* the flag is delivered,
that a parked fiber is resumed, or that an actor gets unbounded time. The drain
window is the primary's, and it closes when the primary returns.
## The bug, as measured
Fresh server, two WebSocket clients, `SIGTERM`, both must receive a close frame:
| Sample | Result |
| --- | --- |
| 5 fresh servers | 1 failure (`eof\|close`) |
| 12 fresh servers | 3 failures, one `eof\|eof` |
| 16 fresh servers | 5 failures |
The failing client's socket reaches EOF with **no close frame and no
diagnostic** — the process exits and the kernel closes the fd.
Traced with instrumentation on the sample's actors: `main` → Registry → Room →
Writer. The Registry runs and sees its room. The **Room never processes the
shutdown message**, so the Writer's close branch never runs. Clients that did
get a frame were saved by their own Reader noticing `env.stopping()`, not by the
room broadcast.
## The design, as built
`runtime/src/vm.c` already encoded the correct contract in `NEXT_RUNNABLE()`:
a worker that takes a stop while it has a live fiber returns 2 and **keeps
draining its inbox** until the primary sets `eng_shutdown`. Its comment says so
in as many words — "queued shutdown messages (close frames!) still run".
`shard_main`'s own idle branch contradicted it. A worker with an empty run queue
waits in `wo_io_wait`, and on `WO_IO_STOP` it called `fib_reap_all` and
**broke** — abandoning whatever was still in its inbox, which `wo_engine_stop`
then freed wholesale during teardown.
So the failure needed a shard that was *idle* at `SIGTERM`. A Room actor between
messages is exactly that, which is why the warm soak server hid it: warm shards
had live fibers and took the correct path.
The fix makes the idle branch obey the same contract: while the primary's drain
window is open, an idle worker adopts its inbox and runs what arrives, yielding
between empty polls so a drain cannot become a hot spin across every core. Only
`eng_shutdown` — set by the primary after `main` returns — ends it.
One branch, in one place, matching a contract the file already stated.
## Progress
| Piece | State |
| --- | --- |
| the idle-worker drain branch in `shard_main` (`runtime/src/vm.c`) | ✅ one branch, matching the contract `NEXT_RUNNABLE()` already stated |
| `sched_yield` on an empty poll so the drain cannot hot-spin | ✅ |
| fresh-server drain, repeated | ✅ **20 of 20**, from 5-in-16 failing |
| chat gate at the default 1k soak | ✅ **11 checks, 0 failures** — 1000/1000 clients, both `WO_IO` backends, ASan clean |
| full runtime battery (this touches every actor program's shard loop) | ✅ **36 suites** (18 × both dispatch flavors), 0 fail, `cli_smoke: OK`; compiler 556 checks 0 fail |
| the regression pin | ✅ the chat gate's drain leg, now that it starts its OWN (cold) server — that decoupling is what caught this. **Not** a corpus fixture or unit test: nothing in `runtime/test/` drives `wo_engine_start`/`wo_engine_stop` today, and no corpus fixture can trigger a stop, so pinning it below the gate means new multithreaded test infrastructure — named as its own cost, not smuggled in here |
**Measured 2026-08-27.** Before: 5 of 16 fresh-server drains left a client at
EOF. After: **20 of 20 clean.** At the observed failure rate, 20 clean runs by
luck would be about 0.04%, so this is the fix rather than a quieter race.
## Acceptance Criteria
Met:
- **Given** a fresh server with two connected WebSocket clients, **when** it is
sent `SIGTERM`, **then** both clients receive a close frame — **repeatedly**,
not once. The bug reproduced at 5 in 16, so a single green run proves nothing;
the criterion is a run of at least 16 with zero failures.
✅ **20 of 20**, from 5-in-16 failing. A single run would have proved nothing.
- **Given** an actor whose shard is idle at the moment of the stop, **when** a
message is sent to it before the stop flag is observed, **then** its
`receive` runs before the engine stops. ✅ this is exactly the case that
failed — the Room between messages — and it is what the branch now covers.
- **Given** the drain window, **when** a worker has nothing to adopt, **then**
it does not hot-spin. ✅ `sched_yield()` on an empty poll; the 1k soak's RSS
and timing legs are unchanged (marker reached all 1000 in 28 ms).
- **Given** `just chat`, **when** it runs at the default soak, **then** all
legs pass on both `WO_IO` backends and under the ASan build with zero leaks.
✅ 11 checks, 0 failures. The fd leg also settled the lazy-init question at
scale: **1000 connections left the count at 44**, unchanged after 20 more.
- **Given** the full runtime battery, **when** it runs, **then** no suite
regresses — this touches the shard loop every actor program uses. ✅ 36 suites
0 fail, plus the compiler's 556 checks.
- **Given** a program with no worker shards (`WO_SHARDS=1`), **when** it stops,
**then** behaviour is unchanged. ✅ the gate's `WO_SHARDS=1` leg passes, and
the branch is unreachable there — `wo_engine_stop` returns early at
`nshards <= 1`, so a single-shard program never enters a worker loop.
Outstanding:
- **A pin below the gate.** The guarantee is currently proven by the chat gate
only. Nothing in `runtime/test/` drives `wo_engine_start`/`wo_engine_stop`,
and no corpus fixture can trigger a stop, so pinning it lower means new
multithreaded test infrastructure. Named as its own cost rather than assumed
cheap.
## Out Of Scope
- **Unbounded drain.** The window is the primary's and closes when `main`
returns. A program that wants longer holds the window open itself.
- **Delivering sends issued *after* the stop flag.** Nothing promises that, and
promising it would mean a program could refuse to exit.
- **Resuming parked fibers on stop.** `WO_SYS_STOPPED` unwinds them; that
contract is iteration 24's and stays.
- **A shutdown acknowledgement in the language surface.** The alternative fix
was a barrier the sample builds itself, rejected below.
- **`main` parking after the stop flag.** Still forbidden — a park after the
flag unwinds. `main` still spins; the point is that spinning now works
because the workers cooperate.
## Info — the forks, settled
1. **Engine guarantee, not a sample barrier.** The alternative was an
acknowledged drain: rooms confirm back to `main`, which waits. Rejected —
`main` cannot park after the stop flag, so it could only spin on the
acknowledgement anyway, and every future actor program would have to
re-implement the same handshake to avoid losing mail. A guarantee is stated
once; a barrier is re-invented per program.
2. **Not the spin budget.** Replacing the sample's `spin < 20000000` with a 1 s
wall-clock deadline still failed 2 of 12. More time cannot help when the
shard is not scheduled at all, and the reverted attempt cost a fixed second
on every shutdown. Recorded because a bigger spin is the obvious wrong fix.
3. **Not `dummy_writer()`.** Hoisting the shutdown message's placeholder actor
out of the drain path (it spawned during shutdown) left 5 of 16 failing.
4. **Yield rather than spin in the idle drain.** A worker polling an empty
inbox in a tight loop would burn a core per shard during the window and
starve the actors being drained.
5. **Chain position 3**, with [31](31-actor-lifecycle.md): it is lifecycle
semantics, and [24](24-chat-websocket-workload.md)'s gate is what proves it.

View file

@ -0,0 +1,265 @@
# databasev2 3 — WAL checkpoint (implementation plan)
> **For agentic workers:** REQUIRED SUB-SKILL: Use
> superpowers:subagent-driven-development (recommended) or
> superpowers:executing-plans to implement this plan task-by-task. Steps
> use checkbox (`- [ ]`) syntax for tracking.
>
> **Style rule (user convention):** concept, reason, and required
> behaviour in words plus verification commands only — no implementation
> or test code blocks; the executor writes the code.
**Goal:** reclaim disk and bound replay by rewriting the log as one record per
live row and swapping it in with `rename`, so boot replays a short log instead
of all history.
**Architecture:** compaction writes the live store into a temporary file using
the existing record grammar and the existing append path, fsyncs it, renames it
over the live WAL, fsyncs the parent directory, and reopens the descriptor.
Recovery is untouched — boot still opens one file and replays it — and every
crash point is safe because `rename` is atomic.
**Tech Stack:** C11, libc only. `pwrite`, `fdatasync`, `rename`, `open`,
`unlink`. No new dependency and no new file format.
**Spec:** [`../specs/2026-08-28-wal-checkpoint-design.md`](../specs/2026-08-28-wal-checkpoint-design.md)
## Global Constraints
- **Recovery must not change.** No second source, no cutoff offset, no control
file. If a task finds itself editing the replay path, something has gone
wrong with the design and it should stop rather than proceed.
- **Every crash point falls back.** Before the rename the live log is untouched;
after it the new log is complete. There must be no window in which a reader
could observe a mixture.
- **The record grammar is frozen.** The whole argument for this design is that
it already suffices. A compacted log is INSERT records for live rows, ids
preserved exactly.
- **Bounded memory.** `stage()` grows the staging buffer by doubling and never
shrinks it, so dumping a whole store through one buffer would hold the entire
store in RAM — the unbounded growth databasev2 1 identified as how this engine
dies. The dump must flush periodically.
- **Compaction may run only where nothing is staged** — in practice immediately
after a barrier. Anywhere else, a staged record lands in a file about to be
replaced.
- **libc only**, no new syscall interface. Gates run through `just`. Never
commit on `master`; branch first.
---
## Task 1 — `wo_wal_compact`: rewrite, fsync, rename, reopen
**Files:**
- Modify: `database/src/wal.c`, `database/src/wal.h`.
- Test: `runtime/test/test_wal.c`.
**Interfaces:**
- Produces: a compaction entry point taking the live WAL and the store, which
replaces the log with one INSERT record per live row and leaves the WAL usable
(descriptor reopened, offset correct). Returns success or failure; a failure
must leave the ORIGINAL log intact and usable, because a failed checkpoint is
not a durability event.
- Consumes: the existing append path and commit routine, and the bitmap walk
that `db.c` already performs in three places.
- [ ] Read three things first and confirm them, because the design rests on
them: `apply_record` implements UPDATE as remove-then-recreate (so records are
full row images), `wo_wal_append_insert` takes an id and reads the row from
the store (so ids are preserved), and the tail scan treats a zero length field
as end-of-log (so the new file must be zero beyond its records).
- [ ] Test first, RED: build a store, age it (insert rows, then update the same
rows repeatedly so history exceeds live data), compact, then assert **both**
that the log got materially shorter AND that a fresh replay of it produces the
same rows with the same ids and the same values. Shorter alone is worthless —
a truncating bug also passes that.
- [ ] Verify RED for the right reason: the entry point does not exist yet.
- [ ] Implement the walk: for each class, iterate slots via the bitmap and
append one INSERT per live row. Reuse the append path; do not write a second
encoder.
- [ ] **Flush every K records rather than staging the whole store.** Point a
scratch WAL at the temp descriptor and commit periodically. State the chosen K
and why in a comment. Without this the dump holds the entire store in RAM.
- [ ] Sequence the switch exactly: fsync the temp file, `rename` over the live
path, **fsync the parent directory** (the rename is atomic in-kernel but the
directory entry is not durable until the parent is synced), then reopen the
descriptor — the old one refers to an unlinked inode — and reset the offset to
the new end of log.
- [ ] Handle failure without losing data: any error before the rename must
unlink the temp file and leave the live log untouched. A failed compaction is
a missed optimisation, **not** a durability failure, so it must NOT take the
fatal path databasev2 4 introduced.
- [ ] GREEN: `just wovm-test`.
- [ ] Commit.
## Task 2 — a stale temp file is removed, never read
**Files:**
- Modify: `database/src/wal.c` (the open path).
- Test: `runtime/test/test_wal.c`.
**Interfaces:**
- Consumes: Task 1's temp-file naming.
- Produces: the guarantee that a crash mid-rewrite leaves nothing that can be
mistaken for data.
- [ ] Test first, RED: place a temp file next to the log containing *plausible,
well-formed records* (not garbage — garbage would be rejected anyway and would
prove nothing), open the store, and assert the temp file is gone and the
replayed store is exactly what the live log said.
- [ ] Verify RED for the right reason.
- [ ] Remove any stale temp file when the WAL is opened. Note in a comment why
this is safe: the only way one exists is a crash before a rename, and its
contents are by definition not yet authoritative.
- [ ] GREEN: `just wovm-test`.
- [ ] Commit.
## Task 3 — the trigger, and the ordering guard
**Files:**
- Modify: `database/src/wal.c`, `database/src/wal.h` (remember the last
compaction's size; the policy decision), `runtime/src/vm.c` (call the check
after the barrier).
- Test: `runtime/test/test_wal.c`.
**Interfaces:**
- Consumes: Task 1's compaction entry point.
- Produces: automatic compaction, and the invariant that it never runs with
records staged.
- [ ] Extract the policy as a **pure decision** — given the log's used bytes,
the bytes the last compaction wrote, and a floor, should we compact? Pure
because it is then unit-testable without a store, which is the only way this
policy gets tested at all.
- [ ] Test the decision directly, RED then GREEN: below the floor it never
fires however bad the ratio; above the floor it fires exactly when used bytes
exceed the multiple; with no prior compaction it uses the floor alone.
- [ ] Record the bytes each compaction wrote, so the denominator is measured
rather than estimated. Estimating the live size would mean estimating Text,
and the compactor already knows the true number.
- [ ] Expose the floor and the ratio as env knobs, matching the existing idiom
(`WO_MAILBOX`, `WO_HEAP_MB`, `WO_SHARDS`, `WO_WAL_STATS`). **This is what
makes the policy testable** — a test sets a tiny floor and forces compaction
in a few writes instead of waiting for megabytes. Document them beside the
others. **Deviation from the spec, disclosed:** the spec spoke of a "manual
trigger for tests"; env-tunable thresholds serve that purpose without adding
language surface, which is the cheaper way to buy the same testability.
- [ ] **No timer.** If the implementer is tempted, the reason is in the spec:
Postgres' `CheckPointTimeout` bounds loss from unflushed buffers, our records
are durable at commit, and an idle log does not grow.
- [ ] Call the check from the one place that is safe — immediately after the
drain's barrier, where nothing is staged. Comment that this is a correctness
requirement and not a scheduling preference.
- [ ] Verify the guard: a test that stages records and then makes the policy
say yes must find compaction deferred, not executed. This is the assertion
that keeps the ordering rule true as the code moves.
- [ ] Verify durability is unaffected: `just db-bench --quick` — the crash and
restart legs must be unchanged, and part A's `wmix` legs must still batch.
- [ ] Commit.
## Task 4 — kill -9 *during* compaction
**Files:**
- Test: `runtime/test/test_wal.c` (extend the existing fork-based crash
battery).
**Interfaces:**
- Consumes: Tasks 1–3.
- Produces: the evidence for the criterion the whole design is shaped around.
- [ ] Read the existing crash battery first: a forked child inserts and acks
each committed id over a pipe while the parent SIGKILLs it mid-stream, then
the parent verifies every acked id survived. Extend that shape rather than
inventing a second harness.
- [ ] Drive compaction repeatedly in the child (a tiny floor makes it fire
often) while it inserts and acks, and kill at many instants so the kill lands
inside a rewrite, at the rename, and after it.
- [ ] Assert the property, not a state: after replay the store must equal
**either** the pre-compaction **or** the post-compaction content — never a
mixture — and **every acked id must be present**. A test that only checks "it
replayed without error" would pass on a silently truncated log.
- [ ] Assert no temp file survives a kill in a way that affects the next boot.
- [ ] Run the battery repeatedly, not once: this is a race, and one green run
proves very little. State how many repetitions were run in the commit message.
- [ ] GREEN: `just wovm-test` plus the repetitions.
- [ ] Commit.
## Task 5 — measure: space, boot, and the pause
**Files:**
- Modify: `scripts/db-bench.py` (a checkpoint leg), `docs/plan/perf-targets.md`,
`bench/baseline.json` (refresh, with the reason in the commit message).
**Interfaces:**
- Consumes: Tasks 1–3.
- Produces: the before/after record, and the pause number the spec deliberately
refused to assume.
- [ ] Capture the before numbers already measured on master, rather than
re-deriving them: `seed 20000` leaves a 986 614-byte log; 20 000 updates take
it to 2 590 262 bytes **with the same live rows**; boot+verify on that aged
store is 155 ms.
- [ ] Add a leg that ages a store, compacts it, and records: bytes before and
after, the ratio reclaimed, and boot time before and after. Age it by
updating the same rows — history must grow while the live set does not, or the
leg is measuring insert throughput instead of compaction.
- [ ] Measure the **stop-the-world pause** on the largest store the harness
builds and record it as a number. State the budget it must meet.
- [ ] **If the pause exceeds the budget, stop and report it.** That is the
finding the spec asked for, and the alternatives (incremental copy,
fork-and-dump) are bought against this number — not before it.
- [ ] Give the new metrics tolerances that match what they are: bytes reclaimed
is structural and can be gated tightly; the pause is wall-clock on a shared
box and cannot. Do not waive them all, which is the mistake part A's task 4
made and had to undo.
- [ ] Verify the gate bites: doctor the reclaimed-bytes metric and confirm the
suite fails on exactly that metric.
- [ ] Refresh the baseline and confirm the **full** campaign passes against it.
The committed baseline is full-mode (`N=20000`, `crash_reps=3`) — writing a
quick-mode baseline over it is a regression, and part A made exactly that
mistake.
- [ ] Commit.
## Task 6 — closeout
**Files:**
- Modify: `docs/stories/databasev2/03-wal-checkpoint.md`,
`docs/stories/00-status.md`, `docs/plan/oop-vm/04-db-binding.md`,
`database/src/CODE-LOGIC.md`, `docs/examples/db-bench/README.md`.
- [ ] `04-db-binding.md`: the normative ordering rule — compaction runs only
where nothing is staged, and what recovery does (unchanged: one file, replayed
from byte 0). This is the doc the spec named for it.
- [ ] `CODE-LOGIC.md`: why one file rather than snapshot-plus-tail, why
`rename` is the crash-safety primitive, why the dump flushes periodically, and
why a failed compaction is not a durability event. Reasoning, not call graph.
- [ ] README: the new env knobs beside the existing ones, and the checkpoint
leg.
- [ ] Story: progress, criteria split met/outstanding, and the measured
before/after.
- [ ] Board: standup entry in the six-question shape, and the chain note —
chain 6 was the last link, so say what the chain's completion means and what
is next.
- [ ] **Record the `resident: keys` obligation prominently, in the story and at
the compactor.** Compaction moves every record, so it invalidates every WAL
offset iteration 2 stores; the compactor must rebuild that map as it writes.
There is nothing to implement today because iteration 2's storage half does
not exist — which is exactly why this must be written where the next
implementer will hit it, not left in a spec they may not read.
- [ ] Full battery: `just wovm-test`, `just woc-test`, `just oop-e2e`,
`just db-bench`, `python3 scripts/linkcheck.py .`
- [ ] Commit.
## Self-review notes
- **Spec coverage.** Compaction and the switch → Task 1. Stale temp → Task 2.
Trigger, no timer, ordering rule → Task 3. Crash safety → Task 4. Space, boot,
pause → Task 5. Normative doc, `resident: keys` obligation → Task 6.
- **The riskiest task is 4**, not 1: Task 1's correctness is a single replay
comparison, while Task 4 is a race and can pass by luck. Hence the explicit
instruction to run it repeatedly and to state the count.
- **Task 2 looks trivial and is not.** A stale temp file containing well-formed
records is the one input that could be mistaken for data, so the test uses
plausible records rather than garbage.
- **One thing deliberately NOT a task:** rebuilding the `resident: keys` offset
map. It cannot be implemented against a feature that does not exist yet.
Recorded as an obligation in Task 6 instead of a stub nobody can test.

View file

@ -0,0 +1,249 @@
# databasev2 4 part A — WAL group commit (implementation plan)
> **For agentic workers:** REQUIRED SUB-SKILL: Use
> superpowers:subagent-driven-development (recommended) or
> superpowers:executing-plans to implement this plan task-by-task. Steps
> use checkbox (`- [ ]`) syntax for tracking.
>
> **Style rule (user convention):** concept, reason, and required
> behaviour in words plus verification commands only — no implementation
> or test code blocks; the executor writes the code.
**Goal:** one durability barrier per drain instead of one per statement, so a
writer is acknowledged after the barrier that carried its record rather than
after a barrier of its own.
**Architecture:** the barrier moves up, not out. Applying to RAM and staging the
record stay exactly where they are in `db.c`; the request path stops committing
after each append and instead holds its reply envelope, and shard 0 issues one
commit when it runs out of queued requests, then releases every held reply. Any
failure between "RAM mutated" and "record durable" ends the process with a
diagnostic.
**Tech Stack:** C11, libc only. `pwrite` + `fdatasync` (unchanged — io_uring is
part B). The existing per-shard envelope inbox carries the requests.
**Spec:** [`../specs/2026-08-28-wal-group-commit-design.md`](../specs/2026-08-28-wal-group-commit-design.md)
## Global Constraints
- **Durability is unchanged.** Every guarantee iterations 9 and 22 proved holds
identically: replay-whole-or-not-at-all, torn-tail drop, no acknowledged
write ever lost. This changes when the barrier runs, never what the log holds.
- **A writer is released only after the barrier carrying its record.** Never
before, and never on the strength of a different batch's barrier.
- **libc only.** No new dependency, no new syscall interface in part A.
- **The payoff metric is `durable.sN.mixwrite`** (today 480 ops/s, p99
5888 µs). `durable.s1.*` and both `seed` legs are regression guards, not
targets — a serial writer and an all-inline shard have nothing to batch with.
- **`WO_T_IO` leaves the write path.** A commit or staging failure is fatal, not
catchable. Exit 1 is a trap and exit 2 is a refusal, so this takes a third
status of its own.
- Gates run through `just`. Never commit on `master`; branch first.
---
## Task 1 — a failed barrier is detected, and fatal
**Files:**
- Modify: `database/src/wal.c` (the commit routine's failure returns; a new
fatal-commit entry point beside it), `database/src/wal.h` (declare it).
- Test: `runtime/test/test_wal.c` (a new case in the existing suite).
**Interfaces:**
- Produces: a commit entry point that takes the WAL and the number of records
in the batch, commits, and on failure writes one stderr line naming the
failing operation, the `errno` text, the WAL path and the record count, then
exits with the durability-failure status. Tasks 2 and 3 call only this.
- Consumes: the existing staging buffer and commit routine.
- [ ] Read the commit routine first and confirm what it already reports: it
loops `pwrite` until the staged buffer is written, then `fdatasync`, and
returns non-zero on either failing. Confirm the WAL struct carries its path,
or add it — the diagnostic is worthless without it.
- [ ] Test first, RED: assert the commit routine reports failure when the
descriptor is unusable (a closed descriptor gives `EBADF`). This proves the
error is *detected*; it does not exercise the exit.
- [ ] Verify RED for the right reason — the case must fail because the
assertion is unmet, not because the suite does not compile.
- [ ] Add the fatal entry point. It must distinguish the two operations in its
message: a `pwrite` failure and an `fdatasync` failure are different
operational problems and the operator needs to know which.
- [ ] GREEN: `just wovm-test`. The new case passes and no existing case moves.
- [ ] **Disclosed gap, record it in the commit message:** the exit path itself
is not exercised. Forcing a real `fdatasync` failure needs a full or
read-only filesystem, which the gate cannot arrange without mount
privileges. Do NOT add a fault-injection switch to buy coverage — shipping a
binary that can be told to kill itself is the worse trade, and the spec
rejected it.
- [ ] Commit.
## Task 2 — the barrier moves to the drain point; replies are held
**Files:**
- Modify: `database/src/db.c` (the request-path arms only — the three commit
calls inside the marshaled-statement executor), `runtime/src/vm.c` (the
envelope drain loop's DB-statement branch and the end of that loop).
- Test: no new fixture; the existing durability battery is the test. It already
covers exactly what could break.
**Interfaces:**
- Consumes: Task 1's fatal commit entry point.
- Produces: the invariant later tasks measure — at most one barrier per drain,
and every held reply released only after it.
- [ ] Read the drain loop's DB-statement branch first. Today it executes the
request, marks it done, then immediately pushes a reply envelope that unparks
the requester. Note that it runs on shard 0's thread, serialized — that is
why no locking is needed anywhere in this task.
- [ ] Remove the three commit calls from the request-path executor in `db.c`.
Leave applying to RAM and staging untouched, and leave the **inline** path's
three commit calls alone — Task 3 owns that path and conflating them is how
this change breaks the single-shard configuration.
- [ ] In the drain loop, collect reply envelopes in a local list instead of
pushing them as each request finishes. A local is correct and deliberate:
nothing needs to survive the loop, and per-shard state would outlive the
batch it describes.
- [ ] At the end of the drain loop, if anything was staged, call Task 1's fatal
commit once, then push every held reply.
- [ ] Handle the empty case: a drain that executed no DB statements must not
commit and must not touch the staging buffer.
- [ ] Verify the ack contract has not moved: `just wovm-test` — the WAL and
table suites must be unchanged, since neither knows about batching.
- [ ] Verify durability end to end: `just db-bench --quick`. The restart-replay
and `kill -9` crash legs are the ones that matter — a kill between staging and
the barrier must lose only unacknowledged writes. **If a crash leg fails here,
stop; do not adjust the test.** That leg failing means the ack contract broke,
which is the one thing this task may not do.
- [ ] Commit.
## Task 3 — the inline path keeps its own barrier, and says why
**Files:**
- Modify: `database/src/db.c` (the inline path's three commit calls — replace
with Task 1's fatal entry point), plus the comment above them.
**Interfaces:**
- Consumes: Task 1's fatal commit entry point.
- Produces: nothing new. This task exists to make the asymmetry deliberate and
legible rather than accidental.
- [ ] Replace the inline path's three commit calls with Task 1's fatal entry
point, batch size one. Behaviour is unchanged — this is the fatal-failure
rule reaching the second path, not batching.
- [ ] Write the comment that explains the asymmetry, because the next reader
will otherwise "fix" it: the inline path cannot hold a reply, because it
returns into its own fiber rather than unparking a requester. Batching it
would require parking that fiber on the barrier, which is part B's machinery
and deliberately out of part A.
- [ ] Confirm the ordering assumption holds: because the drain loop always
commits before it ends, nothing uncommitted is ever left staged when an
inline statement runs. If that stops being true the inline path would commit
another statement's record early — say so in the comment as the reason the
drain must commit unconditionally.
- [ ] Verify: `just wovm-test` and `just db-bench --quick` both green, and
`WO_SHARDS=1` in particular — the single-shard configuration takes this path
exclusively.
- [ ] Commit.
## Task 4 — prove batches actually form
**Files:**
- Modify: `scripts/db-bench.py` (new metrics and their tolerances),
`docs/examples/db-bench/main.wo` only if the batch figures cannot be observed
without the sample reporting them.
- Test: the driver's own gate-bites check.
**Interfaces:**
- Consumes: the batching from Task 2.
- Produces: mean batch size, peak batch size and peak staged bytes as recorded
metrics, so Task 5 measures a mechanism that is known to engage.
- [ ] Decide where the counters live and prefer the smallest surface: the
runtime can report them at exit, or the driver can derive them. Do not add a
builtin for this — the numbers are diagnostic, not part of the language.
- [ ] Record mean and peak batch size under the concurrent multi-shard write
workload. **This is the task's real point:** if batches are always one, the
feature is inert and any throughput change came from somewhere else, so the
measurement in Task 5 would be attributing a win to the wrong cause.
- [ ] Record peak staged bytes. This settles whether the batch needs a cap with
a number instead of a guess — the spec deliberately shipped no cap because the
request queue is already bounded upstream by iteration 24's mailbox caps.
- [ ] Give the new metrics wide tolerances. Batch size is a function of arrival
timing, so gating it tightly would gate the scheduler; what must be gated is
that it is greater than one under contention.
- [ ] Verify the gate bites: doctor the recorded mean batch size to one and
confirm the suite fails on exactly that metric.
- [ ] Commit.
## Task 5 — measure the payoff, gate it, write it down
**Files:**
- Modify: `bench/baseline.json` (refresh, with the reason in the commit
message), `docs/plan/perf-targets.md` (a new section).
**Interfaces:**
- Consumes: Tasks 2 and 4.
- Produces: the before/after record every later optimization argues against.
- [ ] Capture the before numbers from the committed baseline rather than
re-measuring them: `durable.sN.mixwrite` 480 ops/s, p50 538 µs, p99 5888 µs;
`durable.s1.mixwrite` 1023 ops/s, p99 664 µs; `seed` ~4460 ops/s on both.
- [ ] Run the full campaign, not the quick one, and record after numbers for
the same metrics on the same machine. A payoff measured across machines is
not a payoff.
- [ ] Assert the scoped criterion: **`durable.sN.mixwrite` throughput up and
p99 down**, with `durable.s1.*` and both `seed` legs not regressed. Do not
report the s1 seed number as a disappointment — a serial writer has nothing
to batch with, and the spec says so.
- [ ] Write the `perf-targets.md` section: the before/after table, the mean and
peak batch size that produced it, and the peak staged bytes. State the
inversion that motivated the work — multi-shard concurrent writes were 2×
slower than single-shard with a 9× worse p99 — and whether it is now gone.
- [ ] If the payoff is absent or small, **say so and stop.** That is a finding,
not a failure: it would mean the barrier was not the bottleneck the baseline
implied, and part B must not be started on an unproven premise.
- [ ] Refresh the baseline and confirm `just db-bench` passes against it, then
re-confirm the gate bites on a doctored write metric.
- [ ] Commit.
## Task 6 — closeout
**Files:**
- Modify: `docs/stories/databasev2/04-io-uring-commit.md` (progress, criteria
split met/outstanding, the landing banner),
`docs/stories/00-status.md` (standup entry, chain note),
`docs/plan/oop-vm/01-error-catalog.md` (the `WO_T_IO` removal and the new
exit status), `database/src/CODE-LOGIC.md` (a group-commit section).
- [ ] Story: record what landed and what did not. The outstanding items are
single-shard concurrent batching (needs the inline park) and part B itself.
Keep the corrected premise visible — this iteration was written as
"fsync-per-commit" and the engine was fsync-per-statement.
- [ ] Error catalogue: `WO_T_IO` no longer reachable from a write, and the new
durability-failure exit status documented beside the trap and refusal codes.
A language-visible removal that is not written down is a trap for the next
reader.
- [ ] `CODE-LOGIC.md`: the commit path as built — where the barrier runs, why
replies are held, why the inline path is asymmetric, and the one rule for
failure. Explain the reasoning, not the call graph.
- [ ] Board: the standup entry in the six-question shape, and the chain note —
part B's go/no-go now rests on Task 5's number.
- [ ] Full battery after the doc edits: `just wovm-test`, `just woc-test`,
`just oop-e2e`, `just db-bench`, `python3 scripts/linkcheck.py .`
- [ ] Commit.
## Self-review notes
- **Spec coverage.** Queue-drain boundary → Task 2. Fatal failure rule → Tasks 1
and 3. Held replies and the ack contract → Task 2. No batch cap, settled by
measurement → Task 4. Payoff and its scoping → Task 5. `WO_T_IO` removal →
Task 6. The disclosed abort-coverage gap → Task 1's last step.
- **The riskiest task is 2**, and its risk is concentrated in one place: the
crash legs of the durability battery. That is why the plan says stop rather
than adjust if they fail.
- **Task 3 looks like a no-op and is not.** Without it the inline path keeps a
catchable `WO_T_IO` while the request path aborts, which is precisely the
per-path unevenness this spec exists to remove.
- **Task 4 before Task 5 is deliberate.** Measuring a payoff before proving the
mechanism engages is how a win gets attributed to the wrong cause.

View file

@ -0,0 +1,196 @@
# WAL checkpoint — design
> databasev2 [3](../../stories/databasev2/03-wal-checkpoint.md), chain 6.
> Brainstormed and approved 2026-08-28, after
> [databasev2 4 part A](2026-08-28-wal-group-commit-design.md) landed.
>
> **One sentence:** compact the log by rewriting it as one record per live row
> into a temporary file, then `rename` it over the live WAL — so recovery is
> unchanged and crash safety comes from the filesystem.
## Decisions taken (the brainstorm's forks, settled)
| Fork | Decision |
| --- | --- |
| Snapshot format | **None.** The compacted log *is* the snapshot, in the existing record grammar |
| One source or two | **One.** Rewrite + atomic `rename`; boot logic is untouched |
| Trigger | **Volume only**, as a ratio against the last compaction's own size, with an absolute floor. **No timer** — see below |
| Write availability | **Stop-the-world**, measured against a stated budget rather than assumed acceptable |
| Composition with group commit | Compaction runs only where **nothing is staged** — immediately after a barrier |
| `resident: keys` (iteration 2) | Compaction **rebuilds the offset map** as it writes. It cannot be left to discover this later |
## Why one file, and why Postgres cannot do it
Postgres was read for this (`.dev/reference/postgresql`), and the conclusion is
that its design is *unavailable* to us — which is what makes the simpler option
legitimate rather than lazy.
| | PostgreSQL | writeonce |
| --- | --- | --- |
| Where data lives | heap/data files; the WAL is a redo tail | **the WAL is the only durable form**, replayed into RAM |
| WAL contents | page deltas and full-page images | **full row images** — `apply_record` implements UPDATE as remove-then-recreate |
| Compaction | **never**; segments before the redo point are recycled by `rename` or unlinked | possible, because a log of row images *is* a complete store |
| Bounded replay | recovery starts at the redo LSN in the control file | recovery starts at byte 0 of a *shorter* log |
| Crash safety of the switch | control file written in place, full block, torn writes caught by **CRC32C** (`update_controlfile`) | one `rename` |
| Trigger | `CheckPointTimeout` (300 s) **or** WAL volume (`XLogCheckpointNeeded`) | volume only |
| Pause | none; flush is spread over time in a **separate process** | stop-the-world |
Postgres cannot compact its WAL because a compacted redo log is not a store —
its records describe changes to pages that live elsewhere. Ours describe whole
rows, so the compacted log needs no companion. That single difference removes
the control file, the redo pointer, the second recovery source, and the separate
process from our design.
**What is worth porting is not the architecture but the ordering discipline:**
publish the new "recovery starts here" atomically and *last*, so a crash at any
instant falls back to the previous state with nothing to undo. Postgres achieves
that with a redo pointer computed at checkpoint *start* and a control file
updated at the *end*. We achieve the same property with `rename`, in one
syscall, because we can swap the entire data set atomically and Postgres cannot.
**Correction to a prior exploration doc.**
`docs/plan/exploration/postgresql/buffer-and-checkpoint.md` states that Postgres
updates its control file by rename ("the same in `BasicOpenFile` +
`fsync_parent_path`"). It does not — `update_controlfile` opens the existing file
`O_WRONLY`, writes a zero-padded full block in place, and relies on CRC32C to
detect a torn write. That doc also assumes writeonce has **segment files**
("records before that LSN are *known* to be in the segment files"), which it
does not and, per databasev2 2, deliberately will not. The doc predates the
databasev2 direction and should be annotated rather than followed.
## The design
### Compaction
Run on the owner shard, which owns the WAL. Walk each class's live rows — the
bitmap-over-slabs walk that three call sites in `db.c` already perform — and
append one INSERT record per live row to a **new** file, using the existing
append path. No new encoder, no new decoder, no format.
Then: fsync the new file, `rename` it over the live path, fsync the parent
directory (the rename's atomicity is in-kernel; the directory entry is not
durable until the parent is synced — Postgres does the same, and the existing
exploration doc is right about *this* part), and reopen the WAL descriptor,
because the old one now refers to an unlinked inode.
**Every crash point is safe without any recovery logic of ours.** Before the
rename, the live WAL is untouched and the temp file is garbage. After it, the new
log is complete by construction. There is no window in which a reader could see a
mixture, so the acceptance criterion — "recovery produces the same consistent
store as if the checkpoint had never started" — is satisfied by `rename`, not by
code we must get right.
Two obligations follow. Boot must **unlink a stale temp file** if one is present,
because a crash mid-rewrite leaves one behind and it must never be mistaken for
data. And the temp file must be zero-padded beyond its records exactly as the
live WAL is, because the tail scan identifies the end of the log by a zero
length field.
### When it runs, and where in the sequence
**The point matters more than the policy.** The drain stages records into one
buffer and commits them together; compaction rewrites the file those records
would land in. So compaction may run **only when nothing is staged** — in
practice, immediately after a barrier, before the next statement is served.
Anywhere else and a staged record would either be written to a file about to be
replaced, or be lost with it. This is the normative ordering rule that
[`04-db-binding.md`](../../plan/oop-vm/04-db-binding.md) must carry.
**Trigger: volume, as a self-tuning ratio.** Compact when the WAL's used bytes
exceed a multiple of the bytes the *last* compaction wrote, with an absolute
floor so a small store never bothers. The denominator is known exactly — the
compactor wrote it — so this needs no estimate of the live set's size, which is
not cheaply knowable when rows hold Text. The floor exists because a store whose
whole log is a few hundred kilobytes has nothing to reclaim.
**No timer, and that is a deliberate difference from Postgres.** Postgres needs
`CheckPointTimeout` because its dirty buffers are not durable until flushed — an
idle-but-dirty system must still checkpoint or it loses data. Our records are
already durable at commit; a checkpoint reclaims space and shortens boot and
nothing else. An idle system's log does not grow, so a timer would fire with
nothing to do. Adding one would be copying Postgres' mechanism without its
reason.
A manual trigger exists for tests, because a policy that can only be observed by
waiting is a policy that cannot be tested.
### The pause, and how it is judged
Compaction is stop-the-world: the owner shard rewrites the log as one long
operation while no statement is served. This is the simplest correct thing, and
part A's own experience argues for measuring before buying complexity to avoid
it. The dump is O(live rows) encodings plus one write and one barrier, so the
expectation is that it is fast — but an expectation is not a measurement, and
the proof plan below states the budget it must meet.
If the measured pause exceeds the budget, **that is a finding and a follow-up,
not something this iteration solves by adding concurrency.** The alternative
designs (incremental copy, fork-and-dump) cost exactly what Postgres pays, and
should only be bought against a number.
### The interaction that will otherwise be discovered late
**Compaction invalidates every stored WAL offset.** Rewriting the log moves every
record, so any offset captured from the old file is meaningless afterwards — not
stale-but-readable, but pointing at an arbitrary byte of a different file.
[Iteration 2](../../stories/databasev2/02-table-storage-modes.md)'s
`resident: keys` stores exactly such offsets, one per row, and reads rows back
through them.
The compactor therefore **rebuilds the offset map as it writes**: it is emitting
the new records and knows each one's new position, so this is the cheap
direction and the only one that keeps both features usable together. The
alternative — forbidding compaction while any `resident: keys` table is live —
would mean the feature that exists to handle huge tables is incompatible with
the feature that stops their log growing forever.
This is recorded here because iteration 2's storage half is not yet
implemented, so nothing will fail today. It will fail later, in a way that looks
like data corruption rather than a design gap.
## Proof plan
| Claim | How it is proven |
| --- | --- |
| Space is reclaimed | An aged store shrinks. Measured today on master: `seed 20000` gives a 986 614-byte log; 20 000 updates take it to 2 590 262 bytes with **the same live rows**. Compaction must return it to approximately the former |
| Replay is bounded | Boot time on the aged store before and after, recorded. Measured today: boot+verify on that store is 155 ms |
| Crash safety | `kill -9` at many instants *during* compaction, then replay: the store must equal either the pre-compaction or post-compaction state, never a mixture, and no acked write may be missing. This is the criterion the whole design is shaped around, so it gets the crash battery's treatment rather than one case |
| A stale temp file is harmless | Boot with one present, containing plausible records: it is removed and never read |
| The pause is known | The stop-the-world pause measured on the largest store the harness builds, recorded as a number with a stated budget — not asserted to be acceptable |
| The ordering rule holds | Compaction with records staged must be impossible by construction; a test that stages and then requests compaction must find it deferred, not executed |
| Nothing regressed | The full battery, and specifically part A's `wmix` legs: compaction must not change the ack contract or the batching it introduced |
## Out of scope
- **A second file, a control file, or a redo pointer.** The Postgres shape,
priced above and not needed once the log is self-sufficient.
- **Avoiding the pause.** Incremental or forked dumps are bought against a
measurement, not in advance.
- **Per-shard compaction policy.** One owner shard owns the WAL today; when that
changes, this decision is revisited with it.
- **Compacting away tombstones across shards, or any cross-shard coordination.**
There is one log.
- **io_uring for the rewrite** — part B of databasev2 4, whose premise is
already under revision.
- **Changing the record grammar.** The entire argument for this design is that
the grammar already suffices.
## Alternatives rejected
**Snapshot + WAL tail (the Postgres shape).** Rejected because it buys write
availability at the cost of a second recovery source, a cutoff offset, a control
file with its own torn-write detection, and a crash-safety guarantee that
depends on our ordering rather than on `rename`. Postgres pays this because its
log cannot stand alone; ours can.
**Compacting in place.** Rejected outright: there is no crash point at which a
partially rewritten live log is recoverable, and it trades the one property that
makes this design defensible for nothing.
**A timer trigger.** Rejected with a reason rather than on taste: Postgres' timer
exists to bound data loss from unflushed buffers, and we have no unflushed
buffers. An idle log does not grow.
**A ratio against an estimated live-set size.** Rejected in favour of the last
compaction's measured output, because estimating the live size means estimating
Text, and the compactor already knows the true number.

View file

@ -0,0 +1,208 @@
# WAL group commit — design
> databasev2 [4](../../stories/databasev2/04-io-uring-commit.md), part A.
> Brainstormed and approved 2026-08-28.
>
> **This spec covers batching only.** The iteration was split during the
> brainstorm: part A amortises one durability barrier across many statements,
> part B (io_uring submission) is deferred until A's measurement says whether
> the blocking boundary is still the bottleneck. That split matches the
> iteration's own fork 1 — "drop-in behind `wo_wal_commit` first, an async
> variant only if the scheduler proves the blocking boundary is the
> bottleneck" — and it means the throughput win arrives behind a much smaller
> correctness surface.
## Decisions taken (the brainstorm's forks, settled)
| Fork | Decision |
| --- | --- |
| Scope | **Batching first, io_uring later.** Two independent wins were being carried as one; only the first needs a new syscall interface, and it is where most of the number lives |
| Batch boundary | **Queue-drain.** Shard 0 stages every pending write request, then commits once. No timer, no tunable |
| Failure | **Fatal, diagnosed abort.** Any failure between "RAM mutated" and "record durable" ends the process |
| Batch cap | **None initially.** Measure peak staged bytes; add a cap only if the queue's existing upstream bound proves insufficient |
| Abort coverage | Unit-test the failure *return*; the abort path itself stays covered by inspection, and that gap is disclosed |
## The problem, read off the engine
The story says "replace fsync-per-commit with io_uring group-commit". Read
against the code, the premise needed correcting: the engine does not commit per
*commit*, it commits per **statement**. `db.c` calls `wo_wal_commit`
immediately after every append, at all six sites — insert, update and remove,
each on both the inline and the DB-actor path. Every single row change is one
`pwrite` plus one `fdatasync`.
That is what the numbers say too. Iteration 22's baseline records durable writes
at **4460 ops/s** single-shard and mixed writes at **1023 ops/s**, p50 **430 µs**,
p99 **664 µs** — against **1.28M ops/s** for durable reads. Writes are roughly
290× slower than reads, and the barrier is the whole of it.
**The batching machinery already exists and is simply never used.**
`wo_wal_commit` writes `w->buf` for `w->len` bytes — a staged buffer that can
hold any number of records. Today it never holds more than one, because the
caller commits immediately after staging. So part A is closer to removing calls
than to adding a mechanism.
## The design
### The commit path
The six `wo_wal_commit` calls come out of `db.c`. Applying to RAM and staging
the record stay exactly where they are; only the barrier moves, up to the point
where shard 0 runs out of work.
Shard 0 owns the WAL — DB statements from other shards arrive as marshaled
request envelopes and are executed on shard 0's thread, serialized, and a reply
envelope unparks the requester. The change is that **the reply is held rather
than sent**: shard 0 executes and stages each queued request, keeps draining
while requests remain, then issues one barrier, and only then releases every
held reply.
Each requester therefore unparks having been acknowledged after the barrier that
carried *its* record — the ack contract the story states, which today is true
only because every batch has one member.
A statement executing inline on shard 0 (rather than arriving as a request)
stages and commits before returning, as it does now. It has no reply to hold —
it returns into its own fiber — and because the drain always commits before it
ends, nothing uncommitted is ever left staged when the inline path runs.
### Why queue-drain, and what it costs
The batch boundary is the queue going empty, not a tick and not a timer. Two
properties follow, and they are the reason to prefer it:
- **A lone writer pays nothing.** One queued request means a batch of one, which
is today's path at today's latency. Batching engages only under genuine
contention, so an idle system is not taxed to serve a busy one.
- **The batch self-tunes.** Its size is whatever actually accumulated between
drains, so it grows with load rather than with a configured number. There is
nothing to set and nothing to set wrong.
The rejected alternative was the iteration's recorded leaning, the shard tick.
That leaning was recorded when the batch was assumed to ride an io_uring
submission; with batching landing first, a tick boundary would add up to one
quantum of latency even to a lone writer — paying the cost of batching when
there is nothing to batch with.
**No batch cap ships initially, and that is a decision rather than an
oversight.** databasev2 1 established that unbounded growth is precisely how
this engine dies without warning, so the instinct to bound it is right. But the
request queue is already bounded upstream by iteration 24's mailbox caps, and a
second bound on the same quantity is a knob that can only be wrong. The proof
plan measures peak staged bytes so the question is settled by a number.
## Failure: one rule, replacing three behaviours
Today's rollback is uneven, and the code says so. An `insert` whose commit fails
removes the row again, under a comment claiming RAM never claims what disk has
not acknowledged. An `update` or a `delete` whose commit fails does **not** roll
back — its comment admits the state plainly: RAM ahead of disk, trap, do not
ack. Nothing acknowledged is lost, but the process continues with divergent
state, and batching would multiply that from one row to as many as the batch
held.
The rule that replaces it: **once a statement has mutated RAM, the only outcomes
are durable or process death.** It covers both failure points identically —
a staging failure and a barrier failure have the same consequence, RAM ahead of
disk with no way back, and only one of the three verbs can undo itself.
Retrying is not an alternative worth designing for. On Linux a failed `fsync`
may already have discarded the dirty pages, so a second call can report success
having written nothing; the recovery that actually works is replay, which
returns exactly the last durable state. That is what the log is for.
**This removes `WO_T_IO` from the write path.** A program can no longer catch a
disk failure on a write. The removal is deliberate — there was never a
recovery a program could meaningfully perform with its RAM ahead of its disk —
but it is language-visible and must be stated in the story banner and the error
catalogue, not slipped in.
The diagnostic has to earn the abort: the failing operation, the `errno` text,
the WAL path, and the number of records in the batch, on stderr, then exit with
a status of its own. Exit 1 is a trap and exit 2 is a refusal, so a durability
failure takes a third. `abort()` is rejected — a core dump on a full disk is
noise, not evidence.
## What will improve, and what will not
**Corrected 2026-08-28, after reading the baseline properly.** The spec first
pointed at `durable.s1.seed` as the payoff metric. That was wrong, and the
reason is structural rather than a matter of degree.
Worker shards hold no WAL at all — the runtime asserts it — so every DB
statement on a worker marshals to shard 0 and parks, while a statement already
on shard 0 executes inline. **A queue of write requests therefore exists only
when other shards are writing.** Batches form where there is a queue:
| Workload | Today | Batching |
| --- | --- | --- |
| `durable.sN.mixwrite` — concurrent writers across shards | **480 ops/s, p99 5888 µs** | **the target.** N shards marshal N writes and shard 0 pays N barriers serially; one barrier replaces them |
| `durable.s1.mixwrite` — concurrent writers, one shard | 1023 ops/s, p99 664 µs | **no change.** Every write is inline with no queue, so no batch forms |
| `durable.*.seed` — one serial writer | ~4460 ops/s | **no change**, under any batching scheme. There is nothing to batch with |
The inversion in those numbers is the finding worth keeping: **multi-shard
concurrent writes are currently 2× slower than single-shard with a 9× worse
p99.** Adding shards makes durable writing worse today, because every marshaled
statement still buys its own barrier on the owner. That is the pathology group
commit exists to remove, and it is a better argument for this iteration than the
one the story recorded.
**Single-shard concurrent batching is deliberately out of part A.** It would
need the inline path to park its fiber on the barrier rather than commit
synchronously — the same parking machinery part B needs anyway. Deferring it
keeps A to one mechanism, and B inherits the reason to build it.
So the acceptance criterion is scoped: **`durable.sN.mixwrite` throughput up and
its p99 down; `durable.s1.*` and both `seed` legs must not regress.** A plan
that reported "no improvement" against the s1 seed number would be measuring a
workload this change cannot help.
## Proof plan
| Claim | How it is proven |
| --- | --- |
| The payoff is real | **`durable.sN.mixwrite`** before and after on one machine, recorded in `perf-targets.md`. Today 480 ops/s, p99 5888 µs. `durable.s1.*` and both `seed` legs are regression guards, not targets — see the section above |
| Durability is unchanged | Iteration 22's crash battery, unaltered: concurrent writers, `kill -9` mid-stream, replay. **The critical test** — a kill between staging and the barrier must lose only unacknowledged writes |
| Batches actually form | New metrics for mean and peak batch size under contention. If batches are always one, the feature is inert and any throughput change came from somewhere else |
| No idle tax | Single-writer p99 must not regress against the current baseline |
| The cap question is answered | Peak staged bytes recorded per run |
| A failure is detected | `test_wal.c` asserts `wo_wal_commit` reports failure on a bad descriptor |
**One disclosed gap.** Forcing a genuine `fdatasync` failure needs a full or
read-only filesystem, which the gate cannot arrange without mount privileges.
The unit test proves the error is *detected*; the abort that follows it stays
covered by inspection. The alternative — a fault-injection switch — means
shipping a binary that can be told to kill itself, which is a worse trade. This
gap is recorded rather than hidden, because iteration 40 was exactly a fatal
path that nothing exercised.
## Out of scope
- **io_uring submission.** Part B, and it only earns its complexity if A's
measurement shows the blocking boundary still dominating. A's parking and ack
machinery is what B would build on, so nothing here is wasted either way.
- **`transaction { }`** — language iteration 18. A transaction already *is* a
staged batch, so the two compose without either knowing about the other; that
is a reason not to entangle them now.
- **Checkpoint and compaction** — databasev2 3. This changes when the barrier
runs, never what the log contains.
- **The read path.** databasev2 1 measured that appending under memory pressure
costs about 1% while random reads cost 273×, so the pressure is on reads —
but that is iteration 2's `resident: keys` question, not this one.
- **Rollback with pre-images.** Rejected above: it would add per-write cost on
every statement to serve a path that ends the process anyway.
## Alternatives rejected
**Tick-boundary batching** — the iteration's recorded leaning, superseded by
the split. It taxes an idle system to serve a busy one.
**Count-or-timer batching** — two tunables, and the timer reintroduces the tick
problem with extra configuration.
**Full rollback with an undo log** — keeps `WO_T_IO` catchable, at the price of
capturing pre-images for every update and delete, paid on every write, to
support continuing in a state the engine cannot trust.
**Keeping today's per-verb behaviour** — turns a rare one-row divergence into a
routine N-row one, silently.

View file

@ -60,6 +60,13 @@ site:
fibers:
./scripts/fibers-accept.sh
# chat: iteration 24's gate (docs/examples/chat) — rooms/presence/broadcast
# over WebSocket via actors: functional on both WO_IO backends, the
# 1k-clients-one-hot-room soak (fds/RSS accounted), SIGTERM drain with
# close frames, and an ASan leg. `just chat` runs it (CHAT_SOAK=N trims).
chat:
./scripts/chat-accept.sh
# db-actor: arc stage 3's gate (docs/examples/db-actor) — worker-shard
# actors read/write the database through the transparent DB actor; WAL
# replay pair included. `just db-actor` runs it.

View file

@ -257,3 +257,100 @@ layout.
- **`listen_unix` sets O_NONBLOCK on the listener itself** — accept4's
SOCK_NONBLOCK flags the ACCEPTED socket only; a blocking listener
would block the whole shard (found by the seam probe, both backends).
## Actor lifecycle: call, death, monitor, timers (iteration 24, ids 88–90)
Four pieces that together answer "what happens to an actor that is waiting,
that dies, that watches, or that wants to be woken later". All four live in
`vm.c` with their entry points in `builtin.c`; the structures are in `vm.h`.
**`call` (id 88) — a send that waits.** An ordinary `send` returns immediately;
`call` parks the calling fiber and resumes it with the receive's return value.
The reply is a **typed scalar**, which is what let the agreement be checked at
compile time (WO-E226) rather than carried as a tagged value at runtime. The
caller is never left hanging: if the callee dies mid-call, or the address is
already dead, the caller **traps catchably** instead of parking forever. That
is the property worth keeping in mind when reading the code — every path out of
a call either resumes the fiber or traps it.
**Death.** A `receive` that traps uncaught marks the actor dead on its home
thread. From then on sends to it drop silently, calls trap, queued callers are
error-unparked, and its state and mailbox are released. Silent-drop for sends
is deliberate: a sender cannot handle another actor's failure, and making every
`send` fallible would put a `try` on every line.
**The mailbox cap and its counter.** One cap for every mailbox (default 1024,
`WO_MAILBOX` overrides at boot; the chat gate shrinks it to 8 to force the
policy). `pending` counts sent-but-not-delivered. It is incremented by the
**sender**, on any shard, and decremented by the **home thread** at delivery —
so it is touched only through `wo_mbox_reserve`/`wo_mbox_release` and their
`__atomic` builtins. The consequence is disclosed rather than hidden: the cap
can overshoot by at most the number of in-flight sends. Overflow is fail-fast —
the send raises a catchable `WO_T_ACTOR` (trap 13), which is what lets a room
drop a slow member instead of growing without bound.
**`monitor` (id 89) — the death notice.** `wo_monitor` is one registration:
observer, the moved-in notice message, next. The list lives on the **watched**
actor and is owned by its home thread, so the death walk needs no lock — dying
is a home-thread event and the list is right there. The notice is the
observer's own M-typed message, so an observer receives death notices in the
same shape as everything else. Monitoring an already-dead actor fires
immediately rather than silently doing nothing. An observer whose mailbox is
full loses the notice, with a disclosed stderr line — the alternative was
blocking a death walk on a slow observer.
It takes **three arguments** (`watched, observer, msg`), not the two the spec
first proposed, because the caller may be `main`, which has no mailbox and so
cannot be an implicit observer.
**`time.after` (id 90) — one-shot, no cancel.** `wo_timer` is `at` (wall ms),
target, message, next. The list lives on the **arming fiber's shard** and is
scanned by the same deadline machinery that already serves fd-park deadlines,
so timers cost no new wait mechanism. Firing is an ordinary runtime send, which
means it inherits the ordinary rules: a full target drops with a stderr line, a
dead target drops silently. There is no cancel; the idiom is a generation
counter in the message, which the `timer-generation` corpus fixture pins.
**Where to look when a lifecycle thing misbehaves:** `wo_vm_actor_monitor` and
`wo_vm_timer_after` in `vm.c` are the two entry points; `shard_main` and
`NEXT_RUNNABLE()` decide when a shard runs, adopts, or stops. The corpus
fixtures `monitor-death`, `timer-delivery` and `timer-generation` are the
smallest working examples of each.
## The shutdown drain guarantee (iteration 40)
**A message sent before the stop flag is observed is delivered and run before
the engine stops.** Stated because it was once untrue in a way nothing caught.
`wo_engine_stop` sets `eng_shutdown`, wakes every worker, joins them, and only
then tears down — freeing whatever envelopes are still queued. So a worker that
leaves its loop early takes its inbox with it. `NEXT_RUNNABLE()` has always
encoded the right behaviour for a worker holding a live fiber: on a stop it
returns 2 and keeps draining, because "only the PRIMARY's stop ends the
program". `shard_main`'s **idle** branch did the opposite — it reaped and broke
— so a shard whose actors happened to be between messages at `SIGTERM`
abandoned everything still in flight.
It now honours the same contract: while the primary's window is open an idle
worker adopts its inbox and runs what arrives, `sched_yield`ing on an empty
poll so a drain cannot burn a core per shard and starve the actors it exists to
let run. Only `eng_shutdown` — which the primary sets after `main` returns —
ends it.
Two things follow that are easy to get wrong. The window is the **primary's**,
so a program that wants a longer drain holds it open itself; `main` cannot park
after the stop flag, because a park there unwinds. And the whole path is
unreachable at `WO_SHARDS=1`, where `wo_engine_stop` returns at `nshards <= 1`.
## Digests: sha1, sha256, hmac_sha256 (iteration 34, ids 85–87)
`crypto.c` holds SHA-1 and SHA-256 over a single buffer and HMAC-SHA-256 on top
of the latter, each returning a fresh `Bytes`. No streaming API and no other
primitives — these exist because WebSocket's handshake needs SHA-1 and ETags
need SHA-256, and that is the whole of the demand so far.
Correctness is pinned to the published vectors rather than to itself:
RFC 3174 for SHA-1, the FIPS/RFC 6234 vectors for SHA-256, RFC 4231 for HMAC,
in `runtime/test/test_crypto.c` (18 checks). **There is still no RNG anywhere
in the runtime** — HMAC authenticates a token but cannot mint one, which is why
iteration 39 leads with a random-bytes builtin.

View file

@ -200,6 +200,18 @@ int wo_builtin(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
}
case WO_B_CALL: /* iteration 24: park/reply protocol lives in vm.c */
return wo_vm_actor_call(vm, R, ins, msg);
case WO_B_MONITOR: {
int rc = wo_vm_actor_monitor(vm, R[B], R[B + 1], R[B + 2], msg);
if (rc) return rc;
R[A] = 0;
return 0;
}
case WO_B_TIME_AFTER: {
int rc = wo_vm_timer_after(vm, (int64_t)R[B], R[B + 1], R[B + 2], msg);
if (rc) return rc;
R[A] = 0;
return 0;
}
case WO_B_NOW: { /* wall-clock milliseconds */
struct timespec ts;
clock_gettime(CLOCK_REALTIME, &ts);

View file

@ -183,6 +183,17 @@ int wo_load_buf(wo_module *m, const uint8_t *buf, size_t len, char *err,
* nowhere to live. woc refuses this at compile time; the loader
* refuses it again because what the loader accepts, the interpreter
* trusts — this combination must never reach the engine. */
/* databasev2 2: `resident: keys` PARSES and sets this bit, but the
* storage half (tasks 5c/5d) is not implemented — rows are still fully
* resident. Accepting it would be an annotation the compiler honours
* in name only: a developer could declare a 120 GB table keys-resident,
* see it compile, and be OOM-killed. Refuse until the storage lands. */
if (flags & WO_CLASSF_RESIDENT_KEYS)
BAIL("class %u declares `resident: keys`, which is NOT IMPLEMENTED "
"yet — rows are still fully resident, so the annotation would "
"be honoured in name only. Remove it until databasev2 2 tasks "
"5c/5d land; `resident: all` is what actually runs",
(unsigned)i);
if ((flags & WO_CLASSF_VOLATILE) && (flags & WO_CLASSF_RESIDENT_KEYS))
BAIL("class %u: durable:false with resident:keys — rows would have "
"nowhere to be read from", (unsigned)i);

View file

@ -4,6 +4,7 @@
* 2 = usage or load failure (loader's message on stderr)
* Heap cap defaults to 64 MiB, overridable via WO_HEAP_MB. */
#include <fcntl.h>
#include <signal.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
@ -124,6 +125,22 @@ static void gc_pump(wo_vm *vm) {
}
}
/* databasev2 4: one diagnostic line about group commit, opt-in via
* WO_WAL_STATS. Off by default because it would otherwise pollute the output
* of every durable program; a gate that wants the numbers asks for them. */
static void wal_stats_report(const wo_wal *w) {
if (!w || !getenv("WO_WAL_STATS")) return;
fprintf(stderr,
"walstats batches=%llu records=%llu peak_batch=%llu peak_staged=%llu "
"compactions=%llu compact_us_max=%llu compact_us_total=%llu compacted_bytes=%llu\n",
(unsigned long long)w->stat_batches, (unsigned long long)w->stat_records,
(unsigned long long)w->stat_peak_batch, (unsigned long long)w->stat_peak_staged,
(unsigned long long)w->stat_compactions,
(unsigned long long)w->stat_compact_us_max,
(unsigned long long)w->stat_compact_us_total,
(unsigned long long)w->compacted_bytes);
}
int main(int argc, char **argv) {
wo_module mod;
char err[256];
@ -178,6 +195,10 @@ int main(int argc, char **argv) {
return 2;
}
wo_tls_set(&VM);
/* iteration 24: a write to a peer-closed socket must be EPIPE (a
* catchable WO_T_IO), never a process-killing SIGPIPE — every
* serving program writes to sockets whose peers vanish. */
signal(SIGPIPE, SIG_IGN);
/* The database engine boots with the VM: every class IS a table.
* Durability is opt-in — WO_DATA=<dir> opens <dir>/shard-0.wal,
* replays it before the entry runs (boot-before-listeners doctrine),
@ -254,6 +275,25 @@ int main(int argc, char **argv) {
if (v >= 1 && v <= 0x7FFFFFFFul) wo_mailbox_cap = (uint32_t)v;
}
}
/* databasev2 3: the checkpoint policy. WO_CHECKPOINT_BYTES is the floor
* below which a log is too small to bother compacting; WO_CHECKPOINT_RATIO
* is how many times the live set's own size counts as too much history.
* Both exist mainly so the policy is TESTABLE — a gate sets a tiny floor
* and forces compaction in a few writes rather than waiting for megabytes.
* There is no time-based trigger, by design: our records are durable at
* commit, so an idle log does not grow. */
{
const char *cb = getenv("WO_CHECKPOINT_BYTES");
if (cb && cb[0]) {
unsigned long long v = strtoull(cb, NULL, 10);
if (v > 0) wo_wal_ckpt_floor = (uint64_t)v;
}
const char *cr = getenv("WO_CHECKPOINT_RATIO");
if (cr && cr[0]) {
unsigned long v = strtoul(cr, NULL, 10);
if (v <= 0xFFFFFFFFul) wo_wal_ckpt_ratio = (uint32_t)v;
}
}
/* the arc's stage 2: all cores by default (the brave landing), one
* pinned worker vm per extra core; WO_SHARDS caps or forces it */
{
@ -269,7 +309,7 @@ int main(int argc, char **argv) {
if (wo_engine_start(&mod, heap_mb << 20, nshards) != 0) {
fprintf(stderr, "wovm: cannot start %u shards\n", nshards);
wo_engine_stop();
if (VM.rt.wal) wo_wal_close(&WAL);
if (VM.rt.wal) { wal_stats_report(&WAL); wo_wal_close(&WAL); }
wo_db_destroy(&DB);
wo_vm_destroy(&VM);
wo_module_free(&mod);
@ -328,7 +368,7 @@ int main(int argc, char **argv) {
* unwind. */
if (argv_val) wo_drop_kind(&VM.rt, WO_K_MULTI, argv_val);
wo_engine_stop(); /* join + destroy the worker shards before the primary */
if (VM.rt.wal) wo_wal_close(&WAL);
if (VM.rt.wal) { wal_stats_report(&WAL); wo_wal_close(&WAL); }
wo_db_destroy(&DB);
gc_pump(&VM);
wo_vm_destroy(&VM);

View file

@ -296,6 +296,8 @@ static void tick_arm_uring(wo_vm *vm, int64_t now) {
for (wo_fiber *fb = vm->parked; fb; fb = fb->pnext)
if (fb->state == WO_FIB_PARKED && fb->park_fd >= 0 && fb->park_deadline > 0)
if (next == 0 || fb->park_deadline < next) next = fb->park_deadline;
int64_t tn = wo_vm_timers_next(vm); /* iteration 24 T5: armed timers */
if (tn > 0 && (next == 0 || tn < next)) next = tn;
if (next == 0) return;
if (vm->tick_armed && vm->tick_at <= next) return;
int64_t rel = next - now;
@ -323,7 +325,27 @@ static void efd_drain(wo_vm *vm) {
int wo_io_wait(wo_vm *vm) {
for (;;) {
if (wo_sys_stop_pending()) return WO_IO_STOP;
if (wo_sys_stop_pending()) {
/* iteration 24 (the drain): a STOP does not kill parked fibers
* from the outside — it WAKES them all, and each blocking
* builtin resolves per its own stop contract (deadline'd waits
* answer their timeout result, sleeps return early, plain
* waits answer WO_SYS_STOPPED and that fiber unwinds). The
* program's own code then drains and returns. Nothing parked
* = nothing to resolve: the old immediate-stop answer. */
int woke = 0;
wo_fiber *fb = vm->parked;
while (fb) {
wo_fiber *nx = fb->pnext;
if (fb->state == WO_FIB_PARKED) {
wake(vm, fb);
woke = 1;
}
fb = nx;
}
if (woke) return 0;
return WO_IO_STOP;
}
if (vm->io_kind == 0) {
/* keep the wake eventfd armed (oneshot POLL_ADD, re-armed
* after each firing) so inbox pushes interrupt the wait */
@ -371,7 +393,9 @@ int wo_io_wait(wo_vm *vm) {
head++;
}
__atomic_store_n(r.cq_head, head, __ATOMIC_RELEASE);
if (deadline_sweep_uring(vm, now_ms()) && woke != 2) woke = 1;
int64_t swnow = now_ms();
if (wo_vm_timers_fire(vm, swnow) && woke != 2) woke = 1;
if (deadline_sweep_uring(vm, swnow) && woke != 2) woke = 1;
if (woke == 2) return 1; /* adopt-needed */
if (woke) return 0;
continue;
@ -389,6 +413,14 @@ int wo_io_wait(wo_vm *vm) {
}
int timeout = -1;
int64_t now = now_ms();
{
int64_t tn = wo_vm_timers_next(vm); /* iteration 24 T5 */
if (tn > 0) {
int64_t rel = tn - now;
if (rel < 0) rel = 0;
timeout = (int)rel;
}
}
for (wo_fiber *fb = vm->parked; fb; fb = fb->pnext)
if (fb->park_fd == -1
|| (fb->park_fd >= 0 && fb->park_deadline > 0)) {
@ -417,6 +449,7 @@ int wo_io_wait(wo_vm *vm) {
}
}
now = now_ms();
if (wo_vm_timers_fire(vm, now)) woke = 1;
wo_fiber *fb = vm->parked;
while (fb) {
wo_fiber *nx = fb->pnext;

View file

@ -395,6 +395,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
if (stop_pending()) return WO_SYS_STOPPED;
}
if (fd < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) {
if (stop_pending()) return WO_SYS_STOPPED;
/* arc T4: park until the listener is readable, then retry */
vm->cur->park_fd = (int)R[B];
vm->cur->park_deadline = 0;
@ -431,6 +432,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
/* arc T4: nothing readable yet — free the buffer (the retry
* re-allocates) and park until the fd is readable */
wo_str_free(rt, s);
if (stop_pending()) return WO_SYS_STOPPED;
vm->cur->park_fd = (int)R[B];
vm->cur->park_deadline = 0;
vm->cur->park_events = POLLIN;
@ -476,6 +478,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
continue;
}
if (errno == EAGAIN || errno == EWOULDBLOCK) {
if (stop_pending()) return WO_SYS_STOPPED;
vm->cur->park_wr_at = at;
vm->cur->park_fd = (int)R[B];
vm->cur->park_deadline = 0;
@ -534,7 +537,9 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
}
if (n < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) {
wo_str_free(rt, s);
if (fb->dl_at > 0 && dnow >= fb->dl_at) {
if (stop_pending() || (fb->dl_at > 0 && dnow >= fb->dl_at)) {
/* iteration 24: a STOP resolves the wait as its timeout
* result — the program's own drain code decides what next */
fb->dl_active = 0;
R[A] = 0; /* ?Text nil: the deadline expired */
return 0;
@ -584,9 +589,9 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
}
}
if (fd < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) {
if (fb->dl_at > 0 && dnow >= fb->dl_at) {
if (stop_pending() || (fb->dl_at > 0 && dnow >= fb->dl_at)) {
fb->dl_active = 0;
R[A] = WO_NIL_SCALAR; /* ?Int nil: nothing arrived */
R[A] = WO_NIL_SCALAR; /* ?Int nil: nothing arrived (or stop) */
return 0;
}
fb->park_fd = (int)R[B];
@ -631,7 +636,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
continue;
}
if (errno == EAGAIN || errno == EWOULDBLOCK) {
if (fb->dl_at > 0 && dnow >= fb->dl_at) {
if (stop_pending() || (fb->dl_at > 0 && dnow >= fb->dl_at)) {
fb->dl_active = 0;
R[A] = 0; /* false: torn mid-write — close the fd */
return 0;

View file

@ -16,6 +16,7 @@
#include "db.h" /* arc stage 3: the transparent DB RPC (wo_db_req) */
#include "table.h" /* slot encode/decode for the RPC marshaling */
#include "wal.h" /* databasev2 4: the drain issues the barrier */
#include <pthread.h>
#include <poll.h>
@ -78,6 +79,9 @@ static int actor_push(wo_actor *a, wo_msg m);
static void call_reply_to(wo_vm *vm, wo_fiber *caller, uint32_t caller_shard,
uint64_t reply, int status);
static void actor_drop_payload(wo_vm *vm, uint64_t payload);
static void monitors_fire(wo_vm *vm, wo_actor *a);
static void runtime_notify(wo_vm *vm, wo_actor *target, uint64_t msg_val,
const char *what);
/* the owning thread drains its inbox: adopt actors, deliver sends,
* execute home-routed frees. Returns how many envelopes were handled. */
@ -88,6 +92,11 @@ static int wo_vm_adopt(wo_vm *vm) {
ib->head = ib->tail = NULL;
pthread_mutex_unlock(&ib->mu);
int n = 0;
/* databasev2 4 (group commit): DB replies are HELD until one barrier has
* covered the whole drain. Locals, not per-shard state: nothing here needs
* to outlive the batch it describes. */
wo_envelope *rhead = NULL, *rtail = NULL;
uint32_t staged = 0;
while (e) {
wo_envelope *nx = e->next;
switch (e->kind) {
@ -133,6 +142,25 @@ static int wo_vm_adopt(wo_vm *vm) {
}
break;
}
case 7: { /* iteration 24 T4: a cross-shard monitor registration —
WE are the watched actor's home. Dead already = the
notice fires now; else it joins the list. */
wo_actor *ob = (wo_actor *)(uintptr_t)e->from_fiber;
if (e->actor->dead) {
runtime_notify(vm, ob, e->payload, "death notice");
break;
}
wo_monitor *mn = calloc(1, sizeof *mn);
if (!mn) {
actor_drop_payload(vm, e->payload);
break;
}
mn->observer = ob;
mn->msg = e->payload;
mn->next = e->actor->monitors;
e->actor->monitors = mn;
break;
}
case 6: /* iteration 24: a call reply landing on the caller's shard —
fill the slot and wake the parked fiber; the re-executed
builtin consumes it (status != 0 makes it trap). */
@ -149,13 +177,35 @@ static int wo_vm_adopt(wo_vm *vm) {
* the same request back as the reply. */
wo_db_req *q = (wo_db_req *)(uintptr_t)e->payload;
assert(vm->is_primary && "DB requests route to shard 0 only");
wo_wal *dw = (wo_wal *)vm->rt.wal;
size_t before = dw ? dw->len : 0;
wo_db_exec_req(vm, q);
q->done = 1;
/* did this statement actually stage a record? Asking the buffer
* beats guessing from the opcode, and the count is what the
* failure diagnostic reports. */
if (dw && dw->len > before) staged++;
wo_envelope *re = calloc(1, sizeof *re);
if (re) {
re->kind = 4;
re->payload = e->payload;
inbox_push_to(q->from_shard, re);
re->next = NULL;
if (dw && dw->len > before) {
/* This statement STAGED a record, so its reply is HELD:
* pushing it now would unpark the requester before its
* record is durable, which is the ack contract this
* iteration exists to make literally true. FIFO, so the
* first waiter is released first. */
if (rtail) rtail->next = re; else rhead = re;
rtail = re;
} else {
/* A READ (or any statement that staged nothing) has no
* durability to wait for. Holding it too was measurably
* wrong: it parked readers behind an fsync they had no
* stake in, and durable.sN.mixread p99 rose ~4x
* (1043 -> 4057us) until this branch existed. */
inbox_push_to(q->from_shard, re);
}
} /* OOM: the requester stays parked until stop — leak, not UB */
break;
}
@ -170,6 +220,41 @@ static int wo_vm_adopt(wo_vm *vm) {
n++;
e = nx;
}
/* databasev2 4: ONE barrier for everything this drain staged, then every
* held reply. Each requester therefore unparks having been acknowledged
* after the barrier that carried ITS record. Commit unconditionally when
* anything is staged — the inline path relies on finding the buffer empty
* (see db.c), so a drain must never leave a record behind. */
if (staged) {
wo_wal *cw = (wo_wal *)vm->rt.wal;
if (cw) wo_wal_commit_fatal(cw, staged);
}
while (rhead) {
wo_envelope *rn = rhead->next;
wo_db_req *rq = (wo_db_req *)(uintptr_t)rhead->payload;
rhead->next = NULL;
inbox_push_to(rq->from_shard, rhead);
rhead = rn;
}
/* databasev2 3: the ONE point where compaction is safe — the barrier above
* just ran, so the staging buffer is empty. Anywhere else, a staged record
* would be written into a file about to be replaced. This is a correctness
* requirement, not a scheduling preference; wo_wal_compact also refuses a
* non-empty buffer as a backstop.
*
* Replies are released FIRST, deliberately: their records are already
* durable, and holding them across a stop-the-world rewrite would add the
* rewrite's full duration to their latency for no benefit.
*
* The result is ignored because a failed compaction is a missed
* optimisation, not a durability event — the original log is left intact
* and the process carries on. */
if (staged) {
wo_wal *cw = (wo_wal *)vm->rt.wal;
if (cw && wo_wal_should_compact(cw->off, cw->compacted_bytes,
wo_wal_ckpt_floor, wo_wal_ckpt_ratio))
(void)wo_wal_compact(cw, (wo_db *)vm->rt.db);
}
return n;
}
@ -425,6 +510,33 @@ static void *shard_main(void *arg) {
} else {
int rc = wo_io_wait(vm); /* parked fibers AND the wake eventfd */
if (rc == WO_IO_STOP) {
/* iteration 40 — THE DRAIN GUARANTEE. A message sent before
* the stop flag is observed must be delivered and run before
* the engine stops.
*
* NEXT_RUNNABLE() already states this contract for a worker
* holding a live fiber: it returns 2 and keeps draining "so
* queued shutdown messages (close frames!) still run". This
* branch — the IDLE worker, empty run queue, waiting on the
* plane — used to reap and break instead, abandoning whatever
* sat in its inbox for wo_engine_stop() to free wholesale.
*
* An actor between messages is exactly that idle case, which
* is why a WARM server hid the bug: warm shards had live
* fibers and took the correct path. Measured 2026-08-27 on a
* fresh server: 5 of 16 SIGTERM drains left a WebSocket
* client at EOF with no close frame and no diagnostic.
*
* The window belongs to the PRIMARY and closes when it sets
* eng_shutdown (after main returns), so honour it here and
* only exit when the primary says so. Yield on an empty poll:
* a tight loop would burn a core per shard and starve the very
* actors the drain exists to let run. */
if (!eng_shutdown) {
(void)wo_vm_adopt(vm);
if (!vm->qhead) sched_yield();
continue;
}
fib_reap_all(vm);
break;
}
@ -445,6 +557,85 @@ int wo_engine_primary_inbox(int wake_efd) {
return 0;
}
/* iteration 24 teardown phase 1 (single-threaded, BEFORE eng_teardown):
* dismantle one vm's actor world with real drops — container backings are
* malloc'd, so wholesale arena death does NOT cover them (LSan, chat's
* registry map). Cross-shard payloads route home through wo_route_free
* (still live here); the routed kind-2 envelopes are settled by the
* caller's inbox passes. */
static void vm_drop_actor_world(wo_vm *vm) {
wo_actor *a = vm->actors;
vm->actors = NULL;
while (a) {
wo_actor *nx = a->next_all;
if (a->instance) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)a->instance);
for (uint32_t i = 0; i < a->mlen; i++) {
uint64_t m = a->msgs[(a->mhead + i) % a->mcap].payload;
if (m) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)m);
}
wo_monitor *mo = a->monitors;
while (mo) {
wo_monitor *mnx = mo->next;
if (mo->msg) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)mo->msg);
free(mo);
mo = mnx;
}
free(a->msgs);
free(a);
a = nx;
}
wo_timer *tt = vm->timers;
vm->timers = NULL;
while (tt) {
wo_timer *tnx = tt->next;
if (tt->msg) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)tt->msg);
free(tt);
tt = tnx;
}
}
/* Settle every inbox after phase 1: home-routed frees execute on their
* owner vm; payload-carrying strays drop (possibly routing again — the
* outer loop runs until everything is quiet). Node memory always freed. */
static int eng_settle_inboxes(void) {
int moved = 0;
for (uint32_t i = 0; i < wo_eng.nshards && i < WO_ENG_MAX_SHARDS; i++) {
if (!INBOX_READY[i]) continue;
wo_vm *vm = &wo_eng.shards[i];
wo_inbox *ib = &INBOX[i];
wo_envelope *e = ib->head;
ib->head = ib->tail = NULL;
while (e) {
wo_envelope *nx = e->next;
switch (e->kind) {
case 2: /* WE are home: the direct drop is the settlement */
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)e->payload);
break;
case 0:
case 5:
case 7: /* in-flight payloads: drop (may route -> next pass) */
if (e->payload)
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)e->payload);
break;
case 1: /* an unadopted actor shell */
if (e->actor) {
if (e->actor->instance)
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)e->actor->instance);
free(e->actor->msgs);
free(e->actor);
}
break;
default: /* 3/4/6: scalar or engine-side payloads, node-only */
break;
}
free(e);
moved++;
e = nx;
}
}
return moved;
}
int wo_engine_start(const wo_module *mod, size_t heap_cap, uint32_t nshards) {
wo_eng.nshards = nshards;
eng_heap_cap = heap_cap;
@ -489,6 +680,13 @@ void wo_engine_stop(void) {
(void)n;
}
for (uint32_t i = 1; i < wo_eng.nshards; i++) pthread_join(ts[i - 1], NULL);
/* single-threaded from here: PHASE 1 — real drops while every arena
* and the routing fabric are still alive (malloc'd container backings
* inside actor state need them; iteration 24's registry map). Settle
* passes run until routed frees stop appearing. */
for (uint32_t i = 0; i < wo_eng.nshards && i < WO_ENG_MAX_SHARDS; i++)
if (wo_eng.shards[i].rt.arena.base) vm_drop_actor_world(&wo_eng.shards[i]);
while (eng_settle_inboxes() > 0) {}
/* single-threaded from here. Every arena dies wholesale, so routed
* frees and queued payloads need no per-object drops — DISCARD the
* envelopes (freeing the malloc'd nodes/actors) and let the arenas
@ -557,19 +755,42 @@ void wo_vm_destroy(wo_vm *vm) {
free(fb);
}
/* actors first — dropping their state and queued messages needs the
* runtime alive */
* runtime alive. BUT: once the engine is in teardown, arenas die
* WHOLESALE (the standing doctrine) — a moved-in message's home arena
* may belong to an ALREADY-destroyed shard, and even reading its
* header is a use-after-free (ASan, chat's drain). Structures are
* still freed; payload drops are skipped. */
int drops_ok = !eng_teardown;
wo_actor *a = vm->actors;
while (a) {
wo_actor *nx = a->next_all;
if (a->instance) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)a->instance);
for (uint32_t i = 0; i < a->mlen; i++) {
if (drops_ok && a->instance)
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)a->instance);
for (uint32_t i = 0; drops_ok && i < a->mlen; i++) {
uint64_t m = a->msgs[(a->mhead + i) % a->mcap].payload;
if (m) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)m);
}
wo_monitor *mo = a->monitors;
while (mo) { /* undelivered notices are the runtime's to drop */
wo_monitor *mnx = mo->next;
if (drops_ok && mo->msg)
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)mo->msg);
free(mo);
mo = mnx;
}
free(a->msgs);
free(a);
a = nx;
}
wo_timer *tt = vm->timers;
vm->timers = NULL;
while (tt) { /* unfired timers likewise */
wo_timer *tnx = tt->next;
if (drops_ok && tt->msg)
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)tt->msg);
free(tt);
tt = tnx;
}
vm->actors = NULL;
wo_io_destroy(vm);
wo_rt_destroy(&vm->rt);
@ -760,6 +981,7 @@ static void actor_die(wo_vm *vm, wo_actor *a, wo_fiber *delivery) {
a->instance = 0;
}
a->active = NULL;
monitors_fire(vm, a);
}
/* Mailbox nonempty, no delivery fiber: start one on the next message.
@ -877,6 +1099,57 @@ int wo_vm_actor_send(wo_vm *vm, uint64_t addr, uint64_t msg_val, const char **ms
return 0;
}
/* iteration 24 T4/T5: a RUNTIME-sourced delivery (death notice, timer).
* No fiber to trap: a full or dead target drops the message with a
* stderr line (spec'd disclosure), never silently. Runs on any thread —
* cross-shard targets ride the ordinary kind-0 envelope. */
static void runtime_notify(wo_vm *vm, wo_actor *target, uint64_t msg_val,
const char *what) {
if (!target || !msg_val) return;
if (target->dead) {
actor_drop_payload(vm, msg_val);
return; /* send-to-dead: silent by contract */
}
if (wo_mbox_reserve(target) != 0) {
fprintf(stderr, "wovm: %s dropped — the observer's mailbox is full\n", what);
actor_drop_payload(vm, msg_val);
return;
}
if (target->home != vm->shard_id) {
wo_envelope *e = calloc(1, sizeof *e);
if (!e) {
wo_mbox_release(target);
actor_drop_payload(vm, msg_val);
return;
}
e->kind = 0;
e->actor = target;
e->payload = msg_val;
inbox_push_to(target->home, e);
return;
}
wo_msg m0 = { msg_val, NULL, 0 };
if (actor_push(target, m0) != 0) {
wo_mbox_release(target);
actor_drop_payload(vm, msg_val);
return;
}
if (!target->active) (void)actor_activate(vm, target);
}
/* iteration 24 T4: the death walk — every registered observer gets its
* chosen notice, then the list is gone (an actor dies once). */
static void monitors_fire(wo_vm *vm, wo_actor *a) {
wo_monitor *m = a->monitors;
a->monitors = NULL;
while (m) {
wo_monitor *nx = m->next;
runtime_notify(vm, m->observer, m->msg, "death notice");
free(m);
m = nx;
}
}
/* iteration 24: call — send that waits. First entry enqueues with the
* caller attached and parks (WO_PARK_INBOX, the DB-RPC park); the resume
* RE-EXECUTES this builtin and consumes the scalar reply. No hangs, ever:
@ -946,6 +1219,105 @@ int wo_vm_actor_call(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
return WO_SYS_PARKED;
}
int wo_vm_actor_monitor(wo_vm *vm, uint64_t watched, uint64_t observer,
uint64_t msg_val, const char **msg) {
wo_actor *w = (wo_actor *)(uintptr_t)watched;
wo_actor *o = (wo_actor *)(uintptr_t)observer;
if (!w || !o) {
*msg = "monitor: nil actor address";
return WO_T_BOUNDS;
}
if (!msg_val) {
*msg = "monitor: nil notice message";
return WO_T_BOUNDS;
}
/* the registration belongs to the WATCHED actor's home thread */
if (w->home != vm->shard_id) {
wo_envelope *e = calloc(1, sizeof *e);
if (!e) {
actor_drop_payload(vm, msg_val);
*msg = "out of memory";
return WO_T_OOM;
}
e->kind = 7;
e->actor = w;
e->payload = msg_val;
e->from_fiber = (wo_fiber *)o; /* reused slot: the observer */
inbox_push_to(w->home, e);
return 0;
}
if (w->dead) { /* monitoring the dead: the notice fires NOW */
runtime_notify(vm, o, msg_val, "death notice");
return 0;
}
wo_monitor *m = calloc(1, sizeof *m);
if (!m) {
actor_drop_payload(vm, msg_val);
*msg = "out of memory";
return WO_T_OOM;
}
m->observer = o;
m->msg = msg_val;
m->next = w->monitors;
w->monitors = m;
return 0;
}
int wo_vm_timer_after(wo_vm *vm, int64_t ms, uint64_t addr, uint64_t msg_val,
const char **msg) {
wo_actor *a = (wo_actor *)(uintptr_t)addr;
if (!a) {
*msg = "time.after: nil actor address";
return WO_T_BOUNDS;
}
if (!msg_val) {
*msg = "time.after: nil message";
return WO_T_BOUNDS;
}
if (ms <= 0) { /* no wait to arm: deliver now */
runtime_notify(vm, a, msg_val, "timer message");
return 0;
}
wo_timer *t = calloc(1, sizeof *t);
if (!t) {
actor_drop_payload(vm, msg_val);
*msg = "out of memory";
return WO_T_OOM;
}
struct timespec now;
clock_gettime(CLOCK_REALTIME, &now);
t->at = (int64_t)now.tv_sec * 1000 + now.tv_nsec / 1000000 + ms;
t->target = a;
t->msg = msg_val;
t->next = vm->timers;
vm->timers = t;
return 0;
}
int wo_vm_timers_fire(wo_vm *vm, int64_t now) {
int fired = 0;
wo_timer **pp = &vm->timers;
while (*pp) {
wo_timer *t = *pp;
if (t->at <= now) {
*pp = t->next;
runtime_notify(vm, t->target, t->msg, "timer message");
free(t);
fired++;
} else {
pp = &t->next;
}
}
return fired;
}
int64_t wo_vm_timers_next(wo_vm *vm) {
int64_t next = 0;
for (wo_timer *t = vm->timers; t; t = t->next)
if (next == 0 || t->at < next) next = t->at;
return next;
}
/* The drop-table entry governing instruction [pc]: the last one recorded
* at or before it. NULL = nothing live there. */
static const wo_dropent *vm_dropent(const wo_methodrec *me, uint32_t pc) {
@ -1227,6 +1599,15 @@ static int vm_run(wo_vm *vm, uint64_t *ret, wo_err *err) {
} \
int iorc_ = wo_io_wait(vm); \
if (iorc_ == WO_IO_STOP) { \
/* iteration 24: a WORKER on stop keeps DRAINING — its \
* serve loop spins adopting the inbox until the primary \
* finishes the drain window and sets eng_shutdown, so \
* queued shutdown messages (close frames!) still run. \
* Only the PRIMARY's stop ends the program. */ \
if (!vm->is_primary) { \
vm->cur = &vm->f0; \
return 2; \
} \
fib_reap_all(vm); \
vm->cur = &vm->f0; \
return 1; \
@ -1729,8 +2110,31 @@ dispatch:
vm->cur->frames[vm->cur->depth - 1].pc = pc - 1;
vm->cur->ncatch = 0;
vm_unwind(vm, 0);
/* a stop ends the PROGRAM: every fiber — the stopped one,
* queued ones, main wherever it is — unwinds clean */
/* iteration 24 (the drain): a STOPPED wait on a NON-main fiber
* unwinds that fiber ALONE — the rest of the program (main's
* drain code, actors flushing close frames) keeps running.
* Main's own STOPPED still ends the program, as ever. */
if (vm->cur != &vm->f0) {
wo_fiber *dead = vm->cur;
if (dead->actor) {
wo_actor *da = dead->actor;
if (dead->cur_msg) {
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)dead->cur_msg);
dead->cur_msg = 0;
}
call_reply_to(vm, dead->msg_caller, dead->msg_caller_shard,
0, WO_T_ACTOR);
dead->msg_caller = NULL;
da->active = NULL;
}
vm->nfibers--;
fib_retire(vm, dead);
NEXT_RUNNABLE();
RELOAD();
NEXT();
}
/* main: a stop ends the PROGRAM — every remaining fiber
* unwinds clean */
if (vm->cur != &vm->f0) {
wo_fiber *dead = vm->cur;
vm->cur = &vm->f0;

View file

@ -120,6 +120,27 @@ typedef struct wo_msg {
* guarantee). Death (iteration 24): a receive trapping uncaught marks
* the actor dead — sends to it drop silently, calls trap, queued
* callers are error-unparked; the state and mailbox are released. */
/* iteration 24 T4: one death-notice registration. The runtime owns the
* moved-in notice message until delivery (or drops it if the observer is
* unreachable). The list lives on the WATCHED actor, owned by its home
* thread. */
typedef struct wo_monitor {
struct wo_actor *observer;
uint64_t msg;
struct wo_monitor *next;
} wo_monitor;
/* iteration 24 T5: one armed one-shot timer — fires as an ordinary
* runtime send of the moved message when `at` passes. The list lives on
* the ARMING fiber's shard and is scanned by the same deadline machinery
* that serves fd-park deadlines. */
typedef struct wo_timer {
int64_t at; /* wall ms */
struct wo_actor *target;
uint64_t msg;
struct wo_timer *next;
} wo_timer;
typedef struct wo_actor {
uint64_t instance; /* the moved-in state object (runtime-owned) */
uint32_t method; /* receive's method index (self + msg = 2 args) */
@ -134,6 +155,7 @@ typedef struct wo_actor {
* overshoot by at most the number of in-flight sends — disclosed. */
uint32_t pending;
wo_fiber *active; /* the delivery fiber, NULL when idle */
wo_monitor *monitors; /* iteration 24 T4: who wants the death notice */
struct wo_actor *next_all; /* the vm's all-actors list */
} wo_actor;
@ -176,6 +198,9 @@ typedef struct wo_vm {
* freed memory is the UAF this prevents. Steady-state pool size = the
* peak live fiber count; the pool dies with the vm. */
wo_fiber *fib_pool;
/* iteration 24 T5: this shard's armed timers (unsorted list — the
* deadline scan is already linear; a wheel is measured-later work) */
wo_timer *timers;
/* iteration 35, uring backend: the shard's ONE deadline tick — a
* TIMEOUT op with a sentinel user_data armed for the nearest fd-park
* deadline (fd parks keep exactly one POLL op each; expiry wakes them
@ -206,6 +231,20 @@ int wo_vm_actor_send(wo_vm *vm, uint64_t addr, uint64_t msg_val, const char **ms
* caller attached and parks (WO_SYS_PARKED); the re-execution consumes the
* scalar reply into R[A] (vm.c owns the protocol, builtin.c dispatches). */
int wo_vm_actor_call(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg);
/* iteration 24 T4: register a death notice — monitor(watched, observer,
* msg). The msg MOVES to the runtime; an already-dead watched actor
* delivers it immediately. */
int wo_vm_actor_monitor(wo_vm *vm, uint64_t watched, uint64_t observer,
uint64_t msg_val, const char **msg);
/* iteration 24 T5: arm a one-shot timer on THIS shard — time.after(ms,
* addr, msg). ms <= 0 delivers now. */
int wo_vm_timer_after(wo_vm *vm, int64_t ms, uint64_t addr, uint64_t msg_val,
const char **msg);
/* iteration 24 T5: fire every timer at or past `now` (park.c's deadline
* machinery calls this beside the fd-park sweep). Returns fired count. */
int wo_vm_timers_fire(wo_vm *vm, int64_t now);
/* The nearest armed timer's deadline, 0 = none (park.c's tick/timeout). */
int64_t wo_vm_timers_next(wo_vm *vm);
/* ---- the shard engine (arc stage 2) ------------------------------------
* One pinned thread per shard, each a full wo_vm (own arena, GC, I/O
@ -231,7 +270,11 @@ typedef struct wo_envelope {
* from_shard/from_fiber = the parked caller),
* 6 = CALL_REPLY (payload = the SCALAR reply, from_fiber =
* the caller to unpark; status 0 = ok, WO_T_ACTOR =
* the callee was/went dead — the caller traps) */
* the callee was/went dead — the caller traps),
* 7 = MONITOR (iteration 24 T4: actor = the WATCHED one,
* from_fiber REUSED as the observer wo_actor*, payload =
* the moved notice — registered on the watched actor's
* home thread; already-dead delivers the notice now) */
struct wo_actor *actor;
uint64_t payload;
uint32_t from_shard;

View file

@ -478,8 +478,17 @@ enum {
* return value arrives. R is a SCALAR (v1,
* compiler-enforced WO-E226). Dead callee =
* WO_T_ACTOR, immediately or mid-call. */
/* ids 89 (monitor) and 90 (time.after) are RESERVED for the rest of
* the lifecycle slice — do not reuse. */
WO_B_MONITOR = 89, /* (watched, observer, msg) -> (): the
* observer's own M-typed msg is delivered
* when watched dies (trap-death); already
* dead delivers NOW; msg MOVES. A full
* observer's notice is dropped with a
* stderr line (no fiber to trap). */
WO_B_TIME_AFTER = 90, /* (ms, addr, msg) -> (): one-shot timer —
* msg (MOVED) arrives as an ordinary send
* after ms; no cancel (the generation-
* counter idiom is the documented answer);
* ms <= 0 delivers now. */
/* ---- iteration 35: net seams (sysio.c). Deadlines are per-CALL (no
* hidden fd state); a timeout is an EXPECTED outcome, so it answers
* nil/false, never a trap. ms <= 0 = no deadline (the old behavior,

View file

@ -117,6 +117,397 @@ static void test_roundtrip_replay(void) {
wo_rt_destroy(&rt);
}
/* databasev2 4 part A, Task 1: a failed barrier must be DETECTED, and the
* caller must be able to tell WHICH operation failed — a pwrite failure and
* an fdatasync failure are different operational problems and the diagnostic
* has to name the right one. This proves detection only; the fatal exit that
* follows it cannot be exercised in-process. */
static void test_commit_failure_detected(void) {
char path[128];
snprintf(path, sizeof path, "%s/commitfail.wal", g_dir);
wo_rt rt;
T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0);
wo_db db;
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
wo_wal w;
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
const char *msg = "";
/* the WAL remembers where it lives — the abort diagnostic is worthless
* without it */
T_CHECK(w.path != NULL && strstr(w.path, "commitfail.wal") != NULL);
wo_str *s = wo_str_new(&rt, "abc", 3);
uint64_t vals[2] = {7, (uint64_t)(uintptr_t)s};
uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL);
T_CHECK(id != 0);
T_EQ(wo_wal_append_insert(&w, &db, 0, id), 0);
T_CHECK(w.len > 0); /* something really is staged */
/* an unusable descriptor: pwrite reports EBADF. -1 is used rather than
* closing the real fd so the close below cannot double-free it. */
int real = w.fd;
w.fd = -1;
T_EQ(wo_wal_commit(&w), WO_WAL_ERR_WRITE);
T_CHECK(w.len > 0); /* a failed commit consumes nothing */
w.fd = real;
wo_wal_close(&w);
wo_db_destroy(&db);
wo_rt_destroy(&rt);
}
/* databasev2 3 Task 1: compaction rewrites the log as one record per LIVE row.
* Asserts BOTH halves on purpose: "the file got shorter" is also true of a
* truncating bug, so the replay comparison is what actually proves it. */
static void test_compact_shortens_and_replays_equal(void) {
char path[128];
snprintf(path, sizeof path, "%s/compact.wal", g_dir);
wo_rt rt;
T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0);
wo_db db;
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
wo_wal w;
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
const char *msg = "";
uint64_t ids[3];
for (int i = 0; i < 3; i++) {
wo_str *s = wo_str_new(&rt, "abc", 3);
uint64_t vals[2] = {(uint64_t)(i * 10), (uint64_t)(uintptr_t)s};
ids[i] = wo_row_insert(&db, 0, vals, &msg, NULL);
T_CHECK(ids[i] != 0);
T_EQ(wo_wal_append_insert(&w, &db, 0, ids[i]), 0);
T_EQ(wo_wal_commit(&w), 0);
}
/* age it: the SAME row updated repeatedly, so HISTORY grows while the live
* set does not — the exact case checkpoint exists for */
for (int k = 0; k < 40; k++) {
int ek = 0;
T_EQ(wo_row_update_field(&db, 0, ids[0], 0, (uint64_t)(500 + k), &msg, &ek), 0);
T_EQ(wo_wal_append_update(&w, &db, 0, ids[0]), 0);
T_EQ(wo_wal_commit(&w), 0);
}
uint64_t before_bytes = 0;
int64_t before_recs = wo_wal_check(path, &before_bytes);
T_CHECK(before_recs == 43); /* 3 inserts + 40 updates, all history */
T_EQ(wo_wal_compact(&w, &db), 0);
uint64_t after_bytes = 0;
int64_t after_recs = wo_wal_check(path, &after_bytes);
T_CHECK(after_recs == 3); /* one record per LIVE row */
T_CHECK(after_bytes < before_bytes); /* and the file really shrank */
/* the WAL stays usable: the descriptor was reopened and the offset reset,
* so a further write must land AFTER the compacted records, not over them */
wo_str *s4 = wo_str_new(&rt, "xyz", 3);
uint64_t v4[2] = {99, (uint64_t)(uintptr_t)s4};
uint64_t id4 = wo_row_insert(&db, 0, v4, &msg, NULL);
T_CHECK(id4 != 0);
T_EQ(wo_wal_append_insert(&w, &db, 0, id4), 0);
T_EQ(wo_wal_commit(&w), 0);
T_CHECK(wo_wal_check(path, NULL) == 4);
wo_wal_close(&w);
/* the proof: a FRESH store replayed from the compacted log must hold the
* same rows, the same ids, and the LAST value each row had */
wo_db db2;
T_EQ(wo_db_init(&db2, CLASSES, 1, 0, 1), 0);
T_EQ(wo_wal_replay(path, &db2), 4);
uint64_t out[2];
T_EQ(wo_row_read(&db2, &rt, 0, ids[0], out, &msg), 0);
T_CHECK(out[0] == 539); /* the 40th update won, not the original 0 */
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
T_EQ(wo_row_read(&db2, &rt, 0, ids[1], out, &msg), 0);
T_CHECK(out[0] == 10);
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
T_EQ(wo_row_read(&db2, &rt, 0, ids[2], out, &msg), 0);
T_CHECK(out[0] == 20);
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
T_EQ(wo_row_read(&db2, &rt, 0, id4, out, &msg), 0);
T_CHECK(out[0] == 99);
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
wo_db_destroy(&db2);
wo_db_destroy(&db);
wo_rt_destroy(&rt);
}
/* databasev2 3 Task 2: a stale temp file is the one input that could be
* mistaken for data — a crash before the rename leaves one behind, full of
* well-formed records that are NOT yet authoritative. So the fixture uses
* plausible records (a byte copy of a real log), not garbage: garbage would be
* rejected by the CRC anyway and would prove nothing. */
static void test_stale_compact_temp_is_removed(void) {
char path[128], tmp[160];
snprintf(path, sizeof path, "%s/stale.wal", g_dir);
snprintf(tmp, sizeof tmp, "%s%s", path, WO_WAL_TMP_SUFFIX);
wo_rt rt;
T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0);
wo_db db;
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
wo_wal w;
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
const char *msg = "";
/* two live rows in the REAL log */
uint64_t ids[2];
for (int i = 0; i < 2; i++) {
wo_str *s = wo_str_new(&rt, "abc", 3);
uint64_t vals[2] = {(uint64_t)(i + 1), (uint64_t)(uintptr_t)s};
ids[i] = wo_row_insert(&db, 0, vals, &msg, NULL);
T_EQ(wo_wal_append_insert(&w, &db, 0, ids[i]), 0);
T_EQ(wo_wal_commit(&w), 0);
}
wo_wal_close(&w);
/* forge a plausible stale temp: a byte copy of the real log */
{
int src = open(path, O_RDONLY);
int dst = open(tmp, O_WRONLY | O_CREAT | O_TRUNC, 0644);
T_CHECK(src >= 0 && dst >= 0);
char buf[8192];
ssize_t n;
while ((n = read(src, buf, sizeof buf)) > 0) T_CHECK(write(dst, buf, (size_t)n) == n);
close(src);
close(dst);
T_EQ(access(tmp, F_OK), 0); /* it really is there before we open */
}
wo_wal w2;
T_EQ(wo_wal_open(&w2, path, 1 << 16), 0);
T_CHECK(access(tmp, F_OK) != 0); /* gone, and never consulted */
wo_wal_close(&w2);
/* and the live log still says exactly what it said */
wo_db db2;
T_EQ(wo_db_init(&db2, CLASSES, 1, 0, 1), 0);
T_EQ(wo_wal_replay(path, &db2), 2);
uint64_t out[2];
T_EQ(wo_row_read(&db2, &rt, 0, ids[0], out, &msg), 0);
T_CHECK(out[0] == 1);
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
T_EQ(wo_row_read(&db2, &rt, 0, ids[1], out, &msg), 0);
T_CHECK(out[0] == 2);
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
wo_db_destroy(&db2);
wo_db_destroy(&db);
wo_rt_destroy(&rt);
}
/* databasev2 3 Task 3: the trigger, tested as a pure decision. Kept pure
* precisely so it CAN be tested — a policy only observable by writing megabytes
* and waiting is a policy nobody checks. */
static void test_should_compact_policy(void) {
/* below the floor, nothing fires however bad the ratio looks */
T_EQ(wo_wal_should_compact(1000, 10, 4096, 3), 0);
T_EQ(wo_wal_should_compact(4095, 1, 4096, 3), 0);
/* past the floor with no prior compaction: run once to learn the size */
T_EQ(wo_wal_should_compact(4096, 0, 4096, 3), 1);
/* with a known denominator it is a straight ratio test */
T_EQ(wo_wal_should_compact(30000, 10000, 4096, 3), 0); /* exactly 3x is not MORE than 3x */
T_EQ(wo_wal_should_compact(30001, 10000, 4096, 3), 1);
T_EQ(wo_wal_should_compact(19999, 10000, 4096, 2), 0);
T_EQ(wo_wal_should_compact(20001, 10000, 4096, 2), 1);
/* a zero ratio disables the policy rather than dividing by nothing */
T_EQ(wo_wal_should_compact(1u << 30, 10, 4096, 0), 0);
}
/* databasev2 3 Task 3: the ordering rule, asserted rather than trusted.
* Compaction with records staged would write them into a file about to be
* replaced, so it must be REFUSED — and refused without touching the log. */
static void test_compact_refuses_with_staged_records(void) {
char path[128];
snprintf(path, sizeof path, "%s/staged.wal", g_dir);
wo_rt rt;
T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0);
wo_db db;
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
wo_wal w;
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
const char *msg = "";
wo_str *s1 = wo_str_new(&rt, "abc", 3);
uint64_t v1[2] = {7, (uint64_t)(uintptr_t)s1};
uint64_t id1 = wo_row_insert(&db, 0, v1, &msg, NULL);
T_EQ(wo_wal_append_insert(&w, &db, 0, id1), 0);
T_EQ(wo_wal_commit(&w), 0); /* durable, buffer empty */
/* now stage WITHOUT committing */
wo_str *s2 = wo_str_new(&rt, "xyz", 3);
uint64_t v2[2] = {8, (uint64_t)(uintptr_t)s2};
uint64_t id2 = wo_row_insert(&db, 0, v2, &msg, NULL);
T_EQ(wo_wal_append_insert(&w, &db, 0, id2), 0);
T_CHECK(w.len > 0);
uint64_t before = 0;
int64_t recs = wo_wal_check(path, &before);
T_EQ(wo_wal_compact(&w, &db), -1); /* refused */
T_CHECK(w.len > 0); /* and the staged record is still there */
uint64_t after = 0;
T_CHECK(wo_wal_check(path, &after) == recs && after == before); /* log untouched */
/* the staged record still commits normally afterwards */
T_EQ(wo_wal_commit(&w), 0);
T_CHECK(wo_wal_check(path, NULL) == recs + 1);
wo_wal_close(&w);
wo_db_destroy(&db);
wo_rt_destroy(&rt);
}
/* databasev2 3 Task 4: kill -9 DURING compaction.
*
* The existing battery is insert-only, so its "records >= acks" oracle is
* exactly what compaction is allowed to break: collapsing history is the point.
* The invariant that survives is the ACKED LIVE SET — every id acked as
* inserted and not later acked as deleted must be present with its acked value,
* and every id acked as deleted must be absent. Both the pre-compaction and the
* post-compaction log satisfy that identically, which is precisely the
* "never a mixture" property the design is shaped around.
*
* The child deletes as it goes so HISTORY accumulates while the live set stays
* small — without that, compaction would have nothing to collapse and the test
* would prove nothing. */
#define CK_DELETED UINT64_MAX
static void ck_ack(int fd, uint64_t id, uint64_t val) {
uint64_t rec[2] = {id, val};
if (write(fd, rec, sizeof rec) != (ssize_t)sizeof rec) _exit(0); /* parent gone */
}
static void compact_battery_child(const char *path, int ack_fd) {
wo_rt rt;
wo_db db;
wo_wal w;
if (wo_rt_init(&rt, 1 << 20, CLASSES, 1) != 0) _exit(9);
if (wo_db_init(&db, CLASSES, 1, 0, 1) != 0) _exit(9);
if (wo_wal_open(&w, path, 1 << 20) != 0) _exit(9);
const char *msg = "";
uint64_t live[512];
size_t nlive = 0;
for (uint64_t i = 0;; i++) {
uint64_t val = i * 7 + 3;
wo_str *s = wo_str_new(&rt, "r", 1);
uint64_t vals[2] = {val, (uint64_t)(uintptr_t)s};
uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL);
wo_str_free(&rt, s);
if (!id) _exit(9);
if (wo_wal_append_insert(&w, &db, 0, id) != 0) _exit(9);
if (wo_wal_commit(&w) != 0) _exit(9); /* durable BEFORE the ack */
ck_ack(ack_fd, id, val);
if (nlive < 512) live[nlive++] = id;
/* drop the oldest so history grows while the live set does not */
if (nlive > 16) {
uint64_t victim = live[0];
memmove(live, live + 1, (nlive - 1) * sizeof live[0]);
nlive--;
/* INTENT FIRST, deliberately. An ack after the commit would race:
* a kill between them leaves the row legitimately gone on disk
* while the last ack still says "inserted", and the parent would
* demand a row the engine was right to remove. Announcing intent
* makes the row's fate simply UNKNOWN to the parent, which is the
* honest thing to assert about it. */
ck_ack(ack_fd, victim, CK_DELETED);
if (wo_row_remove(&db, 0, victim) != 0) _exit(9);
if (wo_wal_append_remove(&w, 0, victim) != 0) _exit(9);
if (wo_wal_commit(&w) != 0) _exit(9);
}
/* compact often, so a kill has a real chance of landing inside one */
if (i % 24 == 23) (void)wo_wal_compact(&w, &db);
}
}
static void test_compact_crash_battery(void) {
int rounds = 40; /* it is a RACE: one green run proves very little */
for (int round = 0; round < rounds; round++) {
char path[128], tmp[160];
snprintf(path, sizeof path, "%s/ckcrash-%d.wal", g_dir, round);
snprintf(tmp, sizeof tmp, "%s%s", path, WO_WAL_TMP_SUFFIX);
int pipefd[2];
T_EQ(pipe(pipefd), 0);
pid_t pid = fork();
T_CHECK(pid >= 0);
if (pid == 0) {
close(pipefd[0]);
compact_battery_child(path, pipefd[1]);
_exit(0);
}
close(pipefd[1]);
/* vary the instant so kills land before, inside and after rewrites */
struct timespec ts = {0, (7 + round * 3) * 1000000L};
while (nanosleep(&ts, &ts) != 0) {}
kill(pid, SIGKILL);
int status;
waitpid(pid, &status, 0);
/* replay the acks into the expected live set, in order */
uint64_t ids[65536], vals[65536];
size_t n = 0;
for (;;) {
uint64_t rec[2];
ssize_t r = read(pipefd[0], rec, sizeof rec);
if (r != (ssize_t)sizeof rec) break;
if (n < 65536) { ids[n] = rec[0]; vals[n] = rec[1]; n++; }
}
close(pipefd[0]);
T_CHECK(n > 0); /* the child got at least one commit out */
wo_rt rt;
T_EQ(wo_rt_init(&rt, 1 << 22, CLASSES, 1), 0);
wo_db db;
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
int64_t ck_recs = wo_wal_check(path, NULL);
int64_t ck_applied = wo_wal_replay(path, &db);
T_CHECK(ck_applied >= 0); /* never reported as corruption */
/* A stale temp may well EXIST after a kill inside compaction — that is
* the expected debris. The guarantee is that the next OPEN removes it
* and never reads it, so that is what gets asserted here; checking
* merely for its absence after a replay would be asserting something
* the design never promised (wo_wal_replay does not open the WAL). */
{
wo_wal probe;
T_EQ(wo_wal_open(&probe, path, 1 << 20), 0);
T_CHECK(access(tmp, F_OK) != 0);
wo_wal_close(&probe);
}
const char *msg = "";
int bad = 0, checked = 0;
for (size_t k = 0; k < n && !bad; k++) {
if (vals[k] == CK_DELETED) continue; /* intent: fate is unknown */
/* an id ever announced for deletion may legally be gone */
int doomed = 0;
for (size_t j = 0; j < n; j++)
if (ids[j] == ids[k] && vals[j] == CK_DELETED) { doomed = 1; break; }
if (doomed) continue;
uint64_t out[2];
int rc = wo_row_read(&db, &rt, 0, ids[k], out, &msg);
if (0) {
} else if (rc != 0 || out[0] != vals[k]) {
bad = 1; /* an acked insert is missing or wrong */
fprintf(stderr, "CKDIAG round=%d id=%llu rc=%d got=%llu want=%llu ack#%zu/%zu "
"log_records=%lld replay_applied=%lld\n",
round, (unsigned long long)ids[k], rc,
rc == 0 ? (unsigned long long)out[0] : 0ull,
(unsigned long long)vals[k], k, n,
(long long)ck_recs, (long long)ck_applied);
} else {
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
}
checked++;
}
T_CHECK(checked > 0);
T_CHECK(!bad);
wo_db_destroy(&db);
wo_rt_destroy(&rt);
}
}
static void test_torn_tail(void) {
char path[128];
snprintf(path, sizeof path, "%s/torn.wal", g_dir);
@ -534,12 +925,18 @@ int main(void) {
snprintf(g_dir, sizeof g_dir, "/tmp/wo-wal-test-XXXXXX");
if (!mkdtemp(g_dir)) return 1;
test_roundtrip_replay();
test_commit_failure_detected();
test_compact_shortens_and_replays_equal();
test_stale_compact_temp_is_removed();
test_should_compact_policy();
test_compact_refuses_with_staged_records();
test_torn_tail();
test_float_bytes_replay();
test_offset_capture();
test_offset_after_failed_commit();
test_read_row_at();
test_crash_battery();
test_compact_crash_battery();
/* leave the dir for a failed run's forensics only */
if (!t_fail) {
char cmd[128];

433
scripts/chat-accept.sh Executable file
View file

@ -0,0 +1,433 @@
#!/usr/bin/env bash
# scripts/chat-accept.sh — iteration 24's gate. The chat sample serves
# WebSocket rooms through the framework ([deps], file:// remote); a raw
# RFC 6455 python client (stdlib only, INDEPENDENT accept-key check)
# proves: the handshake, broadcast + presence + isolation across rooms,
# the 1k-clients-one-hot-room soak (fds/RSS accounted), and the SIGTERM
# drain (close frames, exit 0) — functional legs on BOTH WO_IO backends
# plus an ASan run.
set -uo pipefail
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
WOC="$ROOT/compiler/_build/default/bin/woc"
WOVM="$ROOT/runtime/wovm"
ASAN="$ROOT/runtime/build/wovm_asan"
PORT0="${CHAT_PORT:-18901}"
PORT="$PORT0"
SOAK_N="${CHAT_SOAK:-1000}"
pass=0; fail=0
ok() { echo "ok $1"; pass=$((pass + 1)); }
bad() { echo "FAIL $1 -- $2"; fail=$((fail + 1)); }
if [[ ! -x "$WOC" || ! -x "$WOVM" ]]; then
echo "chat-accept: build woc and wovm first" >&2; exit 1
fi
ulimit -n 8192 2>/dev/null || true
W="$(mktemp -d "${TMPDIR:-/tmp}/chat-accept.XXXXXX")"
SRV=""
# The example's server log lives at a STABLE path so a developer can
# `tail -F /tmp/chat.log` while this runs. It used to go to the per-run temp
# dir, which cleanup() deletes on exit — so there was nothing left to read and
# nothing to follow live. Truncated once here, then APPENDED by every leg with
# a banner, so one file holds the whole run in order.
SRVLOG="/tmp/chat.log"
: > "$SRVLOG"
LEG=0
LEGFROM=1
echo "server log: $SRVLOG (tail -F \"$SRVLOG\" to follow)"
cleanup() {
# kill EVERY server this run started, not merely the most recent $SRV: a leg
# that dies before clearing SRV used to orphan a listener, which then broke
# the next run on the same port. $W is unique per run, so matching on it
# cannot touch another run's processes.
[[ -n "$SRV" ]] && kill -9 "$SRV" 2>/dev/null
pkill -9 -f "$W/app/target/chat" 2>/dev/null
rm -rf "$W"
}
trap cleanup EXIT
cp -r "$ROOT/docs/examples/porch" "$W/fw"
git -C "$W/fw" init -q && git -C "$W/fw" add -A
git -C "$W/fw" -c user.email=t@t -c user.name=t commit -qm v01 && git -C "$W/fw" tag v0.1.0
cp -r "$ROOT/docs/examples/chat" "$W/app"
sed -i "s|https://github.com/shoneyj/porch|file://$W/fw|" "$W/app/wo.toml"
printf '[build]\nruntime = "%s"\n' "$WOVM" >> "$W/app/wo.toml"
if "$WOC" "$W/app" >"$W/build.out" 2>&1 && [[ -x "$W/app/target/chat" ]]; then
ok "deps chain + build"
else
bad "build" "$(grep -m1 error "$W/build.out" || head -1 "$W/build.out")"
echo "chat-accept: 1 checks, 1 failures"; exit 1
fi
# the raw client, shared by every leg
CLIENT="$W/wsc.py"
cat > "$CLIENT" <<'PYEOF'
import socket, base64, hashlib, os, time
GUID = "258EAFA5-E914-47DA-95CA-C5AB0DC85B11"
BUF = {}
def connect(port, room, name, timeout=8, rcvbuf=None):
# rcvbuf: shrink THIS client's receive buffer so the server's socket fills
# quickly — how the WO_MAILBOX leg manufactures a genuinely slow member
# without sleeping. Must be set before connect() to take effect.
if rcvbuf is None:
s = socket.create_connection(("127.0.0.1", port), timeout=timeout)
else:
s = socket.socket()
s.setsockopt(socket.SOL_SOCKET, socket.SO_RCVBUF, rcvbuf)
s.settimeout(timeout)
s.connect(("127.0.0.1", port))
key = base64.b64encode(os.urandom(16)).decode()
s.sendall((f"GET /ws?room={room}&name={name} HTTP/1.1\r\nhost: a\r\n"
f"upgrade: websocket\r\nconnection: Upgrade\r\n"
f"sec-websocket-key: {key}\r\nsec-websocket-version: 13\r\n\r\n").encode())
d = b""
while b"\r\n\r\n" not in d: d += s.recv(2000)
head, _, rest = d.partition(b"\r\n\r\n")
BUF[s] = rest # a frame may already ride the same segment
head = head.decode()
assert " 101 " in head.splitlines()[0], head.splitlines()[0]
want = base64.b64encode(hashlib.sha1((key + GUID).encode()).digest()).decode()
assert want in head, "accept-key mismatch (independent check)"
return s
def _take(s, n, timeout):
s.settimeout(timeout)
b = BUF.get(s, b"")
while len(b) < n:
c = s.recv(4096)
if not c:
BUF[s] = b
return None
b += c
BUF[s] = b[n:]
return b[:n]
def send(s, text):
p = text.encode(); mask = os.urandom(4)
if len(p) < 126: hdr = bytes([0x81, 0x80 | len(p)])
else: hdr = bytes([0x81, 0x80 | 126, len(p) >> 8, len(p) & 255])
s.sendall(hdr + mask + bytes(b ^ mask[i % 4] for i, b in enumerate(p)))
def recv(s, timeout=5):
h = _take(s, 2, timeout)
if h is None: return (-2, "") # EOF
b0, b1 = h[0], h[1]
ln = b1 & 0x7F
if ln == 126:
e = _take(s, 2, timeout); ln = (e[0] << 8) | e[1]
d = _take(s, ln, timeout) if ln else b""
return (b0 & 0x0F), (d or b"").decode(errors="replace")
PYEOF
serve() { # serve PORT [env...] — start + wait for THIS server's listener line
PORT="$1"; shift
LEG=$((LEG + 1))
printf '\n===== leg %d — port %s — %s =====\n' "$LEG" "$PORT" "${*:-default env}" >>"$SRVLOG"
# readiness is searched only in THIS leg's slice: the log is appended, never
# truncated, so a 'listening' line from an earlier leg would lie
LEGFROM=$(( $(wc -l < "$SRVLOG") + 1 ))
"$@" "$W/app/target/chat" "$PORT" >>"$SRVLOG" 2>&1 &
SRV=$!
for _ in $(seq 1 80); do
tail -n "+$LEGFROM" "$SRVLOG" 2>/dev/null | grep -q listening && return 0
sleep 0.1
done
return 1
}
functional() { # $1 = leg name
timeout 30 python3 - "$PORT" <<'PYEOF'
import sys; sys.path.insert(0, sys.argv[0].rsplit("/",1)[0])
port = int(sys.argv[1])
import importlib.util, os
spec = importlib.util.spec_from_file_location("wsc", os.environ["WSC"])
wsc = importlib.util.module_from_spec(spec); spec.loader.exec_module(wsc)
a = wsc.connect(port, "lobby", "alice")
assert wsc.recv(a) == (1, "* alice joined")
b = wsc.connect(port, "lobby", "bob")
assert wsc.recv(a) == (1, "* bob joined")
assert wsc.recv(b) == (1, "* bob joined")
c = wsc.connect(port, "other", "carol")
assert wsc.recv(c) == (1, "* carol joined")
wsc.send(a, "hello room")
assert wsc.recv(a) == (1, "alice: hello room")
assert wsc.recv(b) == (1, "alice: hello room")
import socket
try:
k, t = wsc.recv(c, timeout=0.8); assert False, f"leak into other room: {t}"
except socket.timeout: pass
b.close()
k, t = wsc.recv(a)
assert (k, t) == (1, "* bob left"), (k, t)
a.close(); c.close()
print("functional-ok")
PYEOF
}
# ---- 2. functional on both backends ----
export WSC="$CLIENT"
serve "$((PORT0 + 0))" env WO_IO=uring || bad "serve-uring" "no listener"
r="$(functional uring)"; [[ "$r" == *functional-ok* ]] \
&& ok "uring: handshake(key verified) + presence + broadcast + isolation + leave" \
|| bad "uring-functional" "$r"
kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV=""
serve "$((PORT0 + 1))" env WO_IO=epoll || bad "serve-epoll" "no listener"
r="$(functional epoll)"; [[ "$r" == *functional-ok* ]] \
&& ok "epoll: the same matrix" || bad "epoll-functional" "$r"
kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV=""
# ---- 3. the soak: N clients, ONE hot room ----
serve "$((PORT0 + 2))" || bad "serve-soak" "no listener"
fds_before="$(ls /proc/$SRV/fd 2>/dev/null | wc -l)"
fds_prev=99999
r="$(timeout 180 python3 - "$PORT" "$SOAK_N" <<'PYEOF'
import asyncio, sys, os, time, base64, hashlib
port, N = int(sys.argv[1]), int(sys.argv[2])
GUID = "258EAFA5-E914-47DA-95CA-C5AB0DC85B11"
MARK = "the-hot-room-marker"
sem = asyncio.Semaphore(100)
async def client(i, results):
async with sem:
r, w = await asyncio.open_connection("127.0.0.1", port)
key = base64.b64encode(os.urandom(16)).decode()
w.write((f"GET /ws?room=hot&name=c{i} HTTP/1.1\r\nhost: a\r\n"
f"upgrade: websocket\r\nconnection: Upgrade\r\n"
f"sec-websocket-key: {key}\r\nsec-websocket-version: 13\r\n\r\n").encode())
await w.drain()
d = b""
while b"\r\n\r\n" not in d: d += await r.read(2000)
if i == 0:
# the sender: wait for the herd, then one marker line
await asyncio.sleep(0)
results["sender_ready"].set()
try:
buf = b""
deadline = time.time() + 150
while time.time() < deadline:
try:
c = await asyncio.wait_for(r.read(8192), timeout=5)
except asyncio.TimeoutError:
if results["sent"].is_set(): break
continue
if not c: break
buf += c
# scan frames for the marker (server frames are unmasked, small)
if MARK.encode() in buf:
results["got"] += 1
return
finally:
w.close()
async def main():
results = {"got": 0, "sender_ready": asyncio.Event(), "sent": asyncio.Event()}
conns = []
# keep the sender's socket outside the tasks: join first
sr, sw = None, None
async def sender():
nonlocal sr, sw
async with sem:
sr, sw = await asyncio.open_connection("127.0.0.1", port)
key = base64.b64encode(os.urandom(16)).decode()
sw.write((f"GET /ws?room=hot&name=sender HTTP/1.1\r\nhost: a\r\n"
f"upgrade: websocket\r\nconnection: Upgrade\r\n"
f"sec-websocket-key: {key}\r\nsec-websocket-version: 13\r\n\r\n").encode())
await sw.drain()
d = b""
while b"\r\n\r\n" not in d: d += await sr.read(2000)
await sender()
tasks = [asyncio.create_task(client(i, results)) for i in range(N)]
await asyncio.sleep(max(2.0, N / 250)) # let the herd join + drain presence
p = MARK.encode(); mask = os.urandom(4)
hdr = bytes([0x81, 0x80 | len(p)])
sw.write(hdr + mask + bytes(b ^ mask[i % 4] for i, b in enumerate(p)))
await sw.drain()
results["sent"].set()
t0 = time.time()
await asyncio.gather(*tasks, return_exceptions=True)
el = int((time.time() - t0) * 1000)
sw.close()
print(f"{results['got']}|{N}|{el}")
asyncio.run(main())
PYEOF
)"
got="${r%%|*}"; rest="${r#*|}"; n="${rest%%|*}"; el="${rest#*|}"
[[ "$got" == "$n" ]] \
&& ok "soak: the marker reached all $got/$n hot-room clients (${el}ms after send)" \
|| bad "soak" "$r"
# leave-broadcast storms take a moment to settle after 1k closes
for _ in $(seq 1 20); do
fds_w1="$(ls /proc/$SRV/fd 2>/dev/null | wc -l)"
[[ "$fds_w1" -le "$fds_prev" ]] && break
fds_prev="$fds_w1"
sleep 0.5
done
# The fd check is for a per-CONNECTION leak, and a fixed tolerance cannot
# express that. Shards initialise LAZILY (runtime/src/vm.c: a worker's vm is
# not paid for until its first fiber arrives), so the first wave legitimately
# adds one io_uring + one eventfd PER SHARD, capped at nproc — on a 20-core
# box that is +18, which the old `fds_before + 8` read as a leak. Measured
# 2026-08-27: 26 -> 44 after 20 clients, then still 44 after 40 more.
#
# So assert the invariant itself: a SECOND wave must not raise the count.
# Core-count independent, and it catches a slow leak that any fixed
# tolerance would hide inside its own slack.
timeout 60 python3 - "$PORT" 20 <<'PYEOF' >/dev/null 2>&1
import socket, base64, os, sys, time
port, n = int(sys.argv[1]), int(sys.argv[2])
socks = []
for i in range(n):
s = socket.create_connection(("127.0.0.1", port), timeout=8)
k = base64.b64encode(os.urandom(16)).decode()
s.sendall((f"GET /ws?room=fdwave&name=w{i} HTTP/1.1\r\nhost: a\r\n"
f"upgrade: websocket\r\nconnection: Upgrade\r\n"
f"sec-websocket-key: {k}\r\nsec-websocket-version: 13\r\n\r\n").encode())
h = b""
while b"\r\n\r\n" not in h:
h += s.recv(4096)
socks.append(s)
time.sleep(0.5)
for s in socks:
s.close()
PYEOF
fds_after="$fds_w1"
for _ in $(seq 1 20); do
fds_after="$(ls /proc/$SRV/fd 2>/dev/null | wc -l)"
[[ "$fds_after" -le "$fds_w1" ]] && break
sleep 0.5
done
rss_kb="$(awk '/VmRSS/{print $2}' /proc/$SRV/status 2>/dev/null)"
[[ "$fds_after" -le "$fds_w1" ]] \
&& ok "no per-connection fd leak (start $fds_before, after $SOAK_N: $fds_w1, after 20 more: $fds_after)" \
|| bad "soak-fds" "second wave grew fds: $fds_w1 -> $fds_after (start $fds_before)"
[[ -n "$rss_kb" && "$rss_kb" -lt 819200 ]] \
&& ok "soak RSS bounded (${rss_kb}KB < 800MB)" || bad "soak-rss" "${rss_kb}KB"
# ---- 4. drain: SIGTERM with clients connected -> close frames, exit 0 ----
# Starts its OWN server. It used to inherit the soak leg's $SRV, which meant
# any leg inserted between them silently handed drain an empty pid: its python
# died on int(""), the leg reported a bare failure, AND the soak server was
# never killed — orphaning a listener that then broke the NEXT run's soak on
# the same port. No leg may depend on another leg's server.
serve "$((PORT0 + 6))" || bad "serve-drain" "no listener"
r="$(timeout 30 python3 - "$PORT" "$SRV" <<'PYEOF'
import sys, os, time, signal, socket
import importlib.util
spec = importlib.util.spec_from_file_location("wsc", os.environ["WSC"])
wsc = importlib.util.module_from_spec(spec); spec.loader.exec_module(wsc)
port, srv = int(sys.argv[1]), int(sys.argv[2])
a = wsc.connect(port, "lobby", "alice"); wsc.recv(a)
b = wsc.connect(port, "lobby", "bob"); wsc.recv(a); wsc.recv(b)
os.kill(srv, signal.SIGTERM)
def drained(s):
try:
while True:
k, _ = wsc.recv(s, timeout=5)
if k == 8: return "close-frame"
if k == -2: return "eof"
except socket.timeout:
return "stuck"
except (ConnectionResetError, BrokenPipeError):
return "reset"
print(drained(a) + "|" + drained(b))
PYEOF
)"
[[ "$r" == "close-frame|close-frame" ]] \
&& ok "drain: both clients got the close frame" || bad "drain" "$r"
stopped=1
for _ in $(seq 1 40); do kill -0 "$SRV" 2>/dev/null || { stopped=0; break; }; sleep 0.1; done
[[ $stopped -eq 0 ]] && ok "SIGTERM exits 0" || bad "stop" "still running"
SRV=""
# ---- 4b. WO_SHARDS=1: the same matrix on one shard ----
# The plan requires `just chat` green at default cores AND on a single shard:
# cross-shard placement is where the actor work can hide a bug, so the
# one-shard run is the control that says a failure is placement's fault.
serve "$((PORT0 + 4))" env WO_SHARDS=1 || bad "serve-shards1" "no listener"
r="$(functional shards1)"; [[ "$r" == *functional-ok* ]] \
&& ok "WO_SHARDS=1: the same matrix on a single shard" \
|| bad "shards1-functional" "$r"
kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV=""
# ---- 4c. WO_MAILBOX=8: the drop-slow-member path FIRES and the room lives ----
# The backpressure policy earning its keep. A member that stops reading makes
# its writer block on write_dl; with the mailbox capped at 8 the room's
# broadcast send traps (WO_T_ACTOR), and the room must CATCH that, drop the
# member, and keep serving everyone else. Asserting the room survives is the
# point — a room that dies with its slowest member is the bug this policy
# exists to prevent.
serve "$((PORT0 + 5))" env WO_MAILBOX=8 || bad "serve-mailbox" "no listener"
r="$(timeout 90 python3 - "$PORT" <<'PYEOF'
import importlib.util, os, socket, sys, time
spec = importlib.util.spec_from_file_location("wsc", os.environ["WSC"])
wsc = importlib.util.module_from_spec(spec); spec.loader.exec_module(wsc)
port = int(sys.argv[1])
fast = wsc.connect(port, "bp", "fast")
wsc.recv(fast) # * fast joined
# the slow member: a tiny receive buffer so the server's socket fills fast,
# and it never reads a single frame
slow = wsc.connect(port, "bp", "slow", rcvbuf=2048)
wsc.recv(fast) # * slow joined
# storm: big frames the slow member never drains
blob = "x" * 1024
for i in range(400):
try:
wsc.send(fast, f"{i}-{blob}")
except OSError:
break
# drain what fast owes us so its own mailbox cannot be the thing that fills
deadline = time.time() + 20
seen = 0
while time.time() < deadline:
try:
k, t = wsc.recv(fast, timeout=0.5)
seen += 1
except Exception:
break
# the room must still be alive and serving the fast member
survivor = wsc.connect(port, "bp", "late")
ok_join = False
deadline = time.time() + 15
while time.time() < deadline:
try:
k, t = wsc.recv(fast, timeout=1.0)
if "late joined" in t:
ok_join = True
break
except Exception:
break
print("mailbox-ok" if ok_join else f"mailbox-dead seen={seen}")
slow.close(); fast.close(); survivor.close()
PYEOF
)"
[[ "$r" == *mailbox-ok* ]] \
&& ok "WO_MAILBOX=8: slow member dropped, room survived and kept serving" \
|| bad "mailbox-backpressure" "$r"
kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV=""
# ---- 5. the ASan leg: functional matrix, zero leaks ----
if [[ -x "$ASAN" ]]; then
sed -i "s|runtime = \".*\"|runtime = \"$ASAN\"|" "$W/app/wo.toml"
rm -rf "$W/app/target"
"$WOC" "$W/app" >/dev/null 2>&1
serve "$((PORT0 + 3))" || bad "serve-asan" "no listener"
r="$(functional asan)"
kill -TERM "$SRV" 2>/dev/null
for _ in $(seq 1 60); do kill -0 "$SRV" 2>/dev/null || break; sleep 0.1; done
SRV=""
if [[ "$r" == *functional-ok* ]] \
&& ! tail -n "+$LEGFROM" "$SRVLOG" | grep -q "AddressSanitizer\|LeakSanitizer"; then
ok "ASan run clean (functional + drain, zero leaks)"
else
bad "asan" "$(tail -n "+$LEGFROM" "$SRVLOG" | grep -m1 -E 'ERROR|SUMMARY' || echo "$r")"
fi
else
bad "asan" "runtime/build/wovm_asan missing — make -C runtime wovm-asan"
fi
echo
printf 'chat-accept: %d checks, %d failures\n' "$((pass + fail))" "$fail"
[[ $fail -eq 0 ]]

View file

@ -26,6 +26,24 @@ QUICK = "--quick" in sys.argv
WRITE_BASELINE = "--write-baseline" in sys.argv
N = 2000 if QUICK else 20000
# databasev2 4: the write-concurrent leg. `mix` writes on one op in ten with
# C=4, so group commit had almost nothing to batch there (measured mean batch
# 1.01, peak 3) — a property of that workload, not of the mechanism. C is high
# on purpose: batching is a function of how many writes are in flight, and
# measured mean batch rose 1.13 -> 1.76 -> 5.35 at C = 4 -> 16 -> 64.
WMIX_N = 4000 if QUICK else 20000
WMIX_C = 32 if QUICK else 64
# databasev2 3: the checkpoint leg. Ages a store by UPDATING the same rows, so
# history grows while the live set does not — otherwise the leg measures insert
# throughput instead of compaction.
CKPT_SEED = 2000 if QUICK else 5000
CKPT_OPS = 8000 if QUICK else 20000
# The stop-the-world budget. 50ms is a stall a serving process can absorb
# without a client noticing a timeout; measured at ~13ms for a 2MB live set,
# so this leaves real headroom while still failing before a stall becomes
# user-visible. Compaction is O(live rows), so this budget is what eventually
# forces the incremental design the spec deliberately did not buy in advance.
CKPT_PAUSE_BUDGET_US = 50000
MSG_N = 20000 if QUICK else 200000
WAL_N = 800 if QUICK else 4000
CRASH_REPS = 1 if QUICK else 3
@ -95,6 +113,55 @@ def parse_metrics(lines, into, prefix):
if m:
into[f"{prefix}.msgrate.msgs_sec"] = int(m.group(2))
def wmix_leg(metrics, tag, env, data):
"""Every op a durable write, WMIX_C at once — the leg that actually
exercises group commit.
It reuses the store the `all` run just seeded (a fresh process replays it,
so `kmod` is there) and asks the runtime for its group-commit counters via
WO_WAL_STATS. The counters matter as much as the throughput: if batches are
always one the mechanism is inert and any throughput change came from
somewhere else, so a payoff would be attributed to the wrong cause."""
e = dict(env); e["WO_WAL_STATS"] = "1"
rc, lines, _, _ = run(["wmix", str(WMIX_N), str(WMIX_C)], e, 1800)
if rc != 0:
bad(f"{tag}.wmix", f"rc={rc} tail={lines[-2:]}")
return
ops = p50 = p99 = None
batches = records = peak_batch = peak_staged = None
for l in lines:
f = l.split()
if f and f[0] == "wmix" and len(f) == 5:
ops, p50, p99 = int(f[2]), int(f[3]), int(f[4])
elif f and f[0] == "walstats":
kv = dict(x.split("=", 1) for x in f[1:] if "=" in x)
batches = int(kv.get("batches", 0)); records = int(kv.get("records", 0))
peak_batch = int(kv.get("peak_batch", 0)); peak_staged = int(kv.get("peak_staged", 0))
if ops is None or batches is None:
bad(f"{tag}.wmix", "no report or no walstats line")
return
metrics[f"{tag}.wmix.ops_sec"] = ops
metrics[f"{tag}.wmix.p50us"] = p50
metrics[f"{tag}.wmix.p99us"] = p99
metrics[f"{tag}.wmix.peak_batch"] = peak_batch
metrics[f"{tag}.wmix.peak_staged"] = peak_staged
mean = round(records / batches, 2) if batches else 0
metrics[f"{tag}.wmix.mean_batch"] = mean
ok(f"{tag}.wmix: {ops} ops/sec, p50 {p50}us p99 {p99}us; "
f"{records} records over {batches} barriers (mean {mean}, peak {peak_batch}), "
f"peak staged {peak_staged}B")
# The gate that matters. Only the MULTI-shard leg can batch: a worker's
# statements marshal to shard 0 and queue, while shard-0 statements run
# inline and commit one at a time by design (see db.c).
if tag.endswith(".sN"):
if mean > 1.0:
ok(f"{tag}.wmix batches form (mean {mean} > 1)")
else:
bad(f"{tag}.wmix-inert",
f"mean batch {mean} — group commit is not engaging, so a "
f"throughput change would not be attributable to it")
def campaign():
metrics = {}
ncores = os.cpu_count() or 1
@ -123,6 +190,8 @@ def campaign():
bad(f"{tag}.mix.fds", f"grew {fdg}")
else:
ok(f"{tag}.mix.fds flat")
if flavor == "durable" and data:
wmix_leg(metrics, tag, env, data)
if data: shutil.rmtree(data, ignore_errors=True)
# msgrate once per shard count, RAM only (no store dependency)
for shards in (1, ncores):
@ -235,6 +304,49 @@ def tolerance_for(key):
if key.startswith("ceiling."): return 100
if key.startswith("randread."): return 100
if key.startswith("replay."): return 100
# databasev2 4: batch SHAPE follows arrival timing, so gating it tightly
# would gate the scheduler — what must hold is that the mean exceeds one
# under contention, which wmix_leg asserts directly against the live run.
# wmix's throughput and latency are NOT waived: they are the payoff, and a
# blanket waiver here would have left the whole leg ungated.
if key.endswith((".wmix.mean_batch", ".wmix.peak_batch", ".wmix.peak_staged")):
return 100
# databasev2 4: DURABLE multi-shard p99 is an fsync TAIL, and group commit
# made it both noisier and legitimately higher. Measured across three full
# runs of the same build, durable.sN.mixread.p99 was 1043 / 2318 / 4147 us
# and wmix.p99 8758 / 20000 — a 2-4x spread with the box near idle, because
# a barrier now blocks the owner shard LONGER (more records per fsync) even
# though it blocks LESS OFTEN. That is the trade group commit makes on a
# single-threaded owner, and part B (async submission) is what would undo
# it. Gating a 2-4x-variable tail at 50% gates the disk, not the engine, so
# the FLOOR is the real guard here — and it is not slack: mixread's floor
# (4172us) came within 25us of tripping on the worst run.
if key.startswith("durable.sN.") and key.endswith(".p99us"):
# Widened again 2026-08-29 with more evidence: mixread p99 was measured
# at 1043 / 2318 / 4147us and mixwrite at 1623 / 4446us across runs of
# the SAME build on a near-idle box — a 3-4x spread. 100% was still
# gating the disk. The FLOOR stays the real guard and is not slack:
# mixread's came within 25us of tripping on the worst run observed.
return 300
# databasev2 3: the RECLAIM ratio is structural and gated tightly — it is
# the feature's whole claim. Boot time and the pause are wall-clock on a
# shared box and are not: waiving them all would have left the leg ungated,
# which is the mistake part A's task 4 made and had to undo.
if key in ("ckpt.boot_off_ms", "ckpt.boot_on_ms", "ckpt.pause_us_max",
"ckpt.compactions", "ckpt.bytes_off", "ckpt.bytes_on"):
return 400
# compaction BANDWIDTH is the engine's own property, so it is gated for
# real — it is what regressed 8x when the dump was fsyncing per flush
if key == "ckpt.pause_us_per_mb":
return 100
# msgrate is actor-to-actor throughput and is scheduling-bound, so its
# run-to-run spread is far wider than its old 15%. MEASURED across the 10
# full runs recorded on 2026-08-28/29 — several of them predating the
# checkpoint work — it ranged 10.7M to 17.9M msgs/sec, a 1.67x spread. A
# 15% gate on that gates the scheduler and fails intermittently whatever
# the engine does. Pre-existing; found while closing databasev2 3, not
# caused by it.
if ".msgrate." in key: return 70
if ".mixread." in key or ".mixwrite." in key: return 50
if ".sN." in key: return 50
if ".read." in key or ".query." in key: return 50
@ -246,7 +358,11 @@ def write_baseline(metrics):
"tolerances come from tolerance_for() in the driver"}}
for k, v in sorted(metrics.items()):
if k.endswith(("rss_growth_kb", "fd_growth")): continue
higher = k.endswith(("ops_sec", "msgs_sec"))
# reclaim_x: MORE reclaimed is better. Recorded as lower-is-better by
# the default detector, which would have passed "no reclaim at all" and
# failed an improvement — the feature's central claim, gated backwards.
higher = k.endswith(("ops_sec", "msgs_sec", "mean_batch", "peak_batch",
"reclaim_x"))
floor_div = 8 if k.endswith("msgs_sec") else 4
# latency floors never sit below 100µs: at post-index µs scale a
# 4×1µs "catastrophe line" is noise; the tripwire means "µs became
@ -520,9 +636,12 @@ def randread(metrics):
def wal_used(data_dir):
"""Bytes actually written across the store's WAL files.
The non-zero prefix, NOT the file size: shard WALs are fallocate'd to
1 MiB up front, so getsize reports 1048576 for an empty store and proves
nothing. Same reason scripts/residency-accept.sh measures it this way."""
The non-zero prefix, NOT the file size: shard WALs are preallocated, so
getsize reports the preallocation (1 MiB) even for an empty store. Same
reason scripts/residency-accept.sh measures it this way.
databasev2 1 and databasev2 3 each grew their own copy of this helper on
separate branches; this is the single one they now share."""
total = 0
for name in sorted(os.listdir(data_dir)):
with open(os.path.join(data_dir, name), "rb") as f:
@ -621,6 +740,96 @@ def replay(metrics):
metrics["replay.history_penalty_x"] = round(penalty, 2)
ok(f"replay: identical dataset, {penalty:.2f}x the boot cost from history alone "
f"({ins_ms:.0f} -> {his_ms:.0f} ms) -- what a checkpoint would collapse")
def checkpoint_leg(metrics):
"""Space reclaimed, boot time, and the stop-the-world PAUSE.
The same workload runs twice, differing only in whether checkpointing can
fire: an enormous floor disables it, a small one lets it. Comparing two runs
of one build is what isolates compaction from everything else the workload
does.
Boot is measured with the sample's `boot` mode, which does nothing at all —
with WO_DATA set the runtime replays the whole log before main runs, so a
mode with no work of its own is the only honest way to price replay."""
ncores = os.cpu_count() or 1
out = {}
for name, knobs in (("off", {"WO_CHECKPOINT_BYTES": "1000000000"}),
("on", {"WO_CHECKPOINT_BYTES": "65536", "WO_CHECKPOINT_RATIO": "2"})):
data = os.path.join(ROOT, "bench", f"tmp.{os.getpid()}.ckpt.{name}")
shutil.rmtree(data, ignore_errors=True); os.makedirs(data, exist_ok=True)
env = {"WO_DATA": data, "WO_SHARDS": str(ncores), "WO_WAL_STATS": "1"}
env.update(knobs)
rc, _, _, _ = run(["seed", str(CKPT_SEED)], env, 1800)
if rc != 0:
bad(f"ckpt.{name}.seed", f"rc={rc}"); shutil.rmtree(data, ignore_errors=True); return
rc, lines, _, _ = run(["wmix", str(CKPT_OPS), "16"], env, 1800)
if rc != 0:
bad(f"ckpt.{name}.age", f"rc={rc}"); shutil.rmtree(data, ignore_errors=True); return
stats = {}
for l in lines:
f = l.split()
if f and f[0] == "walstats":
stats = dict(x.split("=", 1) for x in f[1:] if "=" in x)
used = wal_used(data)
# NOT through run(): it samples RSS on a 250ms poll, so every timing it
# produces floors at the poll quantum — boot measured that way reported
# 251ms both with and without checkpointing, which is the harness's
# clock, not the engine's. Median of 3 because this is wall-clock.
benv = dict(os.environ)
benv.update({"WO_DATA": data, "WO_SHARDS": str(ncores)})
samples = []
brc = 0
for _ in range(3):
t0 = time.monotonic()
pr = subprocess.run([BIN, "boot"], stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL, env=benv, timeout=900)
samples.append((time.monotonic() - t0) * 1000.0)
brc = pr.returncode or brc
boot_ms = sorted(samples)[1]
if brc != 0:
bad(f"ckpt.{name}.boot", f"rc={brc}"); shutil.rmtree(data, ignore_errors=True); return
out[name] = (used, boot_ms, stats)
shutil.rmtree(data, ignore_errors=True)
(off_b, off_boot, _), (on_b, on_boot, st) = out["off"], out["on"]
comps = int(st.get("compactions", 0))
if comps == 0:
bad("ckpt.inert", "no compaction ran — the leg proves nothing about checkpointing")
return
metrics["ckpt.compactions"] = comps
metrics["ckpt.bytes_off"] = off_b
metrics["ckpt.bytes_on"] = on_b
metrics["ckpt.reclaim_x"] = round(off_b / max(on_b, 1), 2)
metrics["ckpt.boot_off_ms"] = int(round(off_boot))
metrics["ckpt.boot_on_ms"] = int(round(on_boot))
metrics["ckpt.pause_us_max"] = int(st.get("compact_us_max", 0))
# The RAW pause scales with the live set, and this workload's live set is
# not fixed: wmix's hist_dump inserts a row per latency bucket, so a noisier
# box produces more buckets, more rows, and a longer pause. Gating the raw
# number against a baseline therefore gates the box. What belongs to the
# ENGINE is the rate, so that is what carries a real tolerance; the raw
# pause keeps the absolute budget assertion below as its guard.
cb = int(st.get("compacted_bytes", 0))
if cb > 0 and metrics["ckpt.pause_us_max"] > 0:
metrics["ckpt.pause_us_per_mb"] = int(round(
metrics["ckpt.pause_us_max"] / (cb / (1024.0 * 1024.0))))
ok(f"ckpt: {off_b} -> {on_b} bytes ({metrics['ckpt.reclaim_x']}x reclaimed) over "
f"{comps} compactions; boot {off_boot:.0f} -> {on_boot:.0f} ms; "
f"stop-the-world pause max {metrics['ckpt.pause_us_max']}us "
f"({metrics.get('ckpt.pause_us_per_mb', 0)}us/MB)")
# the space claim is the point of the feature, so it is asserted, not just recorded
if off_b <= on_b:
bad("ckpt.no-reclaim", f"checkpointing did not shrink the log ({off_b} -> {on_b})")
else:
ok(f"ckpt: the log is smaller with checkpointing on")
# THE BUDGET. Stated, not assumed — the spec refused to assume it.
if metrics["ckpt.pause_us_max"] > CKPT_PAUSE_BUDGET_US:
bad("ckpt.pause-budget",
f"stop-the-world pause {metrics['ckpt.pause_us_max']}us exceeds the stated "
f"{CKPT_PAUSE_BUDGET_US}us budget — alternatives (incremental copy, "
f"fork-and-dump) are bought against THIS number")
else:
ok(f"ckpt: pause within budget ({metrics['ckpt.pause_us_max']} <= {CKPT_PAUSE_BUDGET_US}us)")
def main():
@ -639,6 +848,7 @@ def main():
ceiling(metrics)
randread(metrics)
replay(metrics)
checkpoint_leg(metrics)
os.makedirs(RESULTS_DIR, exist_ok=True)
stamp = time.strftime("%Y%m%d-%H%M%S")
out = os.path.join(RESULTS_DIR, f"run-{stamp}{'-quick' if QUICK else ''}.json")

View file

@ -51,6 +51,14 @@ if [[ ! -x "$WOVM" ]]; then
fi
WORK="$(mktemp -d "${TMPDIR:-/tmp}/lw-accept.XXXXXX")"
# stable, tailable log for the example app: the per-run work dir is deleted on
# exit, so a developer had nothing to follow. `tail -F /tmp/log-watcher.log`.
# Each invocation keeps its own $WORK/*.out (the checks grep those) and is
# ALSO teed here, banner-separated, so one file holds the whole run.
APPLOG="/tmp/log-watcher.log"
: > "$APPLOG"
echo "app log: $APPLOG (tail -F \"$APPLOG\" to follow)"
# LW_ACCEPT_KEEP=1 leaves the work directory (image, logs, cron.d, the
# server's own stdout) in place — what you want the moment a check fails.
cleanup() {
@ -89,7 +97,8 @@ fi
# watcher to decide the burst is over.
LOG="$WORK/app.log"
: >"$LOG"
timeout -k 2 12 "$WOVM" "$IMAGE" watch "$LOG" 2 1 >"$WORK/watch.out" 2>&1 &
printf '\n===== watch =====\n' >>"$APPLOG"
timeout -k 2 12 "$WOVM" "$IMAGE" watch "$LOG" 2 1 > >(tee -a "$APPLOG" >"$WORK/watch.out") 2>&1 &
WATCH_PID=$!
sleep 2
printf 'info service starting\n' >>"$LOG"
@ -110,6 +119,7 @@ CRON="$WORK/cron.d"
mkdir -p "$CRON"
printf '* * * * * root /usr/bin/backup.sh > /var/log/backup.log 2>&1\n' >"$CRON/backup"
timeout 8 "$WOVM" "$IMAGE" run "$CRON" >"$WORK/run.out" 2>&1
{ printf '\n===== run =====\n'; cat "$WORK/run.out"; } >>"$APPLOG"
if grep -q "^SCHEDULE /var/log/backup.log" "$WORK/run.out"; then
ok "run (parsed and scheduled the cron entry)"
else
@ -124,7 +134,8 @@ EOF
# -k: `env.stopping()` installs a SIGTERM handler that only sets a flag, and
# the serve loop is blocked in accept(), so a plain TERM is swallowed — the
# process needs a KILL to actually stop (recorded in docs/00-status.md).
timeout -k 2 20 "$WOVM" "$IMAGE" mcp "$CRON" "$WORK/cfg.json" >"$WORK/mcp.out" 2>&1 &
printf '\n===== mcp =====\n' >>"$APPLOG"
timeout -k 2 20 "$WOVM" "$IMAGE" mcp "$CRON" "$WORK/cfg.json" > >(tee -a "$APPLOG" >"$WORK/mcp.out") 2>&1 &
SRV_PID=$!
sleep 2
@ -256,7 +267,8 @@ if [[ -n "${LW_SOAK:-}" ]]; then
soak_mode() {
local name="$1" load_fn="$2"
shift 2
"$WOVM" "$IMAGE" "$@" >"$WORK/soak-$name.out" 2>&1 &
printf '\n===== soak %s =====\n' "$name" >>"$APPLOG"
"$WOVM" "$IMAGE" "$@" > >(tee -a "$APPLOG" >"$WORK/soak-$name.out") 2>&1 &
local pid=$! rss0 fd0 rss1 fd1 drss dfd deadline i
sleep 3 # first-touch pages and the first work cycle
if ! kill -0 "$pid" 2>/dev/null; then

View file

@ -56,6 +56,11 @@ fi
PORT=$((8500 + RANDOM % 400))
DATA="$W/data"; mkdir -p "$DATA"
# stable, tailable server log — the per-run temp dir is deleted on exit
SRVLOG="/tmp/site.log"
: > "$SRVLOG"
echo "server log: $SRVLOG (tail -F \"$SRVLOG\" to follow)"
hit() { # path [method] [data] [token] -> "STATUS|BODY" (redirects not followed)
python3 - "$PORT" "$1" "${2:-GET}" "${3:-}" "${4:-}" <<'PYEOF'
@ -97,7 +102,8 @@ expect() { # name got want_status want_substr
}
serve() {
SITE_TOKEN=s3cr3t WO_DATA="$DATA" "$W/app/target/site" "$PORT" >>"$W/srv.out" 2>&1 &
printf '\n===== serve — port %s =====\n' "$PORT" >>"$SRVLOG"
SITE_TOKEN=s3cr3t WO_DATA="$DATA" "$W/app/target/site" "$PORT" >>"$SRVLOG" 2>&1 &
SRV=$!
for _ in $(seq 1 40); do
[[ "$(hit /health 2>/dev/null)" == 200* ]] && return 0

View file

@ -83,9 +83,19 @@ else
fi
DATA="$W/data"; mkdir -p "$DATA"
WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >"$W/srv.out" 2>&1 &
# stable, tailable server log: the per-run temp dir is deleted on exit, so a
# developer had nothing to follow. `tail -F /tmp/web-app.log` while this runs.
SRVLOG="/tmp/web-app.log"
: > "$SRVLOG"
echo "server log: $SRVLOG (tail -F \"$SRVLOG\" to follow)"
printf '===== boot — port %s =====\n' "$PORT" >>"$SRVLOG"
LEGFROM=$(( $(wc -l < "$SRVLOG") + 1 ))
WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >>"$SRVLOG" 2>&1 &
SRV=$!
for _ in $(seq 1 40); do grep -q listening "$W/srv.out" 2>/dev/null && break; sleep 0.1; done
for _ in $(seq 1 40); do
tail -n "+$LEGFROM" "$SRVLOG" 2>/dev/null | grep -q listening && break
sleep 0.1
done
# one tiny HTTP client; python is already a repo test dependency
hit() { # method path [body] [auth: yes|no] [content-type] -> "STATUS|BODY"
@ -467,7 +477,8 @@ for _ in $(seq 1 30); do kill -0 "$SRV" 2>/dev/null || { stopped=0; break; }; sl
SRV=""
# ---- 15. restart persistence (WAL replay) ----
WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >>"$W/srv.out" 2>&1 &
printf '\n===== restart (WAL replay) — port %s =====\n' "$PORT" >>"$SRVLOG"
WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >>"$SRVLOG" 2>&1 &
SRV=$!
sleep 0.5
expect "product survives a restart (WAL)" "$(hit GET /products)" 200 '"name":"mug"'

View file

@ -0,0 +1,3 @@
died: boom
died: late
done

View file

@ -0,0 +1,36 @@
use time
-- iteration 24 T4: actor death is OBSERVABLE. The observer names its own
-- notice message; the watched actor trapping uncaught (the runtime's
-- stderr line) delivers it. Monitoring an ALREADY dead actor fires
-- immediately. WO_SHARDS=1 (the runner) keeps the order deterministic.
class Note {
who: Text
}
class Watch {
pad: Int
fn receive(msg: Note) {
print("died: ${msg.who}");
}
}
class Boom {
pad: Int
fn receive(msg: Note) {
let z = len(msg.who) - len(msg.who);
let q = 1 / z;
}
}
fn main() -> Int {
let obs: actor Note = spawn Watch { pad: 0 };
let b: actor Note = spawn Boom { pad: 0 };
monitor(b, obs, Note { who: "boom" });
send(b, Note { who: "x" });
time.sleep(100);
monitor(b, obs, Note { who: "late" });
time.sleep(100);
print("done");
return 0;
}

View file

@ -0,0 +1,3 @@
tick: now
tick: armed
done

View file

@ -0,0 +1,23 @@
use time
-- iteration 24 T5: a timer is a MESSAGE. time.after arms a one-shot on
-- this shard; the target receives it like any send. ms <= 0 delivers now.
class Tick {
tag: Text
}
class Sink {
pad: Int
fn receive(msg: Tick) {
print("tick: ${msg.tag}");
}
}
fn main() -> Int {
let a: actor Tick = spawn Sink { pad: 0 };
time.after(30, a, Tick { tag: "armed" });
time.after(0, a, Tick { tag: "now" });
time.sleep(150);
print("done");
return 0;
}

View file

@ -0,0 +1,3 @@
stale gen 1 ignored
fired gen 2
done

View file

@ -0,0 +1,28 @@
use time
-- iteration 24 T5: the CANCEL idiom — no cancel builtin, a generation
-- counter instead. The actor bumps its generation; a stale timer's
-- message names the old one and is recognized and ignored on arrival.
class Timer {
gen: Int
}
class Gate {
gen: Int
fn receive(msg: Timer) {
if msg.gen == self.gen {
print("fired gen ${msg.gen}");
} else {
print("stale gen ${msg.gen} ignored");
}
}
}
fn main() -> Int {
let g: actor Timer = spawn Gate { gen: 2 };
time.after(30, g, Timer { gen: 1 });
time.after(60, g, Timer { gen: 2 });
time.sleep(200);
print("done");
return 0;
}