Merge master into db-residency-doctrine — and close the two half-exposed features
The branch was 17 ahead / 25 behind with 11 conflicting files, and drifting further: db.c had been rewritten twice on master since (group commit, then compaction). Resolved rather than rebased so both histories stay legible. Conflicts, and how each was settled: - db.c: BOTH semantics kept. Master's fatal path and compaction check now sit behind the branch's `table_is_durable` predicate, in all three inline arms — a volatile table reaches neither the barrier nor the compaction check - db-bench sample: every mode from both sides (growth, growth-verify, randread, replayseed, wmix) and ONE `boot` mode, which both sides had added independently - db-bench.py: all six legs kept. Both sides had also grown the same WAL-size helper under different names; collapsed into one - perf-targets: the branch's §5 (RAM ceiling) then master's §6/§7 — master's numbering had already assumed a §5 it did not have - story frontmatter: master's `status` (the landing truth) plus the branch's `readiness` axis. 03 would have read `done` + `refine`, which is a contradiction — it was brainstormed and landed on master, so `ready` - board: both standup blocks newest-first; master's chain rows (a superset); the branch's databasev2 1-2 rows with master's 3-4. Fixed a stray `|` in master's row 3 - baseline: master's, then REGENERATED from a full campaign — 143 metrics, 132 checks, 0 failures with both sides' legs present TWO HALF-EXPOSED FEATURES FIXED, because the merge rule is that master gets no feature that is honoured in name only: - `resident: keys` PARSED, set a .wob flag, and did nothing: rows stayed fully resident. A developer could declare a 120 GB table keys-resident, watch it compile, and be OOM-killed. The loader now REFUSES it with a message naming what to write instead, until tasks 5c/5d land. The compiler still parses it and its AST golden still passes, so the grammar work stays tested - `durable: false` was honoured ONLY on the inline path. wo_db_exec_req had no guard at all, so a volatile table written from an actor on a worker shard would still be logged — precisely porch's session-table case, and precisely what iteration 2 exists to provide. All three request-path arms now carry the same predicate. Found by reading the merged code, not by a test: the obvious probe runs main() on the primary and therefore only exercises the inline path Verified on the merged tree: wovm-test 0, woc-test 0, oop-e2e 122/0, residency-accept 8/0, db-bench 132/0, linkcheck clean. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
commit
02b4b13a52
56 changed files with 5200 additions and 352 deletions
|
|
@ -1,22 +1,70 @@
|
|||
{
|
||||
"_config": {
|
||||
"N": 2000,
|
||||
"crash_reps": 1,
|
||||
"msg_n": 20000,
|
||||
"N": 20000,
|
||||
"crash_reps": 3,
|
||||
"msg_n": 200000,
|
||||
"note": "refresh only with a commit that says why; tolerances come from tolerance_for() in the driver",
|
||||
"wal_n": 800
|
||||
"wal_n": 4000
|
||||
},
|
||||
"ceiling.rows_recovered": {
|
||||
"dir": "lower",
|
||||
"floor": 159744,
|
||||
"floor": 159492,
|
||||
"tolerance_pct": 100,
|
||||
"value": 39936
|
||||
"value": 39873
|
||||
},
|
||||
"ckpt.boot_off_ms": {
|
||||
"dir": "lower",
|
||||
"floor": 456,
|
||||
"tolerance_pct": 400,
|
||||
"value": 114
|
||||
},
|
||||
"ckpt.boot_on_ms": {
|
||||
"dir": "lower",
|
||||
"floor": 256,
|
||||
"tolerance_pct": 400,
|
||||
"value": 64
|
||||
},
|
||||
"ckpt.bytes_off": {
|
||||
"dir": "lower",
|
||||
"floor": 7876676,
|
||||
"tolerance_pct": 400,
|
||||
"value": 1969169
|
||||
},
|
||||
"ckpt.bytes_on": {
|
||||
"dir": "lower",
|
||||
"floor": 3696192,
|
||||
"tolerance_pct": 400,
|
||||
"value": 924048
|
||||
},
|
||||
"ckpt.compactions": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 400,
|
||||
"value": 6
|
||||
},
|
||||
"ckpt.pause_us_max": {
|
||||
"dir": "lower",
|
||||
"floor": 33912,
|
||||
"tolerance_pct": 400,
|
||||
"value": 8478
|
||||
},
|
||||
"ckpt.pause_us_per_mb": {
|
||||
"dir": "lower",
|
||||
"floor": 65848,
|
||||
"tolerance_pct": 100,
|
||||
"value": 16462
|
||||
},
|
||||
"ckpt.reclaim_x": {
|
||||
"dir": "higher",
|
||||
"floor": 0.0,
|
||||
"tolerance_pct": 15,
|
||||
"value": 2.13
|
||||
},
|
||||
"durable.s1.mixread.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 2237,
|
||||
"floor": 2452,
|
||||
"tolerance_pct": 50,
|
||||
"value": 8949
|
||||
"value": 9809
|
||||
},
|
||||
"durable.s1.mixread.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -28,31 +76,31 @@
|
|||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 2
|
||||
"value": 12
|
||||
},
|
||||
"durable.s1.mixwrite.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 248,
|
||||
"floor": 272,
|
||||
"tolerance_pct": 50,
|
||||
"value": 994
|
||||
"value": 1089
|
||||
},
|
||||
"durable.s1.mixwrite.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 820,
|
||||
"floor": 1704,
|
||||
"tolerance_pct": 50,
|
||||
"value": 205
|
||||
"value": 426
|
||||
},
|
||||
"durable.s1.mixwrite.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 872,
|
||||
"floor": 1984,
|
||||
"tolerance_pct": 50,
|
||||
"value": 218
|
||||
"value": 496
|
||||
},
|
||||
"durable.s1.query.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 335570,
|
||||
"floor": 306372,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1342281
|
||||
"value": 1225490
|
||||
},
|
||||
"durable.s1.query.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -68,9 +116,9 @@
|
|||
},
|
||||
"durable.s1.read.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 347705,
|
||||
"floor": 307389,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1390820
|
||||
"value": 1229558
|
||||
},
|
||||
"durable.s1.read.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -82,85 +130,121 @@
|
|||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 2
|
||||
"value": 1
|
||||
},
|
||||
"durable.s1.seed.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1103,
|
||||
"floor": 1095,
|
||||
"tolerance_pct": 15,
|
||||
"value": 4415
|
||||
"value": 4381
|
||||
},
|
||||
"durable.s1.seed.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 828,
|
||||
"floor": 848,
|
||||
"tolerance_pct": 15,
|
||||
"value": 207
|
||||
"value": 212
|
||||
},
|
||||
"durable.s1.seed.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 2092,
|
||||
"floor": 2432,
|
||||
"tolerance_pct": 15,
|
||||
"value": 523
|
||||
"value": 608
|
||||
},
|
||||
"durable.s1.wmix.mean_batch": {
|
||||
"dir": "higher",
|
||||
"floor": 0.0,
|
||||
"tolerance_pct": 100,
|
||||
"value": 1.0
|
||||
},
|
||||
"durable.s1.wmix.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 402,
|
||||
"tolerance_pct": 15,
|
||||
"value": 1611
|
||||
},
|
||||
"durable.s1.wmix.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 1764,
|
||||
"tolerance_pct": 15,
|
||||
"value": 441
|
||||
},
|
||||
"durable.s1.wmix.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 2684,
|
||||
"tolerance_pct": 15,
|
||||
"value": 671
|
||||
},
|
||||
"durable.s1.wmix.peak_batch": {
|
||||
"dir": "higher",
|
||||
"floor": 0,
|
||||
"tolerance_pct": 100,
|
||||
"value": 1
|
||||
},
|
||||
"durable.s1.wmix.peak_staged": {
|
||||
"dir": "lower",
|
||||
"floor": 196,
|
||||
"tolerance_pct": 100,
|
||||
"value": 49
|
||||
},
|
||||
"durable.s1.write.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1155,
|
||||
"floor": 573,
|
||||
"tolerance_pct": 15,
|
||||
"value": 4620
|
||||
"value": 2294
|
||||
},
|
||||
"durable.s1.write.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 832,
|
||||
"floor": 1760,
|
||||
"tolerance_pct": 15,
|
||||
"value": 208
|
||||
"value": 440
|
||||
},
|
||||
"durable.s1.write.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 1948,
|
||||
"floor": 2716,
|
||||
"tolerance_pct": 15,
|
||||
"value": 487
|
||||
"value": 679
|
||||
},
|
||||
"durable.sN.mixread.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1112,
|
||||
"floor": 1183,
|
||||
"tolerance_pct": 50,
|
||||
"value": 4450
|
||||
"value": 4733
|
||||
},
|
||||
"durable.sN.mixread.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 236,
|
||||
"floor": 244,
|
||||
"tolerance_pct": 50,
|
||||
"value": 59
|
||||
"value": 61
|
||||
},
|
||||
"durable.sN.mixread.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 13100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 3275
|
||||
"floor": 16200,
|
||||
"tolerance_pct": 300,
|
||||
"value": 4050
|
||||
},
|
||||
"durable.sN.mixwrite.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 123,
|
||||
"floor": 131,
|
||||
"tolerance_pct": 50,
|
||||
"value": 494
|
||||
"value": 525
|
||||
},
|
||||
"durable.sN.mixwrite.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 1160,
|
||||
"floor": 2172,
|
||||
"tolerance_pct": 50,
|
||||
"value": 290
|
||||
"value": 543
|
||||
},
|
||||
"durable.sN.mixwrite.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 2944,
|
||||
"tolerance_pct": 50,
|
||||
"value": 736
|
||||
"floor": 16440,
|
||||
"tolerance_pct": 300,
|
||||
"value": 4110
|
||||
},
|
||||
"durable.sN.query.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 287356,
|
||||
"floor": 308451,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1149425
|
||||
"value": 1233806
|
||||
},
|
||||
"durable.sN.query.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -171,14 +255,14 @@
|
|||
"durable.sN.query.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"tolerance_pct": 300,
|
||||
"value": 1
|
||||
},
|
||||
"durable.sN.read.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 192752,
|
||||
"floor": 248188,
|
||||
"tolerance_pct": 50,
|
||||
"value": 771010
|
||||
"value": 992752
|
||||
},
|
||||
"durable.sN.read.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -189,44 +273,80 @@
|
|||
"durable.sN.read.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"tolerance_pct": 300,
|
||||
"value": 2
|
||||
},
|
||||
"durable.sN.seed.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1142,
|
||||
"floor": 1104,
|
||||
"tolerance_pct": 50,
|
||||
"value": 4571
|
||||
"value": 4418
|
||||
},
|
||||
"durable.sN.seed.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 836,
|
||||
"floor": 844,
|
||||
"tolerance_pct": 50,
|
||||
"value": 209
|
||||
"value": 211
|
||||
},
|
||||
"durable.sN.seed.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 1916,
|
||||
"floor": 2188,
|
||||
"tolerance_pct": 300,
|
||||
"value": 547
|
||||
},
|
||||
"durable.sN.wmix.mean_batch": {
|
||||
"dir": "higher",
|
||||
"floor": 1.0,
|
||||
"tolerance_pct": 100,
|
||||
"value": 6.22
|
||||
},
|
||||
"durable.sN.wmix.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1504,
|
||||
"tolerance_pct": 50,
|
||||
"value": 479
|
||||
"value": 6017
|
||||
},
|
||||
"durable.sN.wmix.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 27184,
|
||||
"tolerance_pct": 50,
|
||||
"value": 6796
|
||||
},
|
||||
"durable.sN.wmix.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 37484,
|
||||
"tolerance_pct": 300,
|
||||
"value": 9371
|
||||
},
|
||||
"durable.sN.wmix.peak_batch": {
|
||||
"dir": "higher",
|
||||
"floor": 15,
|
||||
"tolerance_pct": 100,
|
||||
"value": 60
|
||||
},
|
||||
"durable.sN.wmix.peak_staged": {
|
||||
"dir": "lower",
|
||||
"floor": 11760,
|
||||
"tolerance_pct": 100,
|
||||
"value": 2940
|
||||
},
|
||||
"durable.sN.write.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1010,
|
||||
"floor": 580,
|
||||
"tolerance_pct": 50,
|
||||
"value": 4040
|
||||
"value": 2320
|
||||
},
|
||||
"durable.sN.write.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 832,
|
||||
"floor": 1760,
|
||||
"tolerance_pct": 50,
|
||||
"value": 208
|
||||
"value": 440
|
||||
},
|
||||
"durable.sN.write.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 1996,
|
||||
"tolerance_pct": 50,
|
||||
"value": 499
|
||||
"floor": 2688,
|
||||
"tolerance_pct": 300,
|
||||
"value": 672
|
||||
},
|
||||
"growth.available": {
|
||||
"dir": "lower",
|
||||
|
|
@ -236,9 +356,9 @@
|
|||
},
|
||||
"growth.int.noswap.bytes_per_row": {
|
||||
"dir": "lower",
|
||||
"floor": 392,
|
||||
"floor": 440,
|
||||
"tolerance_pct": 10,
|
||||
"value": 98
|
||||
"value": 110
|
||||
},
|
||||
"growth.int.noswap.doublings": {
|
||||
"dir": "lower",
|
||||
|
|
@ -266,21 +386,21 @@
|
|||
},
|
||||
"growth.int.noswap.rows": {
|
||||
"dir": "lower",
|
||||
"floor": 80000,
|
||||
"floor": 800000,
|
||||
"tolerance_pct": 100,
|
||||
"value": 20000
|
||||
"value": 200000
|
||||
},
|
||||
"growth.int.noswap.rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 23968,
|
||||
"floor": 168528,
|
||||
"tolerance_pct": 100,
|
||||
"value": 5992
|
||||
"value": 42132
|
||||
},
|
||||
"growth.int.swap.bytes_per_row": {
|
||||
"dir": "lower",
|
||||
"floor": 392,
|
||||
"floor": 440,
|
||||
"tolerance_pct": 10,
|
||||
"value": 98
|
||||
"value": 110
|
||||
},
|
||||
"growth.int.swap.doublings": {
|
||||
"dir": "lower",
|
||||
|
|
@ -308,21 +428,21 @@
|
|||
},
|
||||
"growth.int.swap.rows": {
|
||||
"dir": "lower",
|
||||
"floor": 80000,
|
||||
"floor": 800000,
|
||||
"tolerance_pct": 100,
|
||||
"value": 20000
|
||||
"value": 200000
|
||||
},
|
||||
"growth.int.swap.rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 23984,
|
||||
"floor": 168576,
|
||||
"tolerance_pct": 100,
|
||||
"value": 5996
|
||||
"value": 42144
|
||||
},
|
||||
"growth.text.noswap.bytes_per_row": {
|
||||
"dir": "lower",
|
||||
"floor": 1288,
|
||||
"floor": 1284,
|
||||
"tolerance_pct": 10,
|
||||
"value": 322
|
||||
"value": 321
|
||||
},
|
||||
"growth.text.noswap.doublings": {
|
||||
"dir": "lower",
|
||||
|
|
@ -350,21 +470,21 @@
|
|||
},
|
||||
"growth.text.noswap.rows": {
|
||||
"dir": "lower",
|
||||
"floor": 80000,
|
||||
"floor": 800000,
|
||||
"tolerance_pct": 100,
|
||||
"value": 20000
|
||||
"value": 200000
|
||||
},
|
||||
"growth.text.noswap.rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 41216,
|
||||
"floor": 343312,
|
||||
"tolerance_pct": 100,
|
||||
"value": 10304
|
||||
"value": 85828
|
||||
},
|
||||
"growth.text.swap.bytes_per_row": {
|
||||
"dir": "lower",
|
||||
"floor": 1288,
|
||||
"floor": 1284,
|
||||
"tolerance_pct": 10,
|
||||
"value": 322
|
||||
"value": 321
|
||||
},
|
||||
"growth.text.swap.doublings": {
|
||||
"dir": "lower",
|
||||
|
|
@ -392,21 +512,21 @@
|
|||
},
|
||||
"growth.text.swap.rows": {
|
||||
"dir": "lower",
|
||||
"floor": 80000,
|
||||
"floor": 800000,
|
||||
"tolerance_pct": 100,
|
||||
"value": 20000
|
||||
"value": 200000
|
||||
},
|
||||
"growth.text.swap.rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 41216,
|
||||
"floor": 343328,
|
||||
"tolerance_pct": 100,
|
||||
"value": 10304
|
||||
"value": 85832
|
||||
},
|
||||
"ram.s1.mixread.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 2236,
|
||||
"floor": 22286,
|
||||
"tolerance_pct": 50,
|
||||
"value": 8947
|
||||
"value": 89144
|
||||
},
|
||||
"ram.s1.mixread.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -422,9 +542,9 @@
|
|||
},
|
||||
"ram.s1.mixwrite.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 248,
|
||||
"floor": 2476,
|
||||
"tolerance_pct": 50,
|
||||
"value": 994
|
||||
"value": 9904
|
||||
},
|
||||
"ram.s1.mixwrite.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -440,15 +560,15 @@
|
|||
},
|
||||
"ram.s1.msgrate.msgs_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 419322,
|
||||
"tolerance_pct": 15,
|
||||
"value": 3354579
|
||||
"floor": 1336469,
|
||||
"tolerance_pct": 70,
|
||||
"value": 10691756
|
||||
},
|
||||
"ram.s1.query.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 324675,
|
||||
"floor": 244857,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1298701
|
||||
"value": 979431
|
||||
},
|
||||
"ram.s1.query.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -464,9 +584,9 @@
|
|||
},
|
||||
"ram.s1.read.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 332889,
|
||||
"floor": 252270,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1331557
|
||||
"value": 1009081
|
||||
},
|
||||
"ram.s1.read.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -482,87 +602,87 @@
|
|||
},
|
||||
"ram.s1.seed.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 375939,
|
||||
"floor": 62904,
|
||||
"tolerance_pct": 15,
|
||||
"value": 1503759
|
||||
"value": 251616
|
||||
},
|
||||
"ram.s1.seed.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 15,
|
||||
"value": 0
|
||||
"value": 4
|
||||
},
|
||||
"ram.s1.seed.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 15,
|
||||
"value": 2
|
||||
"value": 9
|
||||
},
|
||||
"ram.s1.write.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 272628,
|
||||
"floor": 47770,
|
||||
"tolerance_pct": 15,
|
||||
"value": 1090512
|
||||
"value": 191080
|
||||
},
|
||||
"ram.s1.write.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 15,
|
||||
"value": 1
|
||||
"value": 8
|
||||
},
|
||||
"ram.s1.write.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 15,
|
||||
"value": 2
|
||||
"value": 10
|
||||
},
|
||||
"ram.sN.mixread.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 2241,
|
||||
"floor": 11218,
|
||||
"tolerance_pct": 50,
|
||||
"value": 8964
|
||||
"value": 44874
|
||||
},
|
||||
"ram.sN.mixread.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 228,
|
||||
"floor": 240,
|
||||
"tolerance_pct": 50,
|
||||
"value": 57
|
||||
"value": 60
|
||||
},
|
||||
"ram.sN.mixread.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 1412,
|
||||
"floor": 324,
|
||||
"tolerance_pct": 50,
|
||||
"value": 353
|
||||
"value": 81
|
||||
},
|
||||
"ram.sN.mixwrite.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 249,
|
||||
"floor": 1246,
|
||||
"tolerance_pct": 50,
|
||||
"value": 996
|
||||
"value": 4986
|
||||
},
|
||||
"ram.sN.mixwrite.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 252,
|
||||
"floor": 260,
|
||||
"tolerance_pct": 50,
|
||||
"value": 63
|
||||
"value": 65
|
||||
},
|
||||
"ram.sN.mixwrite.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 280,
|
||||
"floor": 356,
|
||||
"tolerance_pct": 50,
|
||||
"value": 70
|
||||
"value": 89
|
||||
},
|
||||
"ram.sN.msgrate.msgs_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 214795,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1718360
|
||||
"floor": 317323,
|
||||
"tolerance_pct": 70,
|
||||
"value": 2538586
|
||||
},
|
||||
"ram.sN.query.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 331125,
|
||||
"floor": 291545,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1324503
|
||||
"value": 1166180
|
||||
},
|
||||
"ram.sN.query.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -578,9 +698,9 @@
|
|||
},
|
||||
"ram.sN.read.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 340599,
|
||||
"floor": 317823,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1362397
|
||||
"value": 1271294
|
||||
},
|
||||
"ram.sN.read.p50us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -596,87 +716,87 @@
|
|||
},
|
||||
"ram.sN.seed.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 353606,
|
||||
"floor": 73305,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1414427
|
||||
"value": 293220
|
||||
},
|
||||
"ram.sN.seed.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1
|
||||
"value": 3
|
||||
},
|
||||
"ram.sN.seed.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 2
|
||||
"value": 7
|
||||
},
|
||||
"ram.sN.write.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 290697,
|
||||
"floor": 56810,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1162790
|
||||
"value": 227241
|
||||
},
|
||||
"ram.sN.write.p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 1
|
||||
"value": 6
|
||||
},
|
||||
"ram.sN.write.p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 50,
|
||||
"value": 2
|
||||
"value": 10
|
||||
},
|
||||
"randread.collapse_x": {
|
||||
"dir": "lower",
|
||||
"floor": 1172,
|
||||
"floor": 1084,
|
||||
"tolerance_pct": 100,
|
||||
"value": 293
|
||||
"value": 271
|
||||
},
|
||||
"randread.overcap.filled_rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 25680,
|
||||
"floor": 58144,
|
||||
"tolerance_pct": 100,
|
||||
"value": 6420
|
||||
"value": 14536
|
||||
},
|
||||
"randread.overcap.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 1665,
|
||||
"floor": 1427,
|
||||
"tolerance_pct": 100,
|
||||
"value": 6661
|
||||
"value": 5711
|
||||
},
|
||||
"randread.overcap.read_p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 556,
|
||||
"floor": 624,
|
||||
"tolerance_pct": 100,
|
||||
"value": 139
|
||||
"value": 156
|
||||
},
|
||||
"randread.overcap.read_p99us": {
|
||||
"dir": "lower",
|
||||
"floor": 1920,
|
||||
"floor": 1628,
|
||||
"tolerance_pct": 100,
|
||||
"value": 480
|
||||
"value": 407
|
||||
},
|
||||
"randread.resident.filled_rss_kb": {
|
||||
"dir": "lower",
|
||||
"floor": 54080,
|
||||
"floor": 168288,
|
||||
"tolerance_pct": 100,
|
||||
"value": 13520
|
||||
"value": 42072
|
||||
},
|
||||
"randread.resident.ops_sec": {
|
||||
"dir": "higher",
|
||||
"floor": 488424,
|
||||
"floor": 387281,
|
||||
"tolerance_pct": 100,
|
||||
"value": 1953697
|
||||
"value": 1549126
|
||||
},
|
||||
"randread.resident.read_p50us": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 0
|
||||
"value": 1
|
||||
},
|
||||
"randread.resident.read_p99us": {
|
||||
"dir": "lower",
|
||||
|
|
@ -686,57 +806,57 @@
|
|||
},
|
||||
"replay.history.ms": {
|
||||
"dir": "lower",
|
||||
"floor": 844,
|
||||
"floor": 12876,
|
||||
"tolerance_pct": 100,
|
||||
"value": 211
|
||||
"value": 3219
|
||||
},
|
||||
"replay.history.ns_per_record": {
|
||||
"dir": "lower",
|
||||
"floor": 21068,
|
||||
"floor": 64384,
|
||||
"tolerance_pct": 100,
|
||||
"value": 5267
|
||||
"value": 16096
|
||||
},
|
||||
"replay.history.records": {
|
||||
"dir": "lower",
|
||||
"floor": 160000,
|
||||
"floor": 800000,
|
||||
"tolerance_pct": 100,
|
||||
"value": 40000
|
||||
"value": 200000
|
||||
},
|
||||
"replay.history.wal_bytes": {
|
||||
"dir": "lower",
|
||||
"floor": 7840140,
|
||||
"floor": 39200140,
|
||||
"tolerance_pct": 100,
|
||||
"value": 1960035
|
||||
"value": 9800035
|
||||
},
|
||||
"replay.history_penalty_x": {
|
||||
"dir": "lower",
|
||||
"floor": 100,
|
||||
"tolerance_pct": 100,
|
||||
"value": 1.9
|
||||
"value": 1.6
|
||||
},
|
||||
"replay.inserts.ms": {
|
||||
"dir": "lower",
|
||||
"floor": 444,
|
||||
"floor": 8064,
|
||||
"tolerance_pct": 100,
|
||||
"value": 111
|
||||
"value": 2016
|
||||
},
|
||||
"replay.inserts.ns_per_record": {
|
||||
"dir": "lower",
|
||||
"floor": 22120,
|
||||
"floor": 80656,
|
||||
"tolerance_pct": 100,
|
||||
"value": 5530
|
||||
"value": 20164
|
||||
},
|
||||
"replay.inserts.records": {
|
||||
"dir": "lower",
|
||||
"floor": 80000,
|
||||
"floor": 400000,
|
||||
"tolerance_pct": 100,
|
||||
"value": 20000
|
||||
"value": 100000
|
||||
},
|
||||
"replay.inserts.wal_bytes": {
|
||||
"dir": "lower",
|
||||
"floor": 3920140,
|
||||
"floor": 19600140,
|
||||
"tolerance_pct": 100,
|
||||
"value": 980035
|
||||
"value": 4900035
|
||||
},
|
||||
"replay.startup_ms": {
|
||||
"dir": "lower",
|
||||
|
|
|
|||
|
|
@ -295,6 +295,7 @@ let b_sha1 = 85
|
|||
let b_sha256 = 86
|
||||
let b_hmac_sha256 = 87
|
||||
let b_call = 88
|
||||
let b_monitor = 89
|
||||
let b_split = 28
|
||||
let b_split_ws = 29
|
||||
let b_join = 30
|
||||
|
|
@ -1110,7 +1111,7 @@ let is_builtin_name (n : string) =
|
|||
"substr"; "trim"; "to_lower"; "char_of"; "parse_int"; "split"; "split_ws"; "join"; "slice";
|
||||
"pop"; "shift"; "sort"; "reverse"; "remove"; "key_at"; "val_at";
|
||||
(* the concurrency arc *)
|
||||
"send"; "call";
|
||||
"send"; "call"; "monitor";
|
||||
(* iteration 19: Float bridges and Bytes surface *)
|
||||
"float"; "trunc"; "parse_float"; "float_to_text"; "float_cmp"; "bytes_len"; "bytes_at";
|
||||
"bytes_slice"; "bytes_eq"; "bytes_concat"; "base64_encode"; "base64_decode";
|
||||
|
|
@ -3441,11 +3442,18 @@ and emit_call (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e : As
|
|||
put f (ins_abc op_builtin dst base sm.Types.sm_builtin);
|
||||
(* every stdlib member only READS its arguments, so one that was
|
||||
freshly built here (`net.write(c, head .. resp.body)`) has no
|
||||
other owner and dies with the call *)
|
||||
other owner and dies with the call. The ONE exception:
|
||||
`time.after`'s message (arg 2) MOVES to the runtime — the
|
||||
timer owns it until delivery (iteration 24 T5). *)
|
||||
let moves i =
|
||||
alias = "time" && mname = "after" && i = 2
|
||||
in
|
||||
List.iteri
|
||||
(fun i (a : Ast.expr) ->
|
||||
drop_fresh_owned ~keep:dst p f (base + i) a;
|
||||
drop_fresh_text ~keep:dst p f (base + i) a)
|
||||
if not (moves i) then begin
|
||||
drop_fresh_owned ~keep:dst p f (base + i) a;
|
||||
drop_fresh_text ~keep:dst p f (base + i) a
|
||||
end)
|
||||
args
|
||||
end)
|
||||
| Some u -> (
|
||||
|
|
@ -3722,7 +3730,7 @@ and emit_builtin (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e :
|
|||
dangle the value just read) and the stores, which either copy (Text,
|
||||
handled by copied_container_call) or take ownership (OWNED/GCREF). *)
|
||||
let reader = List.mem name [ "get"; "latest"; "key_at"; "val_at" ] in
|
||||
(if not (List.mem name [ "push"; "set"; "send"; "call" ]) then
|
||||
(if not (List.mem name [ "push"; "set"; "send"; "call"; "monitor" ]) then
|
||||
List.iteri
|
||||
(fun i (a : Ast.expr) ->
|
||||
(* a reader's result points into arg0 (the container) — dropping
|
||||
|
|
@ -3754,6 +3762,7 @@ and emit_builtin (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e :
|
|||
match name with
|
||||
| "send" -> fixed b_send (* arc: msg (arg1) moved to the runtime — never dropped here *)
|
||||
| "call" -> fixed b_call (* iteration 24: same move; the SCALAR reply lands in dst *)
|
||||
| "monitor" -> fixed b_monitor (* T4: notice msg (arg2) moves to the runtime *)
|
||||
| "now" -> fixed b_now
|
||||
| "print" -> fixed b_print
|
||||
| "print_int" -> fixed b_print_int
|
||||
|
|
|
|||
|
|
@ -1348,6 +1348,10 @@ and analyze_call (ctx : ctx) (call_e : Ast.expr) (callee : Ast.expr) (args : Ast
|
|||
iteration 24: call(addr, msg) moves its message identically. *)
|
||||
| Ident "send" -> i = 1 && Types.StringMap.find_opt "send" ctx.syms.Types.free_fns = None
|
||||
| Ident "call" -> i = 1 && Types.StringMap.find_opt "call" ctx.syms.Types.free_fns = None
|
||||
(* T4/T5: the notice / timer message moves to the runtime too *)
|
||||
| Ident "monitor" ->
|
||||
i = 2 && Types.StringMap.find_opt "monitor" ctx.syms.Types.free_fns = None
|
||||
| Field ({ kind = Ident "time"; _ }, "after") -> i = 2
|
||||
| _ -> false
|
||||
in
|
||||
List.iteri
|
||||
|
|
@ -1365,7 +1369,8 @@ and analyze_call (ctx : ctx) (call_e : Ast.expr) (callee : Ast.expr) (args : Ast
|
|||
transfer ctx p
|
||||
~what:
|
||||
(match callee.kind with
|
||||
| Ident "send" | Ident "call" ->
|
||||
| Ident "send" | Ident "call" | Ident "monitor"
|
||||
| Field ({ kind = Ident "time"; _ }, "after") ->
|
||||
"cannot be sent — a message moves to the receiver"
|
||||
| _ -> "cannot be stored in a container")
|
||||
then record_move ctx p (MvArg "element"))
|
||||
|
|
|
|||
|
|
@ -309,6 +309,8 @@ let stdlib_members : stdlib_member list =
|
|||
m "net" "write_dl" 3 93 (Some (TScalar "Bool")) None;
|
||||
m "net" "listen_unix" 1 94 (Some (TScalar "Int")) None;
|
||||
m "net" "peer" 1 95 (Some (TScalar "Text")) None;
|
||||
(* iteration 24 T5: one-shot timer — the msg MOVES to the runtime *)
|
||||
m "time" "after" 3 90 None None;
|
||||
(* proc *)
|
||||
m "proc" "run" 2 56 (Some (TNullable (TScalar proc_record_name))) (Some proc_record_name);
|
||||
(* json — both members are lowered specially (emit.ml): encode needs its
|
||||
|
|
@ -1917,6 +1919,48 @@ let typecheck_program ~file ~(module_of : string -> string)
|
|||
~message:"`call`'s first argument must be an `actor M` address" ())
|
||||
| None -> ())
|
||||
| _ -> ())
|
||||
| None when name = "monitor" ->
|
||||
(* iteration 24 T4: monitor(watched, observer, msg) — the
|
||||
notice msg is typed against the OBSERVER's mailbox
|
||||
(three-argument form: the caller may be main, which has
|
||||
no mailbox). msg moves like send's. *)
|
||||
(if List.length args <> 3 then
|
||||
Diag.Collector.add collector
|
||||
(Diag.error ~code:bad_arity_code ~file ~line:e.pos.line ~col:e.pos.col
|
||||
~message:
|
||||
(Printf.sprintf
|
||||
"`monitor` takes 3 arguments (watched, observer, notice), given %d"
|
||||
(List.length args))
|
||||
())
|
||||
else
|
||||
match args with
|
||||
| [ w; o; m ] -> (
|
||||
(match confident_typ cenv w with
|
||||
| Some (TActor _) | None -> ()
|
||||
| Some _ ->
|
||||
Diag.Collector.add collector
|
||||
(Diag.error ~code:type_mismatch_code ~file ~line:w.pos.line
|
||||
~col:w.pos.col
|
||||
~message:"`monitor`'s first argument must be an `actor M` address" ()));
|
||||
match confident_typ cenv o with
|
||||
| Some (TActor want) -> (
|
||||
match confident_typ cenv m with
|
||||
| Some (TScalar got) when got <> want ->
|
||||
Diag.Collector.add collector
|
||||
(Diag.error ~code:type_mismatch_code ~file ~line:m.pos.line
|
||||
~col:m.pos.col
|
||||
~message:
|
||||
(Printf.sprintf
|
||||
"the observer receives `%s` — the notice is a `%s`" want got)
|
||||
())
|
||||
| _ -> ())
|
||||
| Some _ ->
|
||||
Diag.Collector.add collector
|
||||
(Diag.error ~code:type_mismatch_code ~file ~line:o.pos.line
|
||||
~col:o.pos.col
|
||||
~message:"`monitor`'s second argument must be an `actor M` address" ())
|
||||
| None -> ())
|
||||
| _ -> ())
|
||||
| None ->
|
||||
let confident_types = List.map (confident_typ cenv) args in
|
||||
check_builtin_call ~file collector name e.pos args confident_types)
|
||||
|
|
|
|||
|
|
@ -119,3 +119,135 @@ rather than acknowledging what disk never got.
|
|||
columns excluded (engine raw-eq is narrower than VM float-eq, and a
|
||||
probe miss cannot be resurrected by a recheck). Pinned by
|
||||
`tests/corpus/run/query-index-probe`.
|
||||
|
||||
## Group commit: one barrier per drain (databasev2 4 part A, 2026-08-28)
|
||||
|
||||
**What changed:** the engine used to commit per *statement*. `db.c` called
|
||||
`wo_wal_commit` immediately after every append, at all six sites, so each row
|
||||
change bought its own `pwrite` and its own `fdatasync`. Now the barrier belongs
|
||||
to the drain, not to the statement.
|
||||
|
||||
**Where the barrier runs, and why there.** A statement on a worker shard has no
|
||||
WAL to write — the runtime asserts workers hold neither `db` nor `wal` — so it
|
||||
marshals to shard 0 and parks. Shard 0 executes those requests in its envelope
|
||||
drain (`wo_vm_adopt`), and the drain now **holds each reply** instead of pushing
|
||||
it as the statement finishes. When the queue empties it issues one barrier, then
|
||||
releases every held reply.
|
||||
|
||||
Holding the reply is the whole mechanism. Pushing it early would unpark the
|
||||
requester before its record was durable; holding it means each writer is
|
||||
acknowledged after the barrier that carried *its own* record. That was always
|
||||
the intended contract — it was simply true by accident before, because every
|
||||
batch had exactly one member.
|
||||
|
||||
**Why the queue is the boundary.** Not a tick, and not a timer. A queue of one
|
||||
gives a batch of one, so a lone writer pays exactly what it paid before; the
|
||||
batch grows only when writes genuinely contend. A tick boundary would have
|
||||
added latency even with nothing to batch against, which is taxing an idle
|
||||
system to serve a busy one. There is nothing to tune, which is the point.
|
||||
|
||||
**Why the inline path is asymmetric.** A statement already on shard 0 stages and
|
||||
commits before returning, batch size one. It cannot hold a reply because there
|
||||
is nobody to reply to — it returns into its own fiber. Batching it would mean
|
||||
parking that fiber on the barrier, which is part B's machinery. Two consequences
|
||||
worth keeping in mind: single-shard configurations get no batching at all, by
|
||||
design; and the inline commit is only safe because the drain commits
|
||||
*unconditionally* whenever anything is staged, so the buffer is empty when an
|
||||
inline statement runs. If that ever stops holding, the inline path would make
|
||||
another statement's record durable early and acknowledge it to the wrong writer.
|
||||
|
||||
**One rule for failure: once a statement has mutated RAM, the outcomes are
|
||||
durable or process death.** It replaced three behaviours that disagreed —
|
||||
`insert` un-applied itself, while `update` and `delete` returned a catchable
|
||||
trap and left RAM ahead of disk, which their own comments said out loud.
|
||||
Batching would have multiplied that from one row to a whole batch. So a failed
|
||||
stage or a failed barrier now prints one diagnostic (operation, log path,
|
||||
`errno`, record count) and exits 3; `WO_T_IO` is unreachable from a write.
|
||||
Retrying is not offered because it is unsound: on Linux a failed `fsync` may
|
||||
already have discarded the dirty pages, so a second call can report success
|
||||
having written nothing. Replay is the recovery that works.
|
||||
|
||||
**Measuring it.** `WO_WAL_STATS=1` makes the runtime print one line at exit —
|
||||
batches, records, peak batch, peak staged bytes. Opt-in, because it would
|
||||
otherwise pollute every durable program's output. The counters live in `wo_wal`
|
||||
rather than behind a builtin: they are diagnostic, not part of the language.
|
||||
`db-bench`'s `wmix N C` leg exists to exercise this at all — `mix` writes on one
|
||||
op in ten with C=4, which produced a measured mean batch of 1.01, so it could
|
||||
never have shown whether batching worked.
|
||||
|
||||
**If you are looking at this because writes got slower**, check the mean batch
|
||||
first. Mean 1.0 means the mechanism is not engaging, which is expected for a
|
||||
serial writer or a single-shard configuration and a bug anywhere else.
|
||||
|
||||
## Checkpoint: compaction by rewrite + rename (databasev2 3, 2026-08-29)
|
||||
|
||||
**The problem:** nothing ever removed superseded records, so the log grew
|
||||
forever and boot replayed all history. Measured before this: 20 000 rows seeded
|
||||
gave a 986 KB log; updating those same rows 20 000 times took it to 2.6 MB with
|
||||
**the same live data**.
|
||||
|
||||
**Why one file and not a snapshot plus a tail.** Postgres does the opposite —
|
||||
its WAL is a redo tail and the data lives in heap files, so a checkpoint flushes
|
||||
pages and then recycles log segments; it never compacts. It cannot: its records
|
||||
are page deltas, so a compacted redo log is not a store. **Ours are full row
|
||||
images** — `apply_record` implements UPDATE as remove-then-recreate — so a log
|
||||
of one record per live row *is* a complete store. That single difference deletes
|
||||
the control file, the redo pointer, the second recovery source and the separate
|
||||
process from this design. Recovery is not merely compatible with compaction; it
|
||||
is completely unaware of it.
|
||||
|
||||
**Why `rename` is the whole crash-safety story.** The dump goes to a temp file,
|
||||
which is fsynced, renamed over the live log, and then the parent directory is
|
||||
fsynced (the rename is atomic in-kernel, but the directory entry is not durable
|
||||
until the parent is — Postgres does the same for the same reason). Before the
|
||||
rename the live log is intact and the temp is not authoritative; after it the new
|
||||
log is complete. There is no instant at which a reader sees a mixture, so this
|
||||
needs no recovery logic of its own. What Postgres achieves with a redo pointer
|
||||
computed at checkpoint start and a control file written at the end, one syscall
|
||||
achieves here — because we can swap the entire data set atomically and Postgres
|
||||
cannot.
|
||||
|
||||
A crash mid-rewrite leaves a temp file. The next open **removes it**, and it is
|
||||
deleted rather than ignored because a file full of well-formed records sitting
|
||||
beside the log is exactly what a later reader mistakes for data.
|
||||
|
||||
**Why the dump flushes periodically, and why it does NOT fsync when it does.**
|
||||
`stage()` grows the staging buffer by doubling and never shrinks it, so pushing a
|
||||
whole store through one buffer would hold the entire store in RAM on top of the
|
||||
store — the unbounded growth databasev2 1 measured as how this engine dies. So
|
||||
the dump flushes every 256 records. It flushes with a plain write, **not** a
|
||||
commit: intermediate durability is worthless because the temp is not
|
||||
authoritative until the rename and is fsynced once immediately before it. Using
|
||||
the committing path cost one barrier per 256 records and made the pause 8×
|
||||
larger — measured 107 649 µs against 13 212 µs for a 2 MB live set, ~22 MB/s
|
||||
against ~181 MB/s.
|
||||
|
||||
**Why the replacement is preallocated like the original.** The WAL is
|
||||
preallocated so that appends never extend the file, which is what lets
|
||||
`fdatasync` alone serve as the ack barrier. A replacement opened without it
|
||||
would silently change that property, and the zero-padded tail the open-time scan
|
||||
relies on.
|
||||
|
||||
**When it runs.** Only where the staging buffer is empty — right after a
|
||||
barrier. Both write paths check: the drain (`vm.c`, after its commit and after
|
||||
releasing held replies, since those records are already durable and should not
|
||||
wait out a rewrite) and the inline path (`db.c`). Wiring only the drain left
|
||||
`WO_SHARDS=1` never compacting, with its log growing forever: measured 536 KB
|
||||
where the multi-shard run held 446 KB.
|
||||
|
||||
**The trigger** compares the log against what the *last* compaction actually
|
||||
wrote, with an absolute floor. The denominator is measured rather than
|
||||
estimated, because estimating the live size means estimating Text and the
|
||||
compactor already knows the true number. There is deliberately **no timer**:
|
||||
Postgres needs one because its dirty buffers are not durable until flushed, and
|
||||
ours are durable at commit — an idle log does not grow.
|
||||
|
||||
**A failed compaction is a missed optimisation, not a durability event.** It
|
||||
leaves the original log intact and returns an error the callers ignore. It must
|
||||
never take `wo_wal_commit_fatal`'s path, which exists for a different problem.
|
||||
|
||||
**If you are here because a checkpoint misbehaved:** `WO_WAL_STATS=1` reports
|
||||
compaction count, the stop-the-world pause (max and total) and the last
|
||||
compaction's size. `WO_CHECKPOINT_BYTES` and `WO_CHECKPOINT_RATIO` move the
|
||||
policy; setting a tiny floor forces compaction in a few writes, which is how the
|
||||
gate tests it at all.
|
||||
|
|
|
|||
|
|
@ -18,6 +18,23 @@ static int table_is_durable(const wo_db *db, uint32_t cid) {
|
|||
return (db->classes[cid].flags & WO_CLASSF_VOLATILE) == 0u;
|
||||
}
|
||||
|
||||
/* databasev2 3: the inline path's compaction check.
|
||||
*
|
||||
* The drain has its own (vm.c, after the barrier). This one exists because a
|
||||
* statement running ON the owner shard never enters that drain, so without it
|
||||
* a single-shard durable program's log grows FOREVER — measured: WO_SHARDS=1
|
||||
* reached 536 KB where the multi-shard run held 446 KB, because the check was
|
||||
* only wired into the drain.
|
||||
*
|
||||
* Safe here for the same reason it is safe there: the commit above just
|
||||
* emptied the staging buffer. The result is ignored because a failed
|
||||
* compaction is a missed optimisation, not a durability event. */
|
||||
static void maybe_compact(wo_db *db, wo_wal *w) {
|
||||
if (wo_wal_should_compact(w->off, w->compacted_bytes, wo_wal_ckpt_floor,
|
||||
wo_wal_ckpt_ratio))
|
||||
(void)wo_wal_compact(w, db);
|
||||
}
|
||||
|
||||
int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
||||
uint32_t A = wo_ins_a(ins), B = wo_ins_b(ins), C = wo_ins_c(ins);
|
||||
wo_db *db = (wo_db *)vm->rt.db;
|
||||
|
|
@ -36,15 +53,26 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
|||
: WO_T_DB;
|
||||
wo_wal *w = (wo_wal *)vm->rt.wal;
|
||||
if (w && table_is_durable(db, cid)) {
|
||||
/* RAM applied, record staged, ONE commit before the ack (the
|
||||
* builtin's return). A failed commit is a failed write: the
|
||||
* row is removed again so RAM never claims what disk never
|
||||
* acknowledged, and the statement traps. */
|
||||
if (wo_wal_append_insert(w, db, cid, id) != 0 || wo_wal_commit(w) != 0) {
|
||||
wo_row_remove(db, cid, id);
|
||||
*msg = "wal commit failed";
|
||||
return WO_T_IO;
|
||||
}
|
||||
/* THE INLINE PATH KEEPS ITS OWN BARRIER, AND THAT ASYMMETRY IS
|
||||
* DELIBERATE (databasev2 4 part A). The request path batches:
|
||||
* wo_vm_adopt holds each reply and commits once per drain. This
|
||||
* path cannot, because it has no reply to hold — it returns into
|
||||
* its OWN fiber rather than unparking a requester. Do not "fix"
|
||||
* this by dropping the commit: without it an inline statement
|
||||
* would never be durable at all.
|
||||
*
|
||||
* Committing here is safe because the drain commits
|
||||
* unconditionally whenever anything is staged, so the buffer is
|
||||
* empty when this runs.
|
||||
*
|
||||
* The `table_is_durable` guard is databasev2 2's: a
|
||||
* `@table(durable: false)` class is never staged, so it reaches
|
||||
* neither this barrier nor the compaction check below.
|
||||
*
|
||||
* Failure is fatal, not a trap: the row is already in RAM. */
|
||||
if (wo_wal_append_insert(w, db, cid, id) != 0) wo_wal_stage_fatal(w);
|
||||
wo_wal_commit_fatal(w, 1);
|
||||
maybe_compact(db, w);
|
||||
}
|
||||
R[A] = id;
|
||||
return 0;
|
||||
|
|
@ -58,10 +86,11 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
|||
return ek == DB_ERR_UNIQUE ? WO_T_UNIQUE : ek == DB_ERR_OOM ? WO_T_OOM : WO_T_DB;
|
||||
wo_wal *w = (wo_wal *)vm->rt.wal;
|
||||
if (w && table_is_durable(db, cid)) {
|
||||
if (wo_wal_append_update(w, db, cid, id) != 0 || wo_wal_commit(w) != 0) {
|
||||
*msg = "wal commit failed"; /* RAM ahead of disk: trap, do not ack */
|
||||
return WO_T_IO;
|
||||
}
|
||||
/* was: trap and leave RAM ahead of disk, which the old comment
|
||||
* admitted. Now fatal — see the insert arm. */
|
||||
if (wo_wal_append_update(w, db, cid, id) != 0) wo_wal_stage_fatal(w);
|
||||
wo_wal_commit_fatal(w, 1);
|
||||
maybe_compact(db, w);
|
||||
}
|
||||
R[A] = 0;
|
||||
return 0;
|
||||
|
|
@ -81,10 +110,9 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
|||
}
|
||||
wo_wal *w = (wo_wal *)vm->rt.wal;
|
||||
if (w && table_is_durable(db, cid)) {
|
||||
if (wo_wal_append_remove(w, cid, id) != 0 || wo_wal_commit(w) != 0) {
|
||||
*msg = "wal commit failed";
|
||||
return WO_T_IO;
|
||||
}
|
||||
if (wo_wal_append_remove(w, cid, id) != 0) wo_wal_stage_fatal(w);
|
||||
wo_wal_commit_fatal(w, 1);
|
||||
maybe_compact(db, w);
|
||||
}
|
||||
R[A] = 0;
|
||||
return 0;
|
||||
|
|
@ -226,13 +254,12 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) {
|
|||
q->msg = m;
|
||||
break;
|
||||
}
|
||||
if (w) {
|
||||
if (wo_wal_append_insert(w, db, q->cid, id) != 0 || wo_wal_commit(w) != 0) {
|
||||
wo_row_remove(db, q->cid, id);
|
||||
q->status = WO_T_IO;
|
||||
q->msg = "wal commit failed";
|
||||
break;
|
||||
}
|
||||
if (w && table_is_durable(db, q->cid)) {
|
||||
/* databasev2 4: staging failure is FATAL, not a trap. The row is
|
||||
* already in RAM; of the three verbs only insert could undo
|
||||
* itself, so continuing means RAM ahead of disk. One rule: once a
|
||||
* statement has mutated RAM, the outcomes are durable or death. */
|
||||
if (wo_wal_append_insert(w, db, q->cid, id) != 0) wo_wal_stage_fatal(w);
|
||||
}
|
||||
q->result = id;
|
||||
break;
|
||||
|
|
@ -244,12 +271,8 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) {
|
|||
q->msg = m;
|
||||
break;
|
||||
}
|
||||
if (w) {
|
||||
if (wo_wal_append_update(w, db, q->cid, q->id) != 0 || wo_wal_commit(w) != 0) {
|
||||
q->status = WO_T_IO;
|
||||
q->msg = "wal commit failed";
|
||||
break;
|
||||
}
|
||||
if (w && table_is_durable(db, q->cid)) {
|
||||
if (wo_wal_append_update(w, db, q->cid, q->id) != 0) wo_wal_stage_fatal(w);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
|
@ -264,12 +287,8 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) {
|
|||
q->msg = "no such row";
|
||||
break;
|
||||
}
|
||||
if (w) {
|
||||
if (wo_wal_append_remove(w, q->cid, q->id) != 0 || wo_wal_commit(w) != 0) {
|
||||
q->status = WO_T_IO;
|
||||
q->msg = "wal commit failed";
|
||||
break;
|
||||
}
|
||||
if (w && table_is_durable(db, q->cid)) {
|
||||
if (wo_wal_append_remove(w, q->cid, q->id) != 0) wo_wal_stage_fatal(w);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -4,7 +4,9 @@
|
|||
#include "wal.h"
|
||||
|
||||
#include <errno.h>
|
||||
#include <time.h>
|
||||
#include <fcntl.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
|
|
@ -302,6 +304,19 @@ int wo_wal_open(wo_wal *w, const char *path, uint64_t prealloc) {
|
|||
memset(w, 0, sizeof(*w));
|
||||
w->fd = open(path, O_RDWR | O_CREAT, 0644);
|
||||
if (w->fd < 0) return -1;
|
||||
w->path = strdup(path); /* NULL is tolerated: the diagnostic degrades */
|
||||
/* databasev2 3: remove a stale compaction temp before doing anything else.
|
||||
* The only way one exists is a crash before the rename, which means its
|
||||
* records were never authoritative — the live log below is the truth. It is
|
||||
* deleted rather than ignored because a file full of well-formed records
|
||||
* sitting beside the log is exactly the thing a future reader mistakes for
|
||||
* data. */
|
||||
{
|
||||
char tmp[4096];
|
||||
if ((size_t)snprintf(tmp, sizeof tmp, "%s%s", path, WO_WAL_TMP_SUFFIX) < sizeof tmp)
|
||||
(void)unlink(tmp);
|
||||
}
|
||||
w->prealloc = prealloc;
|
||||
if (prealloc) {
|
||||
/* best-effort: a filesystem without fallocate still works */
|
||||
(void)posix_fallocate(w->fd, 0, (off_t)prealloc);
|
||||
|
|
@ -317,6 +332,7 @@ int wo_wal_open(wo_wal *w, const char *path, uint64_t prealloc) {
|
|||
|
||||
void wo_wal_close(wo_wal *w) {
|
||||
if (w->fd >= 0) close(w->fd);
|
||||
free(w->path);
|
||||
free(w->buf);
|
||||
memset(w, 0, sizeof(*w));
|
||||
w->fd = -1;
|
||||
|
|
@ -390,7 +406,103 @@ int wo_wal_append_remove(wo_wal *w, uint32_t class_id, uint64_t id) {
|
|||
}
|
||||
|
||||
int wo_wal_commit(wo_wal *w) {
|
||||
if (!w->len) return 0;
|
||||
if (!w->len) return 0; /* empty commits are not batches; do not count them */
|
||||
if (w->len > w->stat_peak_staged) w->stat_peak_staged = w->len;
|
||||
size_t at = 0;
|
||||
while (at < w->len) {
|
||||
ssize_t n = pwrite(w->fd, w->buf + at, w->len - at, (off_t)(w->off + at));
|
||||
if (n < 0) {
|
||||
if (errno == EINTR) continue;
|
||||
return WO_WAL_ERR_WRITE;
|
||||
}
|
||||
at += (size_t)n;
|
||||
}
|
||||
if (fdatasync(w->fd) != 0) return WO_WAL_ERR_SYNC;
|
||||
w->off += w->len;
|
||||
w->len = 0; /* acked: the batch is durable */
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Nothing at either fatal point is recoverable: RAM holds changes the log
|
||||
* does not, and this process can no longer serve reads that would survive a
|
||||
* restart. Name what failed precisely enough to act on, then stop. */
|
||||
static void wal_die(const wo_wal *w, const char *op, uint32_t nrec) {
|
||||
fprintf(stderr,
|
||||
"writeonce: DURABILITY FAILURE — %s failed on %s: %s\n"
|
||||
" %u record(s) were NOT made durable and are not acknowledged.\n"
|
||||
" The process is stopping: replay restores the last durable state.\n",
|
||||
op, w->path ? w->path : "(the write-ahead log)", strerror(errno),
|
||||
nrec);
|
||||
exit(WO_EXIT_DURABILITY);
|
||||
}
|
||||
|
||||
void wo_wal_stage_fatal(const wo_wal *w) { wal_die(w, "staging a record", 1); }
|
||||
|
||||
void wo_wal_commit_fatal(wo_wal *w, uint32_t nrec) {
|
||||
int staged = w->len != 0;
|
||||
int rc = wo_wal_commit(w);
|
||||
if (rc == 0) {
|
||||
if (staged) { /* count the barrier that actually happened */
|
||||
w->stat_batches++;
|
||||
w->stat_records += nrec;
|
||||
if (nrec > w->stat_peak_batch) w->stat_peak_batch = nrec;
|
||||
}
|
||||
return;
|
||||
}
|
||||
wal_die(w, rc == WO_WAL_ERR_SYNC ? "fdatasync" : "pwrite", nrec);
|
||||
}
|
||||
|
||||
uint64_t wo_wal_ckpt_floor = 4u << 20; /* 4 MiB: below this there is nothing worth reclaiming */
|
||||
uint32_t wo_wal_ckpt_ratio = 3u; /* 3x the live-set's own size is enough history */
|
||||
|
||||
int wo_wal_should_compact(uint64_t used, uint64_t last, uint64_t floor, uint32_t ratio) {
|
||||
if (used < floor) return 0; /* a small log has nothing to reclaim */
|
||||
if (last == 0) return 1; /* past the floor and never compacted: do it once
|
||||
* to establish the denominator */
|
||||
if (ratio == 0) return 0; /* a zero ratio disables the policy rather than
|
||||
* dividing by nothing */
|
||||
return used > last * (uint64_t)ratio;
|
||||
}
|
||||
|
||||
/* databasev2 3: how many records the dump stages before flushing.
|
||||
*
|
||||
* NOT unbounded: stage() grows the staging buffer by doubling and never
|
||||
* shrinks it, so appending a whole store through one buffer would hold the
|
||||
* entire store in RAM on top of the store itself — the unbounded growth
|
||||
* databasev2 1 identified as how this engine dies. 256 records is a few tens
|
||||
* of KiB per flush, which is large enough that the syscall cost is amortised
|
||||
* and small enough that the buffer never matters. */
|
||||
#define WO_WAL_COMPACT_FLUSH 256u
|
||||
|
||||
/* rename(2)'s atomicity is in-kernel: the new directory ENTRY is not durable
|
||||
* until the parent directory is synced. Postgres does the same thing for the
|
||||
* same reason. Best-effort — a filesystem that refuses to sync a directory
|
||||
* still leaves a correct log, just one whose swap might not survive a power
|
||||
* cut. */
|
||||
static void sync_parent_dir(const char *path) {
|
||||
char dir[4096];
|
||||
size_t n = strlen(path);
|
||||
if (n >= sizeof dir) return;
|
||||
memcpy(dir, path, n + 1);
|
||||
char *slash = strrchr(dir, '/');
|
||||
if (slash == dir) dir[1] = '\0';
|
||||
else if (slash) *slash = '\0';
|
||||
else memcpy(dir, ".", 2);
|
||||
int fd = open(dir, O_RDONLY);
|
||||
if (fd < 0) return;
|
||||
(void)fsync(fd);
|
||||
close(fd);
|
||||
}
|
||||
|
||||
/* databasev2 3: write the staged bytes WITHOUT a durability barrier.
|
||||
*
|
||||
* Only compaction's dump uses this. Intermediate durability there is worthless:
|
||||
* the temp file is not authoritative until the rename, and it is fsynced once
|
||||
* immediately before that. Using wo_wal_commit for the dump instead cost one
|
||||
* fdatasync per 256 records — measured, that was most of the stop-the-world
|
||||
* pause (~22 MB/s, where the fixed cost plus ~150 redundant syncs dominated a
|
||||
* 2 MB dump). */
|
||||
static int wal_write_nosync(wo_wal *w) {
|
||||
size_t at = 0;
|
||||
while (at < w->len) {
|
||||
ssize_t n = pwrite(w->fd, w->buf + at, w->len - at, (off_t)(w->off + at));
|
||||
|
|
@ -400,12 +512,110 @@ int wo_wal_commit(wo_wal *w) {
|
|||
}
|
||||
at += (size_t)n;
|
||||
}
|
||||
if (fdatasync(w->fd) != 0) return -1;
|
||||
w->off += w->len;
|
||||
w->len = 0; /* acked: the batch is durable */
|
||||
w->len = 0;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static uint64_t mono_us(void) {
|
||||
struct timespec ts;
|
||||
if (clock_gettime(CLOCK_MONOTONIC, &ts) != 0) return 0;
|
||||
return (uint64_t)ts.tv_sec * 1000000ull + (uint64_t)ts.tv_nsec / 1000ull;
|
||||
}
|
||||
|
||||
/* ============================================================================
|
||||
* OBLIGATION FOR WHOEVER IMPLEMENTS `resident: keys` (databasev2 2, tasks
|
||||
* 5c/5d) — READ THIS BEFORE STORING WAL OFFSETS.
|
||||
*
|
||||
* Compaction rewrites the log and MOVES EVERY RECORD. Any WAL byte offset
|
||||
* captured from the old file is meaningless afterwards — not stale-but-
|
||||
* readable, but pointing at an arbitrary byte of a different file.
|
||||
*
|
||||
* `resident: keys` stores exactly such an offset per row and reads rows back
|
||||
* through it. So the loop below, which knows each record's NEW position as it
|
||||
* writes it, MUST also rebuild that map. It is the cheap direction and the only
|
||||
* one that keeps both features usable together; the alternative is forbidding
|
||||
* compaction whenever such a table is live, which would mean the feature for
|
||||
* huge tables is incompatible with the feature that stops their log growing.
|
||||
*
|
||||
* Nothing fails today because that storage half does not exist yet. It will
|
||||
* fail later, and it will look like data corruption rather than a design gap.
|
||||
* ==========================================================================*/
|
||||
int wo_wal_compact(wo_wal *w, wo_db *db) {
|
||||
/* staged records would be written into a file about to be replaced */
|
||||
if (!w->path || w->len != 0) return -1;
|
||||
uint64_t t0 = mono_us();
|
||||
|
||||
char tmp[4096];
|
||||
if ((size_t)snprintf(tmp, sizeof tmp, "%s%s", w->path, WO_WAL_TMP_SUFFIX) >= sizeof tmp)
|
||||
return -1;
|
||||
(void)unlink(tmp); /* a stale one would otherwise be appended to */
|
||||
|
||||
wo_wal nw;
|
||||
/* THE REPLACEMENT MUST BE PREALLOCATED LIKE THE ORIGINAL. The WAL is
|
||||
* preallocated so appends never extend the file, which is precisely what
|
||||
* makes fdatasync sufficient as the ack barrier — no file-size metadata
|
||||
* has to reach disk for an acked record to be readable. Opening the
|
||||
* replacement with prealloc 0 silently removed that property, and the
|
||||
* crash battery caught it: records acked shortly before a kill went
|
||||
* missing, with the log otherwise intact and self-consistent. */
|
||||
if (wo_wal_open(&nw, tmp, w->prealloc) != 0) return -1;
|
||||
|
||||
/* one INSERT per live row, in the existing grammar, through the existing
|
||||
* append path — so replay needs no second decoder and ids are preserved
|
||||
* exactly (wo_wal_append_insert takes the id and reads the row) */
|
||||
uint32_t pending = 0;
|
||||
for (uint32_t cid = 0; cid < db->class_cnt; cid++) {
|
||||
db_table *t = &db->tables[cid];
|
||||
if (!t->slabs) continue; /* tables are created lazily */
|
||||
uint32_t total = t->slab_cnt * DB_SLAB_ROWS;
|
||||
for (uint32_t g = 0; g < total; g++) {
|
||||
if (!(t->bitmap[g >> 6] & (1ull << (g & 63)))) continue;
|
||||
db_row *r = (db_row *)(t->slabs[g / DB_SLAB_ROWS] +
|
||||
(size_t)(g % DB_SLAB_ROWS) * t->row_size);
|
||||
if (wo_wal_append_insert(&nw, db, cid, r->id) != 0) goto fail;
|
||||
if (++pending >= WO_WAL_COMPACT_FLUSH) {
|
||||
if (wal_write_nosync(&nw) != 0) goto fail;
|
||||
pending = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (wal_write_nosync(&nw) != 0) goto fail; /* the tail batch */
|
||||
/* THE dump's one and only barrier: everything above is just bytes in the
|
||||
* page cache until this, and nothing reads the temp before the rename. */
|
||||
if (fsync(nw.fd) != 0) goto fail;
|
||||
|
||||
uint64_t new_bytes = nw.off;
|
||||
wo_wal_close(&nw);
|
||||
|
||||
/* THE SWITCH. Every crash point either side of this is safe. */
|
||||
if (rename(tmp, w->path) != 0) {
|
||||
(void)unlink(tmp);
|
||||
return -1;
|
||||
}
|
||||
sync_parent_dir(w->path);
|
||||
|
||||
/* the old descriptor now refers to an unlinked inode */
|
||||
if (w->fd >= 0) close(w->fd);
|
||||
w->fd = open(w->path, O_RDWR);
|
||||
if (w->fd < 0) return -1; /* the log is correct on disk; this process cannot go on */
|
||||
w->off = new_bytes;
|
||||
w->len = 0;
|
||||
w->compacted_bytes = new_bytes;
|
||||
{ /* the stop-the-world pause: nothing was served while this ran */
|
||||
uint64_t el = mono_us() - t0;
|
||||
w->stat_compactions++;
|
||||
w->stat_compact_us_total += el;
|
||||
if (el > w->stat_compact_us_max) w->stat_compact_us_max = el;
|
||||
}
|
||||
return 0;
|
||||
|
||||
fail:
|
||||
wo_wal_close(&nw);
|
||||
(void)unlink(tmp);
|
||||
return -1; /* the live log is untouched and still usable */
|
||||
}
|
||||
|
||||
static int apply_record(wo_db *db, const uint8_t *payload, uint32_t len) {
|
||||
rbuf r = {payload, payload + len, 0};
|
||||
uint8_t kind = rd_u8(&r);
|
||||
|
|
|
|||
|
|
@ -47,10 +47,38 @@ enum { WO_WAL_INSERT = 1, WO_WAL_REMOVE = 2, WO_WAL_UPDATE = 3 };
|
|||
|
||||
typedef struct wo_wal {
|
||||
int fd;
|
||||
/* databasev2 4: where this WAL lives, so a durability failure can name
|
||||
* the file it could not write. An abort diagnostic without the path
|
||||
* sends an operator hunting. Owned here, freed by wo_wal_close. */
|
||||
char *path;
|
||||
uint64_t off; /* next write offset (the intact tail) */
|
||||
/* staged batch: appended by wal_append_*, flushed by wal_commit */
|
||||
uint8_t *buf;
|
||||
size_t len, cap;
|
||||
/* databasev2 4: group-commit diagnostics. Batching is worthless if
|
||||
* batches are always one, and a throughput change would then have come
|
||||
* from somewhere else — so the mechanism is measured, not assumed.
|
||||
* peak_staged also settles whether the batch needs a cap with a number
|
||||
* instead of a guess. Reported at exit under WO_WAL_STATS. */
|
||||
uint64_t stat_batches; /* non-empty commits */
|
||||
uint64_t stat_records; /* records those commits carried */
|
||||
uint64_t stat_peak_batch; /* most records in one barrier */
|
||||
uint64_t stat_peak_staged; /* most bytes staged behind one barrier */
|
||||
/* databasev2 3: bytes the last compaction wrote. The trigger compares the
|
||||
* log against THIS rather than an estimate of the live set — estimating
|
||||
* would mean estimating Text, and the compactor knows the true number. */
|
||||
uint64_t compacted_bytes;
|
||||
/* databasev2 3: the preallocation this log was opened with. Compaction
|
||||
* MUST give the replacement the same one: the WAL is preallocated so that
|
||||
* appends never extend the file, which is what lets fdatasync alone be the
|
||||
* ack barrier. A replacement without it silently weakens durability. */
|
||||
uint64_t prealloc;
|
||||
/* databasev2 3: what compaction actually did, reported under WO_WAL_STATS.
|
||||
* The PAUSE is the number the spec refused to assume — compaction is
|
||||
* stop-the-world, so its duration is the cost being weighed. */
|
||||
uint64_t stat_compactions;
|
||||
uint64_t stat_compact_us_max;
|
||||
uint64_t stat_compact_us_total;
|
||||
} wo_wal;
|
||||
|
||||
/* databasev2 2: the file offset the NEXT staged record will occupy.
|
||||
|
|
@ -90,10 +118,94 @@ int wo_wal_append_remove(wo_wal *w, uint32_t class_id, uint64_t id);
|
|||
* later optimization, recorded). Call AFTER the RAM update. */
|
||||
int wo_wal_append_update(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id);
|
||||
|
||||
/* databasev2 4: which half of the barrier failed. A pwrite failure and an
|
||||
* fdatasync failure are different operational problems (a short write vs a
|
||||
* device refusing the flush), so the diagnostic must name the right one. */
|
||||
#define WO_WAL_ERR_WRITE (-1)
|
||||
#define WO_WAL_ERR_SYNC (-2)
|
||||
|
||||
/* The process exit status for a durability failure.
|
||||
*
|
||||
* 74 is sysexits' EX_IOERR, chosen deliberately over a small number: 1 is a
|
||||
* trap and 2 is a loader refusal, but 3 and 4 are already used by SAMPLES for
|
||||
* their own meanings — db-bench's own `verify` exits 3 on a checksum mismatch,
|
||||
* and it is the gate that exercises durability, so a durability abort exiting 3
|
||||
* would have been indistinguishable from the mismatch it is supposed to help
|
||||
* diagnose. The low range belongs to programs; the runtime takes a high one. */
|
||||
#define WO_EXIT_DURABILITY 74
|
||||
|
||||
/* Write the staged batch and fdatasync — the ack line. Empty batch = ok,
|
||||
* no syscall. 0 ok, -1 write/sync failure (the batch stays staged). */
|
||||
* no syscall. 0 ok, WO_WAL_ERR_WRITE / WO_WAL_ERR_SYNC on failure (the
|
||||
* batch stays staged: a failed commit consumes nothing). */
|
||||
int wo_wal_commit(wo_wal *w);
|
||||
|
||||
/* databasev2 3: the checkpoint trigger, as a PURE decision so it can be tested
|
||||
* without a store — which is the only way a policy like this gets tested at all.
|
||||
*
|
||||
* [used] the log's used bytes; [last] what the LAST compaction wrote (0 if it
|
||||
* has never run); [floor] the size below which compacting is not worth it;
|
||||
* [ratio] the multiple of [last] that counts as too much history.
|
||||
*
|
||||
* The denominator is the last compaction's MEASURED output rather than an
|
||||
* estimate of the live set: estimating would mean estimating Text, and the
|
||||
* compactor already knows the true number.
|
||||
*
|
||||
* There is deliberately NO TIME component. Postgres' CheckPointTimeout exists
|
||||
* to bound data loss from unflushed buffers; our records are durable at commit,
|
||||
* so a checkpoint only reclaims space and shortens boot. An idle log does not
|
||||
* grow, so a timer would fire with nothing to do.
|
||||
*
|
||||
* 1 = compact now, 0 = leave it. */
|
||||
int wo_wal_should_compact(uint64_t used, uint64_t last, uint64_t floor, uint32_t ratio);
|
||||
|
||||
/* Defaults, overridable at boot by WO_CHECKPOINT_BYTES / WO_CHECKPOINT_RATIO.
|
||||
* The knobs are what make the policy testable: a test sets a tiny floor and
|
||||
* forces compaction in a few writes instead of waiting for megabytes. */
|
||||
extern uint64_t wo_wal_ckpt_floor;
|
||||
extern uint32_t wo_wal_ckpt_ratio;
|
||||
|
||||
/* databasev2 3: the temporary file compaction writes before the swap. Named
|
||||
* next to the log so it lands on the same filesystem — rename(2) is only
|
||||
* atomic within one. Boot removes a stale one (a crash before the rename). */
|
||||
#define WO_WAL_TMP_SUFFIX ".compact"
|
||||
|
||||
/* databasev2 3: rewrite the log as one INSERT record per LIVE row, then swap
|
||||
* it in with rename(2).
|
||||
*
|
||||
* Recovery is deliberately untouched: the result is an ordinary log in the
|
||||
* ordinary grammar, replayed from byte 0. Crash safety comes from rename being
|
||||
* atomic — before it the live log is intact and the temp file is not
|
||||
* authoritative; after it the new log is complete. There is no window in which
|
||||
* a reader sees a mixture, so this needs no recovery logic of its own.
|
||||
*
|
||||
* REFUSES if anything is staged (returns -1 without touching the log): those
|
||||
* records would be written into a file about to be replaced. Callers must
|
||||
* invoke this only where the staging buffer is empty — right after a barrier.
|
||||
*
|
||||
* A failure is a MISSED OPTIMISATION, not a durability event: the original log
|
||||
* is left usable and the process keeps running. It must not take the fatal
|
||||
* path wo_wal_commit_fatal takes.
|
||||
*
|
||||
* 0 ok, -1 on any failure. */
|
||||
int wo_wal_compact(wo_wal *w, wo_db *db);
|
||||
|
||||
/* databasev2 4: a record could not even be STAGED (the row is already in
|
||||
* RAM, so this is the same unrecoverable position as a failed barrier — see
|
||||
* wo_wal_commit_fatal). Never returns. */
|
||||
void wo_wal_stage_fatal(const wo_wal *w);
|
||||
|
||||
/* databasev2 4: commit, or END THE PROCESS.
|
||||
*
|
||||
* The one rule this iteration introduces: once a statement has mutated RAM,
|
||||
* the only outcomes are durable or process death. Retrying is not an
|
||||
* alternative — on Linux a failed fsync may already have discarded the dirty
|
||||
* pages, so a second call can report success having written nothing. The
|
||||
* recovery that works is replay, which returns the last durable state.
|
||||
*
|
||||
* [nrec] is the number of records in the batch, for the diagnostic only.
|
||||
* Returns on success; never returns on failure. */
|
||||
void wo_wal_commit_fatal(wo_wal *w, uint32_t nrec);
|
||||
|
||||
/* Boot replay: apply every intact record to [db] in order. Ids re-enter
|
||||
* exactly as logged; each table's next_id advances past the replayed ids
|
||||
* that belong to this shard. Returns the number of records applied, or -1
|
||||
|
|
|
|||
|
|
@ -128,6 +128,7 @@ flowchart TD
|
|||
classDef rt fill:#8250df,color:#fff,stroke:none
|
||||
classDef gated fill:#eac54f,color:#000,stroke:none
|
||||
classDef v2 fill:#0969da,color:#fff,stroke:none
|
||||
classDef done fill:#1a7f37,color:#fff,stroke:none
|
||||
|
||||
I7b2["7b per-shard collector (done — the precondition 8 waited on)"]:::rt
|
||||
I8x["8 shard-actor runtime: thread-per-core, ownership-move messages"]:::rt
|
||||
|
|
@ -140,7 +141,7 @@ flowchart TD
|
|||
STREAM2["request body streaming + backpressure"]:::gated
|
||||
SRESP2["streaming responses + explicit commit point"]:::gated
|
||||
CANCEL2["per-request cancellation propagation"]:::gated
|
||||
PUBSUB2["pub/sub + WebSockets (rejected until here)"]:::gated
|
||||
PUBSUB2["DONE 2026-08-27 — pub/sub + WebSockets (iteration 24: ws_accept + wsframe + room actors)"]:::done
|
||||
ASYNC9C["20 async attach statements (rejected-for-now alternative)"]:::gated
|
||||
TIMEOUTS2["idle timeouts become schedulable (net seam still needed)"]:::gated
|
||||
|
||||
|
|
|
|||
93
docs/2026-08-27-chat-drain-finding.md
Normal file
93
docs/2026-08-27-chat-drain-finding.md
Normal file
|
|
@ -0,0 +1,93 @@
|
|||
# Iteration 24 T9 — the drain bug the gate was hiding
|
||||
|
||||
**Found 2026-08-27** while finishing T8/T9 on branch `chat-ws-lifecycle`.
|
||||
Not fixed: the fix is an engine-level decision, recorded here so it is not
|
||||
rediscovered.
|
||||
|
||||
## The symptom
|
||||
|
||||
`just chat`'s drain leg asserts both connected clients receive a WebSocket
|
||||
close frame on `SIGTERM`. Against a **fresh** server it is flaky:
|
||||
|
||||
| Sample | Result |
|
||||
| --- | --- |
|
||||
| 5 fresh servers, 2 clients each | 4 × `close\|close`, 1 × `eof\|close` |
|
||||
| 12 fresh servers | 3 failures, one of them `eof\|eof` |
|
||||
| 16 fresh servers | 5 failures |
|
||||
|
||||
A failing client's socket reaches EOF with **no close frame and no
|
||||
diagnostic** — the process exits and the kernel closes the fd.
|
||||
|
||||
## Why the gate never caught it
|
||||
|
||||
The drain leg did not start its own server. It inherited `$SRV` from the soak
|
||||
leg — a server the soak had already pushed 1000 clients through, so every
|
||||
shard was warm and every actor already scheduled. Draining a warm server hides
|
||||
the cold-start race. Fixed in this change: **every leg now starts its own
|
||||
server**, which is what exposed the bug.
|
||||
|
||||
## Root cause, traced
|
||||
|
||||
Instrumented the sample's actors (diagnostics not committed) and correlated
|
||||
against failing runs:
|
||||
|
||||
1. `DIAG registry-shutdown rooms=1` — main's `send(reg, kind: 2)` **is**
|
||||
delivered and the Registry runs.
|
||||
2. `DIAG room-shutdown` — **never printed on a failing run.** The Room never
|
||||
processes the `kind: 4` shutdown the Registry sends it.
|
||||
3. The Writer's close branch never runs for the affected client, so no close
|
||||
frame is written and the fd is never closed by the Writer. Its
|
||||
`try net.write_dl(...)` is **not** failing — a diagnostic on that path
|
||||
printed zero times.
|
||||
4. A client that *does* get a close frame is usually saved by its own
|
||||
**Reader** noticing `env.stopping()` and running its tail
|
||||
(`DIAG reader-tail bob r2=1`), not by the room broadcast.
|
||||
|
||||
So the drain chain is main → Registry → Room → Writer, three hops across
|
||||
shards, and **the Room's shard does not reliably adopt its inbox before the
|
||||
engine stops.**
|
||||
|
||||
## What was ruled out
|
||||
|
||||
- **Not the spin budget.** Replacing `spin < 20000000` with a wall-clock
|
||||
deadline of 1 s (`time.ticks()`) still failed 2 of 12. More time does not
|
||||
help, which is the strongest evidence the room's shard is not being
|
||||
scheduled at all rather than being scheduled late. That change was reverted:
|
||||
it fixed nothing and cost a fixed 1 s on every shutdown.
|
||||
- **Not `dummy_writer()` spawning during shutdown.** Hoisting it to a
|
||||
Registry field spawned once at startup left 5 of 16 failing.
|
||||
- **Not a write failure.** See point 3.
|
||||
|
||||
## The decision this needs
|
||||
|
||||
`main` cannot park after the stop flag (a park unwinds), so it spins — and
|
||||
spinning is not a barrier. Either:
|
||||
|
||||
- **the engine drains pending inboxes before stopping**, so a `send` issued
|
||||
before the stop flag is guaranteed delivered; or
|
||||
- **the sample gets a real barrier** — the drain is acknowledged back to main,
|
||||
which requires main to observe a reply without parking.
|
||||
|
||||
The first is the honest fix and belongs to the actor lifecycle (iteration 31,
|
||||
absorbed into 24). It is a semantic guarantee — "a send before shutdown is
|
||||
delivered" — not a tuning parameter, and it should be stated in the runtime's
|
||||
lifecycle docs and pinned by a corpus fixture, not left to a spin count.
|
||||
|
||||
## Gate defects fixed alongside (all committed)
|
||||
|
||||
1. **fd check was core-count dependent.** `fds_before + 8` read lazy per-shard
|
||||
init as a leak: shards initialise on first fiber, each taking one
|
||||
`io_uring` + one `eventfd`, capped at `nproc`. On a 20-core box the first
|
||||
wave legitimately adds 18. Measured 26 → 44 after 20 clients, then **still
|
||||
44 after 40 more**. Replaced with the invariant the check is actually for:
|
||||
a second wave must not raise the count. Core-count independent, and it
|
||||
catches a slow leak that any fixed slack would hide.
|
||||
2. **A failed leg orphaned its server.** The drain leg's python died on
|
||||
`int("")` when `$SRV` was empty, so the soak server was never killed and
|
||||
its listener broke the *next* run's soak on the same port. `cleanup` now
|
||||
kills every server a run started, matched on the run's unique temp dir.
|
||||
3. **Two legs the plan requires were missing** — `WO_SHARDS=1` (the
|
||||
single-shard control that says a failure is placement's fault) and
|
||||
`WO_MAILBOX=8` (the drop-slow-member backpressure path). Both added, both
|
||||
green. The mailbox leg manufactures a genuinely slow member by shrinking
|
||||
its `SO_RCVBUF`, so it needs no sleeps.
|
||||
|
|
@ -1,60 +0,0 @@
|
|||
---
|
||||
slice: "24" # the story that owns the status; see stories/24-chat-websocket-workload.md
|
||||
status: in-progress
|
||||
---
|
||||
|
||||
# Active slice — chat + actor lifecycle (iteration 24, absorbing 31 + 34)
|
||||
|
||||
Branch `chat-ws-lifecycle`. Spec:
|
||||
[`superpowers/specs/2026-08-23-chat-websocket-actor-lifecycle-design.md`](superpowers/specs/2026-08-23-chat-websocket-actor-lifecycle-design.md)
|
||||
· plan:
|
||||
[`superpowers/plans/2026-08-23-chat-ws-lifecycle.md`](superpowers/plans/2026-08-23-chat-ws-lifecycle.md)
|
||||
· board: [`stories/00-status.md`](stories/00-status.md).
|
||||
|
||||
## Progress (2026-08-23)
|
||||
|
||||
- ✅ **T1 crypto** (`d14fa9f`): sha1/sha256/hmac_sha256, ids 85–87, RFC
|
||||
vectors 18/0, corpus pin. Story 34's C-builtin resolution delivered.
|
||||
- ✅ **T2 bounded mailboxes** (`92754a8`): cap 1024 + `WO_MAILBOX`,
|
||||
sender-side atomic reserve, WO_T_ACTOR (trap 13) catchable. Plus a
|
||||
pre-existing compiler fix: try-arm Text places (bare `e.msg`) now
|
||||
copy before the arm's scope dies (was ASan use-after-free + SEGV).
|
||||
- ✅ **T6 WS upgrade** (`79cfa01`): `ws_accept` + accept-key + the
|
||||
101 hijack sentinel; plain HTTP byte-identical (web-app 26/26).
|
||||
- ✅ **T7 frame codec** (`7ad2ced`): pure-`.wo` RFC 6455 parse/serialize,
|
||||
probe-verified against the RFC's own bytes.
|
||||
- ✅ **T3 call/reply** (`ed69841`): `call` parks + typed scalar reply
|
||||
(WO-E226 through actor-M erasure); actor DEATH landed with it —
|
||||
callers never hang (mid-call + to-dead both trap catchably). Fixed
|
||||
TRAPF's fiber-death leak/dangle en route.
|
||||
|
||||
Every landed task: full battery 12/12, fresh-built.
|
||||
|
||||
## Pending
|
||||
|
||||
- ⬜ **T4 monitor(watched, observer, msg)** — id 89. Most of the death
|
||||
machinery exists (`actor_die`); T4 adds the per-actor monitor list,
|
||||
the death walk delivering the observer's own M-typed notice,
|
||||
monitor-of-already-dead firing immediately, full-observer notice =
|
||||
disclosed stderr drop. Three-argument form (spec deviation, disclosed
|
||||
in the plan: the caller may be `main`, which has no mailbox).
|
||||
- ⬜ **T5 time.after(ms, addr, msg)** — id 90, one-shot, no cancel;
|
||||
rides the T4 deadline plumbing; delivery = runtime send (full = drop
|
||||
+ stderr line, dead = silent). Corpus: timer-delivery,
|
||||
timer-generation (the cancel idiom). Both WO_IO backends.
|
||||
- ⬜ **T8 chat sample** — docs/examples/chat: registry (`call`'s first
|
||||
consumer), room actors (cap-trap drops slow members, `monitor` reaps
|
||||
dead writers), reader/writer actor pair per connection over
|
||||
ws_accept/wsframe; SIGTERM close choreography.
|
||||
- ⬜ **T9 chat gate** — scripts/chat-accept.sh + raw-RFC6455 python
|
||||
client; the spec's five checks (functional cross-shard — also the
|
||||
deferred cross-shard `call` proof — handshake vector, 1k soak with a
|
||||
`WO_MAILBOX=8` sub-run, drain under both backends + ASan, battery).
|
||||
- ⬜ **T10 closeout** — stories 24/31/34 → done/ with banners (note the
|
||||
scalar-reply v1 narrowing + three-argument monitor deviations), board
|
||||
standup entry, graph nodes, framework README ledger rows, runtime +
|
||||
chat CODE-LOGIC sections, delete this marker. Final battery.
|
||||
|
||||
This file is deleted when the slice lands (board convention). It lives flat in
|
||||
`docs/` rather than a status folder — since 2026-08-26 no directory in this repo
|
||||
encodes state; `status:` above is the only place it is recorded.
|
||||
82
docs/examples/chat/CODE-LOGIC.md
Normal file
82
docs/examples/chat/CODE-LOGIC.md
Normal file
|
|
@ -0,0 +1,82 @@
|
|||
# `docs/examples/chat` — how the sample is put together
|
||||
|
||||
Iteration 24's acceptance workload: rooms, presence and broadcast over
|
||||
WebSocket, actors on fibers across shards, one binary, no broker. It exists to
|
||||
*drive* the actor work, so nearly every shape here is chosen to exercise
|
||||
something the runtime claims.
|
||||
|
||||
Gate: `just chat` (`scripts/chat-accept.sh`), which logs to `/tmp/chat.log` —
|
||||
`tail -F` it while the gate runs.
|
||||
|
||||
## The actors
|
||||
|
||||
| Actor | Owns | Answers |
|
||||
| --- | --- | --- |
|
||||
| `Registry` | name → room map, a fallback room | a `call` returning the room's address; spawns rooms on demand |
|
||||
| `Room` | its member list (writer address + name) | join, leave, a text line, shutdown |
|
||||
| `Reader` | the read half of one connection | nothing — it loops on the fd and sends onward |
|
||||
| `Writer` | the **fd**, and the write half | text, pong, close |
|
||||
| `ConnWorker` | one accepted connection | runs the HTTP layer over that fd |
|
||||
|
||||
`Registry` is the first honest consumer of `call`: the handler runs on the
|
||||
connection worker's shard, the registry lives wherever placement put it, and
|
||||
the reply is a scalar — the room's address. That is the cross-shard `call`
|
||||
proof the gate asserts, not a contrivance added for it.
|
||||
|
||||
## Two actors per connection, not one
|
||||
|
||||
One fd, two directions, and they block independently. A single actor would have
|
||||
to be inside `read` to notice the client, and inside `write` to deliver a
|
||||
broadcast — it cannot be in both, so a broadcast would stall behind a quiet
|
||||
client's read. Splitting them buys three things:
|
||||
|
||||
1. **The `Writer` is the sole writer of that fd.** Frames can never interleave,
|
||||
which for a framed protocol is a correctness property and not a nicety.
|
||||
2. **The `Reader` may block as long as it likes.** It sits in `read_dl` with a
|
||||
30 s idle deadline and nothing else is waiting on it.
|
||||
3. **The `Writer`'s mailbox becomes the backpressure point.** A slow client
|
||||
stops draining its socket, its `Writer` blocks in `write_dl`, its mailbox
|
||||
fills, and the room's next broadcast to it raises a catchable `WO_T_ACTOR`.
|
||||
The room catches that and drops the member. **This is the whole reason the
|
||||
mailbox cap is fail-fast** — the room survives its slowest member, and the
|
||||
gate's `WO_MAILBOX=8` leg proves the path fires rather than assuming it.
|
||||
|
||||
`Room.say` is written around that: it shifts every member, tries the send, and
|
||||
keeps only the members whose send succeeded — a failed one is sent a close and
|
||||
dropped. So fan-out and eviction are the same pass.
|
||||
|
||||
## Who owns the fd
|
||||
|
||||
The `Writer`. It closes it, in every branch: a failed write sets `dead` and
|
||||
closes; a close message writes the close frame and closes. The `Reader` closes
|
||||
the fd itself in exactly one case — when its `send_close` to the writer traps,
|
||||
meaning the writer is unreachable and nobody else will. Without that the fd
|
||||
would leak on a dead-writer path.
|
||||
|
||||
`Writer.dead` guards against a second close, which matters because two
|
||||
independent paths can decide a connection is finished (the reader seeing EOF,
|
||||
and the room broadcasting shutdown).
|
||||
|
||||
## Shutdown choreography
|
||||
|
||||
On `env.stopping()` the accept loop stops and `main` sends one message to the
|
||||
`Registry`, which fans out to every room; each room shifts its members and
|
||||
sends each `Writer` a close; each writer writes the close frame and closes the
|
||||
fd. `main` then spins — it may **not** park, because a park after the stop flag
|
||||
unwinds — and returns, which is what stops the engine.
|
||||
|
||||
Independently, every `Reader` notices `env.stopping()` at its loop head and
|
||||
runs its tail: leave the room, close the writer.
|
||||
|
||||
Both paths exist and that is deliberate: the reader path covers a connection
|
||||
whose room is already gone, the room path covers a reader parked in a read that
|
||||
has not come back yet.
|
||||
|
||||
**This is where iteration 40 came from.** The room path used to be unreliable:
|
||||
a `Room` whose shard was idle at `SIGTERM` never adopted the shutdown message,
|
||||
because an idle worker abandoned its inbox on stop. Clients that still got a
|
||||
close frame were being saved by the reader path alone — which is why the
|
||||
failure looked random and why a warmed-up server hid it. The engine now
|
||||
guarantees that a send issued before the stop flag is delivered, so both paths
|
||||
work as written. Nothing in this file changed to fix it, and that is the point:
|
||||
the sample was right and the runtime was not.
|
||||
335
docs/examples/chat/main.wo
Normal file
335
docs/examples/chat/main.wo
Normal file
|
|
@ -0,0 +1,335 @@
|
|||
-- chat — iteration 24's acceptance workload. Rooms, presence and
|
||||
-- broadcast over WebSocket: every connection is a reader actor (sole fd
|
||||
-- reader) plus a writer actor (sole fd writer); rooms and the registry
|
||||
-- are actors; delivery between them is ownership-moving sends, across
|
||||
-- shards when placement lands them there. One binary, no broker.
|
||||
--
|
||||
-- CHAT_TOKEN is not needed — chat is open; the framework serves it
|
||||
-- through [deps] exactly like web-app:
|
||||
-- woc . && ./target/chat 8080
|
||||
-- ws://127.0.0.1:8080/ws?room=lobby&name=alice
|
||||
--
|
||||
-- The actor split exists because an actor takes ONE message at a time:
|
||||
-- a single per-connection actor blocked in net read could never hear a
|
||||
-- broadcast. The reader owns the socket's inbound half and the carry
|
||||
-- buffer; the writer owns the outbound half so frames never interleave.
|
||||
use env
|
||||
use net
|
||||
use time
|
||||
use porch
|
||||
use porch/http
|
||||
use porch/router
|
||||
|
||||
-- ---- message types (one per actor) --------------------------------------
|
||||
|
||||
-- To a writer: 1 = text frame, 2 = close (frame + fd close), 3 = pong.
|
||||
class WriterMsg {
|
||||
kind: Int
|
||||
text: Text
|
||||
}
|
||||
|
||||
-- To a room: 1 = join, 2 = leave, 3 = text, 4 = shutdown (drain).
|
||||
class RoomMsg {
|
||||
kind: Int
|
||||
name: Text
|
||||
text: Text
|
||||
writer: actor WriterMsg
|
||||
}
|
||||
|
||||
-- To the registry: 1 = lookup (a `call` — the reply is the room's
|
||||
-- address), 2 = shutdown every room (a `send` on SIGTERM).
|
||||
class Lookup {
|
||||
kind: Int
|
||||
room: Text
|
||||
}
|
||||
|
||||
-- To a reader: everything the connection's inbound loop needs.
|
||||
class ReaderMsg {
|
||||
fd: net.Conn
|
||||
room: actor RoomMsg
|
||||
writer: actor WriterMsg
|
||||
name: Text
|
||||
}
|
||||
|
||||
-- One connection accepted, one worker: builds its own App and runs the
|
||||
-- framework's keep-alive loop (the serving-slice pattern).
|
||||
class Conn {
|
||||
fd: net.Conn
|
||||
}
|
||||
|
||||
-- ---- the writer: sole owner of the outbound half -------------------------
|
||||
|
||||
class Writer {
|
||||
fd: net.Conn
|
||||
dead: Int
|
||||
fn receive(msg: WriterMsg) {
|
||||
if self.dead == 1 { return; }
|
||||
if msg.kind == 1 {
|
||||
let ok = try net.write_dl(self.fd, ws_text(msg.text), 2000) catch (e) false;
|
||||
if ok == false {
|
||||
-- a stalled or gone client: tear the fd; the reader will see EOF
|
||||
-- and route the leave through the room
|
||||
self.dead = 1;
|
||||
net.close(self.fd);
|
||||
}
|
||||
return;
|
||||
}
|
||||
if msg.kind == 3 {
|
||||
let ok2 = try net.write_dl(self.fd, ws_pong(msg.text), 2000) catch (e) false;
|
||||
if ok2 == false {
|
||||
self.dead = 1;
|
||||
net.close(self.fd);
|
||||
}
|
||||
return;
|
||||
}
|
||||
-- close: the drain path (room shutdown or reader-detected close)
|
||||
self.dead = 1;
|
||||
let ig = try net.write_dl(self.fd, ws_close(), 1000) catch (e) false;
|
||||
net.close(self.fd);
|
||||
}
|
||||
}
|
||||
|
||||
-- ---- the room: members, presence, fan-out --------------------------------
|
||||
|
||||
class Mem {
|
||||
w: actor WriterMsg
|
||||
name: Text
|
||||
}
|
||||
|
||||
class Room {
|
||||
members: multi Mem
|
||||
fn receive(msg: RoomMsg) {
|
||||
if msg.kind == 1 {
|
||||
push(self.members, Mem { w: msg.writer, name: "${msg.name}" });
|
||||
self.say("* ${msg.name} joined");
|
||||
return;
|
||||
}
|
||||
if msg.kind == 2 {
|
||||
let keep: multi Mem = [];
|
||||
while len(self.members) > 0 {
|
||||
let m = shift(self.members);
|
||||
if m.name != msg.name { push(keep, m); }
|
||||
}
|
||||
self.members = keep;
|
||||
self.say("* ${msg.name} left");
|
||||
return;
|
||||
}
|
||||
if msg.kind == 3 {
|
||||
self.say("${msg.name}: ${msg.text}");
|
||||
return;
|
||||
}
|
||||
-- shutdown: every member gets a close frame; the list empties
|
||||
while len(self.members) > 0 {
|
||||
let m = shift(self.members);
|
||||
let r = try send_close(m.w) catch (e) 0;
|
||||
}
|
||||
}
|
||||
|
||||
-- fan-out one line; a member whose mailbox is FULL is a slow client —
|
||||
-- the fail-fast cap turns it into a drop-from-the-room (the backpressure
|
||||
-- policy earning its keep)
|
||||
fn say(line: Text) {
|
||||
let keep: multi Mem = [];
|
||||
while len(self.members) > 0 {
|
||||
let m = shift(self.members);
|
||||
let ok = try send_text(m.w, "${line}") catch (e) 0;
|
||||
if ok == 1 {
|
||||
push(keep, m);
|
||||
} else {
|
||||
let r = try send_close(m.w) catch (e) 0;
|
||||
}
|
||||
}
|
||||
self.members = keep;
|
||||
}
|
||||
}
|
||||
|
||||
-- send wrappers: `try` is an expression, so give it Int results
|
||||
fn send_text(w: actor WriterMsg, line: Text) -> Int {
|
||||
send(w, WriterMsg { kind: 1, text: line });
|
||||
return 1;
|
||||
}
|
||||
|
||||
fn send_close(w: actor WriterMsg) -> Int {
|
||||
send(w, WriterMsg { kind: 2, text: "" });
|
||||
return 1;
|
||||
}
|
||||
|
||||
-- ---- the registry: name -> room, spawn on demand --------------------------
|
||||
|
||||
class RoomRef {
|
||||
r: actor RoomMsg
|
||||
}
|
||||
|
||||
class Registry {
|
||||
rooms: map<Text, RoomRef>
|
||||
fallback: actor RoomMsg
|
||||
fn receive(msg: Lookup) -> actor RoomMsg {
|
||||
if msg.kind == 2 {
|
||||
for k, v in self.rooms {
|
||||
send(v.r, RoomMsg { kind: 4, name: "", text: "", writer: dummy_writer() });
|
||||
}
|
||||
return self.fallback;
|
||||
}
|
||||
if has(self.rooms, msg.room) == 1 {
|
||||
let have = self.rooms[msg.room];
|
||||
if have != nil {
|
||||
return have.r;
|
||||
}
|
||||
}
|
||||
let room: actor RoomMsg = spawn Room { members: [] };
|
||||
self.rooms[msg.room] = RoomRef { r: room };
|
||||
return room;
|
||||
}
|
||||
}
|
||||
|
||||
-- RoomMsg requires a writer field on every construction; the shutdown
|
||||
-- message has no meaningful one, so a throwaway satisfies the shape (it
|
||||
-- never receives anything — kind 4 reads no fields).
|
||||
fn dummy_writer() -> actor WriterMsg {
|
||||
let w: actor WriterMsg = spawn Writer { fd: 0 - 1, dead: 1 };
|
||||
return w;
|
||||
}
|
||||
|
||||
-- ---- the reader: sole owner of the inbound half ---------------------------
|
||||
|
||||
class Reader {
|
||||
pad: Int
|
||||
fn receive(msg: ReaderMsg) {
|
||||
let carry = "";
|
||||
let alive = true;
|
||||
while alive {
|
||||
if env.stopping() { alive = false; continue; }
|
||||
let got = try net.read_dl(msg.fd, 4096, 30000) catch (e) nil;
|
||||
if got == nil {
|
||||
-- idle deadline or I/O trap: this client is done
|
||||
alive = false;
|
||||
continue;
|
||||
}
|
||||
let bytes = "${got}";
|
||||
if len(bytes) == 0 {
|
||||
alive = false;
|
||||
continue;
|
||||
}
|
||||
carry = carry .. bytes;
|
||||
let more = true;
|
||||
while more {
|
||||
let f = ws_parse(carry);
|
||||
if f.kind == 0 {
|
||||
more = false;
|
||||
continue;
|
||||
}
|
||||
carry = f.rest;
|
||||
if f.kind == 1 {
|
||||
send(msg.room, RoomMsg { kind: 3, name: "${msg.name}", text: f.payload, writer: msg.writer });
|
||||
continue;
|
||||
}
|
||||
if f.kind == 9 {
|
||||
send(msg.writer, WriterMsg { kind: 3, text: f.payload });
|
||||
continue;
|
||||
}
|
||||
if f.kind == 10 or f.kind == 2 {
|
||||
continue; -- pongs ignored; binary tolerated (echo is not chat)
|
||||
}
|
||||
-- close frame or protocol error: stop reading
|
||||
alive = false;
|
||||
more = false;
|
||||
}
|
||||
}
|
||||
-- the tail sends must survive full mailboxes (a leave storm after a
|
||||
-- mass close): a trap here would kill the reader and orphan the fd
|
||||
let r1 = try send_leave(msg.room, "${msg.name}", msg.writer) catch (e) 0;
|
||||
let r2 = try send_close(msg.writer) catch (e) 0;
|
||||
if r2 == 0 {
|
||||
-- the writer is unreachable (full/dead): close the fd ourselves
|
||||
net.close(msg.fd);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn send_leave(room: actor RoomMsg, name: Text, w: actor WriterMsg) -> Int {
|
||||
send(room, RoomMsg { kind: 2, name: name, text: "", writer: w });
|
||||
return 1;
|
||||
}
|
||||
|
||||
-- ---- HTTP: the upgrade route + usage --------------------------------------
|
||||
|
||||
class WsRoute {
|
||||
reg: actor Lookup
|
||||
fn handle(req: Req) -> Resp {
|
||||
if ws_upgrade_valid(req) == false {
|
||||
return bad_request("expected a websocket upgrade");
|
||||
}
|
||||
let rname = req.query["room"];
|
||||
if rname == nil { return bad_request("expected ?room=<name>&name=<who>"); }
|
||||
let who = req.query["name"];
|
||||
if who == nil { return bad_request("expected ?room=<name>&name=<who>"); }
|
||||
-- the cross-shard call: this handler runs on the connection worker's
|
||||
-- shard, the registry lives wherever placement put it
|
||||
let room = call(self.reg, Lookup { kind: 1, room: "${rname}" });
|
||||
let fd = ws_accept(req);
|
||||
let w: actor WriterMsg = spawn Writer { fd: fd, dead: 0 };
|
||||
let rd: actor ReaderMsg = spawn Reader { pad: 0 };
|
||||
send(room, RoomMsg { kind: 1, name: "${who}", text: "", writer: w });
|
||||
send(rd, ReaderMsg { fd: fd, room: room, writer: w, name: "${who}" });
|
||||
return hijacked();
|
||||
}
|
||||
}
|
||||
|
||||
class Usage {
|
||||
pad: Int
|
||||
fn handle(req: Req) -> Resp {
|
||||
return ok_json("{\"ws\":\"/ws?room=<name>&name=<who>\"}");
|
||||
}
|
||||
}
|
||||
|
||||
fn build_app(reg: actor Lookup) -> App {
|
||||
let app = App { middleware: [], routes: [] };
|
||||
app.get("/", Usage { pad: 0 });
|
||||
app.get("/ws", WsRoute { reg: reg });
|
||||
return app;
|
||||
}
|
||||
|
||||
class ConnWorker {
|
||||
reg: actor Lookup
|
||||
fn receive(msg: Conn) {
|
||||
let app = build_app(self.reg);
|
||||
app.handle_conn(msg.fd, 10000, 10000);
|
||||
}
|
||||
}
|
||||
|
||||
fn main(args: multi Text) -> Int {
|
||||
if len(args) < 1 {
|
||||
print_err("usage: chat <port>");
|
||||
return 2;
|
||||
}
|
||||
let port = parse_int(args[0]);
|
||||
if port == nil {
|
||||
print_err("chat: <port> must be a number");
|
||||
return 2;
|
||||
}
|
||||
let fb: actor RoomMsg = spawn Room { members: [] };
|
||||
let reg: actor Lookup = spawn Registry { rooms: {}, fallback: fb };
|
||||
let srv = net.listen("127.0.0.1", port);
|
||||
print("listening on 127.0.0.1:${port}");
|
||||
while true {
|
||||
if env.stopping() {
|
||||
-- the drain: every room broadcasts a close frame and writers flush.
|
||||
-- main must NOT park here (a park after the stop flag unwinds), so
|
||||
-- it SPINS — each loop back-edge pays a reduction, and the budget
|
||||
-- hands the shard to the draining actors between slices; worker
|
||||
-- shards keep adopting their inboxes until the engine stops.
|
||||
send(reg, Lookup { kind: 2, room: "" });
|
||||
let spin = 0;
|
||||
while spin < 20000000 {
|
||||
spin = spin + 1;
|
||||
}
|
||||
net.close(srv);
|
||||
return 0;
|
||||
}
|
||||
let c = net.accept_dl(srv, 250);
|
||||
if c != nil {
|
||||
let w: actor Conn = spawn ConnWorker { reg: reg };
|
||||
send(w, Conn { fd: c });
|
||||
}
|
||||
}
|
||||
}
|
||||
9
docs/examples/chat/wo.toml
Normal file
9
docs/examples/chat/wo.toml
Normal file
|
|
@ -0,0 +1,9 @@
|
|||
name = "chat"
|
||||
version = "0.1.0"
|
||||
description = "Iteration 24's acceptance workload: rooms + presence + broadcast over WebSocket — actors on fibers across shards, one binary, no broker"
|
||||
|
||||
[runtime]
|
||||
wo = ">= 0.1"
|
||||
|
||||
[deps]
|
||||
porch = { git = "https://github.com/shoneyj/porch", rev = "v0.1.0" }
|
||||
|
|
@ -56,7 +56,8 @@ place it runs.
|
|||
feature.
|
||||
- **Why `main` waits.** `main` is not an actor and has no mailbox, so it sleeps
|
||||
rather than awaiting — the gap iteration 31's `call` closes for actors and
|
||||
[24's marker](../../active-slice-2026-08-23-chat-ws-lifecycle.md) tracks.
|
||||
[iteration 24](../../stories/language-runtime-database/24-chat-websocket-workload.md)
|
||||
landed 2026-08-27.
|
||||
|
||||
Reasoning under the engine side: [`database/src/CODE-LOGIC.md`](../../../database/src/CODE-LOGIC.md).
|
||||
Contract: [`plan/oop-vm/04-db-binding.md`](../../plan/oop-vm/04-db-binding.md).
|
||||
|
|
|
|||
|
|
@ -24,9 +24,26 @@ strictly better. Recorded as a plan deviation.)
|
|||
| `query N` | full equality probes on the k index (≈10 rows each), materialized and counted. |
|
||||
| `write N` | alternating inserts (disjoint k range 2e6+) and updates through query results. Corrupts the checksum by design — durability legs run on a fresh store. |
|
||||
| `wal N` | the crash battery's vehicle: insert-only (k range 1e6+), `acked <i>` printed AFTER each insert returns — the return IS the ack (RAM applied, WAL record staged, ONE commit done). |
|
||||
| `wmix N C` | **databasev2 4:** every op a durable write (update through a query result), C at once. Exists because `mix` writes on one op in ten with C=4 — 20 writes in a quick run, measured mean batch **1.01** — so no existing leg could show whether group commit engages. Histogram kind 2, because a replayed store still holds the seeding run's kind-0/1 `Hist` rows. Seed first. |
|
||||
| `boot` | **databasev2 3:** does NOTHING. With `WO_DATA` set the runtime replays the whole log before `main` runs, so a mode with no work of its own is the only honest way to price boot |
|
||||
| `verify` | store vs its own Meta rows: count, checksum, one unique probe. Exit 3 on mismatch. |
|
||||
| `verify-acked M` | after kill -9 mid-`wal`: rows 1..M exist with the right v; rows beyond M allowed (acked after the last print flushed). Exit 3 on mismatch. |
|
||||
|
||||
## Env knobs
|
||||
|
||||
| var | effect |
|
||||
| --- | --- |
|
||||
| `WO_DATA=<dir>` | durability on: replay `<dir>/shard-0.wal` at boot, log every write. Without it the store is RAM-only |
|
||||
| `WO_SHARDS=<n>` | shard count. **`1` means every statement runs inline on shard 0 and group commit cannot engage** — batches form only where writes queue from other shards |
|
||||
| `WO_CHECKPOINT_BYTES` / `WO_CHECKPOINT_RATIO` | **databasev2 3:** the checkpoint trigger — the log must exceed the floor AND exceed the ratio times the last compaction's own size. A tiny floor forces compaction in a few writes, which is how the gate tests the policy at all; an enormous one disables it, which is how the checkpoint leg measures the same workload with and without |
|
||||
| `WO_WAL_STATS=1` | **databasev2 4:** print one line at exit — `walstats batches=… records=… peak_batch=… peak_staged=… compactions=… compact_us_max=… compact_us_total=… compacted_bytes=…`. Opt-in so it does not pollute every durable program's output. Mean batch is `records/batches`; **mean 1.0 means group commit is not engaging**, which is expected for a serial writer or `WO_SHARDS=1` and a bug anywhere else |
|
||||
|
||||
**Do not put `WO_DATA` on `/tmp`.** It is `tmpfs` on the reference machine,
|
||||
where `fdatasync` is free: the same `wmix` run measured **195 000 ops/s at p50
|
||||
1 µs** there against **2200 ops/s at p50 7200 µs** on ext4. There is no
|
||||
durability barrier to price on a memory filesystem. The driver keeps its stores
|
||||
under `bench/` for exactly this reason.
|
||||
|
||||
## Coordination idiom (this side of iteration 31)
|
||||
|
||||
There is no request/response surface yet: concurrent modes drive
|
||||
|
|
|
|||
|
|
@ -338,6 +338,91 @@ class Mixer {
|
|||
}
|
||||
}
|
||||
|
||||
-- databasev2 4 part A: every op a durable write, C at once.
|
||||
--
|
||||
-- Why this leg exists. `mix` writes on one op in ten with C=4, so at most a
|
||||
-- handful of writes are ever in flight and group commit has almost nothing to
|
||||
-- batch: measured mean batch 1.01 over 3112 barriers, peak 3. That is a
|
||||
-- property of the WORKLOAD, not of the mechanism, and without a write-
|
||||
-- concurrent leg the iteration's payoff cannot be evaluated either way.
|
||||
--
|
||||
-- Updates rather than inserts: comparable to what `mixwrite` measures, and the
|
||||
-- row count stays flat so a long run does not turn into a growth test.
|
||||
-- Histogram kind 2, because a replayed store still holds the seeding run's
|
||||
-- kind-0/1 Hist rows and merging those would report someone else's latencies.
|
||||
class WJob {
|
||||
ops: Int
|
||||
seed: Int
|
||||
kmod: Int
|
||||
}
|
||||
|
||||
class WMixer {
|
||||
id: Int
|
||||
fn receive(msg: WJob) {
|
||||
let hw: map<Int, Int> = {};
|
||||
let s = msg.seed;
|
||||
let i = 0;
|
||||
while i < msg.ops {
|
||||
s = lcg(s);
|
||||
let key = s % msg.kmod;
|
||||
let o0 = time.ticks();
|
||||
for r in from x in Item where x.k == key take 1 select x {
|
||||
r.v = r.v + 1;
|
||||
}
|
||||
hist_add(hw, time.ticks() - o0);
|
||||
i = i + 1;
|
||||
}
|
||||
hist_dump(hw, 2);
|
||||
insert Meta { tag: "wmixdone${self.id}", val: msg.ops };
|
||||
}
|
||||
}
|
||||
|
||||
fn wmix_mode(total: Int, c: Int) -> Int {
|
||||
let kmod = meta_val("kmod");
|
||||
if kmod < 1 {
|
||||
print_err("wmix: seed first");
|
||||
return 1;
|
||||
}
|
||||
let per = total / c;
|
||||
if per < 1 {
|
||||
per = 1;
|
||||
}
|
||||
let wall0 = time.ticks();
|
||||
let i = 0;
|
||||
while i < c {
|
||||
let a: actor WJob = spawn WMixer { id: i };
|
||||
send(a, WJob { ops: per, seed: 4242 + i * 7919, kmod: kmod });
|
||||
i = i + 1;
|
||||
}
|
||||
let done = 0;
|
||||
while done < c {
|
||||
time.sleep(20);
|
||||
done = 0;
|
||||
i = 0;
|
||||
while i < c {
|
||||
if meta_val("wmixdone${i}") >= 0 {
|
||||
done = done + 1;
|
||||
}
|
||||
i = i + 1;
|
||||
}
|
||||
}
|
||||
let wall = time.ticks() - wall0;
|
||||
let hw: map<Int, Int> = {};
|
||||
let nw = 0;
|
||||
for x in from x in Hist select x {
|
||||
if x.kind == 2 {
|
||||
if has(hw, x.b) {
|
||||
set(hw, x.b, get(hw, x.b) + x.c);
|
||||
} else {
|
||||
set(hw, x.b, x.c);
|
||||
}
|
||||
nw = nw + x.c;
|
||||
}
|
||||
}
|
||||
report("wmix", nw, wall, hw);
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn mix_mode(total: Int, c: Int) -> Int {
|
||||
let kmod = meta_val("kmod");
|
||||
if kmod < 1 {
|
||||
|
|
@ -466,7 +551,7 @@ fn all_mode(n: Int) -> Int {
|
|||
fn usage() -> Int {
|
||||
print_err("usage: db-bench <mode>");
|
||||
print_err(" all N | seed N | read N | query N | write N | wal N");
|
||||
print_err(" mix N C | msgrate N | growth N int|text | growth-verify");
|
||||
print_err(" mix N C | wmix N C | msgrate N | growth N int|text | growth-verify");
|
||||
print_err(" randread N R | replayseed N M | boot");
|
||||
print_err(" verify | verify-acked M");
|
||||
return 2;
|
||||
|
|
@ -683,6 +768,10 @@ fn main(args: multi Text) -> Int {
|
|||
if args[0] == "growth-verify" {
|
||||
return growth_verify();
|
||||
}
|
||||
-- Does NOTHING. With WO_DATA set the runtime replays the whole log before
|
||||
-- main runs, so a mode with no work of its own measures replay plus a fixed
|
||||
-- process start — which is what "boot time" has to mean. Both databasev2 1
|
||||
-- (replay baseline) and databasev2 3 (checkpoint boot) price boot with it.
|
||||
if args[0] == "boot" {
|
||||
return boot_mode();
|
||||
}
|
||||
|
|
@ -746,6 +835,17 @@ fn main(args: multi Text) -> Int {
|
|||
}
|
||||
return randread_mode(n, rr);
|
||||
}
|
||||
if args[0] == "wmix" {
|
||||
if len(args) < 3 {
|
||||
return usage();
|
||||
}
|
||||
let wc = parse_int(args[2]);
|
||||
if wc == nil or wc < 1 {
|
||||
print_err("db-bench: <c> must be a positive number");
|
||||
return 2;
|
||||
}
|
||||
return wmix_mode(n, wc);
|
||||
}
|
||||
if args[0] == "mix" {
|
||||
if len(args) < 3 {
|
||||
return usage();
|
||||
|
|
|
|||
|
|
@ -75,7 +75,11 @@ porch = { git = "https://github.com/shoneyj/porch", rev = "v0.1.0" }
|
|||
- **TLS: none, anywhere.** Deploy behind nginx/caddy; the proxy terminates
|
||||
TLS+ALPN and gives browsers HTTP/2 while this backend speaks HTTP/1.1
|
||||
keep-alive. See the web-app sample's README for the nginx sketch.
|
||||
- `Content-Length` bodies only (no chunked encoding), no WebSockets/SSE,
|
||||
- `Content-Length` bodies only (no chunked encoding); **WebSockets ARE
|
||||
supported since 2026-08-27** — `ws_accept` (`http/ws.wo`) performs the RFC
|
||||
6455 handshake and hands back the hijacked `net.Conn`, and `http/wsframe.wo`
|
||||
is a pure-`.wo` frame codec; `docs/examples/chat` is the worked example and
|
||||
`just chat` its gate. **SSE is still absent**, and so is chunked encoding.
|
||||
JSON-first (no templates). Form-encoded bodies parse through
|
||||
`form_values(req)` (`+` and `%XX` decoded, nil on any other
|
||||
content-type); multipart/form-data through `multipart_parts(req)`
|
||||
|
|
@ -130,6 +134,7 @@ first (pure `.wo` cannot express it yet).
|
|||
| Content negotiation | ✅ `media_type(req)` request-side; `accepts(req, mtype)` response-side (exact, type/*, */*; q-values stripped not ranked — ranking waits for an app serving alternates) — slice 2 |
|
||||
| Trusted-proxy client IP | 🔶 `client_ip(req)` parses X-Forwarded-For; `net.peer(fd)` (iteration 35) exposes the peer — the verify middleware is now a pure-`.wo` candidate slice |
|
||||
| Status/header setting · redirects | ✅ builders + `set_header` |
|
||||
| WebSockets · pub/sub | ✅ **2026-08-27 (iteration 24)** — `ws_accept` does the RFC 6455 handshake and hands back the hijacked `net.Conn`; `http/wsframe.wo` is a pure-`.wo` frame codec. Rooms/presence/broadcast are actors in `docs/examples/chat`, gated by `just chat` (11 checks, 1000-client soak, both `WO_IO` backends, ASan clean). No SSE |
|
||||
| Lazy body streaming + backpressure · streaming responses · explicit commit point | ⏸ UNBLOCKED by the arc (8/11 landed 2026-08-21) — stays parked until its own slice |
|
||||
| ETag + conditional requests | ✅ `etag_for` (quoted base64 SHA-256) + `with_etag` (If-None-Match → 304) over iteration 34's digest builtins — slice 2 |
|
||||
|
||||
|
|
@ -140,7 +145,7 @@ first (pure `.wo` cannot express it yet).
|
|||
| Ordered middleware chain | ✅ registration order, `?Resp` short-circuits |
|
||||
| Request-scoped context | ✅ `req.ctx` map (slice 2): middleware writes, handlers read; identity stays in `principal` |
|
||||
| Guaranteed teardown | 🔶 every fd closes on every path (gate-proven); no user teardown hooks yet |
|
||||
| Cancellation into pending storage ops | ⏸ UNBLOCKED by the arc (8/11 landed 2026-08-21) — stays parked until its own slice |
|
||||
| Cancellation into pending storage ops | ⏸ **unblocked, not built.** The arc landed 2026-08-21 and iteration 24 (2026-08-27) added the lifecycle a cancellation would ride — `call` with a catchable trap when the callee dies, bounded mailboxes, `monitor`, and `time.after` for a deadline. Nothing here consumes them yet; it stays parked until its own slice |
|
||||
| Panic recovery | 🔶 trap = 500 and the server survives ✅; "rolls back the transaction" is framework v2 (needs `transaction { }`, iteration 18) |
|
||||
|
||||
### Storage integration (the differentiator — framework v2 territory)
|
||||
|
|
|
|||
|
|
@ -29,6 +29,31 @@ Writeonce's phase 12 `Engine` keeps an `HashMap<(TypeName, SegmentOffset), Cache
|
|||
The kernel page cache does most of the work. `pread` against an fd that already has its page cached is a memcpy. `pwrite` populates the page cache without going to disk until pressure or `fsync`. This is why writeonce explicitly does NOT use `O_DIRECT` (see [`linux/12-pwrite-fsync.md`](../linux/12-pwrite-fsync.md)) — the page cache is the one cache we want.
|
||||
|
||||
## Checkpoint — the writeonce shape
|
||||
> **⚠ TWO CORRECTIONS, 2026-08-28** (found while brainstorming
|
||||
> [databasev2 3](../../../stories/databasev2/03-wal-checkpoint.md); spec:
|
||||
> [`2026-08-28-wal-checkpoint-design.md`](../../../superpowers/specs/2026-08-28-wal-checkpoint-design.md)).
|
||||
>
|
||||
> 1. **Postgres does NOT update its control file by rename.** The claim below
|
||||
> that "Postgres does the same in `BasicOpenFile` + `fsync_parent_path`" is
|
||||
> wrong: `update_controlfile` (`src/common/controldata_utils.c`) opens the
|
||||
> existing file `O_WRONLY`, writes a zero-padded **full block in place**, and
|
||||
> relies on **CRC32C** over the struct to detect a torn write. The
|
||||
> `fsync(parent_dir)` reasoning below is still correct *for renames* — it is
|
||||
> just not what Postgres does here.
|
||||
> 2. **The checkpoint sketch below assumes writeonce has segment files.** It
|
||||
> says records before the LSN are "*known* to be in the segment files". There
|
||||
> are none: the WAL is writeonce's only durable form, replayed into RAM, and
|
||||
> [databasev2 2](../../../stories/databasev2/02-table-storage-modes.md)
|
||||
> deliberately rejected adding a paged store. This document predates the
|
||||
> databasev2 direction, so read the loop below as a design for an
|
||||
> architecture that was not chosen.
|
||||
>
|
||||
> What survived the comparison is the **ordering discipline**, not the
|
||||
> architecture: publish the new "recovery starts here" atomically and last, so a
|
||||
> crash falls back. writeonce gets that from one `rename` of the whole log —
|
||||
> possible only because its records are full row images, where Postgres' are
|
||||
> page deltas.
|
||||
|
||||
|
||||
Postgres' checkpoint runs in a separate process and signals the postmaster when done. Writeonce's runs as a periodic loop step:
|
||||
|
||||
|
|
|
|||
|
|
@ -65,7 +65,7 @@ The metadata exists for exactly one reason: `json.encode`/`json.decode` are runt
|
|||
- **the OS half** — fs.exists/list/stat/read_all/read_at/append, time.sleep/local/iso, env.get/stopping, net.listen/accept/read/write/close, proc.run. Ids 40–56; `runtime/src/sysio.c`. A member that returns a record takes that record's **class id as its last argument**, so the VM allocates what it fills without knowing any source type name.
|
||||
- **json** — encode (value + the value's static kind), decode (text + the class id to build). Ids 57–58; `runtime/src/json.c`. Decode yields the zero word on malformed input rather than trapping, which is what makes `json.decode(t) as T` a checked decode.
|
||||
- **59 `map_get_opt`** (`m[k]`'s optional read), **60 `text_copy`** (Text's ownership-boundary copy — Task 1 of the executable plan).
|
||||
- **database** — **61 `db_insert`** (iteration 9, Task 3): window is R[B] = class id, R[B+1..] = one slot per **declared** field in declaration order; result R[A] = the new row's id. The loader validates the class-id slot statically (variable window: the field slots are validated at runtime by the engine against the class table). Engine failure traps `WO_T_DB`; a failed WAL commit traps `WO_T_IO` after un-applying the row. `database/src/db.c`.
|
||||
- **database** — **61 `db_insert`** (iteration 9, Task 3): window is R[B] = class id, R[B+1..] = one slot per **declared** field in declaration order; result R[A] = the new row's id. The loader validates the class-id slot statically (variable window: the field slots are validated at runtime by the engine against the class table). Engine failure traps `WO_T_DB`. **A failed WAL commit no longer traps (databasev2 4, 2026-08-28): it ENDS THE PROCESS** with exit status 74 and a diagnostic naming the failing operation, the log path, `errno` and the batch size. `WO_T_IO` is unreachable from any DB write. The reason is that only `insert` could ever un-apply itself — `update` and `delete` never could, and their own comments admitted they left RAM ahead of disk — so continuing after a durability failure meant serving state that would not survive a restart. Retrying is not offered either: on Linux a failed `fsync` may already have discarded the dirty pages, so a second call can report success having written nothing. Replay is the recovery that works. `database/src/db.c`, `database/src/wal.c`.
|
||||
|
||||
**`?T` and nil.** A heap-shaped optional (`?Text`, `?Rec`, `?multi`, `?map`, `?@gc`) stores what `T` stores and spells nil as the **zero word** — every per-kind drop plan already ignores a zero slot, so `?T`'s field kind is `T`'s. A **nullable scalar** (`?Int`, `?Bool`, `?Timestamp`, `?Id`) cannot: `0` is a perfectly good `Int`, and real programs store it in a `?Int`. Its nil is therefore `WO_NIL_SCALAR` = −2^62 (not `INT64_MIN`: the compiler's own integers are 63-bit, so that value is not expressible on the emitting side). Such a field is marked `WOB_FIELD_NIL_SCALAR` in `field_class[i]`, which is how the runtime knows to write that word where it must produce absence itself — today only `json.decode` leaving a key absent, and `parse_int` on unparseable input.
|
||||
|
||||
|
|
|
|||
|
|
@ -118,11 +118,49 @@ R[B+1..] = one slot per declared field in declaration order (the literal's
|
|||
order is irrelevant — slots are the class table's).
|
||||
|
||||
Execution: `wo_row_insert` (RAM, engine copies every value), then — when
|
||||
durability is on — stage + **commit before the builtin returns**: the
|
||||
builtin's return IS the acknowledgment, so ack-after-fsync holds at
|
||||
statement granularity until iteration 8 brings tick-scoped group commit. A
|
||||
failed commit un-applies the row and traps `WO_T_IO`; engine failures trap
|
||||
`WO_T_DB`. Durability is opt-in: `WO_DATA=<dir>` makes the CLI replay
|
||||
durability is on — stage, then a barrier before the acknowledgment. **Updated
|
||||
2026-08-28 (databasev2 4 part A): group commit landed, and the barrier's
|
||||
location now depends on which path the statement takes.**
|
||||
|
||||
A statement arriving from a worker shard marshals to shard 0 and parks; shard 0
|
||||
stages every such request, issues **one** barrier when its queue empties, and
|
||||
only then releases the held replies — so each writer is acknowledged after the
|
||||
barrier that carried *its* record. A statement already running on shard 0 takes
|
||||
the inline path and still commits before the builtin returns, because it has no
|
||||
reply to hold: it returns into its own fiber, and batching it would require
|
||||
parking that fiber on the barrier (deferred to part B). The boundary is the
|
||||
queue draining, **not** the tick this document previously anticipated — a tick
|
||||
would add latency to a lone writer, taxing an idle system to serve a busy one.
|
||||
|
||||
Measured: ~2.9× durable write throughput and ~2.1× lower p50 on a
|
||||
write-concurrent workload; unchanged for a serial writer, which has nothing to
|
||||
batch with.
|
||||
|
||||
**Compaction (databasev2 3, 2026-08-29) may run only where NOTHING IS STAGED.**
|
||||
That is a correctness requirement, not a scheduling preference: the staging
|
||||
buffer holds records destined for a file that compaction is about to replace, so
|
||||
compacting with a non-empty buffer would either write them into a file about to
|
||||
be discarded or lose them with it. In practice the safe points are immediately
|
||||
after a barrier — the drain's, and the inline path's — and both are wired.
|
||||
`wo_wal_compact` refuses a non-empty buffer as a backstop rather than trusting
|
||||
its callers.
|
||||
|
||||
**Recovery is unchanged by compaction.** The result is an ordinary log in the
|
||||
ordinary record grammar, replayed from byte 0; there is no snapshot, no second
|
||||
source, no cutoff offset and no control file. Crash safety comes from `rename`
|
||||
being atomic: before it the live log is intact and the temp file is not
|
||||
authoritative, after it the new log is complete, and no reader can observe a
|
||||
mixture. A crash mid-rewrite leaves a temp file, which the next open removes.
|
||||
|
||||
A failed compaction is a **missed optimisation, not a durability event** — the
|
||||
original log is left usable and the process continues. It must not take the
|
||||
fatal path below.
|
||||
|
||||
A failed commit **no longer traps — it ends the process** (exit 74, with a
|
||||
diagnostic naming the operation, log path, `errno` and batch size). So does a
|
||||
failed staging. `WO_T_IO` is unreachable from a DB write. One rule: once a
|
||||
statement has mutated RAM, the outcomes are durable or death. Engine failures
|
||||
still trap `WO_T_DB`. Durability is opt-in: `WO_DATA=<dir>` makes the CLI replay
|
||||
`<dir>/shard-0.wal` before the entry runs and commit every insert; without
|
||||
it the engine is RAM-only (every corpus fixture runs that way).
|
||||
|
||||
|
|
|
|||
|
|
@ -279,6 +279,8 @@ unset `env.get` are nil.
|
|||
| `sha1(bytes)` | `-> Bytes` | 20-byte digest (id 85, iteration 34) — exists because RFC 6455's Sec-WebSocket-Accept demands SHA-1 |
|
||||
| `sha256(bytes)` | `-> Bytes` | 32-byte digest (id 86, iteration 34) |
|
||||
| `hmac_sha256(key, msg)` | `-> Bytes` | RFC 2104 over SHA-256, both args Bytes (id 87, iteration 34); key > 64 bytes hashed first |
|
||||
| `monitor(watched, observer, msg)` | — | iteration 24 (id 89): the observer's own M-typed msg (MOVED) is delivered when watched dies (trap-death); already-dead delivers now; a full observer's notice is dropped with a stderr line — no fiber to trap |
|
||||
| `time.after(ms, addr, msg)` | — | iteration 24 (id 90): one-shot timer — msg (MOVED) arrives as an ordinary send after ms on the arming shard; ms <= 0 delivers now; NO cancel — the generation-counter idiom (run/timer-generation) is the answer |
|
||||
| `call(addr, msg)` | `-> R` | send that WAITS (id 88, iteration 24): the message moves like `send`'s, the caller's fiber parks until the receive's return value arrives. R = the receive's declared return type — every `receive(msg: M)` program-wide must agree on it and it must be a copyable scalar in v1 (WO-E226 otherwise). A dead callee traps WO_T_ACTOR, immediately or mid-call — a `call` never hangs |
|
||||
| `env.get(name)` | `-> ?Text` | unset is nil |
|
||||
| `env.stopping()` | `-> Bool` | SIGTERM/SIGINT latch, handlers installed on first use |
|
||||
|
|
|
|||
|
|
@ -216,3 +216,187 @@ which is too coarse for the one number a checkpoint is meant to improve.
|
|||
|
||||
WAL bytes are measured as the file's **non-zero prefix**, never its size: shard
|
||||
WALs are `fallocate`'d to 1 MiB, so an empty store reports 1048576.
|
||||
## 6. WAL group commit: one barrier per drain (databasev2 4 part A)
|
||||
|
||||
**Measured 2026-08-28.** Before this, the engine committed per *statement*:
|
||||
`db.c` called `wo_wal_commit` immediately after every append, so each row
|
||||
change bought its own `pwrite` + `fdatasync`. Now shard 0 stages every queued
|
||||
write request, issues one barrier, and only then releases the held replies.
|
||||
|
||||
### The controlled before/after
|
||||
|
||||
Same machine, same workload (`wmix 4000 32` — every op a durable update, 32
|
||||
concurrent), same build except `db.c` and `vm.c`, two runs each, interleaved:
|
||||
|
||||
| | ops/sec | p50 | p99 |
|
||||
| --- | --- | --- | --- |
|
||||
| per-statement barrier | 2213 · 2177 | 7183 · 7251 µs | **20000 · 20000 µs** |
|
||||
| group commit | **6216 · 6525** | **3458 · 3444 µs** | 11139 · 5971 µs |
|
||||
|
||||
**≈2.9× throughput, ≈2.1× lower p50.**
|
||||
|
||||
**The p99 "before" figure is at the histogram ceiling, not a measurement.**
|
||||
`hist_add` clamps at 20000 µs, and both before-runs pinned there — so the true
|
||||
before p99 is ≥20 ms and unknown. The improvement is *at least* 2.3×; the
|
||||
honest statement is that the old p99 was off the end of the instrument.
|
||||
|
||||
### Confirmation from the committed baseline
|
||||
|
||||
The full campaign gives the same answer a second way. `s1` takes the inline
|
||||
path, which commits per statement **by design**, so within one build the two
|
||||
shard configurations are batching-off against batching-on:
|
||||
|
||||
| Leg | ops/sec | p50 | p99 | mean batch | peak batch |
|
||||
| --- | --- | --- | --- | --- | --- |
|
||||
| `durable.s1.wmix` (inline, unbatched) | 1467 | 455 µs | 721 µs | **1.0** | 1 |
|
||||
| `durable.sN.wmix` (batched) | **5117** | 8208 µs | 12169 µs | **5.43** | 57 |
|
||||
|
||||
3.5× throughput, agreeing with the 2.9× above. Note `sN` latency is *higher*
|
||||
while throughput is 3.5× better: 64 writers queueing behind one owner shard
|
||||
trade per-op latency for barrier amortisation, which is what group commit is.
|
||||
|
||||
Batching scales with write concurrency exactly as designed — mean batch at
|
||||
C = 4 / 16 / 64 was **1.13 / 1.76 / 5.35**, peak **3 / 10 / 39**.
|
||||
|
||||
### What did NOT improve, and why that was predicted
|
||||
|
||||
`durable.sN.mixwrite` went **480 → 492 ops/s** — unchanged. That is the metric
|
||||
the spec *originally* named as the payoff, and correcting it was part of the
|
||||
brainstorm: `mix` writes on one op in ten with C=4, so a quick run performs
|
||||
**20 writes** and mean batch measured **1.01** over 3112 barriers. A workload
|
||||
that never has two writes in flight cannot be helped by batching them.
|
||||
`durable.*.seed` is likewise unchanged: a serial single writer has nothing to
|
||||
batch with under any scheme.
|
||||
|
||||
**So the payoff is real but conditional: it appears exactly where concurrent
|
||||
durable writes fan into the owner shard, and nowhere else.**
|
||||
|
||||
### Two traps worth recording
|
||||
|
||||
**Do not benchmark durability on `/tmp`.** It is `tmpfs` here, where
|
||||
`fdatasync` is free — the same `wmix` run reported **195 000 ops/s at p50 1 µs**
|
||||
there against **2200 ops/s at p50 7200 µs** on ext4. There is no barrier to
|
||||
amortise on a memory filesystem, so a group-commit measurement taken there
|
||||
measures nothing. `db-bench` gets this right by keeping its stores under
|
||||
`bench/`.
|
||||
|
||||
**The record count is not the update count.** `wmix` staged 7755 records for
|
||||
4000 updates because the histogram dump and the done-marker are themselves
|
||||
durable inserts. They arrive as an end-of-run burst, which is batch-friendly,
|
||||
so `mean_batch` is not purely update-driven. Peak staged bytes stayed small
|
||||
(2793 B at C=64), which is what settled the decision to ship **no batch cap**:
|
||||
the request queue's existing upstream bound is sufficient.
|
||||
|
||||
### The cost side: tail latency on the owner shard
|
||||
|
||||
Group commit is a trade, and the full battery made the other side of it visible.
|
||||
|
||||
**A bug first, caught by `durable.sN.mixread.p99`.** The drain initially held
|
||||
*every* DB reply until the barrier — including **reads**, which stage nothing and
|
||||
have no stake in durability. That parked readers behind an fsync for no reason
|
||||
and pushed read p99 from ~1043 µs to **4057 µs**. Reads are now released
|
||||
immediately; only a statement that actually staged a record has its reply held.
|
||||
|
||||
**What remains is inherent, not a bug.** A barrier now blocks the owner shard
|
||||
**longer** (more records per fsync) even though it blocks **less often**, so
|
||||
anything arriving during a barrier — reads included — waits behind it. Measured
|
||||
across three full runs of the same build, `durable.sN.mixread.p99` came in at
|
||||
**1043 / 2318 / 4147 µs** and `wmix.p99` at **8758 / 20000 µs**, a 2–4× spread
|
||||
with the box near idle.
|
||||
|
||||
So the honest summary of part A on a single-threaded owner shard: **~3× write
|
||||
throughput, at the price of a longer and noisier tail for everything queued
|
||||
behind a barrier.** That is precisely what part B (async submission — submit the
|
||||
barrier and keep serving) would undo, and it is a better argument for part B than
|
||||
the "close the 66× gap" framing part B was originally given.
|
||||
|
||||
**Gating consequence.** `durable.sN.*.p99us` now carries a 100% tolerance,
|
||||
because a 2–4×-variable tail gated at 50% gates the disk rather than the engine.
|
||||
The **floor** is the real guard there, and it is not slack: `mixread`'s floor
|
||||
(4172 µs) came within 25 µs of tripping on the worst observed run.
|
||||
|
||||
## 7. WAL checkpoint: compaction (databasev2 3)
|
||||
|
||||
**Measured 2026-08-29.** Before this the log grew forever: nothing ever removed
|
||||
superseded records, so boot replayed all history and the file only ever got
|
||||
bigger. Compaction rewrites it as one record per live row and swaps it in with
|
||||
`rename`.
|
||||
|
||||
### Space and boot — the same workload, twice
|
||||
|
||||
Identical work, differing only in whether checkpointing may fire (an enormous
|
||||
floor disables it). Full campaign:
|
||||
|
||||
| | checkpointing off | checkpointing on |
|
||||
| --- | --- | --- |
|
||||
| WAL used | 1 962 358 B | **907 094 B** |
|
||||
| boot (median of 3, `boot` mode) | 114 ms | **64 ms** |
|
||||
| compactions | 0 | 6 |
|
||||
|
||||
**2.16× space reclaimed, 1.78× faster boot.** Boot is measured with a mode that
|
||||
does nothing at all: with `WO_DATA` set the runtime replays the whole log before
|
||||
`main` runs, so a mode with no work of its own is the only honest way to price
|
||||
replay. It is *not* measured through the driver's `run()` helper, which samples
|
||||
RSS on a 250 ms poll — timings taken that way reported "251 ms" both with and
|
||||
without checkpointing, which is the harness's clock rather than the engine's.
|
||||
|
||||
### The stop-the-world pause, and why it stopped being 8× worse
|
||||
|
||||
Compaction blocks the owner shard for its duration. The spec refused to assume
|
||||
that was acceptable, so it is measured and gated against a stated **50 ms**
|
||||
budget: a stall a serving process can absorb without a client seeing a timeout.
|
||||
|
||||
Measured **2 651 µs** on the full campaign — comfortably inside it.
|
||||
|
||||
It was not always. The first implementation flushed the dump through
|
||||
`wo_wal_commit`, which `fdatasync`s, so a dump paid one barrier per 256 records:
|
||||
|
||||
| live set | pause, per-flush fsync | pause, one final fsync |
|
||||
| --- | --- | --- |
|
||||
| ~107 KB | 23 948 µs | **2 903 µs** |
|
||||
| ~500 KB | 36 361 µs | **7 526 µs** |
|
||||
| ~1.98 MB | 107 649 µs | **13 212 µs** |
|
||||
|
||||
Marginal rate went from **~22 MB/s to ~181 MB/s** — from sync-bound to
|
||||
bandwidth-bound. Intermediate durability during a dump is worthless: the temp
|
||||
file is not authoritative until the rename and is fsynced once immediately
|
||||
before it, so those barriers bought nothing and cost 8×.
|
||||
|
||||
**The pause is O(live rows), and that is the number that eventually forces an
|
||||
incremental design.** At ~181 MB/s a 1 GB live set implies roughly 5.5 s — well
|
||||
past any interactive budget. The spec deliberately did not buy incremental
|
||||
copying in advance; this is the measurement it is to be bought against.
|
||||
|
||||
### Gating
|
||||
|
||||
`ckpt.reclaim_x` is the feature's central claim and is gated tightly (15%).
|
||||
Everything else in the leg — boot times, the pause, the byte counts — is
|
||||
wall-clock or workload-shaped on a shared box and carries a wide tolerance,
|
||||
because waiving them *all* would have left the leg ungated. The leg also
|
||||
asserts two things directly rather than trusting a metric: that some compaction
|
||||
actually ran (otherwise it proves nothing), and that the log really is smaller
|
||||
with checkpointing on.
|
||||
|
||||
One direction bug worth recording: `reclaim_x` was first recorded as
|
||||
lower-is-better by the default detector, which would have **passed "reclaimed
|
||||
nothing" and failed an improvement** — the central claim gated backwards.
|
||||
|
||||
**Gate-tolerance corrections made while closing this iteration**, both recorded
|
||||
because a widened tolerance that is not justified is indistinguishable from a
|
||||
silenced regression:
|
||||
|
||||
- **`ckpt.pause_us_max` is no longer gated against a baseline.** The raw pause
|
||||
scales with the live set, and this workload's live set is not fixed —
|
||||
`wmix`'s `hist_dump` inserts a row per latency bucket, so a noisier box makes
|
||||
more buckets, more rows, and a longer pause. What belongs to the engine is the
|
||||
**rate**, so `ckpt.pause_us_per_mb` carries the real tolerance and the raw
|
||||
pause keeps the absolute 50 ms budget as its guard.
|
||||
- **`ram.*.msgrate.msgs_sec` moved from 15% to 70%, and this one is
|
||||
pre-existing.** Across the ten full runs recorded on 2026-08-28/29 — several
|
||||
predating the checkpoint work — it ranged **10.7M to 17.9M msgs/sec, a 1.67×
|
||||
spread**. A 15% gate on a scheduling-bound throughput metric fails
|
||||
intermittently whatever the engine does.
|
||||
- **`durable.sN.*.p99us` moved from 100% to 300%**, with more evidence than the
|
||||
first widening had: mixread p99 measured 1043 / 2318 / 4147 µs and mixwrite
|
||||
1623 / 4446 µs across runs of the same build. The floors remain the real
|
||||
guard, and they are not slack — mixread's came within 25 µs of tripping.
|
||||
|
|
|
|||
|
|
@ -67,6 +67,161 @@ behind this board; live Obsidian Dataview views:
|
|||
|
||||
## ▶ NEXT PLAN
|
||||
|
||||
### Landed 2026-08-29 — databasev2 3, WAL checkpoint (the chain's last link)
|
||||
|
||||
**Implemented last time (2026-08-29):** compaction. The log used to grow forever
|
||||
— nothing removed superseded records, so boot replayed all history. It is now
|
||||
rewritten as one record per live row into a temp file and swapped in with
|
||||
`rename`. Six tasks, brainstormed and spec'd first
|
||||
([spec](../superpowers/specs/2026-08-28-wal-checkpoint-design.md) ·
|
||||
[plan](../superpowers/plans/2026-08-28-wal-checkpoint.md)).
|
||||
|
||||
**Key findings (measured, not asserted):** **2.16× space reclaimed**
|
||||
(1 962 358 → 907 094 B), **boot 114 → 64 ms**, stop-the-world pause **2 651 µs**
|
||||
against a stated 50 ms budget. Reading `.dev/reference/postgresql` was what made
|
||||
the design defensible rather than lazy: **Postgres never compacts its WAL**,
|
||||
because its records are page deltas and a compacted redo log is not a store —
|
||||
hence heap files, a control file, a redo pointer, a second recovery source and a
|
||||
separate checkpointer process. Ours are **full row images**, so a compacted log
|
||||
*is* a complete store, and all of that machinery disappears. What was worth
|
||||
porting is the ordering discipline — publish the switch atomically and last — and
|
||||
one `rename` provides it.
|
||||
|
||||
**Learned — two bugs of mine that measurement found, not review:** wiring the
|
||||
trigger only into the drain left **`WO_SHARDS=1` never compacting**, its log
|
||||
growing forever (536 KB where multi-shard held 446 KB), because a statement on
|
||||
the owner shard never enters that drain. And the dump was **8× slower than
|
||||
necessary**, flushing through the committing path and paying one `fdatasync` per
|
||||
256 records for durability that is worthless before the rename — one final
|
||||
barrier took a 2 MB dump from 107 649 µs to 13 212 µs, ~22 MB/s to ~181 MB/s.
|
||||
Separately, the crash battery's *first* version failed on correct code ~1 run in
|
||||
3: it acked deletes after committing them, so a kill in between made it demand a
|
||||
row the engine was right to remove. Deletes now announce intent first.
|
||||
|
||||
**Dependencies unblocked:** every link in the concurrency + fiber chain has now
|
||||
landed its planned work — stage 3 → 22 → 24 (absorbing 31 + 34) → 40 →
|
||||
databasev2 4 part A → databasev2 3. **Not "complete", precisely:** chain 5 stays
|
||||
`in-progress` because databasev2 4's part B was never done, and its premise was
|
||||
invalidated by part A rather than satisfied. Nothing in the chain is blocked on
|
||||
anything else in it.
|
||||
|
||||
**Next steps:** the honest queue is (1) databasev2 2's outstanding 5c/5d, whose
|
||||
`resident: keys` half is unimplemented and now carries a recorded obligation —
|
||||
compaction invalidates every WAL offset it stores, so the compactor must rebuild
|
||||
that map; (2) databasev2 4 **part B**, whose premise was invalidated by part A
|
||||
and which needs re-brainstorming rather than starting; (3) the O(live rows)
|
||||
pause, ~5.5 s at a 1 GB live set, which is the number an incremental checkpoint
|
||||
must be bought against.
|
||||
|
||||
**`.dev/reference` used:** `postgresql` — `xlog.c` (`CreateCheckPoint`, segment
|
||||
recycling), `checkpointer.c` (the time-or-volume trigger), and
|
||||
`controldata_utils.c`, which also corrected a prior exploration doc: Postgres
|
||||
updates its control file **in place with a CRC**, not by rename.
|
||||
|
||||
---
|
||||
|
||||
### Landed 2026-08-28 — databasev2 4 part A, WAL group commit
|
||||
|
||||
**Implemented last time (2026-08-28):** one durability barrier per drain
|
||||
instead of one per statement. Shard 0 stages every queued write request, holds
|
||||
each reply, commits once when its queue empties, then releases all — so a writer
|
||||
is acknowledged after the barrier that carried *its* record, which was the
|
||||
intended contract all along and was true before only because every batch had one
|
||||
member. Six tasks, brainstormed and spec'd first
|
||||
([spec](../superpowers/specs/2026-08-28-wal-group-commit-design.md) ·
|
||||
[plan](../superpowers/plans/2026-08-28-wal-group-commit.md)).
|
||||
|
||||
**Key findings (measured, not asserted):** **≈2.9× durable write throughput,
|
||||
≈2.1× lower p50** on a write-concurrent workload, confirmed a second way by the
|
||||
`s1`-vs-`sN` split within one build (1467 → 5117 ops/s, mean batch 1.0 → 5.43,
|
||||
peak 57) — 2.9× and 3.5× agreeing. Batching scales with contention: mean batch
|
||||
1.13 / 1.76 / 5.35 at C = 4 / 16 / 64. **The story's premise was wrong**: it
|
||||
said "fsync-per-commit" and the engine was fsync-per-**statement**, committing
|
||||
after every append at all six sites — so part A was closer to deleting calls
|
||||
than adding a mechanism.
|
||||
|
||||
**Learned — three things the measurement corrected, not the code:**
|
||||
(1) **`/tmp` is tmpfs here, where `fdatasync` is free.** The same run reported
|
||||
195 000 ops/s at p50 1 µs there against 2200 at 7200 µs on ext4. A group-commit
|
||||
measurement taken on a memory filesystem measures nothing; `db-bench` is right
|
||||
to keep its stores under `bench/`. (2) **No existing leg could exercise the
|
||||
feature** — `mix` writes on one op in ten with C=4, giving 20 writes and mean
|
||||
batch 1.01, so a `wmix` write-concurrent leg had to be added or the payoff was
|
||||
unevaluable either way. (3) **The before-p99 was off the instrument** —
|
||||
`hist_add` clamps at 20000 µs and both before-runs pinned there, so the gain is
|
||||
*at least* 2.3× and the true old p99 is unknown.
|
||||
|
||||
**Dependencies unblocked — and one dependency invalidated.** `WO_T_IO` is
|
||||
unreachable from a DB write: a failed stage or barrier now ends the process
|
||||
(exit 74, diagnosed), replacing three behaviours that disagreed — `insert`
|
||||
un-applied itself while `update` and `delete` returned a catchable trap and
|
||||
admitted in their own comments that they left RAM ahead of disk. **Part B's
|
||||
premise is invalidated**: it was justified by "close the 66× durable gap", but
|
||||
that gap is two problems. Concurrent fan-in was a batching problem and is now
|
||||
~3× better; a **serial** writer waiting on one barrier is a latency problem that
|
||||
batching cannot touch and io_uring does not obviously help either. Part B should
|
||||
be re-brainstormed, not started.
|
||||
|
||||
**Next steps:** either re-brainstorm part B against its corrected premise, or
|
||||
take chain 6 ([databasev2 3](databasev2/03-wal-checkpoint.md), WAL checkpoint),
|
||||
which now has the replay "before" it lacked. **(Superseded 2026-08-29: it
|
||||
landed.)** Two debts named rather than hidden:
|
||||
the abort path is not exercised (forcing a real `fdatasync` failure needs mount
|
||||
privileges), and single-shard concurrent batching needs the inline-path park —
|
||||
the same machinery part B would need.
|
||||
|
||||
**`.dev/reference` used:** none this slice. The sources were the engine's own
|
||||
code and the Linux `fsync`-failure semantics that make retrying unsound.
|
||||
|
||||
---
|
||||
|
||||
### Landed 2026-08-27 — iteration 24, chat + actor lifecycle (absorbing 31 + 34)
|
||||
|
||||
**Implemented last time (2026-08-27):** the slice closed and merged to master
|
||||
(`ed5334d`, fast-forward). T4 `monitor` + T5 `time.after` (ids 89/90) had
|
||||
landed on the branch; this session merged master in (adopting the `porch`
|
||||
rename), finished T8/T9, fixed the gate, found and fixed a runtime bug, and did
|
||||
T10. Iterations 31 and 34 land inside it.
|
||||
|
||||
**Key findings (measured, not asserted):** finishing the gate mattered more than
|
||||
finishing the sample. Making **every leg start its own server** — instead of the
|
||||
drain leg inheriting the soak's warmed one — exposed that **5 of 16**
|
||||
fresh-server SIGTERM drains left a client at EOF with no close frame and no
|
||||
diagnostic. Traced to `shard_main`: `NEXT_RUNNABLE()` already stated the
|
||||
contract ("a WORKER on stop keeps DRAINING … close frames!") but the **idle**
|
||||
branch reaped and broke, abandoning its inbox. An actor between messages is
|
||||
exactly that idle case. Split out as
|
||||
[40](language-runtime-database/40-shutdown-drain-guarantee.md); **20 of 20
|
||||
clean** after. Also measured: the fd check had been core-count dependent — lazy
|
||||
per-shard init takes one `io_uring` + one `eventfd` per shard, capped at
|
||||
`nproc`, so 26 → 44 on a 20-core box read as a leak. **1000 connections left it
|
||||
at 44**, which settled it.
|
||||
|
||||
**Learned:** three of the four gate failures were **stale build artifacts**, not
|
||||
code. A branch switch leaves `compiler/_build/` and `runtime/build/` holding the
|
||||
other branch's binaries, and a `woc` emitting `.wob` v7 against a v6 runtime
|
||||
surfaces only as "no listener" — rebuild both before believing a gate failure.
|
||||
And a gate that reuses another leg's server is not merely untidy: it hid a real
|
||||
bug, and when its own leg failed it orphaned a listener that broke the *next*
|
||||
run. Example apps now log to `/tmp/<app>.log` so a developer can `tail -F` them.
|
||||
|
||||
**Dependencies unblocked:** PUBSUB2 (WebSockets + pub/sub, rejected until this
|
||||
point) is done; the porch ledger's WebSocket rows are ✅ and its cancellation row
|
||||
is unblocked-not-built. Chain position 4 is complete, so **the chain's next link
|
||||
is [databasev2 4](databasev2/04-io-uring-commit.md)** (io_uring group-commit).
|
||||
Still blocked: CSRF and sessions — iteration 34 shipped HMAC but **there is
|
||||
still no RNG**, and HMAC authenticates a token without being able to mint one,
|
||||
which is [39](language-runtime-database/39-web-framework-parity.md)'s leading
|
||||
item.
|
||||
|
||||
**Next steps:** databasev2 4, or databasev2 2's outstanding 5c/5d. One debt is
|
||||
named rather than hidden: iteration 40's guarantee is proven only by the chat
|
||||
gate — nothing in `runtime/test/` drives `wo_engine_start`/`wo_engine_stop` and
|
||||
no corpus fixture can trigger a stop, so pinning it lower needs new
|
||||
multithreaded test infrastructure.
|
||||
|
||||
**`.dev/reference` used:** none this slice. The sources were RFC 6455, RFC
|
||||
3174/4231 for the digest vectors, and the kernel's own interfaces for the drain.
|
||||
### Landed 2026-08-27 — databasev2 1, the RAM ceiling measured
|
||||
|
||||
**Implemented last time (2026-08-27):** databasev2 1 refined (three forks
|
||||
|
|
@ -226,10 +381,16 @@ no reference project was consulted for the implementation).
|
|||
**The concurrency + fiber chain — ✅ stage 3 → ✅ 22 → 🔄 24 (absorbing
|
||||
31 + 34) → 23 → 32.** The chain's original order put 31 before 24; the
|
||||
2026-08-23 directive absorbed 31 INTO 24, and 34 resolved with it, so
|
||||
those three are one slice. **The live slice is iteration 24** — spec and
|
||||
plan approved 2026-08-23, executing on branch `chat-ws-lifecycle`, five
|
||||
of ten tasks landed. Its running state is the marker doc
|
||||
([`2026-08-23-chat-ws-lifecycle.md`](../active-slice-2026-08-23-chat-ws-lifecycle.md)),
|
||||
those three are one slice. **Iteration 24 is nine of ten tasks landed and MERGED TO MASTER
|
||||
on 2026-08-27** (fast-forward, `ed5334d`): T1 crypto, T2 bounded mailboxes,
|
||||
T3 call/reply, T4 `monitor` + T5 `time.after` (ids 89/90 — the reserved holes
|
||||
are now filled), T6 ws upgrade, T7 frame codec, T8 chat sample, T9 the chat
|
||||
gate. Verified on master: chat 11 checks 0 failures at the full 1000-client
|
||||
soak, runtime battery 36 suites 0 fail, compiler 556 checks 0 fail, corpus
|
||||
119 checks 0 fail. Only **T10 closeout** remains — which is what still holds
|
||||
stories 24/31/34 open. Finishing T9 exposed and fixed a real runtime bug,
|
||||
split out as [40](language-runtime-database/40-shutdown-drain-guarantee.md). Its running state is the marker doc
|
||||
(the marker doc, deleted at closeout per the convention),
|
||||
which is the file to read for what is done and what is next; stories
|
||||
[31](language-runtime-database/31-actor-lifecycle.md) and
|
||||
[34](language-runtime-database/34-crypto-builtins.md) keep
|
||||
|
|
@ -279,6 +440,9 @@ both still literal holes in `wob.h`'s builtin enum; then T8 the chat
|
|||
sample, T9 its gate, T10 closeout setting 24/31/34 to `status: done`) → 23
|
||||
(io_uring group-commit — target: close the 4.5k→297k durable gap) →
|
||||
32 (WAL checkpoint). Held tail resumes on its own precedence notes.
|
||||
> (**Superseded 2026-08-28:** 24 landed, and 23's part A landed with it —
|
||||
> "close the 4.5k→297k durable gap" turned out to be the wrong target; see
|
||||
> the databasev2 4 row.)
|
||||
|
||||
**`.dev/reference` used:** none this slice (the LW_SOAK discipline and
|
||||
linkcheck.py precedent came from in-repo scripts).
|
||||
|
|
@ -436,11 +600,15 @@ that sequences its tasks. Read one, approve, then the next starts.
|
|||
| 19 | [Float + Bytes](language-runtime-database/19-missing-scalar-types.md) | ✅ **landed 2026-08-20** — `.wob` v5: Float constant tag, field kinds 6/7, opcodes 34-41 (IEEE-quiet f64), builtins 70-83. Full stack: literals, arithmetic, `@table` column, WAL bit-exact replay, json fractions in / shortest-round-trip out, `?Float` reserved-NaN nil, total-order index (NaN last, `-0.0` == `+0.0`), Bytes + base64. No implicit Int/Float mixing (WO-E201); `float`/`trunc` are the only bridges. Proof: web-app price is a real Float (`{"price":9.99}`), `just web-app` 23/0; corpus 103/0 |
|
||||
| 11 | [Fibers](language-runtime-database/11-fibers.md) | ✅ **landed 2026-08-21** with the arc (`just fibers` 10/0); fs-park re-scoped out of v1, disclosed in the story |
|
||||
| 22 | [Durability, throughput, scale](language-runtime-database/22-durability-throughput-scale.md) | ✅ **landed 2026-08-21** — db-bench + baseline.json (74 metrics) + restart/kill -9 proofs both shard counts; durable 4.5k vs ram 297k inserts/s, reads O(table), msgrate 13.4M/2.45M |
|
||||
| 31 | [Actor lifecycle](language-runtime-database/31-actor-lifecycle.md) | 🔄 **absorbed into 24** (directive 2026-08-23) and half landed there: `call` request/response with a typed scalar reply (`WO_B_CALL = 88`, WO-E226), bounded mailboxes (`WO_MAILBOX`, cap 1024, catchable `WO_T_ACTOR`), and actor death that traps callers instead of hanging them. Still open: `monitor` and `time.after` — ids **89 and 90 are reserved holes** in `wob.h`, which is the machine-checkable proof of what is left. Supervision trees stay out of v1 |
|
||||
| 24 | [chat: WebSocket workload](language-runtime-database/24-chat-websocket-workload.md) | 🔄 **the live slice** (absorbing 31 + 34, directive 2026-08-23) — branch `chat-ws-lifecycle`, 5/10 tasks landed: crypto, bounded mailboxes, WS upgrade, frame codec, `call`/reply + actor death. Pending: `monitor`, `time.after`, the chat sample, its gate, closeout. State lives in [the marker](../active-slice-2026-08-23-chat-ws-lifecycle.md) |
|
||||
| 31 | [Actor lifecycle](language-runtime-database/31-actor-lifecycle.md) | ✅ **LANDED 2026-08-27 inside 24** (directive 2026-08-23). All four mechanisms: `call`/reply with a typed scalar reply (`WO_B_CALL = 88`, WO-E226), bounded mailboxes (`WO_MAILBOX`, cap 1024, catchable `WO_T_ACTOR`), actor death that traps callers instead of hanging them, **`monitor` (89)** and **`time.after` (90)** — the reserved holes in `wob.h` are filled. A fifth mechanism it did not anticipate came out of proving the gate: the shutdown drain guarantee, [40](language-runtime-database/40-shutdown-drain-guarantee.md). Supervision trees stay out of v1 |
|
||||
| 24 | [chat: WebSocket workload](language-runtime-database/24-chat-websocket-workload.md) | ✅ **LANDED 2026-08-27** (absorbing 31 + 34) — all ten tasks; merged to master `ed5334d`. `just chat` **11 checks, 0 failures** at the full 1000-client soak: handshake, functional matrix on both `WO_IO` backends and on one shard, the soak, the fd invariant, the SIGTERM drain, `WO_MAILBOX=8` backpressure, ASan clean. Finishing its gate found a real runtime bug, split out as [40](language-runtime-database/40-shutdown-drain-guarantee.md) |
|
||||
| 23 | [io_uring group-commit](databasev2/04-io-uring-commit.md) | ✅ **part A LANDED 2026-08-28 — group commit**, one barrier per drain instead of one per statement (the engine was fsync-per-STATEMENT, not per commit; the story's premise was wrong). Shard 0 holds each reply, commits once when its queue empties, releases all — so a writer is acked after the barrier carrying ITS record. **≈2.9× durable write throughput, ≈2.1× lower p50**, two measurement methods agreeing (2.9× controlled, 3.5× s1-vs-sN); mean batch 5.43, peak 57. A durability failure is now **fatal (exit 74), not a catchable `WO_T_IO`** — replacing three behaviours that disagreed, two of which admitted leaving RAM ahead of disk. **What it did NOT do:** `durable.sN.mixwrite` 480→492 (unchanged — that workload does 20 writes at C=4, mean batch 1.01) and `seed` unchanged (serial writers have nothing to batch with). **This row used to say "close the 66× gap"; that target was mis-stated** — the gap is two problems and part A fixes only the concurrent one. ⬜ part B (io_uring) **needs re-brainstorming**, not starting on the old premise |
|
||||
| 32 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) | ✅ **LANDED 2026-08-29 — the chain's last link.** Compaction rewrites the log as one record per live row and swaps it in with `rename`, so **recovery is completely unchanged** and crash safety comes from the filesystem rather than from code. **2.16× space reclaimed** (1 962 358 → 907 094 B), **boot 114 → 64 ms**, stop-the-world pause **2 651 µs** against a stated 50 ms budget. Read `.dev/reference/postgresql` for it: PG *never* compacts its WAL — its records are page deltas, so it needs heap files, a control file, a redo pointer and a separate process. Ours are full row images, so a compacted log IS a store, which deletes all of that. `kill -9` during compaction: 40 rounds/run, 10 clean runs, and **mutation-proven** — against in-place rewrite instead of `rename` the battery fails every time. Outstanding: the **`resident: keys` offset map** (compaction moves every record; the obligation is recorded at the compactor) and the O(live rows) pause, ~5.5 s at 1 GB, which is what an incremental design must be bought against |
|
||||
| 33 | [Single-file store](databasev2/07-single-file-db.md) | ⬜ off-chain, small — `WO_DATA=<path>.db` file form; driver-only (story written 2026-08-22) |
|
||||
| 34 | [Crypto builtins](language-runtime-database/34-crypto-builtins.md) | 🔄 **code landed** as 24's T1 (`d14fa9f`): `sha1`/`sha256`/`hmac_sha256`, ids 85–87 in `wob.h`, `runtime/src/crypto.c`, RFC/FIPS vectors 18/0, corpus pin. The 24 gate that once needed it is cleared. Frontmatter keeps `status: refine` only until 24's T10 closeout sets it to `done` |
|
||||
| 38 | [Content platform capabilities](language-runtime-database/38-content-platform-capabilities.md) | ⬜ off-chain, needs a spec — the two capability families no iteration owns, confirmed against `runtime/src/wob.h`: `fs` mutation verbs (six fs builtins, ids 40–45; `append` creates-if-absent, so nothing is ever replaced, truncated, deleted or renamed) and `net.connect` (ids 51–55 + 91–95, no connect, and no `connect()` anywhere in `runtime/src/` — so no OIDC/SMTP/object-store/webhook/federation). Driven by a `docs/examples/vault` content-collaboration workload, in 28's mould. New builtins from 96 (89/90 reserved for 31); no `.wob` bump (`WOB_VERSION 6u`, last moved by 36). Story written 2026-08-26 from the "can it build a Nextcloud?" ask |
|
||||
| 39 | [Web framework parity](language-runtime-database/39-web-framework-parity.md) | ⬜ off-chain, needs a spec — from [the Fiber v3.5.0 study](../plan/exploration/fiber/00-fiber-parity.md) (all 32 of its middleware read against `porch`; **nine already have a counterpart**). Leads with a **random-bytes builtin**: the framework ledger claimed CSRF/sessions were unblocked by iteration 34's HMAC, but HMAC authenticates a token and cannot mint one — there is no RNG anywhere in the runtime. Then cookies (absent both ways; `Resp.headers` being a map cannot carry two `Set-Cookie` lines), then limiter/idempotency (cheapest wins — `@table` + `time.ticks`, nothing new), sessions, CSRF, and the routing/response sugar. Streaming/SSE/compression, `@derive` binding, TTL cache, `proxy` and metrics all excluded with owners named |
|
||||
| 40 | [Shutdown drain guarantee](language-runtime-database/40-shutdown-drain-guarantee.md) | ✅ **LANDED 2026-08-27 — chain 3, with 31; split out of 24.** One rule: **a message sent before the stop flag is observed must be delivered and run before the engine stops.** Found by measurement, not review: making the chat gate's drain leg start its OWN (cold) server exposed that **5 of 16** fresh-server SIGTERM drains left a WebSocket client at EOF with no close frame and no diagnostic. Traced to `shard_main` — `NEXT_RUNNABLE()` already stated the contract ("a WORKER on stop keeps DRAINING … close frames!") but the IDLE branch reaped and broke, abandoning its inbox for teardown to free. An actor between messages is exactly that idle case, which is why a WARM soak server hid it for so long. Fix is one branch honouring the primary's drain window, yielding on an empty poll. **20 of 20 clean after**; `just chat` 11 checks 0 failures at the full 1000-client soak (which also settled the fd question: 1000 connections left the count at 44); runtime battery 36 suites 0 fail, compiler 556 checks 0 fail. Ruled out: a bigger spin (a 1 s wall-clock deadline still failed 2 of 12) and spawn-during-shutdown. Outstanding: a pin below the gate — nothing in `runtime/test/` drives the engine start/stop and no corpus fixture can trigger a stop |
|
||||
| 37 | [wo-html components](language-runtime-database/37-wo-html-components.md) | ✅ off-chain — LANDED 2026-08-25. Raw text literal (backtick, margin stripped at lex time, `{{ }}` auto-escapes) + the component layer: `Component`/`render_all`/`Layout` in wo-html, `ok_html` moved into the framework, site and shop both migrated |
|
||||
| 35 | [net runtime seams](language-runtime-database/35-net-runtime-seams.md) | ⬜ off-chain — fd deadlines on the park plane, Unix sockets, peer address; owns the ledger's three 🔧 rows (story written 2026-08-22) |
|
||||
| 25 | [HTTP service layer](../superpowers/plans/2026-08-01-http-service-layer.md) | ⏸ hold (2026-08-21) — story file removed; the plan doc remains |
|
||||
|
|
@ -461,13 +629,16 @@ that sequences its tasks. Read one, approve, then the next starts.
|
|||
| Language | 🔄 [iteration 36 — operator parity](language-runtime-database/36-operator-parity.md): `not`, bitwise `& \| ^ << >>`, hex/binary/`_` literals, compound assigns — CODE LANDED 2026-08-22 (branch operator-parity, `.wob` v6, all gates green; reference project `.dev/reference/go` drove the design). Awaiting the developer's MANUAL pass on `docs/examples/operators/` (no test fixtures by directive); unblocks story 34's pure-`.wo` HMAC question | [plan](../superpowers/plans/2026-08-22-operator-parity.md) |
|
||||
| Language | the framework v1-polish slice landed 2026-08-20 (branch framework-v1, awaiting merge); next per the order: brainstorm 20/21's forks | [order](#implementation-order-re-sequenced-2026-08-21--concurrency-chain) |
|
||||
| Runtime | ✅ **iteration 35 landed 2026-08-23** (branch `framework-v1b`, with framework v1 slice 2 + the serving slice): net deadlines/unix/peer (ids 91–95), fiber pooling, serve_conn + web-app fiber-per-connection — web-app gate 41/0, both WO_IO backends | [design](../superpowers/specs/2026-08-23-net-seams-park-design.md) |
|
||||
| Runtime | 🔄 **iteration 24 (absorbing 31 + 34): chat + actor lifecycle** — spec + plan approved 2026-08-23 (24 absorbs 31 by directive; 34 resolved C-builtins); executing on branch `chat-ws-lifecycle` | [marker](../active-slice-2026-08-23-chat-ws-lifecycle.md) · [plan](../superpowers/plans/2026-08-23-chat-ws-lifecycle.md) |
|
||||
|
||||
The active slice's marker doc is
|
||||
[`docs/active-slice-2026-08-23-chat-ws-lifecycle.md`](../active-slice-2026-08-23-chat-ws-lifecycle.md)
|
||||
— one file, deleted when the slice lands. Everything else pending is the
|
||||
concurrency chain (see *Pending* below); the held tail is every story
|
||||
whose frontmatter reads `status: hold`.
|
||||
**No slice is active.** Iteration 24 landed 2026-08-27 and its marker doc was
|
||||
deleted per the convention. Everything pending is the concurrency chain (see
|
||||
*Pending* below) — **the chain's next link is
|
||||
[databasev2 4](databasev2/04-io-uring-commit.md)** (chain 5, the io_uring
|
||||
group-commit write path, `was_language_iteration: 23`), which now has iteration
|
||||
22's fsync-per-commit numbers in hand, plus databasev2 1's finding that the
|
||||
write path is *not* where memory pressure bites (appending under a cap costs
|
||||
~1%, random reads 273×). The held tail is every story whose frontmatter reads
|
||||
`status: hold`.
|
||||
|
||||
### Landed 2026-08-14 — the compile-and-run milestone
|
||||
|
||||
|
|
@ -705,8 +876,8 @@ the language arc as v1 history.
|
|||
| --- | --- | --- |
|
||||
| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | ✅ **MEASURED 2026-08-27** — `readiness: ready`, `status: done`; forks settled, harness landed (**148 checks**). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Also measured: **random reads over an oversized table collapse 273×** (1.85M vs 6 771 reads/s, p99 1 µs vs 487 µs) — so the two access patterns sit ~270× apart under the same pressure, and departure is a **step, not a curve**. Replay measured too: **≈5.5 µs/record, 1.9× history penalty** (10M records ≈ 55 s of boot) — iteration 3's missing "before", now gated. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from |
|
||||
| 2 | [per-table storage: `durable` and `resident`](databasev2/02-table-storage-modes.md) | 🔄 **the language enrichment — the `durable` half is DONE and usable.** Two optional `@table` keys, `durable: true\|false` and `resident: all\|keys`, both defaulting to today's behaviour (all 28 existing declarations compile unchanged, no golden moved). Landed: the grammar, WO-E224 (a durable `ref` into a volatile table is refused), `.wob` v7 carrying both properties in spare `flags` bits, `durable: false` actually skipping the WAL (measured: 50 inserts → 1500 bytes durable, **0** volatile) with a mode-mismatch startup refusal, plus offset capture and read-a-row-from-an-offset. Outstanding: 5c/5d (the id→offset map and rewiring `wo_row_ptr`'s 11 call sites, slab scans and `@unique`/FK across the boundary — not yet written up), the two runtime refusals, and closeout. [spec](../superpowers/specs/2026-08-26-table-residency-design.md) · [plan](../superpowers/plans/2026-08-26-table-residency.md) |
|
||||
| 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ⬜ snapshot + truncate: disk reclaimed, replay bounded |
|
||||
| 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ⬜ **`readiness: ready` — the one startable iteration in the repo** (four forks confirmed settled 2026-08-20). Close the 66× gap iteration 22 measured (durable 4.5k vs ram 297k inserts/s) |
|
||||
| 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ✅ **LANDED 2026-08-29 — the chain's last link.** Compaction rewrites the log as one record per live row and swaps it in with `rename`, so **recovery is completely unchanged** and crash safety comes from the filesystem rather than from code. **2.16× space reclaimed** (1 962 358 → 907 094 B), **boot 114 → 64 ms**, stop-the-world pause **2 651 µs** against a stated 50 ms budget. Read `.dev/reference/postgresql` for it: PG *never* compacts its WAL — its records are page deltas, so it needs heap files, a control file, a redo pointer and a separate process. Ours are full row images, so a compacted log IS a store, which deletes all of that. `kill -9` during compaction: 40 rounds/run, 10 clean runs, and **mutation-proven** — against in-place rewrite instead of `rename` the battery fails every time. Outstanding: the **`resident: keys` offset map** (compaction moves every record; the obligation is recorded at the compactor) and the O(live rows) pause, ~5.5 s at 1 GB, which is what an incremental design must be bought against |
|
||||
| 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ✅ **part A LANDED 2026-08-28 — group commit**, one barrier per drain instead of one per statement (the engine was fsync-per-STATEMENT, not per commit; the story's premise was wrong). Shard 0 holds each reply, commits once when its queue empties, releases all — so a writer is acked after the barrier carrying ITS record. **≈2.9× durable write throughput, ≈2.1× lower p50**, two measurement methods agreeing (2.9× controlled, 3.5× s1-vs-sN); mean batch 5.43, peak 57. A durability failure is now **fatal (exit 74), not a catchable `WO_T_IO`** — replacing three behaviours that disagreed, two of which admitted leaving RAM ahead of disk. **What it did NOT do:** `durable.sN.mixwrite` 480→492 (unchanged — that workload does 20 writes at C=4, mean batch 1.01) and `seed` unchanged (serial writers have nothing to batch with). **This row used to say "close the 66× gap"; that target was mis-stated** — the gap is two problems and part A fixes only the concurrent one. ⬜ part B (io_uring) **needs re-brainstorming**, not starting on the old premise |
|
||||
| 5 | [Bounded tables and eviction](databasev2/05-bounded-tables-eviction.md) | ⬜ a declared capacity + refuse/evict/back-pressure, and a process-level pressure signal that sheds **before** the allocator or OS gets involved — turning the invisible failure into a managed one |
|
||||
| 6 | [Cold tiering](databasev2/06-cold-tiering.md) | ⚠ **largely superseded by 2** — `resident: keys` took the ceiling-raising role; its user-space-working-set premise was rejected for the kernel page cache. Mostly forks: which shape, whether the index itself fits, whether the *language* surfaces the fault cost, and whether `@unique` on a cold table is refused outright. A paged B-tree stays rejected — if tiering needs one, reject tiering |
|
||||
| 7 | [Single-file store](databasev2/07-single-file-db.md) *(was 33)* | ⬜ `WO_DATA=<path>.db`; driver-only, independent |
|
||||
|
|
|
|||
|
|
@ -2,8 +2,8 @@
|
|||
track: databasev2
|
||||
iteration: "3"
|
||||
was_language_iteration: "32"
|
||||
status: pending
|
||||
readiness: refine
|
||||
status: done
|
||||
readiness: ready
|
||||
chain: 6
|
||||
---
|
||||
|
||||
|
|
@ -29,6 +29,113 @@ chain: 6
|
|||
> replay/restart numbers to justify its policy and must compose with
|
||||
> 23's group-commit write path.
|
||||
|
||||
> **BRAINSTORMED 2026-08-28.** Spec:
|
||||
> [`2026-08-28-wal-checkpoint-design.md`](../../superpowers/specs/2026-08-28-wal-checkpoint-design.md)
|
||||
> · plan: [`2026-08-28-wal-checkpoint.md`](../../superpowers/plans/2026-08-28-wal-checkpoint.md)
|
||||
> (6 tasks).
|
||||
> Read `.dev/reference/postgresql` for this — and the conclusion was that
|
||||
> Postgres' design is *unavailable* to us, which is what makes the simpler one
|
||||
> legitimate.
|
||||
>
|
||||
> **The design in one sentence:** compact the log by rewriting it as one record
|
||||
> per live row into a temp file, then `rename` it over the live WAL. Recovery is
|
||||
> **completely unchanged** — boot still opens one file and replays it — and the
|
||||
> crash criterion is satisfied by the filesystem rather than by code we must get
|
||||
> right.
|
||||
>
|
||||
> **Why one file works here and not in Postgres.** Postgres never compacts its
|
||||
> WAL: its records are page deltas, so a compacted redo log is not a store, and
|
||||
> it must keep heap files, a control file, a redo pointer and a second recovery
|
||||
> source. Ours are **full row images** — `apply_record` implements UPDATE as
|
||||
> remove-then-recreate — so a compacted log *is* a complete store. That one
|
||||
> difference deletes the control file, the redo pointer, the cutoff offset and
|
||||
> the separate process from the design.
|
||||
>
|
||||
> **Forks settled:** no snapshot format (the compacted log is the snapshot); one
|
||||
> source, not two; **volume-only trigger** as a ratio against the last
|
||||
> compaction's own measured output, with an absolute floor — **no timer**,
|
||||
> because Postgres' timer exists to bound loss from unflushed buffers and we have
|
||||
> none; stop-the-world, with the pause measured against a stated budget rather
|
||||
> than assumed acceptable.
|
||||
>
|
||||
> **The coupling that would otherwise be found late:** compaction moves every
|
||||
> record, so it **invalidates every WAL offset**
|
||||
> [iteration 2](02-table-storage-modes.md)'s `resident: keys` stores. The
|
||||
> compactor rebuilds the offset map as it writes. Recorded now because iteration
|
||||
> 2's storage half is unimplemented, so nothing breaks today — it would break
|
||||
> later, looking like corruption rather than a design gap.
|
||||
>
|
||||
> **Measured on master 2026-08-28, grounding the whole iteration:** `seed 20000`
|
||||
> leaves a 986 614-byte log; 20 000 updates take it to **2 590 262 bytes with the
|
||||
> same live rows** (2.6× history for no data), and boot+verify on that store is
|
||||
> **155 ms**.
|
||||
|
||||
## Progress — landed 2026-08-29
|
||||
|
||||
| # | Task | State |
|
||||
| --- | --- | --- |
|
||||
| 1 | `wo_wal_compact` — rewrite, fsync, rename, fsync parent, reopen | ✅ `8ea510d` |
|
||||
| 2 | a stale compaction temp is removed at open | ✅ `8bfbd4b` |
|
||||
| 3 | the trigger (pure decision + env knobs) and the ordering guard | ✅ `6dbcb9a` |
|
||||
| 4 | `kill -9` DURING compaction — 40 rounds, mutation-proven | ✅ `9b283d5` |
|
||||
| 5 | measure space, boot and the stop-the-world pause | ✅ `d87f65a` |
|
||||
| 6 | closeout | ✅ this change |
|
||||
|
||||
### Measured
|
||||
|
||||
| | checkpointing off | checkpointing on |
|
||||
| --- | --- | --- |
|
||||
| WAL used | 1 962 358 B | **907 094 B** |
|
||||
| boot | 114 ms | **64 ms** |
|
||||
|
||||
**2.16× space reclaimed, 1.78× faster boot**, stop-the-world pause **2 651 µs**
|
||||
against a stated 50 ms budget. Full details, including the pause's scaling, are
|
||||
in [`perf-targets.md`](../../plan/perf-targets.md) §7.
|
||||
|
||||
### Two bugs the work found, both mine
|
||||
|
||||
**Wiring only the drain left `WO_SHARDS=1` never compacting** — its log grew
|
||||
forever (536 KB where the multi-shard run held 446 KB), because a statement on
|
||||
the owner shard never enters that drain. Both write paths now check.
|
||||
|
||||
**The dump was 8× slower than it needed to be**, flushing through the
|
||||
committing path and so paying one `fdatasync` per 256 records for durability
|
||||
that is worthless before the rename. One final barrier took the pause from
|
||||
107 649 µs to 13 212 µs on a 2 MB live set — ~22 MB/s to ~181 MB/s.
|
||||
|
||||
## Acceptance Criteria
|
||||
|
||||
Met:
|
||||
|
||||
- **Given** an aged store, **when** it is compacted, **then** disk is reclaimed.
|
||||
✅ 2.16× on the full campaign, asserted rather than merely recorded — the leg
|
||||
fails if the log is not smaller with checkpointing on.
|
||||
- **Given** the same store, **when** it boots, **then** replay is bounded by the
|
||||
live set rather than by history. ✅ 114 → 64 ms.
|
||||
- **Given** `kill -9` at ANY instant during a checkpoint, **when** the process
|
||||
restarts, **then** recovery produces the same consistent store as if the
|
||||
checkpoint had never started, with no acknowledged write lost. ✅ 40 rounds
|
||||
per run, 10 consecutive clean runs, and **proven to have teeth**: against the
|
||||
design's rejected alternative (in-place rewrite instead of `rename`) the
|
||||
battery fails every run with the log destroyed.
|
||||
- **Given** the iteration-22 replay numbers, **then** a before/after delta is
|
||||
recorded. ✅ `perf-targets.md` §7.
|
||||
- **Given** writes arriving while a checkpoint runs, **then** the ack contract
|
||||
holds. ✅ compaction runs only where nothing is staged, asserted by a test
|
||||
that stages and requires refusal; `wo_wal_compact` also refuses as a backstop.
|
||||
|
||||
Outstanding:
|
||||
|
||||
- **The `resident: keys` offset map.** Compaction moves every record, so it
|
||||
invalidates every WAL offset [iteration 2](02-table-storage-modes.md) stores.
|
||||
The compactor must rebuild that map as it writes. **Nothing fails today**
|
||||
because iteration 2's storage half is unimplemented — which is exactly why the
|
||||
obligation is written at the compactor in `wal.c`, where the next implementer
|
||||
hits it, rather than only in a spec they may not read.
|
||||
- **The pause is O(live rows).** At ~181 MB/s a 1 GB live set implies ~5.5 s,
|
||||
past any interactive budget. Incremental or forked copying was deliberately
|
||||
not bought in advance; this is the number to buy it against.
|
||||
|
||||
## Goals
|
||||
|
||||
- **Disk space is reclaimed.** A checkpoint writes the live store as a
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
track: databasev2
|
||||
iteration: "4"
|
||||
was_language_iteration: "23"
|
||||
status: pending
|
||||
status: in-progress
|
||||
readiness: ready
|
||||
chain: 5
|
||||
---
|
||||
|
|
@ -59,6 +59,111 @@ chain: 5
|
|||
> batch — under io_uring it becomes exactly one submission, so the two
|
||||
> features compose without either knowing the other.
|
||||
|
||||
> **BRAINSTORMED 2026-08-28 — and SPLIT IN TWO.** Spec for part A:
|
||||
> [`2026-08-28-wal-group-commit-design.md`](../../superpowers/specs/2026-08-28-wal-group-commit-design.md)
|
||||
> · plan: [`2026-08-28-wal-group-commit.md`](../../superpowers/plans/2026-08-28-wal-group-commit.md)
|
||||
> (6 tasks).
|
||||
>
|
||||
> **The payoff metric is `durable.sN.mixwrite`, not the s1 numbers.** Worker
|
||||
> shards hold no WAL — the runtime asserts it — so every statement on a worker
|
||||
> marshals to shard 0 and parks, while a statement already on shard 0 runs
|
||||
> inline. Batches form only where there is a queue, so concurrent multi-shard
|
||||
> writes batch and a single-shard or serial workload does not. The baseline
|
||||
> shows why that is the right target anyway: **multi-shard concurrent writes are
|
||||
> 480 ops/s at p99 5888 µs against single-shard's 1023 at p99 664 — adding
|
||||
> shards makes durable writing WORSE today**, because every marshaled statement
|
||||
> still buys its own barrier on the owner.
|
||||
>
|
||||
> **The premise below needed correcting.** This story says "replace
|
||||
> fsync-per-commit with io_uring group-commit", but the engine does not commit
|
||||
> per commit — it commits per **statement**: `db.c` calls `wo_wal_commit`
|
||||
> immediately after every append, at all six sites, so every row change is one
|
||||
> `pwrite` plus one `fdatasync`. That splits the goal into two independent
|
||||
> wins, and only the second needs io_uring:
|
||||
>
|
||||
> - **Part A — batching.** Let many statements share one barrier. The staging
|
||||
> buffer already holds any number of records; today it never holds more than
|
||||
> one because the caller commits immediately. Mostly a deletion of calls.
|
||||
> - **Part B — async submission.** The shard submits and keeps working instead
|
||||
> of blocking in `fdatasync`. Deferred until A's measurement says whether the
|
||||
> blocking boundary is still the bottleneck.
|
||||
>
|
||||
> **A is where most of the number lives.** Iteration 22 measured durable writes
|
||||
> at 4460 ops/s and mixed writes at 1023 ops/s (p99 664 µs) against 1.28M ops/s
|
||||
> for durable reads — ~290× apart, essentially all of it the per-statement
|
||||
> barrier.
|
||||
>
|
||||
> **Forks settled in the brainstorm:** batch boundary is **queue-drain** (not
|
||||
> the tick this story recorded — a tick taxes an idle system to serve a busy
|
||||
> one); a failure between "RAM mutated" and "record durable" is a **fatal,
|
||||
> diagnosed abort**, replacing today's uneven rollback where `insert` undoes
|
||||
> itself and `update`/`delete` admit in a comment that they leave RAM ahead of
|
||||
> disk. **That removes `WO_T_IO` from the write path** — a language-visible
|
||||
> change, recorded here deliberately.
|
||||
>
|
||||
> `status: in-progress` because the brainstorm is done and the spec is
|
||||
> approved; the plan is next. (The `readiness` axis that would say this
|
||||
> precisely lives on the unmerged `db-residency-doctrine`.)
|
||||
|
||||
## Progress — part A landed 2026-08-28
|
||||
|
||||
| # | Task | State |
|
||||
| --- | --- | --- |
|
||||
| 1 | a failed barrier is detected, and fatal | ✅ `d3ff03e` |
|
||||
| 2 | one barrier per drain; replies held | ✅ `b9b8a45` |
|
||||
| 3 | the inline path takes the fatal rule, asymmetry documented | ✅ `a6ccdbe` |
|
||||
| 4 | prove batches form — the `wmix` write-concurrent leg | ✅ `40d029c` |
|
||||
| 5 | measure the payoff, gate it, record it | ✅ `d52ea8a` |
|
||||
| 6 | closeout | ✅ this change |
|
||||
| — | **part B — io_uring submission** | ⬜ **not started; its premise changed, see below** |
|
||||
|
||||
### The payoff, measured two ways
|
||||
|
||||
| Measurement | Before | After |
|
||||
| --- | --- | --- |
|
||||
| controlled (same build, only `db.c`/`vm.c` swapped; `wmix 4000 32`) | 2213 · 2177 ops/s, p50 7183 · 7251 µs | **6216 · 6525 ops/s, p50 3458 · 3444 µs** |
|
||||
| committed baseline: `s1` inline vs `sN` batched | 1467 ops/s, mean batch 1.0 | **5117 ops/s, mean batch 5.43, peak 57** |
|
||||
|
||||
**≈2.9× throughput, ≈2.1× lower p50**, and the two methods agree (2.9× and
|
||||
3.5×). Batching scales with contention: mean batch **1.13 / 1.76 / 5.35** at
|
||||
C = 4 / 16 / 64.
|
||||
|
||||
### The cost side, and a bug the battery caught
|
||||
|
||||
**Reads were being held behind the barrier.** The drain first held *every* DB
|
||||
reply until the commit — including reads, which stage nothing. `mixread` p99 rose
|
||||
from ~1043 µs to **4057 µs** until only staging statements had their replies
|
||||
held. Caught by the gate, not by review.
|
||||
|
||||
**What remains is inherent:** a barrier blocks the owner shard longer (more
|
||||
records per fsync) though less often, so anything queued behind one waits. Three
|
||||
full runs of the same build gave `durable.sN.mixread.p99` of **1043 / 2318 /
|
||||
4147 µs** — a 2–4× spread near idle. So part A buys ~3× write throughput at the
|
||||
cost of a longer, noisier tail on the owner shard. `durable.sN.*.p99us` was
|
||||
re-baselined at 100% tolerance for that reason, with the floor as the real guard
|
||||
(`mixread`'s came within 25 µs of tripping).
|
||||
|
||||
**This is the strongest argument for part B** — submitting the barrier and
|
||||
continuing to serve is exactly what removes this cost.
|
||||
|
||||
### What did NOT improve — and it was predicted
|
||||
|
||||
- **`durable.sN.mixwrite`: 480 → 492 ops/s, i.e. unchanged.** This was the
|
||||
spec's *original* payoff metric, and correcting it was part of the brainstorm:
|
||||
`mix` writes on one op in ten with C=4, so a quick run performs **20 writes**
|
||||
and measured mean batch **1.01**. A workload that never has two writes in
|
||||
flight cannot be helped by batching them.
|
||||
- **`durable.*.seed`: unchanged.** A serial single writer has nothing to batch
|
||||
with, under any scheme.
|
||||
- **This board's stated target was mis-stated.** It read "close the 66× gap
|
||||
iteration 22 measured (durable 4.5k vs ram 297k inserts/s)". Part A does not
|
||||
close that gap and structurally cannot: `seed` is serial, and one writer
|
||||
waiting on one barrier is a **latency** problem, not a batching one. Recorded
|
||||
rather than quietly renumbered.
|
||||
- **The before-p99 is not a measurement.** `hist_add` clamps at 20000 µs and
|
||||
both before-runs pinned exactly there, so the true value is ≥20 ms and
|
||||
unknown. The gain is *at least* 2.3×.
|
||||
|
||||
## Goals
|
||||
|
||||
- **Replace fsync-per-commit with io_uring group-commit** on the WAL write
|
||||
|
|
@ -77,26 +182,50 @@ chain: 5
|
|||
|
||||
## Acceptance Criteria
|
||||
|
||||
- What to achieve?
|
||||
- **Given** the io_uring write path under the iteration-22 crash battery
|
||||
(concurrent writers, kill -9 mid-stream, reboot, replay),
|
||||
- **when** it runs,
|
||||
- **then** every acknowledged write is present after replay and no
|
||||
unacknowledged partial write is ever visible — the exact result the
|
||||
fsync path gives, so durability is provably unchanged.
|
||||
- What to achieve?
|
||||
- **Given** the iteration-22 durable write benchmark,
|
||||
- **when** it is run on the fsync-per-commit path and then the io_uring
|
||||
group-commit path on the same machine,
|
||||
- **then** the io_uring path's write throughput is materially higher and
|
||||
its p99 commit latency lower, with the before/after numbers recorded —
|
||||
the payoff, measured, not asserted.
|
||||
- What to achieve?
|
||||
- **Given** a kernel without io_uring (old, or restricted by seccomp),
|
||||
- **when** the runtime starts,
|
||||
- **then** it falls back to the pwrite + fdatasync path automatically and
|
||||
correctly — io_uring is an accelerator, never a hard dependency, and a
|
||||
binary that runs everywhere is the whole project's premise.
|
||||
Met:
|
||||
|
||||
- **Given** the io_uring write path under iteration 22's crash battery, **when**
|
||||
it runs, **then** every acknowledged write is present after replay. ✅ — the
|
||||
criterion applies unchanged to part A's batching. `crash.sN` (the batched
|
||||
path) recovered every acked row after `kill -9`, `crash.s1` likewise, and both
|
||||
restart legs replay byte-true. This was the one thing batching could break.
|
||||
- **Given** the durable write benchmark before and after, **then** throughput is
|
||||
materially higher and p99 lower, recorded. ✅ ~2.9× and ~2.1× (p50); see
|
||||
`perf-targets.md` §6. **Scoped honestly:** on a write-concurrent workload
|
||||
only, and p99's "before" is at the histogram ceiling.
|
||||
- **Given** batching, **when** it runs, **then** it is proven to engage rather
|
||||
than assumed. ✅ mean batch 5.43, peak 57 on the gated leg, and the live
|
||||
assertion fails the suite if the mean drops to 1.
|
||||
- **Given** a durability failure, **when** it happens, **then** the engine does
|
||||
not continue with RAM ahead of disk. ✅ fatal, diagnosed, exit 74 — replacing
|
||||
three behaviours that disagreed.
|
||||
|
||||
Outstanding:
|
||||
|
||||
- **Given** a kernel without io_uring, **when** the runtime starts, **then** it
|
||||
falls back automatically. *(part B — part A adds no syscall interface, so
|
||||
nothing to fall back from yet.)*
|
||||
- **Single-shard concurrent batching.** A statement on shard 0 commits inline
|
||||
and cannot batch; doing so needs the inline path to park its fiber on the
|
||||
barrier — the same machinery part B needs. So `WO_SHARDS=1` gets no batching
|
||||
at all, by design and measured (mean batch 1.0).
|
||||
- **The abort path is not exercised.** Forcing a real `fdatasync` failure needs a
|
||||
full or read-only filesystem, which the gate cannot arrange without mount
|
||||
privileges. The unit test proves the error is *detected*; the exit three lines
|
||||
later is covered by inspection. Disclosed rather than papered over — iteration
|
||||
40 was exactly a fatal path nothing exercised.
|
||||
|
||||
## Part B — its premise changed
|
||||
|
||||
Part B was justified by "close the 66× durable gap". Part A shows that framing
|
||||
was wrong: the gap is **two** problems. Concurrent write fan-in was a batching
|
||||
problem and is now ~3× better. What remains is a **serial** writer waiting on a
|
||||
single barrier, which no amount of batching can help — and io_uring does not
|
||||
obviously help it either, since one writer still needs one durable barrier
|
||||
before its ack. Part B's real candidates are overlapping the barrier with other
|
||||
work on the shard, and the inline-path park that single-shard batching also
|
||||
needs. **It should be re-brainstormed against that, not started on the old
|
||||
premise.**
|
||||
|
||||
## Out Of Scope
|
||||
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
---
|
||||
iteration: "24"
|
||||
status: in-progress
|
||||
status: done
|
||||
readiness: ready
|
||||
chain: 4
|
||||
---
|
||||
|
|
@ -20,6 +20,34 @@ chain: 4
|
|||
> bounded mailboxes, actor death, timers). Iteration 19 LANDED
|
||||
> 2026-08-20, so Bytes is available for frame parse/serialize.
|
||||
|
||||
> **✅ LANDED 2026-08-27** (branch `chat-ws-lifecycle`, merged to master
|
||||
> `ed5334d`). Ten tasks: crypto (T1), bounded mailboxes (T2), `call`/reply and
|
||||
> actor death (T3), `monitor` (T4), `time.after` (T5), the WS upgrade seam
|
||||
> (T6), the pure-`.wo` frame codec (T7), the chat sample (T8), the gate (T9),
|
||||
> this closeout (T10). It absorbed [31](31-actor-lifecycle.md) and
|
||||
> [34](34-crypto-builtins.md), which land with it.
|
||||
>
|
||||
> **Gate — `just chat`, 11 checks, 0 failures** at the full 1000-client soak:
|
||||
> handshake with an independently recomputed accept-key, the functional matrix
|
||||
> (presence, broadcast, room isolation, leave) on **both** `WO_IO` backends and
|
||||
> on a single shard, the 1k hot-room soak, the fd invariant, the SIGTERM drain,
|
||||
> `WO_MAILBOX=8` backpressure, and an ASan run with zero leaks. Battery
|
||||
> alongside: runtime 36 suites 0 fail, compiler 556 checks, corpus 119 checks.
|
||||
> The sample logs to `/tmp/chat.log`.
|
||||
>
|
||||
> **Two disclosed deviations from the spec.** `monitor` takes **three**
|
||||
> arguments (`watched, observer, msg`) rather than two, because the caller may
|
||||
> be `main`, which has no mailbox and cannot be an implicit observer. And a
|
||||
> `call` reply is a **typed scalar** in v1 — which is what let the agreement be
|
||||
> checked at compile time (WO-E226) instead of carried as a tagged value.
|
||||
>
|
||||
> **What finishing the gate found.** Making every leg start its own server
|
||||
> exposed a real runtime bug the warmed soak server had been hiding: on a fresh
|
||||
> server, 5 of 16 SIGTERM drains left a client at EOF with no close frame. It
|
||||
> was not this sample's fault — the fix is an engine guarantee, split out as
|
||||
> [40](40-shutdown-drain-guarantee.md). Design notes:
|
||||
> [`docs/examples/chat/CODE-LOGIC.md`](../../examples/chat/CODE-LOGIC.md).
|
||||
|
||||
## Why this iteration exists
|
||||
|
||||
Everything the framework ledger parks behind concurrency — WebSockets,
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
---
|
||||
iteration: "31"
|
||||
status: in-progress
|
||||
status: done
|
||||
readiness: ready
|
||||
chain: 3
|
||||
---
|
||||
|
|
@ -17,6 +17,22 @@ chain: 3
|
|||
> ([iteration 24](24-chat-websocket-workload.md)) cannot be written
|
||||
> honestly without these four mechanisms.
|
||||
|
||||
> **✅ LANDED 2026-08-27 — INSIDE [24](24-chat-websocket-workload.md)**, per
|
||||
> the 2026-08-23 directive that absorbed it. All four mechanisms shipped:
|
||||
> `call`/reply with a typed scalar reply (id 88, WO-E226), **bounded mailboxes**
|
||||
> (`WO_MAILBOX`, default 1024, fail-fast with a catchable `WO_T_ACTOR`),
|
||||
> **actor death** that traps callers instead of hanging them, `monitor`
|
||||
> (id 89) and `time.after` (id 90). Ids 89 and 90 were reserved holes in
|
||||
> `wob.h`; they are filled.
|
||||
>
|
||||
> **A fifth mechanism was added that this story did not anticipate**: the
|
||||
> shutdown drain guarantee, [40](40-shutdown-drain-guarantee.md). It is
|
||||
> lifecycle semantics — this story gave actors a death notice, 40 gives the
|
||||
> program a shutdown that does not lose mail — and it was found by measurement
|
||||
> while proving 24's gate, not by review.
|
||||
>
|
||||
> How each piece works: `runtime/src/CODE-LOGIC.md`, "Actor lifecycle".
|
||||
|
||||
## Why this iteration exists
|
||||
|
||||
The arc's stages 1+2 shipped `spawn`/`send` mechanism without lifecycle:
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
---
|
||||
iteration: "34"
|
||||
status: in-progress
|
||||
status: done
|
||||
readiness: ready
|
||||
---
|
||||
|
||||
|
|
@ -19,6 +19,18 @@ readiness: ready
|
|||
> Off the concurrency chain but **gates chain position 4**: iteration
|
||||
> 24's WebSocket handshake needs SHA-1 before chat can land.
|
||||
|
||||
> **✅ LANDED 2026-08-27 — inside [24](24-chat-websocket-workload.md)** as its
|
||||
> task 1. The fork resolved to **C builtins**: `sha1` (85), `sha256` (86),
|
||||
> `hmac_sha256` (87), each over one buffer returning a fresh `Bytes`. Pinned to
|
||||
> the published vectors — RFC 3174, the SHA-256 vectors, RFC 4231 — in
|
||||
> `runtime/test/test_crypto.c`, 18 checks, plus a corpus fixture hashing "abc"
|
||||
> from `.wo`. This unblocked chain position 4: the WebSocket handshake needs
|
||||
> SHA-1, and `just chat` verifies the accept-key independently.
|
||||
>
|
||||
> **The gap it did NOT close:** there is still no RNG in the runtime. HMAC
|
||||
> authenticates a token and cannot mint one, so CSRF and sessions stay blocked
|
||||
> — which is why [39](39-web-framework-parity.md) leads with a random-bytes
|
||||
> builtin rather than treating them as unblocked.
|
||||
## Why this iteration exists
|
||||
|
||||
Four consumers already wait on it, none able to proceed:
|
||||
|
|
|
|||
|
|
@ -0,0 +1,158 @@
|
|||
---
|
||||
iteration: "40"
|
||||
status: done
|
||||
chain: 3
|
||||
---
|
||||
|
||||
# iteration 40 — the shutdown drain guarantee: a send before the stop flag is delivered
|
||||
|
||||
> Part of [Story — one language, one runtime, one database, one binary](00-story.md).
|
||||
>
|
||||
> **Split out of [24](24-chat-websocket-workload.md) on 2026-08-27** because it
|
||||
> is a runtime *semantic*, not a task in a sample's gate. It belongs to the
|
||||
> actor lifecycle ([31](31-actor-lifecycle.md), absorbed into 24) and it is
|
||||
> the half of "lifecycle" that nothing had stated: 31 gave actors a death
|
||||
> notice, this gives the program a shutdown that does not lose mail.
|
||||
>
|
||||
> **Found by measurement, not review.** The chat gate's drain leg had been
|
||||
> passing only because it drained a server the 1k soak had already warmed.
|
||||
> Making every leg start its own server exposed it:
|
||||
> [`2026-08-27-chat-drain-finding.md`](../../2026-08-27-chat-drain-finding.md).
|
||||
|
||||
## The rule
|
||||
|
||||
**A message sent before the stop flag is observed must be delivered and run
|
||||
before the engine stops.** One sentence, and it is the whole iteration. It is a
|
||||
guarantee, not a tuning parameter — which is why a spin count could never
|
||||
express it.
|
||||
|
||||
What it does *not* promise: that a message sent *after* the flag is delivered,
|
||||
that a parked fiber is resumed, or that an actor gets unbounded time. The drain
|
||||
window is the primary's, and it closes when the primary returns.
|
||||
|
||||
## The bug, as measured
|
||||
|
||||
Fresh server, two WebSocket clients, `SIGTERM`, both must receive a close frame:
|
||||
|
||||
| Sample | Result |
|
||||
| --- | --- |
|
||||
| 5 fresh servers | 1 failure (`eof\|close`) |
|
||||
| 12 fresh servers | 3 failures, one `eof\|eof` |
|
||||
| 16 fresh servers | 5 failures |
|
||||
|
||||
The failing client's socket reaches EOF with **no close frame and no
|
||||
diagnostic** — the process exits and the kernel closes the fd.
|
||||
|
||||
Traced with instrumentation on the sample's actors: `main` → Registry → Room →
|
||||
Writer. The Registry runs and sees its room. The **Room never processes the
|
||||
shutdown message**, so the Writer's close branch never runs. Clients that did
|
||||
get a frame were saved by their own Reader noticing `env.stopping()`, not by the
|
||||
room broadcast.
|
||||
|
||||
## The design, as built
|
||||
|
||||
`runtime/src/vm.c` already encoded the correct contract in `NEXT_RUNNABLE()`:
|
||||
a worker that takes a stop while it has a live fiber returns 2 and **keeps
|
||||
draining its inbox** until the primary sets `eng_shutdown`. Its comment says so
|
||||
in as many words — "queued shutdown messages (close frames!) still run".
|
||||
|
||||
`shard_main`'s own idle branch contradicted it. A worker with an empty run queue
|
||||
waits in `wo_io_wait`, and on `WO_IO_STOP` it called `fib_reap_all` and
|
||||
**broke** — abandoning whatever was still in its inbox, which `wo_engine_stop`
|
||||
then freed wholesale during teardown.
|
||||
|
||||
So the failure needed a shard that was *idle* at `SIGTERM`. A Room actor between
|
||||
messages is exactly that, which is why the warm soak server hid it: warm shards
|
||||
had live fibers and took the correct path.
|
||||
|
||||
The fix makes the idle branch obey the same contract: while the primary's drain
|
||||
window is open, an idle worker adopts its inbox and runs what arrives, yielding
|
||||
between empty polls so a drain cannot become a hot spin across every core. Only
|
||||
`eng_shutdown` — set by the primary after `main` returns — ends it.
|
||||
|
||||
One branch, in one place, matching a contract the file already stated.
|
||||
|
||||
## Progress
|
||||
|
||||
| Piece | State |
|
||||
| --- | --- |
|
||||
| the idle-worker drain branch in `shard_main` (`runtime/src/vm.c`) | ✅ one branch, matching the contract `NEXT_RUNNABLE()` already stated |
|
||||
| `sched_yield` on an empty poll so the drain cannot hot-spin | ✅ |
|
||||
| fresh-server drain, repeated | ✅ **20 of 20**, from 5-in-16 failing |
|
||||
| chat gate at the default 1k soak | ✅ **11 checks, 0 failures** — 1000/1000 clients, both `WO_IO` backends, ASan clean |
|
||||
| full runtime battery (this touches every actor program's shard loop) | ✅ **36 suites** (18 × both dispatch flavors), 0 fail, `cli_smoke: OK`; compiler 556 checks 0 fail |
|
||||
| the regression pin | ✅ the chat gate's drain leg, now that it starts its OWN (cold) server — that decoupling is what caught this. **Not** a corpus fixture or unit test: nothing in `runtime/test/` drives `wo_engine_start`/`wo_engine_stop` today, and no corpus fixture can trigger a stop, so pinning it below the gate means new multithreaded test infrastructure — named as its own cost, not smuggled in here |
|
||||
|
||||
**Measured 2026-08-27.** Before: 5 of 16 fresh-server drains left a client at
|
||||
EOF. After: **20 of 20 clean.** At the observed failure rate, 20 clean runs by
|
||||
luck would be about 0.04%, so this is the fix rather than a quieter race.
|
||||
|
||||
## Acceptance Criteria
|
||||
|
||||
Met:
|
||||
|
||||
- **Given** a fresh server with two connected WebSocket clients, **when** it is
|
||||
sent `SIGTERM`, **then** both clients receive a close frame — **repeatedly**,
|
||||
not once. The bug reproduced at 5 in 16, so a single green run proves nothing;
|
||||
the criterion is a run of at least 16 with zero failures.
|
||||
✅ **20 of 20**, from 5-in-16 failing. A single run would have proved nothing.
|
||||
- **Given** an actor whose shard is idle at the moment of the stop, **when** a
|
||||
message is sent to it before the stop flag is observed, **then** its
|
||||
`receive` runs before the engine stops. ✅ this is exactly the case that
|
||||
failed — the Room between messages — and it is what the branch now covers.
|
||||
- **Given** the drain window, **when** a worker has nothing to adopt, **then**
|
||||
it does not hot-spin. ✅ `sched_yield()` on an empty poll; the 1k soak's RSS
|
||||
and timing legs are unchanged (marker reached all 1000 in 28 ms).
|
||||
- **Given** `just chat`, **when** it runs at the default soak, **then** all
|
||||
legs pass on both `WO_IO` backends and under the ASan build with zero leaks.
|
||||
✅ 11 checks, 0 failures. The fd leg also settled the lazy-init question at
|
||||
scale: **1000 connections left the count at 44**, unchanged after 20 more.
|
||||
- **Given** the full runtime battery, **when** it runs, **then** no suite
|
||||
regresses — this touches the shard loop every actor program uses. ✅ 36 suites
|
||||
0 fail, plus the compiler's 556 checks.
|
||||
- **Given** a program with no worker shards (`WO_SHARDS=1`), **when** it stops,
|
||||
**then** behaviour is unchanged. ✅ the gate's `WO_SHARDS=1` leg passes, and
|
||||
the branch is unreachable there — `wo_engine_stop` returns early at
|
||||
`nshards <= 1`, so a single-shard program never enters a worker loop.
|
||||
|
||||
Outstanding:
|
||||
|
||||
- **A pin below the gate.** The guarantee is currently proven by the chat gate
|
||||
only. Nothing in `runtime/test/` drives `wo_engine_start`/`wo_engine_stop`,
|
||||
and no corpus fixture can trigger a stop, so pinning it lower means new
|
||||
multithreaded test infrastructure. Named as its own cost rather than assumed
|
||||
cheap.
|
||||
|
||||
## Out Of Scope
|
||||
|
||||
- **Unbounded drain.** The window is the primary's and closes when `main`
|
||||
returns. A program that wants longer holds the window open itself.
|
||||
- **Delivering sends issued *after* the stop flag.** Nothing promises that, and
|
||||
promising it would mean a program could refuse to exit.
|
||||
- **Resuming parked fibers on stop.** `WO_SYS_STOPPED` unwinds them; that
|
||||
contract is iteration 24's and stays.
|
||||
- **A shutdown acknowledgement in the language surface.** The alternative fix
|
||||
was a barrier the sample builds itself, rejected below.
|
||||
- **`main` parking after the stop flag.** Still forbidden — a park after the
|
||||
flag unwinds. `main` still spins; the point is that spinning now works
|
||||
because the workers cooperate.
|
||||
|
||||
## Info — the forks, settled
|
||||
|
||||
1. **Engine guarantee, not a sample barrier.** The alternative was an
|
||||
acknowledged drain: rooms confirm back to `main`, which waits. Rejected —
|
||||
`main` cannot park after the stop flag, so it could only spin on the
|
||||
acknowledgement anyway, and every future actor program would have to
|
||||
re-implement the same handshake to avoid losing mail. A guarantee is stated
|
||||
once; a barrier is re-invented per program.
|
||||
2. **Not the spin budget.** Replacing the sample's `spin < 20000000` with a 1 s
|
||||
wall-clock deadline still failed 2 of 12. More time cannot help when the
|
||||
shard is not scheduled at all, and the reverted attempt cost a fixed second
|
||||
on every shutdown. Recorded because a bigger spin is the obvious wrong fix.
|
||||
3. **Not `dummy_writer()`.** Hoisting the shutdown message's placeholder actor
|
||||
out of the drain path (it spawned during shutdown) left 5 of 16 failing.
|
||||
4. **Yield rather than spin in the idle drain.** A worker polling an empty
|
||||
inbox in a tight loop would burn a core per shard during the window and
|
||||
starve the actors being drained.
|
||||
5. **Chain position 3**, with [31](31-actor-lifecycle.md): it is lifecycle
|
||||
semantics, and [24](24-chat-websocket-workload.md)'s gate is what proves it.
|
||||
265
docs/superpowers/plans/2026-08-28-wal-checkpoint.md
Normal file
265
docs/superpowers/plans/2026-08-28-wal-checkpoint.md
Normal file
|
|
@ -0,0 +1,265 @@
|
|||
# databasev2 3 — WAL checkpoint (implementation plan)
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use
|
||||
> superpowers:subagent-driven-development (recommended) or
|
||||
> superpowers:executing-plans to implement this plan task-by-task. Steps
|
||||
> use checkbox (`- [ ]`) syntax for tracking.
|
||||
>
|
||||
> **Style rule (user convention):** concept, reason, and required
|
||||
> behaviour in words plus verification commands only — no implementation
|
||||
> or test code blocks; the executor writes the code.
|
||||
|
||||
**Goal:** reclaim disk and bound replay by rewriting the log as one record per
|
||||
live row and swapping it in with `rename`, so boot replays a short log instead
|
||||
of all history.
|
||||
|
||||
**Architecture:** compaction writes the live store into a temporary file using
|
||||
the existing record grammar and the existing append path, fsyncs it, renames it
|
||||
over the live WAL, fsyncs the parent directory, and reopens the descriptor.
|
||||
Recovery is untouched — boot still opens one file and replays it — and every
|
||||
crash point is safe because `rename` is atomic.
|
||||
|
||||
**Tech Stack:** C11, libc only. `pwrite`, `fdatasync`, `rename`, `open`,
|
||||
`unlink`. No new dependency and no new file format.
|
||||
|
||||
**Spec:** [`../specs/2026-08-28-wal-checkpoint-design.md`](../specs/2026-08-28-wal-checkpoint-design.md)
|
||||
|
||||
## Global Constraints
|
||||
|
||||
- **Recovery must not change.** No second source, no cutoff offset, no control
|
||||
file. If a task finds itself editing the replay path, something has gone
|
||||
wrong with the design and it should stop rather than proceed.
|
||||
- **Every crash point falls back.** Before the rename the live log is untouched;
|
||||
after it the new log is complete. There must be no window in which a reader
|
||||
could observe a mixture.
|
||||
- **The record grammar is frozen.** The whole argument for this design is that
|
||||
it already suffices. A compacted log is INSERT records for live rows, ids
|
||||
preserved exactly.
|
||||
- **Bounded memory.** `stage()` grows the staging buffer by doubling and never
|
||||
shrinks it, so dumping a whole store through one buffer would hold the entire
|
||||
store in RAM — the unbounded growth databasev2 1 identified as how this engine
|
||||
dies. The dump must flush periodically.
|
||||
- **Compaction may run only where nothing is staged** — in practice immediately
|
||||
after a barrier. Anywhere else, a staged record lands in a file about to be
|
||||
replaced.
|
||||
- **libc only**, no new syscall interface. Gates run through `just`. Never
|
||||
commit on `master`; branch first.
|
||||
|
||||
---
|
||||
|
||||
## Task 1 — `wo_wal_compact`: rewrite, fsync, rename, reopen
|
||||
|
||||
**Files:**
|
||||
- Modify: `database/src/wal.c`, `database/src/wal.h`.
|
||||
- Test: `runtime/test/test_wal.c`.
|
||||
|
||||
**Interfaces:**
|
||||
- Produces: a compaction entry point taking the live WAL and the store, which
|
||||
replaces the log with one INSERT record per live row and leaves the WAL usable
|
||||
(descriptor reopened, offset correct). Returns success or failure; a failure
|
||||
must leave the ORIGINAL log intact and usable, because a failed checkpoint is
|
||||
not a durability event.
|
||||
- Consumes: the existing append path and commit routine, and the bitmap walk
|
||||
that `db.c` already performs in three places.
|
||||
|
||||
- [ ] Read three things first and confirm them, because the design rests on
|
||||
them: `apply_record` implements UPDATE as remove-then-recreate (so records are
|
||||
full row images), `wo_wal_append_insert` takes an id and reads the row from
|
||||
the store (so ids are preserved), and the tail scan treats a zero length field
|
||||
as end-of-log (so the new file must be zero beyond its records).
|
||||
- [ ] Test first, RED: build a store, age it (insert rows, then update the same
|
||||
rows repeatedly so history exceeds live data), compact, then assert **both**
|
||||
that the log got materially shorter AND that a fresh replay of it produces the
|
||||
same rows with the same ids and the same values. Shorter alone is worthless —
|
||||
a truncating bug also passes that.
|
||||
- [ ] Verify RED for the right reason: the entry point does not exist yet.
|
||||
- [ ] Implement the walk: for each class, iterate slots via the bitmap and
|
||||
append one INSERT per live row. Reuse the append path; do not write a second
|
||||
encoder.
|
||||
- [ ] **Flush every K records rather than staging the whole store.** Point a
|
||||
scratch WAL at the temp descriptor and commit periodically. State the chosen K
|
||||
and why in a comment. Without this the dump holds the entire store in RAM.
|
||||
- [ ] Sequence the switch exactly: fsync the temp file, `rename` over the live
|
||||
path, **fsync the parent directory** (the rename is atomic in-kernel but the
|
||||
directory entry is not durable until the parent is synced), then reopen the
|
||||
descriptor — the old one refers to an unlinked inode — and reset the offset to
|
||||
the new end of log.
|
||||
- [ ] Handle failure without losing data: any error before the rename must
|
||||
unlink the temp file and leave the live log untouched. A failed compaction is
|
||||
a missed optimisation, **not** a durability failure, so it must NOT take the
|
||||
fatal path databasev2 4 introduced.
|
||||
- [ ] GREEN: `just wovm-test`.
|
||||
- [ ] Commit.
|
||||
|
||||
## Task 2 — a stale temp file is removed, never read
|
||||
|
||||
**Files:**
|
||||
- Modify: `database/src/wal.c` (the open path).
|
||||
- Test: `runtime/test/test_wal.c`.
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: Task 1's temp-file naming.
|
||||
- Produces: the guarantee that a crash mid-rewrite leaves nothing that can be
|
||||
mistaken for data.
|
||||
|
||||
- [ ] Test first, RED: place a temp file next to the log containing *plausible,
|
||||
well-formed records* (not garbage — garbage would be rejected anyway and would
|
||||
prove nothing), open the store, and assert the temp file is gone and the
|
||||
replayed store is exactly what the live log said.
|
||||
- [ ] Verify RED for the right reason.
|
||||
- [ ] Remove any stale temp file when the WAL is opened. Note in a comment why
|
||||
this is safe: the only way one exists is a crash before a rename, and its
|
||||
contents are by definition not yet authoritative.
|
||||
- [ ] GREEN: `just wovm-test`.
|
||||
- [ ] Commit.
|
||||
|
||||
## Task 3 — the trigger, and the ordering guard
|
||||
|
||||
**Files:**
|
||||
- Modify: `database/src/wal.c`, `database/src/wal.h` (remember the last
|
||||
compaction's size; the policy decision), `runtime/src/vm.c` (call the check
|
||||
after the barrier).
|
||||
- Test: `runtime/test/test_wal.c`.
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: Task 1's compaction entry point.
|
||||
- Produces: automatic compaction, and the invariant that it never runs with
|
||||
records staged.
|
||||
|
||||
- [ ] Extract the policy as a **pure decision** — given the log's used bytes,
|
||||
the bytes the last compaction wrote, and a floor, should we compact? Pure
|
||||
because it is then unit-testable without a store, which is the only way this
|
||||
policy gets tested at all.
|
||||
- [ ] Test the decision directly, RED then GREEN: below the floor it never
|
||||
fires however bad the ratio; above the floor it fires exactly when used bytes
|
||||
exceed the multiple; with no prior compaction it uses the floor alone.
|
||||
- [ ] Record the bytes each compaction wrote, so the denominator is measured
|
||||
rather than estimated. Estimating the live size would mean estimating Text,
|
||||
and the compactor already knows the true number.
|
||||
- [ ] Expose the floor and the ratio as env knobs, matching the existing idiom
|
||||
(`WO_MAILBOX`, `WO_HEAP_MB`, `WO_SHARDS`, `WO_WAL_STATS`). **This is what
|
||||
makes the policy testable** — a test sets a tiny floor and forces compaction
|
||||
in a few writes instead of waiting for megabytes. Document them beside the
|
||||
others. **Deviation from the spec, disclosed:** the spec spoke of a "manual
|
||||
trigger for tests"; env-tunable thresholds serve that purpose without adding
|
||||
language surface, which is the cheaper way to buy the same testability.
|
||||
- [ ] **No timer.** If the implementer is tempted, the reason is in the spec:
|
||||
Postgres' `CheckPointTimeout` bounds loss from unflushed buffers, our records
|
||||
are durable at commit, and an idle log does not grow.
|
||||
- [ ] Call the check from the one place that is safe — immediately after the
|
||||
drain's barrier, where nothing is staged. Comment that this is a correctness
|
||||
requirement and not a scheduling preference.
|
||||
- [ ] Verify the guard: a test that stages records and then makes the policy
|
||||
say yes must find compaction deferred, not executed. This is the assertion
|
||||
that keeps the ordering rule true as the code moves.
|
||||
- [ ] Verify durability is unaffected: `just db-bench --quick` — the crash and
|
||||
restart legs must be unchanged, and part A's `wmix` legs must still batch.
|
||||
- [ ] Commit.
|
||||
|
||||
## Task 4 — kill -9 *during* compaction
|
||||
|
||||
**Files:**
|
||||
- Test: `runtime/test/test_wal.c` (extend the existing fork-based crash
|
||||
battery).
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: Tasks 1–3.
|
||||
- Produces: the evidence for the criterion the whole design is shaped around.
|
||||
|
||||
- [ ] Read the existing crash battery first: a forked child inserts and acks
|
||||
each committed id over a pipe while the parent SIGKILLs it mid-stream, then
|
||||
the parent verifies every acked id survived. Extend that shape rather than
|
||||
inventing a second harness.
|
||||
- [ ] Drive compaction repeatedly in the child (a tiny floor makes it fire
|
||||
often) while it inserts and acks, and kill at many instants so the kill lands
|
||||
inside a rewrite, at the rename, and after it.
|
||||
- [ ] Assert the property, not a state: after replay the store must equal
|
||||
**either** the pre-compaction **or** the post-compaction content — never a
|
||||
mixture — and **every acked id must be present**. A test that only checks "it
|
||||
replayed without error" would pass on a silently truncated log.
|
||||
- [ ] Assert no temp file survives a kill in a way that affects the next boot.
|
||||
- [ ] Run the battery repeatedly, not once: this is a race, and one green run
|
||||
proves very little. State how many repetitions were run in the commit message.
|
||||
- [ ] GREEN: `just wovm-test` plus the repetitions.
|
||||
- [ ] Commit.
|
||||
|
||||
## Task 5 — measure: space, boot, and the pause
|
||||
|
||||
**Files:**
|
||||
- Modify: `scripts/db-bench.py` (a checkpoint leg), `docs/plan/perf-targets.md`,
|
||||
`bench/baseline.json` (refresh, with the reason in the commit message).
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: Tasks 1–3.
|
||||
- Produces: the before/after record, and the pause number the spec deliberately
|
||||
refused to assume.
|
||||
|
||||
- [ ] Capture the before numbers already measured on master, rather than
|
||||
re-deriving them: `seed 20000` leaves a 986 614-byte log; 20 000 updates take
|
||||
it to 2 590 262 bytes **with the same live rows**; boot+verify on that aged
|
||||
store is 155 ms.
|
||||
- [ ] Add a leg that ages a store, compacts it, and records: bytes before and
|
||||
after, the ratio reclaimed, and boot time before and after. Age it by
|
||||
updating the same rows — history must grow while the live set does not, or the
|
||||
leg is measuring insert throughput instead of compaction.
|
||||
- [ ] Measure the **stop-the-world pause** on the largest store the harness
|
||||
builds and record it as a number. State the budget it must meet.
|
||||
- [ ] **If the pause exceeds the budget, stop and report it.** That is the
|
||||
finding the spec asked for, and the alternatives (incremental copy,
|
||||
fork-and-dump) are bought against this number — not before it.
|
||||
- [ ] Give the new metrics tolerances that match what they are: bytes reclaimed
|
||||
is structural and can be gated tightly; the pause is wall-clock on a shared
|
||||
box and cannot. Do not waive them all, which is the mistake part A's task 4
|
||||
made and had to undo.
|
||||
- [ ] Verify the gate bites: doctor the reclaimed-bytes metric and confirm the
|
||||
suite fails on exactly that metric.
|
||||
- [ ] Refresh the baseline and confirm the **full** campaign passes against it.
|
||||
The committed baseline is full-mode (`N=20000`, `crash_reps=3`) — writing a
|
||||
quick-mode baseline over it is a regression, and part A made exactly that
|
||||
mistake.
|
||||
- [ ] Commit.
|
||||
|
||||
## Task 6 — closeout
|
||||
|
||||
**Files:**
|
||||
- Modify: `docs/stories/databasev2/03-wal-checkpoint.md`,
|
||||
`docs/stories/00-status.md`, `docs/plan/oop-vm/04-db-binding.md`,
|
||||
`database/src/CODE-LOGIC.md`, `docs/examples/db-bench/README.md`.
|
||||
|
||||
- [ ] `04-db-binding.md`: the normative ordering rule — compaction runs only
|
||||
where nothing is staged, and what recovery does (unchanged: one file, replayed
|
||||
from byte 0). This is the doc the spec named for it.
|
||||
- [ ] `CODE-LOGIC.md`: why one file rather than snapshot-plus-tail, why
|
||||
`rename` is the crash-safety primitive, why the dump flushes periodically, and
|
||||
why a failed compaction is not a durability event. Reasoning, not call graph.
|
||||
- [ ] README: the new env knobs beside the existing ones, and the checkpoint
|
||||
leg.
|
||||
- [ ] Story: progress, criteria split met/outstanding, and the measured
|
||||
before/after.
|
||||
- [ ] Board: standup entry in the six-question shape, and the chain note —
|
||||
chain 6 was the last link, so say what the chain's completion means and what
|
||||
is next.
|
||||
- [ ] **Record the `resident: keys` obligation prominently, in the story and at
|
||||
the compactor.** Compaction moves every record, so it invalidates every WAL
|
||||
offset iteration 2 stores; the compactor must rebuild that map as it writes.
|
||||
There is nothing to implement today because iteration 2's storage half does
|
||||
not exist — which is exactly why this must be written where the next
|
||||
implementer will hit it, not left in a spec they may not read.
|
||||
- [ ] Full battery: `just wovm-test`, `just woc-test`, `just oop-e2e`,
|
||||
`just db-bench`, `python3 scripts/linkcheck.py .`
|
||||
- [ ] Commit.
|
||||
|
||||
## Self-review notes
|
||||
|
||||
- **Spec coverage.** Compaction and the switch → Task 1. Stale temp → Task 2.
|
||||
Trigger, no timer, ordering rule → Task 3. Crash safety → Task 4. Space, boot,
|
||||
pause → Task 5. Normative doc, `resident: keys` obligation → Task 6.
|
||||
- **The riskiest task is 4**, not 1: Task 1's correctness is a single replay
|
||||
comparison, while Task 4 is a race and can pass by luck. Hence the explicit
|
||||
instruction to run it repeatedly and to state the count.
|
||||
- **Task 2 looks trivial and is not.** A stale temp file containing well-formed
|
||||
records is the one input that could be mistaken for data, so the test uses
|
||||
plausible records rather than garbage.
|
||||
- **One thing deliberately NOT a task:** rebuilding the `resident: keys` offset
|
||||
map. It cannot be implemented against a feature that does not exist yet.
|
||||
Recorded as an obligation in Task 6 instead of a stub nobody can test.
|
||||
249
docs/superpowers/plans/2026-08-28-wal-group-commit.md
Normal file
249
docs/superpowers/plans/2026-08-28-wal-group-commit.md
Normal file
|
|
@ -0,0 +1,249 @@
|
|||
# databasev2 4 part A — WAL group commit (implementation plan)
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use
|
||||
> superpowers:subagent-driven-development (recommended) or
|
||||
> superpowers:executing-plans to implement this plan task-by-task. Steps
|
||||
> use checkbox (`- [ ]`) syntax for tracking.
|
||||
>
|
||||
> **Style rule (user convention):** concept, reason, and required
|
||||
> behaviour in words plus verification commands only — no implementation
|
||||
> or test code blocks; the executor writes the code.
|
||||
|
||||
**Goal:** one durability barrier per drain instead of one per statement, so a
|
||||
writer is acknowledged after the barrier that carried its record rather than
|
||||
after a barrier of its own.
|
||||
|
||||
**Architecture:** the barrier moves up, not out. Applying to RAM and staging the
|
||||
record stay exactly where they are in `db.c`; the request path stops committing
|
||||
after each append and instead holds its reply envelope, and shard 0 issues one
|
||||
commit when it runs out of queued requests, then releases every held reply. Any
|
||||
failure between "RAM mutated" and "record durable" ends the process with a
|
||||
diagnostic.
|
||||
|
||||
**Tech Stack:** C11, libc only. `pwrite` + `fdatasync` (unchanged — io_uring is
|
||||
part B). The existing per-shard envelope inbox carries the requests.
|
||||
|
||||
**Spec:** [`../specs/2026-08-28-wal-group-commit-design.md`](../specs/2026-08-28-wal-group-commit-design.md)
|
||||
|
||||
## Global Constraints
|
||||
|
||||
- **Durability is unchanged.** Every guarantee iterations 9 and 22 proved holds
|
||||
identically: replay-whole-or-not-at-all, torn-tail drop, no acknowledged
|
||||
write ever lost. This changes when the barrier runs, never what the log holds.
|
||||
- **A writer is released only after the barrier carrying its record.** Never
|
||||
before, and never on the strength of a different batch's barrier.
|
||||
- **libc only.** No new dependency, no new syscall interface in part A.
|
||||
- **The payoff metric is `durable.sN.mixwrite`** (today 480 ops/s, p99
|
||||
5888 µs). `durable.s1.*` and both `seed` legs are regression guards, not
|
||||
targets — a serial writer and an all-inline shard have nothing to batch with.
|
||||
- **`WO_T_IO` leaves the write path.** A commit or staging failure is fatal, not
|
||||
catchable. Exit 1 is a trap and exit 2 is a refusal, so this takes a third
|
||||
status of its own.
|
||||
- Gates run through `just`. Never commit on `master`; branch first.
|
||||
|
||||
---
|
||||
|
||||
## Task 1 — a failed barrier is detected, and fatal
|
||||
|
||||
**Files:**
|
||||
- Modify: `database/src/wal.c` (the commit routine's failure returns; a new
|
||||
fatal-commit entry point beside it), `database/src/wal.h` (declare it).
|
||||
- Test: `runtime/test/test_wal.c` (a new case in the existing suite).
|
||||
|
||||
**Interfaces:**
|
||||
- Produces: a commit entry point that takes the WAL and the number of records
|
||||
in the batch, commits, and on failure writes one stderr line naming the
|
||||
failing operation, the `errno` text, the WAL path and the record count, then
|
||||
exits with the durability-failure status. Tasks 2 and 3 call only this.
|
||||
- Consumes: the existing staging buffer and commit routine.
|
||||
|
||||
- [ ] Read the commit routine first and confirm what it already reports: it
|
||||
loops `pwrite` until the staged buffer is written, then `fdatasync`, and
|
||||
returns non-zero on either failing. Confirm the WAL struct carries its path,
|
||||
or add it — the diagnostic is worthless without it.
|
||||
- [ ] Test first, RED: assert the commit routine reports failure when the
|
||||
descriptor is unusable (a closed descriptor gives `EBADF`). This proves the
|
||||
error is *detected*; it does not exercise the exit.
|
||||
- [ ] Verify RED for the right reason — the case must fail because the
|
||||
assertion is unmet, not because the suite does not compile.
|
||||
- [ ] Add the fatal entry point. It must distinguish the two operations in its
|
||||
message: a `pwrite` failure and an `fdatasync` failure are different
|
||||
operational problems and the operator needs to know which.
|
||||
- [ ] GREEN: `just wovm-test`. The new case passes and no existing case moves.
|
||||
- [ ] **Disclosed gap, record it in the commit message:** the exit path itself
|
||||
is not exercised. Forcing a real `fdatasync` failure needs a full or
|
||||
read-only filesystem, which the gate cannot arrange without mount
|
||||
privileges. Do NOT add a fault-injection switch to buy coverage — shipping a
|
||||
binary that can be told to kill itself is the worse trade, and the spec
|
||||
rejected it.
|
||||
- [ ] Commit.
|
||||
|
||||
## Task 2 — the barrier moves to the drain point; replies are held
|
||||
|
||||
**Files:**
|
||||
- Modify: `database/src/db.c` (the request-path arms only — the three commit
|
||||
calls inside the marshaled-statement executor), `runtime/src/vm.c` (the
|
||||
envelope drain loop's DB-statement branch and the end of that loop).
|
||||
- Test: no new fixture; the existing durability battery is the test. It already
|
||||
covers exactly what could break.
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: Task 1's fatal commit entry point.
|
||||
- Produces: the invariant later tasks measure — at most one barrier per drain,
|
||||
and every held reply released only after it.
|
||||
|
||||
- [ ] Read the drain loop's DB-statement branch first. Today it executes the
|
||||
request, marks it done, then immediately pushes a reply envelope that unparks
|
||||
the requester. Note that it runs on shard 0's thread, serialized — that is
|
||||
why no locking is needed anywhere in this task.
|
||||
- [ ] Remove the three commit calls from the request-path executor in `db.c`.
|
||||
Leave applying to RAM and staging untouched, and leave the **inline** path's
|
||||
three commit calls alone — Task 3 owns that path and conflating them is how
|
||||
this change breaks the single-shard configuration.
|
||||
- [ ] In the drain loop, collect reply envelopes in a local list instead of
|
||||
pushing them as each request finishes. A local is correct and deliberate:
|
||||
nothing needs to survive the loop, and per-shard state would outlive the
|
||||
batch it describes.
|
||||
- [ ] At the end of the drain loop, if anything was staged, call Task 1's fatal
|
||||
commit once, then push every held reply.
|
||||
- [ ] Handle the empty case: a drain that executed no DB statements must not
|
||||
commit and must not touch the staging buffer.
|
||||
- [ ] Verify the ack contract has not moved: `just wovm-test` — the WAL and
|
||||
table suites must be unchanged, since neither knows about batching.
|
||||
- [ ] Verify durability end to end: `just db-bench --quick`. The restart-replay
|
||||
and `kill -9` crash legs are the ones that matter — a kill between staging and
|
||||
the barrier must lose only unacknowledged writes. **If a crash leg fails here,
|
||||
stop; do not adjust the test.** That leg failing means the ack contract broke,
|
||||
which is the one thing this task may not do.
|
||||
- [ ] Commit.
|
||||
|
||||
## Task 3 — the inline path keeps its own barrier, and says why
|
||||
|
||||
**Files:**
|
||||
- Modify: `database/src/db.c` (the inline path's three commit calls — replace
|
||||
with Task 1's fatal entry point), plus the comment above them.
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: Task 1's fatal commit entry point.
|
||||
- Produces: nothing new. This task exists to make the asymmetry deliberate and
|
||||
legible rather than accidental.
|
||||
|
||||
- [ ] Replace the inline path's three commit calls with Task 1's fatal entry
|
||||
point, batch size one. Behaviour is unchanged — this is the fatal-failure
|
||||
rule reaching the second path, not batching.
|
||||
- [ ] Write the comment that explains the asymmetry, because the next reader
|
||||
will otherwise "fix" it: the inline path cannot hold a reply, because it
|
||||
returns into its own fiber rather than unparking a requester. Batching it
|
||||
would require parking that fiber on the barrier, which is part B's machinery
|
||||
and deliberately out of part A.
|
||||
- [ ] Confirm the ordering assumption holds: because the drain loop always
|
||||
commits before it ends, nothing uncommitted is ever left staged when an
|
||||
inline statement runs. If that stops being true the inline path would commit
|
||||
another statement's record early — say so in the comment as the reason the
|
||||
drain must commit unconditionally.
|
||||
- [ ] Verify: `just wovm-test` and `just db-bench --quick` both green, and
|
||||
`WO_SHARDS=1` in particular — the single-shard configuration takes this path
|
||||
exclusively.
|
||||
- [ ] Commit.
|
||||
|
||||
## Task 4 — prove batches actually form
|
||||
|
||||
**Files:**
|
||||
- Modify: `scripts/db-bench.py` (new metrics and their tolerances),
|
||||
`docs/examples/db-bench/main.wo` only if the batch figures cannot be observed
|
||||
without the sample reporting them.
|
||||
- Test: the driver's own gate-bites check.
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: the batching from Task 2.
|
||||
- Produces: mean batch size, peak batch size and peak staged bytes as recorded
|
||||
metrics, so Task 5 measures a mechanism that is known to engage.
|
||||
|
||||
- [ ] Decide where the counters live and prefer the smallest surface: the
|
||||
runtime can report them at exit, or the driver can derive them. Do not add a
|
||||
builtin for this — the numbers are diagnostic, not part of the language.
|
||||
- [ ] Record mean and peak batch size under the concurrent multi-shard write
|
||||
workload. **This is the task's real point:** if batches are always one, the
|
||||
feature is inert and any throughput change came from somewhere else, so the
|
||||
measurement in Task 5 would be attributing a win to the wrong cause.
|
||||
- [ ] Record peak staged bytes. This settles whether the batch needs a cap with
|
||||
a number instead of a guess — the spec deliberately shipped no cap because the
|
||||
request queue is already bounded upstream by iteration 24's mailbox caps.
|
||||
- [ ] Give the new metrics wide tolerances. Batch size is a function of arrival
|
||||
timing, so gating it tightly would gate the scheduler; what must be gated is
|
||||
that it is greater than one under contention.
|
||||
- [ ] Verify the gate bites: doctor the recorded mean batch size to one and
|
||||
confirm the suite fails on exactly that metric.
|
||||
- [ ] Commit.
|
||||
|
||||
## Task 5 — measure the payoff, gate it, write it down
|
||||
|
||||
**Files:**
|
||||
- Modify: `bench/baseline.json` (refresh, with the reason in the commit
|
||||
message), `docs/plan/perf-targets.md` (a new section).
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: Tasks 2 and 4.
|
||||
- Produces: the before/after record every later optimization argues against.
|
||||
|
||||
- [ ] Capture the before numbers from the committed baseline rather than
|
||||
re-measuring them: `durable.sN.mixwrite` 480 ops/s, p50 538 µs, p99 5888 µs;
|
||||
`durable.s1.mixwrite` 1023 ops/s, p99 664 µs; `seed` ~4460 ops/s on both.
|
||||
- [ ] Run the full campaign, not the quick one, and record after numbers for
|
||||
the same metrics on the same machine. A payoff measured across machines is
|
||||
not a payoff.
|
||||
- [ ] Assert the scoped criterion: **`durable.sN.mixwrite` throughput up and
|
||||
p99 down**, with `durable.s1.*` and both `seed` legs not regressed. Do not
|
||||
report the s1 seed number as a disappointment — a serial writer has nothing
|
||||
to batch with, and the spec says so.
|
||||
- [ ] Write the `perf-targets.md` section: the before/after table, the mean and
|
||||
peak batch size that produced it, and the peak staged bytes. State the
|
||||
inversion that motivated the work — multi-shard concurrent writes were 2×
|
||||
slower than single-shard with a 9× worse p99 — and whether it is now gone.
|
||||
- [ ] If the payoff is absent or small, **say so and stop.** That is a finding,
|
||||
not a failure: it would mean the barrier was not the bottleneck the baseline
|
||||
implied, and part B must not be started on an unproven premise.
|
||||
- [ ] Refresh the baseline and confirm `just db-bench` passes against it, then
|
||||
re-confirm the gate bites on a doctored write metric.
|
||||
- [ ] Commit.
|
||||
|
||||
## Task 6 — closeout
|
||||
|
||||
**Files:**
|
||||
- Modify: `docs/stories/databasev2/04-io-uring-commit.md` (progress, criteria
|
||||
split met/outstanding, the landing banner),
|
||||
`docs/stories/00-status.md` (standup entry, chain note),
|
||||
`docs/plan/oop-vm/01-error-catalog.md` (the `WO_T_IO` removal and the new
|
||||
exit status), `database/src/CODE-LOGIC.md` (a group-commit section).
|
||||
|
||||
- [ ] Story: record what landed and what did not. The outstanding items are
|
||||
single-shard concurrent batching (needs the inline park) and part B itself.
|
||||
Keep the corrected premise visible — this iteration was written as
|
||||
"fsync-per-commit" and the engine was fsync-per-statement.
|
||||
- [ ] Error catalogue: `WO_T_IO` no longer reachable from a write, and the new
|
||||
durability-failure exit status documented beside the trap and refusal codes.
|
||||
A language-visible removal that is not written down is a trap for the next
|
||||
reader.
|
||||
- [ ] `CODE-LOGIC.md`: the commit path as built — where the barrier runs, why
|
||||
replies are held, why the inline path is asymmetric, and the one rule for
|
||||
failure. Explain the reasoning, not the call graph.
|
||||
- [ ] Board: the standup entry in the six-question shape, and the chain note —
|
||||
part B's go/no-go now rests on Task 5's number.
|
||||
- [ ] Full battery after the doc edits: `just wovm-test`, `just woc-test`,
|
||||
`just oop-e2e`, `just db-bench`, `python3 scripts/linkcheck.py .`
|
||||
- [ ] Commit.
|
||||
|
||||
## Self-review notes
|
||||
|
||||
- **Spec coverage.** Queue-drain boundary → Task 2. Fatal failure rule → Tasks 1
|
||||
and 3. Held replies and the ack contract → Task 2. No batch cap, settled by
|
||||
measurement → Task 4. Payoff and its scoping → Task 5. `WO_T_IO` removal →
|
||||
Task 6. The disclosed abort-coverage gap → Task 1's last step.
|
||||
- **The riskiest task is 2**, and its risk is concentrated in one place: the
|
||||
crash legs of the durability battery. That is why the plan says stop rather
|
||||
than adjust if they fail.
|
||||
- **Task 3 looks like a no-op and is not.** Without it the inline path keeps a
|
||||
catchable `WO_T_IO` while the request path aborts, which is precisely the
|
||||
per-path unevenness this spec exists to remove.
|
||||
- **Task 4 before Task 5 is deliberate.** Measuring a payoff before proving the
|
||||
mechanism engages is how a win gets attributed to the wrong cause.
|
||||
196
docs/superpowers/specs/2026-08-28-wal-checkpoint-design.md
Normal file
196
docs/superpowers/specs/2026-08-28-wal-checkpoint-design.md
Normal file
|
|
@ -0,0 +1,196 @@
|
|||
# WAL checkpoint — design
|
||||
|
||||
> databasev2 [3](../../stories/databasev2/03-wal-checkpoint.md), chain 6.
|
||||
> Brainstormed and approved 2026-08-28, after
|
||||
> [databasev2 4 part A](2026-08-28-wal-group-commit-design.md) landed.
|
||||
>
|
||||
> **One sentence:** compact the log by rewriting it as one record per live row
|
||||
> into a temporary file, then `rename` it over the live WAL — so recovery is
|
||||
> unchanged and crash safety comes from the filesystem.
|
||||
|
||||
## Decisions taken (the brainstorm's forks, settled)
|
||||
|
||||
| Fork | Decision |
|
||||
| --- | --- |
|
||||
| Snapshot format | **None.** The compacted log *is* the snapshot, in the existing record grammar |
|
||||
| One source or two | **One.** Rewrite + atomic `rename`; boot logic is untouched |
|
||||
| Trigger | **Volume only**, as a ratio against the last compaction's own size, with an absolute floor. **No timer** — see below |
|
||||
| Write availability | **Stop-the-world**, measured against a stated budget rather than assumed acceptable |
|
||||
| Composition with group commit | Compaction runs only where **nothing is staged** — immediately after a barrier |
|
||||
| `resident: keys` (iteration 2) | Compaction **rebuilds the offset map** as it writes. It cannot be left to discover this later |
|
||||
|
||||
## Why one file, and why Postgres cannot do it
|
||||
|
||||
Postgres was read for this (`.dev/reference/postgresql`), and the conclusion is
|
||||
that its design is *unavailable* to us — which is what makes the simpler option
|
||||
legitimate rather than lazy.
|
||||
|
||||
| | PostgreSQL | writeonce |
|
||||
| --- | --- | --- |
|
||||
| Where data lives | heap/data files; the WAL is a redo tail | **the WAL is the only durable form**, replayed into RAM |
|
||||
| WAL contents | page deltas and full-page images | **full row images** — `apply_record` implements UPDATE as remove-then-recreate |
|
||||
| Compaction | **never**; segments before the redo point are recycled by `rename` or unlinked | possible, because a log of row images *is* a complete store |
|
||||
| Bounded replay | recovery starts at the redo LSN in the control file | recovery starts at byte 0 of a *shorter* log |
|
||||
| Crash safety of the switch | control file written in place, full block, torn writes caught by **CRC32C** (`update_controlfile`) | one `rename` |
|
||||
| Trigger | `CheckPointTimeout` (300 s) **or** WAL volume (`XLogCheckpointNeeded`) | volume only |
|
||||
| Pause | none; flush is spread over time in a **separate process** | stop-the-world |
|
||||
|
||||
Postgres cannot compact its WAL because a compacted redo log is not a store —
|
||||
its records describe changes to pages that live elsewhere. Ours describe whole
|
||||
rows, so the compacted log needs no companion. That single difference removes
|
||||
the control file, the redo pointer, the second recovery source, and the separate
|
||||
process from our design.
|
||||
|
||||
**What is worth porting is not the architecture but the ordering discipline:**
|
||||
publish the new "recovery starts here" atomically and *last*, so a crash at any
|
||||
instant falls back to the previous state with nothing to undo. Postgres achieves
|
||||
that with a redo pointer computed at checkpoint *start* and a control file
|
||||
updated at the *end*. We achieve the same property with `rename`, in one
|
||||
syscall, because we can swap the entire data set atomically and Postgres cannot.
|
||||
|
||||
**Correction to a prior exploration doc.**
|
||||
`docs/plan/exploration/postgresql/buffer-and-checkpoint.md` states that Postgres
|
||||
updates its control file by rename ("the same in `BasicOpenFile` +
|
||||
`fsync_parent_path`"). It does not — `update_controlfile` opens the existing file
|
||||
`O_WRONLY`, writes a zero-padded full block in place, and relies on CRC32C to
|
||||
detect a torn write. That doc also assumes writeonce has **segment files**
|
||||
("records before that LSN are *known* to be in the segment files"), which it
|
||||
does not and, per databasev2 2, deliberately will not. The doc predates the
|
||||
databasev2 direction and should be annotated rather than followed.
|
||||
|
||||
## The design
|
||||
|
||||
### Compaction
|
||||
|
||||
Run on the owner shard, which owns the WAL. Walk each class's live rows — the
|
||||
bitmap-over-slabs walk that three call sites in `db.c` already perform — and
|
||||
append one INSERT record per live row to a **new** file, using the existing
|
||||
append path. No new encoder, no new decoder, no format.
|
||||
|
||||
Then: fsync the new file, `rename` it over the live path, fsync the parent
|
||||
directory (the rename's atomicity is in-kernel; the directory entry is not
|
||||
durable until the parent is synced — Postgres does the same, and the existing
|
||||
exploration doc is right about *this* part), and reopen the WAL descriptor,
|
||||
because the old one now refers to an unlinked inode.
|
||||
|
||||
**Every crash point is safe without any recovery logic of ours.** Before the
|
||||
rename, the live WAL is untouched and the temp file is garbage. After it, the new
|
||||
log is complete by construction. There is no window in which a reader could see a
|
||||
mixture, so the acceptance criterion — "recovery produces the same consistent
|
||||
store as if the checkpoint had never started" — is satisfied by `rename`, not by
|
||||
code we must get right.
|
||||
|
||||
Two obligations follow. Boot must **unlink a stale temp file** if one is present,
|
||||
because a crash mid-rewrite leaves one behind and it must never be mistaken for
|
||||
data. And the temp file must be zero-padded beyond its records exactly as the
|
||||
live WAL is, because the tail scan identifies the end of the log by a zero
|
||||
length field.
|
||||
|
||||
### When it runs, and where in the sequence
|
||||
|
||||
**The point matters more than the policy.** The drain stages records into one
|
||||
buffer and commits them together; compaction rewrites the file those records
|
||||
would land in. So compaction may run **only when nothing is staged** — in
|
||||
practice, immediately after a barrier, before the next statement is served.
|
||||
Anywhere else and a staged record would either be written to a file about to be
|
||||
replaced, or be lost with it. This is the normative ordering rule that
|
||||
[`04-db-binding.md`](../../plan/oop-vm/04-db-binding.md) must carry.
|
||||
|
||||
**Trigger: volume, as a self-tuning ratio.** Compact when the WAL's used bytes
|
||||
exceed a multiple of the bytes the *last* compaction wrote, with an absolute
|
||||
floor so a small store never bothers. The denominator is known exactly — the
|
||||
compactor wrote it — so this needs no estimate of the live set's size, which is
|
||||
not cheaply knowable when rows hold Text. The floor exists because a store whose
|
||||
whole log is a few hundred kilobytes has nothing to reclaim.
|
||||
|
||||
**No timer, and that is a deliberate difference from Postgres.** Postgres needs
|
||||
`CheckPointTimeout` because its dirty buffers are not durable until flushed — an
|
||||
idle-but-dirty system must still checkpoint or it loses data. Our records are
|
||||
already durable at commit; a checkpoint reclaims space and shortens boot and
|
||||
nothing else. An idle system's log does not grow, so a timer would fire with
|
||||
nothing to do. Adding one would be copying Postgres' mechanism without its
|
||||
reason.
|
||||
|
||||
A manual trigger exists for tests, because a policy that can only be observed by
|
||||
waiting is a policy that cannot be tested.
|
||||
|
||||
### The pause, and how it is judged
|
||||
|
||||
Compaction is stop-the-world: the owner shard rewrites the log as one long
|
||||
operation while no statement is served. This is the simplest correct thing, and
|
||||
part A's own experience argues for measuring before buying complexity to avoid
|
||||
it. The dump is O(live rows) encodings plus one write and one barrier, so the
|
||||
expectation is that it is fast — but an expectation is not a measurement, and
|
||||
the proof plan below states the budget it must meet.
|
||||
|
||||
If the measured pause exceeds the budget, **that is a finding and a follow-up,
|
||||
not something this iteration solves by adding concurrency.** The alternative
|
||||
designs (incremental copy, fork-and-dump) cost exactly what Postgres pays, and
|
||||
should only be bought against a number.
|
||||
|
||||
### The interaction that will otherwise be discovered late
|
||||
|
||||
**Compaction invalidates every stored WAL offset.** Rewriting the log moves every
|
||||
record, so any offset captured from the old file is meaningless afterwards — not
|
||||
stale-but-readable, but pointing at an arbitrary byte of a different file.
|
||||
[Iteration 2](../../stories/databasev2/02-table-storage-modes.md)'s
|
||||
`resident: keys` stores exactly such offsets, one per row, and reads rows back
|
||||
through them.
|
||||
|
||||
The compactor therefore **rebuilds the offset map as it writes**: it is emitting
|
||||
the new records and knows each one's new position, so this is the cheap
|
||||
direction and the only one that keeps both features usable together. The
|
||||
alternative — forbidding compaction while any `resident: keys` table is live —
|
||||
would mean the feature that exists to handle huge tables is incompatible with
|
||||
the feature that stops their log growing forever.
|
||||
|
||||
This is recorded here because iteration 2's storage half is not yet
|
||||
implemented, so nothing will fail today. It will fail later, in a way that looks
|
||||
like data corruption rather than a design gap.
|
||||
|
||||
## Proof plan
|
||||
|
||||
| Claim | How it is proven |
|
||||
| --- | --- |
|
||||
| Space is reclaimed | An aged store shrinks. Measured today on master: `seed 20000` gives a 986 614-byte log; 20 000 updates take it to 2 590 262 bytes with **the same live rows**. Compaction must return it to approximately the former |
|
||||
| Replay is bounded | Boot time on the aged store before and after, recorded. Measured today: boot+verify on that store is 155 ms |
|
||||
| Crash safety | `kill -9` at many instants *during* compaction, then replay: the store must equal either the pre-compaction or post-compaction state, never a mixture, and no acked write may be missing. This is the criterion the whole design is shaped around, so it gets the crash battery's treatment rather than one case |
|
||||
| A stale temp file is harmless | Boot with one present, containing plausible records: it is removed and never read |
|
||||
| The pause is known | The stop-the-world pause measured on the largest store the harness builds, recorded as a number with a stated budget — not asserted to be acceptable |
|
||||
| The ordering rule holds | Compaction with records staged must be impossible by construction; a test that stages and then requests compaction must find it deferred, not executed |
|
||||
| Nothing regressed | The full battery, and specifically part A's `wmix` legs: compaction must not change the ack contract or the batching it introduced |
|
||||
|
||||
## Out of scope
|
||||
|
||||
- **A second file, a control file, or a redo pointer.** The Postgres shape,
|
||||
priced above and not needed once the log is self-sufficient.
|
||||
- **Avoiding the pause.** Incremental or forked dumps are bought against a
|
||||
measurement, not in advance.
|
||||
- **Per-shard compaction policy.** One owner shard owns the WAL today; when that
|
||||
changes, this decision is revisited with it.
|
||||
- **Compacting away tombstones across shards, or any cross-shard coordination.**
|
||||
There is one log.
|
||||
- **io_uring for the rewrite** — part B of databasev2 4, whose premise is
|
||||
already under revision.
|
||||
- **Changing the record grammar.** The entire argument for this design is that
|
||||
the grammar already suffices.
|
||||
|
||||
## Alternatives rejected
|
||||
|
||||
**Snapshot + WAL tail (the Postgres shape).** Rejected because it buys write
|
||||
availability at the cost of a second recovery source, a cutoff offset, a control
|
||||
file with its own torn-write detection, and a crash-safety guarantee that
|
||||
depends on our ordering rather than on `rename`. Postgres pays this because its
|
||||
log cannot stand alone; ours can.
|
||||
|
||||
**Compacting in place.** Rejected outright: there is no crash point at which a
|
||||
partially rewritten live log is recoverable, and it trades the one property that
|
||||
makes this design defensible for nothing.
|
||||
|
||||
**A timer trigger.** Rejected with a reason rather than on taste: Postgres' timer
|
||||
exists to bound data loss from unflushed buffers, and we have no unflushed
|
||||
buffers. An idle log does not grow.
|
||||
|
||||
**A ratio against an estimated live-set size.** Rejected in favour of the last
|
||||
compaction's measured output, because estimating the live size means estimating
|
||||
Text, and the compactor already knows the true number.
|
||||
208
docs/superpowers/specs/2026-08-28-wal-group-commit-design.md
Normal file
208
docs/superpowers/specs/2026-08-28-wal-group-commit-design.md
Normal file
|
|
@ -0,0 +1,208 @@
|
|||
# WAL group commit — design
|
||||
|
||||
> databasev2 [4](../../stories/databasev2/04-io-uring-commit.md), part A.
|
||||
> Brainstormed and approved 2026-08-28.
|
||||
>
|
||||
> **This spec covers batching only.** The iteration was split during the
|
||||
> brainstorm: part A amortises one durability barrier across many statements,
|
||||
> part B (io_uring submission) is deferred until A's measurement says whether
|
||||
> the blocking boundary is still the bottleneck. That split matches the
|
||||
> iteration's own fork 1 — "drop-in behind `wo_wal_commit` first, an async
|
||||
> variant only if the scheduler proves the blocking boundary is the
|
||||
> bottleneck" — and it means the throughput win arrives behind a much smaller
|
||||
> correctness surface.
|
||||
|
||||
## Decisions taken (the brainstorm's forks, settled)
|
||||
|
||||
| Fork | Decision |
|
||||
| --- | --- |
|
||||
| Scope | **Batching first, io_uring later.** Two independent wins were being carried as one; only the first needs a new syscall interface, and it is where most of the number lives |
|
||||
| Batch boundary | **Queue-drain.** Shard 0 stages every pending write request, then commits once. No timer, no tunable |
|
||||
| Failure | **Fatal, diagnosed abort.** Any failure between "RAM mutated" and "record durable" ends the process |
|
||||
| Batch cap | **None initially.** Measure peak staged bytes; add a cap only if the queue's existing upstream bound proves insufficient |
|
||||
| Abort coverage | Unit-test the failure *return*; the abort path itself stays covered by inspection, and that gap is disclosed |
|
||||
|
||||
## The problem, read off the engine
|
||||
|
||||
The story says "replace fsync-per-commit with io_uring group-commit". Read
|
||||
against the code, the premise needed correcting: the engine does not commit per
|
||||
*commit*, it commits per **statement**. `db.c` calls `wo_wal_commit`
|
||||
immediately after every append, at all six sites — insert, update and remove,
|
||||
each on both the inline and the DB-actor path. Every single row change is one
|
||||
`pwrite` plus one `fdatasync`.
|
||||
|
||||
That is what the numbers say too. Iteration 22's baseline records durable writes
|
||||
at **4460 ops/s** single-shard and mixed writes at **1023 ops/s**, p50 **430 µs**,
|
||||
p99 **664 µs** — against **1.28M ops/s** for durable reads. Writes are roughly
|
||||
290× slower than reads, and the barrier is the whole of it.
|
||||
|
||||
**The batching machinery already exists and is simply never used.**
|
||||
`wo_wal_commit` writes `w->buf` for `w->len` bytes — a staged buffer that can
|
||||
hold any number of records. Today it never holds more than one, because the
|
||||
caller commits immediately after staging. So part A is closer to removing calls
|
||||
than to adding a mechanism.
|
||||
|
||||
## The design
|
||||
|
||||
### The commit path
|
||||
|
||||
The six `wo_wal_commit` calls come out of `db.c`. Applying to RAM and staging
|
||||
the record stay exactly where they are; only the barrier moves, up to the point
|
||||
where shard 0 runs out of work.
|
||||
|
||||
Shard 0 owns the WAL — DB statements from other shards arrive as marshaled
|
||||
request envelopes and are executed on shard 0's thread, serialized, and a reply
|
||||
envelope unparks the requester. The change is that **the reply is held rather
|
||||
than sent**: shard 0 executes and stages each queued request, keeps draining
|
||||
while requests remain, then issues one barrier, and only then releases every
|
||||
held reply.
|
||||
|
||||
Each requester therefore unparks having been acknowledged after the barrier that
|
||||
carried *its* record — the ack contract the story states, which today is true
|
||||
only because every batch has one member.
|
||||
|
||||
A statement executing inline on shard 0 (rather than arriving as a request)
|
||||
stages and commits before returning, as it does now. It has no reply to hold —
|
||||
it returns into its own fiber — and because the drain always commits before it
|
||||
ends, nothing uncommitted is ever left staged when the inline path runs.
|
||||
|
||||
### Why queue-drain, and what it costs
|
||||
|
||||
The batch boundary is the queue going empty, not a tick and not a timer. Two
|
||||
properties follow, and they are the reason to prefer it:
|
||||
|
||||
- **A lone writer pays nothing.** One queued request means a batch of one, which
|
||||
is today's path at today's latency. Batching engages only under genuine
|
||||
contention, so an idle system is not taxed to serve a busy one.
|
||||
- **The batch self-tunes.** Its size is whatever actually accumulated between
|
||||
drains, so it grows with load rather than with a configured number. There is
|
||||
nothing to set and nothing to set wrong.
|
||||
|
||||
The rejected alternative was the iteration's recorded leaning, the shard tick.
|
||||
That leaning was recorded when the batch was assumed to ride an io_uring
|
||||
submission; with batching landing first, a tick boundary would add up to one
|
||||
quantum of latency even to a lone writer — paying the cost of batching when
|
||||
there is nothing to batch with.
|
||||
|
||||
**No batch cap ships initially, and that is a decision rather than an
|
||||
oversight.** databasev2 1 established that unbounded growth is precisely how
|
||||
this engine dies without warning, so the instinct to bound it is right. But the
|
||||
request queue is already bounded upstream by iteration 24's mailbox caps, and a
|
||||
second bound on the same quantity is a knob that can only be wrong. The proof
|
||||
plan measures peak staged bytes so the question is settled by a number.
|
||||
|
||||
## Failure: one rule, replacing three behaviours
|
||||
|
||||
Today's rollback is uneven, and the code says so. An `insert` whose commit fails
|
||||
removes the row again, under a comment claiming RAM never claims what disk has
|
||||
not acknowledged. An `update` or a `delete` whose commit fails does **not** roll
|
||||
back — its comment admits the state plainly: RAM ahead of disk, trap, do not
|
||||
ack. Nothing acknowledged is lost, but the process continues with divergent
|
||||
state, and batching would multiply that from one row to as many as the batch
|
||||
held.
|
||||
|
||||
The rule that replaces it: **once a statement has mutated RAM, the only outcomes
|
||||
are durable or process death.** It covers both failure points identically —
|
||||
a staging failure and a barrier failure have the same consequence, RAM ahead of
|
||||
disk with no way back, and only one of the three verbs can undo itself.
|
||||
|
||||
Retrying is not an alternative worth designing for. On Linux a failed `fsync`
|
||||
may already have discarded the dirty pages, so a second call can report success
|
||||
having written nothing; the recovery that actually works is replay, which
|
||||
returns exactly the last durable state. That is what the log is for.
|
||||
|
||||
**This removes `WO_T_IO` from the write path.** A program can no longer catch a
|
||||
disk failure on a write. The removal is deliberate — there was never a
|
||||
recovery a program could meaningfully perform with its RAM ahead of its disk —
|
||||
but it is language-visible and must be stated in the story banner and the error
|
||||
catalogue, not slipped in.
|
||||
|
||||
The diagnostic has to earn the abort: the failing operation, the `errno` text,
|
||||
the WAL path, and the number of records in the batch, on stderr, then exit with
|
||||
a status of its own. Exit 1 is a trap and exit 2 is a refusal, so a durability
|
||||
failure takes a third. `abort()` is rejected — a core dump on a full disk is
|
||||
noise, not evidence.
|
||||
|
||||
## What will improve, and what will not
|
||||
|
||||
**Corrected 2026-08-28, after reading the baseline properly.** The spec first
|
||||
pointed at `durable.s1.seed` as the payoff metric. That was wrong, and the
|
||||
reason is structural rather than a matter of degree.
|
||||
|
||||
Worker shards hold no WAL at all — the runtime asserts it — so every DB
|
||||
statement on a worker marshals to shard 0 and parks, while a statement already
|
||||
on shard 0 executes inline. **A queue of write requests therefore exists only
|
||||
when other shards are writing.** Batches form where there is a queue:
|
||||
|
||||
| Workload | Today | Batching |
|
||||
| --- | --- | --- |
|
||||
| `durable.sN.mixwrite` — concurrent writers across shards | **480 ops/s, p99 5888 µs** | **the target.** N shards marshal N writes and shard 0 pays N barriers serially; one barrier replaces them |
|
||||
| `durable.s1.mixwrite` — concurrent writers, one shard | 1023 ops/s, p99 664 µs | **no change.** Every write is inline with no queue, so no batch forms |
|
||||
| `durable.*.seed` — one serial writer | ~4460 ops/s | **no change**, under any batching scheme. There is nothing to batch with |
|
||||
|
||||
The inversion in those numbers is the finding worth keeping: **multi-shard
|
||||
concurrent writes are currently 2× slower than single-shard with a 9× worse
|
||||
p99.** Adding shards makes durable writing worse today, because every marshaled
|
||||
statement still buys its own barrier on the owner. That is the pathology group
|
||||
commit exists to remove, and it is a better argument for this iteration than the
|
||||
one the story recorded.
|
||||
|
||||
**Single-shard concurrent batching is deliberately out of part A.** It would
|
||||
need the inline path to park its fiber on the barrier rather than commit
|
||||
synchronously — the same parking machinery part B needs anyway. Deferring it
|
||||
keeps A to one mechanism, and B inherits the reason to build it.
|
||||
|
||||
So the acceptance criterion is scoped: **`durable.sN.mixwrite` throughput up and
|
||||
its p99 down; `durable.s1.*` and both `seed` legs must not regress.** A plan
|
||||
that reported "no improvement" against the s1 seed number would be measuring a
|
||||
workload this change cannot help.
|
||||
|
||||
## Proof plan
|
||||
|
||||
| Claim | How it is proven |
|
||||
| --- | --- |
|
||||
| The payoff is real | **`durable.sN.mixwrite`** before and after on one machine, recorded in `perf-targets.md`. Today 480 ops/s, p99 5888 µs. `durable.s1.*` and both `seed` legs are regression guards, not targets — see the section above |
|
||||
| Durability is unchanged | Iteration 22's crash battery, unaltered: concurrent writers, `kill -9` mid-stream, replay. **The critical test** — a kill between staging and the barrier must lose only unacknowledged writes |
|
||||
| Batches actually form | New metrics for mean and peak batch size under contention. If batches are always one, the feature is inert and any throughput change came from somewhere else |
|
||||
| No idle tax | Single-writer p99 must not regress against the current baseline |
|
||||
| The cap question is answered | Peak staged bytes recorded per run |
|
||||
| A failure is detected | `test_wal.c` asserts `wo_wal_commit` reports failure on a bad descriptor |
|
||||
|
||||
**One disclosed gap.** Forcing a genuine `fdatasync` failure needs a full or
|
||||
read-only filesystem, which the gate cannot arrange without mount privileges.
|
||||
The unit test proves the error is *detected*; the abort that follows it stays
|
||||
covered by inspection. The alternative — a fault-injection switch — means
|
||||
shipping a binary that can be told to kill itself, which is a worse trade. This
|
||||
gap is recorded rather than hidden, because iteration 40 was exactly a fatal
|
||||
path that nothing exercised.
|
||||
|
||||
## Out of scope
|
||||
|
||||
- **io_uring submission.** Part B, and it only earns its complexity if A's
|
||||
measurement shows the blocking boundary still dominating. A's parking and ack
|
||||
machinery is what B would build on, so nothing here is wasted either way.
|
||||
- **`transaction { }`** — language iteration 18. A transaction already *is* a
|
||||
staged batch, so the two compose without either knowing about the other; that
|
||||
is a reason not to entangle them now.
|
||||
- **Checkpoint and compaction** — databasev2 3. This changes when the barrier
|
||||
runs, never what the log contains.
|
||||
- **The read path.** databasev2 1 measured that appending under memory pressure
|
||||
costs about 1% while random reads cost 273×, so the pressure is on reads —
|
||||
but that is iteration 2's `resident: keys` question, not this one.
|
||||
- **Rollback with pre-images.** Rejected above: it would add per-write cost on
|
||||
every statement to serve a path that ends the process anyway.
|
||||
|
||||
## Alternatives rejected
|
||||
|
||||
**Tick-boundary batching** — the iteration's recorded leaning, superseded by
|
||||
the split. It taxes an idle system to serve a busy one.
|
||||
|
||||
**Count-or-timer batching** — two tunables, and the timer reintroduces the tick
|
||||
problem with extra configuration.
|
||||
|
||||
**Full rollback with an undo log** — keeps `WO_T_IO` catchable, at the price of
|
||||
capturing pre-images for every update and delete, paid on every write, to
|
||||
support continuing in a state the engine cannot trust.
|
||||
|
||||
**Keeping today's per-verb behaviour** — turns a rare one-row divergence into a
|
||||
routine N-row one, silently.
|
||||
7
justfile
7
justfile
|
|
@ -60,6 +60,13 @@ site:
|
|||
fibers:
|
||||
./scripts/fibers-accept.sh
|
||||
|
||||
# chat: iteration 24's gate (docs/examples/chat) — rooms/presence/broadcast
|
||||
# over WebSocket via actors: functional on both WO_IO backends, the
|
||||
# 1k-clients-one-hot-room soak (fds/RSS accounted), SIGTERM drain with
|
||||
# close frames, and an ASan leg. `just chat` runs it (CHAT_SOAK=N trims).
|
||||
chat:
|
||||
./scripts/chat-accept.sh
|
||||
|
||||
# db-actor: arc stage 3's gate (docs/examples/db-actor) — worker-shard
|
||||
# actors read/write the database through the transparent DB actor; WAL
|
||||
# replay pair included. `just db-actor` runs it.
|
||||
|
|
|
|||
|
|
@ -257,3 +257,100 @@ layout.
|
|||
- **`listen_unix` sets O_NONBLOCK on the listener itself** — accept4's
|
||||
SOCK_NONBLOCK flags the ACCEPTED socket only; a blocking listener
|
||||
would block the whole shard (found by the seam probe, both backends).
|
||||
|
||||
## Actor lifecycle: call, death, monitor, timers (iteration 24, ids 88–90)
|
||||
|
||||
Four pieces that together answer "what happens to an actor that is waiting,
|
||||
that dies, that watches, or that wants to be woken later". All four live in
|
||||
`vm.c` with their entry points in `builtin.c`; the structures are in `vm.h`.
|
||||
|
||||
**`call` (id 88) — a send that waits.** An ordinary `send` returns immediately;
|
||||
`call` parks the calling fiber and resumes it with the receive's return value.
|
||||
The reply is a **typed scalar**, which is what let the agreement be checked at
|
||||
compile time (WO-E226) rather than carried as a tagged value at runtime. The
|
||||
caller is never left hanging: if the callee dies mid-call, or the address is
|
||||
already dead, the caller **traps catchably** instead of parking forever. That
|
||||
is the property worth keeping in mind when reading the code — every path out of
|
||||
a call either resumes the fiber or traps it.
|
||||
|
||||
**Death.** A `receive` that traps uncaught marks the actor dead on its home
|
||||
thread. From then on sends to it drop silently, calls trap, queued callers are
|
||||
error-unparked, and its state and mailbox are released. Silent-drop for sends
|
||||
is deliberate: a sender cannot handle another actor's failure, and making every
|
||||
`send` fallible would put a `try` on every line.
|
||||
|
||||
**The mailbox cap and its counter.** One cap for every mailbox (default 1024,
|
||||
`WO_MAILBOX` overrides at boot; the chat gate shrinks it to 8 to force the
|
||||
policy). `pending` counts sent-but-not-delivered. It is incremented by the
|
||||
**sender**, on any shard, and decremented by the **home thread** at delivery —
|
||||
so it is touched only through `wo_mbox_reserve`/`wo_mbox_release` and their
|
||||
`__atomic` builtins. The consequence is disclosed rather than hidden: the cap
|
||||
can overshoot by at most the number of in-flight sends. Overflow is fail-fast —
|
||||
the send raises a catchable `WO_T_ACTOR` (trap 13), which is what lets a room
|
||||
drop a slow member instead of growing without bound.
|
||||
|
||||
**`monitor` (id 89) — the death notice.** `wo_monitor` is one registration:
|
||||
observer, the moved-in notice message, next. The list lives on the **watched**
|
||||
actor and is owned by its home thread, so the death walk needs no lock — dying
|
||||
is a home-thread event and the list is right there. The notice is the
|
||||
observer's own M-typed message, so an observer receives death notices in the
|
||||
same shape as everything else. Monitoring an already-dead actor fires
|
||||
immediately rather than silently doing nothing. An observer whose mailbox is
|
||||
full loses the notice, with a disclosed stderr line — the alternative was
|
||||
blocking a death walk on a slow observer.
|
||||
|
||||
It takes **three arguments** (`watched, observer, msg`), not the two the spec
|
||||
first proposed, because the caller may be `main`, which has no mailbox and so
|
||||
cannot be an implicit observer.
|
||||
|
||||
**`time.after` (id 90) — one-shot, no cancel.** `wo_timer` is `at` (wall ms),
|
||||
target, message, next. The list lives on the **arming fiber's shard** and is
|
||||
scanned by the same deadline machinery that already serves fd-park deadlines,
|
||||
so timers cost no new wait mechanism. Firing is an ordinary runtime send, which
|
||||
means it inherits the ordinary rules: a full target drops with a stderr line, a
|
||||
dead target drops silently. There is no cancel; the idiom is a generation
|
||||
counter in the message, which the `timer-generation` corpus fixture pins.
|
||||
|
||||
**Where to look when a lifecycle thing misbehaves:** `wo_vm_actor_monitor` and
|
||||
`wo_vm_timer_after` in `vm.c` are the two entry points; `shard_main` and
|
||||
`NEXT_RUNNABLE()` decide when a shard runs, adopts, or stops. The corpus
|
||||
fixtures `monitor-death`, `timer-delivery` and `timer-generation` are the
|
||||
smallest working examples of each.
|
||||
|
||||
## The shutdown drain guarantee (iteration 40)
|
||||
|
||||
**A message sent before the stop flag is observed is delivered and run before
|
||||
the engine stops.** Stated because it was once untrue in a way nothing caught.
|
||||
|
||||
`wo_engine_stop` sets `eng_shutdown`, wakes every worker, joins them, and only
|
||||
then tears down — freeing whatever envelopes are still queued. So a worker that
|
||||
leaves its loop early takes its inbox with it. `NEXT_RUNNABLE()` has always
|
||||
encoded the right behaviour for a worker holding a live fiber: on a stop it
|
||||
returns 2 and keeps draining, because "only the PRIMARY's stop ends the
|
||||
program". `shard_main`'s **idle** branch did the opposite — it reaped and broke
|
||||
— so a shard whose actors happened to be between messages at `SIGTERM`
|
||||
abandoned everything still in flight.
|
||||
|
||||
It now honours the same contract: while the primary's window is open an idle
|
||||
worker adopts its inbox and runs what arrives, `sched_yield`ing on an empty
|
||||
poll so a drain cannot burn a core per shard and starve the actors it exists to
|
||||
let run. Only `eng_shutdown` — which the primary sets after `main` returns —
|
||||
ends it.
|
||||
|
||||
Two things follow that are easy to get wrong. The window is the **primary's**,
|
||||
so a program that wants a longer drain holds it open itself; `main` cannot park
|
||||
after the stop flag, because a park there unwinds. And the whole path is
|
||||
unreachable at `WO_SHARDS=1`, where `wo_engine_stop` returns at `nshards <= 1`.
|
||||
|
||||
## Digests: sha1, sha256, hmac_sha256 (iteration 34, ids 85–87)
|
||||
|
||||
`crypto.c` holds SHA-1 and SHA-256 over a single buffer and HMAC-SHA-256 on top
|
||||
of the latter, each returning a fresh `Bytes`. No streaming API and no other
|
||||
primitives — these exist because WebSocket's handshake needs SHA-1 and ETags
|
||||
need SHA-256, and that is the whole of the demand so far.
|
||||
|
||||
Correctness is pinned to the published vectors rather than to itself:
|
||||
RFC 3174 for SHA-1, the FIPS/RFC 6234 vectors for SHA-256, RFC 4231 for HMAC,
|
||||
in `runtime/test/test_crypto.c` (18 checks). **There is still no RNG anywhere
|
||||
in the runtime** — HMAC authenticates a token but cannot mint one, which is why
|
||||
iteration 39 leads with a random-bytes builtin.
|
||||
|
|
|
|||
|
|
@ -200,6 +200,18 @@ int wo_builtin(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
|||
}
|
||||
case WO_B_CALL: /* iteration 24: park/reply protocol lives in vm.c */
|
||||
return wo_vm_actor_call(vm, R, ins, msg);
|
||||
case WO_B_MONITOR: {
|
||||
int rc = wo_vm_actor_monitor(vm, R[B], R[B + 1], R[B + 2], msg);
|
||||
if (rc) return rc;
|
||||
R[A] = 0;
|
||||
return 0;
|
||||
}
|
||||
case WO_B_TIME_AFTER: {
|
||||
int rc = wo_vm_timer_after(vm, (int64_t)R[B], R[B + 1], R[B + 2], msg);
|
||||
if (rc) return rc;
|
||||
R[A] = 0;
|
||||
return 0;
|
||||
}
|
||||
case WO_B_NOW: { /* wall-clock milliseconds */
|
||||
struct timespec ts;
|
||||
clock_gettime(CLOCK_REALTIME, &ts);
|
||||
|
|
|
|||
|
|
@ -183,6 +183,17 @@ int wo_load_buf(wo_module *m, const uint8_t *buf, size_t len, char *err,
|
|||
* nowhere to live. woc refuses this at compile time; the loader
|
||||
* refuses it again because what the loader accepts, the interpreter
|
||||
* trusts — this combination must never reach the engine. */
|
||||
/* databasev2 2: `resident: keys` PARSES and sets this bit, but the
|
||||
* storage half (tasks 5c/5d) is not implemented — rows are still fully
|
||||
* resident. Accepting it would be an annotation the compiler honours
|
||||
* in name only: a developer could declare a 120 GB table keys-resident,
|
||||
* see it compile, and be OOM-killed. Refuse until the storage lands. */
|
||||
if (flags & WO_CLASSF_RESIDENT_KEYS)
|
||||
BAIL("class %u declares `resident: keys`, which is NOT IMPLEMENTED "
|
||||
"yet — rows are still fully resident, so the annotation would "
|
||||
"be honoured in name only. Remove it until databasev2 2 tasks "
|
||||
"5c/5d land; `resident: all` is what actually runs",
|
||||
(unsigned)i);
|
||||
if ((flags & WO_CLASSF_VOLATILE) && (flags & WO_CLASSF_RESIDENT_KEYS))
|
||||
BAIL("class %u: durable:false with resident:keys — rows would have "
|
||||
"nowhere to be read from", (unsigned)i);
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@
|
|||
* 2 = usage or load failure (loader's message on stderr)
|
||||
* Heap cap defaults to 64 MiB, overridable via WO_HEAP_MB. */
|
||||
#include <fcntl.h>
|
||||
#include <signal.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
|
@ -124,6 +125,22 @@ static void gc_pump(wo_vm *vm) {
|
|||
}
|
||||
}
|
||||
|
||||
/* databasev2 4: one diagnostic line about group commit, opt-in via
|
||||
* WO_WAL_STATS. Off by default because it would otherwise pollute the output
|
||||
* of every durable program; a gate that wants the numbers asks for them. */
|
||||
static void wal_stats_report(const wo_wal *w) {
|
||||
if (!w || !getenv("WO_WAL_STATS")) return;
|
||||
fprintf(stderr,
|
||||
"walstats batches=%llu records=%llu peak_batch=%llu peak_staged=%llu "
|
||||
"compactions=%llu compact_us_max=%llu compact_us_total=%llu compacted_bytes=%llu\n",
|
||||
(unsigned long long)w->stat_batches, (unsigned long long)w->stat_records,
|
||||
(unsigned long long)w->stat_peak_batch, (unsigned long long)w->stat_peak_staged,
|
||||
(unsigned long long)w->stat_compactions,
|
||||
(unsigned long long)w->stat_compact_us_max,
|
||||
(unsigned long long)w->stat_compact_us_total,
|
||||
(unsigned long long)w->compacted_bytes);
|
||||
}
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
wo_module mod;
|
||||
char err[256];
|
||||
|
|
@ -178,6 +195,10 @@ int main(int argc, char **argv) {
|
|||
return 2;
|
||||
}
|
||||
wo_tls_set(&VM);
|
||||
/* iteration 24: a write to a peer-closed socket must be EPIPE (a
|
||||
* catchable WO_T_IO), never a process-killing SIGPIPE — every
|
||||
* serving program writes to sockets whose peers vanish. */
|
||||
signal(SIGPIPE, SIG_IGN);
|
||||
/* The database engine boots with the VM: every class IS a table.
|
||||
* Durability is opt-in — WO_DATA=<dir> opens <dir>/shard-0.wal,
|
||||
* replays it before the entry runs (boot-before-listeners doctrine),
|
||||
|
|
@ -254,6 +275,25 @@ int main(int argc, char **argv) {
|
|||
if (v >= 1 && v <= 0x7FFFFFFFul) wo_mailbox_cap = (uint32_t)v;
|
||||
}
|
||||
}
|
||||
/* databasev2 3: the checkpoint policy. WO_CHECKPOINT_BYTES is the floor
|
||||
* below which a log is too small to bother compacting; WO_CHECKPOINT_RATIO
|
||||
* is how many times the live set's own size counts as too much history.
|
||||
* Both exist mainly so the policy is TESTABLE — a gate sets a tiny floor
|
||||
* and forces compaction in a few writes rather than waiting for megabytes.
|
||||
* There is no time-based trigger, by design: our records are durable at
|
||||
* commit, so an idle log does not grow. */
|
||||
{
|
||||
const char *cb = getenv("WO_CHECKPOINT_BYTES");
|
||||
if (cb && cb[0]) {
|
||||
unsigned long long v = strtoull(cb, NULL, 10);
|
||||
if (v > 0) wo_wal_ckpt_floor = (uint64_t)v;
|
||||
}
|
||||
const char *cr = getenv("WO_CHECKPOINT_RATIO");
|
||||
if (cr && cr[0]) {
|
||||
unsigned long v = strtoul(cr, NULL, 10);
|
||||
if (v <= 0xFFFFFFFFul) wo_wal_ckpt_ratio = (uint32_t)v;
|
||||
}
|
||||
}
|
||||
/* the arc's stage 2: all cores by default (the brave landing), one
|
||||
* pinned worker vm per extra core; WO_SHARDS caps or forces it */
|
||||
{
|
||||
|
|
@ -269,7 +309,7 @@ int main(int argc, char **argv) {
|
|||
if (wo_engine_start(&mod, heap_mb << 20, nshards) != 0) {
|
||||
fprintf(stderr, "wovm: cannot start %u shards\n", nshards);
|
||||
wo_engine_stop();
|
||||
if (VM.rt.wal) wo_wal_close(&WAL);
|
||||
if (VM.rt.wal) { wal_stats_report(&WAL); wo_wal_close(&WAL); }
|
||||
wo_db_destroy(&DB);
|
||||
wo_vm_destroy(&VM);
|
||||
wo_module_free(&mod);
|
||||
|
|
@ -328,7 +368,7 @@ int main(int argc, char **argv) {
|
|||
* unwind. */
|
||||
if (argv_val) wo_drop_kind(&VM.rt, WO_K_MULTI, argv_val);
|
||||
wo_engine_stop(); /* join + destroy the worker shards before the primary */
|
||||
if (VM.rt.wal) wo_wal_close(&WAL);
|
||||
if (VM.rt.wal) { wal_stats_report(&WAL); wo_wal_close(&WAL); }
|
||||
wo_db_destroy(&DB);
|
||||
gc_pump(&VM);
|
||||
wo_vm_destroy(&VM);
|
||||
|
|
|
|||
|
|
@ -296,6 +296,8 @@ static void tick_arm_uring(wo_vm *vm, int64_t now) {
|
|||
for (wo_fiber *fb = vm->parked; fb; fb = fb->pnext)
|
||||
if (fb->state == WO_FIB_PARKED && fb->park_fd >= 0 && fb->park_deadline > 0)
|
||||
if (next == 0 || fb->park_deadline < next) next = fb->park_deadline;
|
||||
int64_t tn = wo_vm_timers_next(vm); /* iteration 24 T5: armed timers */
|
||||
if (tn > 0 && (next == 0 || tn < next)) next = tn;
|
||||
if (next == 0) return;
|
||||
if (vm->tick_armed && vm->tick_at <= next) return;
|
||||
int64_t rel = next - now;
|
||||
|
|
@ -323,7 +325,27 @@ static void efd_drain(wo_vm *vm) {
|
|||
|
||||
int wo_io_wait(wo_vm *vm) {
|
||||
for (;;) {
|
||||
if (wo_sys_stop_pending()) return WO_IO_STOP;
|
||||
if (wo_sys_stop_pending()) {
|
||||
/* iteration 24 (the drain): a STOP does not kill parked fibers
|
||||
* from the outside — it WAKES them all, and each blocking
|
||||
* builtin resolves per its own stop contract (deadline'd waits
|
||||
* answer their timeout result, sleeps return early, plain
|
||||
* waits answer WO_SYS_STOPPED and that fiber unwinds). The
|
||||
* program's own code then drains and returns. Nothing parked
|
||||
* = nothing to resolve: the old immediate-stop answer. */
|
||||
int woke = 0;
|
||||
wo_fiber *fb = vm->parked;
|
||||
while (fb) {
|
||||
wo_fiber *nx = fb->pnext;
|
||||
if (fb->state == WO_FIB_PARKED) {
|
||||
wake(vm, fb);
|
||||
woke = 1;
|
||||
}
|
||||
fb = nx;
|
||||
}
|
||||
if (woke) return 0;
|
||||
return WO_IO_STOP;
|
||||
}
|
||||
if (vm->io_kind == 0) {
|
||||
/* keep the wake eventfd armed (oneshot POLL_ADD, re-armed
|
||||
* after each firing) so inbox pushes interrupt the wait */
|
||||
|
|
@ -371,7 +393,9 @@ int wo_io_wait(wo_vm *vm) {
|
|||
head++;
|
||||
}
|
||||
__atomic_store_n(r.cq_head, head, __ATOMIC_RELEASE);
|
||||
if (deadline_sweep_uring(vm, now_ms()) && woke != 2) woke = 1;
|
||||
int64_t swnow = now_ms();
|
||||
if (wo_vm_timers_fire(vm, swnow) && woke != 2) woke = 1;
|
||||
if (deadline_sweep_uring(vm, swnow) && woke != 2) woke = 1;
|
||||
if (woke == 2) return 1; /* adopt-needed */
|
||||
if (woke) return 0;
|
||||
continue;
|
||||
|
|
@ -389,6 +413,14 @@ int wo_io_wait(wo_vm *vm) {
|
|||
}
|
||||
int timeout = -1;
|
||||
int64_t now = now_ms();
|
||||
{
|
||||
int64_t tn = wo_vm_timers_next(vm); /* iteration 24 T5 */
|
||||
if (tn > 0) {
|
||||
int64_t rel = tn - now;
|
||||
if (rel < 0) rel = 0;
|
||||
timeout = (int)rel;
|
||||
}
|
||||
}
|
||||
for (wo_fiber *fb = vm->parked; fb; fb = fb->pnext)
|
||||
if (fb->park_fd == -1
|
||||
|| (fb->park_fd >= 0 && fb->park_deadline > 0)) {
|
||||
|
|
@ -417,6 +449,7 @@ int wo_io_wait(wo_vm *vm) {
|
|||
}
|
||||
}
|
||||
now = now_ms();
|
||||
if (wo_vm_timers_fire(vm, now)) woke = 1;
|
||||
wo_fiber *fb = vm->parked;
|
||||
while (fb) {
|
||||
wo_fiber *nx = fb->pnext;
|
||||
|
|
|
|||
|
|
@ -395,6 +395,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
|||
if (stop_pending()) return WO_SYS_STOPPED;
|
||||
}
|
||||
if (fd < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) {
|
||||
if (stop_pending()) return WO_SYS_STOPPED;
|
||||
/* arc T4: park until the listener is readable, then retry */
|
||||
vm->cur->park_fd = (int)R[B];
|
||||
vm->cur->park_deadline = 0;
|
||||
|
|
@ -431,6 +432,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
|||
/* arc T4: nothing readable yet — free the buffer (the retry
|
||||
* re-allocates) and park until the fd is readable */
|
||||
wo_str_free(rt, s);
|
||||
if (stop_pending()) return WO_SYS_STOPPED;
|
||||
vm->cur->park_fd = (int)R[B];
|
||||
vm->cur->park_deadline = 0;
|
||||
vm->cur->park_events = POLLIN;
|
||||
|
|
@ -476,6 +478,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
|||
continue;
|
||||
}
|
||||
if (errno == EAGAIN || errno == EWOULDBLOCK) {
|
||||
if (stop_pending()) return WO_SYS_STOPPED;
|
||||
vm->cur->park_wr_at = at;
|
||||
vm->cur->park_fd = (int)R[B];
|
||||
vm->cur->park_deadline = 0;
|
||||
|
|
@ -534,7 +537,9 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
|||
}
|
||||
if (n < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) {
|
||||
wo_str_free(rt, s);
|
||||
if (fb->dl_at > 0 && dnow >= fb->dl_at) {
|
||||
if (stop_pending() || (fb->dl_at > 0 && dnow >= fb->dl_at)) {
|
||||
/* iteration 24: a STOP resolves the wait as its timeout
|
||||
* result — the program's own drain code decides what next */
|
||||
fb->dl_active = 0;
|
||||
R[A] = 0; /* ?Text nil: the deadline expired */
|
||||
return 0;
|
||||
|
|
@ -584,9 +589,9 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
|||
}
|
||||
}
|
||||
if (fd < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) {
|
||||
if (fb->dl_at > 0 && dnow >= fb->dl_at) {
|
||||
if (stop_pending() || (fb->dl_at > 0 && dnow >= fb->dl_at)) {
|
||||
fb->dl_active = 0;
|
||||
R[A] = WO_NIL_SCALAR; /* ?Int nil: nothing arrived */
|
||||
R[A] = WO_NIL_SCALAR; /* ?Int nil: nothing arrived (or stop) */
|
||||
return 0;
|
||||
}
|
||||
fb->park_fd = (int)R[B];
|
||||
|
|
@ -631,7 +636,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
|||
continue;
|
||||
}
|
||||
if (errno == EAGAIN || errno == EWOULDBLOCK) {
|
||||
if (fb->dl_at > 0 && dnow >= fb->dl_at) {
|
||||
if (stop_pending() || (fb->dl_at > 0 && dnow >= fb->dl_at)) {
|
||||
fb->dl_active = 0;
|
||||
R[A] = 0; /* false: torn mid-write — close the fd */
|
||||
return 0;
|
||||
|
|
|
|||
416
runtime/src/vm.c
416
runtime/src/vm.c
|
|
@ -16,6 +16,7 @@
|
|||
|
||||
#include "db.h" /* arc stage 3: the transparent DB RPC (wo_db_req) */
|
||||
#include "table.h" /* slot encode/decode for the RPC marshaling */
|
||||
#include "wal.h" /* databasev2 4: the drain issues the barrier */
|
||||
|
||||
#include <pthread.h>
|
||||
#include <poll.h>
|
||||
|
|
@ -78,6 +79,9 @@ static int actor_push(wo_actor *a, wo_msg m);
|
|||
static void call_reply_to(wo_vm *vm, wo_fiber *caller, uint32_t caller_shard,
|
||||
uint64_t reply, int status);
|
||||
static void actor_drop_payload(wo_vm *vm, uint64_t payload);
|
||||
static void monitors_fire(wo_vm *vm, wo_actor *a);
|
||||
static void runtime_notify(wo_vm *vm, wo_actor *target, uint64_t msg_val,
|
||||
const char *what);
|
||||
|
||||
/* the owning thread drains its inbox: adopt actors, deliver sends,
|
||||
* execute home-routed frees. Returns how many envelopes were handled. */
|
||||
|
|
@ -88,6 +92,11 @@ static int wo_vm_adopt(wo_vm *vm) {
|
|||
ib->head = ib->tail = NULL;
|
||||
pthread_mutex_unlock(&ib->mu);
|
||||
int n = 0;
|
||||
/* databasev2 4 (group commit): DB replies are HELD until one barrier has
|
||||
* covered the whole drain. Locals, not per-shard state: nothing here needs
|
||||
* to outlive the batch it describes. */
|
||||
wo_envelope *rhead = NULL, *rtail = NULL;
|
||||
uint32_t staged = 0;
|
||||
while (e) {
|
||||
wo_envelope *nx = e->next;
|
||||
switch (e->kind) {
|
||||
|
|
@ -133,6 +142,25 @@ static int wo_vm_adopt(wo_vm *vm) {
|
|||
}
|
||||
break;
|
||||
}
|
||||
case 7: { /* iteration 24 T4: a cross-shard monitor registration —
|
||||
WE are the watched actor's home. Dead already = the
|
||||
notice fires now; else it joins the list. */
|
||||
wo_actor *ob = (wo_actor *)(uintptr_t)e->from_fiber;
|
||||
if (e->actor->dead) {
|
||||
runtime_notify(vm, ob, e->payload, "death notice");
|
||||
break;
|
||||
}
|
||||
wo_monitor *mn = calloc(1, sizeof *mn);
|
||||
if (!mn) {
|
||||
actor_drop_payload(vm, e->payload);
|
||||
break;
|
||||
}
|
||||
mn->observer = ob;
|
||||
mn->msg = e->payload;
|
||||
mn->next = e->actor->monitors;
|
||||
e->actor->monitors = mn;
|
||||
break;
|
||||
}
|
||||
case 6: /* iteration 24: a call reply landing on the caller's shard —
|
||||
fill the slot and wake the parked fiber; the re-executed
|
||||
builtin consumes it (status != 0 makes it trap). */
|
||||
|
|
@ -149,13 +177,35 @@ static int wo_vm_adopt(wo_vm *vm) {
|
|||
* the same request back as the reply. */
|
||||
wo_db_req *q = (wo_db_req *)(uintptr_t)e->payload;
|
||||
assert(vm->is_primary && "DB requests route to shard 0 only");
|
||||
wo_wal *dw = (wo_wal *)vm->rt.wal;
|
||||
size_t before = dw ? dw->len : 0;
|
||||
wo_db_exec_req(vm, q);
|
||||
q->done = 1;
|
||||
/* did this statement actually stage a record? Asking the buffer
|
||||
* beats guessing from the opcode, and the count is what the
|
||||
* failure diagnostic reports. */
|
||||
if (dw && dw->len > before) staged++;
|
||||
wo_envelope *re = calloc(1, sizeof *re);
|
||||
if (re) {
|
||||
re->kind = 4;
|
||||
re->payload = e->payload;
|
||||
inbox_push_to(q->from_shard, re);
|
||||
re->next = NULL;
|
||||
if (dw && dw->len > before) {
|
||||
/* This statement STAGED a record, so its reply is HELD:
|
||||
* pushing it now would unpark the requester before its
|
||||
* record is durable, which is the ack contract this
|
||||
* iteration exists to make literally true. FIFO, so the
|
||||
* first waiter is released first. */
|
||||
if (rtail) rtail->next = re; else rhead = re;
|
||||
rtail = re;
|
||||
} else {
|
||||
/* A READ (or any statement that staged nothing) has no
|
||||
* durability to wait for. Holding it too was measurably
|
||||
* wrong: it parked readers behind an fsync they had no
|
||||
* stake in, and durable.sN.mixread p99 rose ~4x
|
||||
* (1043 -> 4057us) until this branch existed. */
|
||||
inbox_push_to(q->from_shard, re);
|
||||
}
|
||||
} /* OOM: the requester stays parked until stop — leak, not UB */
|
||||
break;
|
||||
}
|
||||
|
|
@ -170,6 +220,41 @@ static int wo_vm_adopt(wo_vm *vm) {
|
|||
n++;
|
||||
e = nx;
|
||||
}
|
||||
/* databasev2 4: ONE barrier for everything this drain staged, then every
|
||||
* held reply. Each requester therefore unparks having been acknowledged
|
||||
* after the barrier that carried ITS record. Commit unconditionally when
|
||||
* anything is staged — the inline path relies on finding the buffer empty
|
||||
* (see db.c), so a drain must never leave a record behind. */
|
||||
if (staged) {
|
||||
wo_wal *cw = (wo_wal *)vm->rt.wal;
|
||||
if (cw) wo_wal_commit_fatal(cw, staged);
|
||||
}
|
||||
while (rhead) {
|
||||
wo_envelope *rn = rhead->next;
|
||||
wo_db_req *rq = (wo_db_req *)(uintptr_t)rhead->payload;
|
||||
rhead->next = NULL;
|
||||
inbox_push_to(rq->from_shard, rhead);
|
||||
rhead = rn;
|
||||
}
|
||||
/* databasev2 3: the ONE point where compaction is safe — the barrier above
|
||||
* just ran, so the staging buffer is empty. Anywhere else, a staged record
|
||||
* would be written into a file about to be replaced. This is a correctness
|
||||
* requirement, not a scheduling preference; wo_wal_compact also refuses a
|
||||
* non-empty buffer as a backstop.
|
||||
*
|
||||
* Replies are released FIRST, deliberately: their records are already
|
||||
* durable, and holding them across a stop-the-world rewrite would add the
|
||||
* rewrite's full duration to their latency for no benefit.
|
||||
*
|
||||
* The result is ignored because a failed compaction is a missed
|
||||
* optimisation, not a durability event — the original log is left intact
|
||||
* and the process carries on. */
|
||||
if (staged) {
|
||||
wo_wal *cw = (wo_wal *)vm->rt.wal;
|
||||
if (cw && wo_wal_should_compact(cw->off, cw->compacted_bytes,
|
||||
wo_wal_ckpt_floor, wo_wal_ckpt_ratio))
|
||||
(void)wo_wal_compact(cw, (wo_db *)vm->rt.db);
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
|
|
@ -425,6 +510,33 @@ static void *shard_main(void *arg) {
|
|||
} else {
|
||||
int rc = wo_io_wait(vm); /* parked fibers AND the wake eventfd */
|
||||
if (rc == WO_IO_STOP) {
|
||||
/* iteration 40 — THE DRAIN GUARANTEE. A message sent before
|
||||
* the stop flag is observed must be delivered and run before
|
||||
* the engine stops.
|
||||
*
|
||||
* NEXT_RUNNABLE() already states this contract for a worker
|
||||
* holding a live fiber: it returns 2 and keeps draining "so
|
||||
* queued shutdown messages (close frames!) still run". This
|
||||
* branch — the IDLE worker, empty run queue, waiting on the
|
||||
* plane — used to reap and break instead, abandoning whatever
|
||||
* sat in its inbox for wo_engine_stop() to free wholesale.
|
||||
*
|
||||
* An actor between messages is exactly that idle case, which
|
||||
* is why a WARM server hid the bug: warm shards had live
|
||||
* fibers and took the correct path. Measured 2026-08-27 on a
|
||||
* fresh server: 5 of 16 SIGTERM drains left a WebSocket
|
||||
* client at EOF with no close frame and no diagnostic.
|
||||
*
|
||||
* The window belongs to the PRIMARY and closes when it sets
|
||||
* eng_shutdown (after main returns), so honour it here and
|
||||
* only exit when the primary says so. Yield on an empty poll:
|
||||
* a tight loop would burn a core per shard and starve the very
|
||||
* actors the drain exists to let run. */
|
||||
if (!eng_shutdown) {
|
||||
(void)wo_vm_adopt(vm);
|
||||
if (!vm->qhead) sched_yield();
|
||||
continue;
|
||||
}
|
||||
fib_reap_all(vm);
|
||||
break;
|
||||
}
|
||||
|
|
@ -445,6 +557,85 @@ int wo_engine_primary_inbox(int wake_efd) {
|
|||
return 0;
|
||||
}
|
||||
|
||||
/* iteration 24 teardown phase 1 (single-threaded, BEFORE eng_teardown):
|
||||
* dismantle one vm's actor world with real drops — container backings are
|
||||
* malloc'd, so wholesale arena death does NOT cover them (LSan, chat's
|
||||
* registry map). Cross-shard payloads route home through wo_route_free
|
||||
* (still live here); the routed kind-2 envelopes are settled by the
|
||||
* caller's inbox passes. */
|
||||
static void vm_drop_actor_world(wo_vm *vm) {
|
||||
wo_actor *a = vm->actors;
|
||||
vm->actors = NULL;
|
||||
while (a) {
|
||||
wo_actor *nx = a->next_all;
|
||||
if (a->instance) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)a->instance);
|
||||
for (uint32_t i = 0; i < a->mlen; i++) {
|
||||
uint64_t m = a->msgs[(a->mhead + i) % a->mcap].payload;
|
||||
if (m) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)m);
|
||||
}
|
||||
wo_monitor *mo = a->monitors;
|
||||
while (mo) {
|
||||
wo_monitor *mnx = mo->next;
|
||||
if (mo->msg) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)mo->msg);
|
||||
free(mo);
|
||||
mo = mnx;
|
||||
}
|
||||
free(a->msgs);
|
||||
free(a);
|
||||
a = nx;
|
||||
}
|
||||
wo_timer *tt = vm->timers;
|
||||
vm->timers = NULL;
|
||||
while (tt) {
|
||||
wo_timer *tnx = tt->next;
|
||||
if (tt->msg) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)tt->msg);
|
||||
free(tt);
|
||||
tt = tnx;
|
||||
}
|
||||
}
|
||||
|
||||
/* Settle every inbox after phase 1: home-routed frees execute on their
|
||||
* owner vm; payload-carrying strays drop (possibly routing again — the
|
||||
* outer loop runs until everything is quiet). Node memory always freed. */
|
||||
static int eng_settle_inboxes(void) {
|
||||
int moved = 0;
|
||||
for (uint32_t i = 0; i < wo_eng.nshards && i < WO_ENG_MAX_SHARDS; i++) {
|
||||
if (!INBOX_READY[i]) continue;
|
||||
wo_vm *vm = &wo_eng.shards[i];
|
||||
wo_inbox *ib = &INBOX[i];
|
||||
wo_envelope *e = ib->head;
|
||||
ib->head = ib->tail = NULL;
|
||||
while (e) {
|
||||
wo_envelope *nx = e->next;
|
||||
switch (e->kind) {
|
||||
case 2: /* WE are home: the direct drop is the settlement */
|
||||
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)e->payload);
|
||||
break;
|
||||
case 0:
|
||||
case 5:
|
||||
case 7: /* in-flight payloads: drop (may route -> next pass) */
|
||||
if (e->payload)
|
||||
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)e->payload);
|
||||
break;
|
||||
case 1: /* an unadopted actor shell */
|
||||
if (e->actor) {
|
||||
if (e->actor->instance)
|
||||
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)e->actor->instance);
|
||||
free(e->actor->msgs);
|
||||
free(e->actor);
|
||||
}
|
||||
break;
|
||||
default: /* 3/4/6: scalar or engine-side payloads, node-only */
|
||||
break;
|
||||
}
|
||||
free(e);
|
||||
moved++;
|
||||
e = nx;
|
||||
}
|
||||
}
|
||||
return moved;
|
||||
}
|
||||
|
||||
int wo_engine_start(const wo_module *mod, size_t heap_cap, uint32_t nshards) {
|
||||
wo_eng.nshards = nshards;
|
||||
eng_heap_cap = heap_cap;
|
||||
|
|
@ -489,6 +680,13 @@ void wo_engine_stop(void) {
|
|||
(void)n;
|
||||
}
|
||||
for (uint32_t i = 1; i < wo_eng.nshards; i++) pthread_join(ts[i - 1], NULL);
|
||||
/* single-threaded from here: PHASE 1 — real drops while every arena
|
||||
* and the routing fabric are still alive (malloc'd container backings
|
||||
* inside actor state need them; iteration 24's registry map). Settle
|
||||
* passes run until routed frees stop appearing. */
|
||||
for (uint32_t i = 0; i < wo_eng.nshards && i < WO_ENG_MAX_SHARDS; i++)
|
||||
if (wo_eng.shards[i].rt.arena.base) vm_drop_actor_world(&wo_eng.shards[i]);
|
||||
while (eng_settle_inboxes() > 0) {}
|
||||
/* single-threaded from here. Every arena dies wholesale, so routed
|
||||
* frees and queued payloads need no per-object drops — DISCARD the
|
||||
* envelopes (freeing the malloc'd nodes/actors) and let the arenas
|
||||
|
|
@ -557,19 +755,42 @@ void wo_vm_destroy(wo_vm *vm) {
|
|||
free(fb);
|
||||
}
|
||||
/* actors first — dropping their state and queued messages needs the
|
||||
* runtime alive */
|
||||
* runtime alive. BUT: once the engine is in teardown, arenas die
|
||||
* WHOLESALE (the standing doctrine) — a moved-in message's home arena
|
||||
* may belong to an ALREADY-destroyed shard, and even reading its
|
||||
* header is a use-after-free (ASan, chat's drain). Structures are
|
||||
* still freed; payload drops are skipped. */
|
||||
int drops_ok = !eng_teardown;
|
||||
wo_actor *a = vm->actors;
|
||||
while (a) {
|
||||
wo_actor *nx = a->next_all;
|
||||
if (a->instance) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)a->instance);
|
||||
for (uint32_t i = 0; i < a->mlen; i++) {
|
||||
if (drops_ok && a->instance)
|
||||
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)a->instance);
|
||||
for (uint32_t i = 0; drops_ok && i < a->mlen; i++) {
|
||||
uint64_t m = a->msgs[(a->mhead + i) % a->mcap].payload;
|
||||
if (m) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)m);
|
||||
}
|
||||
wo_monitor *mo = a->monitors;
|
||||
while (mo) { /* undelivered notices are the runtime's to drop */
|
||||
wo_monitor *mnx = mo->next;
|
||||
if (drops_ok && mo->msg)
|
||||
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)mo->msg);
|
||||
free(mo);
|
||||
mo = mnx;
|
||||
}
|
||||
free(a->msgs);
|
||||
free(a);
|
||||
a = nx;
|
||||
}
|
||||
wo_timer *tt = vm->timers;
|
||||
vm->timers = NULL;
|
||||
while (tt) { /* unfired timers likewise */
|
||||
wo_timer *tnx = tt->next;
|
||||
if (drops_ok && tt->msg)
|
||||
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)tt->msg);
|
||||
free(tt);
|
||||
tt = tnx;
|
||||
}
|
||||
vm->actors = NULL;
|
||||
wo_io_destroy(vm);
|
||||
wo_rt_destroy(&vm->rt);
|
||||
|
|
@ -760,6 +981,7 @@ static void actor_die(wo_vm *vm, wo_actor *a, wo_fiber *delivery) {
|
|||
a->instance = 0;
|
||||
}
|
||||
a->active = NULL;
|
||||
monitors_fire(vm, a);
|
||||
}
|
||||
|
||||
/* Mailbox nonempty, no delivery fiber: start one on the next message.
|
||||
|
|
@ -877,6 +1099,57 @@ int wo_vm_actor_send(wo_vm *vm, uint64_t addr, uint64_t msg_val, const char **ms
|
|||
return 0;
|
||||
}
|
||||
|
||||
/* iteration 24 T4/T5: a RUNTIME-sourced delivery (death notice, timer).
|
||||
* No fiber to trap: a full or dead target drops the message with a
|
||||
* stderr line (spec'd disclosure), never silently. Runs on any thread —
|
||||
* cross-shard targets ride the ordinary kind-0 envelope. */
|
||||
static void runtime_notify(wo_vm *vm, wo_actor *target, uint64_t msg_val,
|
||||
const char *what) {
|
||||
if (!target || !msg_val) return;
|
||||
if (target->dead) {
|
||||
actor_drop_payload(vm, msg_val);
|
||||
return; /* send-to-dead: silent by contract */
|
||||
}
|
||||
if (wo_mbox_reserve(target) != 0) {
|
||||
fprintf(stderr, "wovm: %s dropped — the observer's mailbox is full\n", what);
|
||||
actor_drop_payload(vm, msg_val);
|
||||
return;
|
||||
}
|
||||
if (target->home != vm->shard_id) {
|
||||
wo_envelope *e = calloc(1, sizeof *e);
|
||||
if (!e) {
|
||||
wo_mbox_release(target);
|
||||
actor_drop_payload(vm, msg_val);
|
||||
return;
|
||||
}
|
||||
e->kind = 0;
|
||||
e->actor = target;
|
||||
e->payload = msg_val;
|
||||
inbox_push_to(target->home, e);
|
||||
return;
|
||||
}
|
||||
wo_msg m0 = { msg_val, NULL, 0 };
|
||||
if (actor_push(target, m0) != 0) {
|
||||
wo_mbox_release(target);
|
||||
actor_drop_payload(vm, msg_val);
|
||||
return;
|
||||
}
|
||||
if (!target->active) (void)actor_activate(vm, target);
|
||||
}
|
||||
|
||||
/* iteration 24 T4: the death walk — every registered observer gets its
|
||||
* chosen notice, then the list is gone (an actor dies once). */
|
||||
static void monitors_fire(wo_vm *vm, wo_actor *a) {
|
||||
wo_monitor *m = a->monitors;
|
||||
a->monitors = NULL;
|
||||
while (m) {
|
||||
wo_monitor *nx = m->next;
|
||||
runtime_notify(vm, m->observer, m->msg, "death notice");
|
||||
free(m);
|
||||
m = nx;
|
||||
}
|
||||
}
|
||||
|
||||
/* iteration 24: call — send that waits. First entry enqueues with the
|
||||
* caller attached and parks (WO_PARK_INBOX, the DB-RPC park); the resume
|
||||
* RE-EXECUTES this builtin and consumes the scalar reply. No hangs, ever:
|
||||
|
|
@ -946,6 +1219,105 @@ int wo_vm_actor_call(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
|||
return WO_SYS_PARKED;
|
||||
}
|
||||
|
||||
int wo_vm_actor_monitor(wo_vm *vm, uint64_t watched, uint64_t observer,
|
||||
uint64_t msg_val, const char **msg) {
|
||||
wo_actor *w = (wo_actor *)(uintptr_t)watched;
|
||||
wo_actor *o = (wo_actor *)(uintptr_t)observer;
|
||||
if (!w || !o) {
|
||||
*msg = "monitor: nil actor address";
|
||||
return WO_T_BOUNDS;
|
||||
}
|
||||
if (!msg_val) {
|
||||
*msg = "monitor: nil notice message";
|
||||
return WO_T_BOUNDS;
|
||||
}
|
||||
/* the registration belongs to the WATCHED actor's home thread */
|
||||
if (w->home != vm->shard_id) {
|
||||
wo_envelope *e = calloc(1, sizeof *e);
|
||||
if (!e) {
|
||||
actor_drop_payload(vm, msg_val);
|
||||
*msg = "out of memory";
|
||||
return WO_T_OOM;
|
||||
}
|
||||
e->kind = 7;
|
||||
e->actor = w;
|
||||
e->payload = msg_val;
|
||||
e->from_fiber = (wo_fiber *)o; /* reused slot: the observer */
|
||||
inbox_push_to(w->home, e);
|
||||
return 0;
|
||||
}
|
||||
if (w->dead) { /* monitoring the dead: the notice fires NOW */
|
||||
runtime_notify(vm, o, msg_val, "death notice");
|
||||
return 0;
|
||||
}
|
||||
wo_monitor *m = calloc(1, sizeof *m);
|
||||
if (!m) {
|
||||
actor_drop_payload(vm, msg_val);
|
||||
*msg = "out of memory";
|
||||
return WO_T_OOM;
|
||||
}
|
||||
m->observer = o;
|
||||
m->msg = msg_val;
|
||||
m->next = w->monitors;
|
||||
w->monitors = m;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int wo_vm_timer_after(wo_vm *vm, int64_t ms, uint64_t addr, uint64_t msg_val,
|
||||
const char **msg) {
|
||||
wo_actor *a = (wo_actor *)(uintptr_t)addr;
|
||||
if (!a) {
|
||||
*msg = "time.after: nil actor address";
|
||||
return WO_T_BOUNDS;
|
||||
}
|
||||
if (!msg_val) {
|
||||
*msg = "time.after: nil message";
|
||||
return WO_T_BOUNDS;
|
||||
}
|
||||
if (ms <= 0) { /* no wait to arm: deliver now */
|
||||
runtime_notify(vm, a, msg_val, "timer message");
|
||||
return 0;
|
||||
}
|
||||
wo_timer *t = calloc(1, sizeof *t);
|
||||
if (!t) {
|
||||
actor_drop_payload(vm, msg_val);
|
||||
*msg = "out of memory";
|
||||
return WO_T_OOM;
|
||||
}
|
||||
struct timespec now;
|
||||
clock_gettime(CLOCK_REALTIME, &now);
|
||||
t->at = (int64_t)now.tv_sec * 1000 + now.tv_nsec / 1000000 + ms;
|
||||
t->target = a;
|
||||
t->msg = msg_val;
|
||||
t->next = vm->timers;
|
||||
vm->timers = t;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int wo_vm_timers_fire(wo_vm *vm, int64_t now) {
|
||||
int fired = 0;
|
||||
wo_timer **pp = &vm->timers;
|
||||
while (*pp) {
|
||||
wo_timer *t = *pp;
|
||||
if (t->at <= now) {
|
||||
*pp = t->next;
|
||||
runtime_notify(vm, t->target, t->msg, "timer message");
|
||||
free(t);
|
||||
fired++;
|
||||
} else {
|
||||
pp = &t->next;
|
||||
}
|
||||
}
|
||||
return fired;
|
||||
}
|
||||
|
||||
int64_t wo_vm_timers_next(wo_vm *vm) {
|
||||
int64_t next = 0;
|
||||
for (wo_timer *t = vm->timers; t; t = t->next)
|
||||
if (next == 0 || t->at < next) next = t->at;
|
||||
return next;
|
||||
}
|
||||
|
||||
/* The drop-table entry governing instruction [pc]: the last one recorded
|
||||
* at or before it. NULL = nothing live there. */
|
||||
static const wo_dropent *vm_dropent(const wo_methodrec *me, uint32_t pc) {
|
||||
|
|
@ -1227,6 +1599,15 @@ static int vm_run(wo_vm *vm, uint64_t *ret, wo_err *err) {
|
|||
} \
|
||||
int iorc_ = wo_io_wait(vm); \
|
||||
if (iorc_ == WO_IO_STOP) { \
|
||||
/* iteration 24: a WORKER on stop keeps DRAINING — its \
|
||||
* serve loop spins adopting the inbox until the primary \
|
||||
* finishes the drain window and sets eng_shutdown, so \
|
||||
* queued shutdown messages (close frames!) still run. \
|
||||
* Only the PRIMARY's stop ends the program. */ \
|
||||
if (!vm->is_primary) { \
|
||||
vm->cur = &vm->f0; \
|
||||
return 2; \
|
||||
} \
|
||||
fib_reap_all(vm); \
|
||||
vm->cur = &vm->f0; \
|
||||
return 1; \
|
||||
|
|
@ -1729,8 +2110,31 @@ dispatch:
|
|||
vm->cur->frames[vm->cur->depth - 1].pc = pc - 1;
|
||||
vm->cur->ncatch = 0;
|
||||
vm_unwind(vm, 0);
|
||||
/* a stop ends the PROGRAM: every fiber — the stopped one,
|
||||
* queued ones, main wherever it is — unwinds clean */
|
||||
/* iteration 24 (the drain): a STOPPED wait on a NON-main fiber
|
||||
* unwinds that fiber ALONE — the rest of the program (main's
|
||||
* drain code, actors flushing close frames) keeps running.
|
||||
* Main's own STOPPED still ends the program, as ever. */
|
||||
if (vm->cur != &vm->f0) {
|
||||
wo_fiber *dead = vm->cur;
|
||||
if (dead->actor) {
|
||||
wo_actor *da = dead->actor;
|
||||
if (dead->cur_msg) {
|
||||
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)dead->cur_msg);
|
||||
dead->cur_msg = 0;
|
||||
}
|
||||
call_reply_to(vm, dead->msg_caller, dead->msg_caller_shard,
|
||||
0, WO_T_ACTOR);
|
||||
dead->msg_caller = NULL;
|
||||
da->active = NULL;
|
||||
}
|
||||
vm->nfibers--;
|
||||
fib_retire(vm, dead);
|
||||
NEXT_RUNNABLE();
|
||||
RELOAD();
|
||||
NEXT();
|
||||
}
|
||||
/* main: a stop ends the PROGRAM — every remaining fiber
|
||||
* unwinds clean */
|
||||
if (vm->cur != &vm->f0) {
|
||||
wo_fiber *dead = vm->cur;
|
||||
vm->cur = &vm->f0;
|
||||
|
|
|
|||
|
|
@ -120,6 +120,27 @@ typedef struct wo_msg {
|
|||
* guarantee). Death (iteration 24): a receive trapping uncaught marks
|
||||
* the actor dead — sends to it drop silently, calls trap, queued
|
||||
* callers are error-unparked; the state and mailbox are released. */
|
||||
/* iteration 24 T4: one death-notice registration. The runtime owns the
|
||||
* moved-in notice message until delivery (or drops it if the observer is
|
||||
* unreachable). The list lives on the WATCHED actor, owned by its home
|
||||
* thread. */
|
||||
typedef struct wo_monitor {
|
||||
struct wo_actor *observer;
|
||||
uint64_t msg;
|
||||
struct wo_monitor *next;
|
||||
} wo_monitor;
|
||||
|
||||
/* iteration 24 T5: one armed one-shot timer — fires as an ordinary
|
||||
* runtime send of the moved message when `at` passes. The list lives on
|
||||
* the ARMING fiber's shard and is scanned by the same deadline machinery
|
||||
* that serves fd-park deadlines. */
|
||||
typedef struct wo_timer {
|
||||
int64_t at; /* wall ms */
|
||||
struct wo_actor *target;
|
||||
uint64_t msg;
|
||||
struct wo_timer *next;
|
||||
} wo_timer;
|
||||
|
||||
typedef struct wo_actor {
|
||||
uint64_t instance; /* the moved-in state object (runtime-owned) */
|
||||
uint32_t method; /* receive's method index (self + msg = 2 args) */
|
||||
|
|
@ -134,6 +155,7 @@ typedef struct wo_actor {
|
|||
* overshoot by at most the number of in-flight sends — disclosed. */
|
||||
uint32_t pending;
|
||||
wo_fiber *active; /* the delivery fiber, NULL when idle */
|
||||
wo_monitor *monitors; /* iteration 24 T4: who wants the death notice */
|
||||
struct wo_actor *next_all; /* the vm's all-actors list */
|
||||
} wo_actor;
|
||||
|
||||
|
|
@ -176,6 +198,9 @@ typedef struct wo_vm {
|
|||
* freed memory is the UAF this prevents. Steady-state pool size = the
|
||||
* peak live fiber count; the pool dies with the vm. */
|
||||
wo_fiber *fib_pool;
|
||||
/* iteration 24 T5: this shard's armed timers (unsorted list — the
|
||||
* deadline scan is already linear; a wheel is measured-later work) */
|
||||
wo_timer *timers;
|
||||
/* iteration 35, uring backend: the shard's ONE deadline tick — a
|
||||
* TIMEOUT op with a sentinel user_data armed for the nearest fd-park
|
||||
* deadline (fd parks keep exactly one POLL op each; expiry wakes them
|
||||
|
|
@ -206,6 +231,20 @@ int wo_vm_actor_send(wo_vm *vm, uint64_t addr, uint64_t msg_val, const char **ms
|
|||
* caller attached and parks (WO_SYS_PARKED); the re-execution consumes the
|
||||
* scalar reply into R[A] (vm.c owns the protocol, builtin.c dispatches). */
|
||||
int wo_vm_actor_call(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg);
|
||||
/* iteration 24 T4: register a death notice — monitor(watched, observer,
|
||||
* msg). The msg MOVES to the runtime; an already-dead watched actor
|
||||
* delivers it immediately. */
|
||||
int wo_vm_actor_monitor(wo_vm *vm, uint64_t watched, uint64_t observer,
|
||||
uint64_t msg_val, const char **msg);
|
||||
/* iteration 24 T5: arm a one-shot timer on THIS shard — time.after(ms,
|
||||
* addr, msg). ms <= 0 delivers now. */
|
||||
int wo_vm_timer_after(wo_vm *vm, int64_t ms, uint64_t addr, uint64_t msg_val,
|
||||
const char **msg);
|
||||
/* iteration 24 T5: fire every timer at or past `now` (park.c's deadline
|
||||
* machinery calls this beside the fd-park sweep). Returns fired count. */
|
||||
int wo_vm_timers_fire(wo_vm *vm, int64_t now);
|
||||
/* The nearest armed timer's deadline, 0 = none (park.c's tick/timeout). */
|
||||
int64_t wo_vm_timers_next(wo_vm *vm);
|
||||
|
||||
/* ---- the shard engine (arc stage 2) ------------------------------------
|
||||
* One pinned thread per shard, each a full wo_vm (own arena, GC, I/O
|
||||
|
|
@ -231,7 +270,11 @@ typedef struct wo_envelope {
|
|||
* from_shard/from_fiber = the parked caller),
|
||||
* 6 = CALL_REPLY (payload = the SCALAR reply, from_fiber =
|
||||
* the caller to unpark; status 0 = ok, WO_T_ACTOR =
|
||||
* the callee was/went dead — the caller traps) */
|
||||
* the callee was/went dead — the caller traps),
|
||||
* 7 = MONITOR (iteration 24 T4: actor = the WATCHED one,
|
||||
* from_fiber REUSED as the observer wo_actor*, payload =
|
||||
* the moved notice — registered on the watched actor's
|
||||
* home thread; already-dead delivers the notice now) */
|
||||
struct wo_actor *actor;
|
||||
uint64_t payload;
|
||||
uint32_t from_shard;
|
||||
|
|
|
|||
|
|
@ -478,8 +478,17 @@ enum {
|
|||
* return value arrives. R is a SCALAR (v1,
|
||||
* compiler-enforced WO-E226). Dead callee =
|
||||
* WO_T_ACTOR, immediately or mid-call. */
|
||||
/* ids 89 (monitor) and 90 (time.after) are RESERVED for the rest of
|
||||
* the lifecycle slice — do not reuse. */
|
||||
WO_B_MONITOR = 89, /* (watched, observer, msg) -> (): the
|
||||
* observer's own M-typed msg is delivered
|
||||
* when watched dies (trap-death); already
|
||||
* dead delivers NOW; msg MOVES. A full
|
||||
* observer's notice is dropped with a
|
||||
* stderr line (no fiber to trap). */
|
||||
WO_B_TIME_AFTER = 90, /* (ms, addr, msg) -> (): one-shot timer —
|
||||
* msg (MOVED) arrives as an ordinary send
|
||||
* after ms; no cancel (the generation-
|
||||
* counter idiom is the documented answer);
|
||||
* ms <= 0 delivers now. */
|
||||
/* ---- iteration 35: net seams (sysio.c). Deadlines are per-CALL (no
|
||||
* hidden fd state); a timeout is an EXPECTED outcome, so it answers
|
||||
* nil/false, never a trap. ms <= 0 = no deadline (the old behavior,
|
||||
|
|
|
|||
|
|
@ -117,6 +117,397 @@ static void test_roundtrip_replay(void) {
|
|||
wo_rt_destroy(&rt);
|
||||
}
|
||||
|
||||
/* databasev2 4 part A, Task 1: a failed barrier must be DETECTED, and the
|
||||
* caller must be able to tell WHICH operation failed — a pwrite failure and
|
||||
* an fdatasync failure are different operational problems and the diagnostic
|
||||
* has to name the right one. This proves detection only; the fatal exit that
|
||||
* follows it cannot be exercised in-process. */
|
||||
static void test_commit_failure_detected(void) {
|
||||
char path[128];
|
||||
snprintf(path, sizeof path, "%s/commitfail.wal", g_dir);
|
||||
wo_rt rt;
|
||||
T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0);
|
||||
wo_db db;
|
||||
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
|
||||
wo_wal w;
|
||||
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
|
||||
const char *msg = "";
|
||||
|
||||
/* the WAL remembers where it lives — the abort diagnostic is worthless
|
||||
* without it */
|
||||
T_CHECK(w.path != NULL && strstr(w.path, "commitfail.wal") != NULL);
|
||||
|
||||
wo_str *s = wo_str_new(&rt, "abc", 3);
|
||||
uint64_t vals[2] = {7, (uint64_t)(uintptr_t)s};
|
||||
uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL);
|
||||
T_CHECK(id != 0);
|
||||
T_EQ(wo_wal_append_insert(&w, &db, 0, id), 0);
|
||||
T_CHECK(w.len > 0); /* something really is staged */
|
||||
|
||||
/* an unusable descriptor: pwrite reports EBADF. -1 is used rather than
|
||||
* closing the real fd so the close below cannot double-free it. */
|
||||
int real = w.fd;
|
||||
w.fd = -1;
|
||||
T_EQ(wo_wal_commit(&w), WO_WAL_ERR_WRITE);
|
||||
T_CHECK(w.len > 0); /* a failed commit consumes nothing */
|
||||
w.fd = real;
|
||||
|
||||
wo_wal_close(&w);
|
||||
wo_db_destroy(&db);
|
||||
wo_rt_destroy(&rt);
|
||||
}
|
||||
|
||||
/* databasev2 3 Task 1: compaction rewrites the log as one record per LIVE row.
|
||||
* Asserts BOTH halves on purpose: "the file got shorter" is also true of a
|
||||
* truncating bug, so the replay comparison is what actually proves it. */
|
||||
static void test_compact_shortens_and_replays_equal(void) {
|
||||
char path[128];
|
||||
snprintf(path, sizeof path, "%s/compact.wal", g_dir);
|
||||
wo_rt rt;
|
||||
T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0);
|
||||
wo_db db;
|
||||
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
|
||||
wo_wal w;
|
||||
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
|
||||
const char *msg = "";
|
||||
|
||||
uint64_t ids[3];
|
||||
for (int i = 0; i < 3; i++) {
|
||||
wo_str *s = wo_str_new(&rt, "abc", 3);
|
||||
uint64_t vals[2] = {(uint64_t)(i * 10), (uint64_t)(uintptr_t)s};
|
||||
ids[i] = wo_row_insert(&db, 0, vals, &msg, NULL);
|
||||
T_CHECK(ids[i] != 0);
|
||||
T_EQ(wo_wal_append_insert(&w, &db, 0, ids[i]), 0);
|
||||
T_EQ(wo_wal_commit(&w), 0);
|
||||
}
|
||||
/* age it: the SAME row updated repeatedly, so HISTORY grows while the live
|
||||
* set does not — the exact case checkpoint exists for */
|
||||
for (int k = 0; k < 40; k++) {
|
||||
int ek = 0;
|
||||
T_EQ(wo_row_update_field(&db, 0, ids[0], 0, (uint64_t)(500 + k), &msg, &ek), 0);
|
||||
T_EQ(wo_wal_append_update(&w, &db, 0, ids[0]), 0);
|
||||
T_EQ(wo_wal_commit(&w), 0);
|
||||
}
|
||||
uint64_t before_bytes = 0;
|
||||
int64_t before_recs = wo_wal_check(path, &before_bytes);
|
||||
T_CHECK(before_recs == 43); /* 3 inserts + 40 updates, all history */
|
||||
|
||||
T_EQ(wo_wal_compact(&w, &db), 0);
|
||||
|
||||
uint64_t after_bytes = 0;
|
||||
int64_t after_recs = wo_wal_check(path, &after_bytes);
|
||||
T_CHECK(after_recs == 3); /* one record per LIVE row */
|
||||
T_CHECK(after_bytes < before_bytes); /* and the file really shrank */
|
||||
|
||||
/* the WAL stays usable: the descriptor was reopened and the offset reset,
|
||||
* so a further write must land AFTER the compacted records, not over them */
|
||||
wo_str *s4 = wo_str_new(&rt, "xyz", 3);
|
||||
uint64_t v4[2] = {99, (uint64_t)(uintptr_t)s4};
|
||||
uint64_t id4 = wo_row_insert(&db, 0, v4, &msg, NULL);
|
||||
T_CHECK(id4 != 0);
|
||||
T_EQ(wo_wal_append_insert(&w, &db, 0, id4), 0);
|
||||
T_EQ(wo_wal_commit(&w), 0);
|
||||
T_CHECK(wo_wal_check(path, NULL) == 4);
|
||||
wo_wal_close(&w);
|
||||
|
||||
/* the proof: a FRESH store replayed from the compacted log must hold the
|
||||
* same rows, the same ids, and the LAST value each row had */
|
||||
wo_db db2;
|
||||
T_EQ(wo_db_init(&db2, CLASSES, 1, 0, 1), 0);
|
||||
T_EQ(wo_wal_replay(path, &db2), 4);
|
||||
uint64_t out[2];
|
||||
T_EQ(wo_row_read(&db2, &rt, 0, ids[0], out, &msg), 0);
|
||||
T_CHECK(out[0] == 539); /* the 40th update won, not the original 0 */
|
||||
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
|
||||
T_EQ(wo_row_read(&db2, &rt, 0, ids[1], out, &msg), 0);
|
||||
T_CHECK(out[0] == 10);
|
||||
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
|
||||
T_EQ(wo_row_read(&db2, &rt, 0, ids[2], out, &msg), 0);
|
||||
T_CHECK(out[0] == 20);
|
||||
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
|
||||
T_EQ(wo_row_read(&db2, &rt, 0, id4, out, &msg), 0);
|
||||
T_CHECK(out[0] == 99);
|
||||
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
|
||||
|
||||
wo_db_destroy(&db2);
|
||||
wo_db_destroy(&db);
|
||||
wo_rt_destroy(&rt);
|
||||
}
|
||||
|
||||
/* databasev2 3 Task 2: a stale temp file is the one input that could be
|
||||
* mistaken for data — a crash before the rename leaves one behind, full of
|
||||
* well-formed records that are NOT yet authoritative. So the fixture uses
|
||||
* plausible records (a byte copy of a real log), not garbage: garbage would be
|
||||
* rejected by the CRC anyway and would prove nothing. */
|
||||
static void test_stale_compact_temp_is_removed(void) {
|
||||
char path[128], tmp[160];
|
||||
snprintf(path, sizeof path, "%s/stale.wal", g_dir);
|
||||
snprintf(tmp, sizeof tmp, "%s%s", path, WO_WAL_TMP_SUFFIX);
|
||||
wo_rt rt;
|
||||
T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0);
|
||||
wo_db db;
|
||||
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
|
||||
wo_wal w;
|
||||
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
|
||||
const char *msg = "";
|
||||
|
||||
/* two live rows in the REAL log */
|
||||
uint64_t ids[2];
|
||||
for (int i = 0; i < 2; i++) {
|
||||
wo_str *s = wo_str_new(&rt, "abc", 3);
|
||||
uint64_t vals[2] = {(uint64_t)(i + 1), (uint64_t)(uintptr_t)s};
|
||||
ids[i] = wo_row_insert(&db, 0, vals, &msg, NULL);
|
||||
T_EQ(wo_wal_append_insert(&w, &db, 0, ids[i]), 0);
|
||||
T_EQ(wo_wal_commit(&w), 0);
|
||||
}
|
||||
wo_wal_close(&w);
|
||||
|
||||
/* forge a plausible stale temp: a byte copy of the real log */
|
||||
{
|
||||
int src = open(path, O_RDONLY);
|
||||
int dst = open(tmp, O_WRONLY | O_CREAT | O_TRUNC, 0644);
|
||||
T_CHECK(src >= 0 && dst >= 0);
|
||||
char buf[8192];
|
||||
ssize_t n;
|
||||
while ((n = read(src, buf, sizeof buf)) > 0) T_CHECK(write(dst, buf, (size_t)n) == n);
|
||||
close(src);
|
||||
close(dst);
|
||||
T_EQ(access(tmp, F_OK), 0); /* it really is there before we open */
|
||||
}
|
||||
|
||||
wo_wal w2;
|
||||
T_EQ(wo_wal_open(&w2, path, 1 << 16), 0);
|
||||
T_CHECK(access(tmp, F_OK) != 0); /* gone, and never consulted */
|
||||
wo_wal_close(&w2);
|
||||
|
||||
/* and the live log still says exactly what it said */
|
||||
wo_db db2;
|
||||
T_EQ(wo_db_init(&db2, CLASSES, 1, 0, 1), 0);
|
||||
T_EQ(wo_wal_replay(path, &db2), 2);
|
||||
uint64_t out[2];
|
||||
T_EQ(wo_row_read(&db2, &rt, 0, ids[0], out, &msg), 0);
|
||||
T_CHECK(out[0] == 1);
|
||||
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
|
||||
T_EQ(wo_row_read(&db2, &rt, 0, ids[1], out, &msg), 0);
|
||||
T_CHECK(out[0] == 2);
|
||||
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
|
||||
|
||||
wo_db_destroy(&db2);
|
||||
wo_db_destroy(&db);
|
||||
wo_rt_destroy(&rt);
|
||||
}
|
||||
|
||||
/* databasev2 3 Task 3: the trigger, tested as a pure decision. Kept pure
|
||||
* precisely so it CAN be tested — a policy only observable by writing megabytes
|
||||
* and waiting is a policy nobody checks. */
|
||||
static void test_should_compact_policy(void) {
|
||||
/* below the floor, nothing fires however bad the ratio looks */
|
||||
T_EQ(wo_wal_should_compact(1000, 10, 4096, 3), 0);
|
||||
T_EQ(wo_wal_should_compact(4095, 1, 4096, 3), 0);
|
||||
/* past the floor with no prior compaction: run once to learn the size */
|
||||
T_EQ(wo_wal_should_compact(4096, 0, 4096, 3), 1);
|
||||
/* with a known denominator it is a straight ratio test */
|
||||
T_EQ(wo_wal_should_compact(30000, 10000, 4096, 3), 0); /* exactly 3x is not MORE than 3x */
|
||||
T_EQ(wo_wal_should_compact(30001, 10000, 4096, 3), 1);
|
||||
T_EQ(wo_wal_should_compact(19999, 10000, 4096, 2), 0);
|
||||
T_EQ(wo_wal_should_compact(20001, 10000, 4096, 2), 1);
|
||||
/* a zero ratio disables the policy rather than dividing by nothing */
|
||||
T_EQ(wo_wal_should_compact(1u << 30, 10, 4096, 0), 0);
|
||||
}
|
||||
|
||||
/* databasev2 3 Task 3: the ordering rule, asserted rather than trusted.
|
||||
* Compaction with records staged would write them into a file about to be
|
||||
* replaced, so it must be REFUSED — and refused without touching the log. */
|
||||
static void test_compact_refuses_with_staged_records(void) {
|
||||
char path[128];
|
||||
snprintf(path, sizeof path, "%s/staged.wal", g_dir);
|
||||
wo_rt rt;
|
||||
T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0);
|
||||
wo_db db;
|
||||
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
|
||||
wo_wal w;
|
||||
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
|
||||
const char *msg = "";
|
||||
|
||||
wo_str *s1 = wo_str_new(&rt, "abc", 3);
|
||||
uint64_t v1[2] = {7, (uint64_t)(uintptr_t)s1};
|
||||
uint64_t id1 = wo_row_insert(&db, 0, v1, &msg, NULL);
|
||||
T_EQ(wo_wal_append_insert(&w, &db, 0, id1), 0);
|
||||
T_EQ(wo_wal_commit(&w), 0); /* durable, buffer empty */
|
||||
|
||||
/* now stage WITHOUT committing */
|
||||
wo_str *s2 = wo_str_new(&rt, "xyz", 3);
|
||||
uint64_t v2[2] = {8, (uint64_t)(uintptr_t)s2};
|
||||
uint64_t id2 = wo_row_insert(&db, 0, v2, &msg, NULL);
|
||||
T_EQ(wo_wal_append_insert(&w, &db, 0, id2), 0);
|
||||
T_CHECK(w.len > 0);
|
||||
|
||||
uint64_t before = 0;
|
||||
int64_t recs = wo_wal_check(path, &before);
|
||||
T_EQ(wo_wal_compact(&w, &db), -1); /* refused */
|
||||
T_CHECK(w.len > 0); /* and the staged record is still there */
|
||||
uint64_t after = 0;
|
||||
T_CHECK(wo_wal_check(path, &after) == recs && after == before); /* log untouched */
|
||||
|
||||
/* the staged record still commits normally afterwards */
|
||||
T_EQ(wo_wal_commit(&w), 0);
|
||||
T_CHECK(wo_wal_check(path, NULL) == recs + 1);
|
||||
|
||||
wo_wal_close(&w);
|
||||
wo_db_destroy(&db);
|
||||
wo_rt_destroy(&rt);
|
||||
}
|
||||
|
||||
/* databasev2 3 Task 4: kill -9 DURING compaction.
|
||||
*
|
||||
* The existing battery is insert-only, so its "records >= acks" oracle is
|
||||
* exactly what compaction is allowed to break: collapsing history is the point.
|
||||
* The invariant that survives is the ACKED LIVE SET — every id acked as
|
||||
* inserted and not later acked as deleted must be present with its acked value,
|
||||
* and every id acked as deleted must be absent. Both the pre-compaction and the
|
||||
* post-compaction log satisfy that identically, which is precisely the
|
||||
* "never a mixture" property the design is shaped around.
|
||||
*
|
||||
* The child deletes as it goes so HISTORY accumulates while the live set stays
|
||||
* small — without that, compaction would have nothing to collapse and the test
|
||||
* would prove nothing. */
|
||||
#define CK_DELETED UINT64_MAX
|
||||
|
||||
static void ck_ack(int fd, uint64_t id, uint64_t val) {
|
||||
uint64_t rec[2] = {id, val};
|
||||
if (write(fd, rec, sizeof rec) != (ssize_t)sizeof rec) _exit(0); /* parent gone */
|
||||
}
|
||||
|
||||
static void compact_battery_child(const char *path, int ack_fd) {
|
||||
wo_rt rt;
|
||||
wo_db db;
|
||||
wo_wal w;
|
||||
if (wo_rt_init(&rt, 1 << 20, CLASSES, 1) != 0) _exit(9);
|
||||
if (wo_db_init(&db, CLASSES, 1, 0, 1) != 0) _exit(9);
|
||||
if (wo_wal_open(&w, path, 1 << 20) != 0) _exit(9);
|
||||
const char *msg = "";
|
||||
uint64_t live[512];
|
||||
size_t nlive = 0;
|
||||
for (uint64_t i = 0;; i++) {
|
||||
uint64_t val = i * 7 + 3;
|
||||
wo_str *s = wo_str_new(&rt, "r", 1);
|
||||
uint64_t vals[2] = {val, (uint64_t)(uintptr_t)s};
|
||||
uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL);
|
||||
wo_str_free(&rt, s);
|
||||
if (!id) _exit(9);
|
||||
if (wo_wal_append_insert(&w, &db, 0, id) != 0) _exit(9);
|
||||
if (wo_wal_commit(&w) != 0) _exit(9); /* durable BEFORE the ack */
|
||||
ck_ack(ack_fd, id, val);
|
||||
if (nlive < 512) live[nlive++] = id;
|
||||
|
||||
/* drop the oldest so history grows while the live set does not */
|
||||
if (nlive > 16) {
|
||||
uint64_t victim = live[0];
|
||||
memmove(live, live + 1, (nlive - 1) * sizeof live[0]);
|
||||
nlive--;
|
||||
/* INTENT FIRST, deliberately. An ack after the commit would race:
|
||||
* a kill between them leaves the row legitimately gone on disk
|
||||
* while the last ack still says "inserted", and the parent would
|
||||
* demand a row the engine was right to remove. Announcing intent
|
||||
* makes the row's fate simply UNKNOWN to the parent, which is the
|
||||
* honest thing to assert about it. */
|
||||
ck_ack(ack_fd, victim, CK_DELETED);
|
||||
if (wo_row_remove(&db, 0, victim) != 0) _exit(9);
|
||||
if (wo_wal_append_remove(&w, 0, victim) != 0) _exit(9);
|
||||
if (wo_wal_commit(&w) != 0) _exit(9);
|
||||
}
|
||||
/* compact often, so a kill has a real chance of landing inside one */
|
||||
if (i % 24 == 23) (void)wo_wal_compact(&w, &db);
|
||||
}
|
||||
}
|
||||
|
||||
static void test_compact_crash_battery(void) {
|
||||
int rounds = 40; /* it is a RACE: one green run proves very little */
|
||||
for (int round = 0; round < rounds; round++) {
|
||||
char path[128], tmp[160];
|
||||
snprintf(path, sizeof path, "%s/ckcrash-%d.wal", g_dir, round);
|
||||
snprintf(tmp, sizeof tmp, "%s%s", path, WO_WAL_TMP_SUFFIX);
|
||||
int pipefd[2];
|
||||
T_EQ(pipe(pipefd), 0);
|
||||
pid_t pid = fork();
|
||||
T_CHECK(pid >= 0);
|
||||
if (pid == 0) {
|
||||
close(pipefd[0]);
|
||||
compact_battery_child(path, pipefd[1]);
|
||||
_exit(0);
|
||||
}
|
||||
close(pipefd[1]);
|
||||
/* vary the instant so kills land before, inside and after rewrites */
|
||||
struct timespec ts = {0, (7 + round * 3) * 1000000L};
|
||||
while (nanosleep(&ts, &ts) != 0) {}
|
||||
kill(pid, SIGKILL);
|
||||
int status;
|
||||
waitpid(pid, &status, 0);
|
||||
|
||||
/* replay the acks into the expected live set, in order */
|
||||
uint64_t ids[65536], vals[65536];
|
||||
size_t n = 0;
|
||||
for (;;) {
|
||||
uint64_t rec[2];
|
||||
ssize_t r = read(pipefd[0], rec, sizeof rec);
|
||||
if (r != (ssize_t)sizeof rec) break;
|
||||
if (n < 65536) { ids[n] = rec[0]; vals[n] = rec[1]; n++; }
|
||||
}
|
||||
close(pipefd[0]);
|
||||
T_CHECK(n > 0); /* the child got at least one commit out */
|
||||
|
||||
wo_rt rt;
|
||||
T_EQ(wo_rt_init(&rt, 1 << 22, CLASSES, 1), 0);
|
||||
wo_db db;
|
||||
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
|
||||
int64_t ck_recs = wo_wal_check(path, NULL);
|
||||
int64_t ck_applied = wo_wal_replay(path, &db);
|
||||
T_CHECK(ck_applied >= 0); /* never reported as corruption */
|
||||
|
||||
/* A stale temp may well EXIST after a kill inside compaction — that is
|
||||
* the expected debris. The guarantee is that the next OPEN removes it
|
||||
* and never reads it, so that is what gets asserted here; checking
|
||||
* merely for its absence after a replay would be asserting something
|
||||
* the design never promised (wo_wal_replay does not open the WAL). */
|
||||
{
|
||||
wo_wal probe;
|
||||
T_EQ(wo_wal_open(&probe, path, 1 << 20), 0);
|
||||
T_CHECK(access(tmp, F_OK) != 0);
|
||||
wo_wal_close(&probe);
|
||||
}
|
||||
|
||||
const char *msg = "";
|
||||
int bad = 0, checked = 0;
|
||||
for (size_t k = 0; k < n && !bad; k++) {
|
||||
if (vals[k] == CK_DELETED) continue; /* intent: fate is unknown */
|
||||
/* an id ever announced for deletion may legally be gone */
|
||||
int doomed = 0;
|
||||
for (size_t j = 0; j < n; j++)
|
||||
if (ids[j] == ids[k] && vals[j] == CK_DELETED) { doomed = 1; break; }
|
||||
if (doomed) continue;
|
||||
uint64_t out[2];
|
||||
int rc = wo_row_read(&db, &rt, 0, ids[k], out, &msg);
|
||||
if (0) {
|
||||
} else if (rc != 0 || out[0] != vals[k]) {
|
||||
bad = 1; /* an acked insert is missing or wrong */
|
||||
fprintf(stderr, "CKDIAG round=%d id=%llu rc=%d got=%llu want=%llu ack#%zu/%zu "
|
||||
"log_records=%lld replay_applied=%lld\n",
|
||||
round, (unsigned long long)ids[k], rc,
|
||||
rc == 0 ? (unsigned long long)out[0] : 0ull,
|
||||
(unsigned long long)vals[k], k, n,
|
||||
(long long)ck_recs, (long long)ck_applied);
|
||||
} else {
|
||||
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
|
||||
}
|
||||
checked++;
|
||||
}
|
||||
T_CHECK(checked > 0);
|
||||
T_CHECK(!bad);
|
||||
wo_db_destroy(&db);
|
||||
wo_rt_destroy(&rt);
|
||||
}
|
||||
}
|
||||
|
||||
static void test_torn_tail(void) {
|
||||
char path[128];
|
||||
snprintf(path, sizeof path, "%s/torn.wal", g_dir);
|
||||
|
|
@ -534,12 +925,18 @@ int main(void) {
|
|||
snprintf(g_dir, sizeof g_dir, "/tmp/wo-wal-test-XXXXXX");
|
||||
if (!mkdtemp(g_dir)) return 1;
|
||||
test_roundtrip_replay();
|
||||
test_commit_failure_detected();
|
||||
test_compact_shortens_and_replays_equal();
|
||||
test_stale_compact_temp_is_removed();
|
||||
test_should_compact_policy();
|
||||
test_compact_refuses_with_staged_records();
|
||||
test_torn_tail();
|
||||
test_float_bytes_replay();
|
||||
test_offset_capture();
|
||||
test_offset_after_failed_commit();
|
||||
test_read_row_at();
|
||||
test_crash_battery();
|
||||
test_compact_crash_battery();
|
||||
/* leave the dir for a failed run's forensics only */
|
||||
if (!t_fail) {
|
||||
char cmd[128];
|
||||
|
|
|
|||
433
scripts/chat-accept.sh
Executable file
433
scripts/chat-accept.sh
Executable file
|
|
@ -0,0 +1,433 @@
|
|||
#!/usr/bin/env bash
|
||||
# scripts/chat-accept.sh — iteration 24's gate. The chat sample serves
|
||||
# WebSocket rooms through the framework ([deps], file:// remote); a raw
|
||||
# RFC 6455 python client (stdlib only, INDEPENDENT accept-key check)
|
||||
# proves: the handshake, broadcast + presence + isolation across rooms,
|
||||
# the 1k-clients-one-hot-room soak (fds/RSS accounted), and the SIGTERM
|
||||
# drain (close frames, exit 0) — functional legs on BOTH WO_IO backends
|
||||
# plus an ASan run.
|
||||
set -uo pipefail
|
||||
|
||||
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
WOC="$ROOT/compiler/_build/default/bin/woc"
|
||||
WOVM="$ROOT/runtime/wovm"
|
||||
ASAN="$ROOT/runtime/build/wovm_asan"
|
||||
PORT0="${CHAT_PORT:-18901}"
|
||||
PORT="$PORT0"
|
||||
SOAK_N="${CHAT_SOAK:-1000}"
|
||||
|
||||
pass=0; fail=0
|
||||
ok() { echo "ok $1"; pass=$((pass + 1)); }
|
||||
bad() { echo "FAIL $1 -- $2"; fail=$((fail + 1)); }
|
||||
|
||||
if [[ ! -x "$WOC" || ! -x "$WOVM" ]]; then
|
||||
echo "chat-accept: build woc and wovm first" >&2; exit 1
|
||||
fi
|
||||
ulimit -n 8192 2>/dev/null || true
|
||||
|
||||
W="$(mktemp -d "${TMPDIR:-/tmp}/chat-accept.XXXXXX")"
|
||||
SRV=""
|
||||
# The example's server log lives at a STABLE path so a developer can
|
||||
# `tail -F /tmp/chat.log` while this runs. It used to go to the per-run temp
|
||||
# dir, which cleanup() deletes on exit — so there was nothing left to read and
|
||||
# nothing to follow live. Truncated once here, then APPENDED by every leg with
|
||||
# a banner, so one file holds the whole run in order.
|
||||
SRVLOG="/tmp/chat.log"
|
||||
: > "$SRVLOG"
|
||||
LEG=0
|
||||
LEGFROM=1
|
||||
echo "server log: $SRVLOG (tail -F \"$SRVLOG\" to follow)"
|
||||
|
||||
cleanup() {
|
||||
# kill EVERY server this run started, not merely the most recent $SRV: a leg
|
||||
# that dies before clearing SRV used to orphan a listener, which then broke
|
||||
# the next run on the same port. $W is unique per run, so matching on it
|
||||
# cannot touch another run's processes.
|
||||
[[ -n "$SRV" ]] && kill -9 "$SRV" 2>/dev/null
|
||||
pkill -9 -f "$W/app/target/chat" 2>/dev/null
|
||||
rm -rf "$W"
|
||||
}
|
||||
trap cleanup EXIT
|
||||
|
||||
cp -r "$ROOT/docs/examples/porch" "$W/fw"
|
||||
git -C "$W/fw" init -q && git -C "$W/fw" add -A
|
||||
git -C "$W/fw" -c user.email=t@t -c user.name=t commit -qm v01 && git -C "$W/fw" tag v0.1.0
|
||||
cp -r "$ROOT/docs/examples/chat" "$W/app"
|
||||
sed -i "s|https://github.com/shoneyj/porch|file://$W/fw|" "$W/app/wo.toml"
|
||||
printf '[build]\nruntime = "%s"\n' "$WOVM" >> "$W/app/wo.toml"
|
||||
|
||||
if "$WOC" "$W/app" >"$W/build.out" 2>&1 && [[ -x "$W/app/target/chat" ]]; then
|
||||
ok "deps chain + build"
|
||||
else
|
||||
bad "build" "$(grep -m1 error "$W/build.out" || head -1 "$W/build.out")"
|
||||
echo "chat-accept: 1 checks, 1 failures"; exit 1
|
||||
fi
|
||||
|
||||
# the raw client, shared by every leg
|
||||
CLIENT="$W/wsc.py"
|
||||
cat > "$CLIENT" <<'PYEOF'
|
||||
import socket, base64, hashlib, os, time
|
||||
GUID = "258EAFA5-E914-47DA-95CA-C5AB0DC85B11"
|
||||
BUF = {}
|
||||
def connect(port, room, name, timeout=8, rcvbuf=None):
|
||||
# rcvbuf: shrink THIS client's receive buffer so the server's socket fills
|
||||
# quickly — how the WO_MAILBOX leg manufactures a genuinely slow member
|
||||
# without sleeping. Must be set before connect() to take effect.
|
||||
if rcvbuf is None:
|
||||
s = socket.create_connection(("127.0.0.1", port), timeout=timeout)
|
||||
else:
|
||||
s = socket.socket()
|
||||
s.setsockopt(socket.SOL_SOCKET, socket.SO_RCVBUF, rcvbuf)
|
||||
s.settimeout(timeout)
|
||||
s.connect(("127.0.0.1", port))
|
||||
key = base64.b64encode(os.urandom(16)).decode()
|
||||
s.sendall((f"GET /ws?room={room}&name={name} HTTP/1.1\r\nhost: a\r\n"
|
||||
f"upgrade: websocket\r\nconnection: Upgrade\r\n"
|
||||
f"sec-websocket-key: {key}\r\nsec-websocket-version: 13\r\n\r\n").encode())
|
||||
d = b""
|
||||
while b"\r\n\r\n" not in d: d += s.recv(2000)
|
||||
head, _, rest = d.partition(b"\r\n\r\n")
|
||||
BUF[s] = rest # a frame may already ride the same segment
|
||||
head = head.decode()
|
||||
assert " 101 " in head.splitlines()[0], head.splitlines()[0]
|
||||
want = base64.b64encode(hashlib.sha1((key + GUID).encode()).digest()).decode()
|
||||
assert want in head, "accept-key mismatch (independent check)"
|
||||
return s
|
||||
def _take(s, n, timeout):
|
||||
s.settimeout(timeout)
|
||||
b = BUF.get(s, b"")
|
||||
while len(b) < n:
|
||||
c = s.recv(4096)
|
||||
if not c:
|
||||
BUF[s] = b
|
||||
return None
|
||||
b += c
|
||||
BUF[s] = b[n:]
|
||||
return b[:n]
|
||||
def send(s, text):
|
||||
p = text.encode(); mask = os.urandom(4)
|
||||
if len(p) < 126: hdr = bytes([0x81, 0x80 | len(p)])
|
||||
else: hdr = bytes([0x81, 0x80 | 126, len(p) >> 8, len(p) & 255])
|
||||
s.sendall(hdr + mask + bytes(b ^ mask[i % 4] for i, b in enumerate(p)))
|
||||
def recv(s, timeout=5):
|
||||
h = _take(s, 2, timeout)
|
||||
if h is None: return (-2, "") # EOF
|
||||
b0, b1 = h[0], h[1]
|
||||
ln = b1 & 0x7F
|
||||
if ln == 126:
|
||||
e = _take(s, 2, timeout); ln = (e[0] << 8) | e[1]
|
||||
d = _take(s, ln, timeout) if ln else b""
|
||||
return (b0 & 0x0F), (d or b"").decode(errors="replace")
|
||||
PYEOF
|
||||
|
||||
serve() { # serve PORT [env...] — start + wait for THIS server's listener line
|
||||
PORT="$1"; shift
|
||||
LEG=$((LEG + 1))
|
||||
printf '\n===== leg %d — port %s — %s =====\n' "$LEG" "$PORT" "${*:-default env}" >>"$SRVLOG"
|
||||
# readiness is searched only in THIS leg's slice: the log is appended, never
|
||||
# truncated, so a 'listening' line from an earlier leg would lie
|
||||
LEGFROM=$(( $(wc -l < "$SRVLOG") + 1 ))
|
||||
"$@" "$W/app/target/chat" "$PORT" >>"$SRVLOG" 2>&1 &
|
||||
SRV=$!
|
||||
for _ in $(seq 1 80); do
|
||||
tail -n "+$LEGFROM" "$SRVLOG" 2>/dev/null | grep -q listening && return 0
|
||||
sleep 0.1
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
functional() { # $1 = leg name
|
||||
timeout 30 python3 - "$PORT" <<'PYEOF'
|
||||
import sys; sys.path.insert(0, sys.argv[0].rsplit("/",1)[0])
|
||||
port = int(sys.argv[1])
|
||||
import importlib.util, os
|
||||
spec = importlib.util.spec_from_file_location("wsc", os.environ["WSC"])
|
||||
wsc = importlib.util.module_from_spec(spec); spec.loader.exec_module(wsc)
|
||||
a = wsc.connect(port, "lobby", "alice")
|
||||
assert wsc.recv(a) == (1, "* alice joined")
|
||||
b = wsc.connect(port, "lobby", "bob")
|
||||
assert wsc.recv(a) == (1, "* bob joined")
|
||||
assert wsc.recv(b) == (1, "* bob joined")
|
||||
c = wsc.connect(port, "other", "carol")
|
||||
assert wsc.recv(c) == (1, "* carol joined")
|
||||
wsc.send(a, "hello room")
|
||||
assert wsc.recv(a) == (1, "alice: hello room")
|
||||
assert wsc.recv(b) == (1, "alice: hello room")
|
||||
import socket
|
||||
try:
|
||||
k, t = wsc.recv(c, timeout=0.8); assert False, f"leak into other room: {t}"
|
||||
except socket.timeout: pass
|
||||
b.close()
|
||||
k, t = wsc.recv(a)
|
||||
assert (k, t) == (1, "* bob left"), (k, t)
|
||||
a.close(); c.close()
|
||||
print("functional-ok")
|
||||
PYEOF
|
||||
}
|
||||
|
||||
# ---- 2. functional on both backends ----
|
||||
export WSC="$CLIENT"
|
||||
serve "$((PORT0 + 0))" env WO_IO=uring || bad "serve-uring" "no listener"
|
||||
r="$(functional uring)"; [[ "$r" == *functional-ok* ]] \
|
||||
&& ok "uring: handshake(key verified) + presence + broadcast + isolation + leave" \
|
||||
|| bad "uring-functional" "$r"
|
||||
kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV=""
|
||||
|
||||
serve "$((PORT0 + 1))" env WO_IO=epoll || bad "serve-epoll" "no listener"
|
||||
r="$(functional epoll)"; [[ "$r" == *functional-ok* ]] \
|
||||
&& ok "epoll: the same matrix" || bad "epoll-functional" "$r"
|
||||
kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV=""
|
||||
|
||||
# ---- 3. the soak: N clients, ONE hot room ----
|
||||
serve "$((PORT0 + 2))" || bad "serve-soak" "no listener"
|
||||
fds_before="$(ls /proc/$SRV/fd 2>/dev/null | wc -l)"
|
||||
fds_prev=99999
|
||||
r="$(timeout 180 python3 - "$PORT" "$SOAK_N" <<'PYEOF'
|
||||
import asyncio, sys, os, time, base64, hashlib
|
||||
port, N = int(sys.argv[1]), int(sys.argv[2])
|
||||
GUID = "258EAFA5-E914-47DA-95CA-C5AB0DC85B11"
|
||||
MARK = "the-hot-room-marker"
|
||||
sem = asyncio.Semaphore(100)
|
||||
async def client(i, results):
|
||||
async with sem:
|
||||
r, w = await asyncio.open_connection("127.0.0.1", port)
|
||||
key = base64.b64encode(os.urandom(16)).decode()
|
||||
w.write((f"GET /ws?room=hot&name=c{i} HTTP/1.1\r\nhost: a\r\n"
|
||||
f"upgrade: websocket\r\nconnection: Upgrade\r\n"
|
||||
f"sec-websocket-key: {key}\r\nsec-websocket-version: 13\r\n\r\n").encode())
|
||||
await w.drain()
|
||||
d = b""
|
||||
while b"\r\n\r\n" not in d: d += await r.read(2000)
|
||||
if i == 0:
|
||||
# the sender: wait for the herd, then one marker line
|
||||
await asyncio.sleep(0)
|
||||
results["sender_ready"].set()
|
||||
try:
|
||||
buf = b""
|
||||
deadline = time.time() + 150
|
||||
while time.time() < deadline:
|
||||
try:
|
||||
c = await asyncio.wait_for(r.read(8192), timeout=5)
|
||||
except asyncio.TimeoutError:
|
||||
if results["sent"].is_set(): break
|
||||
continue
|
||||
if not c: break
|
||||
buf += c
|
||||
# scan frames for the marker (server frames are unmasked, small)
|
||||
if MARK.encode() in buf:
|
||||
results["got"] += 1
|
||||
return
|
||||
finally:
|
||||
w.close()
|
||||
async def main():
|
||||
results = {"got": 0, "sender_ready": asyncio.Event(), "sent": asyncio.Event()}
|
||||
conns = []
|
||||
# keep the sender's socket outside the tasks: join first
|
||||
sr, sw = None, None
|
||||
async def sender():
|
||||
nonlocal sr, sw
|
||||
async with sem:
|
||||
sr, sw = await asyncio.open_connection("127.0.0.1", port)
|
||||
key = base64.b64encode(os.urandom(16)).decode()
|
||||
sw.write((f"GET /ws?room=hot&name=sender HTTP/1.1\r\nhost: a\r\n"
|
||||
f"upgrade: websocket\r\nconnection: Upgrade\r\n"
|
||||
f"sec-websocket-key: {key}\r\nsec-websocket-version: 13\r\n\r\n").encode())
|
||||
await sw.drain()
|
||||
d = b""
|
||||
while b"\r\n\r\n" not in d: d += await sr.read(2000)
|
||||
await sender()
|
||||
tasks = [asyncio.create_task(client(i, results)) for i in range(N)]
|
||||
await asyncio.sleep(max(2.0, N / 250)) # let the herd join + drain presence
|
||||
p = MARK.encode(); mask = os.urandom(4)
|
||||
hdr = bytes([0x81, 0x80 | len(p)])
|
||||
sw.write(hdr + mask + bytes(b ^ mask[i % 4] for i, b in enumerate(p)))
|
||||
await sw.drain()
|
||||
results["sent"].set()
|
||||
t0 = time.time()
|
||||
await asyncio.gather(*tasks, return_exceptions=True)
|
||||
el = int((time.time() - t0) * 1000)
|
||||
sw.close()
|
||||
print(f"{results['got']}|{N}|{el}")
|
||||
asyncio.run(main())
|
||||
PYEOF
|
||||
)"
|
||||
got="${r%%|*}"; rest="${r#*|}"; n="${rest%%|*}"; el="${rest#*|}"
|
||||
[[ "$got" == "$n" ]] \
|
||||
&& ok "soak: the marker reached all $got/$n hot-room clients (${el}ms after send)" \
|
||||
|| bad "soak" "$r"
|
||||
# leave-broadcast storms take a moment to settle after 1k closes
|
||||
for _ in $(seq 1 20); do
|
||||
fds_w1="$(ls /proc/$SRV/fd 2>/dev/null | wc -l)"
|
||||
[[ "$fds_w1" -le "$fds_prev" ]] && break
|
||||
fds_prev="$fds_w1"
|
||||
sleep 0.5
|
||||
done
|
||||
# The fd check is for a per-CONNECTION leak, and a fixed tolerance cannot
|
||||
# express that. Shards initialise LAZILY (runtime/src/vm.c: a worker's vm is
|
||||
# not paid for until its first fiber arrives), so the first wave legitimately
|
||||
# adds one io_uring + one eventfd PER SHARD, capped at nproc — on a 20-core
|
||||
# box that is +18, which the old `fds_before + 8` read as a leak. Measured
|
||||
# 2026-08-27: 26 -> 44 after 20 clients, then still 44 after 40 more.
|
||||
#
|
||||
# So assert the invariant itself: a SECOND wave must not raise the count.
|
||||
# Core-count independent, and it catches a slow leak that any fixed
|
||||
# tolerance would hide inside its own slack.
|
||||
timeout 60 python3 - "$PORT" 20 <<'PYEOF' >/dev/null 2>&1
|
||||
import socket, base64, os, sys, time
|
||||
port, n = int(sys.argv[1]), int(sys.argv[2])
|
||||
socks = []
|
||||
for i in range(n):
|
||||
s = socket.create_connection(("127.0.0.1", port), timeout=8)
|
||||
k = base64.b64encode(os.urandom(16)).decode()
|
||||
s.sendall((f"GET /ws?room=fdwave&name=w{i} HTTP/1.1\r\nhost: a\r\n"
|
||||
f"upgrade: websocket\r\nconnection: Upgrade\r\n"
|
||||
f"sec-websocket-key: {k}\r\nsec-websocket-version: 13\r\n\r\n").encode())
|
||||
h = b""
|
||||
while b"\r\n\r\n" not in h:
|
||||
h += s.recv(4096)
|
||||
socks.append(s)
|
||||
time.sleep(0.5)
|
||||
for s in socks:
|
||||
s.close()
|
||||
PYEOF
|
||||
fds_after="$fds_w1"
|
||||
for _ in $(seq 1 20); do
|
||||
fds_after="$(ls /proc/$SRV/fd 2>/dev/null | wc -l)"
|
||||
[[ "$fds_after" -le "$fds_w1" ]] && break
|
||||
sleep 0.5
|
||||
done
|
||||
rss_kb="$(awk '/VmRSS/{print $2}' /proc/$SRV/status 2>/dev/null)"
|
||||
[[ "$fds_after" -le "$fds_w1" ]] \
|
||||
&& ok "no per-connection fd leak (start $fds_before, after $SOAK_N: $fds_w1, after 20 more: $fds_after)" \
|
||||
|| bad "soak-fds" "second wave grew fds: $fds_w1 -> $fds_after (start $fds_before)"
|
||||
[[ -n "$rss_kb" && "$rss_kb" -lt 819200 ]] \
|
||||
&& ok "soak RSS bounded (${rss_kb}KB < 800MB)" || bad "soak-rss" "${rss_kb}KB"
|
||||
|
||||
# ---- 4. drain: SIGTERM with clients connected -> close frames, exit 0 ----
|
||||
# Starts its OWN server. It used to inherit the soak leg's $SRV, which meant
|
||||
# any leg inserted between them silently handed drain an empty pid: its python
|
||||
# died on int(""), the leg reported a bare failure, AND the soak server was
|
||||
# never killed — orphaning a listener that then broke the NEXT run's soak on
|
||||
# the same port. No leg may depend on another leg's server.
|
||||
serve "$((PORT0 + 6))" || bad "serve-drain" "no listener"
|
||||
r="$(timeout 30 python3 - "$PORT" "$SRV" <<'PYEOF'
|
||||
import sys, os, time, signal, socket
|
||||
import importlib.util
|
||||
spec = importlib.util.spec_from_file_location("wsc", os.environ["WSC"])
|
||||
wsc = importlib.util.module_from_spec(spec); spec.loader.exec_module(wsc)
|
||||
port, srv = int(sys.argv[1]), int(sys.argv[2])
|
||||
a = wsc.connect(port, "lobby", "alice"); wsc.recv(a)
|
||||
b = wsc.connect(port, "lobby", "bob"); wsc.recv(a); wsc.recv(b)
|
||||
os.kill(srv, signal.SIGTERM)
|
||||
def drained(s):
|
||||
try:
|
||||
while True:
|
||||
k, _ = wsc.recv(s, timeout=5)
|
||||
if k == 8: return "close-frame"
|
||||
if k == -2: return "eof"
|
||||
except socket.timeout:
|
||||
return "stuck"
|
||||
except (ConnectionResetError, BrokenPipeError):
|
||||
return "reset"
|
||||
print(drained(a) + "|" + drained(b))
|
||||
PYEOF
|
||||
)"
|
||||
[[ "$r" == "close-frame|close-frame" ]] \
|
||||
&& ok "drain: both clients got the close frame" || bad "drain" "$r"
|
||||
stopped=1
|
||||
for _ in $(seq 1 40); do kill -0 "$SRV" 2>/dev/null || { stopped=0; break; }; sleep 0.1; done
|
||||
[[ $stopped -eq 0 ]] && ok "SIGTERM exits 0" || bad "stop" "still running"
|
||||
SRV=""
|
||||
|
||||
# ---- 4b. WO_SHARDS=1: the same matrix on one shard ----
|
||||
# The plan requires `just chat` green at default cores AND on a single shard:
|
||||
# cross-shard placement is where the actor work can hide a bug, so the
|
||||
# one-shard run is the control that says a failure is placement's fault.
|
||||
serve "$((PORT0 + 4))" env WO_SHARDS=1 || bad "serve-shards1" "no listener"
|
||||
r="$(functional shards1)"; [[ "$r" == *functional-ok* ]] \
|
||||
&& ok "WO_SHARDS=1: the same matrix on a single shard" \
|
||||
|| bad "shards1-functional" "$r"
|
||||
kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV=""
|
||||
|
||||
# ---- 4c. WO_MAILBOX=8: the drop-slow-member path FIRES and the room lives ----
|
||||
# The backpressure policy earning its keep. A member that stops reading makes
|
||||
# its writer block on write_dl; with the mailbox capped at 8 the room's
|
||||
# broadcast send traps (WO_T_ACTOR), and the room must CATCH that, drop the
|
||||
# member, and keep serving everyone else. Asserting the room survives is the
|
||||
# point — a room that dies with its slowest member is the bug this policy
|
||||
# exists to prevent.
|
||||
serve "$((PORT0 + 5))" env WO_MAILBOX=8 || bad "serve-mailbox" "no listener"
|
||||
r="$(timeout 90 python3 - "$PORT" <<'PYEOF'
|
||||
import importlib.util, os, socket, sys, time
|
||||
spec = importlib.util.spec_from_file_location("wsc", os.environ["WSC"])
|
||||
wsc = importlib.util.module_from_spec(spec); spec.loader.exec_module(wsc)
|
||||
port = int(sys.argv[1])
|
||||
|
||||
fast = wsc.connect(port, "bp", "fast")
|
||||
wsc.recv(fast) # * fast joined
|
||||
# the slow member: a tiny receive buffer so the server's socket fills fast,
|
||||
# and it never reads a single frame
|
||||
slow = wsc.connect(port, "bp", "slow", rcvbuf=2048)
|
||||
wsc.recv(fast) # * slow joined
|
||||
|
||||
# storm: big frames the slow member never drains
|
||||
blob = "x" * 1024
|
||||
for i in range(400):
|
||||
try:
|
||||
wsc.send(fast, f"{i}-{blob}")
|
||||
except OSError:
|
||||
break
|
||||
# drain what fast owes us so its own mailbox cannot be the thing that fills
|
||||
deadline = time.time() + 20
|
||||
seen = 0
|
||||
while time.time() < deadline:
|
||||
try:
|
||||
k, t = wsc.recv(fast, timeout=0.5)
|
||||
seen += 1
|
||||
except Exception:
|
||||
break
|
||||
|
||||
# the room must still be alive and serving the fast member
|
||||
survivor = wsc.connect(port, "bp", "late")
|
||||
ok_join = False
|
||||
deadline = time.time() + 15
|
||||
while time.time() < deadline:
|
||||
try:
|
||||
k, t = wsc.recv(fast, timeout=1.0)
|
||||
if "late joined" in t:
|
||||
ok_join = True
|
||||
break
|
||||
except Exception:
|
||||
break
|
||||
print("mailbox-ok" if ok_join else f"mailbox-dead seen={seen}")
|
||||
slow.close(); fast.close(); survivor.close()
|
||||
PYEOF
|
||||
)"
|
||||
[[ "$r" == *mailbox-ok* ]] \
|
||||
&& ok "WO_MAILBOX=8: slow member dropped, room survived and kept serving" \
|
||||
|| bad "mailbox-backpressure" "$r"
|
||||
kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV=""
|
||||
|
||||
# ---- 5. the ASan leg: functional matrix, zero leaks ----
|
||||
if [[ -x "$ASAN" ]]; then
|
||||
sed -i "s|runtime = \".*\"|runtime = \"$ASAN\"|" "$W/app/wo.toml"
|
||||
rm -rf "$W/app/target"
|
||||
"$WOC" "$W/app" >/dev/null 2>&1
|
||||
serve "$((PORT0 + 3))" || bad "serve-asan" "no listener"
|
||||
r="$(functional asan)"
|
||||
kill -TERM "$SRV" 2>/dev/null
|
||||
for _ in $(seq 1 60); do kill -0 "$SRV" 2>/dev/null || break; sleep 0.1; done
|
||||
SRV=""
|
||||
if [[ "$r" == *functional-ok* ]] \
|
||||
&& ! tail -n "+$LEGFROM" "$SRVLOG" | grep -q "AddressSanitizer\|LeakSanitizer"; then
|
||||
ok "ASan run clean (functional + drain, zero leaks)"
|
||||
else
|
||||
bad "asan" "$(tail -n "+$LEGFROM" "$SRVLOG" | grep -m1 -E 'ERROR|SUMMARY' || echo "$r")"
|
||||
fi
|
||||
else
|
||||
bad "asan" "runtime/build/wovm_asan missing — make -C runtime wovm-asan"
|
||||
fi
|
||||
|
||||
echo
|
||||
printf 'chat-accept: %d checks, %d failures\n' "$((pass + fail))" "$fail"
|
||||
[[ $fail -eq 0 ]]
|
||||
|
|
@ -26,6 +26,24 @@ QUICK = "--quick" in sys.argv
|
|||
WRITE_BASELINE = "--write-baseline" in sys.argv
|
||||
|
||||
N = 2000 if QUICK else 20000
|
||||
# databasev2 4: the write-concurrent leg. `mix` writes on one op in ten with
|
||||
# C=4, so group commit had almost nothing to batch there (measured mean batch
|
||||
# 1.01, peak 3) — a property of that workload, not of the mechanism. C is high
|
||||
# on purpose: batching is a function of how many writes are in flight, and
|
||||
# measured mean batch rose 1.13 -> 1.76 -> 5.35 at C = 4 -> 16 -> 64.
|
||||
WMIX_N = 4000 if QUICK else 20000
|
||||
WMIX_C = 32 if QUICK else 64
|
||||
# databasev2 3: the checkpoint leg. Ages a store by UPDATING the same rows, so
|
||||
# history grows while the live set does not — otherwise the leg measures insert
|
||||
# throughput instead of compaction.
|
||||
CKPT_SEED = 2000 if QUICK else 5000
|
||||
CKPT_OPS = 8000 if QUICK else 20000
|
||||
# The stop-the-world budget. 50ms is a stall a serving process can absorb
|
||||
# without a client noticing a timeout; measured at ~13ms for a 2MB live set,
|
||||
# so this leaves real headroom while still failing before a stall becomes
|
||||
# user-visible. Compaction is O(live rows), so this budget is what eventually
|
||||
# forces the incremental design the spec deliberately did not buy in advance.
|
||||
CKPT_PAUSE_BUDGET_US = 50000
|
||||
MSG_N = 20000 if QUICK else 200000
|
||||
WAL_N = 800 if QUICK else 4000
|
||||
CRASH_REPS = 1 if QUICK else 3
|
||||
|
|
@ -95,6 +113,55 @@ def parse_metrics(lines, into, prefix):
|
|||
if m:
|
||||
into[f"{prefix}.msgrate.msgs_sec"] = int(m.group(2))
|
||||
|
||||
def wmix_leg(metrics, tag, env, data):
|
||||
"""Every op a durable write, WMIX_C at once — the leg that actually
|
||||
exercises group commit.
|
||||
|
||||
It reuses the store the `all` run just seeded (a fresh process replays it,
|
||||
so `kmod` is there) and asks the runtime for its group-commit counters via
|
||||
WO_WAL_STATS. The counters matter as much as the throughput: if batches are
|
||||
always one the mechanism is inert and any throughput change came from
|
||||
somewhere else, so a payoff would be attributed to the wrong cause."""
|
||||
e = dict(env); e["WO_WAL_STATS"] = "1"
|
||||
rc, lines, _, _ = run(["wmix", str(WMIX_N), str(WMIX_C)], e, 1800)
|
||||
if rc != 0:
|
||||
bad(f"{tag}.wmix", f"rc={rc} tail={lines[-2:]}")
|
||||
return
|
||||
ops = p50 = p99 = None
|
||||
batches = records = peak_batch = peak_staged = None
|
||||
for l in lines:
|
||||
f = l.split()
|
||||
if f and f[0] == "wmix" and len(f) == 5:
|
||||
ops, p50, p99 = int(f[2]), int(f[3]), int(f[4])
|
||||
elif f and f[0] == "walstats":
|
||||
kv = dict(x.split("=", 1) for x in f[1:] if "=" in x)
|
||||
batches = int(kv.get("batches", 0)); records = int(kv.get("records", 0))
|
||||
peak_batch = int(kv.get("peak_batch", 0)); peak_staged = int(kv.get("peak_staged", 0))
|
||||
if ops is None or batches is None:
|
||||
bad(f"{tag}.wmix", "no report or no walstats line")
|
||||
return
|
||||
metrics[f"{tag}.wmix.ops_sec"] = ops
|
||||
metrics[f"{tag}.wmix.p50us"] = p50
|
||||
metrics[f"{tag}.wmix.p99us"] = p99
|
||||
metrics[f"{tag}.wmix.peak_batch"] = peak_batch
|
||||
metrics[f"{tag}.wmix.peak_staged"] = peak_staged
|
||||
mean = round(records / batches, 2) if batches else 0
|
||||
metrics[f"{tag}.wmix.mean_batch"] = mean
|
||||
ok(f"{tag}.wmix: {ops} ops/sec, p50 {p50}us p99 {p99}us; "
|
||||
f"{records} records over {batches} barriers (mean {mean}, peak {peak_batch}), "
|
||||
f"peak staged {peak_staged}B")
|
||||
# The gate that matters. Only the MULTI-shard leg can batch: a worker's
|
||||
# statements marshal to shard 0 and queue, while shard-0 statements run
|
||||
# inline and commit one at a time by design (see db.c).
|
||||
if tag.endswith(".sN"):
|
||||
if mean > 1.0:
|
||||
ok(f"{tag}.wmix batches form (mean {mean} > 1)")
|
||||
else:
|
||||
bad(f"{tag}.wmix-inert",
|
||||
f"mean batch {mean} — group commit is not engaging, so a "
|
||||
f"throughput change would not be attributable to it")
|
||||
|
||||
|
||||
def campaign():
|
||||
metrics = {}
|
||||
ncores = os.cpu_count() or 1
|
||||
|
|
@ -123,6 +190,8 @@ def campaign():
|
|||
bad(f"{tag}.mix.fds", f"grew {fdg}")
|
||||
else:
|
||||
ok(f"{tag}.mix.fds flat")
|
||||
if flavor == "durable" and data:
|
||||
wmix_leg(metrics, tag, env, data)
|
||||
if data: shutil.rmtree(data, ignore_errors=True)
|
||||
# msgrate once per shard count, RAM only (no store dependency)
|
||||
for shards in (1, ncores):
|
||||
|
|
@ -235,6 +304,49 @@ def tolerance_for(key):
|
|||
if key.startswith("ceiling."): return 100
|
||||
if key.startswith("randread."): return 100
|
||||
if key.startswith("replay."): return 100
|
||||
# databasev2 4: batch SHAPE follows arrival timing, so gating it tightly
|
||||
# would gate the scheduler — what must hold is that the mean exceeds one
|
||||
# under contention, which wmix_leg asserts directly against the live run.
|
||||
# wmix's throughput and latency are NOT waived: they are the payoff, and a
|
||||
# blanket waiver here would have left the whole leg ungated.
|
||||
if key.endswith((".wmix.mean_batch", ".wmix.peak_batch", ".wmix.peak_staged")):
|
||||
return 100
|
||||
# databasev2 4: DURABLE multi-shard p99 is an fsync TAIL, and group commit
|
||||
# made it both noisier and legitimately higher. Measured across three full
|
||||
# runs of the same build, durable.sN.mixread.p99 was 1043 / 2318 / 4147 us
|
||||
# and wmix.p99 8758 / 20000 — a 2-4x spread with the box near idle, because
|
||||
# a barrier now blocks the owner shard LONGER (more records per fsync) even
|
||||
# though it blocks LESS OFTEN. That is the trade group commit makes on a
|
||||
# single-threaded owner, and part B (async submission) is what would undo
|
||||
# it. Gating a 2-4x-variable tail at 50% gates the disk, not the engine, so
|
||||
# the FLOOR is the real guard here — and it is not slack: mixread's floor
|
||||
# (4172us) came within 25us of tripping on the worst run.
|
||||
if key.startswith("durable.sN.") and key.endswith(".p99us"):
|
||||
# Widened again 2026-08-29 with more evidence: mixread p99 was measured
|
||||
# at 1043 / 2318 / 4147us and mixwrite at 1623 / 4446us across runs of
|
||||
# the SAME build on a near-idle box — a 3-4x spread. 100% was still
|
||||
# gating the disk. The FLOOR stays the real guard and is not slack:
|
||||
# mixread's came within 25us of tripping on the worst run observed.
|
||||
return 300
|
||||
# databasev2 3: the RECLAIM ratio is structural and gated tightly — it is
|
||||
# the feature's whole claim. Boot time and the pause are wall-clock on a
|
||||
# shared box and are not: waiving them all would have left the leg ungated,
|
||||
# which is the mistake part A's task 4 made and had to undo.
|
||||
if key in ("ckpt.boot_off_ms", "ckpt.boot_on_ms", "ckpt.pause_us_max",
|
||||
"ckpt.compactions", "ckpt.bytes_off", "ckpt.bytes_on"):
|
||||
return 400
|
||||
# compaction BANDWIDTH is the engine's own property, so it is gated for
|
||||
# real — it is what regressed 8x when the dump was fsyncing per flush
|
||||
if key == "ckpt.pause_us_per_mb":
|
||||
return 100
|
||||
# msgrate is actor-to-actor throughput and is scheduling-bound, so its
|
||||
# run-to-run spread is far wider than its old 15%. MEASURED across the 10
|
||||
# full runs recorded on 2026-08-28/29 — several of them predating the
|
||||
# checkpoint work — it ranged 10.7M to 17.9M msgs/sec, a 1.67x spread. A
|
||||
# 15% gate on that gates the scheduler and fails intermittently whatever
|
||||
# the engine does. Pre-existing; found while closing databasev2 3, not
|
||||
# caused by it.
|
||||
if ".msgrate." in key: return 70
|
||||
if ".mixread." in key or ".mixwrite." in key: return 50
|
||||
if ".sN." in key: return 50
|
||||
if ".read." in key or ".query." in key: return 50
|
||||
|
|
@ -246,7 +358,11 @@ def write_baseline(metrics):
|
|||
"tolerances come from tolerance_for() in the driver"}}
|
||||
for k, v in sorted(metrics.items()):
|
||||
if k.endswith(("rss_growth_kb", "fd_growth")): continue
|
||||
higher = k.endswith(("ops_sec", "msgs_sec"))
|
||||
# reclaim_x: MORE reclaimed is better. Recorded as lower-is-better by
|
||||
# the default detector, which would have passed "no reclaim at all" and
|
||||
# failed an improvement — the feature's central claim, gated backwards.
|
||||
higher = k.endswith(("ops_sec", "msgs_sec", "mean_batch", "peak_batch",
|
||||
"reclaim_x"))
|
||||
floor_div = 8 if k.endswith("msgs_sec") else 4
|
||||
# latency floors never sit below 100µs: at post-index µs scale a
|
||||
# 4×1µs "catastrophe line" is noise; the tripwire means "µs became
|
||||
|
|
@ -520,9 +636,12 @@ def randread(metrics):
|
|||
def wal_used(data_dir):
|
||||
"""Bytes actually written across the store's WAL files.
|
||||
|
||||
The non-zero prefix, NOT the file size: shard WALs are fallocate'd to
|
||||
1 MiB up front, so getsize reports 1048576 for an empty store and proves
|
||||
nothing. Same reason scripts/residency-accept.sh measures it this way."""
|
||||
The non-zero prefix, NOT the file size: shard WALs are preallocated, so
|
||||
getsize reports the preallocation (1 MiB) even for an empty store. Same
|
||||
reason scripts/residency-accept.sh measures it this way.
|
||||
|
||||
databasev2 1 and databasev2 3 each grew their own copy of this helper on
|
||||
separate branches; this is the single one they now share."""
|
||||
total = 0
|
||||
for name in sorted(os.listdir(data_dir)):
|
||||
with open(os.path.join(data_dir, name), "rb") as f:
|
||||
|
|
@ -621,6 +740,96 @@ def replay(metrics):
|
|||
metrics["replay.history_penalty_x"] = round(penalty, 2)
|
||||
ok(f"replay: identical dataset, {penalty:.2f}x the boot cost from history alone "
|
||||
f"({ins_ms:.0f} -> {his_ms:.0f} ms) -- what a checkpoint would collapse")
|
||||
def checkpoint_leg(metrics):
|
||||
"""Space reclaimed, boot time, and the stop-the-world PAUSE.
|
||||
|
||||
The same workload runs twice, differing only in whether checkpointing can
|
||||
fire: an enormous floor disables it, a small one lets it. Comparing two runs
|
||||
of one build is what isolates compaction from everything else the workload
|
||||
does.
|
||||
|
||||
Boot is measured with the sample's `boot` mode, which does nothing at all —
|
||||
with WO_DATA set the runtime replays the whole log before main runs, so a
|
||||
mode with no work of its own is the only honest way to price replay."""
|
||||
ncores = os.cpu_count() or 1
|
||||
out = {}
|
||||
for name, knobs in (("off", {"WO_CHECKPOINT_BYTES": "1000000000"}),
|
||||
("on", {"WO_CHECKPOINT_BYTES": "65536", "WO_CHECKPOINT_RATIO": "2"})):
|
||||
data = os.path.join(ROOT, "bench", f"tmp.{os.getpid()}.ckpt.{name}")
|
||||
shutil.rmtree(data, ignore_errors=True); os.makedirs(data, exist_ok=True)
|
||||
env = {"WO_DATA": data, "WO_SHARDS": str(ncores), "WO_WAL_STATS": "1"}
|
||||
env.update(knobs)
|
||||
rc, _, _, _ = run(["seed", str(CKPT_SEED)], env, 1800)
|
||||
if rc != 0:
|
||||
bad(f"ckpt.{name}.seed", f"rc={rc}"); shutil.rmtree(data, ignore_errors=True); return
|
||||
rc, lines, _, _ = run(["wmix", str(CKPT_OPS), "16"], env, 1800)
|
||||
if rc != 0:
|
||||
bad(f"ckpt.{name}.age", f"rc={rc}"); shutil.rmtree(data, ignore_errors=True); return
|
||||
stats = {}
|
||||
for l in lines:
|
||||
f = l.split()
|
||||
if f and f[0] == "walstats":
|
||||
stats = dict(x.split("=", 1) for x in f[1:] if "=" in x)
|
||||
used = wal_used(data)
|
||||
# NOT through run(): it samples RSS on a 250ms poll, so every timing it
|
||||
# produces floors at the poll quantum — boot measured that way reported
|
||||
# 251ms both with and without checkpointing, which is the harness's
|
||||
# clock, not the engine's. Median of 3 because this is wall-clock.
|
||||
benv = dict(os.environ)
|
||||
benv.update({"WO_DATA": data, "WO_SHARDS": str(ncores)})
|
||||
samples = []
|
||||
brc = 0
|
||||
for _ in range(3):
|
||||
t0 = time.monotonic()
|
||||
pr = subprocess.run([BIN, "boot"], stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL, env=benv, timeout=900)
|
||||
samples.append((time.monotonic() - t0) * 1000.0)
|
||||
brc = pr.returncode or brc
|
||||
boot_ms = sorted(samples)[1]
|
||||
if brc != 0:
|
||||
bad(f"ckpt.{name}.boot", f"rc={brc}"); shutil.rmtree(data, ignore_errors=True); return
|
||||
out[name] = (used, boot_ms, stats)
|
||||
shutil.rmtree(data, ignore_errors=True)
|
||||
|
||||
(off_b, off_boot, _), (on_b, on_boot, st) = out["off"], out["on"]
|
||||
comps = int(st.get("compactions", 0))
|
||||
if comps == 0:
|
||||
bad("ckpt.inert", "no compaction ran — the leg proves nothing about checkpointing")
|
||||
return
|
||||
metrics["ckpt.compactions"] = comps
|
||||
metrics["ckpt.bytes_off"] = off_b
|
||||
metrics["ckpt.bytes_on"] = on_b
|
||||
metrics["ckpt.reclaim_x"] = round(off_b / max(on_b, 1), 2)
|
||||
metrics["ckpt.boot_off_ms"] = int(round(off_boot))
|
||||
metrics["ckpt.boot_on_ms"] = int(round(on_boot))
|
||||
metrics["ckpt.pause_us_max"] = int(st.get("compact_us_max", 0))
|
||||
# The RAW pause scales with the live set, and this workload's live set is
|
||||
# not fixed: wmix's hist_dump inserts a row per latency bucket, so a noisier
|
||||
# box produces more buckets, more rows, and a longer pause. Gating the raw
|
||||
# number against a baseline therefore gates the box. What belongs to the
|
||||
# ENGINE is the rate, so that is what carries a real tolerance; the raw
|
||||
# pause keeps the absolute budget assertion below as its guard.
|
||||
cb = int(st.get("compacted_bytes", 0))
|
||||
if cb > 0 and metrics["ckpt.pause_us_max"] > 0:
|
||||
metrics["ckpt.pause_us_per_mb"] = int(round(
|
||||
metrics["ckpt.pause_us_max"] / (cb / (1024.0 * 1024.0))))
|
||||
ok(f"ckpt: {off_b} -> {on_b} bytes ({metrics['ckpt.reclaim_x']}x reclaimed) over "
|
||||
f"{comps} compactions; boot {off_boot:.0f} -> {on_boot:.0f} ms; "
|
||||
f"stop-the-world pause max {metrics['ckpt.pause_us_max']}us "
|
||||
f"({metrics.get('ckpt.pause_us_per_mb', 0)}us/MB)")
|
||||
# the space claim is the point of the feature, so it is asserted, not just recorded
|
||||
if off_b <= on_b:
|
||||
bad("ckpt.no-reclaim", f"checkpointing did not shrink the log ({off_b} -> {on_b})")
|
||||
else:
|
||||
ok(f"ckpt: the log is smaller with checkpointing on")
|
||||
# THE BUDGET. Stated, not assumed — the spec refused to assume it.
|
||||
if metrics["ckpt.pause_us_max"] > CKPT_PAUSE_BUDGET_US:
|
||||
bad("ckpt.pause-budget",
|
||||
f"stop-the-world pause {metrics['ckpt.pause_us_max']}us exceeds the stated "
|
||||
f"{CKPT_PAUSE_BUDGET_US}us budget — alternatives (incremental copy, "
|
||||
f"fork-and-dump) are bought against THIS number")
|
||||
else:
|
||||
ok(f"ckpt: pause within budget ({metrics['ckpt.pause_us_max']} <= {CKPT_PAUSE_BUDGET_US}us)")
|
||||
|
||||
|
||||
def main():
|
||||
|
|
@ -639,6 +848,7 @@ def main():
|
|||
ceiling(metrics)
|
||||
randread(metrics)
|
||||
replay(metrics)
|
||||
checkpoint_leg(metrics)
|
||||
os.makedirs(RESULTS_DIR, exist_ok=True)
|
||||
stamp = time.strftime("%Y%m%d-%H%M%S")
|
||||
out = os.path.join(RESULTS_DIR, f"run-{stamp}{'-quick' if QUICK else ''}.json")
|
||||
|
|
|
|||
|
|
@ -51,6 +51,14 @@ if [[ ! -x "$WOVM" ]]; then
|
|||
fi
|
||||
|
||||
WORK="$(mktemp -d "${TMPDIR:-/tmp}/lw-accept.XXXXXX")"
|
||||
# stable, tailable log for the example app: the per-run work dir is deleted on
|
||||
# exit, so a developer had nothing to follow. `tail -F /tmp/log-watcher.log`.
|
||||
# Each invocation keeps its own $WORK/*.out (the checks grep those) and is
|
||||
# ALSO teed here, banner-separated, so one file holds the whole run.
|
||||
APPLOG="/tmp/log-watcher.log"
|
||||
: > "$APPLOG"
|
||||
echo "app log: $APPLOG (tail -F \"$APPLOG\" to follow)"
|
||||
|
||||
# LW_ACCEPT_KEEP=1 leaves the work directory (image, logs, cron.d, the
|
||||
# server's own stdout) in place — what you want the moment a check fails.
|
||||
cleanup() {
|
||||
|
|
@ -89,7 +97,8 @@ fi
|
|||
# watcher to decide the burst is over.
|
||||
LOG="$WORK/app.log"
|
||||
: >"$LOG"
|
||||
timeout -k 2 12 "$WOVM" "$IMAGE" watch "$LOG" 2 1 >"$WORK/watch.out" 2>&1 &
|
||||
printf '\n===== watch =====\n' >>"$APPLOG"
|
||||
timeout -k 2 12 "$WOVM" "$IMAGE" watch "$LOG" 2 1 > >(tee -a "$APPLOG" >"$WORK/watch.out") 2>&1 &
|
||||
WATCH_PID=$!
|
||||
sleep 2
|
||||
printf 'info service starting\n' >>"$LOG"
|
||||
|
|
@ -110,6 +119,7 @@ CRON="$WORK/cron.d"
|
|||
mkdir -p "$CRON"
|
||||
printf '* * * * * root /usr/bin/backup.sh > /var/log/backup.log 2>&1\n' >"$CRON/backup"
|
||||
timeout 8 "$WOVM" "$IMAGE" run "$CRON" >"$WORK/run.out" 2>&1
|
||||
{ printf '\n===== run =====\n'; cat "$WORK/run.out"; } >>"$APPLOG"
|
||||
if grep -q "^SCHEDULE /var/log/backup.log" "$WORK/run.out"; then
|
||||
ok "run (parsed and scheduled the cron entry)"
|
||||
else
|
||||
|
|
@ -124,7 +134,8 @@ EOF
|
|||
# -k: `env.stopping()` installs a SIGTERM handler that only sets a flag, and
|
||||
# the serve loop is blocked in accept(), so a plain TERM is swallowed — the
|
||||
# process needs a KILL to actually stop (recorded in docs/00-status.md).
|
||||
timeout -k 2 20 "$WOVM" "$IMAGE" mcp "$CRON" "$WORK/cfg.json" >"$WORK/mcp.out" 2>&1 &
|
||||
printf '\n===== mcp =====\n' >>"$APPLOG"
|
||||
timeout -k 2 20 "$WOVM" "$IMAGE" mcp "$CRON" "$WORK/cfg.json" > >(tee -a "$APPLOG" >"$WORK/mcp.out") 2>&1 &
|
||||
SRV_PID=$!
|
||||
sleep 2
|
||||
|
||||
|
|
@ -256,7 +267,8 @@ if [[ -n "${LW_SOAK:-}" ]]; then
|
|||
soak_mode() {
|
||||
local name="$1" load_fn="$2"
|
||||
shift 2
|
||||
"$WOVM" "$IMAGE" "$@" >"$WORK/soak-$name.out" 2>&1 &
|
||||
printf '\n===== soak %s =====\n' "$name" >>"$APPLOG"
|
||||
"$WOVM" "$IMAGE" "$@" > >(tee -a "$APPLOG" >"$WORK/soak-$name.out") 2>&1 &
|
||||
local pid=$! rss0 fd0 rss1 fd1 drss dfd deadline i
|
||||
sleep 3 # first-touch pages and the first work cycle
|
||||
if ! kill -0 "$pid" 2>/dev/null; then
|
||||
|
|
|
|||
|
|
@ -56,6 +56,11 @@ fi
|
|||
|
||||
PORT=$((8500 + RANDOM % 400))
|
||||
DATA="$W/data"; mkdir -p "$DATA"
|
||||
# stable, tailable server log — the per-run temp dir is deleted on exit
|
||||
SRVLOG="/tmp/site.log"
|
||||
: > "$SRVLOG"
|
||||
echo "server log: $SRVLOG (tail -F \"$SRVLOG\" to follow)"
|
||||
|
||||
|
||||
hit() { # path [method] [data] [token] -> "STATUS|BODY" (redirects not followed)
|
||||
python3 - "$PORT" "$1" "${2:-GET}" "${3:-}" "${4:-}" <<'PYEOF'
|
||||
|
|
@ -97,7 +102,8 @@ expect() { # name got want_status want_substr
|
|||
}
|
||||
|
||||
serve() {
|
||||
SITE_TOKEN=s3cr3t WO_DATA="$DATA" "$W/app/target/site" "$PORT" >>"$W/srv.out" 2>&1 &
|
||||
printf '\n===== serve — port %s =====\n' "$PORT" >>"$SRVLOG"
|
||||
SITE_TOKEN=s3cr3t WO_DATA="$DATA" "$W/app/target/site" "$PORT" >>"$SRVLOG" 2>&1 &
|
||||
SRV=$!
|
||||
for _ in $(seq 1 40); do
|
||||
[[ "$(hit /health 2>/dev/null)" == 200* ]] && return 0
|
||||
|
|
|
|||
|
|
@ -83,9 +83,19 @@ else
|
|||
fi
|
||||
|
||||
DATA="$W/data"; mkdir -p "$DATA"
|
||||
WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >"$W/srv.out" 2>&1 &
|
||||
# stable, tailable server log: the per-run temp dir is deleted on exit, so a
|
||||
# developer had nothing to follow. `tail -F /tmp/web-app.log` while this runs.
|
||||
SRVLOG="/tmp/web-app.log"
|
||||
: > "$SRVLOG"
|
||||
echo "server log: $SRVLOG (tail -F \"$SRVLOG\" to follow)"
|
||||
printf '===== boot — port %s =====\n' "$PORT" >>"$SRVLOG"
|
||||
LEGFROM=$(( $(wc -l < "$SRVLOG") + 1 ))
|
||||
WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >>"$SRVLOG" 2>&1 &
|
||||
SRV=$!
|
||||
for _ in $(seq 1 40); do grep -q listening "$W/srv.out" 2>/dev/null && break; sleep 0.1; done
|
||||
for _ in $(seq 1 40); do
|
||||
tail -n "+$LEGFROM" "$SRVLOG" 2>/dev/null | grep -q listening && break
|
||||
sleep 0.1
|
||||
done
|
||||
|
||||
# one tiny HTTP client; python is already a repo test dependency
|
||||
hit() { # method path [body] [auth: yes|no] [content-type] -> "STATUS|BODY"
|
||||
|
|
@ -467,7 +477,8 @@ for _ in $(seq 1 30); do kill -0 "$SRV" 2>/dev/null || { stopped=0; break; }; sl
|
|||
SRV=""
|
||||
|
||||
# ---- 15. restart persistence (WAL replay) ----
|
||||
WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >>"$W/srv.out" 2>&1 &
|
||||
printf '\n===== restart (WAL replay) — port %s =====\n' "$PORT" >>"$SRVLOG"
|
||||
WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >>"$SRVLOG" 2>&1 &
|
||||
SRV=$!
|
||||
sleep 0.5
|
||||
expect "product survives a restart (WAL)" "$(hit GET /products)" 200 '"name":"mug"'
|
||||
|
|
|
|||
3
tests/corpus/run/monitor-death/fixture.out
Normal file
3
tests/corpus/run/monitor-death/fixture.out
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
died: boom
|
||||
died: late
|
||||
done
|
||||
36
tests/corpus/run/monitor-death/fixture.wo
Normal file
36
tests/corpus/run/monitor-death/fixture.wo
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
use time
|
||||
|
||||
-- iteration 24 T4: actor death is OBSERVABLE. The observer names its own
|
||||
-- notice message; the watched actor trapping uncaught (the runtime's
|
||||
-- stderr line) delivers it. Monitoring an ALREADY dead actor fires
|
||||
-- immediately. WO_SHARDS=1 (the runner) keeps the order deterministic.
|
||||
class Note {
|
||||
who: Text
|
||||
}
|
||||
|
||||
class Watch {
|
||||
pad: Int
|
||||
fn receive(msg: Note) {
|
||||
print("died: ${msg.who}");
|
||||
}
|
||||
}
|
||||
|
||||
class Boom {
|
||||
pad: Int
|
||||
fn receive(msg: Note) {
|
||||
let z = len(msg.who) - len(msg.who);
|
||||
let q = 1 / z;
|
||||
}
|
||||
}
|
||||
|
||||
fn main() -> Int {
|
||||
let obs: actor Note = spawn Watch { pad: 0 };
|
||||
let b: actor Note = spawn Boom { pad: 0 };
|
||||
monitor(b, obs, Note { who: "boom" });
|
||||
send(b, Note { who: "x" });
|
||||
time.sleep(100);
|
||||
monitor(b, obs, Note { who: "late" });
|
||||
time.sleep(100);
|
||||
print("done");
|
||||
return 0;
|
||||
}
|
||||
3
tests/corpus/run/timer-delivery/fixture.out
Normal file
3
tests/corpus/run/timer-delivery/fixture.out
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
tick: now
|
||||
tick: armed
|
||||
done
|
||||
23
tests/corpus/run/timer-delivery/fixture.wo
Normal file
23
tests/corpus/run/timer-delivery/fixture.wo
Normal file
|
|
@ -0,0 +1,23 @@
|
|||
use time
|
||||
|
||||
-- iteration 24 T5: a timer is a MESSAGE. time.after arms a one-shot on
|
||||
-- this shard; the target receives it like any send. ms <= 0 delivers now.
|
||||
class Tick {
|
||||
tag: Text
|
||||
}
|
||||
|
||||
class Sink {
|
||||
pad: Int
|
||||
fn receive(msg: Tick) {
|
||||
print("tick: ${msg.tag}");
|
||||
}
|
||||
}
|
||||
|
||||
fn main() -> Int {
|
||||
let a: actor Tick = spawn Sink { pad: 0 };
|
||||
time.after(30, a, Tick { tag: "armed" });
|
||||
time.after(0, a, Tick { tag: "now" });
|
||||
time.sleep(150);
|
||||
print("done");
|
||||
return 0;
|
||||
}
|
||||
3
tests/corpus/run/timer-generation/fixture.out
Normal file
3
tests/corpus/run/timer-generation/fixture.out
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
stale gen 1 ignored
|
||||
fired gen 2
|
||||
done
|
||||
28
tests/corpus/run/timer-generation/fixture.wo
Normal file
28
tests/corpus/run/timer-generation/fixture.wo
Normal file
|
|
@ -0,0 +1,28 @@
|
|||
use time
|
||||
|
||||
-- iteration 24 T5: the CANCEL idiom — no cancel builtin, a generation
|
||||
-- counter instead. The actor bumps its generation; a stale timer's
|
||||
-- message names the old one and is recognized and ignored on arrival.
|
||||
class Timer {
|
||||
gen: Int
|
||||
}
|
||||
|
||||
class Gate {
|
||||
gen: Int
|
||||
fn receive(msg: Timer) {
|
||||
if msg.gen == self.gen {
|
||||
print("fired gen ${msg.gen}");
|
||||
} else {
|
||||
print("stale gen ${msg.gen} ignored");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn main() -> Int {
|
||||
let g: actor Timer = spawn Gate { gen: 2 };
|
||||
time.after(30, g, Timer { gen: 1 });
|
||||
time.after(60, g, Timer { gen: 2 });
|
||||
time.sleep(200);
|
||||
print("done");
|
||||
return 0;
|
||||
}
|
||||
Loading…
Reference in a new issue