Merge master into db-residency-doctrine — and close the two half-exposed features
The branch was 17 ahead / 25 behind with 11 conflicting files, and drifting further: db.c had been rewritten twice on master since (group commit, then compaction). Resolved rather than rebased so both histories stay legible. Conflicts, and how each was settled: - db.c: BOTH semantics kept. Master's fatal path and compaction check now sit behind the branch's `table_is_durable` predicate, in all three inline arms — a volatile table reaches neither the barrier nor the compaction check - db-bench sample: every mode from both sides (growth, growth-verify, randread, replayseed, wmix) and ONE `boot` mode, which both sides had added independently - db-bench.py: all six legs kept. Both sides had also grown the same WAL-size helper under different names; collapsed into one - perf-targets: the branch's §5 (RAM ceiling) then master's §6/§7 — master's numbering had already assumed a §5 it did not have - story frontmatter: master's `status` (the landing truth) plus the branch's `readiness` axis. 03 would have read `done` + `refine`, which is a contradiction — it was brainstormed and landed on master, so `ready` - board: both standup blocks newest-first; master's chain rows (a superset); the branch's databasev2 1-2 rows with master's 3-4. Fixed a stray `|` in master's row 3 - baseline: master's, then REGENERATED from a full campaign — 143 metrics, 132 checks, 0 failures with both sides' legs present TWO HALF-EXPOSED FEATURES FIXED, because the merge rule is that master gets no feature that is honoured in name only: - `resident: keys` PARSED, set a .wob flag, and did nothing: rows stayed fully resident. A developer could declare a 120 GB table keys-resident, watch it compile, and be OOM-killed. The loader now REFUSES it with a message naming what to write instead, until tasks 5c/5d land. The compiler still parses it and its AST golden still passes, so the grammar work stays tested - `durable: false` was honoured ONLY on the inline path. wo_db_exec_req had no guard at all, so a volatile table written from an actor on a worker shard would still be logged — precisely porch's session-table case, and precisely what iteration 2 exists to provide. All three request-path arms now carry the same predicate. Found by reading the merged code, not by a test: the obvious probe runs main() on the primary and therefore only exercises the inline path Verified on the merged tree: wovm-test 0, woc-test 0, oop-e2e 122/0, residency-accept 8/0, db-bench 132/0, linkcheck clean. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
commit
02b4b13a52
56 changed files with 5200 additions and 352 deletions
|
|
@ -1,22 +1,70 @@
|
||||||
{
|
{
|
||||||
"_config": {
|
"_config": {
|
||||||
"N": 2000,
|
"N": 20000,
|
||||||
"crash_reps": 1,
|
"crash_reps": 3,
|
||||||
"msg_n": 20000,
|
"msg_n": 200000,
|
||||||
"note": "refresh only with a commit that says why; tolerances come from tolerance_for() in the driver",
|
"note": "refresh only with a commit that says why; tolerances come from tolerance_for() in the driver",
|
||||||
"wal_n": 800
|
"wal_n": 4000
|
||||||
},
|
},
|
||||||
"ceiling.rows_recovered": {
|
"ceiling.rows_recovered": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 159744,
|
"floor": 159492,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 39936
|
"value": 39873
|
||||||
|
},
|
||||||
|
"ckpt.boot_off_ms": {
|
||||||
|
"dir": "lower",
|
||||||
|
"floor": 456,
|
||||||
|
"tolerance_pct": 400,
|
||||||
|
"value": 114
|
||||||
|
},
|
||||||
|
"ckpt.boot_on_ms": {
|
||||||
|
"dir": "lower",
|
||||||
|
"floor": 256,
|
||||||
|
"tolerance_pct": 400,
|
||||||
|
"value": 64
|
||||||
|
},
|
||||||
|
"ckpt.bytes_off": {
|
||||||
|
"dir": "lower",
|
||||||
|
"floor": 7876676,
|
||||||
|
"tolerance_pct": 400,
|
||||||
|
"value": 1969169
|
||||||
|
},
|
||||||
|
"ckpt.bytes_on": {
|
||||||
|
"dir": "lower",
|
||||||
|
"floor": 3696192,
|
||||||
|
"tolerance_pct": 400,
|
||||||
|
"value": 924048
|
||||||
|
},
|
||||||
|
"ckpt.compactions": {
|
||||||
|
"dir": "lower",
|
||||||
|
"floor": 100,
|
||||||
|
"tolerance_pct": 400,
|
||||||
|
"value": 6
|
||||||
|
},
|
||||||
|
"ckpt.pause_us_max": {
|
||||||
|
"dir": "lower",
|
||||||
|
"floor": 33912,
|
||||||
|
"tolerance_pct": 400,
|
||||||
|
"value": 8478
|
||||||
|
},
|
||||||
|
"ckpt.pause_us_per_mb": {
|
||||||
|
"dir": "lower",
|
||||||
|
"floor": 65848,
|
||||||
|
"tolerance_pct": 100,
|
||||||
|
"value": 16462
|
||||||
|
},
|
||||||
|
"ckpt.reclaim_x": {
|
||||||
|
"dir": "higher",
|
||||||
|
"floor": 0.0,
|
||||||
|
"tolerance_pct": 15,
|
||||||
|
"value": 2.13
|
||||||
},
|
},
|
||||||
"durable.s1.mixread.ops_sec": {
|
"durable.s1.mixread.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 2237,
|
"floor": 2452,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 8949
|
"value": 9809
|
||||||
},
|
},
|
||||||
"durable.s1.mixread.p50us": {
|
"durable.s1.mixread.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -28,31 +76,31 @@
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 100,
|
"floor": 100,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 2
|
"value": 12
|
||||||
},
|
},
|
||||||
"durable.s1.mixwrite.ops_sec": {
|
"durable.s1.mixwrite.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 248,
|
"floor": 272,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 994
|
"value": 1089
|
||||||
},
|
},
|
||||||
"durable.s1.mixwrite.p50us": {
|
"durable.s1.mixwrite.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 820,
|
"floor": 1704,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 205
|
"value": 426
|
||||||
},
|
},
|
||||||
"durable.s1.mixwrite.p99us": {
|
"durable.s1.mixwrite.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 872,
|
"floor": 1984,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 218
|
"value": 496
|
||||||
},
|
},
|
||||||
"durable.s1.query.ops_sec": {
|
"durable.s1.query.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 335570,
|
"floor": 306372,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 1342281
|
"value": 1225490
|
||||||
},
|
},
|
||||||
"durable.s1.query.p50us": {
|
"durable.s1.query.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -68,9 +116,9 @@
|
||||||
},
|
},
|
||||||
"durable.s1.read.ops_sec": {
|
"durable.s1.read.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 347705,
|
"floor": 307389,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 1390820
|
"value": 1229558
|
||||||
},
|
},
|
||||||
"durable.s1.read.p50us": {
|
"durable.s1.read.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -82,85 +130,121 @@
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 100,
|
"floor": 100,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 2
|
"value": 1
|
||||||
},
|
},
|
||||||
"durable.s1.seed.ops_sec": {
|
"durable.s1.seed.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 1103,
|
"floor": 1095,
|
||||||
"tolerance_pct": 15,
|
"tolerance_pct": 15,
|
||||||
"value": 4415
|
"value": 4381
|
||||||
},
|
},
|
||||||
"durable.s1.seed.p50us": {
|
"durable.s1.seed.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 828,
|
"floor": 848,
|
||||||
"tolerance_pct": 15,
|
"tolerance_pct": 15,
|
||||||
"value": 207
|
"value": 212
|
||||||
},
|
},
|
||||||
"durable.s1.seed.p99us": {
|
"durable.s1.seed.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 2092,
|
"floor": 2432,
|
||||||
"tolerance_pct": 15,
|
"tolerance_pct": 15,
|
||||||
"value": 523
|
"value": 608
|
||||||
|
},
|
||||||
|
"durable.s1.wmix.mean_batch": {
|
||||||
|
"dir": "higher",
|
||||||
|
"floor": 0.0,
|
||||||
|
"tolerance_pct": 100,
|
||||||
|
"value": 1.0
|
||||||
|
},
|
||||||
|
"durable.s1.wmix.ops_sec": {
|
||||||
|
"dir": "higher",
|
||||||
|
"floor": 402,
|
||||||
|
"tolerance_pct": 15,
|
||||||
|
"value": 1611
|
||||||
|
},
|
||||||
|
"durable.s1.wmix.p50us": {
|
||||||
|
"dir": "lower",
|
||||||
|
"floor": 1764,
|
||||||
|
"tolerance_pct": 15,
|
||||||
|
"value": 441
|
||||||
|
},
|
||||||
|
"durable.s1.wmix.p99us": {
|
||||||
|
"dir": "lower",
|
||||||
|
"floor": 2684,
|
||||||
|
"tolerance_pct": 15,
|
||||||
|
"value": 671
|
||||||
|
},
|
||||||
|
"durable.s1.wmix.peak_batch": {
|
||||||
|
"dir": "higher",
|
||||||
|
"floor": 0,
|
||||||
|
"tolerance_pct": 100,
|
||||||
|
"value": 1
|
||||||
|
},
|
||||||
|
"durable.s1.wmix.peak_staged": {
|
||||||
|
"dir": "lower",
|
||||||
|
"floor": 196,
|
||||||
|
"tolerance_pct": 100,
|
||||||
|
"value": 49
|
||||||
},
|
},
|
||||||
"durable.s1.write.ops_sec": {
|
"durable.s1.write.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 1155,
|
"floor": 573,
|
||||||
"tolerance_pct": 15,
|
"tolerance_pct": 15,
|
||||||
"value": 4620
|
"value": 2294
|
||||||
},
|
},
|
||||||
"durable.s1.write.p50us": {
|
"durable.s1.write.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 832,
|
"floor": 1760,
|
||||||
"tolerance_pct": 15,
|
"tolerance_pct": 15,
|
||||||
"value": 208
|
"value": 440
|
||||||
},
|
},
|
||||||
"durable.s1.write.p99us": {
|
"durable.s1.write.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 1948,
|
"floor": 2716,
|
||||||
"tolerance_pct": 15,
|
"tolerance_pct": 15,
|
||||||
"value": 487
|
"value": 679
|
||||||
},
|
},
|
||||||
"durable.sN.mixread.ops_sec": {
|
"durable.sN.mixread.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 1112,
|
"floor": 1183,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 4450
|
"value": 4733
|
||||||
},
|
},
|
||||||
"durable.sN.mixread.p50us": {
|
"durable.sN.mixread.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 236,
|
"floor": 244,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 59
|
"value": 61
|
||||||
},
|
},
|
||||||
"durable.sN.mixread.p99us": {
|
"durable.sN.mixread.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 13100,
|
"floor": 16200,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 300,
|
||||||
"value": 3275
|
"value": 4050
|
||||||
},
|
},
|
||||||
"durable.sN.mixwrite.ops_sec": {
|
"durable.sN.mixwrite.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 123,
|
"floor": 131,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 494
|
"value": 525
|
||||||
},
|
},
|
||||||
"durable.sN.mixwrite.p50us": {
|
"durable.sN.mixwrite.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 1160,
|
"floor": 2172,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 290
|
"value": 543
|
||||||
},
|
},
|
||||||
"durable.sN.mixwrite.p99us": {
|
"durable.sN.mixwrite.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 2944,
|
"floor": 16440,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 300,
|
||||||
"value": 736
|
"value": 4110
|
||||||
},
|
},
|
||||||
"durable.sN.query.ops_sec": {
|
"durable.sN.query.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 287356,
|
"floor": 308451,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 1149425
|
"value": 1233806
|
||||||
},
|
},
|
||||||
"durable.sN.query.p50us": {
|
"durable.sN.query.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -171,14 +255,14 @@
|
||||||
"durable.sN.query.p99us": {
|
"durable.sN.query.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 100,
|
"floor": 100,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 300,
|
||||||
"value": 1
|
"value": 1
|
||||||
},
|
},
|
||||||
"durable.sN.read.ops_sec": {
|
"durable.sN.read.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 192752,
|
"floor": 248188,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 771010
|
"value": 992752
|
||||||
},
|
},
|
||||||
"durable.sN.read.p50us": {
|
"durable.sN.read.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -189,44 +273,80 @@
|
||||||
"durable.sN.read.p99us": {
|
"durable.sN.read.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 100,
|
"floor": 100,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 300,
|
||||||
"value": 2
|
"value": 2
|
||||||
},
|
},
|
||||||
"durable.sN.seed.ops_sec": {
|
"durable.sN.seed.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 1142,
|
"floor": 1104,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 4571
|
"value": 4418
|
||||||
},
|
},
|
||||||
"durable.sN.seed.p50us": {
|
"durable.sN.seed.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 836,
|
"floor": 844,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 209
|
"value": 211
|
||||||
},
|
},
|
||||||
"durable.sN.seed.p99us": {
|
"durable.sN.seed.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 1916,
|
"floor": 2188,
|
||||||
|
"tolerance_pct": 300,
|
||||||
|
"value": 547
|
||||||
|
},
|
||||||
|
"durable.sN.wmix.mean_batch": {
|
||||||
|
"dir": "higher",
|
||||||
|
"floor": 1.0,
|
||||||
|
"tolerance_pct": 100,
|
||||||
|
"value": 6.22
|
||||||
|
},
|
||||||
|
"durable.sN.wmix.ops_sec": {
|
||||||
|
"dir": "higher",
|
||||||
|
"floor": 1504,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 479
|
"value": 6017
|
||||||
|
},
|
||||||
|
"durable.sN.wmix.p50us": {
|
||||||
|
"dir": "lower",
|
||||||
|
"floor": 27184,
|
||||||
|
"tolerance_pct": 50,
|
||||||
|
"value": 6796
|
||||||
|
},
|
||||||
|
"durable.sN.wmix.p99us": {
|
||||||
|
"dir": "lower",
|
||||||
|
"floor": 37484,
|
||||||
|
"tolerance_pct": 300,
|
||||||
|
"value": 9371
|
||||||
|
},
|
||||||
|
"durable.sN.wmix.peak_batch": {
|
||||||
|
"dir": "higher",
|
||||||
|
"floor": 15,
|
||||||
|
"tolerance_pct": 100,
|
||||||
|
"value": 60
|
||||||
|
},
|
||||||
|
"durable.sN.wmix.peak_staged": {
|
||||||
|
"dir": "lower",
|
||||||
|
"floor": 11760,
|
||||||
|
"tolerance_pct": 100,
|
||||||
|
"value": 2940
|
||||||
},
|
},
|
||||||
"durable.sN.write.ops_sec": {
|
"durable.sN.write.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 1010,
|
"floor": 580,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 4040
|
"value": 2320
|
||||||
},
|
},
|
||||||
"durable.sN.write.p50us": {
|
"durable.sN.write.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 832,
|
"floor": 1760,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 208
|
"value": 440
|
||||||
},
|
},
|
||||||
"durable.sN.write.p99us": {
|
"durable.sN.write.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 1996,
|
"floor": 2688,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 300,
|
||||||
"value": 499
|
"value": 672
|
||||||
},
|
},
|
||||||
"growth.available": {
|
"growth.available": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -236,9 +356,9 @@
|
||||||
},
|
},
|
||||||
"growth.int.noswap.bytes_per_row": {
|
"growth.int.noswap.bytes_per_row": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 392,
|
"floor": 440,
|
||||||
"tolerance_pct": 10,
|
"tolerance_pct": 10,
|
||||||
"value": 98
|
"value": 110
|
||||||
},
|
},
|
||||||
"growth.int.noswap.doublings": {
|
"growth.int.noswap.doublings": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -266,21 +386,21 @@
|
||||||
},
|
},
|
||||||
"growth.int.noswap.rows": {
|
"growth.int.noswap.rows": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 80000,
|
"floor": 800000,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 20000
|
"value": 200000
|
||||||
},
|
},
|
||||||
"growth.int.noswap.rss_kb": {
|
"growth.int.noswap.rss_kb": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 23968,
|
"floor": 168528,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 5992
|
"value": 42132
|
||||||
},
|
},
|
||||||
"growth.int.swap.bytes_per_row": {
|
"growth.int.swap.bytes_per_row": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 392,
|
"floor": 440,
|
||||||
"tolerance_pct": 10,
|
"tolerance_pct": 10,
|
||||||
"value": 98
|
"value": 110
|
||||||
},
|
},
|
||||||
"growth.int.swap.doublings": {
|
"growth.int.swap.doublings": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -308,21 +428,21 @@
|
||||||
},
|
},
|
||||||
"growth.int.swap.rows": {
|
"growth.int.swap.rows": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 80000,
|
"floor": 800000,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 20000
|
"value": 200000
|
||||||
},
|
},
|
||||||
"growth.int.swap.rss_kb": {
|
"growth.int.swap.rss_kb": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 23984,
|
"floor": 168576,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 5996
|
"value": 42144
|
||||||
},
|
},
|
||||||
"growth.text.noswap.bytes_per_row": {
|
"growth.text.noswap.bytes_per_row": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 1288,
|
"floor": 1284,
|
||||||
"tolerance_pct": 10,
|
"tolerance_pct": 10,
|
||||||
"value": 322
|
"value": 321
|
||||||
},
|
},
|
||||||
"growth.text.noswap.doublings": {
|
"growth.text.noswap.doublings": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -350,21 +470,21 @@
|
||||||
},
|
},
|
||||||
"growth.text.noswap.rows": {
|
"growth.text.noswap.rows": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 80000,
|
"floor": 800000,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 20000
|
"value": 200000
|
||||||
},
|
},
|
||||||
"growth.text.noswap.rss_kb": {
|
"growth.text.noswap.rss_kb": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 41216,
|
"floor": 343312,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 10304
|
"value": 85828
|
||||||
},
|
},
|
||||||
"growth.text.swap.bytes_per_row": {
|
"growth.text.swap.bytes_per_row": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 1288,
|
"floor": 1284,
|
||||||
"tolerance_pct": 10,
|
"tolerance_pct": 10,
|
||||||
"value": 322
|
"value": 321
|
||||||
},
|
},
|
||||||
"growth.text.swap.doublings": {
|
"growth.text.swap.doublings": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -392,21 +512,21 @@
|
||||||
},
|
},
|
||||||
"growth.text.swap.rows": {
|
"growth.text.swap.rows": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 80000,
|
"floor": 800000,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 20000
|
"value": 200000
|
||||||
},
|
},
|
||||||
"growth.text.swap.rss_kb": {
|
"growth.text.swap.rss_kb": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 41216,
|
"floor": 343328,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 10304
|
"value": 85832
|
||||||
},
|
},
|
||||||
"ram.s1.mixread.ops_sec": {
|
"ram.s1.mixread.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 2236,
|
"floor": 22286,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 8947
|
"value": 89144
|
||||||
},
|
},
|
||||||
"ram.s1.mixread.p50us": {
|
"ram.s1.mixread.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -422,9 +542,9 @@
|
||||||
},
|
},
|
||||||
"ram.s1.mixwrite.ops_sec": {
|
"ram.s1.mixwrite.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 248,
|
"floor": 2476,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 994
|
"value": 9904
|
||||||
},
|
},
|
||||||
"ram.s1.mixwrite.p50us": {
|
"ram.s1.mixwrite.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -440,15 +560,15 @@
|
||||||
},
|
},
|
||||||
"ram.s1.msgrate.msgs_sec": {
|
"ram.s1.msgrate.msgs_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 419322,
|
"floor": 1336469,
|
||||||
"tolerance_pct": 15,
|
"tolerance_pct": 70,
|
||||||
"value": 3354579
|
"value": 10691756
|
||||||
},
|
},
|
||||||
"ram.s1.query.ops_sec": {
|
"ram.s1.query.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 324675,
|
"floor": 244857,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 1298701
|
"value": 979431
|
||||||
},
|
},
|
||||||
"ram.s1.query.p50us": {
|
"ram.s1.query.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -464,9 +584,9 @@
|
||||||
},
|
},
|
||||||
"ram.s1.read.ops_sec": {
|
"ram.s1.read.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 332889,
|
"floor": 252270,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 1331557
|
"value": 1009081
|
||||||
},
|
},
|
||||||
"ram.s1.read.p50us": {
|
"ram.s1.read.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -482,87 +602,87 @@
|
||||||
},
|
},
|
||||||
"ram.s1.seed.ops_sec": {
|
"ram.s1.seed.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 375939,
|
"floor": 62904,
|
||||||
"tolerance_pct": 15,
|
"tolerance_pct": 15,
|
||||||
"value": 1503759
|
"value": 251616
|
||||||
},
|
},
|
||||||
"ram.s1.seed.p50us": {
|
"ram.s1.seed.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 100,
|
"floor": 100,
|
||||||
"tolerance_pct": 15,
|
"tolerance_pct": 15,
|
||||||
"value": 0
|
"value": 4
|
||||||
},
|
},
|
||||||
"ram.s1.seed.p99us": {
|
"ram.s1.seed.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 100,
|
"floor": 100,
|
||||||
"tolerance_pct": 15,
|
"tolerance_pct": 15,
|
||||||
"value": 2
|
"value": 9
|
||||||
},
|
},
|
||||||
"ram.s1.write.ops_sec": {
|
"ram.s1.write.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 272628,
|
"floor": 47770,
|
||||||
"tolerance_pct": 15,
|
"tolerance_pct": 15,
|
||||||
"value": 1090512
|
"value": 191080
|
||||||
},
|
},
|
||||||
"ram.s1.write.p50us": {
|
"ram.s1.write.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 100,
|
"floor": 100,
|
||||||
"tolerance_pct": 15,
|
"tolerance_pct": 15,
|
||||||
"value": 1
|
"value": 8
|
||||||
},
|
},
|
||||||
"ram.s1.write.p99us": {
|
"ram.s1.write.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 100,
|
"floor": 100,
|
||||||
"tolerance_pct": 15,
|
"tolerance_pct": 15,
|
||||||
"value": 2
|
"value": 10
|
||||||
},
|
},
|
||||||
"ram.sN.mixread.ops_sec": {
|
"ram.sN.mixread.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 2241,
|
"floor": 11218,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 8964
|
"value": 44874
|
||||||
},
|
},
|
||||||
"ram.sN.mixread.p50us": {
|
"ram.sN.mixread.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 228,
|
"floor": 240,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 57
|
"value": 60
|
||||||
},
|
},
|
||||||
"ram.sN.mixread.p99us": {
|
"ram.sN.mixread.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 1412,
|
"floor": 324,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 353
|
"value": 81
|
||||||
},
|
},
|
||||||
"ram.sN.mixwrite.ops_sec": {
|
"ram.sN.mixwrite.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 249,
|
"floor": 1246,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 996
|
"value": 4986
|
||||||
},
|
},
|
||||||
"ram.sN.mixwrite.p50us": {
|
"ram.sN.mixwrite.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 252,
|
"floor": 260,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 63
|
"value": 65
|
||||||
},
|
},
|
||||||
"ram.sN.mixwrite.p99us": {
|
"ram.sN.mixwrite.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 280,
|
"floor": 356,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 70
|
"value": 89
|
||||||
},
|
},
|
||||||
"ram.sN.msgrate.msgs_sec": {
|
"ram.sN.msgrate.msgs_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 214795,
|
"floor": 317323,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 70,
|
||||||
"value": 1718360
|
"value": 2538586
|
||||||
},
|
},
|
||||||
"ram.sN.query.ops_sec": {
|
"ram.sN.query.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 331125,
|
"floor": 291545,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 1324503
|
"value": 1166180
|
||||||
},
|
},
|
||||||
"ram.sN.query.p50us": {
|
"ram.sN.query.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -578,9 +698,9 @@
|
||||||
},
|
},
|
||||||
"ram.sN.read.ops_sec": {
|
"ram.sN.read.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 340599,
|
"floor": 317823,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 1362397
|
"value": 1271294
|
||||||
},
|
},
|
||||||
"ram.sN.read.p50us": {
|
"ram.sN.read.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -596,87 +716,87 @@
|
||||||
},
|
},
|
||||||
"ram.sN.seed.ops_sec": {
|
"ram.sN.seed.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 353606,
|
"floor": 73305,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 1414427
|
"value": 293220
|
||||||
},
|
},
|
||||||
"ram.sN.seed.p50us": {
|
"ram.sN.seed.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 100,
|
"floor": 100,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 1
|
"value": 3
|
||||||
},
|
},
|
||||||
"ram.sN.seed.p99us": {
|
"ram.sN.seed.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 100,
|
"floor": 100,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 2
|
"value": 7
|
||||||
},
|
},
|
||||||
"ram.sN.write.ops_sec": {
|
"ram.sN.write.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 290697,
|
"floor": 56810,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 1162790
|
"value": 227241
|
||||||
},
|
},
|
||||||
"ram.sN.write.p50us": {
|
"ram.sN.write.p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 100,
|
"floor": 100,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 1
|
"value": 6
|
||||||
},
|
},
|
||||||
"ram.sN.write.p99us": {
|
"ram.sN.write.p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 100,
|
"floor": 100,
|
||||||
"tolerance_pct": 50,
|
"tolerance_pct": 50,
|
||||||
"value": 2
|
"value": 10
|
||||||
},
|
},
|
||||||
"randread.collapse_x": {
|
"randread.collapse_x": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 1172,
|
"floor": 1084,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 293
|
"value": 271
|
||||||
},
|
},
|
||||||
"randread.overcap.filled_rss_kb": {
|
"randread.overcap.filled_rss_kb": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 25680,
|
"floor": 58144,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 6420
|
"value": 14536
|
||||||
},
|
},
|
||||||
"randread.overcap.ops_sec": {
|
"randread.overcap.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 1665,
|
"floor": 1427,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 6661
|
"value": 5711
|
||||||
},
|
},
|
||||||
"randread.overcap.read_p50us": {
|
"randread.overcap.read_p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 556,
|
"floor": 624,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 139
|
"value": 156
|
||||||
},
|
},
|
||||||
"randread.overcap.read_p99us": {
|
"randread.overcap.read_p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 1920,
|
"floor": 1628,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 480
|
"value": 407
|
||||||
},
|
},
|
||||||
"randread.resident.filled_rss_kb": {
|
"randread.resident.filled_rss_kb": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 54080,
|
"floor": 168288,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 13520
|
"value": 42072
|
||||||
},
|
},
|
||||||
"randread.resident.ops_sec": {
|
"randread.resident.ops_sec": {
|
||||||
"dir": "higher",
|
"dir": "higher",
|
||||||
"floor": 488424,
|
"floor": 387281,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 1953697
|
"value": 1549126
|
||||||
},
|
},
|
||||||
"randread.resident.read_p50us": {
|
"randread.resident.read_p50us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 100,
|
"floor": 100,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 0
|
"value": 1
|
||||||
},
|
},
|
||||||
"randread.resident.read_p99us": {
|
"randread.resident.read_p99us": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
@ -686,57 +806,57 @@
|
||||||
},
|
},
|
||||||
"replay.history.ms": {
|
"replay.history.ms": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 844,
|
"floor": 12876,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 211
|
"value": 3219
|
||||||
},
|
},
|
||||||
"replay.history.ns_per_record": {
|
"replay.history.ns_per_record": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 21068,
|
"floor": 64384,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 5267
|
"value": 16096
|
||||||
},
|
},
|
||||||
"replay.history.records": {
|
"replay.history.records": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 160000,
|
"floor": 800000,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 40000
|
"value": 200000
|
||||||
},
|
},
|
||||||
"replay.history.wal_bytes": {
|
"replay.history.wal_bytes": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 7840140,
|
"floor": 39200140,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 1960035
|
"value": 9800035
|
||||||
},
|
},
|
||||||
"replay.history_penalty_x": {
|
"replay.history_penalty_x": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 100,
|
"floor": 100,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 1.9
|
"value": 1.6
|
||||||
},
|
},
|
||||||
"replay.inserts.ms": {
|
"replay.inserts.ms": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 444,
|
"floor": 8064,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 111
|
"value": 2016
|
||||||
},
|
},
|
||||||
"replay.inserts.ns_per_record": {
|
"replay.inserts.ns_per_record": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 22120,
|
"floor": 80656,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 5530
|
"value": 20164
|
||||||
},
|
},
|
||||||
"replay.inserts.records": {
|
"replay.inserts.records": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 80000,
|
"floor": 400000,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 20000
|
"value": 100000
|
||||||
},
|
},
|
||||||
"replay.inserts.wal_bytes": {
|
"replay.inserts.wal_bytes": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
"floor": 3920140,
|
"floor": 19600140,
|
||||||
"tolerance_pct": 100,
|
"tolerance_pct": 100,
|
||||||
"value": 980035
|
"value": 4900035
|
||||||
},
|
},
|
||||||
"replay.startup_ms": {
|
"replay.startup_ms": {
|
||||||
"dir": "lower",
|
"dir": "lower",
|
||||||
|
|
|
||||||
|
|
@ -295,6 +295,7 @@ let b_sha1 = 85
|
||||||
let b_sha256 = 86
|
let b_sha256 = 86
|
||||||
let b_hmac_sha256 = 87
|
let b_hmac_sha256 = 87
|
||||||
let b_call = 88
|
let b_call = 88
|
||||||
|
let b_monitor = 89
|
||||||
let b_split = 28
|
let b_split = 28
|
||||||
let b_split_ws = 29
|
let b_split_ws = 29
|
||||||
let b_join = 30
|
let b_join = 30
|
||||||
|
|
@ -1110,7 +1111,7 @@ let is_builtin_name (n : string) =
|
||||||
"substr"; "trim"; "to_lower"; "char_of"; "parse_int"; "split"; "split_ws"; "join"; "slice";
|
"substr"; "trim"; "to_lower"; "char_of"; "parse_int"; "split"; "split_ws"; "join"; "slice";
|
||||||
"pop"; "shift"; "sort"; "reverse"; "remove"; "key_at"; "val_at";
|
"pop"; "shift"; "sort"; "reverse"; "remove"; "key_at"; "val_at";
|
||||||
(* the concurrency arc *)
|
(* the concurrency arc *)
|
||||||
"send"; "call";
|
"send"; "call"; "monitor";
|
||||||
(* iteration 19: Float bridges and Bytes surface *)
|
(* iteration 19: Float bridges and Bytes surface *)
|
||||||
"float"; "trunc"; "parse_float"; "float_to_text"; "float_cmp"; "bytes_len"; "bytes_at";
|
"float"; "trunc"; "parse_float"; "float_to_text"; "float_cmp"; "bytes_len"; "bytes_at";
|
||||||
"bytes_slice"; "bytes_eq"; "bytes_concat"; "base64_encode"; "base64_decode";
|
"bytes_slice"; "bytes_eq"; "bytes_concat"; "base64_encode"; "base64_decode";
|
||||||
|
|
@ -3441,11 +3442,18 @@ and emit_call (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e : As
|
||||||
put f (ins_abc op_builtin dst base sm.Types.sm_builtin);
|
put f (ins_abc op_builtin dst base sm.Types.sm_builtin);
|
||||||
(* every stdlib member only READS its arguments, so one that was
|
(* every stdlib member only READS its arguments, so one that was
|
||||||
freshly built here (`net.write(c, head .. resp.body)`) has no
|
freshly built here (`net.write(c, head .. resp.body)`) has no
|
||||||
other owner and dies with the call *)
|
other owner and dies with the call. The ONE exception:
|
||||||
|
`time.after`'s message (arg 2) MOVES to the runtime — the
|
||||||
|
timer owns it until delivery (iteration 24 T5). *)
|
||||||
|
let moves i =
|
||||||
|
alias = "time" && mname = "after" && i = 2
|
||||||
|
in
|
||||||
List.iteri
|
List.iteri
|
||||||
(fun i (a : Ast.expr) ->
|
(fun i (a : Ast.expr) ->
|
||||||
drop_fresh_owned ~keep:dst p f (base + i) a;
|
if not (moves i) then begin
|
||||||
drop_fresh_text ~keep:dst p f (base + i) a)
|
drop_fresh_owned ~keep:dst p f (base + i) a;
|
||||||
|
drop_fresh_text ~keep:dst p f (base + i) a
|
||||||
|
end)
|
||||||
args
|
args
|
||||||
end)
|
end)
|
||||||
| Some u -> (
|
| Some u -> (
|
||||||
|
|
@ -3722,7 +3730,7 @@ and emit_builtin (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e :
|
||||||
dangle the value just read) and the stores, which either copy (Text,
|
dangle the value just read) and the stores, which either copy (Text,
|
||||||
handled by copied_container_call) or take ownership (OWNED/GCREF). *)
|
handled by copied_container_call) or take ownership (OWNED/GCREF). *)
|
||||||
let reader = List.mem name [ "get"; "latest"; "key_at"; "val_at" ] in
|
let reader = List.mem name [ "get"; "latest"; "key_at"; "val_at" ] in
|
||||||
(if not (List.mem name [ "push"; "set"; "send"; "call" ]) then
|
(if not (List.mem name [ "push"; "set"; "send"; "call"; "monitor" ]) then
|
||||||
List.iteri
|
List.iteri
|
||||||
(fun i (a : Ast.expr) ->
|
(fun i (a : Ast.expr) ->
|
||||||
(* a reader's result points into arg0 (the container) — dropping
|
(* a reader's result points into arg0 (the container) — dropping
|
||||||
|
|
@ -3754,6 +3762,7 @@ and emit_builtin (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e :
|
||||||
match name with
|
match name with
|
||||||
| "send" -> fixed b_send (* arc: msg (arg1) moved to the runtime — never dropped here *)
|
| "send" -> fixed b_send (* arc: msg (arg1) moved to the runtime — never dropped here *)
|
||||||
| "call" -> fixed b_call (* iteration 24: same move; the SCALAR reply lands in dst *)
|
| "call" -> fixed b_call (* iteration 24: same move; the SCALAR reply lands in dst *)
|
||||||
|
| "monitor" -> fixed b_monitor (* T4: notice msg (arg2) moves to the runtime *)
|
||||||
| "now" -> fixed b_now
|
| "now" -> fixed b_now
|
||||||
| "print" -> fixed b_print
|
| "print" -> fixed b_print
|
||||||
| "print_int" -> fixed b_print_int
|
| "print_int" -> fixed b_print_int
|
||||||
|
|
|
||||||
|
|
@ -1348,6 +1348,10 @@ and analyze_call (ctx : ctx) (call_e : Ast.expr) (callee : Ast.expr) (args : Ast
|
||||||
iteration 24: call(addr, msg) moves its message identically. *)
|
iteration 24: call(addr, msg) moves its message identically. *)
|
||||||
| Ident "send" -> i = 1 && Types.StringMap.find_opt "send" ctx.syms.Types.free_fns = None
|
| Ident "send" -> i = 1 && Types.StringMap.find_opt "send" ctx.syms.Types.free_fns = None
|
||||||
| Ident "call" -> i = 1 && Types.StringMap.find_opt "call" ctx.syms.Types.free_fns = None
|
| Ident "call" -> i = 1 && Types.StringMap.find_opt "call" ctx.syms.Types.free_fns = None
|
||||||
|
(* T4/T5: the notice / timer message moves to the runtime too *)
|
||||||
|
| Ident "monitor" ->
|
||||||
|
i = 2 && Types.StringMap.find_opt "monitor" ctx.syms.Types.free_fns = None
|
||||||
|
| Field ({ kind = Ident "time"; _ }, "after") -> i = 2
|
||||||
| _ -> false
|
| _ -> false
|
||||||
in
|
in
|
||||||
List.iteri
|
List.iteri
|
||||||
|
|
@ -1365,7 +1369,8 @@ and analyze_call (ctx : ctx) (call_e : Ast.expr) (callee : Ast.expr) (args : Ast
|
||||||
transfer ctx p
|
transfer ctx p
|
||||||
~what:
|
~what:
|
||||||
(match callee.kind with
|
(match callee.kind with
|
||||||
| Ident "send" | Ident "call" ->
|
| Ident "send" | Ident "call" | Ident "monitor"
|
||||||
|
| Field ({ kind = Ident "time"; _ }, "after") ->
|
||||||
"cannot be sent — a message moves to the receiver"
|
"cannot be sent — a message moves to the receiver"
|
||||||
| _ -> "cannot be stored in a container")
|
| _ -> "cannot be stored in a container")
|
||||||
then record_move ctx p (MvArg "element"))
|
then record_move ctx p (MvArg "element"))
|
||||||
|
|
|
||||||
|
|
@ -309,6 +309,8 @@ let stdlib_members : stdlib_member list =
|
||||||
m "net" "write_dl" 3 93 (Some (TScalar "Bool")) None;
|
m "net" "write_dl" 3 93 (Some (TScalar "Bool")) None;
|
||||||
m "net" "listen_unix" 1 94 (Some (TScalar "Int")) None;
|
m "net" "listen_unix" 1 94 (Some (TScalar "Int")) None;
|
||||||
m "net" "peer" 1 95 (Some (TScalar "Text")) None;
|
m "net" "peer" 1 95 (Some (TScalar "Text")) None;
|
||||||
|
(* iteration 24 T5: one-shot timer — the msg MOVES to the runtime *)
|
||||||
|
m "time" "after" 3 90 None None;
|
||||||
(* proc *)
|
(* proc *)
|
||||||
m "proc" "run" 2 56 (Some (TNullable (TScalar proc_record_name))) (Some proc_record_name);
|
m "proc" "run" 2 56 (Some (TNullable (TScalar proc_record_name))) (Some proc_record_name);
|
||||||
(* json — both members are lowered specially (emit.ml): encode needs its
|
(* json — both members are lowered specially (emit.ml): encode needs its
|
||||||
|
|
@ -1917,6 +1919,48 @@ let typecheck_program ~file ~(module_of : string -> string)
|
||||||
~message:"`call`'s first argument must be an `actor M` address" ())
|
~message:"`call`'s first argument must be an `actor M` address" ())
|
||||||
| None -> ())
|
| None -> ())
|
||||||
| _ -> ())
|
| _ -> ())
|
||||||
|
| None when name = "monitor" ->
|
||||||
|
(* iteration 24 T4: monitor(watched, observer, msg) — the
|
||||||
|
notice msg is typed against the OBSERVER's mailbox
|
||||||
|
(three-argument form: the caller may be main, which has
|
||||||
|
no mailbox). msg moves like send's. *)
|
||||||
|
(if List.length args <> 3 then
|
||||||
|
Diag.Collector.add collector
|
||||||
|
(Diag.error ~code:bad_arity_code ~file ~line:e.pos.line ~col:e.pos.col
|
||||||
|
~message:
|
||||||
|
(Printf.sprintf
|
||||||
|
"`monitor` takes 3 arguments (watched, observer, notice), given %d"
|
||||||
|
(List.length args))
|
||||||
|
())
|
||||||
|
else
|
||||||
|
match args with
|
||||||
|
| [ w; o; m ] -> (
|
||||||
|
(match confident_typ cenv w with
|
||||||
|
| Some (TActor _) | None -> ()
|
||||||
|
| Some _ ->
|
||||||
|
Diag.Collector.add collector
|
||||||
|
(Diag.error ~code:type_mismatch_code ~file ~line:w.pos.line
|
||||||
|
~col:w.pos.col
|
||||||
|
~message:"`monitor`'s first argument must be an `actor M` address" ()));
|
||||||
|
match confident_typ cenv o with
|
||||||
|
| Some (TActor want) -> (
|
||||||
|
match confident_typ cenv m with
|
||||||
|
| Some (TScalar got) when got <> want ->
|
||||||
|
Diag.Collector.add collector
|
||||||
|
(Diag.error ~code:type_mismatch_code ~file ~line:m.pos.line
|
||||||
|
~col:m.pos.col
|
||||||
|
~message:
|
||||||
|
(Printf.sprintf
|
||||||
|
"the observer receives `%s` — the notice is a `%s`" want got)
|
||||||
|
())
|
||||||
|
| _ -> ())
|
||||||
|
| Some _ ->
|
||||||
|
Diag.Collector.add collector
|
||||||
|
(Diag.error ~code:type_mismatch_code ~file ~line:o.pos.line
|
||||||
|
~col:o.pos.col
|
||||||
|
~message:"`monitor`'s second argument must be an `actor M` address" ())
|
||||||
|
| None -> ())
|
||||||
|
| _ -> ())
|
||||||
| None ->
|
| None ->
|
||||||
let confident_types = List.map (confident_typ cenv) args in
|
let confident_types = List.map (confident_typ cenv) args in
|
||||||
check_builtin_call ~file collector name e.pos args confident_types)
|
check_builtin_call ~file collector name e.pos args confident_types)
|
||||||
|
|
|
||||||
|
|
@ -119,3 +119,135 @@ rather than acknowledging what disk never got.
|
||||||
columns excluded (engine raw-eq is narrower than VM float-eq, and a
|
columns excluded (engine raw-eq is narrower than VM float-eq, and a
|
||||||
probe miss cannot be resurrected by a recheck). Pinned by
|
probe miss cannot be resurrected by a recheck). Pinned by
|
||||||
`tests/corpus/run/query-index-probe`.
|
`tests/corpus/run/query-index-probe`.
|
||||||
|
|
||||||
|
## Group commit: one barrier per drain (databasev2 4 part A, 2026-08-28)
|
||||||
|
|
||||||
|
**What changed:** the engine used to commit per *statement*. `db.c` called
|
||||||
|
`wo_wal_commit` immediately after every append, at all six sites, so each row
|
||||||
|
change bought its own `pwrite` and its own `fdatasync`. Now the barrier belongs
|
||||||
|
to the drain, not to the statement.
|
||||||
|
|
||||||
|
**Where the barrier runs, and why there.** A statement on a worker shard has no
|
||||||
|
WAL to write — the runtime asserts workers hold neither `db` nor `wal` — so it
|
||||||
|
marshals to shard 0 and parks. Shard 0 executes those requests in its envelope
|
||||||
|
drain (`wo_vm_adopt`), and the drain now **holds each reply** instead of pushing
|
||||||
|
it as the statement finishes. When the queue empties it issues one barrier, then
|
||||||
|
releases every held reply.
|
||||||
|
|
||||||
|
Holding the reply is the whole mechanism. Pushing it early would unpark the
|
||||||
|
requester before its record was durable; holding it means each writer is
|
||||||
|
acknowledged after the barrier that carried *its own* record. That was always
|
||||||
|
the intended contract — it was simply true by accident before, because every
|
||||||
|
batch had exactly one member.
|
||||||
|
|
||||||
|
**Why the queue is the boundary.** Not a tick, and not a timer. A queue of one
|
||||||
|
gives a batch of one, so a lone writer pays exactly what it paid before; the
|
||||||
|
batch grows only when writes genuinely contend. A tick boundary would have
|
||||||
|
added latency even with nothing to batch against, which is taxing an idle
|
||||||
|
system to serve a busy one. There is nothing to tune, which is the point.
|
||||||
|
|
||||||
|
**Why the inline path is asymmetric.** A statement already on shard 0 stages and
|
||||||
|
commits before returning, batch size one. It cannot hold a reply because there
|
||||||
|
is nobody to reply to — it returns into its own fiber. Batching it would mean
|
||||||
|
parking that fiber on the barrier, which is part B's machinery. Two consequences
|
||||||
|
worth keeping in mind: single-shard configurations get no batching at all, by
|
||||||
|
design; and the inline commit is only safe because the drain commits
|
||||||
|
*unconditionally* whenever anything is staged, so the buffer is empty when an
|
||||||
|
inline statement runs. If that ever stops holding, the inline path would make
|
||||||
|
another statement's record durable early and acknowledge it to the wrong writer.
|
||||||
|
|
||||||
|
**One rule for failure: once a statement has mutated RAM, the outcomes are
|
||||||
|
durable or process death.** It replaced three behaviours that disagreed —
|
||||||
|
`insert` un-applied itself, while `update` and `delete` returned a catchable
|
||||||
|
trap and left RAM ahead of disk, which their own comments said out loud.
|
||||||
|
Batching would have multiplied that from one row to a whole batch. So a failed
|
||||||
|
stage or a failed barrier now prints one diagnostic (operation, log path,
|
||||||
|
`errno`, record count) and exits 3; `WO_T_IO` is unreachable from a write.
|
||||||
|
Retrying is not offered because it is unsound: on Linux a failed `fsync` may
|
||||||
|
already have discarded the dirty pages, so a second call can report success
|
||||||
|
having written nothing. Replay is the recovery that works.
|
||||||
|
|
||||||
|
**Measuring it.** `WO_WAL_STATS=1` makes the runtime print one line at exit —
|
||||||
|
batches, records, peak batch, peak staged bytes. Opt-in, because it would
|
||||||
|
otherwise pollute every durable program's output. The counters live in `wo_wal`
|
||||||
|
rather than behind a builtin: they are diagnostic, not part of the language.
|
||||||
|
`db-bench`'s `wmix N C` leg exists to exercise this at all — `mix` writes on one
|
||||||
|
op in ten with C=4, which produced a measured mean batch of 1.01, so it could
|
||||||
|
never have shown whether batching worked.
|
||||||
|
|
||||||
|
**If you are looking at this because writes got slower**, check the mean batch
|
||||||
|
first. Mean 1.0 means the mechanism is not engaging, which is expected for a
|
||||||
|
serial writer or a single-shard configuration and a bug anywhere else.
|
||||||
|
|
||||||
|
## Checkpoint: compaction by rewrite + rename (databasev2 3, 2026-08-29)
|
||||||
|
|
||||||
|
**The problem:** nothing ever removed superseded records, so the log grew
|
||||||
|
forever and boot replayed all history. Measured before this: 20 000 rows seeded
|
||||||
|
gave a 986 KB log; updating those same rows 20 000 times took it to 2.6 MB with
|
||||||
|
**the same live data**.
|
||||||
|
|
||||||
|
**Why one file and not a snapshot plus a tail.** Postgres does the opposite —
|
||||||
|
its WAL is a redo tail and the data lives in heap files, so a checkpoint flushes
|
||||||
|
pages and then recycles log segments; it never compacts. It cannot: its records
|
||||||
|
are page deltas, so a compacted redo log is not a store. **Ours are full row
|
||||||
|
images** — `apply_record` implements UPDATE as remove-then-recreate — so a log
|
||||||
|
of one record per live row *is* a complete store. That single difference deletes
|
||||||
|
the control file, the redo pointer, the second recovery source and the separate
|
||||||
|
process from this design. Recovery is not merely compatible with compaction; it
|
||||||
|
is completely unaware of it.
|
||||||
|
|
||||||
|
**Why `rename` is the whole crash-safety story.** The dump goes to a temp file,
|
||||||
|
which is fsynced, renamed over the live log, and then the parent directory is
|
||||||
|
fsynced (the rename is atomic in-kernel, but the directory entry is not durable
|
||||||
|
until the parent is — Postgres does the same for the same reason). Before the
|
||||||
|
rename the live log is intact and the temp is not authoritative; after it the new
|
||||||
|
log is complete. There is no instant at which a reader sees a mixture, so this
|
||||||
|
needs no recovery logic of its own. What Postgres achieves with a redo pointer
|
||||||
|
computed at checkpoint start and a control file written at the end, one syscall
|
||||||
|
achieves here — because we can swap the entire data set atomically and Postgres
|
||||||
|
cannot.
|
||||||
|
|
||||||
|
A crash mid-rewrite leaves a temp file. The next open **removes it**, and it is
|
||||||
|
deleted rather than ignored because a file full of well-formed records sitting
|
||||||
|
beside the log is exactly what a later reader mistakes for data.
|
||||||
|
|
||||||
|
**Why the dump flushes periodically, and why it does NOT fsync when it does.**
|
||||||
|
`stage()` grows the staging buffer by doubling and never shrinks it, so pushing a
|
||||||
|
whole store through one buffer would hold the entire store in RAM on top of the
|
||||||
|
store — the unbounded growth databasev2 1 measured as how this engine dies. So
|
||||||
|
the dump flushes every 256 records. It flushes with a plain write, **not** a
|
||||||
|
commit: intermediate durability is worthless because the temp is not
|
||||||
|
authoritative until the rename and is fsynced once immediately before it. Using
|
||||||
|
the committing path cost one barrier per 256 records and made the pause 8×
|
||||||
|
larger — measured 107 649 µs against 13 212 µs for a 2 MB live set, ~22 MB/s
|
||||||
|
against ~181 MB/s.
|
||||||
|
|
||||||
|
**Why the replacement is preallocated like the original.** The WAL is
|
||||||
|
preallocated so that appends never extend the file, which is what lets
|
||||||
|
`fdatasync` alone serve as the ack barrier. A replacement opened without it
|
||||||
|
would silently change that property, and the zero-padded tail the open-time scan
|
||||||
|
relies on.
|
||||||
|
|
||||||
|
**When it runs.** Only where the staging buffer is empty — right after a
|
||||||
|
barrier. Both write paths check: the drain (`vm.c`, after its commit and after
|
||||||
|
releasing held replies, since those records are already durable and should not
|
||||||
|
wait out a rewrite) and the inline path (`db.c`). Wiring only the drain left
|
||||||
|
`WO_SHARDS=1` never compacting, with its log growing forever: measured 536 KB
|
||||||
|
where the multi-shard run held 446 KB.
|
||||||
|
|
||||||
|
**The trigger** compares the log against what the *last* compaction actually
|
||||||
|
wrote, with an absolute floor. The denominator is measured rather than
|
||||||
|
estimated, because estimating the live size means estimating Text and the
|
||||||
|
compactor already knows the true number. There is deliberately **no timer**:
|
||||||
|
Postgres needs one because its dirty buffers are not durable until flushed, and
|
||||||
|
ours are durable at commit — an idle log does not grow.
|
||||||
|
|
||||||
|
**A failed compaction is a missed optimisation, not a durability event.** It
|
||||||
|
leaves the original log intact and returns an error the callers ignore. It must
|
||||||
|
never take `wo_wal_commit_fatal`'s path, which exists for a different problem.
|
||||||
|
|
||||||
|
**If you are here because a checkpoint misbehaved:** `WO_WAL_STATS=1` reports
|
||||||
|
compaction count, the stop-the-world pause (max and total) and the last
|
||||||
|
compaction's size. `WO_CHECKPOINT_BYTES` and `WO_CHECKPOINT_RATIO` move the
|
||||||
|
policy; setting a tiny floor forces compaction in a few writes, which is how the
|
||||||
|
gate tests it at all.
|
||||||
|
|
|
||||||
|
|
@ -18,6 +18,23 @@ static int table_is_durable(const wo_db *db, uint32_t cid) {
|
||||||
return (db->classes[cid].flags & WO_CLASSF_VOLATILE) == 0u;
|
return (db->classes[cid].flags & WO_CLASSF_VOLATILE) == 0u;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* databasev2 3: the inline path's compaction check.
|
||||||
|
*
|
||||||
|
* The drain has its own (vm.c, after the barrier). This one exists because a
|
||||||
|
* statement running ON the owner shard never enters that drain, so without it
|
||||||
|
* a single-shard durable program's log grows FOREVER — measured: WO_SHARDS=1
|
||||||
|
* reached 536 KB where the multi-shard run held 446 KB, because the check was
|
||||||
|
* only wired into the drain.
|
||||||
|
*
|
||||||
|
* Safe here for the same reason it is safe there: the commit above just
|
||||||
|
* emptied the staging buffer. The result is ignored because a failed
|
||||||
|
* compaction is a missed optimisation, not a durability event. */
|
||||||
|
static void maybe_compact(wo_db *db, wo_wal *w) {
|
||||||
|
if (wo_wal_should_compact(w->off, w->compacted_bytes, wo_wal_ckpt_floor,
|
||||||
|
wo_wal_ckpt_ratio))
|
||||||
|
(void)wo_wal_compact(w, db);
|
||||||
|
}
|
||||||
|
|
||||||
int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
||||||
uint32_t A = wo_ins_a(ins), B = wo_ins_b(ins), C = wo_ins_c(ins);
|
uint32_t A = wo_ins_a(ins), B = wo_ins_b(ins), C = wo_ins_c(ins);
|
||||||
wo_db *db = (wo_db *)vm->rt.db;
|
wo_db *db = (wo_db *)vm->rt.db;
|
||||||
|
|
@ -36,15 +53,26 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
||||||
: WO_T_DB;
|
: WO_T_DB;
|
||||||
wo_wal *w = (wo_wal *)vm->rt.wal;
|
wo_wal *w = (wo_wal *)vm->rt.wal;
|
||||||
if (w && table_is_durable(db, cid)) {
|
if (w && table_is_durable(db, cid)) {
|
||||||
/* RAM applied, record staged, ONE commit before the ack (the
|
/* THE INLINE PATH KEEPS ITS OWN BARRIER, AND THAT ASYMMETRY IS
|
||||||
* builtin's return). A failed commit is a failed write: the
|
* DELIBERATE (databasev2 4 part A). The request path batches:
|
||||||
* row is removed again so RAM never claims what disk never
|
* wo_vm_adopt holds each reply and commits once per drain. This
|
||||||
* acknowledged, and the statement traps. */
|
* path cannot, because it has no reply to hold — it returns into
|
||||||
if (wo_wal_append_insert(w, db, cid, id) != 0 || wo_wal_commit(w) != 0) {
|
* its OWN fiber rather than unparking a requester. Do not "fix"
|
||||||
wo_row_remove(db, cid, id);
|
* this by dropping the commit: without it an inline statement
|
||||||
*msg = "wal commit failed";
|
* would never be durable at all.
|
||||||
return WO_T_IO;
|
*
|
||||||
}
|
* Committing here is safe because the drain commits
|
||||||
|
* unconditionally whenever anything is staged, so the buffer is
|
||||||
|
* empty when this runs.
|
||||||
|
*
|
||||||
|
* The `table_is_durable` guard is databasev2 2's: a
|
||||||
|
* `@table(durable: false)` class is never staged, so it reaches
|
||||||
|
* neither this barrier nor the compaction check below.
|
||||||
|
*
|
||||||
|
* Failure is fatal, not a trap: the row is already in RAM. */
|
||||||
|
if (wo_wal_append_insert(w, db, cid, id) != 0) wo_wal_stage_fatal(w);
|
||||||
|
wo_wal_commit_fatal(w, 1);
|
||||||
|
maybe_compact(db, w);
|
||||||
}
|
}
|
||||||
R[A] = id;
|
R[A] = id;
|
||||||
return 0;
|
return 0;
|
||||||
|
|
@ -58,10 +86,11 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
||||||
return ek == DB_ERR_UNIQUE ? WO_T_UNIQUE : ek == DB_ERR_OOM ? WO_T_OOM : WO_T_DB;
|
return ek == DB_ERR_UNIQUE ? WO_T_UNIQUE : ek == DB_ERR_OOM ? WO_T_OOM : WO_T_DB;
|
||||||
wo_wal *w = (wo_wal *)vm->rt.wal;
|
wo_wal *w = (wo_wal *)vm->rt.wal;
|
||||||
if (w && table_is_durable(db, cid)) {
|
if (w && table_is_durable(db, cid)) {
|
||||||
if (wo_wal_append_update(w, db, cid, id) != 0 || wo_wal_commit(w) != 0) {
|
/* was: trap and leave RAM ahead of disk, which the old comment
|
||||||
*msg = "wal commit failed"; /* RAM ahead of disk: trap, do not ack */
|
* admitted. Now fatal — see the insert arm. */
|
||||||
return WO_T_IO;
|
if (wo_wal_append_update(w, db, cid, id) != 0) wo_wal_stage_fatal(w);
|
||||||
}
|
wo_wal_commit_fatal(w, 1);
|
||||||
|
maybe_compact(db, w);
|
||||||
}
|
}
|
||||||
R[A] = 0;
|
R[A] = 0;
|
||||||
return 0;
|
return 0;
|
||||||
|
|
@ -81,10 +110,9 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
||||||
}
|
}
|
||||||
wo_wal *w = (wo_wal *)vm->rt.wal;
|
wo_wal *w = (wo_wal *)vm->rt.wal;
|
||||||
if (w && table_is_durable(db, cid)) {
|
if (w && table_is_durable(db, cid)) {
|
||||||
if (wo_wal_append_remove(w, cid, id) != 0 || wo_wal_commit(w) != 0) {
|
if (wo_wal_append_remove(w, cid, id) != 0) wo_wal_stage_fatal(w);
|
||||||
*msg = "wal commit failed";
|
wo_wal_commit_fatal(w, 1);
|
||||||
return WO_T_IO;
|
maybe_compact(db, w);
|
||||||
}
|
|
||||||
}
|
}
|
||||||
R[A] = 0;
|
R[A] = 0;
|
||||||
return 0;
|
return 0;
|
||||||
|
|
@ -226,13 +254,12 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) {
|
||||||
q->msg = m;
|
q->msg = m;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
if (w) {
|
if (w && table_is_durable(db, q->cid)) {
|
||||||
if (wo_wal_append_insert(w, db, q->cid, id) != 0 || wo_wal_commit(w) != 0) {
|
/* databasev2 4: staging failure is FATAL, not a trap. The row is
|
||||||
wo_row_remove(db, q->cid, id);
|
* already in RAM; of the three verbs only insert could undo
|
||||||
q->status = WO_T_IO;
|
* itself, so continuing means RAM ahead of disk. One rule: once a
|
||||||
q->msg = "wal commit failed";
|
* statement has mutated RAM, the outcomes are durable or death. */
|
||||||
break;
|
if (wo_wal_append_insert(w, db, q->cid, id) != 0) wo_wal_stage_fatal(w);
|
||||||
}
|
|
||||||
}
|
}
|
||||||
q->result = id;
|
q->result = id;
|
||||||
break;
|
break;
|
||||||
|
|
@ -244,12 +271,8 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) {
|
||||||
q->msg = m;
|
q->msg = m;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
if (w) {
|
if (w && table_is_durable(db, q->cid)) {
|
||||||
if (wo_wal_append_update(w, db, q->cid, q->id) != 0 || wo_wal_commit(w) != 0) {
|
if (wo_wal_append_update(w, db, q->cid, q->id) != 0) wo_wal_stage_fatal(w);
|
||||||
q->status = WO_T_IO;
|
|
||||||
q->msg = "wal commit failed";
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
@ -264,12 +287,8 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) {
|
||||||
q->msg = "no such row";
|
q->msg = "no such row";
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
if (w) {
|
if (w && table_is_durable(db, q->cid)) {
|
||||||
if (wo_wal_append_remove(w, q->cid, q->id) != 0 || wo_wal_commit(w) != 0) {
|
if (wo_wal_append_remove(w, q->cid, q->id) != 0) wo_wal_stage_fatal(w);
|
||||||
q->status = WO_T_IO;
|
|
||||||
q->msg = "wal commit failed";
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -4,7 +4,9 @@
|
||||||
#include "wal.h"
|
#include "wal.h"
|
||||||
|
|
||||||
#include <errno.h>
|
#include <errno.h>
|
||||||
|
#include <time.h>
|
||||||
#include <fcntl.h>
|
#include <fcntl.h>
|
||||||
|
#include <stdio.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
#include <unistd.h>
|
#include <unistd.h>
|
||||||
|
|
@ -302,6 +304,19 @@ int wo_wal_open(wo_wal *w, const char *path, uint64_t prealloc) {
|
||||||
memset(w, 0, sizeof(*w));
|
memset(w, 0, sizeof(*w));
|
||||||
w->fd = open(path, O_RDWR | O_CREAT, 0644);
|
w->fd = open(path, O_RDWR | O_CREAT, 0644);
|
||||||
if (w->fd < 0) return -1;
|
if (w->fd < 0) return -1;
|
||||||
|
w->path = strdup(path); /* NULL is tolerated: the diagnostic degrades */
|
||||||
|
/* databasev2 3: remove a stale compaction temp before doing anything else.
|
||||||
|
* The only way one exists is a crash before the rename, which means its
|
||||||
|
* records were never authoritative — the live log below is the truth. It is
|
||||||
|
* deleted rather than ignored because a file full of well-formed records
|
||||||
|
* sitting beside the log is exactly the thing a future reader mistakes for
|
||||||
|
* data. */
|
||||||
|
{
|
||||||
|
char tmp[4096];
|
||||||
|
if ((size_t)snprintf(tmp, sizeof tmp, "%s%s", path, WO_WAL_TMP_SUFFIX) < sizeof tmp)
|
||||||
|
(void)unlink(tmp);
|
||||||
|
}
|
||||||
|
w->prealloc = prealloc;
|
||||||
if (prealloc) {
|
if (prealloc) {
|
||||||
/* best-effort: a filesystem without fallocate still works */
|
/* best-effort: a filesystem without fallocate still works */
|
||||||
(void)posix_fallocate(w->fd, 0, (off_t)prealloc);
|
(void)posix_fallocate(w->fd, 0, (off_t)prealloc);
|
||||||
|
|
@ -317,6 +332,7 @@ int wo_wal_open(wo_wal *w, const char *path, uint64_t prealloc) {
|
||||||
|
|
||||||
void wo_wal_close(wo_wal *w) {
|
void wo_wal_close(wo_wal *w) {
|
||||||
if (w->fd >= 0) close(w->fd);
|
if (w->fd >= 0) close(w->fd);
|
||||||
|
free(w->path);
|
||||||
free(w->buf);
|
free(w->buf);
|
||||||
memset(w, 0, sizeof(*w));
|
memset(w, 0, sizeof(*w));
|
||||||
w->fd = -1;
|
w->fd = -1;
|
||||||
|
|
@ -390,7 +406,103 @@ int wo_wal_append_remove(wo_wal *w, uint32_t class_id, uint64_t id) {
|
||||||
}
|
}
|
||||||
|
|
||||||
int wo_wal_commit(wo_wal *w) {
|
int wo_wal_commit(wo_wal *w) {
|
||||||
if (!w->len) return 0;
|
if (!w->len) return 0; /* empty commits are not batches; do not count them */
|
||||||
|
if (w->len > w->stat_peak_staged) w->stat_peak_staged = w->len;
|
||||||
|
size_t at = 0;
|
||||||
|
while (at < w->len) {
|
||||||
|
ssize_t n = pwrite(w->fd, w->buf + at, w->len - at, (off_t)(w->off + at));
|
||||||
|
if (n < 0) {
|
||||||
|
if (errno == EINTR) continue;
|
||||||
|
return WO_WAL_ERR_WRITE;
|
||||||
|
}
|
||||||
|
at += (size_t)n;
|
||||||
|
}
|
||||||
|
if (fdatasync(w->fd) != 0) return WO_WAL_ERR_SYNC;
|
||||||
|
w->off += w->len;
|
||||||
|
w->len = 0; /* acked: the batch is durable */
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Nothing at either fatal point is recoverable: RAM holds changes the log
|
||||||
|
* does not, and this process can no longer serve reads that would survive a
|
||||||
|
* restart. Name what failed precisely enough to act on, then stop. */
|
||||||
|
static void wal_die(const wo_wal *w, const char *op, uint32_t nrec) {
|
||||||
|
fprintf(stderr,
|
||||||
|
"writeonce: DURABILITY FAILURE — %s failed on %s: %s\n"
|
||||||
|
" %u record(s) were NOT made durable and are not acknowledged.\n"
|
||||||
|
" The process is stopping: replay restores the last durable state.\n",
|
||||||
|
op, w->path ? w->path : "(the write-ahead log)", strerror(errno),
|
||||||
|
nrec);
|
||||||
|
exit(WO_EXIT_DURABILITY);
|
||||||
|
}
|
||||||
|
|
||||||
|
void wo_wal_stage_fatal(const wo_wal *w) { wal_die(w, "staging a record", 1); }
|
||||||
|
|
||||||
|
void wo_wal_commit_fatal(wo_wal *w, uint32_t nrec) {
|
||||||
|
int staged = w->len != 0;
|
||||||
|
int rc = wo_wal_commit(w);
|
||||||
|
if (rc == 0) {
|
||||||
|
if (staged) { /* count the barrier that actually happened */
|
||||||
|
w->stat_batches++;
|
||||||
|
w->stat_records += nrec;
|
||||||
|
if (nrec > w->stat_peak_batch) w->stat_peak_batch = nrec;
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
wal_die(w, rc == WO_WAL_ERR_SYNC ? "fdatasync" : "pwrite", nrec);
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t wo_wal_ckpt_floor = 4u << 20; /* 4 MiB: below this there is nothing worth reclaiming */
|
||||||
|
uint32_t wo_wal_ckpt_ratio = 3u; /* 3x the live-set's own size is enough history */
|
||||||
|
|
||||||
|
int wo_wal_should_compact(uint64_t used, uint64_t last, uint64_t floor, uint32_t ratio) {
|
||||||
|
if (used < floor) return 0; /* a small log has nothing to reclaim */
|
||||||
|
if (last == 0) return 1; /* past the floor and never compacted: do it once
|
||||||
|
* to establish the denominator */
|
||||||
|
if (ratio == 0) return 0; /* a zero ratio disables the policy rather than
|
||||||
|
* dividing by nothing */
|
||||||
|
return used > last * (uint64_t)ratio;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* databasev2 3: how many records the dump stages before flushing.
|
||||||
|
*
|
||||||
|
* NOT unbounded: stage() grows the staging buffer by doubling and never
|
||||||
|
* shrinks it, so appending a whole store through one buffer would hold the
|
||||||
|
* entire store in RAM on top of the store itself — the unbounded growth
|
||||||
|
* databasev2 1 identified as how this engine dies. 256 records is a few tens
|
||||||
|
* of KiB per flush, which is large enough that the syscall cost is amortised
|
||||||
|
* and small enough that the buffer never matters. */
|
||||||
|
#define WO_WAL_COMPACT_FLUSH 256u
|
||||||
|
|
||||||
|
/* rename(2)'s atomicity is in-kernel: the new directory ENTRY is not durable
|
||||||
|
* until the parent directory is synced. Postgres does the same thing for the
|
||||||
|
* same reason. Best-effort — a filesystem that refuses to sync a directory
|
||||||
|
* still leaves a correct log, just one whose swap might not survive a power
|
||||||
|
* cut. */
|
||||||
|
static void sync_parent_dir(const char *path) {
|
||||||
|
char dir[4096];
|
||||||
|
size_t n = strlen(path);
|
||||||
|
if (n >= sizeof dir) return;
|
||||||
|
memcpy(dir, path, n + 1);
|
||||||
|
char *slash = strrchr(dir, '/');
|
||||||
|
if (slash == dir) dir[1] = '\0';
|
||||||
|
else if (slash) *slash = '\0';
|
||||||
|
else memcpy(dir, ".", 2);
|
||||||
|
int fd = open(dir, O_RDONLY);
|
||||||
|
if (fd < 0) return;
|
||||||
|
(void)fsync(fd);
|
||||||
|
close(fd);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* databasev2 3: write the staged bytes WITHOUT a durability barrier.
|
||||||
|
*
|
||||||
|
* Only compaction's dump uses this. Intermediate durability there is worthless:
|
||||||
|
* the temp file is not authoritative until the rename, and it is fsynced once
|
||||||
|
* immediately before that. Using wo_wal_commit for the dump instead cost one
|
||||||
|
* fdatasync per 256 records — measured, that was most of the stop-the-world
|
||||||
|
* pause (~22 MB/s, where the fixed cost plus ~150 redundant syncs dominated a
|
||||||
|
* 2 MB dump). */
|
||||||
|
static int wal_write_nosync(wo_wal *w) {
|
||||||
size_t at = 0;
|
size_t at = 0;
|
||||||
while (at < w->len) {
|
while (at < w->len) {
|
||||||
ssize_t n = pwrite(w->fd, w->buf + at, w->len - at, (off_t)(w->off + at));
|
ssize_t n = pwrite(w->fd, w->buf + at, w->len - at, (off_t)(w->off + at));
|
||||||
|
|
@ -400,12 +512,110 @@ int wo_wal_commit(wo_wal *w) {
|
||||||
}
|
}
|
||||||
at += (size_t)n;
|
at += (size_t)n;
|
||||||
}
|
}
|
||||||
if (fdatasync(w->fd) != 0) return -1;
|
|
||||||
w->off += w->len;
|
w->off += w->len;
|
||||||
w->len = 0; /* acked: the batch is durable */
|
w->len = 0;
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static uint64_t mono_us(void) {
|
||||||
|
struct timespec ts;
|
||||||
|
if (clock_gettime(CLOCK_MONOTONIC, &ts) != 0) return 0;
|
||||||
|
return (uint64_t)ts.tv_sec * 1000000ull + (uint64_t)ts.tv_nsec / 1000ull;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* ============================================================================
|
||||||
|
* OBLIGATION FOR WHOEVER IMPLEMENTS `resident: keys` (databasev2 2, tasks
|
||||||
|
* 5c/5d) — READ THIS BEFORE STORING WAL OFFSETS.
|
||||||
|
*
|
||||||
|
* Compaction rewrites the log and MOVES EVERY RECORD. Any WAL byte offset
|
||||||
|
* captured from the old file is meaningless afterwards — not stale-but-
|
||||||
|
* readable, but pointing at an arbitrary byte of a different file.
|
||||||
|
*
|
||||||
|
* `resident: keys` stores exactly such an offset per row and reads rows back
|
||||||
|
* through it. So the loop below, which knows each record's NEW position as it
|
||||||
|
* writes it, MUST also rebuild that map. It is the cheap direction and the only
|
||||||
|
* one that keeps both features usable together; the alternative is forbidding
|
||||||
|
* compaction whenever such a table is live, which would mean the feature for
|
||||||
|
* huge tables is incompatible with the feature that stops their log growing.
|
||||||
|
*
|
||||||
|
* Nothing fails today because that storage half does not exist yet. It will
|
||||||
|
* fail later, and it will look like data corruption rather than a design gap.
|
||||||
|
* ==========================================================================*/
|
||||||
|
int wo_wal_compact(wo_wal *w, wo_db *db) {
|
||||||
|
/* staged records would be written into a file about to be replaced */
|
||||||
|
if (!w->path || w->len != 0) return -1;
|
||||||
|
uint64_t t0 = mono_us();
|
||||||
|
|
||||||
|
char tmp[4096];
|
||||||
|
if ((size_t)snprintf(tmp, sizeof tmp, "%s%s", w->path, WO_WAL_TMP_SUFFIX) >= sizeof tmp)
|
||||||
|
return -1;
|
||||||
|
(void)unlink(tmp); /* a stale one would otherwise be appended to */
|
||||||
|
|
||||||
|
wo_wal nw;
|
||||||
|
/* THE REPLACEMENT MUST BE PREALLOCATED LIKE THE ORIGINAL. The WAL is
|
||||||
|
* preallocated so appends never extend the file, which is precisely what
|
||||||
|
* makes fdatasync sufficient as the ack barrier — no file-size metadata
|
||||||
|
* has to reach disk for an acked record to be readable. Opening the
|
||||||
|
* replacement with prealloc 0 silently removed that property, and the
|
||||||
|
* crash battery caught it: records acked shortly before a kill went
|
||||||
|
* missing, with the log otherwise intact and self-consistent. */
|
||||||
|
if (wo_wal_open(&nw, tmp, w->prealloc) != 0) return -1;
|
||||||
|
|
||||||
|
/* one INSERT per live row, in the existing grammar, through the existing
|
||||||
|
* append path — so replay needs no second decoder and ids are preserved
|
||||||
|
* exactly (wo_wal_append_insert takes the id and reads the row) */
|
||||||
|
uint32_t pending = 0;
|
||||||
|
for (uint32_t cid = 0; cid < db->class_cnt; cid++) {
|
||||||
|
db_table *t = &db->tables[cid];
|
||||||
|
if (!t->slabs) continue; /* tables are created lazily */
|
||||||
|
uint32_t total = t->slab_cnt * DB_SLAB_ROWS;
|
||||||
|
for (uint32_t g = 0; g < total; g++) {
|
||||||
|
if (!(t->bitmap[g >> 6] & (1ull << (g & 63)))) continue;
|
||||||
|
db_row *r = (db_row *)(t->slabs[g / DB_SLAB_ROWS] +
|
||||||
|
(size_t)(g % DB_SLAB_ROWS) * t->row_size);
|
||||||
|
if (wo_wal_append_insert(&nw, db, cid, r->id) != 0) goto fail;
|
||||||
|
if (++pending >= WO_WAL_COMPACT_FLUSH) {
|
||||||
|
if (wal_write_nosync(&nw) != 0) goto fail;
|
||||||
|
pending = 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (wal_write_nosync(&nw) != 0) goto fail; /* the tail batch */
|
||||||
|
/* THE dump's one and only barrier: everything above is just bytes in the
|
||||||
|
* page cache until this, and nothing reads the temp before the rename. */
|
||||||
|
if (fsync(nw.fd) != 0) goto fail;
|
||||||
|
|
||||||
|
uint64_t new_bytes = nw.off;
|
||||||
|
wo_wal_close(&nw);
|
||||||
|
|
||||||
|
/* THE SWITCH. Every crash point either side of this is safe. */
|
||||||
|
if (rename(tmp, w->path) != 0) {
|
||||||
|
(void)unlink(tmp);
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
sync_parent_dir(w->path);
|
||||||
|
|
||||||
|
/* the old descriptor now refers to an unlinked inode */
|
||||||
|
if (w->fd >= 0) close(w->fd);
|
||||||
|
w->fd = open(w->path, O_RDWR);
|
||||||
|
if (w->fd < 0) return -1; /* the log is correct on disk; this process cannot go on */
|
||||||
|
w->off = new_bytes;
|
||||||
|
w->len = 0;
|
||||||
|
w->compacted_bytes = new_bytes;
|
||||||
|
{ /* the stop-the-world pause: nothing was served while this ran */
|
||||||
|
uint64_t el = mono_us() - t0;
|
||||||
|
w->stat_compactions++;
|
||||||
|
w->stat_compact_us_total += el;
|
||||||
|
if (el > w->stat_compact_us_max) w->stat_compact_us_max = el;
|
||||||
|
}
|
||||||
|
return 0;
|
||||||
|
|
||||||
|
fail:
|
||||||
|
wo_wal_close(&nw);
|
||||||
|
(void)unlink(tmp);
|
||||||
|
return -1; /* the live log is untouched and still usable */
|
||||||
|
}
|
||||||
|
|
||||||
static int apply_record(wo_db *db, const uint8_t *payload, uint32_t len) {
|
static int apply_record(wo_db *db, const uint8_t *payload, uint32_t len) {
|
||||||
rbuf r = {payload, payload + len, 0};
|
rbuf r = {payload, payload + len, 0};
|
||||||
uint8_t kind = rd_u8(&r);
|
uint8_t kind = rd_u8(&r);
|
||||||
|
|
|
||||||
|
|
@ -47,10 +47,38 @@ enum { WO_WAL_INSERT = 1, WO_WAL_REMOVE = 2, WO_WAL_UPDATE = 3 };
|
||||||
|
|
||||||
typedef struct wo_wal {
|
typedef struct wo_wal {
|
||||||
int fd;
|
int fd;
|
||||||
|
/* databasev2 4: where this WAL lives, so a durability failure can name
|
||||||
|
* the file it could not write. An abort diagnostic without the path
|
||||||
|
* sends an operator hunting. Owned here, freed by wo_wal_close. */
|
||||||
|
char *path;
|
||||||
uint64_t off; /* next write offset (the intact tail) */
|
uint64_t off; /* next write offset (the intact tail) */
|
||||||
/* staged batch: appended by wal_append_*, flushed by wal_commit */
|
/* staged batch: appended by wal_append_*, flushed by wal_commit */
|
||||||
uint8_t *buf;
|
uint8_t *buf;
|
||||||
size_t len, cap;
|
size_t len, cap;
|
||||||
|
/* databasev2 4: group-commit diagnostics. Batching is worthless if
|
||||||
|
* batches are always one, and a throughput change would then have come
|
||||||
|
* from somewhere else — so the mechanism is measured, not assumed.
|
||||||
|
* peak_staged also settles whether the batch needs a cap with a number
|
||||||
|
* instead of a guess. Reported at exit under WO_WAL_STATS. */
|
||||||
|
uint64_t stat_batches; /* non-empty commits */
|
||||||
|
uint64_t stat_records; /* records those commits carried */
|
||||||
|
uint64_t stat_peak_batch; /* most records in one barrier */
|
||||||
|
uint64_t stat_peak_staged; /* most bytes staged behind one barrier */
|
||||||
|
/* databasev2 3: bytes the last compaction wrote. The trigger compares the
|
||||||
|
* log against THIS rather than an estimate of the live set — estimating
|
||||||
|
* would mean estimating Text, and the compactor knows the true number. */
|
||||||
|
uint64_t compacted_bytes;
|
||||||
|
/* databasev2 3: the preallocation this log was opened with. Compaction
|
||||||
|
* MUST give the replacement the same one: the WAL is preallocated so that
|
||||||
|
* appends never extend the file, which is what lets fdatasync alone be the
|
||||||
|
* ack barrier. A replacement without it silently weakens durability. */
|
||||||
|
uint64_t prealloc;
|
||||||
|
/* databasev2 3: what compaction actually did, reported under WO_WAL_STATS.
|
||||||
|
* The PAUSE is the number the spec refused to assume — compaction is
|
||||||
|
* stop-the-world, so its duration is the cost being weighed. */
|
||||||
|
uint64_t stat_compactions;
|
||||||
|
uint64_t stat_compact_us_max;
|
||||||
|
uint64_t stat_compact_us_total;
|
||||||
} wo_wal;
|
} wo_wal;
|
||||||
|
|
||||||
/* databasev2 2: the file offset the NEXT staged record will occupy.
|
/* databasev2 2: the file offset the NEXT staged record will occupy.
|
||||||
|
|
@ -90,10 +118,94 @@ int wo_wal_append_remove(wo_wal *w, uint32_t class_id, uint64_t id);
|
||||||
* later optimization, recorded). Call AFTER the RAM update. */
|
* later optimization, recorded). Call AFTER the RAM update. */
|
||||||
int wo_wal_append_update(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id);
|
int wo_wal_append_update(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id);
|
||||||
|
|
||||||
|
/* databasev2 4: which half of the barrier failed. A pwrite failure and an
|
||||||
|
* fdatasync failure are different operational problems (a short write vs a
|
||||||
|
* device refusing the flush), so the diagnostic must name the right one. */
|
||||||
|
#define WO_WAL_ERR_WRITE (-1)
|
||||||
|
#define WO_WAL_ERR_SYNC (-2)
|
||||||
|
|
||||||
|
/* The process exit status for a durability failure.
|
||||||
|
*
|
||||||
|
* 74 is sysexits' EX_IOERR, chosen deliberately over a small number: 1 is a
|
||||||
|
* trap and 2 is a loader refusal, but 3 and 4 are already used by SAMPLES for
|
||||||
|
* their own meanings — db-bench's own `verify` exits 3 on a checksum mismatch,
|
||||||
|
* and it is the gate that exercises durability, so a durability abort exiting 3
|
||||||
|
* would have been indistinguishable from the mismatch it is supposed to help
|
||||||
|
* diagnose. The low range belongs to programs; the runtime takes a high one. */
|
||||||
|
#define WO_EXIT_DURABILITY 74
|
||||||
|
|
||||||
/* Write the staged batch and fdatasync — the ack line. Empty batch = ok,
|
/* Write the staged batch and fdatasync — the ack line. Empty batch = ok,
|
||||||
* no syscall. 0 ok, -1 write/sync failure (the batch stays staged). */
|
* no syscall. 0 ok, WO_WAL_ERR_WRITE / WO_WAL_ERR_SYNC on failure (the
|
||||||
|
* batch stays staged: a failed commit consumes nothing). */
|
||||||
int wo_wal_commit(wo_wal *w);
|
int wo_wal_commit(wo_wal *w);
|
||||||
|
|
||||||
|
/* databasev2 3: the checkpoint trigger, as a PURE decision so it can be tested
|
||||||
|
* without a store — which is the only way a policy like this gets tested at all.
|
||||||
|
*
|
||||||
|
* [used] the log's used bytes; [last] what the LAST compaction wrote (0 if it
|
||||||
|
* has never run); [floor] the size below which compacting is not worth it;
|
||||||
|
* [ratio] the multiple of [last] that counts as too much history.
|
||||||
|
*
|
||||||
|
* The denominator is the last compaction's MEASURED output rather than an
|
||||||
|
* estimate of the live set: estimating would mean estimating Text, and the
|
||||||
|
* compactor already knows the true number.
|
||||||
|
*
|
||||||
|
* There is deliberately NO TIME component. Postgres' CheckPointTimeout exists
|
||||||
|
* to bound data loss from unflushed buffers; our records are durable at commit,
|
||||||
|
* so a checkpoint only reclaims space and shortens boot. An idle log does not
|
||||||
|
* grow, so a timer would fire with nothing to do.
|
||||||
|
*
|
||||||
|
* 1 = compact now, 0 = leave it. */
|
||||||
|
int wo_wal_should_compact(uint64_t used, uint64_t last, uint64_t floor, uint32_t ratio);
|
||||||
|
|
||||||
|
/* Defaults, overridable at boot by WO_CHECKPOINT_BYTES / WO_CHECKPOINT_RATIO.
|
||||||
|
* The knobs are what make the policy testable: a test sets a tiny floor and
|
||||||
|
* forces compaction in a few writes instead of waiting for megabytes. */
|
||||||
|
extern uint64_t wo_wal_ckpt_floor;
|
||||||
|
extern uint32_t wo_wal_ckpt_ratio;
|
||||||
|
|
||||||
|
/* databasev2 3: the temporary file compaction writes before the swap. Named
|
||||||
|
* next to the log so it lands on the same filesystem — rename(2) is only
|
||||||
|
* atomic within one. Boot removes a stale one (a crash before the rename). */
|
||||||
|
#define WO_WAL_TMP_SUFFIX ".compact"
|
||||||
|
|
||||||
|
/* databasev2 3: rewrite the log as one INSERT record per LIVE row, then swap
|
||||||
|
* it in with rename(2).
|
||||||
|
*
|
||||||
|
* Recovery is deliberately untouched: the result is an ordinary log in the
|
||||||
|
* ordinary grammar, replayed from byte 0. Crash safety comes from rename being
|
||||||
|
* atomic — before it the live log is intact and the temp file is not
|
||||||
|
* authoritative; after it the new log is complete. There is no window in which
|
||||||
|
* a reader sees a mixture, so this needs no recovery logic of its own.
|
||||||
|
*
|
||||||
|
* REFUSES if anything is staged (returns -1 without touching the log): those
|
||||||
|
* records would be written into a file about to be replaced. Callers must
|
||||||
|
* invoke this only where the staging buffer is empty — right after a barrier.
|
||||||
|
*
|
||||||
|
* A failure is a MISSED OPTIMISATION, not a durability event: the original log
|
||||||
|
* is left usable and the process keeps running. It must not take the fatal
|
||||||
|
* path wo_wal_commit_fatal takes.
|
||||||
|
*
|
||||||
|
* 0 ok, -1 on any failure. */
|
||||||
|
int wo_wal_compact(wo_wal *w, wo_db *db);
|
||||||
|
|
||||||
|
/* databasev2 4: a record could not even be STAGED (the row is already in
|
||||||
|
* RAM, so this is the same unrecoverable position as a failed barrier — see
|
||||||
|
* wo_wal_commit_fatal). Never returns. */
|
||||||
|
void wo_wal_stage_fatal(const wo_wal *w);
|
||||||
|
|
||||||
|
/* databasev2 4: commit, or END THE PROCESS.
|
||||||
|
*
|
||||||
|
* The one rule this iteration introduces: once a statement has mutated RAM,
|
||||||
|
* the only outcomes are durable or process death. Retrying is not an
|
||||||
|
* alternative — on Linux a failed fsync may already have discarded the dirty
|
||||||
|
* pages, so a second call can report success having written nothing. The
|
||||||
|
* recovery that works is replay, which returns the last durable state.
|
||||||
|
*
|
||||||
|
* [nrec] is the number of records in the batch, for the diagnostic only.
|
||||||
|
* Returns on success; never returns on failure. */
|
||||||
|
void wo_wal_commit_fatal(wo_wal *w, uint32_t nrec);
|
||||||
|
|
||||||
/* Boot replay: apply every intact record to [db] in order. Ids re-enter
|
/* Boot replay: apply every intact record to [db] in order. Ids re-enter
|
||||||
* exactly as logged; each table's next_id advances past the replayed ids
|
* exactly as logged; each table's next_id advances past the replayed ids
|
||||||
* that belong to this shard. Returns the number of records applied, or -1
|
* that belong to this shard. Returns the number of records applied, or -1
|
||||||
|
|
|
||||||
|
|
@ -128,6 +128,7 @@ flowchart TD
|
||||||
classDef rt fill:#8250df,color:#fff,stroke:none
|
classDef rt fill:#8250df,color:#fff,stroke:none
|
||||||
classDef gated fill:#eac54f,color:#000,stroke:none
|
classDef gated fill:#eac54f,color:#000,stroke:none
|
||||||
classDef v2 fill:#0969da,color:#fff,stroke:none
|
classDef v2 fill:#0969da,color:#fff,stroke:none
|
||||||
|
classDef done fill:#1a7f37,color:#fff,stroke:none
|
||||||
|
|
||||||
I7b2["7b per-shard collector (done — the precondition 8 waited on)"]:::rt
|
I7b2["7b per-shard collector (done — the precondition 8 waited on)"]:::rt
|
||||||
I8x["8 shard-actor runtime: thread-per-core, ownership-move messages"]:::rt
|
I8x["8 shard-actor runtime: thread-per-core, ownership-move messages"]:::rt
|
||||||
|
|
@ -140,7 +141,7 @@ flowchart TD
|
||||||
STREAM2["request body streaming + backpressure"]:::gated
|
STREAM2["request body streaming + backpressure"]:::gated
|
||||||
SRESP2["streaming responses + explicit commit point"]:::gated
|
SRESP2["streaming responses + explicit commit point"]:::gated
|
||||||
CANCEL2["per-request cancellation propagation"]:::gated
|
CANCEL2["per-request cancellation propagation"]:::gated
|
||||||
PUBSUB2["pub/sub + WebSockets (rejected until here)"]:::gated
|
PUBSUB2["DONE 2026-08-27 — pub/sub + WebSockets (iteration 24: ws_accept + wsframe + room actors)"]:::done
|
||||||
ASYNC9C["20 async attach statements (rejected-for-now alternative)"]:::gated
|
ASYNC9C["20 async attach statements (rejected-for-now alternative)"]:::gated
|
||||||
TIMEOUTS2["idle timeouts become schedulable (net seam still needed)"]:::gated
|
TIMEOUTS2["idle timeouts become schedulable (net seam still needed)"]:::gated
|
||||||
|
|
||||||
|
|
|
||||||
93
docs/2026-08-27-chat-drain-finding.md
Normal file
93
docs/2026-08-27-chat-drain-finding.md
Normal file
|
|
@ -0,0 +1,93 @@
|
||||||
|
# Iteration 24 T9 — the drain bug the gate was hiding
|
||||||
|
|
||||||
|
**Found 2026-08-27** while finishing T8/T9 on branch `chat-ws-lifecycle`.
|
||||||
|
Not fixed: the fix is an engine-level decision, recorded here so it is not
|
||||||
|
rediscovered.
|
||||||
|
|
||||||
|
## The symptom
|
||||||
|
|
||||||
|
`just chat`'s drain leg asserts both connected clients receive a WebSocket
|
||||||
|
close frame on `SIGTERM`. Against a **fresh** server it is flaky:
|
||||||
|
|
||||||
|
| Sample | Result |
|
||||||
|
| --- | --- |
|
||||||
|
| 5 fresh servers, 2 clients each | 4 × `close\|close`, 1 × `eof\|close` |
|
||||||
|
| 12 fresh servers | 3 failures, one of them `eof\|eof` |
|
||||||
|
| 16 fresh servers | 5 failures |
|
||||||
|
|
||||||
|
A failing client's socket reaches EOF with **no close frame and no
|
||||||
|
diagnostic** — the process exits and the kernel closes the fd.
|
||||||
|
|
||||||
|
## Why the gate never caught it
|
||||||
|
|
||||||
|
The drain leg did not start its own server. It inherited `$SRV` from the soak
|
||||||
|
leg — a server the soak had already pushed 1000 clients through, so every
|
||||||
|
shard was warm and every actor already scheduled. Draining a warm server hides
|
||||||
|
the cold-start race. Fixed in this change: **every leg now starts its own
|
||||||
|
server**, which is what exposed the bug.
|
||||||
|
|
||||||
|
## Root cause, traced
|
||||||
|
|
||||||
|
Instrumented the sample's actors (diagnostics not committed) and correlated
|
||||||
|
against failing runs:
|
||||||
|
|
||||||
|
1. `DIAG registry-shutdown rooms=1` — main's `send(reg, kind: 2)` **is**
|
||||||
|
delivered and the Registry runs.
|
||||||
|
2. `DIAG room-shutdown` — **never printed on a failing run.** The Room never
|
||||||
|
processes the `kind: 4` shutdown the Registry sends it.
|
||||||
|
3. The Writer's close branch never runs for the affected client, so no close
|
||||||
|
frame is written and the fd is never closed by the Writer. Its
|
||||||
|
`try net.write_dl(...)` is **not** failing — a diagnostic on that path
|
||||||
|
printed zero times.
|
||||||
|
4. A client that *does* get a close frame is usually saved by its own
|
||||||
|
**Reader** noticing `env.stopping()` and running its tail
|
||||||
|
(`DIAG reader-tail bob r2=1`), not by the room broadcast.
|
||||||
|
|
||||||
|
So the drain chain is main → Registry → Room → Writer, three hops across
|
||||||
|
shards, and **the Room's shard does not reliably adopt its inbox before the
|
||||||
|
engine stops.**
|
||||||
|
|
||||||
|
## What was ruled out
|
||||||
|
|
||||||
|
- **Not the spin budget.** Replacing `spin < 20000000` with a wall-clock
|
||||||
|
deadline of 1 s (`time.ticks()`) still failed 2 of 12. More time does not
|
||||||
|
help, which is the strongest evidence the room's shard is not being
|
||||||
|
scheduled at all rather than being scheduled late. That change was reverted:
|
||||||
|
it fixed nothing and cost a fixed 1 s on every shutdown.
|
||||||
|
- **Not `dummy_writer()` spawning during shutdown.** Hoisting it to a
|
||||||
|
Registry field spawned once at startup left 5 of 16 failing.
|
||||||
|
- **Not a write failure.** See point 3.
|
||||||
|
|
||||||
|
## The decision this needs
|
||||||
|
|
||||||
|
`main` cannot park after the stop flag (a park unwinds), so it spins — and
|
||||||
|
spinning is not a barrier. Either:
|
||||||
|
|
||||||
|
- **the engine drains pending inboxes before stopping**, so a `send` issued
|
||||||
|
before the stop flag is guaranteed delivered; or
|
||||||
|
- **the sample gets a real barrier** — the drain is acknowledged back to main,
|
||||||
|
which requires main to observe a reply without parking.
|
||||||
|
|
||||||
|
The first is the honest fix and belongs to the actor lifecycle (iteration 31,
|
||||||
|
absorbed into 24). It is a semantic guarantee — "a send before shutdown is
|
||||||
|
delivered" — not a tuning parameter, and it should be stated in the runtime's
|
||||||
|
lifecycle docs and pinned by a corpus fixture, not left to a spin count.
|
||||||
|
|
||||||
|
## Gate defects fixed alongside (all committed)
|
||||||
|
|
||||||
|
1. **fd check was core-count dependent.** `fds_before + 8` read lazy per-shard
|
||||||
|
init as a leak: shards initialise on first fiber, each taking one
|
||||||
|
`io_uring` + one `eventfd`, capped at `nproc`. On a 20-core box the first
|
||||||
|
wave legitimately adds 18. Measured 26 → 44 after 20 clients, then **still
|
||||||
|
44 after 40 more**. Replaced with the invariant the check is actually for:
|
||||||
|
a second wave must not raise the count. Core-count independent, and it
|
||||||
|
catches a slow leak that any fixed slack would hide.
|
||||||
|
2. **A failed leg orphaned its server.** The drain leg's python died on
|
||||||
|
`int("")` when `$SRV` was empty, so the soak server was never killed and
|
||||||
|
its listener broke the *next* run's soak on the same port. `cleanup` now
|
||||||
|
kills every server a run started, matched on the run's unique temp dir.
|
||||||
|
3. **Two legs the plan requires were missing** — `WO_SHARDS=1` (the
|
||||||
|
single-shard control that says a failure is placement's fault) and
|
||||||
|
`WO_MAILBOX=8` (the drop-slow-member backpressure path). Both added, both
|
||||||
|
green. The mailbox leg manufactures a genuinely slow member by shrinking
|
||||||
|
its `SO_RCVBUF`, so it needs no sleeps.
|
||||||
|
|
@ -1,60 +0,0 @@
|
||||||
---
|
|
||||||
slice: "24" # the story that owns the status; see stories/24-chat-websocket-workload.md
|
|
||||||
status: in-progress
|
|
||||||
---
|
|
||||||
|
|
||||||
# Active slice — chat + actor lifecycle (iteration 24, absorbing 31 + 34)
|
|
||||||
|
|
||||||
Branch `chat-ws-lifecycle`. Spec:
|
|
||||||
[`superpowers/specs/2026-08-23-chat-websocket-actor-lifecycle-design.md`](superpowers/specs/2026-08-23-chat-websocket-actor-lifecycle-design.md)
|
|
||||||
· plan:
|
|
||||||
[`superpowers/plans/2026-08-23-chat-ws-lifecycle.md`](superpowers/plans/2026-08-23-chat-ws-lifecycle.md)
|
|
||||||
· board: [`stories/00-status.md`](stories/00-status.md).
|
|
||||||
|
|
||||||
## Progress (2026-08-23)
|
|
||||||
|
|
||||||
- ✅ **T1 crypto** (`d14fa9f`): sha1/sha256/hmac_sha256, ids 85–87, RFC
|
|
||||||
vectors 18/0, corpus pin. Story 34's C-builtin resolution delivered.
|
|
||||||
- ✅ **T2 bounded mailboxes** (`92754a8`): cap 1024 + `WO_MAILBOX`,
|
|
||||||
sender-side atomic reserve, WO_T_ACTOR (trap 13) catchable. Plus a
|
|
||||||
pre-existing compiler fix: try-arm Text places (bare `e.msg`) now
|
|
||||||
copy before the arm's scope dies (was ASan use-after-free + SEGV).
|
|
||||||
- ✅ **T6 WS upgrade** (`79cfa01`): `ws_accept` + accept-key + the
|
|
||||||
101 hijack sentinel; plain HTTP byte-identical (web-app 26/26).
|
|
||||||
- ✅ **T7 frame codec** (`7ad2ced`): pure-`.wo` RFC 6455 parse/serialize,
|
|
||||||
probe-verified against the RFC's own bytes.
|
|
||||||
- ✅ **T3 call/reply** (`ed69841`): `call` parks + typed scalar reply
|
|
||||||
(WO-E226 through actor-M erasure); actor DEATH landed with it —
|
|
||||||
callers never hang (mid-call + to-dead both trap catchably). Fixed
|
|
||||||
TRAPF's fiber-death leak/dangle en route.
|
|
||||||
|
|
||||||
Every landed task: full battery 12/12, fresh-built.
|
|
||||||
|
|
||||||
## Pending
|
|
||||||
|
|
||||||
- ⬜ **T4 monitor(watched, observer, msg)** — id 89. Most of the death
|
|
||||||
machinery exists (`actor_die`); T4 adds the per-actor monitor list,
|
|
||||||
the death walk delivering the observer's own M-typed notice,
|
|
||||||
monitor-of-already-dead firing immediately, full-observer notice =
|
|
||||||
disclosed stderr drop. Three-argument form (spec deviation, disclosed
|
|
||||||
in the plan: the caller may be `main`, which has no mailbox).
|
|
||||||
- ⬜ **T5 time.after(ms, addr, msg)** — id 90, one-shot, no cancel;
|
|
||||||
rides the T4 deadline plumbing; delivery = runtime send (full = drop
|
|
||||||
+ stderr line, dead = silent). Corpus: timer-delivery,
|
|
||||||
timer-generation (the cancel idiom). Both WO_IO backends.
|
|
||||||
- ⬜ **T8 chat sample** — docs/examples/chat: registry (`call`'s first
|
|
||||||
consumer), room actors (cap-trap drops slow members, `monitor` reaps
|
|
||||||
dead writers), reader/writer actor pair per connection over
|
|
||||||
ws_accept/wsframe; SIGTERM close choreography.
|
|
||||||
- ⬜ **T9 chat gate** — scripts/chat-accept.sh + raw-RFC6455 python
|
|
||||||
client; the spec's five checks (functional cross-shard — also the
|
|
||||||
deferred cross-shard `call` proof — handshake vector, 1k soak with a
|
|
||||||
`WO_MAILBOX=8` sub-run, drain under both backends + ASan, battery).
|
|
||||||
- ⬜ **T10 closeout** — stories 24/31/34 → done/ with banners (note the
|
|
||||||
scalar-reply v1 narrowing + three-argument monitor deviations), board
|
|
||||||
standup entry, graph nodes, framework README ledger rows, runtime +
|
|
||||||
chat CODE-LOGIC sections, delete this marker. Final battery.
|
|
||||||
|
|
||||||
This file is deleted when the slice lands (board convention). It lives flat in
|
|
||||||
`docs/` rather than a status folder — since 2026-08-26 no directory in this repo
|
|
||||||
encodes state; `status:` above is the only place it is recorded.
|
|
||||||
82
docs/examples/chat/CODE-LOGIC.md
Normal file
82
docs/examples/chat/CODE-LOGIC.md
Normal file
|
|
@ -0,0 +1,82 @@
|
||||||
|
# `docs/examples/chat` — how the sample is put together
|
||||||
|
|
||||||
|
Iteration 24's acceptance workload: rooms, presence and broadcast over
|
||||||
|
WebSocket, actors on fibers across shards, one binary, no broker. It exists to
|
||||||
|
*drive* the actor work, so nearly every shape here is chosen to exercise
|
||||||
|
something the runtime claims.
|
||||||
|
|
||||||
|
Gate: `just chat` (`scripts/chat-accept.sh`), which logs to `/tmp/chat.log` —
|
||||||
|
`tail -F` it while the gate runs.
|
||||||
|
|
||||||
|
## The actors
|
||||||
|
|
||||||
|
| Actor | Owns | Answers |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `Registry` | name → room map, a fallback room | a `call` returning the room's address; spawns rooms on demand |
|
||||||
|
| `Room` | its member list (writer address + name) | join, leave, a text line, shutdown |
|
||||||
|
| `Reader` | the read half of one connection | nothing — it loops on the fd and sends onward |
|
||||||
|
| `Writer` | the **fd**, and the write half | text, pong, close |
|
||||||
|
| `ConnWorker` | one accepted connection | runs the HTTP layer over that fd |
|
||||||
|
|
||||||
|
`Registry` is the first honest consumer of `call`: the handler runs on the
|
||||||
|
connection worker's shard, the registry lives wherever placement put it, and
|
||||||
|
the reply is a scalar — the room's address. That is the cross-shard `call`
|
||||||
|
proof the gate asserts, not a contrivance added for it.
|
||||||
|
|
||||||
|
## Two actors per connection, not one
|
||||||
|
|
||||||
|
One fd, two directions, and they block independently. A single actor would have
|
||||||
|
to be inside `read` to notice the client, and inside `write` to deliver a
|
||||||
|
broadcast — it cannot be in both, so a broadcast would stall behind a quiet
|
||||||
|
client's read. Splitting them buys three things:
|
||||||
|
|
||||||
|
1. **The `Writer` is the sole writer of that fd.** Frames can never interleave,
|
||||||
|
which for a framed protocol is a correctness property and not a nicety.
|
||||||
|
2. **The `Reader` may block as long as it likes.** It sits in `read_dl` with a
|
||||||
|
30 s idle deadline and nothing else is waiting on it.
|
||||||
|
3. **The `Writer`'s mailbox becomes the backpressure point.** A slow client
|
||||||
|
stops draining its socket, its `Writer` blocks in `write_dl`, its mailbox
|
||||||
|
fills, and the room's next broadcast to it raises a catchable `WO_T_ACTOR`.
|
||||||
|
The room catches that and drops the member. **This is the whole reason the
|
||||||
|
mailbox cap is fail-fast** — the room survives its slowest member, and the
|
||||||
|
gate's `WO_MAILBOX=8` leg proves the path fires rather than assuming it.
|
||||||
|
|
||||||
|
`Room.say` is written around that: it shifts every member, tries the send, and
|
||||||
|
keeps only the members whose send succeeded — a failed one is sent a close and
|
||||||
|
dropped. So fan-out and eviction are the same pass.
|
||||||
|
|
||||||
|
## Who owns the fd
|
||||||
|
|
||||||
|
The `Writer`. It closes it, in every branch: a failed write sets `dead` and
|
||||||
|
closes; a close message writes the close frame and closes. The `Reader` closes
|
||||||
|
the fd itself in exactly one case — when its `send_close` to the writer traps,
|
||||||
|
meaning the writer is unreachable and nobody else will. Without that the fd
|
||||||
|
would leak on a dead-writer path.
|
||||||
|
|
||||||
|
`Writer.dead` guards against a second close, which matters because two
|
||||||
|
independent paths can decide a connection is finished (the reader seeing EOF,
|
||||||
|
and the room broadcasting shutdown).
|
||||||
|
|
||||||
|
## Shutdown choreography
|
||||||
|
|
||||||
|
On `env.stopping()` the accept loop stops and `main` sends one message to the
|
||||||
|
`Registry`, which fans out to every room; each room shifts its members and
|
||||||
|
sends each `Writer` a close; each writer writes the close frame and closes the
|
||||||
|
fd. `main` then spins — it may **not** park, because a park after the stop flag
|
||||||
|
unwinds — and returns, which is what stops the engine.
|
||||||
|
|
||||||
|
Independently, every `Reader` notices `env.stopping()` at its loop head and
|
||||||
|
runs its tail: leave the room, close the writer.
|
||||||
|
|
||||||
|
Both paths exist and that is deliberate: the reader path covers a connection
|
||||||
|
whose room is already gone, the room path covers a reader parked in a read that
|
||||||
|
has not come back yet.
|
||||||
|
|
||||||
|
**This is where iteration 40 came from.** The room path used to be unreliable:
|
||||||
|
a `Room` whose shard was idle at `SIGTERM` never adopted the shutdown message,
|
||||||
|
because an idle worker abandoned its inbox on stop. Clients that still got a
|
||||||
|
close frame were being saved by the reader path alone — which is why the
|
||||||
|
failure looked random and why a warmed-up server hid it. The engine now
|
||||||
|
guarantees that a send issued before the stop flag is delivered, so both paths
|
||||||
|
work as written. Nothing in this file changed to fix it, and that is the point:
|
||||||
|
the sample was right and the runtime was not.
|
||||||
335
docs/examples/chat/main.wo
Normal file
335
docs/examples/chat/main.wo
Normal file
|
|
@ -0,0 +1,335 @@
|
||||||
|
-- chat — iteration 24's acceptance workload. Rooms, presence and
|
||||||
|
-- broadcast over WebSocket: every connection is a reader actor (sole fd
|
||||||
|
-- reader) plus a writer actor (sole fd writer); rooms and the registry
|
||||||
|
-- are actors; delivery between them is ownership-moving sends, across
|
||||||
|
-- shards when placement lands them there. One binary, no broker.
|
||||||
|
--
|
||||||
|
-- CHAT_TOKEN is not needed — chat is open; the framework serves it
|
||||||
|
-- through [deps] exactly like web-app:
|
||||||
|
-- woc . && ./target/chat 8080
|
||||||
|
-- ws://127.0.0.1:8080/ws?room=lobby&name=alice
|
||||||
|
--
|
||||||
|
-- The actor split exists because an actor takes ONE message at a time:
|
||||||
|
-- a single per-connection actor blocked in net read could never hear a
|
||||||
|
-- broadcast. The reader owns the socket's inbound half and the carry
|
||||||
|
-- buffer; the writer owns the outbound half so frames never interleave.
|
||||||
|
use env
|
||||||
|
use net
|
||||||
|
use time
|
||||||
|
use porch
|
||||||
|
use porch/http
|
||||||
|
use porch/router
|
||||||
|
|
||||||
|
-- ---- message types (one per actor) --------------------------------------
|
||||||
|
|
||||||
|
-- To a writer: 1 = text frame, 2 = close (frame + fd close), 3 = pong.
|
||||||
|
class WriterMsg {
|
||||||
|
kind: Int
|
||||||
|
text: Text
|
||||||
|
}
|
||||||
|
|
||||||
|
-- To a room: 1 = join, 2 = leave, 3 = text, 4 = shutdown (drain).
|
||||||
|
class RoomMsg {
|
||||||
|
kind: Int
|
||||||
|
name: Text
|
||||||
|
text: Text
|
||||||
|
writer: actor WriterMsg
|
||||||
|
}
|
||||||
|
|
||||||
|
-- To the registry: 1 = lookup (a `call` — the reply is the room's
|
||||||
|
-- address), 2 = shutdown every room (a `send` on SIGTERM).
|
||||||
|
class Lookup {
|
||||||
|
kind: Int
|
||||||
|
room: Text
|
||||||
|
}
|
||||||
|
|
||||||
|
-- To a reader: everything the connection's inbound loop needs.
|
||||||
|
class ReaderMsg {
|
||||||
|
fd: net.Conn
|
||||||
|
room: actor RoomMsg
|
||||||
|
writer: actor WriterMsg
|
||||||
|
name: Text
|
||||||
|
}
|
||||||
|
|
||||||
|
-- One connection accepted, one worker: builds its own App and runs the
|
||||||
|
-- framework's keep-alive loop (the serving-slice pattern).
|
||||||
|
class Conn {
|
||||||
|
fd: net.Conn
|
||||||
|
}
|
||||||
|
|
||||||
|
-- ---- the writer: sole owner of the outbound half -------------------------
|
||||||
|
|
||||||
|
class Writer {
|
||||||
|
fd: net.Conn
|
||||||
|
dead: Int
|
||||||
|
fn receive(msg: WriterMsg) {
|
||||||
|
if self.dead == 1 { return; }
|
||||||
|
if msg.kind == 1 {
|
||||||
|
let ok = try net.write_dl(self.fd, ws_text(msg.text), 2000) catch (e) false;
|
||||||
|
if ok == false {
|
||||||
|
-- a stalled or gone client: tear the fd; the reader will see EOF
|
||||||
|
-- and route the leave through the room
|
||||||
|
self.dead = 1;
|
||||||
|
net.close(self.fd);
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if msg.kind == 3 {
|
||||||
|
let ok2 = try net.write_dl(self.fd, ws_pong(msg.text), 2000) catch (e) false;
|
||||||
|
if ok2 == false {
|
||||||
|
self.dead = 1;
|
||||||
|
net.close(self.fd);
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
-- close: the drain path (room shutdown or reader-detected close)
|
||||||
|
self.dead = 1;
|
||||||
|
let ig = try net.write_dl(self.fd, ws_close(), 1000) catch (e) false;
|
||||||
|
net.close(self.fd);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
-- ---- the room: members, presence, fan-out --------------------------------
|
||||||
|
|
||||||
|
class Mem {
|
||||||
|
w: actor WriterMsg
|
||||||
|
name: Text
|
||||||
|
}
|
||||||
|
|
||||||
|
class Room {
|
||||||
|
members: multi Mem
|
||||||
|
fn receive(msg: RoomMsg) {
|
||||||
|
if msg.kind == 1 {
|
||||||
|
push(self.members, Mem { w: msg.writer, name: "${msg.name}" });
|
||||||
|
self.say("* ${msg.name} joined");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if msg.kind == 2 {
|
||||||
|
let keep: multi Mem = [];
|
||||||
|
while len(self.members) > 0 {
|
||||||
|
let m = shift(self.members);
|
||||||
|
if m.name != msg.name { push(keep, m); }
|
||||||
|
}
|
||||||
|
self.members = keep;
|
||||||
|
self.say("* ${msg.name} left");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if msg.kind == 3 {
|
||||||
|
self.say("${msg.name}: ${msg.text}");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
-- shutdown: every member gets a close frame; the list empties
|
||||||
|
while len(self.members) > 0 {
|
||||||
|
let m = shift(self.members);
|
||||||
|
let r = try send_close(m.w) catch (e) 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
-- fan-out one line; a member whose mailbox is FULL is a slow client —
|
||||||
|
-- the fail-fast cap turns it into a drop-from-the-room (the backpressure
|
||||||
|
-- policy earning its keep)
|
||||||
|
fn say(line: Text) {
|
||||||
|
let keep: multi Mem = [];
|
||||||
|
while len(self.members) > 0 {
|
||||||
|
let m = shift(self.members);
|
||||||
|
let ok = try send_text(m.w, "${line}") catch (e) 0;
|
||||||
|
if ok == 1 {
|
||||||
|
push(keep, m);
|
||||||
|
} else {
|
||||||
|
let r = try send_close(m.w) catch (e) 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
self.members = keep;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
-- send wrappers: `try` is an expression, so give it Int results
|
||||||
|
fn send_text(w: actor WriterMsg, line: Text) -> Int {
|
||||||
|
send(w, WriterMsg { kind: 1, text: line });
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn send_close(w: actor WriterMsg) -> Int {
|
||||||
|
send(w, WriterMsg { kind: 2, text: "" });
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
-- ---- the registry: name -> room, spawn on demand --------------------------
|
||||||
|
|
||||||
|
class RoomRef {
|
||||||
|
r: actor RoomMsg
|
||||||
|
}
|
||||||
|
|
||||||
|
class Registry {
|
||||||
|
rooms: map<Text, RoomRef>
|
||||||
|
fallback: actor RoomMsg
|
||||||
|
fn receive(msg: Lookup) -> actor RoomMsg {
|
||||||
|
if msg.kind == 2 {
|
||||||
|
for k, v in self.rooms {
|
||||||
|
send(v.r, RoomMsg { kind: 4, name: "", text: "", writer: dummy_writer() });
|
||||||
|
}
|
||||||
|
return self.fallback;
|
||||||
|
}
|
||||||
|
if has(self.rooms, msg.room) == 1 {
|
||||||
|
let have = self.rooms[msg.room];
|
||||||
|
if have != nil {
|
||||||
|
return have.r;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let room: actor RoomMsg = spawn Room { members: [] };
|
||||||
|
self.rooms[msg.room] = RoomRef { r: room };
|
||||||
|
return room;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
-- RoomMsg requires a writer field on every construction; the shutdown
|
||||||
|
-- message has no meaningful one, so a throwaway satisfies the shape (it
|
||||||
|
-- never receives anything — kind 4 reads no fields).
|
||||||
|
fn dummy_writer() -> actor WriterMsg {
|
||||||
|
let w: actor WriterMsg = spawn Writer { fd: 0 - 1, dead: 1 };
|
||||||
|
return w;
|
||||||
|
}
|
||||||
|
|
||||||
|
-- ---- the reader: sole owner of the inbound half ---------------------------
|
||||||
|
|
||||||
|
class Reader {
|
||||||
|
pad: Int
|
||||||
|
fn receive(msg: ReaderMsg) {
|
||||||
|
let carry = "";
|
||||||
|
let alive = true;
|
||||||
|
while alive {
|
||||||
|
if env.stopping() { alive = false; continue; }
|
||||||
|
let got = try net.read_dl(msg.fd, 4096, 30000) catch (e) nil;
|
||||||
|
if got == nil {
|
||||||
|
-- idle deadline or I/O trap: this client is done
|
||||||
|
alive = false;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let bytes = "${got}";
|
||||||
|
if len(bytes) == 0 {
|
||||||
|
alive = false;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
carry = carry .. bytes;
|
||||||
|
let more = true;
|
||||||
|
while more {
|
||||||
|
let f = ws_parse(carry);
|
||||||
|
if f.kind == 0 {
|
||||||
|
more = false;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
carry = f.rest;
|
||||||
|
if f.kind == 1 {
|
||||||
|
send(msg.room, RoomMsg { kind: 3, name: "${msg.name}", text: f.payload, writer: msg.writer });
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if f.kind == 9 {
|
||||||
|
send(msg.writer, WriterMsg { kind: 3, text: f.payload });
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if f.kind == 10 or f.kind == 2 {
|
||||||
|
continue; -- pongs ignored; binary tolerated (echo is not chat)
|
||||||
|
}
|
||||||
|
-- close frame or protocol error: stop reading
|
||||||
|
alive = false;
|
||||||
|
more = false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
-- the tail sends must survive full mailboxes (a leave storm after a
|
||||||
|
-- mass close): a trap here would kill the reader and orphan the fd
|
||||||
|
let r1 = try send_leave(msg.room, "${msg.name}", msg.writer) catch (e) 0;
|
||||||
|
let r2 = try send_close(msg.writer) catch (e) 0;
|
||||||
|
if r2 == 0 {
|
||||||
|
-- the writer is unreachable (full/dead): close the fd ourselves
|
||||||
|
net.close(msg.fd);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn send_leave(room: actor RoomMsg, name: Text, w: actor WriterMsg) -> Int {
|
||||||
|
send(room, RoomMsg { kind: 2, name: name, text: "", writer: w });
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
-- ---- HTTP: the upgrade route + usage --------------------------------------
|
||||||
|
|
||||||
|
class WsRoute {
|
||||||
|
reg: actor Lookup
|
||||||
|
fn handle(req: Req) -> Resp {
|
||||||
|
if ws_upgrade_valid(req) == false {
|
||||||
|
return bad_request("expected a websocket upgrade");
|
||||||
|
}
|
||||||
|
let rname = req.query["room"];
|
||||||
|
if rname == nil { return bad_request("expected ?room=<name>&name=<who>"); }
|
||||||
|
let who = req.query["name"];
|
||||||
|
if who == nil { return bad_request("expected ?room=<name>&name=<who>"); }
|
||||||
|
-- the cross-shard call: this handler runs on the connection worker's
|
||||||
|
-- shard, the registry lives wherever placement put it
|
||||||
|
let room = call(self.reg, Lookup { kind: 1, room: "${rname}" });
|
||||||
|
let fd = ws_accept(req);
|
||||||
|
let w: actor WriterMsg = spawn Writer { fd: fd, dead: 0 };
|
||||||
|
let rd: actor ReaderMsg = spawn Reader { pad: 0 };
|
||||||
|
send(room, RoomMsg { kind: 1, name: "${who}", text: "", writer: w });
|
||||||
|
send(rd, ReaderMsg { fd: fd, room: room, writer: w, name: "${who}" });
|
||||||
|
return hijacked();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
class Usage {
|
||||||
|
pad: Int
|
||||||
|
fn handle(req: Req) -> Resp {
|
||||||
|
return ok_json("{\"ws\":\"/ws?room=<name>&name=<who>\"}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn build_app(reg: actor Lookup) -> App {
|
||||||
|
let app = App { middleware: [], routes: [] };
|
||||||
|
app.get("/", Usage { pad: 0 });
|
||||||
|
app.get("/ws", WsRoute { reg: reg });
|
||||||
|
return app;
|
||||||
|
}
|
||||||
|
|
||||||
|
class ConnWorker {
|
||||||
|
reg: actor Lookup
|
||||||
|
fn receive(msg: Conn) {
|
||||||
|
let app = build_app(self.reg);
|
||||||
|
app.handle_conn(msg.fd, 10000, 10000);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn main(args: multi Text) -> Int {
|
||||||
|
if len(args) < 1 {
|
||||||
|
print_err("usage: chat <port>");
|
||||||
|
return 2;
|
||||||
|
}
|
||||||
|
let port = parse_int(args[0]);
|
||||||
|
if port == nil {
|
||||||
|
print_err("chat: <port> must be a number");
|
||||||
|
return 2;
|
||||||
|
}
|
||||||
|
let fb: actor RoomMsg = spawn Room { members: [] };
|
||||||
|
let reg: actor Lookup = spawn Registry { rooms: {}, fallback: fb };
|
||||||
|
let srv = net.listen("127.0.0.1", port);
|
||||||
|
print("listening on 127.0.0.1:${port}");
|
||||||
|
while true {
|
||||||
|
if env.stopping() {
|
||||||
|
-- the drain: every room broadcasts a close frame and writers flush.
|
||||||
|
-- main must NOT park here (a park after the stop flag unwinds), so
|
||||||
|
-- it SPINS — each loop back-edge pays a reduction, and the budget
|
||||||
|
-- hands the shard to the draining actors between slices; worker
|
||||||
|
-- shards keep adopting their inboxes until the engine stops.
|
||||||
|
send(reg, Lookup { kind: 2, room: "" });
|
||||||
|
let spin = 0;
|
||||||
|
while spin < 20000000 {
|
||||||
|
spin = spin + 1;
|
||||||
|
}
|
||||||
|
net.close(srv);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
let c = net.accept_dl(srv, 250);
|
||||||
|
if c != nil {
|
||||||
|
let w: actor Conn = spawn ConnWorker { reg: reg };
|
||||||
|
send(w, Conn { fd: c });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
9
docs/examples/chat/wo.toml
Normal file
9
docs/examples/chat/wo.toml
Normal file
|
|
@ -0,0 +1,9 @@
|
||||||
|
name = "chat"
|
||||||
|
version = "0.1.0"
|
||||||
|
description = "Iteration 24's acceptance workload: rooms + presence + broadcast over WebSocket — actors on fibers across shards, one binary, no broker"
|
||||||
|
|
||||||
|
[runtime]
|
||||||
|
wo = ">= 0.1"
|
||||||
|
|
||||||
|
[deps]
|
||||||
|
porch = { git = "https://github.com/shoneyj/porch", rev = "v0.1.0" }
|
||||||
|
|
@ -56,7 +56,8 @@ place it runs.
|
||||||
feature.
|
feature.
|
||||||
- **Why `main` waits.** `main` is not an actor and has no mailbox, so it sleeps
|
- **Why `main` waits.** `main` is not an actor and has no mailbox, so it sleeps
|
||||||
rather than awaiting — the gap iteration 31's `call` closes for actors and
|
rather than awaiting — the gap iteration 31's `call` closes for actors and
|
||||||
[24's marker](../../active-slice-2026-08-23-chat-ws-lifecycle.md) tracks.
|
[iteration 24](../../stories/language-runtime-database/24-chat-websocket-workload.md)
|
||||||
|
landed 2026-08-27.
|
||||||
|
|
||||||
Reasoning under the engine side: [`database/src/CODE-LOGIC.md`](../../../database/src/CODE-LOGIC.md).
|
Reasoning under the engine side: [`database/src/CODE-LOGIC.md`](../../../database/src/CODE-LOGIC.md).
|
||||||
Contract: [`plan/oop-vm/04-db-binding.md`](../../plan/oop-vm/04-db-binding.md).
|
Contract: [`plan/oop-vm/04-db-binding.md`](../../plan/oop-vm/04-db-binding.md).
|
||||||
|
|
|
||||||
|
|
@ -24,9 +24,26 @@ strictly better. Recorded as a plan deviation.)
|
||||||
| `query N` | full equality probes on the k index (≈10 rows each), materialized and counted. |
|
| `query N` | full equality probes on the k index (≈10 rows each), materialized and counted. |
|
||||||
| `write N` | alternating inserts (disjoint k range 2e6+) and updates through query results. Corrupts the checksum by design — durability legs run on a fresh store. |
|
| `write N` | alternating inserts (disjoint k range 2e6+) and updates through query results. Corrupts the checksum by design — durability legs run on a fresh store. |
|
||||||
| `wal N` | the crash battery's vehicle: insert-only (k range 1e6+), `acked <i>` printed AFTER each insert returns — the return IS the ack (RAM applied, WAL record staged, ONE commit done). |
|
| `wal N` | the crash battery's vehicle: insert-only (k range 1e6+), `acked <i>` printed AFTER each insert returns — the return IS the ack (RAM applied, WAL record staged, ONE commit done). |
|
||||||
|
| `wmix N C` | **databasev2 4:** every op a durable write (update through a query result), C at once. Exists because `mix` writes on one op in ten with C=4 — 20 writes in a quick run, measured mean batch **1.01** — so no existing leg could show whether group commit engages. Histogram kind 2, because a replayed store still holds the seeding run's kind-0/1 `Hist` rows. Seed first. |
|
||||||
|
| `boot` | **databasev2 3:** does NOTHING. With `WO_DATA` set the runtime replays the whole log before `main` runs, so a mode with no work of its own is the only honest way to price boot |
|
||||||
| `verify` | store vs its own Meta rows: count, checksum, one unique probe. Exit 3 on mismatch. |
|
| `verify` | store vs its own Meta rows: count, checksum, one unique probe. Exit 3 on mismatch. |
|
||||||
| `verify-acked M` | after kill -9 mid-`wal`: rows 1..M exist with the right v; rows beyond M allowed (acked after the last print flushed). Exit 3 on mismatch. |
|
| `verify-acked M` | after kill -9 mid-`wal`: rows 1..M exist with the right v; rows beyond M allowed (acked after the last print flushed). Exit 3 on mismatch. |
|
||||||
|
|
||||||
|
## Env knobs
|
||||||
|
|
||||||
|
| var | effect |
|
||||||
|
| --- | --- |
|
||||||
|
| `WO_DATA=<dir>` | durability on: replay `<dir>/shard-0.wal` at boot, log every write. Without it the store is RAM-only |
|
||||||
|
| `WO_SHARDS=<n>` | shard count. **`1` means every statement runs inline on shard 0 and group commit cannot engage** — batches form only where writes queue from other shards |
|
||||||
|
| `WO_CHECKPOINT_BYTES` / `WO_CHECKPOINT_RATIO` | **databasev2 3:** the checkpoint trigger — the log must exceed the floor AND exceed the ratio times the last compaction's own size. A tiny floor forces compaction in a few writes, which is how the gate tests the policy at all; an enormous one disables it, which is how the checkpoint leg measures the same workload with and without |
|
||||||
|
| `WO_WAL_STATS=1` | **databasev2 4:** print one line at exit — `walstats batches=… records=… peak_batch=… peak_staged=… compactions=… compact_us_max=… compact_us_total=… compacted_bytes=…`. Opt-in so it does not pollute every durable program's output. Mean batch is `records/batches`; **mean 1.0 means group commit is not engaging**, which is expected for a serial writer or `WO_SHARDS=1` and a bug anywhere else |
|
||||||
|
|
||||||
|
**Do not put `WO_DATA` on `/tmp`.** It is `tmpfs` on the reference machine,
|
||||||
|
where `fdatasync` is free: the same `wmix` run measured **195 000 ops/s at p50
|
||||||
|
1 µs** there against **2200 ops/s at p50 7200 µs** on ext4. There is no
|
||||||
|
durability barrier to price on a memory filesystem. The driver keeps its stores
|
||||||
|
under `bench/` for exactly this reason.
|
||||||
|
|
||||||
## Coordination idiom (this side of iteration 31)
|
## Coordination idiom (this side of iteration 31)
|
||||||
|
|
||||||
There is no request/response surface yet: concurrent modes drive
|
There is no request/response surface yet: concurrent modes drive
|
||||||
|
|
|
||||||
|
|
@ -338,6 +338,91 @@ class Mixer {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
-- databasev2 4 part A: every op a durable write, C at once.
|
||||||
|
--
|
||||||
|
-- Why this leg exists. `mix` writes on one op in ten with C=4, so at most a
|
||||||
|
-- handful of writes are ever in flight and group commit has almost nothing to
|
||||||
|
-- batch: measured mean batch 1.01 over 3112 barriers, peak 3. That is a
|
||||||
|
-- property of the WORKLOAD, not of the mechanism, and without a write-
|
||||||
|
-- concurrent leg the iteration's payoff cannot be evaluated either way.
|
||||||
|
--
|
||||||
|
-- Updates rather than inserts: comparable to what `mixwrite` measures, and the
|
||||||
|
-- row count stays flat so a long run does not turn into a growth test.
|
||||||
|
-- Histogram kind 2, because a replayed store still holds the seeding run's
|
||||||
|
-- kind-0/1 Hist rows and merging those would report someone else's latencies.
|
||||||
|
class WJob {
|
||||||
|
ops: Int
|
||||||
|
seed: Int
|
||||||
|
kmod: Int
|
||||||
|
}
|
||||||
|
|
||||||
|
class WMixer {
|
||||||
|
id: Int
|
||||||
|
fn receive(msg: WJob) {
|
||||||
|
let hw: map<Int, Int> = {};
|
||||||
|
let s = msg.seed;
|
||||||
|
let i = 0;
|
||||||
|
while i < msg.ops {
|
||||||
|
s = lcg(s);
|
||||||
|
let key = s % msg.kmod;
|
||||||
|
let o0 = time.ticks();
|
||||||
|
for r in from x in Item where x.k == key take 1 select x {
|
||||||
|
r.v = r.v + 1;
|
||||||
|
}
|
||||||
|
hist_add(hw, time.ticks() - o0);
|
||||||
|
i = i + 1;
|
||||||
|
}
|
||||||
|
hist_dump(hw, 2);
|
||||||
|
insert Meta { tag: "wmixdone${self.id}", val: msg.ops };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn wmix_mode(total: Int, c: Int) -> Int {
|
||||||
|
let kmod = meta_val("kmod");
|
||||||
|
if kmod < 1 {
|
||||||
|
print_err("wmix: seed first");
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
let per = total / c;
|
||||||
|
if per < 1 {
|
||||||
|
per = 1;
|
||||||
|
}
|
||||||
|
let wall0 = time.ticks();
|
||||||
|
let i = 0;
|
||||||
|
while i < c {
|
||||||
|
let a: actor WJob = spawn WMixer { id: i };
|
||||||
|
send(a, WJob { ops: per, seed: 4242 + i * 7919, kmod: kmod });
|
||||||
|
i = i + 1;
|
||||||
|
}
|
||||||
|
let done = 0;
|
||||||
|
while done < c {
|
||||||
|
time.sleep(20);
|
||||||
|
done = 0;
|
||||||
|
i = 0;
|
||||||
|
while i < c {
|
||||||
|
if meta_val("wmixdone${i}") >= 0 {
|
||||||
|
done = done + 1;
|
||||||
|
}
|
||||||
|
i = i + 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let wall = time.ticks() - wall0;
|
||||||
|
let hw: map<Int, Int> = {};
|
||||||
|
let nw = 0;
|
||||||
|
for x in from x in Hist select x {
|
||||||
|
if x.kind == 2 {
|
||||||
|
if has(hw, x.b) {
|
||||||
|
set(hw, x.b, get(hw, x.b) + x.c);
|
||||||
|
} else {
|
||||||
|
set(hw, x.b, x.c);
|
||||||
|
}
|
||||||
|
nw = nw + x.c;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
report("wmix", nw, wall, hw);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
fn mix_mode(total: Int, c: Int) -> Int {
|
fn mix_mode(total: Int, c: Int) -> Int {
|
||||||
let kmod = meta_val("kmod");
|
let kmod = meta_val("kmod");
|
||||||
if kmod < 1 {
|
if kmod < 1 {
|
||||||
|
|
@ -466,7 +551,7 @@ fn all_mode(n: Int) -> Int {
|
||||||
fn usage() -> Int {
|
fn usage() -> Int {
|
||||||
print_err("usage: db-bench <mode>");
|
print_err("usage: db-bench <mode>");
|
||||||
print_err(" all N | seed N | read N | query N | write N | wal N");
|
print_err(" all N | seed N | read N | query N | write N | wal N");
|
||||||
print_err(" mix N C | msgrate N | growth N int|text | growth-verify");
|
print_err(" mix N C | wmix N C | msgrate N | growth N int|text | growth-verify");
|
||||||
print_err(" randread N R | replayseed N M | boot");
|
print_err(" randread N R | replayseed N M | boot");
|
||||||
print_err(" verify | verify-acked M");
|
print_err(" verify | verify-acked M");
|
||||||
return 2;
|
return 2;
|
||||||
|
|
@ -683,6 +768,10 @@ fn main(args: multi Text) -> Int {
|
||||||
if args[0] == "growth-verify" {
|
if args[0] == "growth-verify" {
|
||||||
return growth_verify();
|
return growth_verify();
|
||||||
}
|
}
|
||||||
|
-- Does NOTHING. With WO_DATA set the runtime replays the whole log before
|
||||||
|
-- main runs, so a mode with no work of its own measures replay plus a fixed
|
||||||
|
-- process start — which is what "boot time" has to mean. Both databasev2 1
|
||||||
|
-- (replay baseline) and databasev2 3 (checkpoint boot) price boot with it.
|
||||||
if args[0] == "boot" {
|
if args[0] == "boot" {
|
||||||
return boot_mode();
|
return boot_mode();
|
||||||
}
|
}
|
||||||
|
|
@ -746,6 +835,17 @@ fn main(args: multi Text) -> Int {
|
||||||
}
|
}
|
||||||
return randread_mode(n, rr);
|
return randread_mode(n, rr);
|
||||||
}
|
}
|
||||||
|
if args[0] == "wmix" {
|
||||||
|
if len(args) < 3 {
|
||||||
|
return usage();
|
||||||
|
}
|
||||||
|
let wc = parse_int(args[2]);
|
||||||
|
if wc == nil or wc < 1 {
|
||||||
|
print_err("db-bench: <c> must be a positive number");
|
||||||
|
return 2;
|
||||||
|
}
|
||||||
|
return wmix_mode(n, wc);
|
||||||
|
}
|
||||||
if args[0] == "mix" {
|
if args[0] == "mix" {
|
||||||
if len(args) < 3 {
|
if len(args) < 3 {
|
||||||
return usage();
|
return usage();
|
||||||
|
|
|
||||||
|
|
@ -75,7 +75,11 @@ porch = { git = "https://github.com/shoneyj/porch", rev = "v0.1.0" }
|
||||||
- **TLS: none, anywhere.** Deploy behind nginx/caddy; the proxy terminates
|
- **TLS: none, anywhere.** Deploy behind nginx/caddy; the proxy terminates
|
||||||
TLS+ALPN and gives browsers HTTP/2 while this backend speaks HTTP/1.1
|
TLS+ALPN and gives browsers HTTP/2 while this backend speaks HTTP/1.1
|
||||||
keep-alive. See the web-app sample's README for the nginx sketch.
|
keep-alive. See the web-app sample's README for the nginx sketch.
|
||||||
- `Content-Length` bodies only (no chunked encoding), no WebSockets/SSE,
|
- `Content-Length` bodies only (no chunked encoding); **WebSockets ARE
|
||||||
|
supported since 2026-08-27** — `ws_accept` (`http/ws.wo`) performs the RFC
|
||||||
|
6455 handshake and hands back the hijacked `net.Conn`, and `http/wsframe.wo`
|
||||||
|
is a pure-`.wo` frame codec; `docs/examples/chat` is the worked example and
|
||||||
|
`just chat` its gate. **SSE is still absent**, and so is chunked encoding.
|
||||||
JSON-first (no templates). Form-encoded bodies parse through
|
JSON-first (no templates). Form-encoded bodies parse through
|
||||||
`form_values(req)` (`+` and `%XX` decoded, nil on any other
|
`form_values(req)` (`+` and `%XX` decoded, nil on any other
|
||||||
content-type); multipart/form-data through `multipart_parts(req)`
|
content-type); multipart/form-data through `multipart_parts(req)`
|
||||||
|
|
@ -130,6 +134,7 @@ first (pure `.wo` cannot express it yet).
|
||||||
| Content negotiation | ✅ `media_type(req)` request-side; `accepts(req, mtype)` response-side (exact, type/*, */*; q-values stripped not ranked — ranking waits for an app serving alternates) — slice 2 |
|
| Content negotiation | ✅ `media_type(req)` request-side; `accepts(req, mtype)` response-side (exact, type/*, */*; q-values stripped not ranked — ranking waits for an app serving alternates) — slice 2 |
|
||||||
| Trusted-proxy client IP | 🔶 `client_ip(req)` parses X-Forwarded-For; `net.peer(fd)` (iteration 35) exposes the peer — the verify middleware is now a pure-`.wo` candidate slice |
|
| Trusted-proxy client IP | 🔶 `client_ip(req)` parses X-Forwarded-For; `net.peer(fd)` (iteration 35) exposes the peer — the verify middleware is now a pure-`.wo` candidate slice |
|
||||||
| Status/header setting · redirects | ✅ builders + `set_header` |
|
| Status/header setting · redirects | ✅ builders + `set_header` |
|
||||||
|
| WebSockets · pub/sub | ✅ **2026-08-27 (iteration 24)** — `ws_accept` does the RFC 6455 handshake and hands back the hijacked `net.Conn`; `http/wsframe.wo` is a pure-`.wo` frame codec. Rooms/presence/broadcast are actors in `docs/examples/chat`, gated by `just chat` (11 checks, 1000-client soak, both `WO_IO` backends, ASan clean). No SSE |
|
||||||
| Lazy body streaming + backpressure · streaming responses · explicit commit point | ⏸ UNBLOCKED by the arc (8/11 landed 2026-08-21) — stays parked until its own slice |
|
| Lazy body streaming + backpressure · streaming responses · explicit commit point | ⏸ UNBLOCKED by the arc (8/11 landed 2026-08-21) — stays parked until its own slice |
|
||||||
| ETag + conditional requests | ✅ `etag_for` (quoted base64 SHA-256) + `with_etag` (If-None-Match → 304) over iteration 34's digest builtins — slice 2 |
|
| ETag + conditional requests | ✅ `etag_for` (quoted base64 SHA-256) + `with_etag` (If-None-Match → 304) over iteration 34's digest builtins — slice 2 |
|
||||||
|
|
||||||
|
|
@ -140,7 +145,7 @@ first (pure `.wo` cannot express it yet).
|
||||||
| Ordered middleware chain | ✅ registration order, `?Resp` short-circuits |
|
| Ordered middleware chain | ✅ registration order, `?Resp` short-circuits |
|
||||||
| Request-scoped context | ✅ `req.ctx` map (slice 2): middleware writes, handlers read; identity stays in `principal` |
|
| Request-scoped context | ✅ `req.ctx` map (slice 2): middleware writes, handlers read; identity stays in `principal` |
|
||||||
| Guaranteed teardown | 🔶 every fd closes on every path (gate-proven); no user teardown hooks yet |
|
| Guaranteed teardown | 🔶 every fd closes on every path (gate-proven); no user teardown hooks yet |
|
||||||
| Cancellation into pending storage ops | ⏸ UNBLOCKED by the arc (8/11 landed 2026-08-21) — stays parked until its own slice |
|
| Cancellation into pending storage ops | ⏸ **unblocked, not built.** The arc landed 2026-08-21 and iteration 24 (2026-08-27) added the lifecycle a cancellation would ride — `call` with a catchable trap when the callee dies, bounded mailboxes, `monitor`, and `time.after` for a deadline. Nothing here consumes them yet; it stays parked until its own slice |
|
||||||
| Panic recovery | 🔶 trap = 500 and the server survives ✅; "rolls back the transaction" is framework v2 (needs `transaction { }`, iteration 18) |
|
| Panic recovery | 🔶 trap = 500 and the server survives ✅; "rolls back the transaction" is framework v2 (needs `transaction { }`, iteration 18) |
|
||||||
|
|
||||||
### Storage integration (the differentiator — framework v2 territory)
|
### Storage integration (the differentiator — framework v2 territory)
|
||||||
|
|
|
||||||
|
|
@ -29,6 +29,31 @@ Writeonce's phase 12 `Engine` keeps an `HashMap<(TypeName, SegmentOffset), Cache
|
||||||
The kernel page cache does most of the work. `pread` against an fd that already has its page cached is a memcpy. `pwrite` populates the page cache without going to disk until pressure or `fsync`. This is why writeonce explicitly does NOT use `O_DIRECT` (see [`linux/12-pwrite-fsync.md`](../linux/12-pwrite-fsync.md)) — the page cache is the one cache we want.
|
The kernel page cache does most of the work. `pread` against an fd that already has its page cached is a memcpy. `pwrite` populates the page cache without going to disk until pressure or `fsync`. This is why writeonce explicitly does NOT use `O_DIRECT` (see [`linux/12-pwrite-fsync.md`](../linux/12-pwrite-fsync.md)) — the page cache is the one cache we want.
|
||||||
|
|
||||||
## Checkpoint — the writeonce shape
|
## Checkpoint — the writeonce shape
|
||||||
|
> **⚠ TWO CORRECTIONS, 2026-08-28** (found while brainstorming
|
||||||
|
> [databasev2 3](../../../stories/databasev2/03-wal-checkpoint.md); spec:
|
||||||
|
> [`2026-08-28-wal-checkpoint-design.md`](../../../superpowers/specs/2026-08-28-wal-checkpoint-design.md)).
|
||||||
|
>
|
||||||
|
> 1. **Postgres does NOT update its control file by rename.** The claim below
|
||||||
|
> that "Postgres does the same in `BasicOpenFile` + `fsync_parent_path`" is
|
||||||
|
> wrong: `update_controlfile` (`src/common/controldata_utils.c`) opens the
|
||||||
|
> existing file `O_WRONLY`, writes a zero-padded **full block in place**, and
|
||||||
|
> relies on **CRC32C** over the struct to detect a torn write. The
|
||||||
|
> `fsync(parent_dir)` reasoning below is still correct *for renames* — it is
|
||||||
|
> just not what Postgres does here.
|
||||||
|
> 2. **The checkpoint sketch below assumes writeonce has segment files.** It
|
||||||
|
> says records before the LSN are "*known* to be in the segment files". There
|
||||||
|
> are none: the WAL is writeonce's only durable form, replayed into RAM, and
|
||||||
|
> [databasev2 2](../../../stories/databasev2/02-table-storage-modes.md)
|
||||||
|
> deliberately rejected adding a paged store. This document predates the
|
||||||
|
> databasev2 direction, so read the loop below as a design for an
|
||||||
|
> architecture that was not chosen.
|
||||||
|
>
|
||||||
|
> What survived the comparison is the **ordering discipline**, not the
|
||||||
|
> architecture: publish the new "recovery starts here" atomically and last, so a
|
||||||
|
> crash falls back. writeonce gets that from one `rename` of the whole log —
|
||||||
|
> possible only because its records are full row images, where Postgres' are
|
||||||
|
> page deltas.
|
||||||
|
|
||||||
|
|
||||||
Postgres' checkpoint runs in a separate process and signals the postmaster when done. Writeonce's runs as a periodic loop step:
|
Postgres' checkpoint runs in a separate process and signals the postmaster when done. Writeonce's runs as a periodic loop step:
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -65,7 +65,7 @@ The metadata exists for exactly one reason: `json.encode`/`json.decode` are runt
|
||||||
- **the OS half** — fs.exists/list/stat/read_all/read_at/append, time.sleep/local/iso, env.get/stopping, net.listen/accept/read/write/close, proc.run. Ids 40–56; `runtime/src/sysio.c`. A member that returns a record takes that record's **class id as its last argument**, so the VM allocates what it fills without knowing any source type name.
|
- **the OS half** — fs.exists/list/stat/read_all/read_at/append, time.sleep/local/iso, env.get/stopping, net.listen/accept/read/write/close, proc.run. Ids 40–56; `runtime/src/sysio.c`. A member that returns a record takes that record's **class id as its last argument**, so the VM allocates what it fills without knowing any source type name.
|
||||||
- **json** — encode (value + the value's static kind), decode (text + the class id to build). Ids 57–58; `runtime/src/json.c`. Decode yields the zero word on malformed input rather than trapping, which is what makes `json.decode(t) as T` a checked decode.
|
- **json** — encode (value + the value's static kind), decode (text + the class id to build). Ids 57–58; `runtime/src/json.c`. Decode yields the zero word on malformed input rather than trapping, which is what makes `json.decode(t) as T` a checked decode.
|
||||||
- **59 `map_get_opt`** (`m[k]`'s optional read), **60 `text_copy`** (Text's ownership-boundary copy — Task 1 of the executable plan).
|
- **59 `map_get_opt`** (`m[k]`'s optional read), **60 `text_copy`** (Text's ownership-boundary copy — Task 1 of the executable plan).
|
||||||
- **database** — **61 `db_insert`** (iteration 9, Task 3): window is R[B] = class id, R[B+1..] = one slot per **declared** field in declaration order; result R[A] = the new row's id. The loader validates the class-id slot statically (variable window: the field slots are validated at runtime by the engine against the class table). Engine failure traps `WO_T_DB`; a failed WAL commit traps `WO_T_IO` after un-applying the row. `database/src/db.c`.
|
- **database** — **61 `db_insert`** (iteration 9, Task 3): window is R[B] = class id, R[B+1..] = one slot per **declared** field in declaration order; result R[A] = the new row's id. The loader validates the class-id slot statically (variable window: the field slots are validated at runtime by the engine against the class table). Engine failure traps `WO_T_DB`. **A failed WAL commit no longer traps (databasev2 4, 2026-08-28): it ENDS THE PROCESS** with exit status 74 and a diagnostic naming the failing operation, the log path, `errno` and the batch size. `WO_T_IO` is unreachable from any DB write. The reason is that only `insert` could ever un-apply itself — `update` and `delete` never could, and their own comments admitted they left RAM ahead of disk — so continuing after a durability failure meant serving state that would not survive a restart. Retrying is not offered either: on Linux a failed `fsync` may already have discarded the dirty pages, so a second call can report success having written nothing. Replay is the recovery that works. `database/src/db.c`, `database/src/wal.c`.
|
||||||
|
|
||||||
**`?T` and nil.** A heap-shaped optional (`?Text`, `?Rec`, `?multi`, `?map`, `?@gc`) stores what `T` stores and spells nil as the **zero word** — every per-kind drop plan already ignores a zero slot, so `?T`'s field kind is `T`'s. A **nullable scalar** (`?Int`, `?Bool`, `?Timestamp`, `?Id`) cannot: `0` is a perfectly good `Int`, and real programs store it in a `?Int`. Its nil is therefore `WO_NIL_SCALAR` = −2^62 (not `INT64_MIN`: the compiler's own integers are 63-bit, so that value is not expressible on the emitting side). Such a field is marked `WOB_FIELD_NIL_SCALAR` in `field_class[i]`, which is how the runtime knows to write that word where it must produce absence itself — today only `json.decode` leaving a key absent, and `parse_int` on unparseable input.
|
**`?T` and nil.** A heap-shaped optional (`?Text`, `?Rec`, `?multi`, `?map`, `?@gc`) stores what `T` stores and spells nil as the **zero word** — every per-kind drop plan already ignores a zero slot, so `?T`'s field kind is `T`'s. A **nullable scalar** (`?Int`, `?Bool`, `?Timestamp`, `?Id`) cannot: `0` is a perfectly good `Int`, and real programs store it in a `?Int`. Its nil is therefore `WO_NIL_SCALAR` = −2^62 (not `INT64_MIN`: the compiler's own integers are 63-bit, so that value is not expressible on the emitting side). Such a field is marked `WOB_FIELD_NIL_SCALAR` in `field_class[i]`, which is how the runtime knows to write that word where it must produce absence itself — today only `json.decode` leaving a key absent, and `parse_int` on unparseable input.
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -118,11 +118,49 @@ R[B+1..] = one slot per declared field in declaration order (the literal's
|
||||||
order is irrelevant — slots are the class table's).
|
order is irrelevant — slots are the class table's).
|
||||||
|
|
||||||
Execution: `wo_row_insert` (RAM, engine copies every value), then — when
|
Execution: `wo_row_insert` (RAM, engine copies every value), then — when
|
||||||
durability is on — stage + **commit before the builtin returns**: the
|
durability is on — stage, then a barrier before the acknowledgment. **Updated
|
||||||
builtin's return IS the acknowledgment, so ack-after-fsync holds at
|
2026-08-28 (databasev2 4 part A): group commit landed, and the barrier's
|
||||||
statement granularity until iteration 8 brings tick-scoped group commit. A
|
location now depends on which path the statement takes.**
|
||||||
failed commit un-applies the row and traps `WO_T_IO`; engine failures trap
|
|
||||||
`WO_T_DB`. Durability is opt-in: `WO_DATA=<dir>` makes the CLI replay
|
A statement arriving from a worker shard marshals to shard 0 and parks; shard 0
|
||||||
|
stages every such request, issues **one** barrier when its queue empties, and
|
||||||
|
only then releases the held replies — so each writer is acknowledged after the
|
||||||
|
barrier that carried *its* record. A statement already running on shard 0 takes
|
||||||
|
the inline path and still commits before the builtin returns, because it has no
|
||||||
|
reply to hold: it returns into its own fiber, and batching it would require
|
||||||
|
parking that fiber on the barrier (deferred to part B). The boundary is the
|
||||||
|
queue draining, **not** the tick this document previously anticipated — a tick
|
||||||
|
would add latency to a lone writer, taxing an idle system to serve a busy one.
|
||||||
|
|
||||||
|
Measured: ~2.9× durable write throughput and ~2.1× lower p50 on a
|
||||||
|
write-concurrent workload; unchanged for a serial writer, which has nothing to
|
||||||
|
batch with.
|
||||||
|
|
||||||
|
**Compaction (databasev2 3, 2026-08-29) may run only where NOTHING IS STAGED.**
|
||||||
|
That is a correctness requirement, not a scheduling preference: the staging
|
||||||
|
buffer holds records destined for a file that compaction is about to replace, so
|
||||||
|
compacting with a non-empty buffer would either write them into a file about to
|
||||||
|
be discarded or lose them with it. In practice the safe points are immediately
|
||||||
|
after a barrier — the drain's, and the inline path's — and both are wired.
|
||||||
|
`wo_wal_compact` refuses a non-empty buffer as a backstop rather than trusting
|
||||||
|
its callers.
|
||||||
|
|
||||||
|
**Recovery is unchanged by compaction.** The result is an ordinary log in the
|
||||||
|
ordinary record grammar, replayed from byte 0; there is no snapshot, no second
|
||||||
|
source, no cutoff offset and no control file. Crash safety comes from `rename`
|
||||||
|
being atomic: before it the live log is intact and the temp file is not
|
||||||
|
authoritative, after it the new log is complete, and no reader can observe a
|
||||||
|
mixture. A crash mid-rewrite leaves a temp file, which the next open removes.
|
||||||
|
|
||||||
|
A failed compaction is a **missed optimisation, not a durability event** — the
|
||||||
|
original log is left usable and the process continues. It must not take the
|
||||||
|
fatal path below.
|
||||||
|
|
||||||
|
A failed commit **no longer traps — it ends the process** (exit 74, with a
|
||||||
|
diagnostic naming the operation, log path, `errno` and batch size). So does a
|
||||||
|
failed staging. `WO_T_IO` is unreachable from a DB write. One rule: once a
|
||||||
|
statement has mutated RAM, the outcomes are durable or death. Engine failures
|
||||||
|
still trap `WO_T_DB`. Durability is opt-in: `WO_DATA=<dir>` makes the CLI replay
|
||||||
`<dir>/shard-0.wal` before the entry runs and commit every insert; without
|
`<dir>/shard-0.wal` before the entry runs and commit every insert; without
|
||||||
it the engine is RAM-only (every corpus fixture runs that way).
|
it the engine is RAM-only (every corpus fixture runs that way).
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -279,6 +279,8 @@ unset `env.get` are nil.
|
||||||
| `sha1(bytes)` | `-> Bytes` | 20-byte digest (id 85, iteration 34) — exists because RFC 6455's Sec-WebSocket-Accept demands SHA-1 |
|
| `sha1(bytes)` | `-> Bytes` | 20-byte digest (id 85, iteration 34) — exists because RFC 6455's Sec-WebSocket-Accept demands SHA-1 |
|
||||||
| `sha256(bytes)` | `-> Bytes` | 32-byte digest (id 86, iteration 34) |
|
| `sha256(bytes)` | `-> Bytes` | 32-byte digest (id 86, iteration 34) |
|
||||||
| `hmac_sha256(key, msg)` | `-> Bytes` | RFC 2104 over SHA-256, both args Bytes (id 87, iteration 34); key > 64 bytes hashed first |
|
| `hmac_sha256(key, msg)` | `-> Bytes` | RFC 2104 over SHA-256, both args Bytes (id 87, iteration 34); key > 64 bytes hashed first |
|
||||||
|
| `monitor(watched, observer, msg)` | — | iteration 24 (id 89): the observer's own M-typed msg (MOVED) is delivered when watched dies (trap-death); already-dead delivers now; a full observer's notice is dropped with a stderr line — no fiber to trap |
|
||||||
|
| `time.after(ms, addr, msg)` | — | iteration 24 (id 90): one-shot timer — msg (MOVED) arrives as an ordinary send after ms on the arming shard; ms <= 0 delivers now; NO cancel — the generation-counter idiom (run/timer-generation) is the answer |
|
||||||
| `call(addr, msg)` | `-> R` | send that WAITS (id 88, iteration 24): the message moves like `send`'s, the caller's fiber parks until the receive's return value arrives. R = the receive's declared return type — every `receive(msg: M)` program-wide must agree on it and it must be a copyable scalar in v1 (WO-E226 otherwise). A dead callee traps WO_T_ACTOR, immediately or mid-call — a `call` never hangs |
|
| `call(addr, msg)` | `-> R` | send that WAITS (id 88, iteration 24): the message moves like `send`'s, the caller's fiber parks until the receive's return value arrives. R = the receive's declared return type — every `receive(msg: M)` program-wide must agree on it and it must be a copyable scalar in v1 (WO-E226 otherwise). A dead callee traps WO_T_ACTOR, immediately or mid-call — a `call` never hangs |
|
||||||
| `env.get(name)` | `-> ?Text` | unset is nil |
|
| `env.get(name)` | `-> ?Text` | unset is nil |
|
||||||
| `env.stopping()` | `-> Bool` | SIGTERM/SIGINT latch, handlers installed on first use |
|
| `env.stopping()` | `-> Bool` | SIGTERM/SIGINT latch, handlers installed on first use |
|
||||||
|
|
|
||||||
|
|
@ -216,3 +216,187 @@ which is too coarse for the one number a checkpoint is meant to improve.
|
||||||
|
|
||||||
WAL bytes are measured as the file's **non-zero prefix**, never its size: shard
|
WAL bytes are measured as the file's **non-zero prefix**, never its size: shard
|
||||||
WALs are `fallocate`'d to 1 MiB, so an empty store reports 1048576.
|
WALs are `fallocate`'d to 1 MiB, so an empty store reports 1048576.
|
||||||
|
## 6. WAL group commit: one barrier per drain (databasev2 4 part A)
|
||||||
|
|
||||||
|
**Measured 2026-08-28.** Before this, the engine committed per *statement*:
|
||||||
|
`db.c` called `wo_wal_commit` immediately after every append, so each row
|
||||||
|
change bought its own `pwrite` + `fdatasync`. Now shard 0 stages every queued
|
||||||
|
write request, issues one barrier, and only then releases the held replies.
|
||||||
|
|
||||||
|
### The controlled before/after
|
||||||
|
|
||||||
|
Same machine, same workload (`wmix 4000 32` — every op a durable update, 32
|
||||||
|
concurrent), same build except `db.c` and `vm.c`, two runs each, interleaved:
|
||||||
|
|
||||||
|
| | ops/sec | p50 | p99 |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| per-statement barrier | 2213 · 2177 | 7183 · 7251 µs | **20000 · 20000 µs** |
|
||||||
|
| group commit | **6216 · 6525** | **3458 · 3444 µs** | 11139 · 5971 µs |
|
||||||
|
|
||||||
|
**≈2.9× throughput, ≈2.1× lower p50.**
|
||||||
|
|
||||||
|
**The p99 "before" figure is at the histogram ceiling, not a measurement.**
|
||||||
|
`hist_add` clamps at 20000 µs, and both before-runs pinned there — so the true
|
||||||
|
before p99 is ≥20 ms and unknown. The improvement is *at least* 2.3×; the
|
||||||
|
honest statement is that the old p99 was off the end of the instrument.
|
||||||
|
|
||||||
|
### Confirmation from the committed baseline
|
||||||
|
|
||||||
|
The full campaign gives the same answer a second way. `s1` takes the inline
|
||||||
|
path, which commits per statement **by design**, so within one build the two
|
||||||
|
shard configurations are batching-off against batching-on:
|
||||||
|
|
||||||
|
| Leg | ops/sec | p50 | p99 | mean batch | peak batch |
|
||||||
|
| --- | --- | --- | --- | --- | --- |
|
||||||
|
| `durable.s1.wmix` (inline, unbatched) | 1467 | 455 µs | 721 µs | **1.0** | 1 |
|
||||||
|
| `durable.sN.wmix` (batched) | **5117** | 8208 µs | 12169 µs | **5.43** | 57 |
|
||||||
|
|
||||||
|
3.5× throughput, agreeing with the 2.9× above. Note `sN` latency is *higher*
|
||||||
|
while throughput is 3.5× better: 64 writers queueing behind one owner shard
|
||||||
|
trade per-op latency for barrier amortisation, which is what group commit is.
|
||||||
|
|
||||||
|
Batching scales with write concurrency exactly as designed — mean batch at
|
||||||
|
C = 4 / 16 / 64 was **1.13 / 1.76 / 5.35**, peak **3 / 10 / 39**.
|
||||||
|
|
||||||
|
### What did NOT improve, and why that was predicted
|
||||||
|
|
||||||
|
`durable.sN.mixwrite` went **480 → 492 ops/s** — unchanged. That is the metric
|
||||||
|
the spec *originally* named as the payoff, and correcting it was part of the
|
||||||
|
brainstorm: `mix` writes on one op in ten with C=4, so a quick run performs
|
||||||
|
**20 writes** and mean batch measured **1.01** over 3112 barriers. A workload
|
||||||
|
that never has two writes in flight cannot be helped by batching them.
|
||||||
|
`durable.*.seed` is likewise unchanged: a serial single writer has nothing to
|
||||||
|
batch with under any scheme.
|
||||||
|
|
||||||
|
**So the payoff is real but conditional: it appears exactly where concurrent
|
||||||
|
durable writes fan into the owner shard, and nowhere else.**
|
||||||
|
|
||||||
|
### Two traps worth recording
|
||||||
|
|
||||||
|
**Do not benchmark durability on `/tmp`.** It is `tmpfs` here, where
|
||||||
|
`fdatasync` is free — the same `wmix` run reported **195 000 ops/s at p50 1 µs**
|
||||||
|
there against **2200 ops/s at p50 7200 µs** on ext4. There is no barrier to
|
||||||
|
amortise on a memory filesystem, so a group-commit measurement taken there
|
||||||
|
measures nothing. `db-bench` gets this right by keeping its stores under
|
||||||
|
`bench/`.
|
||||||
|
|
||||||
|
**The record count is not the update count.** `wmix` staged 7755 records for
|
||||||
|
4000 updates because the histogram dump and the done-marker are themselves
|
||||||
|
durable inserts. They arrive as an end-of-run burst, which is batch-friendly,
|
||||||
|
so `mean_batch` is not purely update-driven. Peak staged bytes stayed small
|
||||||
|
(2793 B at C=64), which is what settled the decision to ship **no batch cap**:
|
||||||
|
the request queue's existing upstream bound is sufficient.
|
||||||
|
|
||||||
|
### The cost side: tail latency on the owner shard
|
||||||
|
|
||||||
|
Group commit is a trade, and the full battery made the other side of it visible.
|
||||||
|
|
||||||
|
**A bug first, caught by `durable.sN.mixread.p99`.** The drain initially held
|
||||||
|
*every* DB reply until the barrier — including **reads**, which stage nothing and
|
||||||
|
have no stake in durability. That parked readers behind an fsync for no reason
|
||||||
|
and pushed read p99 from ~1043 µs to **4057 µs**. Reads are now released
|
||||||
|
immediately; only a statement that actually staged a record has its reply held.
|
||||||
|
|
||||||
|
**What remains is inherent, not a bug.** A barrier now blocks the owner shard
|
||||||
|
**longer** (more records per fsync) even though it blocks **less often**, so
|
||||||
|
anything arriving during a barrier — reads included — waits behind it. Measured
|
||||||
|
across three full runs of the same build, `durable.sN.mixread.p99` came in at
|
||||||
|
**1043 / 2318 / 4147 µs** and `wmix.p99` at **8758 / 20000 µs**, a 2–4× spread
|
||||||
|
with the box near idle.
|
||||||
|
|
||||||
|
So the honest summary of part A on a single-threaded owner shard: **~3× write
|
||||||
|
throughput, at the price of a longer and noisier tail for everything queued
|
||||||
|
behind a barrier.** That is precisely what part B (async submission — submit the
|
||||||
|
barrier and keep serving) would undo, and it is a better argument for part B than
|
||||||
|
the "close the 66× gap" framing part B was originally given.
|
||||||
|
|
||||||
|
**Gating consequence.** `durable.sN.*.p99us` now carries a 100% tolerance,
|
||||||
|
because a 2–4×-variable tail gated at 50% gates the disk rather than the engine.
|
||||||
|
The **floor** is the real guard there, and it is not slack: `mixread`'s floor
|
||||||
|
(4172 µs) came within 25 µs of tripping on the worst observed run.
|
||||||
|
|
||||||
|
## 7. WAL checkpoint: compaction (databasev2 3)
|
||||||
|
|
||||||
|
**Measured 2026-08-29.** Before this the log grew forever: nothing ever removed
|
||||||
|
superseded records, so boot replayed all history and the file only ever got
|
||||||
|
bigger. Compaction rewrites it as one record per live row and swaps it in with
|
||||||
|
`rename`.
|
||||||
|
|
||||||
|
### Space and boot — the same workload, twice
|
||||||
|
|
||||||
|
Identical work, differing only in whether checkpointing may fire (an enormous
|
||||||
|
floor disables it). Full campaign:
|
||||||
|
|
||||||
|
| | checkpointing off | checkpointing on |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| WAL used | 1 962 358 B | **907 094 B** |
|
||||||
|
| boot (median of 3, `boot` mode) | 114 ms | **64 ms** |
|
||||||
|
| compactions | 0 | 6 |
|
||||||
|
|
||||||
|
**2.16× space reclaimed, 1.78× faster boot.** Boot is measured with a mode that
|
||||||
|
does nothing at all: with `WO_DATA` set the runtime replays the whole log before
|
||||||
|
`main` runs, so a mode with no work of its own is the only honest way to price
|
||||||
|
replay. It is *not* measured through the driver's `run()` helper, which samples
|
||||||
|
RSS on a 250 ms poll — timings taken that way reported "251 ms" both with and
|
||||||
|
without checkpointing, which is the harness's clock rather than the engine's.
|
||||||
|
|
||||||
|
### The stop-the-world pause, and why it stopped being 8× worse
|
||||||
|
|
||||||
|
Compaction blocks the owner shard for its duration. The spec refused to assume
|
||||||
|
that was acceptable, so it is measured and gated against a stated **50 ms**
|
||||||
|
budget: a stall a serving process can absorb without a client seeing a timeout.
|
||||||
|
|
||||||
|
Measured **2 651 µs** on the full campaign — comfortably inside it.
|
||||||
|
|
||||||
|
It was not always. The first implementation flushed the dump through
|
||||||
|
`wo_wal_commit`, which `fdatasync`s, so a dump paid one barrier per 256 records:
|
||||||
|
|
||||||
|
| live set | pause, per-flush fsync | pause, one final fsync |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| ~107 KB | 23 948 µs | **2 903 µs** |
|
||||||
|
| ~500 KB | 36 361 µs | **7 526 µs** |
|
||||||
|
| ~1.98 MB | 107 649 µs | **13 212 µs** |
|
||||||
|
|
||||||
|
Marginal rate went from **~22 MB/s to ~181 MB/s** — from sync-bound to
|
||||||
|
bandwidth-bound. Intermediate durability during a dump is worthless: the temp
|
||||||
|
file is not authoritative until the rename and is fsynced once immediately
|
||||||
|
before it, so those barriers bought nothing and cost 8×.
|
||||||
|
|
||||||
|
**The pause is O(live rows), and that is the number that eventually forces an
|
||||||
|
incremental design.** At ~181 MB/s a 1 GB live set implies roughly 5.5 s — well
|
||||||
|
past any interactive budget. The spec deliberately did not buy incremental
|
||||||
|
copying in advance; this is the measurement it is to be bought against.
|
||||||
|
|
||||||
|
### Gating
|
||||||
|
|
||||||
|
`ckpt.reclaim_x` is the feature's central claim and is gated tightly (15%).
|
||||||
|
Everything else in the leg — boot times, the pause, the byte counts — is
|
||||||
|
wall-clock or workload-shaped on a shared box and carries a wide tolerance,
|
||||||
|
because waiving them *all* would have left the leg ungated. The leg also
|
||||||
|
asserts two things directly rather than trusting a metric: that some compaction
|
||||||
|
actually ran (otherwise it proves nothing), and that the log really is smaller
|
||||||
|
with checkpointing on.
|
||||||
|
|
||||||
|
One direction bug worth recording: `reclaim_x` was first recorded as
|
||||||
|
lower-is-better by the default detector, which would have **passed "reclaimed
|
||||||
|
nothing" and failed an improvement** — the central claim gated backwards.
|
||||||
|
|
||||||
|
**Gate-tolerance corrections made while closing this iteration**, both recorded
|
||||||
|
because a widened tolerance that is not justified is indistinguishable from a
|
||||||
|
silenced regression:
|
||||||
|
|
||||||
|
- **`ckpt.pause_us_max` is no longer gated against a baseline.** The raw pause
|
||||||
|
scales with the live set, and this workload's live set is not fixed —
|
||||||
|
`wmix`'s `hist_dump` inserts a row per latency bucket, so a noisier box makes
|
||||||
|
more buckets, more rows, and a longer pause. What belongs to the engine is the
|
||||||
|
**rate**, so `ckpt.pause_us_per_mb` carries the real tolerance and the raw
|
||||||
|
pause keeps the absolute 50 ms budget as its guard.
|
||||||
|
- **`ram.*.msgrate.msgs_sec` moved from 15% to 70%, and this one is
|
||||||
|
pre-existing.** Across the ten full runs recorded on 2026-08-28/29 — several
|
||||||
|
predating the checkpoint work — it ranged **10.7M to 17.9M msgs/sec, a 1.67×
|
||||||
|
spread**. A 15% gate on a scheduling-bound throughput metric fails
|
||||||
|
intermittently whatever the engine does.
|
||||||
|
- **`durable.sN.*.p99us` moved from 100% to 300%**, with more evidence than the
|
||||||
|
first widening had: mixread p99 measured 1043 / 2318 / 4147 µs and mixwrite
|
||||||
|
1623 / 4446 µs across runs of the same build. The floors remain the real
|
||||||
|
guard, and they are not slack — mixread's came within 25 µs of tripping.
|
||||||
|
|
|
||||||
|
|
@ -67,6 +67,161 @@ behind this board; live Obsidian Dataview views:
|
||||||
|
|
||||||
## ▶ NEXT PLAN
|
## ▶ NEXT PLAN
|
||||||
|
|
||||||
|
### Landed 2026-08-29 — databasev2 3, WAL checkpoint (the chain's last link)
|
||||||
|
|
||||||
|
**Implemented last time (2026-08-29):** compaction. The log used to grow forever
|
||||||
|
— nothing removed superseded records, so boot replayed all history. It is now
|
||||||
|
rewritten as one record per live row into a temp file and swapped in with
|
||||||
|
`rename`. Six tasks, brainstormed and spec'd first
|
||||||
|
([spec](../superpowers/specs/2026-08-28-wal-checkpoint-design.md) ·
|
||||||
|
[plan](../superpowers/plans/2026-08-28-wal-checkpoint.md)).
|
||||||
|
|
||||||
|
**Key findings (measured, not asserted):** **2.16× space reclaimed**
|
||||||
|
(1 962 358 → 907 094 B), **boot 114 → 64 ms**, stop-the-world pause **2 651 µs**
|
||||||
|
against a stated 50 ms budget. Reading `.dev/reference/postgresql` was what made
|
||||||
|
the design defensible rather than lazy: **Postgres never compacts its WAL**,
|
||||||
|
because its records are page deltas and a compacted redo log is not a store —
|
||||||
|
hence heap files, a control file, a redo pointer, a second recovery source and a
|
||||||
|
separate checkpointer process. Ours are **full row images**, so a compacted log
|
||||||
|
*is* a complete store, and all of that machinery disappears. What was worth
|
||||||
|
porting is the ordering discipline — publish the switch atomically and last — and
|
||||||
|
one `rename` provides it.
|
||||||
|
|
||||||
|
**Learned — two bugs of mine that measurement found, not review:** wiring the
|
||||||
|
trigger only into the drain left **`WO_SHARDS=1` never compacting**, its log
|
||||||
|
growing forever (536 KB where multi-shard held 446 KB), because a statement on
|
||||||
|
the owner shard never enters that drain. And the dump was **8× slower than
|
||||||
|
necessary**, flushing through the committing path and paying one `fdatasync` per
|
||||||
|
256 records for durability that is worthless before the rename — one final
|
||||||
|
barrier took a 2 MB dump from 107 649 µs to 13 212 µs, ~22 MB/s to ~181 MB/s.
|
||||||
|
Separately, the crash battery's *first* version failed on correct code ~1 run in
|
||||||
|
3: it acked deletes after committing them, so a kill in between made it demand a
|
||||||
|
row the engine was right to remove. Deletes now announce intent first.
|
||||||
|
|
||||||
|
**Dependencies unblocked:** every link in the concurrency + fiber chain has now
|
||||||
|
landed its planned work — stage 3 → 22 → 24 (absorbing 31 + 34) → 40 →
|
||||||
|
databasev2 4 part A → databasev2 3. **Not "complete", precisely:** chain 5 stays
|
||||||
|
`in-progress` because databasev2 4's part B was never done, and its premise was
|
||||||
|
invalidated by part A rather than satisfied. Nothing in the chain is blocked on
|
||||||
|
anything else in it.
|
||||||
|
|
||||||
|
**Next steps:** the honest queue is (1) databasev2 2's outstanding 5c/5d, whose
|
||||||
|
`resident: keys` half is unimplemented and now carries a recorded obligation —
|
||||||
|
compaction invalidates every WAL offset it stores, so the compactor must rebuild
|
||||||
|
that map; (2) databasev2 4 **part B**, whose premise was invalidated by part A
|
||||||
|
and which needs re-brainstorming rather than starting; (3) the O(live rows)
|
||||||
|
pause, ~5.5 s at a 1 GB live set, which is the number an incremental checkpoint
|
||||||
|
must be bought against.
|
||||||
|
|
||||||
|
**`.dev/reference` used:** `postgresql` — `xlog.c` (`CreateCheckPoint`, segment
|
||||||
|
recycling), `checkpointer.c` (the time-or-volume trigger), and
|
||||||
|
`controldata_utils.c`, which also corrected a prior exploration doc: Postgres
|
||||||
|
updates its control file **in place with a CRC**, not by rename.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Landed 2026-08-28 — databasev2 4 part A, WAL group commit
|
||||||
|
|
||||||
|
**Implemented last time (2026-08-28):** one durability barrier per drain
|
||||||
|
instead of one per statement. Shard 0 stages every queued write request, holds
|
||||||
|
each reply, commits once when its queue empties, then releases all — so a writer
|
||||||
|
is acknowledged after the barrier that carried *its* record, which was the
|
||||||
|
intended contract all along and was true before only because every batch had one
|
||||||
|
member. Six tasks, brainstormed and spec'd first
|
||||||
|
([spec](../superpowers/specs/2026-08-28-wal-group-commit-design.md) ·
|
||||||
|
[plan](../superpowers/plans/2026-08-28-wal-group-commit.md)).
|
||||||
|
|
||||||
|
**Key findings (measured, not asserted):** **≈2.9× durable write throughput,
|
||||||
|
≈2.1× lower p50** on a write-concurrent workload, confirmed a second way by the
|
||||||
|
`s1`-vs-`sN` split within one build (1467 → 5117 ops/s, mean batch 1.0 → 5.43,
|
||||||
|
peak 57) — 2.9× and 3.5× agreeing. Batching scales with contention: mean batch
|
||||||
|
1.13 / 1.76 / 5.35 at C = 4 / 16 / 64. **The story's premise was wrong**: it
|
||||||
|
said "fsync-per-commit" and the engine was fsync-per-**statement**, committing
|
||||||
|
after every append at all six sites — so part A was closer to deleting calls
|
||||||
|
than adding a mechanism.
|
||||||
|
|
||||||
|
**Learned — three things the measurement corrected, not the code:**
|
||||||
|
(1) **`/tmp` is tmpfs here, where `fdatasync` is free.** The same run reported
|
||||||
|
195 000 ops/s at p50 1 µs there against 2200 at 7200 µs on ext4. A group-commit
|
||||||
|
measurement taken on a memory filesystem measures nothing; `db-bench` is right
|
||||||
|
to keep its stores under `bench/`. (2) **No existing leg could exercise the
|
||||||
|
feature** — `mix` writes on one op in ten with C=4, giving 20 writes and mean
|
||||||
|
batch 1.01, so a `wmix` write-concurrent leg had to be added or the payoff was
|
||||||
|
unevaluable either way. (3) **The before-p99 was off the instrument** —
|
||||||
|
`hist_add` clamps at 20000 µs and both before-runs pinned there, so the gain is
|
||||||
|
*at least* 2.3× and the true old p99 is unknown.
|
||||||
|
|
||||||
|
**Dependencies unblocked — and one dependency invalidated.** `WO_T_IO` is
|
||||||
|
unreachable from a DB write: a failed stage or barrier now ends the process
|
||||||
|
(exit 74, diagnosed), replacing three behaviours that disagreed — `insert`
|
||||||
|
un-applied itself while `update` and `delete` returned a catchable trap and
|
||||||
|
admitted in their own comments that they left RAM ahead of disk. **Part B's
|
||||||
|
premise is invalidated**: it was justified by "close the 66× durable gap", but
|
||||||
|
that gap is two problems. Concurrent fan-in was a batching problem and is now
|
||||||
|
~3× better; a **serial** writer waiting on one barrier is a latency problem that
|
||||||
|
batching cannot touch and io_uring does not obviously help either. Part B should
|
||||||
|
be re-brainstormed, not started.
|
||||||
|
|
||||||
|
**Next steps:** either re-brainstorm part B against its corrected premise, or
|
||||||
|
take chain 6 ([databasev2 3](databasev2/03-wal-checkpoint.md), WAL checkpoint),
|
||||||
|
which now has the replay "before" it lacked. **(Superseded 2026-08-29: it
|
||||||
|
landed.)** Two debts named rather than hidden:
|
||||||
|
the abort path is not exercised (forcing a real `fdatasync` failure needs mount
|
||||||
|
privileges), and single-shard concurrent batching needs the inline-path park —
|
||||||
|
the same machinery part B would need.
|
||||||
|
|
||||||
|
**`.dev/reference` used:** none this slice. The sources were the engine's own
|
||||||
|
code and the Linux `fsync`-failure semantics that make retrying unsound.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Landed 2026-08-27 — iteration 24, chat + actor lifecycle (absorbing 31 + 34)
|
||||||
|
|
||||||
|
**Implemented last time (2026-08-27):** the slice closed and merged to master
|
||||||
|
(`ed5334d`, fast-forward). T4 `monitor` + T5 `time.after` (ids 89/90) had
|
||||||
|
landed on the branch; this session merged master in (adopting the `porch`
|
||||||
|
rename), finished T8/T9, fixed the gate, found and fixed a runtime bug, and did
|
||||||
|
T10. Iterations 31 and 34 land inside it.
|
||||||
|
|
||||||
|
**Key findings (measured, not asserted):** finishing the gate mattered more than
|
||||||
|
finishing the sample. Making **every leg start its own server** — instead of the
|
||||||
|
drain leg inheriting the soak's warmed one — exposed that **5 of 16**
|
||||||
|
fresh-server SIGTERM drains left a client at EOF with no close frame and no
|
||||||
|
diagnostic. Traced to `shard_main`: `NEXT_RUNNABLE()` already stated the
|
||||||
|
contract ("a WORKER on stop keeps DRAINING … close frames!") but the **idle**
|
||||||
|
branch reaped and broke, abandoning its inbox. An actor between messages is
|
||||||
|
exactly that idle case. Split out as
|
||||||
|
[40](language-runtime-database/40-shutdown-drain-guarantee.md); **20 of 20
|
||||||
|
clean** after. Also measured: the fd check had been core-count dependent — lazy
|
||||||
|
per-shard init takes one `io_uring` + one `eventfd` per shard, capped at
|
||||||
|
`nproc`, so 26 → 44 on a 20-core box read as a leak. **1000 connections left it
|
||||||
|
at 44**, which settled it.
|
||||||
|
|
||||||
|
**Learned:** three of the four gate failures were **stale build artifacts**, not
|
||||||
|
code. A branch switch leaves `compiler/_build/` and `runtime/build/` holding the
|
||||||
|
other branch's binaries, and a `woc` emitting `.wob` v7 against a v6 runtime
|
||||||
|
surfaces only as "no listener" — rebuild both before believing a gate failure.
|
||||||
|
And a gate that reuses another leg's server is not merely untidy: it hid a real
|
||||||
|
bug, and when its own leg failed it orphaned a listener that broke the *next*
|
||||||
|
run. Example apps now log to `/tmp/<app>.log` so a developer can `tail -F` them.
|
||||||
|
|
||||||
|
**Dependencies unblocked:** PUBSUB2 (WebSockets + pub/sub, rejected until this
|
||||||
|
point) is done; the porch ledger's WebSocket rows are ✅ and its cancellation row
|
||||||
|
is unblocked-not-built. Chain position 4 is complete, so **the chain's next link
|
||||||
|
is [databasev2 4](databasev2/04-io-uring-commit.md)** (io_uring group-commit).
|
||||||
|
Still blocked: CSRF and sessions — iteration 34 shipped HMAC but **there is
|
||||||
|
still no RNG**, and HMAC authenticates a token without being able to mint one,
|
||||||
|
which is [39](language-runtime-database/39-web-framework-parity.md)'s leading
|
||||||
|
item.
|
||||||
|
|
||||||
|
**Next steps:** databasev2 4, or databasev2 2's outstanding 5c/5d. One debt is
|
||||||
|
named rather than hidden: iteration 40's guarantee is proven only by the chat
|
||||||
|
gate — nothing in `runtime/test/` drives `wo_engine_start`/`wo_engine_stop` and
|
||||||
|
no corpus fixture can trigger a stop, so pinning it lower needs new
|
||||||
|
multithreaded test infrastructure.
|
||||||
|
|
||||||
|
**`.dev/reference` used:** none this slice. The sources were RFC 6455, RFC
|
||||||
|
3174/4231 for the digest vectors, and the kernel's own interfaces for the drain.
|
||||||
### Landed 2026-08-27 — databasev2 1, the RAM ceiling measured
|
### Landed 2026-08-27 — databasev2 1, the RAM ceiling measured
|
||||||
|
|
||||||
**Implemented last time (2026-08-27):** databasev2 1 refined (three forks
|
**Implemented last time (2026-08-27):** databasev2 1 refined (three forks
|
||||||
|
|
@ -226,10 +381,16 @@ no reference project was consulted for the implementation).
|
||||||
**The concurrency + fiber chain — ✅ stage 3 → ✅ 22 → 🔄 24 (absorbing
|
**The concurrency + fiber chain — ✅ stage 3 → ✅ 22 → 🔄 24 (absorbing
|
||||||
31 + 34) → 23 → 32.** The chain's original order put 31 before 24; the
|
31 + 34) → 23 → 32.** The chain's original order put 31 before 24; the
|
||||||
2026-08-23 directive absorbed 31 INTO 24, and 34 resolved with it, so
|
2026-08-23 directive absorbed 31 INTO 24, and 34 resolved with it, so
|
||||||
those three are one slice. **The live slice is iteration 24** — spec and
|
those three are one slice. **Iteration 24 is nine of ten tasks landed and MERGED TO MASTER
|
||||||
plan approved 2026-08-23, executing on branch `chat-ws-lifecycle`, five
|
on 2026-08-27** (fast-forward, `ed5334d`): T1 crypto, T2 bounded mailboxes,
|
||||||
of ten tasks landed. Its running state is the marker doc
|
T3 call/reply, T4 `monitor` + T5 `time.after` (ids 89/90 — the reserved holes
|
||||||
([`2026-08-23-chat-ws-lifecycle.md`](../active-slice-2026-08-23-chat-ws-lifecycle.md)),
|
are now filled), T6 ws upgrade, T7 frame codec, T8 chat sample, T9 the chat
|
||||||
|
gate. Verified on master: chat 11 checks 0 failures at the full 1000-client
|
||||||
|
soak, runtime battery 36 suites 0 fail, compiler 556 checks 0 fail, corpus
|
||||||
|
119 checks 0 fail. Only **T10 closeout** remains — which is what still holds
|
||||||
|
stories 24/31/34 open. Finishing T9 exposed and fixed a real runtime bug,
|
||||||
|
split out as [40](language-runtime-database/40-shutdown-drain-guarantee.md). Its running state is the marker doc
|
||||||
|
(the marker doc, deleted at closeout per the convention),
|
||||||
which is the file to read for what is done and what is next; stories
|
which is the file to read for what is done and what is next; stories
|
||||||
[31](language-runtime-database/31-actor-lifecycle.md) and
|
[31](language-runtime-database/31-actor-lifecycle.md) and
|
||||||
[34](language-runtime-database/34-crypto-builtins.md) keep
|
[34](language-runtime-database/34-crypto-builtins.md) keep
|
||||||
|
|
@ -279,6 +440,9 @@ both still literal holes in `wob.h`'s builtin enum; then T8 the chat
|
||||||
sample, T9 its gate, T10 closeout setting 24/31/34 to `status: done`) → 23
|
sample, T9 its gate, T10 closeout setting 24/31/34 to `status: done`) → 23
|
||||||
(io_uring group-commit — target: close the 4.5k→297k durable gap) →
|
(io_uring group-commit — target: close the 4.5k→297k durable gap) →
|
||||||
32 (WAL checkpoint). Held tail resumes on its own precedence notes.
|
32 (WAL checkpoint). Held tail resumes on its own precedence notes.
|
||||||
|
> (**Superseded 2026-08-28:** 24 landed, and 23's part A landed with it —
|
||||||
|
> "close the 4.5k→297k durable gap" turned out to be the wrong target; see
|
||||||
|
> the databasev2 4 row.)
|
||||||
|
|
||||||
**`.dev/reference` used:** none this slice (the LW_SOAK discipline and
|
**`.dev/reference` used:** none this slice (the LW_SOAK discipline and
|
||||||
linkcheck.py precedent came from in-repo scripts).
|
linkcheck.py precedent came from in-repo scripts).
|
||||||
|
|
@ -436,11 +600,15 @@ that sequences its tasks. Read one, approve, then the next starts.
|
||||||
| 19 | [Float + Bytes](language-runtime-database/19-missing-scalar-types.md) | ✅ **landed 2026-08-20** — `.wob` v5: Float constant tag, field kinds 6/7, opcodes 34-41 (IEEE-quiet f64), builtins 70-83. Full stack: literals, arithmetic, `@table` column, WAL bit-exact replay, json fractions in / shortest-round-trip out, `?Float` reserved-NaN nil, total-order index (NaN last, `-0.0` == `+0.0`), Bytes + base64. No implicit Int/Float mixing (WO-E201); `float`/`trunc` are the only bridges. Proof: web-app price is a real Float (`{"price":9.99}`), `just web-app` 23/0; corpus 103/0 |
|
| 19 | [Float + Bytes](language-runtime-database/19-missing-scalar-types.md) | ✅ **landed 2026-08-20** — `.wob` v5: Float constant tag, field kinds 6/7, opcodes 34-41 (IEEE-quiet f64), builtins 70-83. Full stack: literals, arithmetic, `@table` column, WAL bit-exact replay, json fractions in / shortest-round-trip out, `?Float` reserved-NaN nil, total-order index (NaN last, `-0.0` == `+0.0`), Bytes + base64. No implicit Int/Float mixing (WO-E201); `float`/`trunc` are the only bridges. Proof: web-app price is a real Float (`{"price":9.99}`), `just web-app` 23/0; corpus 103/0 |
|
||||||
| 11 | [Fibers](language-runtime-database/11-fibers.md) | ✅ **landed 2026-08-21** with the arc (`just fibers` 10/0); fs-park re-scoped out of v1, disclosed in the story |
|
| 11 | [Fibers](language-runtime-database/11-fibers.md) | ✅ **landed 2026-08-21** with the arc (`just fibers` 10/0); fs-park re-scoped out of v1, disclosed in the story |
|
||||||
| 22 | [Durability, throughput, scale](language-runtime-database/22-durability-throughput-scale.md) | ✅ **landed 2026-08-21** — db-bench + baseline.json (74 metrics) + restart/kill -9 proofs both shard counts; durable 4.5k vs ram 297k inserts/s, reads O(table), msgrate 13.4M/2.45M |
|
| 22 | [Durability, throughput, scale](language-runtime-database/22-durability-throughput-scale.md) | ✅ **landed 2026-08-21** — db-bench + baseline.json (74 metrics) + restart/kill -9 proofs both shard counts; durable 4.5k vs ram 297k inserts/s, reads O(table), msgrate 13.4M/2.45M |
|
||||||
| 31 | [Actor lifecycle](language-runtime-database/31-actor-lifecycle.md) | 🔄 **absorbed into 24** (directive 2026-08-23) and half landed there: `call` request/response with a typed scalar reply (`WO_B_CALL = 88`, WO-E226), bounded mailboxes (`WO_MAILBOX`, cap 1024, catchable `WO_T_ACTOR`), and actor death that traps callers instead of hanging them. Still open: `monitor` and `time.after` — ids **89 and 90 are reserved holes** in `wob.h`, which is the machine-checkable proof of what is left. Supervision trees stay out of v1 |
|
| 31 | [Actor lifecycle](language-runtime-database/31-actor-lifecycle.md) | ✅ **LANDED 2026-08-27 inside 24** (directive 2026-08-23). All four mechanisms: `call`/reply with a typed scalar reply (`WO_B_CALL = 88`, WO-E226), bounded mailboxes (`WO_MAILBOX`, cap 1024, catchable `WO_T_ACTOR`), actor death that traps callers instead of hanging them, **`monitor` (89)** and **`time.after` (90)** — the reserved holes in `wob.h` are filled. A fifth mechanism it did not anticipate came out of proving the gate: the shutdown drain guarantee, [40](language-runtime-database/40-shutdown-drain-guarantee.md). Supervision trees stay out of v1 |
|
||||||
| 24 | [chat: WebSocket workload](language-runtime-database/24-chat-websocket-workload.md) | 🔄 **the live slice** (absorbing 31 + 34, directive 2026-08-23) — branch `chat-ws-lifecycle`, 5/10 tasks landed: crypto, bounded mailboxes, WS upgrade, frame codec, `call`/reply + actor death. Pending: `monitor`, `time.after`, the chat sample, its gate, closeout. State lives in [the marker](../active-slice-2026-08-23-chat-ws-lifecycle.md) |
|
| 24 | [chat: WebSocket workload](language-runtime-database/24-chat-websocket-workload.md) | ✅ **LANDED 2026-08-27** (absorbing 31 + 34) — all ten tasks; merged to master `ed5334d`. `just chat` **11 checks, 0 failures** at the full 1000-client soak: handshake, functional matrix on both `WO_IO` backends and on one shard, the soak, the fd invariant, the SIGTERM drain, `WO_MAILBOX=8` backpressure, ASan clean. Finishing its gate found a real runtime bug, split out as [40](language-runtime-database/40-shutdown-drain-guarantee.md) |
|
||||||
|
| 23 | [io_uring group-commit](databasev2/04-io-uring-commit.md) | ✅ **part A LANDED 2026-08-28 — group commit**, one barrier per drain instead of one per statement (the engine was fsync-per-STATEMENT, not per commit; the story's premise was wrong). Shard 0 holds each reply, commits once when its queue empties, releases all — so a writer is acked after the barrier carrying ITS record. **≈2.9× durable write throughput, ≈2.1× lower p50**, two measurement methods agreeing (2.9× controlled, 3.5× s1-vs-sN); mean batch 5.43, peak 57. A durability failure is now **fatal (exit 74), not a catchable `WO_T_IO`** — replacing three behaviours that disagreed, two of which admitted leaving RAM ahead of disk. **What it did NOT do:** `durable.sN.mixwrite` 480→492 (unchanged — that workload does 20 writes at C=4, mean batch 1.01) and `seed` unchanged (serial writers have nothing to batch with). **This row used to say "close the 66× gap"; that target was mis-stated** — the gap is two problems and part A fixes only the concurrent one. ⬜ part B (io_uring) **needs re-brainstorming**, not starting on the old premise |
|
||||||
|
| 32 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) | ✅ **LANDED 2026-08-29 — the chain's last link.** Compaction rewrites the log as one record per live row and swaps it in with `rename`, so **recovery is completely unchanged** and crash safety comes from the filesystem rather than from code. **2.16× space reclaimed** (1 962 358 → 907 094 B), **boot 114 → 64 ms**, stop-the-world pause **2 651 µs** against a stated 50 ms budget. Read `.dev/reference/postgresql` for it: PG *never* compacts its WAL — its records are page deltas, so it needs heap files, a control file, a redo pointer and a separate process. Ours are full row images, so a compacted log IS a store, which deletes all of that. `kill -9` during compaction: 40 rounds/run, 10 clean runs, and **mutation-proven** — against in-place rewrite instead of `rename` the battery fails every time. Outstanding: the **`resident: keys` offset map** (compaction moves every record; the obligation is recorded at the compactor) and the O(live rows) pause, ~5.5 s at 1 GB, which is what an incremental design must be bought against |
|
||||||
|
| 33 | [Single-file store](databasev2/07-single-file-db.md) | ⬜ off-chain, small — `WO_DATA=<path>.db` file form; driver-only (story written 2026-08-22) |
|
||||||
| 34 | [Crypto builtins](language-runtime-database/34-crypto-builtins.md) | 🔄 **code landed** as 24's T1 (`d14fa9f`): `sha1`/`sha256`/`hmac_sha256`, ids 85–87 in `wob.h`, `runtime/src/crypto.c`, RFC/FIPS vectors 18/0, corpus pin. The 24 gate that once needed it is cleared. Frontmatter keeps `status: refine` only until 24's T10 closeout sets it to `done` |
|
| 34 | [Crypto builtins](language-runtime-database/34-crypto-builtins.md) | 🔄 **code landed** as 24's T1 (`d14fa9f`): `sha1`/`sha256`/`hmac_sha256`, ids 85–87 in `wob.h`, `runtime/src/crypto.c`, RFC/FIPS vectors 18/0, corpus pin. The 24 gate that once needed it is cleared. Frontmatter keeps `status: refine` only until 24's T10 closeout sets it to `done` |
|
||||||
| 38 | [Content platform capabilities](language-runtime-database/38-content-platform-capabilities.md) | ⬜ off-chain, needs a spec — the two capability families no iteration owns, confirmed against `runtime/src/wob.h`: `fs` mutation verbs (six fs builtins, ids 40–45; `append` creates-if-absent, so nothing is ever replaced, truncated, deleted or renamed) and `net.connect` (ids 51–55 + 91–95, no connect, and no `connect()` anywhere in `runtime/src/` — so no OIDC/SMTP/object-store/webhook/federation). Driven by a `docs/examples/vault` content-collaboration workload, in 28's mould. New builtins from 96 (89/90 reserved for 31); no `.wob` bump (`WOB_VERSION 6u`, last moved by 36). Story written 2026-08-26 from the "can it build a Nextcloud?" ask |
|
| 38 | [Content platform capabilities](language-runtime-database/38-content-platform-capabilities.md) | ⬜ off-chain, needs a spec — the two capability families no iteration owns, confirmed against `runtime/src/wob.h`: `fs` mutation verbs (six fs builtins, ids 40–45; `append` creates-if-absent, so nothing is ever replaced, truncated, deleted or renamed) and `net.connect` (ids 51–55 + 91–95, no connect, and no `connect()` anywhere in `runtime/src/` — so no OIDC/SMTP/object-store/webhook/federation). Driven by a `docs/examples/vault` content-collaboration workload, in 28's mould. New builtins from 96 (89/90 reserved for 31); no `.wob` bump (`WOB_VERSION 6u`, last moved by 36). Story written 2026-08-26 from the "can it build a Nextcloud?" ask |
|
||||||
| 39 | [Web framework parity](language-runtime-database/39-web-framework-parity.md) | ⬜ off-chain, needs a spec — from [the Fiber v3.5.0 study](../plan/exploration/fiber/00-fiber-parity.md) (all 32 of its middleware read against `porch`; **nine already have a counterpart**). Leads with a **random-bytes builtin**: the framework ledger claimed CSRF/sessions were unblocked by iteration 34's HMAC, but HMAC authenticates a token and cannot mint one — there is no RNG anywhere in the runtime. Then cookies (absent both ways; `Resp.headers` being a map cannot carry two `Set-Cookie` lines), then limiter/idempotency (cheapest wins — `@table` + `time.ticks`, nothing new), sessions, CSRF, and the routing/response sugar. Streaming/SSE/compression, `@derive` binding, TTL cache, `proxy` and metrics all excluded with owners named |
|
| 39 | [Web framework parity](language-runtime-database/39-web-framework-parity.md) | ⬜ off-chain, needs a spec — from [the Fiber v3.5.0 study](../plan/exploration/fiber/00-fiber-parity.md) (all 32 of its middleware read against `porch`; **nine already have a counterpart**). Leads with a **random-bytes builtin**: the framework ledger claimed CSRF/sessions were unblocked by iteration 34's HMAC, but HMAC authenticates a token and cannot mint one — there is no RNG anywhere in the runtime. Then cookies (absent both ways; `Resp.headers` being a map cannot carry two `Set-Cookie` lines), then limiter/idempotency (cheapest wins — `@table` + `time.ticks`, nothing new), sessions, CSRF, and the routing/response sugar. Streaming/SSE/compression, `@derive` binding, TTL cache, `proxy` and metrics all excluded with owners named |
|
||||||
|
| 40 | [Shutdown drain guarantee](language-runtime-database/40-shutdown-drain-guarantee.md) | ✅ **LANDED 2026-08-27 — chain 3, with 31; split out of 24.** One rule: **a message sent before the stop flag is observed must be delivered and run before the engine stops.** Found by measurement, not review: making the chat gate's drain leg start its OWN (cold) server exposed that **5 of 16** fresh-server SIGTERM drains left a WebSocket client at EOF with no close frame and no diagnostic. Traced to `shard_main` — `NEXT_RUNNABLE()` already stated the contract ("a WORKER on stop keeps DRAINING … close frames!") but the IDLE branch reaped and broke, abandoning its inbox for teardown to free. An actor between messages is exactly that idle case, which is why a WARM soak server hid it for so long. Fix is one branch honouring the primary's drain window, yielding on an empty poll. **20 of 20 clean after**; `just chat` 11 checks 0 failures at the full 1000-client soak (which also settled the fd question: 1000 connections left the count at 44); runtime battery 36 suites 0 fail, compiler 556 checks 0 fail. Ruled out: a bigger spin (a 1 s wall-clock deadline still failed 2 of 12) and spawn-during-shutdown. Outstanding: a pin below the gate — nothing in `runtime/test/` drives the engine start/stop and no corpus fixture can trigger a stop |
|
||||||
| 37 | [wo-html components](language-runtime-database/37-wo-html-components.md) | ✅ off-chain — LANDED 2026-08-25. Raw text literal (backtick, margin stripped at lex time, `{{ }}` auto-escapes) + the component layer: `Component`/`render_all`/`Layout` in wo-html, `ok_html` moved into the framework, site and shop both migrated |
|
| 37 | [wo-html components](language-runtime-database/37-wo-html-components.md) | ✅ off-chain — LANDED 2026-08-25. Raw text literal (backtick, margin stripped at lex time, `{{ }}` auto-escapes) + the component layer: `Component`/`render_all`/`Layout` in wo-html, `ok_html` moved into the framework, site and shop both migrated |
|
||||||
| 35 | [net runtime seams](language-runtime-database/35-net-runtime-seams.md) | ⬜ off-chain — fd deadlines on the park plane, Unix sockets, peer address; owns the ledger's three 🔧 rows (story written 2026-08-22) |
|
| 35 | [net runtime seams](language-runtime-database/35-net-runtime-seams.md) | ⬜ off-chain — fd deadlines on the park plane, Unix sockets, peer address; owns the ledger's three 🔧 rows (story written 2026-08-22) |
|
||||||
| 25 | [HTTP service layer](../superpowers/plans/2026-08-01-http-service-layer.md) | ⏸ hold (2026-08-21) — story file removed; the plan doc remains |
|
| 25 | [HTTP service layer](../superpowers/plans/2026-08-01-http-service-layer.md) | ⏸ hold (2026-08-21) — story file removed; the plan doc remains |
|
||||||
|
|
@ -461,13 +629,16 @@ that sequences its tasks. Read one, approve, then the next starts.
|
||||||
| Language | 🔄 [iteration 36 — operator parity](language-runtime-database/36-operator-parity.md): `not`, bitwise `& \| ^ << >>`, hex/binary/`_` literals, compound assigns — CODE LANDED 2026-08-22 (branch operator-parity, `.wob` v6, all gates green; reference project `.dev/reference/go` drove the design). Awaiting the developer's MANUAL pass on `docs/examples/operators/` (no test fixtures by directive); unblocks story 34's pure-`.wo` HMAC question | [plan](../superpowers/plans/2026-08-22-operator-parity.md) |
|
| Language | 🔄 [iteration 36 — operator parity](language-runtime-database/36-operator-parity.md): `not`, bitwise `& \| ^ << >>`, hex/binary/`_` literals, compound assigns — CODE LANDED 2026-08-22 (branch operator-parity, `.wob` v6, all gates green; reference project `.dev/reference/go` drove the design). Awaiting the developer's MANUAL pass on `docs/examples/operators/` (no test fixtures by directive); unblocks story 34's pure-`.wo` HMAC question | [plan](../superpowers/plans/2026-08-22-operator-parity.md) |
|
||||||
| Language | the framework v1-polish slice landed 2026-08-20 (branch framework-v1, awaiting merge); next per the order: brainstorm 20/21's forks | [order](#implementation-order-re-sequenced-2026-08-21--concurrency-chain) |
|
| Language | the framework v1-polish slice landed 2026-08-20 (branch framework-v1, awaiting merge); next per the order: brainstorm 20/21's forks | [order](#implementation-order-re-sequenced-2026-08-21--concurrency-chain) |
|
||||||
| Runtime | ✅ **iteration 35 landed 2026-08-23** (branch `framework-v1b`, with framework v1 slice 2 + the serving slice): net deadlines/unix/peer (ids 91–95), fiber pooling, serve_conn + web-app fiber-per-connection — web-app gate 41/0, both WO_IO backends | [design](../superpowers/specs/2026-08-23-net-seams-park-design.md) |
|
| Runtime | ✅ **iteration 35 landed 2026-08-23** (branch `framework-v1b`, with framework v1 slice 2 + the serving slice): net deadlines/unix/peer (ids 91–95), fiber pooling, serve_conn + web-app fiber-per-connection — web-app gate 41/0, both WO_IO backends | [design](../superpowers/specs/2026-08-23-net-seams-park-design.md) |
|
||||||
| Runtime | 🔄 **iteration 24 (absorbing 31 + 34): chat + actor lifecycle** — spec + plan approved 2026-08-23 (24 absorbs 31 by directive; 34 resolved C-builtins); executing on branch `chat-ws-lifecycle` | [marker](../active-slice-2026-08-23-chat-ws-lifecycle.md) · [plan](../superpowers/plans/2026-08-23-chat-ws-lifecycle.md) |
|
|
||||||
|
|
||||||
The active slice's marker doc is
|
**No slice is active.** Iteration 24 landed 2026-08-27 and its marker doc was
|
||||||
[`docs/active-slice-2026-08-23-chat-ws-lifecycle.md`](../active-slice-2026-08-23-chat-ws-lifecycle.md)
|
deleted per the convention. Everything pending is the concurrency chain (see
|
||||||
— one file, deleted when the slice lands. Everything else pending is the
|
*Pending* below) — **the chain's next link is
|
||||||
concurrency chain (see *Pending* below); the held tail is every story
|
[databasev2 4](databasev2/04-io-uring-commit.md)** (chain 5, the io_uring
|
||||||
whose frontmatter reads `status: hold`.
|
group-commit write path, `was_language_iteration: 23`), which now has iteration
|
||||||
|
22's fsync-per-commit numbers in hand, plus databasev2 1's finding that the
|
||||||
|
write path is *not* where memory pressure bites (appending under a cap costs
|
||||||
|
~1%, random reads 273×). The held tail is every story whose frontmatter reads
|
||||||
|
`status: hold`.
|
||||||
|
|
||||||
### Landed 2026-08-14 — the compile-and-run milestone
|
### Landed 2026-08-14 — the compile-and-run milestone
|
||||||
|
|
||||||
|
|
@ -705,8 +876,8 @@ the language arc as v1 history.
|
||||||
| --- | --- | --- |
|
| --- | --- | --- |
|
||||||
| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | ✅ **MEASURED 2026-08-27** — `readiness: ready`, `status: done`; forks settled, harness landed (**148 checks**). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Also measured: **random reads over an oversized table collapse 273×** (1.85M vs 6 771 reads/s, p99 1 µs vs 487 µs) — so the two access patterns sit ~270× apart under the same pressure, and departure is a **step, not a curve**. Replay measured too: **≈5.5 µs/record, 1.9× history penalty** (10M records ≈ 55 s of boot) — iteration 3's missing "before", now gated. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from |
|
| 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | ✅ **MEASURED 2026-08-27** — `readiness: ready`, `status: done`; forks settled, harness landed (**148 checks**). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Also measured: **random reads over an oversized table collapse 273×** (1.85M vs 6 771 reads/s, p99 1 µs vs 487 µs) — so the two access patterns sit ~270× apart under the same pressure, and departure is a **step, not a curve**. Replay measured too: **≈5.5 µs/record, 1.9× history penalty** (10M records ≈ 55 s of boot) — iteration 3's missing "before", now gated. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from |
|
||||||
| 2 | [per-table storage: `durable` and `resident`](databasev2/02-table-storage-modes.md) | 🔄 **the language enrichment — the `durable` half is DONE and usable.** Two optional `@table` keys, `durable: true\|false` and `resident: all\|keys`, both defaulting to today's behaviour (all 28 existing declarations compile unchanged, no golden moved). Landed: the grammar, WO-E224 (a durable `ref` into a volatile table is refused), `.wob` v7 carrying both properties in spare `flags` bits, `durable: false` actually skipping the WAL (measured: 50 inserts → 1500 bytes durable, **0** volatile) with a mode-mismatch startup refusal, plus offset capture and read-a-row-from-an-offset. Outstanding: 5c/5d (the id→offset map and rewiring `wo_row_ptr`'s 11 call sites, slab scans and `@unique`/FK across the boundary — not yet written up), the two runtime refusals, and closeout. [spec](../superpowers/specs/2026-08-26-table-residency-design.md) · [plan](../superpowers/plans/2026-08-26-table-residency.md) |
|
| 2 | [per-table storage: `durable` and `resident`](databasev2/02-table-storage-modes.md) | 🔄 **the language enrichment — the `durable` half is DONE and usable.** Two optional `@table` keys, `durable: true\|false` and `resident: all\|keys`, both defaulting to today's behaviour (all 28 existing declarations compile unchanged, no golden moved). Landed: the grammar, WO-E224 (a durable `ref` into a volatile table is refused), `.wob` v7 carrying both properties in spare `flags` bits, `durable: false` actually skipping the WAL (measured: 50 inserts → 1500 bytes durable, **0** volatile) with a mode-mismatch startup refusal, plus offset capture and read-a-row-from-an-offset. Outstanding: 5c/5d (the id→offset map and rewiring `wo_row_ptr`'s 11 call sites, slab scans and `@unique`/FK across the boundary — not yet written up), the two runtime refusals, and closeout. [spec](../superpowers/specs/2026-08-26-table-residency-design.md) · [plan](../superpowers/plans/2026-08-26-table-residency.md) |
|
||||||
| 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ⬜ snapshot + truncate: disk reclaimed, replay bounded |
|
| 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ✅ **LANDED 2026-08-29 — the chain's last link.** Compaction rewrites the log as one record per live row and swaps it in with `rename`, so **recovery is completely unchanged** and crash safety comes from the filesystem rather than from code. **2.16× space reclaimed** (1 962 358 → 907 094 B), **boot 114 → 64 ms**, stop-the-world pause **2 651 µs** against a stated 50 ms budget. Read `.dev/reference/postgresql` for it: PG *never* compacts its WAL — its records are page deltas, so it needs heap files, a control file, a redo pointer and a separate process. Ours are full row images, so a compacted log IS a store, which deletes all of that. `kill -9` during compaction: 40 rounds/run, 10 clean runs, and **mutation-proven** — against in-place rewrite instead of `rename` the battery fails every time. Outstanding: the **`resident: keys` offset map** (compaction moves every record; the obligation is recorded at the compactor) and the O(live rows) pause, ~5.5 s at 1 GB, which is what an incremental design must be bought against |
|
||||||
| 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ⬜ **`readiness: ready` — the one startable iteration in the repo** (four forks confirmed settled 2026-08-20). Close the 66× gap iteration 22 measured (durable 4.5k vs ram 297k inserts/s) |
|
| 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ✅ **part A LANDED 2026-08-28 — group commit**, one barrier per drain instead of one per statement (the engine was fsync-per-STATEMENT, not per commit; the story's premise was wrong). Shard 0 holds each reply, commits once when its queue empties, releases all — so a writer is acked after the barrier carrying ITS record. **≈2.9× durable write throughput, ≈2.1× lower p50**, two measurement methods agreeing (2.9× controlled, 3.5× s1-vs-sN); mean batch 5.43, peak 57. A durability failure is now **fatal (exit 74), not a catchable `WO_T_IO`** — replacing three behaviours that disagreed, two of which admitted leaving RAM ahead of disk. **What it did NOT do:** `durable.sN.mixwrite` 480→492 (unchanged — that workload does 20 writes at C=4, mean batch 1.01) and `seed` unchanged (serial writers have nothing to batch with). **This row used to say "close the 66× gap"; that target was mis-stated** — the gap is two problems and part A fixes only the concurrent one. ⬜ part B (io_uring) **needs re-brainstorming**, not starting on the old premise |
|
||||||
| 5 | [Bounded tables and eviction](databasev2/05-bounded-tables-eviction.md) | ⬜ a declared capacity + refuse/evict/back-pressure, and a process-level pressure signal that sheds **before** the allocator or OS gets involved — turning the invisible failure into a managed one |
|
| 5 | [Bounded tables and eviction](databasev2/05-bounded-tables-eviction.md) | ⬜ a declared capacity + refuse/evict/back-pressure, and a process-level pressure signal that sheds **before** the allocator or OS gets involved — turning the invisible failure into a managed one |
|
||||||
| 6 | [Cold tiering](databasev2/06-cold-tiering.md) | ⚠ **largely superseded by 2** — `resident: keys` took the ceiling-raising role; its user-space-working-set premise was rejected for the kernel page cache. Mostly forks: which shape, whether the index itself fits, whether the *language* surfaces the fault cost, and whether `@unique` on a cold table is refused outright. A paged B-tree stays rejected — if tiering needs one, reject tiering |
|
| 6 | [Cold tiering](databasev2/06-cold-tiering.md) | ⚠ **largely superseded by 2** — `resident: keys` took the ceiling-raising role; its user-space-working-set premise was rejected for the kernel page cache. Mostly forks: which shape, whether the index itself fits, whether the *language* surfaces the fault cost, and whether `@unique` on a cold table is refused outright. A paged B-tree stays rejected — if tiering needs one, reject tiering |
|
||||||
| 7 | [Single-file store](databasev2/07-single-file-db.md) *(was 33)* | ⬜ `WO_DATA=<path>.db`; driver-only, independent |
|
| 7 | [Single-file store](databasev2/07-single-file-db.md) *(was 33)* | ⬜ `WO_DATA=<path>.db`; driver-only, independent |
|
||||||
|
|
|
||||||
|
|
@ -2,8 +2,8 @@
|
||||||
track: databasev2
|
track: databasev2
|
||||||
iteration: "3"
|
iteration: "3"
|
||||||
was_language_iteration: "32"
|
was_language_iteration: "32"
|
||||||
status: pending
|
status: done
|
||||||
readiness: refine
|
readiness: ready
|
||||||
chain: 6
|
chain: 6
|
||||||
---
|
---
|
||||||
|
|
||||||
|
|
@ -29,6 +29,113 @@ chain: 6
|
||||||
> replay/restart numbers to justify its policy and must compose with
|
> replay/restart numbers to justify its policy and must compose with
|
||||||
> 23's group-commit write path.
|
> 23's group-commit write path.
|
||||||
|
|
||||||
|
> **BRAINSTORMED 2026-08-28.** Spec:
|
||||||
|
> [`2026-08-28-wal-checkpoint-design.md`](../../superpowers/specs/2026-08-28-wal-checkpoint-design.md)
|
||||||
|
> · plan: [`2026-08-28-wal-checkpoint.md`](../../superpowers/plans/2026-08-28-wal-checkpoint.md)
|
||||||
|
> (6 tasks).
|
||||||
|
> Read `.dev/reference/postgresql` for this — and the conclusion was that
|
||||||
|
> Postgres' design is *unavailable* to us, which is what makes the simpler one
|
||||||
|
> legitimate.
|
||||||
|
>
|
||||||
|
> **The design in one sentence:** compact the log by rewriting it as one record
|
||||||
|
> per live row into a temp file, then `rename` it over the live WAL. Recovery is
|
||||||
|
> **completely unchanged** — boot still opens one file and replays it — and the
|
||||||
|
> crash criterion is satisfied by the filesystem rather than by code we must get
|
||||||
|
> right.
|
||||||
|
>
|
||||||
|
> **Why one file works here and not in Postgres.** Postgres never compacts its
|
||||||
|
> WAL: its records are page deltas, so a compacted redo log is not a store, and
|
||||||
|
> it must keep heap files, a control file, a redo pointer and a second recovery
|
||||||
|
> source. Ours are **full row images** — `apply_record` implements UPDATE as
|
||||||
|
> remove-then-recreate — so a compacted log *is* a complete store. That one
|
||||||
|
> difference deletes the control file, the redo pointer, the cutoff offset and
|
||||||
|
> the separate process from the design.
|
||||||
|
>
|
||||||
|
> **Forks settled:** no snapshot format (the compacted log is the snapshot); one
|
||||||
|
> source, not two; **volume-only trigger** as a ratio against the last
|
||||||
|
> compaction's own measured output, with an absolute floor — **no timer**,
|
||||||
|
> because Postgres' timer exists to bound loss from unflushed buffers and we have
|
||||||
|
> none; stop-the-world, with the pause measured against a stated budget rather
|
||||||
|
> than assumed acceptable.
|
||||||
|
>
|
||||||
|
> **The coupling that would otherwise be found late:** compaction moves every
|
||||||
|
> record, so it **invalidates every WAL offset**
|
||||||
|
> [iteration 2](02-table-storage-modes.md)'s `resident: keys` stores. The
|
||||||
|
> compactor rebuilds the offset map as it writes. Recorded now because iteration
|
||||||
|
> 2's storage half is unimplemented, so nothing breaks today — it would break
|
||||||
|
> later, looking like corruption rather than a design gap.
|
||||||
|
>
|
||||||
|
> **Measured on master 2026-08-28, grounding the whole iteration:** `seed 20000`
|
||||||
|
> leaves a 986 614-byte log; 20 000 updates take it to **2 590 262 bytes with the
|
||||||
|
> same live rows** (2.6× history for no data), and boot+verify on that store is
|
||||||
|
> **155 ms**.
|
||||||
|
|
||||||
|
## Progress — landed 2026-08-29
|
||||||
|
|
||||||
|
| # | Task | State |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| 1 | `wo_wal_compact` — rewrite, fsync, rename, fsync parent, reopen | ✅ `8ea510d` |
|
||||||
|
| 2 | a stale compaction temp is removed at open | ✅ `8bfbd4b` |
|
||||||
|
| 3 | the trigger (pure decision + env knobs) and the ordering guard | ✅ `6dbcb9a` |
|
||||||
|
| 4 | `kill -9` DURING compaction — 40 rounds, mutation-proven | ✅ `9b283d5` |
|
||||||
|
| 5 | measure space, boot and the stop-the-world pause | ✅ `d87f65a` |
|
||||||
|
| 6 | closeout | ✅ this change |
|
||||||
|
|
||||||
|
### Measured
|
||||||
|
|
||||||
|
| | checkpointing off | checkpointing on |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| WAL used | 1 962 358 B | **907 094 B** |
|
||||||
|
| boot | 114 ms | **64 ms** |
|
||||||
|
|
||||||
|
**2.16× space reclaimed, 1.78× faster boot**, stop-the-world pause **2 651 µs**
|
||||||
|
against a stated 50 ms budget. Full details, including the pause's scaling, are
|
||||||
|
in [`perf-targets.md`](../../plan/perf-targets.md) §7.
|
||||||
|
|
||||||
|
### Two bugs the work found, both mine
|
||||||
|
|
||||||
|
**Wiring only the drain left `WO_SHARDS=1` never compacting** — its log grew
|
||||||
|
forever (536 KB where the multi-shard run held 446 KB), because a statement on
|
||||||
|
the owner shard never enters that drain. Both write paths now check.
|
||||||
|
|
||||||
|
**The dump was 8× slower than it needed to be**, flushing through the
|
||||||
|
committing path and so paying one `fdatasync` per 256 records for durability
|
||||||
|
that is worthless before the rename. One final barrier took the pause from
|
||||||
|
107 649 µs to 13 212 µs on a 2 MB live set — ~22 MB/s to ~181 MB/s.
|
||||||
|
|
||||||
|
## Acceptance Criteria
|
||||||
|
|
||||||
|
Met:
|
||||||
|
|
||||||
|
- **Given** an aged store, **when** it is compacted, **then** disk is reclaimed.
|
||||||
|
✅ 2.16× on the full campaign, asserted rather than merely recorded — the leg
|
||||||
|
fails if the log is not smaller with checkpointing on.
|
||||||
|
- **Given** the same store, **when** it boots, **then** replay is bounded by the
|
||||||
|
live set rather than by history. ✅ 114 → 64 ms.
|
||||||
|
- **Given** `kill -9` at ANY instant during a checkpoint, **when** the process
|
||||||
|
restarts, **then** recovery produces the same consistent store as if the
|
||||||
|
checkpoint had never started, with no acknowledged write lost. ✅ 40 rounds
|
||||||
|
per run, 10 consecutive clean runs, and **proven to have teeth**: against the
|
||||||
|
design's rejected alternative (in-place rewrite instead of `rename`) the
|
||||||
|
battery fails every run with the log destroyed.
|
||||||
|
- **Given** the iteration-22 replay numbers, **then** a before/after delta is
|
||||||
|
recorded. ✅ `perf-targets.md` §7.
|
||||||
|
- **Given** writes arriving while a checkpoint runs, **then** the ack contract
|
||||||
|
holds. ✅ compaction runs only where nothing is staged, asserted by a test
|
||||||
|
that stages and requires refusal; `wo_wal_compact` also refuses as a backstop.
|
||||||
|
|
||||||
|
Outstanding:
|
||||||
|
|
||||||
|
- **The `resident: keys` offset map.** Compaction moves every record, so it
|
||||||
|
invalidates every WAL offset [iteration 2](02-table-storage-modes.md) stores.
|
||||||
|
The compactor must rebuild that map as it writes. **Nothing fails today**
|
||||||
|
because iteration 2's storage half is unimplemented — which is exactly why the
|
||||||
|
obligation is written at the compactor in `wal.c`, where the next implementer
|
||||||
|
hits it, rather than only in a spec they may not read.
|
||||||
|
- **The pause is O(live rows).** At ~181 MB/s a 1 GB live set implies ~5.5 s,
|
||||||
|
past any interactive budget. Incremental or forked copying was deliberately
|
||||||
|
not bought in advance; this is the number to buy it against.
|
||||||
|
|
||||||
## Goals
|
## Goals
|
||||||
|
|
||||||
- **Disk space is reclaimed.** A checkpoint writes the live store as a
|
- **Disk space is reclaimed.** A checkpoint writes the live store as a
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,7 @@
|
||||||
track: databasev2
|
track: databasev2
|
||||||
iteration: "4"
|
iteration: "4"
|
||||||
was_language_iteration: "23"
|
was_language_iteration: "23"
|
||||||
status: pending
|
status: in-progress
|
||||||
readiness: ready
|
readiness: ready
|
||||||
chain: 5
|
chain: 5
|
||||||
---
|
---
|
||||||
|
|
@ -59,6 +59,111 @@ chain: 5
|
||||||
> batch — under io_uring it becomes exactly one submission, so the two
|
> batch — under io_uring it becomes exactly one submission, so the two
|
||||||
> features compose without either knowing the other.
|
> features compose without either knowing the other.
|
||||||
|
|
||||||
|
> **BRAINSTORMED 2026-08-28 — and SPLIT IN TWO.** Spec for part A:
|
||||||
|
> [`2026-08-28-wal-group-commit-design.md`](../../superpowers/specs/2026-08-28-wal-group-commit-design.md)
|
||||||
|
> · plan: [`2026-08-28-wal-group-commit.md`](../../superpowers/plans/2026-08-28-wal-group-commit.md)
|
||||||
|
> (6 tasks).
|
||||||
|
>
|
||||||
|
> **The payoff metric is `durable.sN.mixwrite`, not the s1 numbers.** Worker
|
||||||
|
> shards hold no WAL — the runtime asserts it — so every statement on a worker
|
||||||
|
> marshals to shard 0 and parks, while a statement already on shard 0 runs
|
||||||
|
> inline. Batches form only where there is a queue, so concurrent multi-shard
|
||||||
|
> writes batch and a single-shard or serial workload does not. The baseline
|
||||||
|
> shows why that is the right target anyway: **multi-shard concurrent writes are
|
||||||
|
> 480 ops/s at p99 5888 µs against single-shard's 1023 at p99 664 — adding
|
||||||
|
> shards makes durable writing WORSE today**, because every marshaled statement
|
||||||
|
> still buys its own barrier on the owner.
|
||||||
|
>
|
||||||
|
> **The premise below needed correcting.** This story says "replace
|
||||||
|
> fsync-per-commit with io_uring group-commit", but the engine does not commit
|
||||||
|
> per commit — it commits per **statement**: `db.c` calls `wo_wal_commit`
|
||||||
|
> immediately after every append, at all six sites, so every row change is one
|
||||||
|
> `pwrite` plus one `fdatasync`. That splits the goal into two independent
|
||||||
|
> wins, and only the second needs io_uring:
|
||||||
|
>
|
||||||
|
> - **Part A — batching.** Let many statements share one barrier. The staging
|
||||||
|
> buffer already holds any number of records; today it never holds more than
|
||||||
|
> one because the caller commits immediately. Mostly a deletion of calls.
|
||||||
|
> - **Part B — async submission.** The shard submits and keeps working instead
|
||||||
|
> of blocking in `fdatasync`. Deferred until A's measurement says whether the
|
||||||
|
> blocking boundary is still the bottleneck.
|
||||||
|
>
|
||||||
|
> **A is where most of the number lives.** Iteration 22 measured durable writes
|
||||||
|
> at 4460 ops/s and mixed writes at 1023 ops/s (p99 664 µs) against 1.28M ops/s
|
||||||
|
> for durable reads — ~290× apart, essentially all of it the per-statement
|
||||||
|
> barrier.
|
||||||
|
>
|
||||||
|
> **Forks settled in the brainstorm:** batch boundary is **queue-drain** (not
|
||||||
|
> the tick this story recorded — a tick taxes an idle system to serve a busy
|
||||||
|
> one); a failure between "RAM mutated" and "record durable" is a **fatal,
|
||||||
|
> diagnosed abort**, replacing today's uneven rollback where `insert` undoes
|
||||||
|
> itself and `update`/`delete` admit in a comment that they leave RAM ahead of
|
||||||
|
> disk. **That removes `WO_T_IO` from the write path** — a language-visible
|
||||||
|
> change, recorded here deliberately.
|
||||||
|
>
|
||||||
|
> `status: in-progress` because the brainstorm is done and the spec is
|
||||||
|
> approved; the plan is next. (The `readiness` axis that would say this
|
||||||
|
> precisely lives on the unmerged `db-residency-doctrine`.)
|
||||||
|
|
||||||
|
## Progress — part A landed 2026-08-28
|
||||||
|
|
||||||
|
| # | Task | State |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| 1 | a failed barrier is detected, and fatal | ✅ `d3ff03e` |
|
||||||
|
| 2 | one barrier per drain; replies held | ✅ `b9b8a45` |
|
||||||
|
| 3 | the inline path takes the fatal rule, asymmetry documented | ✅ `a6ccdbe` |
|
||||||
|
| 4 | prove batches form — the `wmix` write-concurrent leg | ✅ `40d029c` |
|
||||||
|
| 5 | measure the payoff, gate it, record it | ✅ `d52ea8a` |
|
||||||
|
| 6 | closeout | ✅ this change |
|
||||||
|
| — | **part B — io_uring submission** | ⬜ **not started; its premise changed, see below** |
|
||||||
|
|
||||||
|
### The payoff, measured two ways
|
||||||
|
|
||||||
|
| Measurement | Before | After |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| controlled (same build, only `db.c`/`vm.c` swapped; `wmix 4000 32`) | 2213 · 2177 ops/s, p50 7183 · 7251 µs | **6216 · 6525 ops/s, p50 3458 · 3444 µs** |
|
||||||
|
| committed baseline: `s1` inline vs `sN` batched | 1467 ops/s, mean batch 1.0 | **5117 ops/s, mean batch 5.43, peak 57** |
|
||||||
|
|
||||||
|
**≈2.9× throughput, ≈2.1× lower p50**, and the two methods agree (2.9× and
|
||||||
|
3.5×). Batching scales with contention: mean batch **1.13 / 1.76 / 5.35** at
|
||||||
|
C = 4 / 16 / 64.
|
||||||
|
|
||||||
|
### The cost side, and a bug the battery caught
|
||||||
|
|
||||||
|
**Reads were being held behind the barrier.** The drain first held *every* DB
|
||||||
|
reply until the commit — including reads, which stage nothing. `mixread` p99 rose
|
||||||
|
from ~1043 µs to **4057 µs** until only staging statements had their replies
|
||||||
|
held. Caught by the gate, not by review.
|
||||||
|
|
||||||
|
**What remains is inherent:** a barrier blocks the owner shard longer (more
|
||||||
|
records per fsync) though less often, so anything queued behind one waits. Three
|
||||||
|
full runs of the same build gave `durable.sN.mixread.p99` of **1043 / 2318 /
|
||||||
|
4147 µs** — a 2–4× spread near idle. So part A buys ~3× write throughput at the
|
||||||
|
cost of a longer, noisier tail on the owner shard. `durable.sN.*.p99us` was
|
||||||
|
re-baselined at 100% tolerance for that reason, with the floor as the real guard
|
||||||
|
(`mixread`'s came within 25 µs of tripping).
|
||||||
|
|
||||||
|
**This is the strongest argument for part B** — submitting the barrier and
|
||||||
|
continuing to serve is exactly what removes this cost.
|
||||||
|
|
||||||
|
### What did NOT improve — and it was predicted
|
||||||
|
|
||||||
|
- **`durable.sN.mixwrite`: 480 → 492 ops/s, i.e. unchanged.** This was the
|
||||||
|
spec's *original* payoff metric, and correcting it was part of the brainstorm:
|
||||||
|
`mix` writes on one op in ten with C=4, so a quick run performs **20 writes**
|
||||||
|
and measured mean batch **1.01**. A workload that never has two writes in
|
||||||
|
flight cannot be helped by batching them.
|
||||||
|
- **`durable.*.seed`: unchanged.** A serial single writer has nothing to batch
|
||||||
|
with, under any scheme.
|
||||||
|
- **This board's stated target was mis-stated.** It read "close the 66× gap
|
||||||
|
iteration 22 measured (durable 4.5k vs ram 297k inserts/s)". Part A does not
|
||||||
|
close that gap and structurally cannot: `seed` is serial, and one writer
|
||||||
|
waiting on one barrier is a **latency** problem, not a batching one. Recorded
|
||||||
|
rather than quietly renumbered.
|
||||||
|
- **The before-p99 is not a measurement.** `hist_add` clamps at 20000 µs and
|
||||||
|
both before-runs pinned exactly there, so the true value is ≥20 ms and
|
||||||
|
unknown. The gain is *at least* 2.3×.
|
||||||
|
|
||||||
## Goals
|
## Goals
|
||||||
|
|
||||||
- **Replace fsync-per-commit with io_uring group-commit** on the WAL write
|
- **Replace fsync-per-commit with io_uring group-commit** on the WAL write
|
||||||
|
|
@ -77,26 +182,50 @@ chain: 5
|
||||||
|
|
||||||
## Acceptance Criteria
|
## Acceptance Criteria
|
||||||
|
|
||||||
- What to achieve?
|
Met:
|
||||||
- **Given** the io_uring write path under the iteration-22 crash battery
|
|
||||||
(concurrent writers, kill -9 mid-stream, reboot, replay),
|
- **Given** the io_uring write path under iteration 22's crash battery, **when**
|
||||||
- **when** it runs,
|
it runs, **then** every acknowledged write is present after replay. ✅ — the
|
||||||
- **then** every acknowledged write is present after replay and no
|
criterion applies unchanged to part A's batching. `crash.sN` (the batched
|
||||||
unacknowledged partial write is ever visible — the exact result the
|
path) recovered every acked row after `kill -9`, `crash.s1` likewise, and both
|
||||||
fsync path gives, so durability is provably unchanged.
|
restart legs replay byte-true. This was the one thing batching could break.
|
||||||
- What to achieve?
|
- **Given** the durable write benchmark before and after, **then** throughput is
|
||||||
- **Given** the iteration-22 durable write benchmark,
|
materially higher and p99 lower, recorded. ✅ ~2.9× and ~2.1× (p50); see
|
||||||
- **when** it is run on the fsync-per-commit path and then the io_uring
|
`perf-targets.md` §6. **Scoped honestly:** on a write-concurrent workload
|
||||||
group-commit path on the same machine,
|
only, and p99's "before" is at the histogram ceiling.
|
||||||
- **then** the io_uring path's write throughput is materially higher and
|
- **Given** batching, **when** it runs, **then** it is proven to engage rather
|
||||||
its p99 commit latency lower, with the before/after numbers recorded —
|
than assumed. ✅ mean batch 5.43, peak 57 on the gated leg, and the live
|
||||||
the payoff, measured, not asserted.
|
assertion fails the suite if the mean drops to 1.
|
||||||
- What to achieve?
|
- **Given** a durability failure, **when** it happens, **then** the engine does
|
||||||
- **Given** a kernel without io_uring (old, or restricted by seccomp),
|
not continue with RAM ahead of disk. ✅ fatal, diagnosed, exit 74 — replacing
|
||||||
- **when** the runtime starts,
|
three behaviours that disagreed.
|
||||||
- **then** it falls back to the pwrite + fdatasync path automatically and
|
|
||||||
correctly — io_uring is an accelerator, never a hard dependency, and a
|
Outstanding:
|
||||||
binary that runs everywhere is the whole project's premise.
|
|
||||||
|
- **Given** a kernel without io_uring, **when** the runtime starts, **then** it
|
||||||
|
falls back automatically. *(part B — part A adds no syscall interface, so
|
||||||
|
nothing to fall back from yet.)*
|
||||||
|
- **Single-shard concurrent batching.** A statement on shard 0 commits inline
|
||||||
|
and cannot batch; doing so needs the inline path to park its fiber on the
|
||||||
|
barrier — the same machinery part B needs. So `WO_SHARDS=1` gets no batching
|
||||||
|
at all, by design and measured (mean batch 1.0).
|
||||||
|
- **The abort path is not exercised.** Forcing a real `fdatasync` failure needs a
|
||||||
|
full or read-only filesystem, which the gate cannot arrange without mount
|
||||||
|
privileges. The unit test proves the error is *detected*; the exit three lines
|
||||||
|
later is covered by inspection. Disclosed rather than papered over — iteration
|
||||||
|
40 was exactly a fatal path nothing exercised.
|
||||||
|
|
||||||
|
## Part B — its premise changed
|
||||||
|
|
||||||
|
Part B was justified by "close the 66× durable gap". Part A shows that framing
|
||||||
|
was wrong: the gap is **two** problems. Concurrent write fan-in was a batching
|
||||||
|
problem and is now ~3× better. What remains is a **serial** writer waiting on a
|
||||||
|
single barrier, which no amount of batching can help — and io_uring does not
|
||||||
|
obviously help it either, since one writer still needs one durable barrier
|
||||||
|
before its ack. Part B's real candidates are overlapping the barrier with other
|
||||||
|
work on the shard, and the inline-path park that single-shard batching also
|
||||||
|
needs. **It should be re-brainstormed against that, not started on the old
|
||||||
|
premise.**
|
||||||
|
|
||||||
## Out Of Scope
|
## Out Of Scope
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,6 @@
|
||||||
---
|
---
|
||||||
iteration: "24"
|
iteration: "24"
|
||||||
status: in-progress
|
status: done
|
||||||
readiness: ready
|
readiness: ready
|
||||||
chain: 4
|
chain: 4
|
||||||
---
|
---
|
||||||
|
|
@ -20,6 +20,34 @@ chain: 4
|
||||||
> bounded mailboxes, actor death, timers). Iteration 19 LANDED
|
> bounded mailboxes, actor death, timers). Iteration 19 LANDED
|
||||||
> 2026-08-20, so Bytes is available for frame parse/serialize.
|
> 2026-08-20, so Bytes is available for frame parse/serialize.
|
||||||
|
|
||||||
|
> **✅ LANDED 2026-08-27** (branch `chat-ws-lifecycle`, merged to master
|
||||||
|
> `ed5334d`). Ten tasks: crypto (T1), bounded mailboxes (T2), `call`/reply and
|
||||||
|
> actor death (T3), `monitor` (T4), `time.after` (T5), the WS upgrade seam
|
||||||
|
> (T6), the pure-`.wo` frame codec (T7), the chat sample (T8), the gate (T9),
|
||||||
|
> this closeout (T10). It absorbed [31](31-actor-lifecycle.md) and
|
||||||
|
> [34](34-crypto-builtins.md), which land with it.
|
||||||
|
>
|
||||||
|
> **Gate — `just chat`, 11 checks, 0 failures** at the full 1000-client soak:
|
||||||
|
> handshake with an independently recomputed accept-key, the functional matrix
|
||||||
|
> (presence, broadcast, room isolation, leave) on **both** `WO_IO` backends and
|
||||||
|
> on a single shard, the 1k hot-room soak, the fd invariant, the SIGTERM drain,
|
||||||
|
> `WO_MAILBOX=8` backpressure, and an ASan run with zero leaks. Battery
|
||||||
|
> alongside: runtime 36 suites 0 fail, compiler 556 checks, corpus 119 checks.
|
||||||
|
> The sample logs to `/tmp/chat.log`.
|
||||||
|
>
|
||||||
|
> **Two disclosed deviations from the spec.** `monitor` takes **three**
|
||||||
|
> arguments (`watched, observer, msg`) rather than two, because the caller may
|
||||||
|
> be `main`, which has no mailbox and cannot be an implicit observer. And a
|
||||||
|
> `call` reply is a **typed scalar** in v1 — which is what let the agreement be
|
||||||
|
> checked at compile time (WO-E226) instead of carried as a tagged value.
|
||||||
|
>
|
||||||
|
> **What finishing the gate found.** Making every leg start its own server
|
||||||
|
> exposed a real runtime bug the warmed soak server had been hiding: on a fresh
|
||||||
|
> server, 5 of 16 SIGTERM drains left a client at EOF with no close frame. It
|
||||||
|
> was not this sample's fault — the fix is an engine guarantee, split out as
|
||||||
|
> [40](40-shutdown-drain-guarantee.md). Design notes:
|
||||||
|
> [`docs/examples/chat/CODE-LOGIC.md`](../../examples/chat/CODE-LOGIC.md).
|
||||||
|
|
||||||
## Why this iteration exists
|
## Why this iteration exists
|
||||||
|
|
||||||
Everything the framework ledger parks behind concurrency — WebSockets,
|
Everything the framework ledger parks behind concurrency — WebSockets,
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,6 @@
|
||||||
---
|
---
|
||||||
iteration: "31"
|
iteration: "31"
|
||||||
status: in-progress
|
status: done
|
||||||
readiness: ready
|
readiness: ready
|
||||||
chain: 3
|
chain: 3
|
||||||
---
|
---
|
||||||
|
|
@ -17,6 +17,22 @@ chain: 3
|
||||||
> ([iteration 24](24-chat-websocket-workload.md)) cannot be written
|
> ([iteration 24](24-chat-websocket-workload.md)) cannot be written
|
||||||
> honestly without these four mechanisms.
|
> honestly without these four mechanisms.
|
||||||
|
|
||||||
|
> **✅ LANDED 2026-08-27 — INSIDE [24](24-chat-websocket-workload.md)**, per
|
||||||
|
> the 2026-08-23 directive that absorbed it. All four mechanisms shipped:
|
||||||
|
> `call`/reply with a typed scalar reply (id 88, WO-E226), **bounded mailboxes**
|
||||||
|
> (`WO_MAILBOX`, default 1024, fail-fast with a catchable `WO_T_ACTOR`),
|
||||||
|
> **actor death** that traps callers instead of hanging them, `monitor`
|
||||||
|
> (id 89) and `time.after` (id 90). Ids 89 and 90 were reserved holes in
|
||||||
|
> `wob.h`; they are filled.
|
||||||
|
>
|
||||||
|
> **A fifth mechanism was added that this story did not anticipate**: the
|
||||||
|
> shutdown drain guarantee, [40](40-shutdown-drain-guarantee.md). It is
|
||||||
|
> lifecycle semantics — this story gave actors a death notice, 40 gives the
|
||||||
|
> program a shutdown that does not lose mail — and it was found by measurement
|
||||||
|
> while proving 24's gate, not by review.
|
||||||
|
>
|
||||||
|
> How each piece works: `runtime/src/CODE-LOGIC.md`, "Actor lifecycle".
|
||||||
|
|
||||||
## Why this iteration exists
|
## Why this iteration exists
|
||||||
|
|
||||||
The arc's stages 1+2 shipped `spawn`/`send` mechanism without lifecycle:
|
The arc's stages 1+2 shipped `spawn`/`send` mechanism without lifecycle:
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,6 @@
|
||||||
---
|
---
|
||||||
iteration: "34"
|
iteration: "34"
|
||||||
status: in-progress
|
status: done
|
||||||
readiness: ready
|
readiness: ready
|
||||||
---
|
---
|
||||||
|
|
||||||
|
|
@ -19,6 +19,18 @@ readiness: ready
|
||||||
> Off the concurrency chain but **gates chain position 4**: iteration
|
> Off the concurrency chain but **gates chain position 4**: iteration
|
||||||
> 24's WebSocket handshake needs SHA-1 before chat can land.
|
> 24's WebSocket handshake needs SHA-1 before chat can land.
|
||||||
|
|
||||||
|
> **✅ LANDED 2026-08-27 — inside [24](24-chat-websocket-workload.md)** as its
|
||||||
|
> task 1. The fork resolved to **C builtins**: `sha1` (85), `sha256` (86),
|
||||||
|
> `hmac_sha256` (87), each over one buffer returning a fresh `Bytes`. Pinned to
|
||||||
|
> the published vectors — RFC 3174, the SHA-256 vectors, RFC 4231 — in
|
||||||
|
> `runtime/test/test_crypto.c`, 18 checks, plus a corpus fixture hashing "abc"
|
||||||
|
> from `.wo`. This unblocked chain position 4: the WebSocket handshake needs
|
||||||
|
> SHA-1, and `just chat` verifies the accept-key independently.
|
||||||
|
>
|
||||||
|
> **The gap it did NOT close:** there is still no RNG in the runtime. HMAC
|
||||||
|
> authenticates a token and cannot mint one, so CSRF and sessions stay blocked
|
||||||
|
> — which is why [39](39-web-framework-parity.md) leads with a random-bytes
|
||||||
|
> builtin rather than treating them as unblocked.
|
||||||
## Why this iteration exists
|
## Why this iteration exists
|
||||||
|
|
||||||
Four consumers already wait on it, none able to proceed:
|
Four consumers already wait on it, none able to proceed:
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,158 @@
|
||||||
|
---
|
||||||
|
iteration: "40"
|
||||||
|
status: done
|
||||||
|
chain: 3
|
||||||
|
---
|
||||||
|
|
||||||
|
# iteration 40 — the shutdown drain guarantee: a send before the stop flag is delivered
|
||||||
|
|
||||||
|
> Part of [Story — one language, one runtime, one database, one binary](00-story.md).
|
||||||
|
>
|
||||||
|
> **Split out of [24](24-chat-websocket-workload.md) on 2026-08-27** because it
|
||||||
|
> is a runtime *semantic*, not a task in a sample's gate. It belongs to the
|
||||||
|
> actor lifecycle ([31](31-actor-lifecycle.md), absorbed into 24) and it is
|
||||||
|
> the half of "lifecycle" that nothing had stated: 31 gave actors a death
|
||||||
|
> notice, this gives the program a shutdown that does not lose mail.
|
||||||
|
>
|
||||||
|
> **Found by measurement, not review.** The chat gate's drain leg had been
|
||||||
|
> passing only because it drained a server the 1k soak had already warmed.
|
||||||
|
> Making every leg start its own server exposed it:
|
||||||
|
> [`2026-08-27-chat-drain-finding.md`](../../2026-08-27-chat-drain-finding.md).
|
||||||
|
|
||||||
|
## The rule
|
||||||
|
|
||||||
|
**A message sent before the stop flag is observed must be delivered and run
|
||||||
|
before the engine stops.** One sentence, and it is the whole iteration. It is a
|
||||||
|
guarantee, not a tuning parameter — which is why a spin count could never
|
||||||
|
express it.
|
||||||
|
|
||||||
|
What it does *not* promise: that a message sent *after* the flag is delivered,
|
||||||
|
that a parked fiber is resumed, or that an actor gets unbounded time. The drain
|
||||||
|
window is the primary's, and it closes when the primary returns.
|
||||||
|
|
||||||
|
## The bug, as measured
|
||||||
|
|
||||||
|
Fresh server, two WebSocket clients, `SIGTERM`, both must receive a close frame:
|
||||||
|
|
||||||
|
| Sample | Result |
|
||||||
|
| --- | --- |
|
||||||
|
| 5 fresh servers | 1 failure (`eof\|close`) |
|
||||||
|
| 12 fresh servers | 3 failures, one `eof\|eof` |
|
||||||
|
| 16 fresh servers | 5 failures |
|
||||||
|
|
||||||
|
The failing client's socket reaches EOF with **no close frame and no
|
||||||
|
diagnostic** — the process exits and the kernel closes the fd.
|
||||||
|
|
||||||
|
Traced with instrumentation on the sample's actors: `main` → Registry → Room →
|
||||||
|
Writer. The Registry runs and sees its room. The **Room never processes the
|
||||||
|
shutdown message**, so the Writer's close branch never runs. Clients that did
|
||||||
|
get a frame were saved by their own Reader noticing `env.stopping()`, not by the
|
||||||
|
room broadcast.
|
||||||
|
|
||||||
|
## The design, as built
|
||||||
|
|
||||||
|
`runtime/src/vm.c` already encoded the correct contract in `NEXT_RUNNABLE()`:
|
||||||
|
a worker that takes a stop while it has a live fiber returns 2 and **keeps
|
||||||
|
draining its inbox** until the primary sets `eng_shutdown`. Its comment says so
|
||||||
|
in as many words — "queued shutdown messages (close frames!) still run".
|
||||||
|
|
||||||
|
`shard_main`'s own idle branch contradicted it. A worker with an empty run queue
|
||||||
|
waits in `wo_io_wait`, and on `WO_IO_STOP` it called `fib_reap_all` and
|
||||||
|
**broke** — abandoning whatever was still in its inbox, which `wo_engine_stop`
|
||||||
|
then freed wholesale during teardown.
|
||||||
|
|
||||||
|
So the failure needed a shard that was *idle* at `SIGTERM`. A Room actor between
|
||||||
|
messages is exactly that, which is why the warm soak server hid it: warm shards
|
||||||
|
had live fibers and took the correct path.
|
||||||
|
|
||||||
|
The fix makes the idle branch obey the same contract: while the primary's drain
|
||||||
|
window is open, an idle worker adopts its inbox and runs what arrives, yielding
|
||||||
|
between empty polls so a drain cannot become a hot spin across every core. Only
|
||||||
|
`eng_shutdown` — set by the primary after `main` returns — ends it.
|
||||||
|
|
||||||
|
One branch, in one place, matching a contract the file already stated.
|
||||||
|
|
||||||
|
## Progress
|
||||||
|
|
||||||
|
| Piece | State |
|
||||||
|
| --- | --- |
|
||||||
|
| the idle-worker drain branch in `shard_main` (`runtime/src/vm.c`) | ✅ one branch, matching the contract `NEXT_RUNNABLE()` already stated |
|
||||||
|
| `sched_yield` on an empty poll so the drain cannot hot-spin | ✅ |
|
||||||
|
| fresh-server drain, repeated | ✅ **20 of 20**, from 5-in-16 failing |
|
||||||
|
| chat gate at the default 1k soak | ✅ **11 checks, 0 failures** — 1000/1000 clients, both `WO_IO` backends, ASan clean |
|
||||||
|
| full runtime battery (this touches every actor program's shard loop) | ✅ **36 suites** (18 × both dispatch flavors), 0 fail, `cli_smoke: OK`; compiler 556 checks 0 fail |
|
||||||
|
| the regression pin | ✅ the chat gate's drain leg, now that it starts its OWN (cold) server — that decoupling is what caught this. **Not** a corpus fixture or unit test: nothing in `runtime/test/` drives `wo_engine_start`/`wo_engine_stop` today, and no corpus fixture can trigger a stop, so pinning it below the gate means new multithreaded test infrastructure — named as its own cost, not smuggled in here |
|
||||||
|
|
||||||
|
**Measured 2026-08-27.** Before: 5 of 16 fresh-server drains left a client at
|
||||||
|
EOF. After: **20 of 20 clean.** At the observed failure rate, 20 clean runs by
|
||||||
|
luck would be about 0.04%, so this is the fix rather than a quieter race.
|
||||||
|
|
||||||
|
## Acceptance Criteria
|
||||||
|
|
||||||
|
Met:
|
||||||
|
|
||||||
|
- **Given** a fresh server with two connected WebSocket clients, **when** it is
|
||||||
|
sent `SIGTERM`, **then** both clients receive a close frame — **repeatedly**,
|
||||||
|
not once. The bug reproduced at 5 in 16, so a single green run proves nothing;
|
||||||
|
the criterion is a run of at least 16 with zero failures.
|
||||||
|
✅ **20 of 20**, from 5-in-16 failing. A single run would have proved nothing.
|
||||||
|
- **Given** an actor whose shard is idle at the moment of the stop, **when** a
|
||||||
|
message is sent to it before the stop flag is observed, **then** its
|
||||||
|
`receive` runs before the engine stops. ✅ this is exactly the case that
|
||||||
|
failed — the Room between messages — and it is what the branch now covers.
|
||||||
|
- **Given** the drain window, **when** a worker has nothing to adopt, **then**
|
||||||
|
it does not hot-spin. ✅ `sched_yield()` on an empty poll; the 1k soak's RSS
|
||||||
|
and timing legs are unchanged (marker reached all 1000 in 28 ms).
|
||||||
|
- **Given** `just chat`, **when** it runs at the default soak, **then** all
|
||||||
|
legs pass on both `WO_IO` backends and under the ASan build with zero leaks.
|
||||||
|
✅ 11 checks, 0 failures. The fd leg also settled the lazy-init question at
|
||||||
|
scale: **1000 connections left the count at 44**, unchanged after 20 more.
|
||||||
|
- **Given** the full runtime battery, **when** it runs, **then** no suite
|
||||||
|
regresses — this touches the shard loop every actor program uses. ✅ 36 suites
|
||||||
|
0 fail, plus the compiler's 556 checks.
|
||||||
|
- **Given** a program with no worker shards (`WO_SHARDS=1`), **when** it stops,
|
||||||
|
**then** behaviour is unchanged. ✅ the gate's `WO_SHARDS=1` leg passes, and
|
||||||
|
the branch is unreachable there — `wo_engine_stop` returns early at
|
||||||
|
`nshards <= 1`, so a single-shard program never enters a worker loop.
|
||||||
|
|
||||||
|
Outstanding:
|
||||||
|
|
||||||
|
- **A pin below the gate.** The guarantee is currently proven by the chat gate
|
||||||
|
only. Nothing in `runtime/test/` drives `wo_engine_start`/`wo_engine_stop`,
|
||||||
|
and no corpus fixture can trigger a stop, so pinning it lower means new
|
||||||
|
multithreaded test infrastructure. Named as its own cost rather than assumed
|
||||||
|
cheap.
|
||||||
|
|
||||||
|
## Out Of Scope
|
||||||
|
|
||||||
|
- **Unbounded drain.** The window is the primary's and closes when `main`
|
||||||
|
returns. A program that wants longer holds the window open itself.
|
||||||
|
- **Delivering sends issued *after* the stop flag.** Nothing promises that, and
|
||||||
|
promising it would mean a program could refuse to exit.
|
||||||
|
- **Resuming parked fibers on stop.** `WO_SYS_STOPPED` unwinds them; that
|
||||||
|
contract is iteration 24's and stays.
|
||||||
|
- **A shutdown acknowledgement in the language surface.** The alternative fix
|
||||||
|
was a barrier the sample builds itself, rejected below.
|
||||||
|
- **`main` parking after the stop flag.** Still forbidden — a park after the
|
||||||
|
flag unwinds. `main` still spins; the point is that spinning now works
|
||||||
|
because the workers cooperate.
|
||||||
|
|
||||||
|
## Info — the forks, settled
|
||||||
|
|
||||||
|
1. **Engine guarantee, not a sample barrier.** The alternative was an
|
||||||
|
acknowledged drain: rooms confirm back to `main`, which waits. Rejected —
|
||||||
|
`main` cannot park after the stop flag, so it could only spin on the
|
||||||
|
acknowledgement anyway, and every future actor program would have to
|
||||||
|
re-implement the same handshake to avoid losing mail. A guarantee is stated
|
||||||
|
once; a barrier is re-invented per program.
|
||||||
|
2. **Not the spin budget.** Replacing the sample's `spin < 20000000` with a 1 s
|
||||||
|
wall-clock deadline still failed 2 of 12. More time cannot help when the
|
||||||
|
shard is not scheduled at all, and the reverted attempt cost a fixed second
|
||||||
|
on every shutdown. Recorded because a bigger spin is the obvious wrong fix.
|
||||||
|
3. **Not `dummy_writer()`.** Hoisting the shutdown message's placeholder actor
|
||||||
|
out of the drain path (it spawned during shutdown) left 5 of 16 failing.
|
||||||
|
4. **Yield rather than spin in the idle drain.** A worker polling an empty
|
||||||
|
inbox in a tight loop would burn a core per shard during the window and
|
||||||
|
starve the actors being drained.
|
||||||
|
5. **Chain position 3**, with [31](31-actor-lifecycle.md): it is lifecycle
|
||||||
|
semantics, and [24](24-chat-websocket-workload.md)'s gate is what proves it.
|
||||||
265
docs/superpowers/plans/2026-08-28-wal-checkpoint.md
Normal file
265
docs/superpowers/plans/2026-08-28-wal-checkpoint.md
Normal file
|
|
@ -0,0 +1,265 @@
|
||||||
|
# databasev2 3 — WAL checkpoint (implementation plan)
|
||||||
|
|
||||||
|
> **For agentic workers:** REQUIRED SUB-SKILL: Use
|
||||||
|
> superpowers:subagent-driven-development (recommended) or
|
||||||
|
> superpowers:executing-plans to implement this plan task-by-task. Steps
|
||||||
|
> use checkbox (`- [ ]`) syntax for tracking.
|
||||||
|
>
|
||||||
|
> **Style rule (user convention):** concept, reason, and required
|
||||||
|
> behaviour in words plus verification commands only — no implementation
|
||||||
|
> or test code blocks; the executor writes the code.
|
||||||
|
|
||||||
|
**Goal:** reclaim disk and bound replay by rewriting the log as one record per
|
||||||
|
live row and swapping it in with `rename`, so boot replays a short log instead
|
||||||
|
of all history.
|
||||||
|
|
||||||
|
**Architecture:** compaction writes the live store into a temporary file using
|
||||||
|
the existing record grammar and the existing append path, fsyncs it, renames it
|
||||||
|
over the live WAL, fsyncs the parent directory, and reopens the descriptor.
|
||||||
|
Recovery is untouched — boot still opens one file and replays it — and every
|
||||||
|
crash point is safe because `rename` is atomic.
|
||||||
|
|
||||||
|
**Tech Stack:** C11, libc only. `pwrite`, `fdatasync`, `rename`, `open`,
|
||||||
|
`unlink`. No new dependency and no new file format.
|
||||||
|
|
||||||
|
**Spec:** [`../specs/2026-08-28-wal-checkpoint-design.md`](../specs/2026-08-28-wal-checkpoint-design.md)
|
||||||
|
|
||||||
|
## Global Constraints
|
||||||
|
|
||||||
|
- **Recovery must not change.** No second source, no cutoff offset, no control
|
||||||
|
file. If a task finds itself editing the replay path, something has gone
|
||||||
|
wrong with the design and it should stop rather than proceed.
|
||||||
|
- **Every crash point falls back.** Before the rename the live log is untouched;
|
||||||
|
after it the new log is complete. There must be no window in which a reader
|
||||||
|
could observe a mixture.
|
||||||
|
- **The record grammar is frozen.** The whole argument for this design is that
|
||||||
|
it already suffices. A compacted log is INSERT records for live rows, ids
|
||||||
|
preserved exactly.
|
||||||
|
- **Bounded memory.** `stage()` grows the staging buffer by doubling and never
|
||||||
|
shrinks it, so dumping a whole store through one buffer would hold the entire
|
||||||
|
store in RAM — the unbounded growth databasev2 1 identified as how this engine
|
||||||
|
dies. The dump must flush periodically.
|
||||||
|
- **Compaction may run only where nothing is staged** — in practice immediately
|
||||||
|
after a barrier. Anywhere else, a staged record lands in a file about to be
|
||||||
|
replaced.
|
||||||
|
- **libc only**, no new syscall interface. Gates run through `just`. Never
|
||||||
|
commit on `master`; branch first.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Task 1 — `wo_wal_compact`: rewrite, fsync, rename, reopen
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `database/src/wal.c`, `database/src/wal.h`.
|
||||||
|
- Test: `runtime/test/test_wal.c`.
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Produces: a compaction entry point taking the live WAL and the store, which
|
||||||
|
replaces the log with one INSERT record per live row and leaves the WAL usable
|
||||||
|
(descriptor reopened, offset correct). Returns success or failure; a failure
|
||||||
|
must leave the ORIGINAL log intact and usable, because a failed checkpoint is
|
||||||
|
not a durability event.
|
||||||
|
- Consumes: the existing append path and commit routine, and the bitmap walk
|
||||||
|
that `db.c` already performs in three places.
|
||||||
|
|
||||||
|
- [ ] Read three things first and confirm them, because the design rests on
|
||||||
|
them: `apply_record` implements UPDATE as remove-then-recreate (so records are
|
||||||
|
full row images), `wo_wal_append_insert` takes an id and reads the row from
|
||||||
|
the store (so ids are preserved), and the tail scan treats a zero length field
|
||||||
|
as end-of-log (so the new file must be zero beyond its records).
|
||||||
|
- [ ] Test first, RED: build a store, age it (insert rows, then update the same
|
||||||
|
rows repeatedly so history exceeds live data), compact, then assert **both**
|
||||||
|
that the log got materially shorter AND that a fresh replay of it produces the
|
||||||
|
same rows with the same ids and the same values. Shorter alone is worthless —
|
||||||
|
a truncating bug also passes that.
|
||||||
|
- [ ] Verify RED for the right reason: the entry point does not exist yet.
|
||||||
|
- [ ] Implement the walk: for each class, iterate slots via the bitmap and
|
||||||
|
append one INSERT per live row. Reuse the append path; do not write a second
|
||||||
|
encoder.
|
||||||
|
- [ ] **Flush every K records rather than staging the whole store.** Point a
|
||||||
|
scratch WAL at the temp descriptor and commit periodically. State the chosen K
|
||||||
|
and why in a comment. Without this the dump holds the entire store in RAM.
|
||||||
|
- [ ] Sequence the switch exactly: fsync the temp file, `rename` over the live
|
||||||
|
path, **fsync the parent directory** (the rename is atomic in-kernel but the
|
||||||
|
directory entry is not durable until the parent is synced), then reopen the
|
||||||
|
descriptor — the old one refers to an unlinked inode — and reset the offset to
|
||||||
|
the new end of log.
|
||||||
|
- [ ] Handle failure without losing data: any error before the rename must
|
||||||
|
unlink the temp file and leave the live log untouched. A failed compaction is
|
||||||
|
a missed optimisation, **not** a durability failure, so it must NOT take the
|
||||||
|
fatal path databasev2 4 introduced.
|
||||||
|
- [ ] GREEN: `just wovm-test`.
|
||||||
|
- [ ] Commit.
|
||||||
|
|
||||||
|
## Task 2 — a stale temp file is removed, never read
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `database/src/wal.c` (the open path).
|
||||||
|
- Test: `runtime/test/test_wal.c`.
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Consumes: Task 1's temp-file naming.
|
||||||
|
- Produces: the guarantee that a crash mid-rewrite leaves nothing that can be
|
||||||
|
mistaken for data.
|
||||||
|
|
||||||
|
- [ ] Test first, RED: place a temp file next to the log containing *plausible,
|
||||||
|
well-formed records* (not garbage — garbage would be rejected anyway and would
|
||||||
|
prove nothing), open the store, and assert the temp file is gone and the
|
||||||
|
replayed store is exactly what the live log said.
|
||||||
|
- [ ] Verify RED for the right reason.
|
||||||
|
- [ ] Remove any stale temp file when the WAL is opened. Note in a comment why
|
||||||
|
this is safe: the only way one exists is a crash before a rename, and its
|
||||||
|
contents are by definition not yet authoritative.
|
||||||
|
- [ ] GREEN: `just wovm-test`.
|
||||||
|
- [ ] Commit.
|
||||||
|
|
||||||
|
## Task 3 — the trigger, and the ordering guard
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `database/src/wal.c`, `database/src/wal.h` (remember the last
|
||||||
|
compaction's size; the policy decision), `runtime/src/vm.c` (call the check
|
||||||
|
after the barrier).
|
||||||
|
- Test: `runtime/test/test_wal.c`.
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Consumes: Task 1's compaction entry point.
|
||||||
|
- Produces: automatic compaction, and the invariant that it never runs with
|
||||||
|
records staged.
|
||||||
|
|
||||||
|
- [ ] Extract the policy as a **pure decision** — given the log's used bytes,
|
||||||
|
the bytes the last compaction wrote, and a floor, should we compact? Pure
|
||||||
|
because it is then unit-testable without a store, which is the only way this
|
||||||
|
policy gets tested at all.
|
||||||
|
- [ ] Test the decision directly, RED then GREEN: below the floor it never
|
||||||
|
fires however bad the ratio; above the floor it fires exactly when used bytes
|
||||||
|
exceed the multiple; with no prior compaction it uses the floor alone.
|
||||||
|
- [ ] Record the bytes each compaction wrote, so the denominator is measured
|
||||||
|
rather than estimated. Estimating the live size would mean estimating Text,
|
||||||
|
and the compactor already knows the true number.
|
||||||
|
- [ ] Expose the floor and the ratio as env knobs, matching the existing idiom
|
||||||
|
(`WO_MAILBOX`, `WO_HEAP_MB`, `WO_SHARDS`, `WO_WAL_STATS`). **This is what
|
||||||
|
makes the policy testable** — a test sets a tiny floor and forces compaction
|
||||||
|
in a few writes instead of waiting for megabytes. Document them beside the
|
||||||
|
others. **Deviation from the spec, disclosed:** the spec spoke of a "manual
|
||||||
|
trigger for tests"; env-tunable thresholds serve that purpose without adding
|
||||||
|
language surface, which is the cheaper way to buy the same testability.
|
||||||
|
- [ ] **No timer.** If the implementer is tempted, the reason is in the spec:
|
||||||
|
Postgres' `CheckPointTimeout` bounds loss from unflushed buffers, our records
|
||||||
|
are durable at commit, and an idle log does not grow.
|
||||||
|
- [ ] Call the check from the one place that is safe — immediately after the
|
||||||
|
drain's barrier, where nothing is staged. Comment that this is a correctness
|
||||||
|
requirement and not a scheduling preference.
|
||||||
|
- [ ] Verify the guard: a test that stages records and then makes the policy
|
||||||
|
say yes must find compaction deferred, not executed. This is the assertion
|
||||||
|
that keeps the ordering rule true as the code moves.
|
||||||
|
- [ ] Verify durability is unaffected: `just db-bench --quick` — the crash and
|
||||||
|
restart legs must be unchanged, and part A's `wmix` legs must still batch.
|
||||||
|
- [ ] Commit.
|
||||||
|
|
||||||
|
## Task 4 — kill -9 *during* compaction
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Test: `runtime/test/test_wal.c` (extend the existing fork-based crash
|
||||||
|
battery).
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Consumes: Tasks 1–3.
|
||||||
|
- Produces: the evidence for the criterion the whole design is shaped around.
|
||||||
|
|
||||||
|
- [ ] Read the existing crash battery first: a forked child inserts and acks
|
||||||
|
each committed id over a pipe while the parent SIGKILLs it mid-stream, then
|
||||||
|
the parent verifies every acked id survived. Extend that shape rather than
|
||||||
|
inventing a second harness.
|
||||||
|
- [ ] Drive compaction repeatedly in the child (a tiny floor makes it fire
|
||||||
|
often) while it inserts and acks, and kill at many instants so the kill lands
|
||||||
|
inside a rewrite, at the rename, and after it.
|
||||||
|
- [ ] Assert the property, not a state: after replay the store must equal
|
||||||
|
**either** the pre-compaction **or** the post-compaction content — never a
|
||||||
|
mixture — and **every acked id must be present**. A test that only checks "it
|
||||||
|
replayed without error" would pass on a silently truncated log.
|
||||||
|
- [ ] Assert no temp file survives a kill in a way that affects the next boot.
|
||||||
|
- [ ] Run the battery repeatedly, not once: this is a race, and one green run
|
||||||
|
proves very little. State how many repetitions were run in the commit message.
|
||||||
|
- [ ] GREEN: `just wovm-test` plus the repetitions.
|
||||||
|
- [ ] Commit.
|
||||||
|
|
||||||
|
## Task 5 — measure: space, boot, and the pause
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `scripts/db-bench.py` (a checkpoint leg), `docs/plan/perf-targets.md`,
|
||||||
|
`bench/baseline.json` (refresh, with the reason in the commit message).
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Consumes: Tasks 1–3.
|
||||||
|
- Produces: the before/after record, and the pause number the spec deliberately
|
||||||
|
refused to assume.
|
||||||
|
|
||||||
|
- [ ] Capture the before numbers already measured on master, rather than
|
||||||
|
re-deriving them: `seed 20000` leaves a 986 614-byte log; 20 000 updates take
|
||||||
|
it to 2 590 262 bytes **with the same live rows**; boot+verify on that aged
|
||||||
|
store is 155 ms.
|
||||||
|
- [ ] Add a leg that ages a store, compacts it, and records: bytes before and
|
||||||
|
after, the ratio reclaimed, and boot time before and after. Age it by
|
||||||
|
updating the same rows — history must grow while the live set does not, or the
|
||||||
|
leg is measuring insert throughput instead of compaction.
|
||||||
|
- [ ] Measure the **stop-the-world pause** on the largest store the harness
|
||||||
|
builds and record it as a number. State the budget it must meet.
|
||||||
|
- [ ] **If the pause exceeds the budget, stop and report it.** That is the
|
||||||
|
finding the spec asked for, and the alternatives (incremental copy,
|
||||||
|
fork-and-dump) are bought against this number — not before it.
|
||||||
|
- [ ] Give the new metrics tolerances that match what they are: bytes reclaimed
|
||||||
|
is structural and can be gated tightly; the pause is wall-clock on a shared
|
||||||
|
box and cannot. Do not waive them all, which is the mistake part A's task 4
|
||||||
|
made and had to undo.
|
||||||
|
- [ ] Verify the gate bites: doctor the reclaimed-bytes metric and confirm the
|
||||||
|
suite fails on exactly that metric.
|
||||||
|
- [ ] Refresh the baseline and confirm the **full** campaign passes against it.
|
||||||
|
The committed baseline is full-mode (`N=20000`, `crash_reps=3`) — writing a
|
||||||
|
quick-mode baseline over it is a regression, and part A made exactly that
|
||||||
|
mistake.
|
||||||
|
- [ ] Commit.
|
||||||
|
|
||||||
|
## Task 6 — closeout
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `docs/stories/databasev2/03-wal-checkpoint.md`,
|
||||||
|
`docs/stories/00-status.md`, `docs/plan/oop-vm/04-db-binding.md`,
|
||||||
|
`database/src/CODE-LOGIC.md`, `docs/examples/db-bench/README.md`.
|
||||||
|
|
||||||
|
- [ ] `04-db-binding.md`: the normative ordering rule — compaction runs only
|
||||||
|
where nothing is staged, and what recovery does (unchanged: one file, replayed
|
||||||
|
from byte 0). This is the doc the spec named for it.
|
||||||
|
- [ ] `CODE-LOGIC.md`: why one file rather than snapshot-plus-tail, why
|
||||||
|
`rename` is the crash-safety primitive, why the dump flushes periodically, and
|
||||||
|
why a failed compaction is not a durability event. Reasoning, not call graph.
|
||||||
|
- [ ] README: the new env knobs beside the existing ones, and the checkpoint
|
||||||
|
leg.
|
||||||
|
- [ ] Story: progress, criteria split met/outstanding, and the measured
|
||||||
|
before/after.
|
||||||
|
- [ ] Board: standup entry in the six-question shape, and the chain note —
|
||||||
|
chain 6 was the last link, so say what the chain's completion means and what
|
||||||
|
is next.
|
||||||
|
- [ ] **Record the `resident: keys` obligation prominently, in the story and at
|
||||||
|
the compactor.** Compaction moves every record, so it invalidates every WAL
|
||||||
|
offset iteration 2 stores; the compactor must rebuild that map as it writes.
|
||||||
|
There is nothing to implement today because iteration 2's storage half does
|
||||||
|
not exist — which is exactly why this must be written where the next
|
||||||
|
implementer will hit it, not left in a spec they may not read.
|
||||||
|
- [ ] Full battery: `just wovm-test`, `just woc-test`, `just oop-e2e`,
|
||||||
|
`just db-bench`, `python3 scripts/linkcheck.py .`
|
||||||
|
- [ ] Commit.
|
||||||
|
|
||||||
|
## Self-review notes
|
||||||
|
|
||||||
|
- **Spec coverage.** Compaction and the switch → Task 1. Stale temp → Task 2.
|
||||||
|
Trigger, no timer, ordering rule → Task 3. Crash safety → Task 4. Space, boot,
|
||||||
|
pause → Task 5. Normative doc, `resident: keys` obligation → Task 6.
|
||||||
|
- **The riskiest task is 4**, not 1: Task 1's correctness is a single replay
|
||||||
|
comparison, while Task 4 is a race and can pass by luck. Hence the explicit
|
||||||
|
instruction to run it repeatedly and to state the count.
|
||||||
|
- **Task 2 looks trivial and is not.** A stale temp file containing well-formed
|
||||||
|
records is the one input that could be mistaken for data, so the test uses
|
||||||
|
plausible records rather than garbage.
|
||||||
|
- **One thing deliberately NOT a task:** rebuilding the `resident: keys` offset
|
||||||
|
map. It cannot be implemented against a feature that does not exist yet.
|
||||||
|
Recorded as an obligation in Task 6 instead of a stub nobody can test.
|
||||||
249
docs/superpowers/plans/2026-08-28-wal-group-commit.md
Normal file
249
docs/superpowers/plans/2026-08-28-wal-group-commit.md
Normal file
|
|
@ -0,0 +1,249 @@
|
||||||
|
# databasev2 4 part A — WAL group commit (implementation plan)
|
||||||
|
|
||||||
|
> **For agentic workers:** REQUIRED SUB-SKILL: Use
|
||||||
|
> superpowers:subagent-driven-development (recommended) or
|
||||||
|
> superpowers:executing-plans to implement this plan task-by-task. Steps
|
||||||
|
> use checkbox (`- [ ]`) syntax for tracking.
|
||||||
|
>
|
||||||
|
> **Style rule (user convention):** concept, reason, and required
|
||||||
|
> behaviour in words plus verification commands only — no implementation
|
||||||
|
> or test code blocks; the executor writes the code.
|
||||||
|
|
||||||
|
**Goal:** one durability barrier per drain instead of one per statement, so a
|
||||||
|
writer is acknowledged after the barrier that carried its record rather than
|
||||||
|
after a barrier of its own.
|
||||||
|
|
||||||
|
**Architecture:** the barrier moves up, not out. Applying to RAM and staging the
|
||||||
|
record stay exactly where they are in `db.c`; the request path stops committing
|
||||||
|
after each append and instead holds its reply envelope, and shard 0 issues one
|
||||||
|
commit when it runs out of queued requests, then releases every held reply. Any
|
||||||
|
failure between "RAM mutated" and "record durable" ends the process with a
|
||||||
|
diagnostic.
|
||||||
|
|
||||||
|
**Tech Stack:** C11, libc only. `pwrite` + `fdatasync` (unchanged — io_uring is
|
||||||
|
part B). The existing per-shard envelope inbox carries the requests.
|
||||||
|
|
||||||
|
**Spec:** [`../specs/2026-08-28-wal-group-commit-design.md`](../specs/2026-08-28-wal-group-commit-design.md)
|
||||||
|
|
||||||
|
## Global Constraints
|
||||||
|
|
||||||
|
- **Durability is unchanged.** Every guarantee iterations 9 and 22 proved holds
|
||||||
|
identically: replay-whole-or-not-at-all, torn-tail drop, no acknowledged
|
||||||
|
write ever lost. This changes when the barrier runs, never what the log holds.
|
||||||
|
- **A writer is released only after the barrier carrying its record.** Never
|
||||||
|
before, and never on the strength of a different batch's barrier.
|
||||||
|
- **libc only.** No new dependency, no new syscall interface in part A.
|
||||||
|
- **The payoff metric is `durable.sN.mixwrite`** (today 480 ops/s, p99
|
||||||
|
5888 µs). `durable.s1.*` and both `seed` legs are regression guards, not
|
||||||
|
targets — a serial writer and an all-inline shard have nothing to batch with.
|
||||||
|
- **`WO_T_IO` leaves the write path.** A commit or staging failure is fatal, not
|
||||||
|
catchable. Exit 1 is a trap and exit 2 is a refusal, so this takes a third
|
||||||
|
status of its own.
|
||||||
|
- Gates run through `just`. Never commit on `master`; branch first.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Task 1 — a failed barrier is detected, and fatal
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `database/src/wal.c` (the commit routine's failure returns; a new
|
||||||
|
fatal-commit entry point beside it), `database/src/wal.h` (declare it).
|
||||||
|
- Test: `runtime/test/test_wal.c` (a new case in the existing suite).
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Produces: a commit entry point that takes the WAL and the number of records
|
||||||
|
in the batch, commits, and on failure writes one stderr line naming the
|
||||||
|
failing operation, the `errno` text, the WAL path and the record count, then
|
||||||
|
exits with the durability-failure status. Tasks 2 and 3 call only this.
|
||||||
|
- Consumes: the existing staging buffer and commit routine.
|
||||||
|
|
||||||
|
- [ ] Read the commit routine first and confirm what it already reports: it
|
||||||
|
loops `pwrite` until the staged buffer is written, then `fdatasync`, and
|
||||||
|
returns non-zero on either failing. Confirm the WAL struct carries its path,
|
||||||
|
or add it — the diagnostic is worthless without it.
|
||||||
|
- [ ] Test first, RED: assert the commit routine reports failure when the
|
||||||
|
descriptor is unusable (a closed descriptor gives `EBADF`). This proves the
|
||||||
|
error is *detected*; it does not exercise the exit.
|
||||||
|
- [ ] Verify RED for the right reason — the case must fail because the
|
||||||
|
assertion is unmet, not because the suite does not compile.
|
||||||
|
- [ ] Add the fatal entry point. It must distinguish the two operations in its
|
||||||
|
message: a `pwrite` failure and an `fdatasync` failure are different
|
||||||
|
operational problems and the operator needs to know which.
|
||||||
|
- [ ] GREEN: `just wovm-test`. The new case passes and no existing case moves.
|
||||||
|
- [ ] **Disclosed gap, record it in the commit message:** the exit path itself
|
||||||
|
is not exercised. Forcing a real `fdatasync` failure needs a full or
|
||||||
|
read-only filesystem, which the gate cannot arrange without mount
|
||||||
|
privileges. Do NOT add a fault-injection switch to buy coverage — shipping a
|
||||||
|
binary that can be told to kill itself is the worse trade, and the spec
|
||||||
|
rejected it.
|
||||||
|
- [ ] Commit.
|
||||||
|
|
||||||
|
## Task 2 — the barrier moves to the drain point; replies are held
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `database/src/db.c` (the request-path arms only — the three commit
|
||||||
|
calls inside the marshaled-statement executor), `runtime/src/vm.c` (the
|
||||||
|
envelope drain loop's DB-statement branch and the end of that loop).
|
||||||
|
- Test: no new fixture; the existing durability battery is the test. It already
|
||||||
|
covers exactly what could break.
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Consumes: Task 1's fatal commit entry point.
|
||||||
|
- Produces: the invariant later tasks measure — at most one barrier per drain,
|
||||||
|
and every held reply released only after it.
|
||||||
|
|
||||||
|
- [ ] Read the drain loop's DB-statement branch first. Today it executes the
|
||||||
|
request, marks it done, then immediately pushes a reply envelope that unparks
|
||||||
|
the requester. Note that it runs on shard 0's thread, serialized — that is
|
||||||
|
why no locking is needed anywhere in this task.
|
||||||
|
- [ ] Remove the three commit calls from the request-path executor in `db.c`.
|
||||||
|
Leave applying to RAM and staging untouched, and leave the **inline** path's
|
||||||
|
three commit calls alone — Task 3 owns that path and conflating them is how
|
||||||
|
this change breaks the single-shard configuration.
|
||||||
|
- [ ] In the drain loop, collect reply envelopes in a local list instead of
|
||||||
|
pushing them as each request finishes. A local is correct and deliberate:
|
||||||
|
nothing needs to survive the loop, and per-shard state would outlive the
|
||||||
|
batch it describes.
|
||||||
|
- [ ] At the end of the drain loop, if anything was staged, call Task 1's fatal
|
||||||
|
commit once, then push every held reply.
|
||||||
|
- [ ] Handle the empty case: a drain that executed no DB statements must not
|
||||||
|
commit and must not touch the staging buffer.
|
||||||
|
- [ ] Verify the ack contract has not moved: `just wovm-test` — the WAL and
|
||||||
|
table suites must be unchanged, since neither knows about batching.
|
||||||
|
- [ ] Verify durability end to end: `just db-bench --quick`. The restart-replay
|
||||||
|
and `kill -9` crash legs are the ones that matter — a kill between staging and
|
||||||
|
the barrier must lose only unacknowledged writes. **If a crash leg fails here,
|
||||||
|
stop; do not adjust the test.** That leg failing means the ack contract broke,
|
||||||
|
which is the one thing this task may not do.
|
||||||
|
- [ ] Commit.
|
||||||
|
|
||||||
|
## Task 3 — the inline path keeps its own barrier, and says why
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `database/src/db.c` (the inline path's three commit calls — replace
|
||||||
|
with Task 1's fatal entry point), plus the comment above them.
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Consumes: Task 1's fatal commit entry point.
|
||||||
|
- Produces: nothing new. This task exists to make the asymmetry deliberate and
|
||||||
|
legible rather than accidental.
|
||||||
|
|
||||||
|
- [ ] Replace the inline path's three commit calls with Task 1's fatal entry
|
||||||
|
point, batch size one. Behaviour is unchanged — this is the fatal-failure
|
||||||
|
rule reaching the second path, not batching.
|
||||||
|
- [ ] Write the comment that explains the asymmetry, because the next reader
|
||||||
|
will otherwise "fix" it: the inline path cannot hold a reply, because it
|
||||||
|
returns into its own fiber rather than unparking a requester. Batching it
|
||||||
|
would require parking that fiber on the barrier, which is part B's machinery
|
||||||
|
and deliberately out of part A.
|
||||||
|
- [ ] Confirm the ordering assumption holds: because the drain loop always
|
||||||
|
commits before it ends, nothing uncommitted is ever left staged when an
|
||||||
|
inline statement runs. If that stops being true the inline path would commit
|
||||||
|
another statement's record early — say so in the comment as the reason the
|
||||||
|
drain must commit unconditionally.
|
||||||
|
- [ ] Verify: `just wovm-test` and `just db-bench --quick` both green, and
|
||||||
|
`WO_SHARDS=1` in particular — the single-shard configuration takes this path
|
||||||
|
exclusively.
|
||||||
|
- [ ] Commit.
|
||||||
|
|
||||||
|
## Task 4 — prove batches actually form
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `scripts/db-bench.py` (new metrics and their tolerances),
|
||||||
|
`docs/examples/db-bench/main.wo` only if the batch figures cannot be observed
|
||||||
|
without the sample reporting them.
|
||||||
|
- Test: the driver's own gate-bites check.
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Consumes: the batching from Task 2.
|
||||||
|
- Produces: mean batch size, peak batch size and peak staged bytes as recorded
|
||||||
|
metrics, so Task 5 measures a mechanism that is known to engage.
|
||||||
|
|
||||||
|
- [ ] Decide where the counters live and prefer the smallest surface: the
|
||||||
|
runtime can report them at exit, or the driver can derive them. Do not add a
|
||||||
|
builtin for this — the numbers are diagnostic, not part of the language.
|
||||||
|
- [ ] Record mean and peak batch size under the concurrent multi-shard write
|
||||||
|
workload. **This is the task's real point:** if batches are always one, the
|
||||||
|
feature is inert and any throughput change came from somewhere else, so the
|
||||||
|
measurement in Task 5 would be attributing a win to the wrong cause.
|
||||||
|
- [ ] Record peak staged bytes. This settles whether the batch needs a cap with
|
||||||
|
a number instead of a guess — the spec deliberately shipped no cap because the
|
||||||
|
request queue is already bounded upstream by iteration 24's mailbox caps.
|
||||||
|
- [ ] Give the new metrics wide tolerances. Batch size is a function of arrival
|
||||||
|
timing, so gating it tightly would gate the scheduler; what must be gated is
|
||||||
|
that it is greater than one under contention.
|
||||||
|
- [ ] Verify the gate bites: doctor the recorded mean batch size to one and
|
||||||
|
confirm the suite fails on exactly that metric.
|
||||||
|
- [ ] Commit.
|
||||||
|
|
||||||
|
## Task 5 — measure the payoff, gate it, write it down
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `bench/baseline.json` (refresh, with the reason in the commit
|
||||||
|
message), `docs/plan/perf-targets.md` (a new section).
|
||||||
|
|
||||||
|
**Interfaces:**
|
||||||
|
- Consumes: Tasks 2 and 4.
|
||||||
|
- Produces: the before/after record every later optimization argues against.
|
||||||
|
|
||||||
|
- [ ] Capture the before numbers from the committed baseline rather than
|
||||||
|
re-measuring them: `durable.sN.mixwrite` 480 ops/s, p50 538 µs, p99 5888 µs;
|
||||||
|
`durable.s1.mixwrite` 1023 ops/s, p99 664 µs; `seed` ~4460 ops/s on both.
|
||||||
|
- [ ] Run the full campaign, not the quick one, and record after numbers for
|
||||||
|
the same metrics on the same machine. A payoff measured across machines is
|
||||||
|
not a payoff.
|
||||||
|
- [ ] Assert the scoped criterion: **`durable.sN.mixwrite` throughput up and
|
||||||
|
p99 down**, with `durable.s1.*` and both `seed` legs not regressed. Do not
|
||||||
|
report the s1 seed number as a disappointment — a serial writer has nothing
|
||||||
|
to batch with, and the spec says so.
|
||||||
|
- [ ] Write the `perf-targets.md` section: the before/after table, the mean and
|
||||||
|
peak batch size that produced it, and the peak staged bytes. State the
|
||||||
|
inversion that motivated the work — multi-shard concurrent writes were 2×
|
||||||
|
slower than single-shard with a 9× worse p99 — and whether it is now gone.
|
||||||
|
- [ ] If the payoff is absent or small, **say so and stop.** That is a finding,
|
||||||
|
not a failure: it would mean the barrier was not the bottleneck the baseline
|
||||||
|
implied, and part B must not be started on an unproven premise.
|
||||||
|
- [ ] Refresh the baseline and confirm `just db-bench` passes against it, then
|
||||||
|
re-confirm the gate bites on a doctored write metric.
|
||||||
|
- [ ] Commit.
|
||||||
|
|
||||||
|
## Task 6 — closeout
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
- Modify: `docs/stories/databasev2/04-io-uring-commit.md` (progress, criteria
|
||||||
|
split met/outstanding, the landing banner),
|
||||||
|
`docs/stories/00-status.md` (standup entry, chain note),
|
||||||
|
`docs/plan/oop-vm/01-error-catalog.md` (the `WO_T_IO` removal and the new
|
||||||
|
exit status), `database/src/CODE-LOGIC.md` (a group-commit section).
|
||||||
|
|
||||||
|
- [ ] Story: record what landed and what did not. The outstanding items are
|
||||||
|
single-shard concurrent batching (needs the inline park) and part B itself.
|
||||||
|
Keep the corrected premise visible — this iteration was written as
|
||||||
|
"fsync-per-commit" and the engine was fsync-per-statement.
|
||||||
|
- [ ] Error catalogue: `WO_T_IO` no longer reachable from a write, and the new
|
||||||
|
durability-failure exit status documented beside the trap and refusal codes.
|
||||||
|
A language-visible removal that is not written down is a trap for the next
|
||||||
|
reader.
|
||||||
|
- [ ] `CODE-LOGIC.md`: the commit path as built — where the barrier runs, why
|
||||||
|
replies are held, why the inline path is asymmetric, and the one rule for
|
||||||
|
failure. Explain the reasoning, not the call graph.
|
||||||
|
- [ ] Board: the standup entry in the six-question shape, and the chain note —
|
||||||
|
part B's go/no-go now rests on Task 5's number.
|
||||||
|
- [ ] Full battery after the doc edits: `just wovm-test`, `just woc-test`,
|
||||||
|
`just oop-e2e`, `just db-bench`, `python3 scripts/linkcheck.py .`
|
||||||
|
- [ ] Commit.
|
||||||
|
|
||||||
|
## Self-review notes
|
||||||
|
|
||||||
|
- **Spec coverage.** Queue-drain boundary → Task 2. Fatal failure rule → Tasks 1
|
||||||
|
and 3. Held replies and the ack contract → Task 2. No batch cap, settled by
|
||||||
|
measurement → Task 4. Payoff and its scoping → Task 5. `WO_T_IO` removal →
|
||||||
|
Task 6. The disclosed abort-coverage gap → Task 1's last step.
|
||||||
|
- **The riskiest task is 2**, and its risk is concentrated in one place: the
|
||||||
|
crash legs of the durability battery. That is why the plan says stop rather
|
||||||
|
than adjust if they fail.
|
||||||
|
- **Task 3 looks like a no-op and is not.** Without it the inline path keeps a
|
||||||
|
catchable `WO_T_IO` while the request path aborts, which is precisely the
|
||||||
|
per-path unevenness this spec exists to remove.
|
||||||
|
- **Task 4 before Task 5 is deliberate.** Measuring a payoff before proving the
|
||||||
|
mechanism engages is how a win gets attributed to the wrong cause.
|
||||||
196
docs/superpowers/specs/2026-08-28-wal-checkpoint-design.md
Normal file
196
docs/superpowers/specs/2026-08-28-wal-checkpoint-design.md
Normal file
|
|
@ -0,0 +1,196 @@
|
||||||
|
# WAL checkpoint — design
|
||||||
|
|
||||||
|
> databasev2 [3](../../stories/databasev2/03-wal-checkpoint.md), chain 6.
|
||||||
|
> Brainstormed and approved 2026-08-28, after
|
||||||
|
> [databasev2 4 part A](2026-08-28-wal-group-commit-design.md) landed.
|
||||||
|
>
|
||||||
|
> **One sentence:** compact the log by rewriting it as one record per live row
|
||||||
|
> into a temporary file, then `rename` it over the live WAL — so recovery is
|
||||||
|
> unchanged and crash safety comes from the filesystem.
|
||||||
|
|
||||||
|
## Decisions taken (the brainstorm's forks, settled)
|
||||||
|
|
||||||
|
| Fork | Decision |
|
||||||
|
| --- | --- |
|
||||||
|
| Snapshot format | **None.** The compacted log *is* the snapshot, in the existing record grammar |
|
||||||
|
| One source or two | **One.** Rewrite + atomic `rename`; boot logic is untouched |
|
||||||
|
| Trigger | **Volume only**, as a ratio against the last compaction's own size, with an absolute floor. **No timer** — see below |
|
||||||
|
| Write availability | **Stop-the-world**, measured against a stated budget rather than assumed acceptable |
|
||||||
|
| Composition with group commit | Compaction runs only where **nothing is staged** — immediately after a barrier |
|
||||||
|
| `resident: keys` (iteration 2) | Compaction **rebuilds the offset map** as it writes. It cannot be left to discover this later |
|
||||||
|
|
||||||
|
## Why one file, and why Postgres cannot do it
|
||||||
|
|
||||||
|
Postgres was read for this (`.dev/reference/postgresql`), and the conclusion is
|
||||||
|
that its design is *unavailable* to us — which is what makes the simpler option
|
||||||
|
legitimate rather than lazy.
|
||||||
|
|
||||||
|
| | PostgreSQL | writeonce |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| Where data lives | heap/data files; the WAL is a redo tail | **the WAL is the only durable form**, replayed into RAM |
|
||||||
|
| WAL contents | page deltas and full-page images | **full row images** — `apply_record` implements UPDATE as remove-then-recreate |
|
||||||
|
| Compaction | **never**; segments before the redo point are recycled by `rename` or unlinked | possible, because a log of row images *is* a complete store |
|
||||||
|
| Bounded replay | recovery starts at the redo LSN in the control file | recovery starts at byte 0 of a *shorter* log |
|
||||||
|
| Crash safety of the switch | control file written in place, full block, torn writes caught by **CRC32C** (`update_controlfile`) | one `rename` |
|
||||||
|
| Trigger | `CheckPointTimeout` (300 s) **or** WAL volume (`XLogCheckpointNeeded`) | volume only |
|
||||||
|
| Pause | none; flush is spread over time in a **separate process** | stop-the-world |
|
||||||
|
|
||||||
|
Postgres cannot compact its WAL because a compacted redo log is not a store —
|
||||||
|
its records describe changes to pages that live elsewhere. Ours describe whole
|
||||||
|
rows, so the compacted log needs no companion. That single difference removes
|
||||||
|
the control file, the redo pointer, the second recovery source, and the separate
|
||||||
|
process from our design.
|
||||||
|
|
||||||
|
**What is worth porting is not the architecture but the ordering discipline:**
|
||||||
|
publish the new "recovery starts here" atomically and *last*, so a crash at any
|
||||||
|
instant falls back to the previous state with nothing to undo. Postgres achieves
|
||||||
|
that with a redo pointer computed at checkpoint *start* and a control file
|
||||||
|
updated at the *end*. We achieve the same property with `rename`, in one
|
||||||
|
syscall, because we can swap the entire data set atomically and Postgres cannot.
|
||||||
|
|
||||||
|
**Correction to a prior exploration doc.**
|
||||||
|
`docs/plan/exploration/postgresql/buffer-and-checkpoint.md` states that Postgres
|
||||||
|
updates its control file by rename ("the same in `BasicOpenFile` +
|
||||||
|
`fsync_parent_path`"). It does not — `update_controlfile` opens the existing file
|
||||||
|
`O_WRONLY`, writes a zero-padded full block in place, and relies on CRC32C to
|
||||||
|
detect a torn write. That doc also assumes writeonce has **segment files**
|
||||||
|
("records before that LSN are *known* to be in the segment files"), which it
|
||||||
|
does not and, per databasev2 2, deliberately will not. The doc predates the
|
||||||
|
databasev2 direction and should be annotated rather than followed.
|
||||||
|
|
||||||
|
## The design
|
||||||
|
|
||||||
|
### Compaction
|
||||||
|
|
||||||
|
Run on the owner shard, which owns the WAL. Walk each class's live rows — the
|
||||||
|
bitmap-over-slabs walk that three call sites in `db.c` already perform — and
|
||||||
|
append one INSERT record per live row to a **new** file, using the existing
|
||||||
|
append path. No new encoder, no new decoder, no format.
|
||||||
|
|
||||||
|
Then: fsync the new file, `rename` it over the live path, fsync the parent
|
||||||
|
directory (the rename's atomicity is in-kernel; the directory entry is not
|
||||||
|
durable until the parent is synced — Postgres does the same, and the existing
|
||||||
|
exploration doc is right about *this* part), and reopen the WAL descriptor,
|
||||||
|
because the old one now refers to an unlinked inode.
|
||||||
|
|
||||||
|
**Every crash point is safe without any recovery logic of ours.** Before the
|
||||||
|
rename, the live WAL is untouched and the temp file is garbage. After it, the new
|
||||||
|
log is complete by construction. There is no window in which a reader could see a
|
||||||
|
mixture, so the acceptance criterion — "recovery produces the same consistent
|
||||||
|
store as if the checkpoint had never started" — is satisfied by `rename`, not by
|
||||||
|
code we must get right.
|
||||||
|
|
||||||
|
Two obligations follow. Boot must **unlink a stale temp file** if one is present,
|
||||||
|
because a crash mid-rewrite leaves one behind and it must never be mistaken for
|
||||||
|
data. And the temp file must be zero-padded beyond its records exactly as the
|
||||||
|
live WAL is, because the tail scan identifies the end of the log by a zero
|
||||||
|
length field.
|
||||||
|
|
||||||
|
### When it runs, and where in the sequence
|
||||||
|
|
||||||
|
**The point matters more than the policy.** The drain stages records into one
|
||||||
|
buffer and commits them together; compaction rewrites the file those records
|
||||||
|
would land in. So compaction may run **only when nothing is staged** — in
|
||||||
|
practice, immediately after a barrier, before the next statement is served.
|
||||||
|
Anywhere else and a staged record would either be written to a file about to be
|
||||||
|
replaced, or be lost with it. This is the normative ordering rule that
|
||||||
|
[`04-db-binding.md`](../../plan/oop-vm/04-db-binding.md) must carry.
|
||||||
|
|
||||||
|
**Trigger: volume, as a self-tuning ratio.** Compact when the WAL's used bytes
|
||||||
|
exceed a multiple of the bytes the *last* compaction wrote, with an absolute
|
||||||
|
floor so a small store never bothers. The denominator is known exactly — the
|
||||||
|
compactor wrote it — so this needs no estimate of the live set's size, which is
|
||||||
|
not cheaply knowable when rows hold Text. The floor exists because a store whose
|
||||||
|
whole log is a few hundred kilobytes has nothing to reclaim.
|
||||||
|
|
||||||
|
**No timer, and that is a deliberate difference from Postgres.** Postgres needs
|
||||||
|
`CheckPointTimeout` because its dirty buffers are not durable until flushed — an
|
||||||
|
idle-but-dirty system must still checkpoint or it loses data. Our records are
|
||||||
|
already durable at commit; a checkpoint reclaims space and shortens boot and
|
||||||
|
nothing else. An idle system's log does not grow, so a timer would fire with
|
||||||
|
nothing to do. Adding one would be copying Postgres' mechanism without its
|
||||||
|
reason.
|
||||||
|
|
||||||
|
A manual trigger exists for tests, because a policy that can only be observed by
|
||||||
|
waiting is a policy that cannot be tested.
|
||||||
|
|
||||||
|
### The pause, and how it is judged
|
||||||
|
|
||||||
|
Compaction is stop-the-world: the owner shard rewrites the log as one long
|
||||||
|
operation while no statement is served. This is the simplest correct thing, and
|
||||||
|
part A's own experience argues for measuring before buying complexity to avoid
|
||||||
|
it. The dump is O(live rows) encodings plus one write and one barrier, so the
|
||||||
|
expectation is that it is fast — but an expectation is not a measurement, and
|
||||||
|
the proof plan below states the budget it must meet.
|
||||||
|
|
||||||
|
If the measured pause exceeds the budget, **that is a finding and a follow-up,
|
||||||
|
not something this iteration solves by adding concurrency.** The alternative
|
||||||
|
designs (incremental copy, fork-and-dump) cost exactly what Postgres pays, and
|
||||||
|
should only be bought against a number.
|
||||||
|
|
||||||
|
### The interaction that will otherwise be discovered late
|
||||||
|
|
||||||
|
**Compaction invalidates every stored WAL offset.** Rewriting the log moves every
|
||||||
|
record, so any offset captured from the old file is meaningless afterwards — not
|
||||||
|
stale-but-readable, but pointing at an arbitrary byte of a different file.
|
||||||
|
[Iteration 2](../../stories/databasev2/02-table-storage-modes.md)'s
|
||||||
|
`resident: keys` stores exactly such offsets, one per row, and reads rows back
|
||||||
|
through them.
|
||||||
|
|
||||||
|
The compactor therefore **rebuilds the offset map as it writes**: it is emitting
|
||||||
|
the new records and knows each one's new position, so this is the cheap
|
||||||
|
direction and the only one that keeps both features usable together. The
|
||||||
|
alternative — forbidding compaction while any `resident: keys` table is live —
|
||||||
|
would mean the feature that exists to handle huge tables is incompatible with
|
||||||
|
the feature that stops their log growing forever.
|
||||||
|
|
||||||
|
This is recorded here because iteration 2's storage half is not yet
|
||||||
|
implemented, so nothing will fail today. It will fail later, in a way that looks
|
||||||
|
like data corruption rather than a design gap.
|
||||||
|
|
||||||
|
## Proof plan
|
||||||
|
|
||||||
|
| Claim | How it is proven |
|
||||||
|
| --- | --- |
|
||||||
|
| Space is reclaimed | An aged store shrinks. Measured today on master: `seed 20000` gives a 986 614-byte log; 20 000 updates take it to 2 590 262 bytes with **the same live rows**. Compaction must return it to approximately the former |
|
||||||
|
| Replay is bounded | Boot time on the aged store before and after, recorded. Measured today: boot+verify on that store is 155 ms |
|
||||||
|
| Crash safety | `kill -9` at many instants *during* compaction, then replay: the store must equal either the pre-compaction or post-compaction state, never a mixture, and no acked write may be missing. This is the criterion the whole design is shaped around, so it gets the crash battery's treatment rather than one case |
|
||||||
|
| A stale temp file is harmless | Boot with one present, containing plausible records: it is removed and never read |
|
||||||
|
| The pause is known | The stop-the-world pause measured on the largest store the harness builds, recorded as a number with a stated budget — not asserted to be acceptable |
|
||||||
|
| The ordering rule holds | Compaction with records staged must be impossible by construction; a test that stages and then requests compaction must find it deferred, not executed |
|
||||||
|
| Nothing regressed | The full battery, and specifically part A's `wmix` legs: compaction must not change the ack contract or the batching it introduced |
|
||||||
|
|
||||||
|
## Out of scope
|
||||||
|
|
||||||
|
- **A second file, a control file, or a redo pointer.** The Postgres shape,
|
||||||
|
priced above and not needed once the log is self-sufficient.
|
||||||
|
- **Avoiding the pause.** Incremental or forked dumps are bought against a
|
||||||
|
measurement, not in advance.
|
||||||
|
- **Per-shard compaction policy.** One owner shard owns the WAL today; when that
|
||||||
|
changes, this decision is revisited with it.
|
||||||
|
- **Compacting away tombstones across shards, or any cross-shard coordination.**
|
||||||
|
There is one log.
|
||||||
|
- **io_uring for the rewrite** — part B of databasev2 4, whose premise is
|
||||||
|
already under revision.
|
||||||
|
- **Changing the record grammar.** The entire argument for this design is that
|
||||||
|
the grammar already suffices.
|
||||||
|
|
||||||
|
## Alternatives rejected
|
||||||
|
|
||||||
|
**Snapshot + WAL tail (the Postgres shape).** Rejected because it buys write
|
||||||
|
availability at the cost of a second recovery source, a cutoff offset, a control
|
||||||
|
file with its own torn-write detection, and a crash-safety guarantee that
|
||||||
|
depends on our ordering rather than on `rename`. Postgres pays this because its
|
||||||
|
log cannot stand alone; ours can.
|
||||||
|
|
||||||
|
**Compacting in place.** Rejected outright: there is no crash point at which a
|
||||||
|
partially rewritten live log is recoverable, and it trades the one property that
|
||||||
|
makes this design defensible for nothing.
|
||||||
|
|
||||||
|
**A timer trigger.** Rejected with a reason rather than on taste: Postgres' timer
|
||||||
|
exists to bound data loss from unflushed buffers, and we have no unflushed
|
||||||
|
buffers. An idle log does not grow.
|
||||||
|
|
||||||
|
**A ratio against an estimated live-set size.** Rejected in favour of the last
|
||||||
|
compaction's measured output, because estimating the live size means estimating
|
||||||
|
Text, and the compactor already knows the true number.
|
||||||
208
docs/superpowers/specs/2026-08-28-wal-group-commit-design.md
Normal file
208
docs/superpowers/specs/2026-08-28-wal-group-commit-design.md
Normal file
|
|
@ -0,0 +1,208 @@
|
||||||
|
# WAL group commit — design
|
||||||
|
|
||||||
|
> databasev2 [4](../../stories/databasev2/04-io-uring-commit.md), part A.
|
||||||
|
> Brainstormed and approved 2026-08-28.
|
||||||
|
>
|
||||||
|
> **This spec covers batching only.** The iteration was split during the
|
||||||
|
> brainstorm: part A amortises one durability barrier across many statements,
|
||||||
|
> part B (io_uring submission) is deferred until A's measurement says whether
|
||||||
|
> the blocking boundary is still the bottleneck. That split matches the
|
||||||
|
> iteration's own fork 1 — "drop-in behind `wo_wal_commit` first, an async
|
||||||
|
> variant only if the scheduler proves the blocking boundary is the
|
||||||
|
> bottleneck" — and it means the throughput win arrives behind a much smaller
|
||||||
|
> correctness surface.
|
||||||
|
|
||||||
|
## Decisions taken (the brainstorm's forks, settled)
|
||||||
|
|
||||||
|
| Fork | Decision |
|
||||||
|
| --- | --- |
|
||||||
|
| Scope | **Batching first, io_uring later.** Two independent wins were being carried as one; only the first needs a new syscall interface, and it is where most of the number lives |
|
||||||
|
| Batch boundary | **Queue-drain.** Shard 0 stages every pending write request, then commits once. No timer, no tunable |
|
||||||
|
| Failure | **Fatal, diagnosed abort.** Any failure between "RAM mutated" and "record durable" ends the process |
|
||||||
|
| Batch cap | **None initially.** Measure peak staged bytes; add a cap only if the queue's existing upstream bound proves insufficient |
|
||||||
|
| Abort coverage | Unit-test the failure *return*; the abort path itself stays covered by inspection, and that gap is disclosed |
|
||||||
|
|
||||||
|
## The problem, read off the engine
|
||||||
|
|
||||||
|
The story says "replace fsync-per-commit with io_uring group-commit". Read
|
||||||
|
against the code, the premise needed correcting: the engine does not commit per
|
||||||
|
*commit*, it commits per **statement**. `db.c` calls `wo_wal_commit`
|
||||||
|
immediately after every append, at all six sites — insert, update and remove,
|
||||||
|
each on both the inline and the DB-actor path. Every single row change is one
|
||||||
|
`pwrite` plus one `fdatasync`.
|
||||||
|
|
||||||
|
That is what the numbers say too. Iteration 22's baseline records durable writes
|
||||||
|
at **4460 ops/s** single-shard and mixed writes at **1023 ops/s**, p50 **430 µs**,
|
||||||
|
p99 **664 µs** — against **1.28M ops/s** for durable reads. Writes are roughly
|
||||||
|
290× slower than reads, and the barrier is the whole of it.
|
||||||
|
|
||||||
|
**The batching machinery already exists and is simply never used.**
|
||||||
|
`wo_wal_commit` writes `w->buf` for `w->len` bytes — a staged buffer that can
|
||||||
|
hold any number of records. Today it never holds more than one, because the
|
||||||
|
caller commits immediately after staging. So part A is closer to removing calls
|
||||||
|
than to adding a mechanism.
|
||||||
|
|
||||||
|
## The design
|
||||||
|
|
||||||
|
### The commit path
|
||||||
|
|
||||||
|
The six `wo_wal_commit` calls come out of `db.c`. Applying to RAM and staging
|
||||||
|
the record stay exactly where they are; only the barrier moves, up to the point
|
||||||
|
where shard 0 runs out of work.
|
||||||
|
|
||||||
|
Shard 0 owns the WAL — DB statements from other shards arrive as marshaled
|
||||||
|
request envelopes and are executed on shard 0's thread, serialized, and a reply
|
||||||
|
envelope unparks the requester. The change is that **the reply is held rather
|
||||||
|
than sent**: shard 0 executes and stages each queued request, keeps draining
|
||||||
|
while requests remain, then issues one barrier, and only then releases every
|
||||||
|
held reply.
|
||||||
|
|
||||||
|
Each requester therefore unparks having been acknowledged after the barrier that
|
||||||
|
carried *its* record — the ack contract the story states, which today is true
|
||||||
|
only because every batch has one member.
|
||||||
|
|
||||||
|
A statement executing inline on shard 0 (rather than arriving as a request)
|
||||||
|
stages and commits before returning, as it does now. It has no reply to hold —
|
||||||
|
it returns into its own fiber — and because the drain always commits before it
|
||||||
|
ends, nothing uncommitted is ever left staged when the inline path runs.
|
||||||
|
|
||||||
|
### Why queue-drain, and what it costs
|
||||||
|
|
||||||
|
The batch boundary is the queue going empty, not a tick and not a timer. Two
|
||||||
|
properties follow, and they are the reason to prefer it:
|
||||||
|
|
||||||
|
- **A lone writer pays nothing.** One queued request means a batch of one, which
|
||||||
|
is today's path at today's latency. Batching engages only under genuine
|
||||||
|
contention, so an idle system is not taxed to serve a busy one.
|
||||||
|
- **The batch self-tunes.** Its size is whatever actually accumulated between
|
||||||
|
drains, so it grows with load rather than with a configured number. There is
|
||||||
|
nothing to set and nothing to set wrong.
|
||||||
|
|
||||||
|
The rejected alternative was the iteration's recorded leaning, the shard tick.
|
||||||
|
That leaning was recorded when the batch was assumed to ride an io_uring
|
||||||
|
submission; with batching landing first, a tick boundary would add up to one
|
||||||
|
quantum of latency even to a lone writer — paying the cost of batching when
|
||||||
|
there is nothing to batch with.
|
||||||
|
|
||||||
|
**No batch cap ships initially, and that is a decision rather than an
|
||||||
|
oversight.** databasev2 1 established that unbounded growth is precisely how
|
||||||
|
this engine dies without warning, so the instinct to bound it is right. But the
|
||||||
|
request queue is already bounded upstream by iteration 24's mailbox caps, and a
|
||||||
|
second bound on the same quantity is a knob that can only be wrong. The proof
|
||||||
|
plan measures peak staged bytes so the question is settled by a number.
|
||||||
|
|
||||||
|
## Failure: one rule, replacing three behaviours
|
||||||
|
|
||||||
|
Today's rollback is uneven, and the code says so. An `insert` whose commit fails
|
||||||
|
removes the row again, under a comment claiming RAM never claims what disk has
|
||||||
|
not acknowledged. An `update` or a `delete` whose commit fails does **not** roll
|
||||||
|
back — its comment admits the state plainly: RAM ahead of disk, trap, do not
|
||||||
|
ack. Nothing acknowledged is lost, but the process continues with divergent
|
||||||
|
state, and batching would multiply that from one row to as many as the batch
|
||||||
|
held.
|
||||||
|
|
||||||
|
The rule that replaces it: **once a statement has mutated RAM, the only outcomes
|
||||||
|
are durable or process death.** It covers both failure points identically —
|
||||||
|
a staging failure and a barrier failure have the same consequence, RAM ahead of
|
||||||
|
disk with no way back, and only one of the three verbs can undo itself.
|
||||||
|
|
||||||
|
Retrying is not an alternative worth designing for. On Linux a failed `fsync`
|
||||||
|
may already have discarded the dirty pages, so a second call can report success
|
||||||
|
having written nothing; the recovery that actually works is replay, which
|
||||||
|
returns exactly the last durable state. That is what the log is for.
|
||||||
|
|
||||||
|
**This removes `WO_T_IO` from the write path.** A program can no longer catch a
|
||||||
|
disk failure on a write. The removal is deliberate — there was never a
|
||||||
|
recovery a program could meaningfully perform with its RAM ahead of its disk —
|
||||||
|
but it is language-visible and must be stated in the story banner and the error
|
||||||
|
catalogue, not slipped in.
|
||||||
|
|
||||||
|
The diagnostic has to earn the abort: the failing operation, the `errno` text,
|
||||||
|
the WAL path, and the number of records in the batch, on stderr, then exit with
|
||||||
|
a status of its own. Exit 1 is a trap and exit 2 is a refusal, so a durability
|
||||||
|
failure takes a third. `abort()` is rejected — a core dump on a full disk is
|
||||||
|
noise, not evidence.
|
||||||
|
|
||||||
|
## What will improve, and what will not
|
||||||
|
|
||||||
|
**Corrected 2026-08-28, after reading the baseline properly.** The spec first
|
||||||
|
pointed at `durable.s1.seed` as the payoff metric. That was wrong, and the
|
||||||
|
reason is structural rather than a matter of degree.
|
||||||
|
|
||||||
|
Worker shards hold no WAL at all — the runtime asserts it — so every DB
|
||||||
|
statement on a worker marshals to shard 0 and parks, while a statement already
|
||||||
|
on shard 0 executes inline. **A queue of write requests therefore exists only
|
||||||
|
when other shards are writing.** Batches form where there is a queue:
|
||||||
|
|
||||||
|
| Workload | Today | Batching |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `durable.sN.mixwrite` — concurrent writers across shards | **480 ops/s, p99 5888 µs** | **the target.** N shards marshal N writes and shard 0 pays N barriers serially; one barrier replaces them |
|
||||||
|
| `durable.s1.mixwrite` — concurrent writers, one shard | 1023 ops/s, p99 664 µs | **no change.** Every write is inline with no queue, so no batch forms |
|
||||||
|
| `durable.*.seed` — one serial writer | ~4460 ops/s | **no change**, under any batching scheme. There is nothing to batch with |
|
||||||
|
|
||||||
|
The inversion in those numbers is the finding worth keeping: **multi-shard
|
||||||
|
concurrent writes are currently 2× slower than single-shard with a 9× worse
|
||||||
|
p99.** Adding shards makes durable writing worse today, because every marshaled
|
||||||
|
statement still buys its own barrier on the owner. That is the pathology group
|
||||||
|
commit exists to remove, and it is a better argument for this iteration than the
|
||||||
|
one the story recorded.
|
||||||
|
|
||||||
|
**Single-shard concurrent batching is deliberately out of part A.** It would
|
||||||
|
need the inline path to park its fiber on the barrier rather than commit
|
||||||
|
synchronously — the same parking machinery part B needs anyway. Deferring it
|
||||||
|
keeps A to one mechanism, and B inherits the reason to build it.
|
||||||
|
|
||||||
|
So the acceptance criterion is scoped: **`durable.sN.mixwrite` throughput up and
|
||||||
|
its p99 down; `durable.s1.*` and both `seed` legs must not regress.** A plan
|
||||||
|
that reported "no improvement" against the s1 seed number would be measuring a
|
||||||
|
workload this change cannot help.
|
||||||
|
|
||||||
|
## Proof plan
|
||||||
|
|
||||||
|
| Claim | How it is proven |
|
||||||
|
| --- | --- |
|
||||||
|
| The payoff is real | **`durable.sN.mixwrite`** before and after on one machine, recorded in `perf-targets.md`. Today 480 ops/s, p99 5888 µs. `durable.s1.*` and both `seed` legs are regression guards, not targets — see the section above |
|
||||||
|
| Durability is unchanged | Iteration 22's crash battery, unaltered: concurrent writers, `kill -9` mid-stream, replay. **The critical test** — a kill between staging and the barrier must lose only unacknowledged writes |
|
||||||
|
| Batches actually form | New metrics for mean and peak batch size under contention. If batches are always one, the feature is inert and any throughput change came from somewhere else |
|
||||||
|
| No idle tax | Single-writer p99 must not regress against the current baseline |
|
||||||
|
| The cap question is answered | Peak staged bytes recorded per run |
|
||||||
|
| A failure is detected | `test_wal.c` asserts `wo_wal_commit` reports failure on a bad descriptor |
|
||||||
|
|
||||||
|
**One disclosed gap.** Forcing a genuine `fdatasync` failure needs a full or
|
||||||
|
read-only filesystem, which the gate cannot arrange without mount privileges.
|
||||||
|
The unit test proves the error is *detected*; the abort that follows it stays
|
||||||
|
covered by inspection. The alternative — a fault-injection switch — means
|
||||||
|
shipping a binary that can be told to kill itself, which is a worse trade. This
|
||||||
|
gap is recorded rather than hidden, because iteration 40 was exactly a fatal
|
||||||
|
path that nothing exercised.
|
||||||
|
|
||||||
|
## Out of scope
|
||||||
|
|
||||||
|
- **io_uring submission.** Part B, and it only earns its complexity if A's
|
||||||
|
measurement shows the blocking boundary still dominating. A's parking and ack
|
||||||
|
machinery is what B would build on, so nothing here is wasted either way.
|
||||||
|
- **`transaction { }`** — language iteration 18. A transaction already *is* a
|
||||||
|
staged batch, so the two compose without either knowing about the other; that
|
||||||
|
is a reason not to entangle them now.
|
||||||
|
- **Checkpoint and compaction** — databasev2 3. This changes when the barrier
|
||||||
|
runs, never what the log contains.
|
||||||
|
- **The read path.** databasev2 1 measured that appending under memory pressure
|
||||||
|
costs about 1% while random reads cost 273×, so the pressure is on reads —
|
||||||
|
but that is iteration 2's `resident: keys` question, not this one.
|
||||||
|
- **Rollback with pre-images.** Rejected above: it would add per-write cost on
|
||||||
|
every statement to serve a path that ends the process anyway.
|
||||||
|
|
||||||
|
## Alternatives rejected
|
||||||
|
|
||||||
|
**Tick-boundary batching** — the iteration's recorded leaning, superseded by
|
||||||
|
the split. It taxes an idle system to serve a busy one.
|
||||||
|
|
||||||
|
**Count-or-timer batching** — two tunables, and the timer reintroduces the tick
|
||||||
|
problem with extra configuration.
|
||||||
|
|
||||||
|
**Full rollback with an undo log** — keeps `WO_T_IO` catchable, at the price of
|
||||||
|
capturing pre-images for every update and delete, paid on every write, to
|
||||||
|
support continuing in a state the engine cannot trust.
|
||||||
|
|
||||||
|
**Keeping today's per-verb behaviour** — turns a rare one-row divergence into a
|
||||||
|
routine N-row one, silently.
|
||||||
7
justfile
7
justfile
|
|
@ -60,6 +60,13 @@ site:
|
||||||
fibers:
|
fibers:
|
||||||
./scripts/fibers-accept.sh
|
./scripts/fibers-accept.sh
|
||||||
|
|
||||||
|
# chat: iteration 24's gate (docs/examples/chat) — rooms/presence/broadcast
|
||||||
|
# over WebSocket via actors: functional on both WO_IO backends, the
|
||||||
|
# 1k-clients-one-hot-room soak (fds/RSS accounted), SIGTERM drain with
|
||||||
|
# close frames, and an ASan leg. `just chat` runs it (CHAT_SOAK=N trims).
|
||||||
|
chat:
|
||||||
|
./scripts/chat-accept.sh
|
||||||
|
|
||||||
# db-actor: arc stage 3's gate (docs/examples/db-actor) — worker-shard
|
# db-actor: arc stage 3's gate (docs/examples/db-actor) — worker-shard
|
||||||
# actors read/write the database through the transparent DB actor; WAL
|
# actors read/write the database through the transparent DB actor; WAL
|
||||||
# replay pair included. `just db-actor` runs it.
|
# replay pair included. `just db-actor` runs it.
|
||||||
|
|
|
||||||
|
|
@ -257,3 +257,100 @@ layout.
|
||||||
- **`listen_unix` sets O_NONBLOCK on the listener itself** — accept4's
|
- **`listen_unix` sets O_NONBLOCK on the listener itself** — accept4's
|
||||||
SOCK_NONBLOCK flags the ACCEPTED socket only; a blocking listener
|
SOCK_NONBLOCK flags the ACCEPTED socket only; a blocking listener
|
||||||
would block the whole shard (found by the seam probe, both backends).
|
would block the whole shard (found by the seam probe, both backends).
|
||||||
|
|
||||||
|
## Actor lifecycle: call, death, monitor, timers (iteration 24, ids 88–90)
|
||||||
|
|
||||||
|
Four pieces that together answer "what happens to an actor that is waiting,
|
||||||
|
that dies, that watches, or that wants to be woken later". All four live in
|
||||||
|
`vm.c` with their entry points in `builtin.c`; the structures are in `vm.h`.
|
||||||
|
|
||||||
|
**`call` (id 88) — a send that waits.** An ordinary `send` returns immediately;
|
||||||
|
`call` parks the calling fiber and resumes it with the receive's return value.
|
||||||
|
The reply is a **typed scalar**, which is what let the agreement be checked at
|
||||||
|
compile time (WO-E226) rather than carried as a tagged value at runtime. The
|
||||||
|
caller is never left hanging: if the callee dies mid-call, or the address is
|
||||||
|
already dead, the caller **traps catchably** instead of parking forever. That
|
||||||
|
is the property worth keeping in mind when reading the code — every path out of
|
||||||
|
a call either resumes the fiber or traps it.
|
||||||
|
|
||||||
|
**Death.** A `receive` that traps uncaught marks the actor dead on its home
|
||||||
|
thread. From then on sends to it drop silently, calls trap, queued callers are
|
||||||
|
error-unparked, and its state and mailbox are released. Silent-drop for sends
|
||||||
|
is deliberate: a sender cannot handle another actor's failure, and making every
|
||||||
|
`send` fallible would put a `try` on every line.
|
||||||
|
|
||||||
|
**The mailbox cap and its counter.** One cap for every mailbox (default 1024,
|
||||||
|
`WO_MAILBOX` overrides at boot; the chat gate shrinks it to 8 to force the
|
||||||
|
policy). `pending` counts sent-but-not-delivered. It is incremented by the
|
||||||
|
**sender**, on any shard, and decremented by the **home thread** at delivery —
|
||||||
|
so it is touched only through `wo_mbox_reserve`/`wo_mbox_release` and their
|
||||||
|
`__atomic` builtins. The consequence is disclosed rather than hidden: the cap
|
||||||
|
can overshoot by at most the number of in-flight sends. Overflow is fail-fast —
|
||||||
|
the send raises a catchable `WO_T_ACTOR` (trap 13), which is what lets a room
|
||||||
|
drop a slow member instead of growing without bound.
|
||||||
|
|
||||||
|
**`monitor` (id 89) — the death notice.** `wo_monitor` is one registration:
|
||||||
|
observer, the moved-in notice message, next. The list lives on the **watched**
|
||||||
|
actor and is owned by its home thread, so the death walk needs no lock — dying
|
||||||
|
is a home-thread event and the list is right there. The notice is the
|
||||||
|
observer's own M-typed message, so an observer receives death notices in the
|
||||||
|
same shape as everything else. Monitoring an already-dead actor fires
|
||||||
|
immediately rather than silently doing nothing. An observer whose mailbox is
|
||||||
|
full loses the notice, with a disclosed stderr line — the alternative was
|
||||||
|
blocking a death walk on a slow observer.
|
||||||
|
|
||||||
|
It takes **three arguments** (`watched, observer, msg`), not the two the spec
|
||||||
|
first proposed, because the caller may be `main`, which has no mailbox and so
|
||||||
|
cannot be an implicit observer.
|
||||||
|
|
||||||
|
**`time.after` (id 90) — one-shot, no cancel.** `wo_timer` is `at` (wall ms),
|
||||||
|
target, message, next. The list lives on the **arming fiber's shard** and is
|
||||||
|
scanned by the same deadline machinery that already serves fd-park deadlines,
|
||||||
|
so timers cost no new wait mechanism. Firing is an ordinary runtime send, which
|
||||||
|
means it inherits the ordinary rules: a full target drops with a stderr line, a
|
||||||
|
dead target drops silently. There is no cancel; the idiom is a generation
|
||||||
|
counter in the message, which the `timer-generation` corpus fixture pins.
|
||||||
|
|
||||||
|
**Where to look when a lifecycle thing misbehaves:** `wo_vm_actor_monitor` and
|
||||||
|
`wo_vm_timer_after` in `vm.c` are the two entry points; `shard_main` and
|
||||||
|
`NEXT_RUNNABLE()` decide when a shard runs, adopts, or stops. The corpus
|
||||||
|
fixtures `monitor-death`, `timer-delivery` and `timer-generation` are the
|
||||||
|
smallest working examples of each.
|
||||||
|
|
||||||
|
## The shutdown drain guarantee (iteration 40)
|
||||||
|
|
||||||
|
**A message sent before the stop flag is observed is delivered and run before
|
||||||
|
the engine stops.** Stated because it was once untrue in a way nothing caught.
|
||||||
|
|
||||||
|
`wo_engine_stop` sets `eng_shutdown`, wakes every worker, joins them, and only
|
||||||
|
then tears down — freeing whatever envelopes are still queued. So a worker that
|
||||||
|
leaves its loop early takes its inbox with it. `NEXT_RUNNABLE()` has always
|
||||||
|
encoded the right behaviour for a worker holding a live fiber: on a stop it
|
||||||
|
returns 2 and keeps draining, because "only the PRIMARY's stop ends the
|
||||||
|
program". `shard_main`'s **idle** branch did the opposite — it reaped and broke
|
||||||
|
— so a shard whose actors happened to be between messages at `SIGTERM`
|
||||||
|
abandoned everything still in flight.
|
||||||
|
|
||||||
|
It now honours the same contract: while the primary's window is open an idle
|
||||||
|
worker adopts its inbox and runs what arrives, `sched_yield`ing on an empty
|
||||||
|
poll so a drain cannot burn a core per shard and starve the actors it exists to
|
||||||
|
let run. Only `eng_shutdown` — which the primary sets after `main` returns —
|
||||||
|
ends it.
|
||||||
|
|
||||||
|
Two things follow that are easy to get wrong. The window is the **primary's**,
|
||||||
|
so a program that wants a longer drain holds it open itself; `main` cannot park
|
||||||
|
after the stop flag, because a park there unwinds. And the whole path is
|
||||||
|
unreachable at `WO_SHARDS=1`, where `wo_engine_stop` returns at `nshards <= 1`.
|
||||||
|
|
||||||
|
## Digests: sha1, sha256, hmac_sha256 (iteration 34, ids 85–87)
|
||||||
|
|
||||||
|
`crypto.c` holds SHA-1 and SHA-256 over a single buffer and HMAC-SHA-256 on top
|
||||||
|
of the latter, each returning a fresh `Bytes`. No streaming API and no other
|
||||||
|
primitives — these exist because WebSocket's handshake needs SHA-1 and ETags
|
||||||
|
need SHA-256, and that is the whole of the demand so far.
|
||||||
|
|
||||||
|
Correctness is pinned to the published vectors rather than to itself:
|
||||||
|
RFC 3174 for SHA-1, the FIPS/RFC 6234 vectors for SHA-256, RFC 4231 for HMAC,
|
||||||
|
in `runtime/test/test_crypto.c` (18 checks). **There is still no RNG anywhere
|
||||||
|
in the runtime** — HMAC authenticates a token but cannot mint one, which is why
|
||||||
|
iteration 39 leads with a random-bytes builtin.
|
||||||
|
|
|
||||||
|
|
@ -200,6 +200,18 @@ int wo_builtin(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
||||||
}
|
}
|
||||||
case WO_B_CALL: /* iteration 24: park/reply protocol lives in vm.c */
|
case WO_B_CALL: /* iteration 24: park/reply protocol lives in vm.c */
|
||||||
return wo_vm_actor_call(vm, R, ins, msg);
|
return wo_vm_actor_call(vm, R, ins, msg);
|
||||||
|
case WO_B_MONITOR: {
|
||||||
|
int rc = wo_vm_actor_monitor(vm, R[B], R[B + 1], R[B + 2], msg);
|
||||||
|
if (rc) return rc;
|
||||||
|
R[A] = 0;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
case WO_B_TIME_AFTER: {
|
||||||
|
int rc = wo_vm_timer_after(vm, (int64_t)R[B], R[B + 1], R[B + 2], msg);
|
||||||
|
if (rc) return rc;
|
||||||
|
R[A] = 0;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
case WO_B_NOW: { /* wall-clock milliseconds */
|
case WO_B_NOW: { /* wall-clock milliseconds */
|
||||||
struct timespec ts;
|
struct timespec ts;
|
||||||
clock_gettime(CLOCK_REALTIME, &ts);
|
clock_gettime(CLOCK_REALTIME, &ts);
|
||||||
|
|
|
||||||
|
|
@ -183,6 +183,17 @@ int wo_load_buf(wo_module *m, const uint8_t *buf, size_t len, char *err,
|
||||||
* nowhere to live. woc refuses this at compile time; the loader
|
* nowhere to live. woc refuses this at compile time; the loader
|
||||||
* refuses it again because what the loader accepts, the interpreter
|
* refuses it again because what the loader accepts, the interpreter
|
||||||
* trusts — this combination must never reach the engine. */
|
* trusts — this combination must never reach the engine. */
|
||||||
|
/* databasev2 2: `resident: keys` PARSES and sets this bit, but the
|
||||||
|
* storage half (tasks 5c/5d) is not implemented — rows are still fully
|
||||||
|
* resident. Accepting it would be an annotation the compiler honours
|
||||||
|
* in name only: a developer could declare a 120 GB table keys-resident,
|
||||||
|
* see it compile, and be OOM-killed. Refuse until the storage lands. */
|
||||||
|
if (flags & WO_CLASSF_RESIDENT_KEYS)
|
||||||
|
BAIL("class %u declares `resident: keys`, which is NOT IMPLEMENTED "
|
||||||
|
"yet — rows are still fully resident, so the annotation would "
|
||||||
|
"be honoured in name only. Remove it until databasev2 2 tasks "
|
||||||
|
"5c/5d land; `resident: all` is what actually runs",
|
||||||
|
(unsigned)i);
|
||||||
if ((flags & WO_CLASSF_VOLATILE) && (flags & WO_CLASSF_RESIDENT_KEYS))
|
if ((flags & WO_CLASSF_VOLATILE) && (flags & WO_CLASSF_RESIDENT_KEYS))
|
||||||
BAIL("class %u: durable:false with resident:keys — rows would have "
|
BAIL("class %u: durable:false with resident:keys — rows would have "
|
||||||
"nowhere to be read from", (unsigned)i);
|
"nowhere to be read from", (unsigned)i);
|
||||||
|
|
|
||||||
|
|
@ -4,6 +4,7 @@
|
||||||
* 2 = usage or load failure (loader's message on stderr)
|
* 2 = usage or load failure (loader's message on stderr)
|
||||||
* Heap cap defaults to 64 MiB, overridable via WO_HEAP_MB. */
|
* Heap cap defaults to 64 MiB, overridable via WO_HEAP_MB. */
|
||||||
#include <fcntl.h>
|
#include <fcntl.h>
|
||||||
|
#include <signal.h>
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
|
|
@ -124,6 +125,22 @@ static void gc_pump(wo_vm *vm) {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* databasev2 4: one diagnostic line about group commit, opt-in via
|
||||||
|
* WO_WAL_STATS. Off by default because it would otherwise pollute the output
|
||||||
|
* of every durable program; a gate that wants the numbers asks for them. */
|
||||||
|
static void wal_stats_report(const wo_wal *w) {
|
||||||
|
if (!w || !getenv("WO_WAL_STATS")) return;
|
||||||
|
fprintf(stderr,
|
||||||
|
"walstats batches=%llu records=%llu peak_batch=%llu peak_staged=%llu "
|
||||||
|
"compactions=%llu compact_us_max=%llu compact_us_total=%llu compacted_bytes=%llu\n",
|
||||||
|
(unsigned long long)w->stat_batches, (unsigned long long)w->stat_records,
|
||||||
|
(unsigned long long)w->stat_peak_batch, (unsigned long long)w->stat_peak_staged,
|
||||||
|
(unsigned long long)w->stat_compactions,
|
||||||
|
(unsigned long long)w->stat_compact_us_max,
|
||||||
|
(unsigned long long)w->stat_compact_us_total,
|
||||||
|
(unsigned long long)w->compacted_bytes);
|
||||||
|
}
|
||||||
|
|
||||||
int main(int argc, char **argv) {
|
int main(int argc, char **argv) {
|
||||||
wo_module mod;
|
wo_module mod;
|
||||||
char err[256];
|
char err[256];
|
||||||
|
|
@ -178,6 +195,10 @@ int main(int argc, char **argv) {
|
||||||
return 2;
|
return 2;
|
||||||
}
|
}
|
||||||
wo_tls_set(&VM);
|
wo_tls_set(&VM);
|
||||||
|
/* iteration 24: a write to a peer-closed socket must be EPIPE (a
|
||||||
|
* catchable WO_T_IO), never a process-killing SIGPIPE — every
|
||||||
|
* serving program writes to sockets whose peers vanish. */
|
||||||
|
signal(SIGPIPE, SIG_IGN);
|
||||||
/* The database engine boots with the VM: every class IS a table.
|
/* The database engine boots with the VM: every class IS a table.
|
||||||
* Durability is opt-in — WO_DATA=<dir> opens <dir>/shard-0.wal,
|
* Durability is opt-in — WO_DATA=<dir> opens <dir>/shard-0.wal,
|
||||||
* replays it before the entry runs (boot-before-listeners doctrine),
|
* replays it before the entry runs (boot-before-listeners doctrine),
|
||||||
|
|
@ -254,6 +275,25 @@ int main(int argc, char **argv) {
|
||||||
if (v >= 1 && v <= 0x7FFFFFFFul) wo_mailbox_cap = (uint32_t)v;
|
if (v >= 1 && v <= 0x7FFFFFFFul) wo_mailbox_cap = (uint32_t)v;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
/* databasev2 3: the checkpoint policy. WO_CHECKPOINT_BYTES is the floor
|
||||||
|
* below which a log is too small to bother compacting; WO_CHECKPOINT_RATIO
|
||||||
|
* is how many times the live set's own size counts as too much history.
|
||||||
|
* Both exist mainly so the policy is TESTABLE — a gate sets a tiny floor
|
||||||
|
* and forces compaction in a few writes rather than waiting for megabytes.
|
||||||
|
* There is no time-based trigger, by design: our records are durable at
|
||||||
|
* commit, so an idle log does not grow. */
|
||||||
|
{
|
||||||
|
const char *cb = getenv("WO_CHECKPOINT_BYTES");
|
||||||
|
if (cb && cb[0]) {
|
||||||
|
unsigned long long v = strtoull(cb, NULL, 10);
|
||||||
|
if (v > 0) wo_wal_ckpt_floor = (uint64_t)v;
|
||||||
|
}
|
||||||
|
const char *cr = getenv("WO_CHECKPOINT_RATIO");
|
||||||
|
if (cr && cr[0]) {
|
||||||
|
unsigned long v = strtoul(cr, NULL, 10);
|
||||||
|
if (v <= 0xFFFFFFFFul) wo_wal_ckpt_ratio = (uint32_t)v;
|
||||||
|
}
|
||||||
|
}
|
||||||
/* the arc's stage 2: all cores by default (the brave landing), one
|
/* the arc's stage 2: all cores by default (the brave landing), one
|
||||||
* pinned worker vm per extra core; WO_SHARDS caps or forces it */
|
* pinned worker vm per extra core; WO_SHARDS caps or forces it */
|
||||||
{
|
{
|
||||||
|
|
@ -269,7 +309,7 @@ int main(int argc, char **argv) {
|
||||||
if (wo_engine_start(&mod, heap_mb << 20, nshards) != 0) {
|
if (wo_engine_start(&mod, heap_mb << 20, nshards) != 0) {
|
||||||
fprintf(stderr, "wovm: cannot start %u shards\n", nshards);
|
fprintf(stderr, "wovm: cannot start %u shards\n", nshards);
|
||||||
wo_engine_stop();
|
wo_engine_stop();
|
||||||
if (VM.rt.wal) wo_wal_close(&WAL);
|
if (VM.rt.wal) { wal_stats_report(&WAL); wo_wal_close(&WAL); }
|
||||||
wo_db_destroy(&DB);
|
wo_db_destroy(&DB);
|
||||||
wo_vm_destroy(&VM);
|
wo_vm_destroy(&VM);
|
||||||
wo_module_free(&mod);
|
wo_module_free(&mod);
|
||||||
|
|
@ -328,7 +368,7 @@ int main(int argc, char **argv) {
|
||||||
* unwind. */
|
* unwind. */
|
||||||
if (argv_val) wo_drop_kind(&VM.rt, WO_K_MULTI, argv_val);
|
if (argv_val) wo_drop_kind(&VM.rt, WO_K_MULTI, argv_val);
|
||||||
wo_engine_stop(); /* join + destroy the worker shards before the primary */
|
wo_engine_stop(); /* join + destroy the worker shards before the primary */
|
||||||
if (VM.rt.wal) wo_wal_close(&WAL);
|
if (VM.rt.wal) { wal_stats_report(&WAL); wo_wal_close(&WAL); }
|
||||||
wo_db_destroy(&DB);
|
wo_db_destroy(&DB);
|
||||||
gc_pump(&VM);
|
gc_pump(&VM);
|
||||||
wo_vm_destroy(&VM);
|
wo_vm_destroy(&VM);
|
||||||
|
|
|
||||||
|
|
@ -296,6 +296,8 @@ static void tick_arm_uring(wo_vm *vm, int64_t now) {
|
||||||
for (wo_fiber *fb = vm->parked; fb; fb = fb->pnext)
|
for (wo_fiber *fb = vm->parked; fb; fb = fb->pnext)
|
||||||
if (fb->state == WO_FIB_PARKED && fb->park_fd >= 0 && fb->park_deadline > 0)
|
if (fb->state == WO_FIB_PARKED && fb->park_fd >= 0 && fb->park_deadline > 0)
|
||||||
if (next == 0 || fb->park_deadline < next) next = fb->park_deadline;
|
if (next == 0 || fb->park_deadline < next) next = fb->park_deadline;
|
||||||
|
int64_t tn = wo_vm_timers_next(vm); /* iteration 24 T5: armed timers */
|
||||||
|
if (tn > 0 && (next == 0 || tn < next)) next = tn;
|
||||||
if (next == 0) return;
|
if (next == 0) return;
|
||||||
if (vm->tick_armed && vm->tick_at <= next) return;
|
if (vm->tick_armed && vm->tick_at <= next) return;
|
||||||
int64_t rel = next - now;
|
int64_t rel = next - now;
|
||||||
|
|
@ -323,7 +325,27 @@ static void efd_drain(wo_vm *vm) {
|
||||||
|
|
||||||
int wo_io_wait(wo_vm *vm) {
|
int wo_io_wait(wo_vm *vm) {
|
||||||
for (;;) {
|
for (;;) {
|
||||||
if (wo_sys_stop_pending()) return WO_IO_STOP;
|
if (wo_sys_stop_pending()) {
|
||||||
|
/* iteration 24 (the drain): a STOP does not kill parked fibers
|
||||||
|
* from the outside — it WAKES them all, and each blocking
|
||||||
|
* builtin resolves per its own stop contract (deadline'd waits
|
||||||
|
* answer their timeout result, sleeps return early, plain
|
||||||
|
* waits answer WO_SYS_STOPPED and that fiber unwinds). The
|
||||||
|
* program's own code then drains and returns. Nothing parked
|
||||||
|
* = nothing to resolve: the old immediate-stop answer. */
|
||||||
|
int woke = 0;
|
||||||
|
wo_fiber *fb = vm->parked;
|
||||||
|
while (fb) {
|
||||||
|
wo_fiber *nx = fb->pnext;
|
||||||
|
if (fb->state == WO_FIB_PARKED) {
|
||||||
|
wake(vm, fb);
|
||||||
|
woke = 1;
|
||||||
|
}
|
||||||
|
fb = nx;
|
||||||
|
}
|
||||||
|
if (woke) return 0;
|
||||||
|
return WO_IO_STOP;
|
||||||
|
}
|
||||||
if (vm->io_kind == 0) {
|
if (vm->io_kind == 0) {
|
||||||
/* keep the wake eventfd armed (oneshot POLL_ADD, re-armed
|
/* keep the wake eventfd armed (oneshot POLL_ADD, re-armed
|
||||||
* after each firing) so inbox pushes interrupt the wait */
|
* after each firing) so inbox pushes interrupt the wait */
|
||||||
|
|
@ -371,7 +393,9 @@ int wo_io_wait(wo_vm *vm) {
|
||||||
head++;
|
head++;
|
||||||
}
|
}
|
||||||
__atomic_store_n(r.cq_head, head, __ATOMIC_RELEASE);
|
__atomic_store_n(r.cq_head, head, __ATOMIC_RELEASE);
|
||||||
if (deadline_sweep_uring(vm, now_ms()) && woke != 2) woke = 1;
|
int64_t swnow = now_ms();
|
||||||
|
if (wo_vm_timers_fire(vm, swnow) && woke != 2) woke = 1;
|
||||||
|
if (deadline_sweep_uring(vm, swnow) && woke != 2) woke = 1;
|
||||||
if (woke == 2) return 1; /* adopt-needed */
|
if (woke == 2) return 1; /* adopt-needed */
|
||||||
if (woke) return 0;
|
if (woke) return 0;
|
||||||
continue;
|
continue;
|
||||||
|
|
@ -389,6 +413,14 @@ int wo_io_wait(wo_vm *vm) {
|
||||||
}
|
}
|
||||||
int timeout = -1;
|
int timeout = -1;
|
||||||
int64_t now = now_ms();
|
int64_t now = now_ms();
|
||||||
|
{
|
||||||
|
int64_t tn = wo_vm_timers_next(vm); /* iteration 24 T5 */
|
||||||
|
if (tn > 0) {
|
||||||
|
int64_t rel = tn - now;
|
||||||
|
if (rel < 0) rel = 0;
|
||||||
|
timeout = (int)rel;
|
||||||
|
}
|
||||||
|
}
|
||||||
for (wo_fiber *fb = vm->parked; fb; fb = fb->pnext)
|
for (wo_fiber *fb = vm->parked; fb; fb = fb->pnext)
|
||||||
if (fb->park_fd == -1
|
if (fb->park_fd == -1
|
||||||
|| (fb->park_fd >= 0 && fb->park_deadline > 0)) {
|
|| (fb->park_fd >= 0 && fb->park_deadline > 0)) {
|
||||||
|
|
@ -417,6 +449,7 @@ int wo_io_wait(wo_vm *vm) {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
now = now_ms();
|
now = now_ms();
|
||||||
|
if (wo_vm_timers_fire(vm, now)) woke = 1;
|
||||||
wo_fiber *fb = vm->parked;
|
wo_fiber *fb = vm->parked;
|
||||||
while (fb) {
|
while (fb) {
|
||||||
wo_fiber *nx = fb->pnext;
|
wo_fiber *nx = fb->pnext;
|
||||||
|
|
|
||||||
|
|
@ -395,6 +395,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
||||||
if (stop_pending()) return WO_SYS_STOPPED;
|
if (stop_pending()) return WO_SYS_STOPPED;
|
||||||
}
|
}
|
||||||
if (fd < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) {
|
if (fd < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) {
|
||||||
|
if (stop_pending()) return WO_SYS_STOPPED;
|
||||||
/* arc T4: park until the listener is readable, then retry */
|
/* arc T4: park until the listener is readable, then retry */
|
||||||
vm->cur->park_fd = (int)R[B];
|
vm->cur->park_fd = (int)R[B];
|
||||||
vm->cur->park_deadline = 0;
|
vm->cur->park_deadline = 0;
|
||||||
|
|
@ -431,6 +432,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
||||||
/* arc T4: nothing readable yet — free the buffer (the retry
|
/* arc T4: nothing readable yet — free the buffer (the retry
|
||||||
* re-allocates) and park until the fd is readable */
|
* re-allocates) and park until the fd is readable */
|
||||||
wo_str_free(rt, s);
|
wo_str_free(rt, s);
|
||||||
|
if (stop_pending()) return WO_SYS_STOPPED;
|
||||||
vm->cur->park_fd = (int)R[B];
|
vm->cur->park_fd = (int)R[B];
|
||||||
vm->cur->park_deadline = 0;
|
vm->cur->park_deadline = 0;
|
||||||
vm->cur->park_events = POLLIN;
|
vm->cur->park_events = POLLIN;
|
||||||
|
|
@ -476,6 +478,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if (errno == EAGAIN || errno == EWOULDBLOCK) {
|
if (errno == EAGAIN || errno == EWOULDBLOCK) {
|
||||||
|
if (stop_pending()) return WO_SYS_STOPPED;
|
||||||
vm->cur->park_wr_at = at;
|
vm->cur->park_wr_at = at;
|
||||||
vm->cur->park_fd = (int)R[B];
|
vm->cur->park_fd = (int)R[B];
|
||||||
vm->cur->park_deadline = 0;
|
vm->cur->park_deadline = 0;
|
||||||
|
|
@ -534,7 +537,9 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
||||||
}
|
}
|
||||||
if (n < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) {
|
if (n < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) {
|
||||||
wo_str_free(rt, s);
|
wo_str_free(rt, s);
|
||||||
if (fb->dl_at > 0 && dnow >= fb->dl_at) {
|
if (stop_pending() || (fb->dl_at > 0 && dnow >= fb->dl_at)) {
|
||||||
|
/* iteration 24: a STOP resolves the wait as its timeout
|
||||||
|
* result — the program's own drain code decides what next */
|
||||||
fb->dl_active = 0;
|
fb->dl_active = 0;
|
||||||
R[A] = 0; /* ?Text nil: the deadline expired */
|
R[A] = 0; /* ?Text nil: the deadline expired */
|
||||||
return 0;
|
return 0;
|
||||||
|
|
@ -584,9 +589,9 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (fd < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) {
|
if (fd < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) {
|
||||||
if (fb->dl_at > 0 && dnow >= fb->dl_at) {
|
if (stop_pending() || (fb->dl_at > 0 && dnow >= fb->dl_at)) {
|
||||||
fb->dl_active = 0;
|
fb->dl_active = 0;
|
||||||
R[A] = WO_NIL_SCALAR; /* ?Int nil: nothing arrived */
|
R[A] = WO_NIL_SCALAR; /* ?Int nil: nothing arrived (or stop) */
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
fb->park_fd = (int)R[B];
|
fb->park_fd = (int)R[B];
|
||||||
|
|
@ -631,7 +636,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if (errno == EAGAIN || errno == EWOULDBLOCK) {
|
if (errno == EAGAIN || errno == EWOULDBLOCK) {
|
||||||
if (fb->dl_at > 0 && dnow >= fb->dl_at) {
|
if (stop_pending() || (fb->dl_at > 0 && dnow >= fb->dl_at)) {
|
||||||
fb->dl_active = 0;
|
fb->dl_active = 0;
|
||||||
R[A] = 0; /* false: torn mid-write — close the fd */
|
R[A] = 0; /* false: torn mid-write — close the fd */
|
||||||
return 0;
|
return 0;
|
||||||
|
|
|
||||||
416
runtime/src/vm.c
416
runtime/src/vm.c
|
|
@ -16,6 +16,7 @@
|
||||||
|
|
||||||
#include "db.h" /* arc stage 3: the transparent DB RPC (wo_db_req) */
|
#include "db.h" /* arc stage 3: the transparent DB RPC (wo_db_req) */
|
||||||
#include "table.h" /* slot encode/decode for the RPC marshaling */
|
#include "table.h" /* slot encode/decode for the RPC marshaling */
|
||||||
|
#include "wal.h" /* databasev2 4: the drain issues the barrier */
|
||||||
|
|
||||||
#include <pthread.h>
|
#include <pthread.h>
|
||||||
#include <poll.h>
|
#include <poll.h>
|
||||||
|
|
@ -78,6 +79,9 @@ static int actor_push(wo_actor *a, wo_msg m);
|
||||||
static void call_reply_to(wo_vm *vm, wo_fiber *caller, uint32_t caller_shard,
|
static void call_reply_to(wo_vm *vm, wo_fiber *caller, uint32_t caller_shard,
|
||||||
uint64_t reply, int status);
|
uint64_t reply, int status);
|
||||||
static void actor_drop_payload(wo_vm *vm, uint64_t payload);
|
static void actor_drop_payload(wo_vm *vm, uint64_t payload);
|
||||||
|
static void monitors_fire(wo_vm *vm, wo_actor *a);
|
||||||
|
static void runtime_notify(wo_vm *vm, wo_actor *target, uint64_t msg_val,
|
||||||
|
const char *what);
|
||||||
|
|
||||||
/* the owning thread drains its inbox: adopt actors, deliver sends,
|
/* the owning thread drains its inbox: adopt actors, deliver sends,
|
||||||
* execute home-routed frees. Returns how many envelopes were handled. */
|
* execute home-routed frees. Returns how many envelopes were handled. */
|
||||||
|
|
@ -88,6 +92,11 @@ static int wo_vm_adopt(wo_vm *vm) {
|
||||||
ib->head = ib->tail = NULL;
|
ib->head = ib->tail = NULL;
|
||||||
pthread_mutex_unlock(&ib->mu);
|
pthread_mutex_unlock(&ib->mu);
|
||||||
int n = 0;
|
int n = 0;
|
||||||
|
/* databasev2 4 (group commit): DB replies are HELD until one barrier has
|
||||||
|
* covered the whole drain. Locals, not per-shard state: nothing here needs
|
||||||
|
* to outlive the batch it describes. */
|
||||||
|
wo_envelope *rhead = NULL, *rtail = NULL;
|
||||||
|
uint32_t staged = 0;
|
||||||
while (e) {
|
while (e) {
|
||||||
wo_envelope *nx = e->next;
|
wo_envelope *nx = e->next;
|
||||||
switch (e->kind) {
|
switch (e->kind) {
|
||||||
|
|
@ -133,6 +142,25 @@ static int wo_vm_adopt(wo_vm *vm) {
|
||||||
}
|
}
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
case 7: { /* iteration 24 T4: a cross-shard monitor registration —
|
||||||
|
WE are the watched actor's home. Dead already = the
|
||||||
|
notice fires now; else it joins the list. */
|
||||||
|
wo_actor *ob = (wo_actor *)(uintptr_t)e->from_fiber;
|
||||||
|
if (e->actor->dead) {
|
||||||
|
runtime_notify(vm, ob, e->payload, "death notice");
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
wo_monitor *mn = calloc(1, sizeof *mn);
|
||||||
|
if (!mn) {
|
||||||
|
actor_drop_payload(vm, e->payload);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
mn->observer = ob;
|
||||||
|
mn->msg = e->payload;
|
||||||
|
mn->next = e->actor->monitors;
|
||||||
|
e->actor->monitors = mn;
|
||||||
|
break;
|
||||||
|
}
|
||||||
case 6: /* iteration 24: a call reply landing on the caller's shard —
|
case 6: /* iteration 24: a call reply landing on the caller's shard —
|
||||||
fill the slot and wake the parked fiber; the re-executed
|
fill the slot and wake the parked fiber; the re-executed
|
||||||
builtin consumes it (status != 0 makes it trap). */
|
builtin consumes it (status != 0 makes it trap). */
|
||||||
|
|
@ -149,13 +177,35 @@ static int wo_vm_adopt(wo_vm *vm) {
|
||||||
* the same request back as the reply. */
|
* the same request back as the reply. */
|
||||||
wo_db_req *q = (wo_db_req *)(uintptr_t)e->payload;
|
wo_db_req *q = (wo_db_req *)(uintptr_t)e->payload;
|
||||||
assert(vm->is_primary && "DB requests route to shard 0 only");
|
assert(vm->is_primary && "DB requests route to shard 0 only");
|
||||||
|
wo_wal *dw = (wo_wal *)vm->rt.wal;
|
||||||
|
size_t before = dw ? dw->len : 0;
|
||||||
wo_db_exec_req(vm, q);
|
wo_db_exec_req(vm, q);
|
||||||
q->done = 1;
|
q->done = 1;
|
||||||
|
/* did this statement actually stage a record? Asking the buffer
|
||||||
|
* beats guessing from the opcode, and the count is what the
|
||||||
|
* failure diagnostic reports. */
|
||||||
|
if (dw && dw->len > before) staged++;
|
||||||
wo_envelope *re = calloc(1, sizeof *re);
|
wo_envelope *re = calloc(1, sizeof *re);
|
||||||
if (re) {
|
if (re) {
|
||||||
re->kind = 4;
|
re->kind = 4;
|
||||||
re->payload = e->payload;
|
re->payload = e->payload;
|
||||||
inbox_push_to(q->from_shard, re);
|
re->next = NULL;
|
||||||
|
if (dw && dw->len > before) {
|
||||||
|
/* This statement STAGED a record, so its reply is HELD:
|
||||||
|
* pushing it now would unpark the requester before its
|
||||||
|
* record is durable, which is the ack contract this
|
||||||
|
* iteration exists to make literally true. FIFO, so the
|
||||||
|
* first waiter is released first. */
|
||||||
|
if (rtail) rtail->next = re; else rhead = re;
|
||||||
|
rtail = re;
|
||||||
|
} else {
|
||||||
|
/* A READ (or any statement that staged nothing) has no
|
||||||
|
* durability to wait for. Holding it too was measurably
|
||||||
|
* wrong: it parked readers behind an fsync they had no
|
||||||
|
* stake in, and durable.sN.mixread p99 rose ~4x
|
||||||
|
* (1043 -> 4057us) until this branch existed. */
|
||||||
|
inbox_push_to(q->from_shard, re);
|
||||||
|
}
|
||||||
} /* OOM: the requester stays parked until stop — leak, not UB */
|
} /* OOM: the requester stays parked until stop — leak, not UB */
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
@ -170,6 +220,41 @@ static int wo_vm_adopt(wo_vm *vm) {
|
||||||
n++;
|
n++;
|
||||||
e = nx;
|
e = nx;
|
||||||
}
|
}
|
||||||
|
/* databasev2 4: ONE barrier for everything this drain staged, then every
|
||||||
|
* held reply. Each requester therefore unparks having been acknowledged
|
||||||
|
* after the barrier that carried ITS record. Commit unconditionally when
|
||||||
|
* anything is staged — the inline path relies on finding the buffer empty
|
||||||
|
* (see db.c), so a drain must never leave a record behind. */
|
||||||
|
if (staged) {
|
||||||
|
wo_wal *cw = (wo_wal *)vm->rt.wal;
|
||||||
|
if (cw) wo_wal_commit_fatal(cw, staged);
|
||||||
|
}
|
||||||
|
while (rhead) {
|
||||||
|
wo_envelope *rn = rhead->next;
|
||||||
|
wo_db_req *rq = (wo_db_req *)(uintptr_t)rhead->payload;
|
||||||
|
rhead->next = NULL;
|
||||||
|
inbox_push_to(rq->from_shard, rhead);
|
||||||
|
rhead = rn;
|
||||||
|
}
|
||||||
|
/* databasev2 3: the ONE point where compaction is safe — the barrier above
|
||||||
|
* just ran, so the staging buffer is empty. Anywhere else, a staged record
|
||||||
|
* would be written into a file about to be replaced. This is a correctness
|
||||||
|
* requirement, not a scheduling preference; wo_wal_compact also refuses a
|
||||||
|
* non-empty buffer as a backstop.
|
||||||
|
*
|
||||||
|
* Replies are released FIRST, deliberately: their records are already
|
||||||
|
* durable, and holding them across a stop-the-world rewrite would add the
|
||||||
|
* rewrite's full duration to their latency for no benefit.
|
||||||
|
*
|
||||||
|
* The result is ignored because a failed compaction is a missed
|
||||||
|
* optimisation, not a durability event — the original log is left intact
|
||||||
|
* and the process carries on. */
|
||||||
|
if (staged) {
|
||||||
|
wo_wal *cw = (wo_wal *)vm->rt.wal;
|
||||||
|
if (cw && wo_wal_should_compact(cw->off, cw->compacted_bytes,
|
||||||
|
wo_wal_ckpt_floor, wo_wal_ckpt_ratio))
|
||||||
|
(void)wo_wal_compact(cw, (wo_db *)vm->rt.db);
|
||||||
|
}
|
||||||
return n;
|
return n;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -425,6 +510,33 @@ static void *shard_main(void *arg) {
|
||||||
} else {
|
} else {
|
||||||
int rc = wo_io_wait(vm); /* parked fibers AND the wake eventfd */
|
int rc = wo_io_wait(vm); /* parked fibers AND the wake eventfd */
|
||||||
if (rc == WO_IO_STOP) {
|
if (rc == WO_IO_STOP) {
|
||||||
|
/* iteration 40 — THE DRAIN GUARANTEE. A message sent before
|
||||||
|
* the stop flag is observed must be delivered and run before
|
||||||
|
* the engine stops.
|
||||||
|
*
|
||||||
|
* NEXT_RUNNABLE() already states this contract for a worker
|
||||||
|
* holding a live fiber: it returns 2 and keeps draining "so
|
||||||
|
* queued shutdown messages (close frames!) still run". This
|
||||||
|
* branch — the IDLE worker, empty run queue, waiting on the
|
||||||
|
* plane — used to reap and break instead, abandoning whatever
|
||||||
|
* sat in its inbox for wo_engine_stop() to free wholesale.
|
||||||
|
*
|
||||||
|
* An actor between messages is exactly that idle case, which
|
||||||
|
* is why a WARM server hid the bug: warm shards had live
|
||||||
|
* fibers and took the correct path. Measured 2026-08-27 on a
|
||||||
|
* fresh server: 5 of 16 SIGTERM drains left a WebSocket
|
||||||
|
* client at EOF with no close frame and no diagnostic.
|
||||||
|
*
|
||||||
|
* The window belongs to the PRIMARY and closes when it sets
|
||||||
|
* eng_shutdown (after main returns), so honour it here and
|
||||||
|
* only exit when the primary says so. Yield on an empty poll:
|
||||||
|
* a tight loop would burn a core per shard and starve the very
|
||||||
|
* actors the drain exists to let run. */
|
||||||
|
if (!eng_shutdown) {
|
||||||
|
(void)wo_vm_adopt(vm);
|
||||||
|
if (!vm->qhead) sched_yield();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
fib_reap_all(vm);
|
fib_reap_all(vm);
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
@ -445,6 +557,85 @@ int wo_engine_primary_inbox(int wake_efd) {
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* iteration 24 teardown phase 1 (single-threaded, BEFORE eng_teardown):
|
||||||
|
* dismantle one vm's actor world with real drops — container backings are
|
||||||
|
* malloc'd, so wholesale arena death does NOT cover them (LSan, chat's
|
||||||
|
* registry map). Cross-shard payloads route home through wo_route_free
|
||||||
|
* (still live here); the routed kind-2 envelopes are settled by the
|
||||||
|
* caller's inbox passes. */
|
||||||
|
static void vm_drop_actor_world(wo_vm *vm) {
|
||||||
|
wo_actor *a = vm->actors;
|
||||||
|
vm->actors = NULL;
|
||||||
|
while (a) {
|
||||||
|
wo_actor *nx = a->next_all;
|
||||||
|
if (a->instance) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)a->instance);
|
||||||
|
for (uint32_t i = 0; i < a->mlen; i++) {
|
||||||
|
uint64_t m = a->msgs[(a->mhead + i) % a->mcap].payload;
|
||||||
|
if (m) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)m);
|
||||||
|
}
|
||||||
|
wo_monitor *mo = a->monitors;
|
||||||
|
while (mo) {
|
||||||
|
wo_monitor *mnx = mo->next;
|
||||||
|
if (mo->msg) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)mo->msg);
|
||||||
|
free(mo);
|
||||||
|
mo = mnx;
|
||||||
|
}
|
||||||
|
free(a->msgs);
|
||||||
|
free(a);
|
||||||
|
a = nx;
|
||||||
|
}
|
||||||
|
wo_timer *tt = vm->timers;
|
||||||
|
vm->timers = NULL;
|
||||||
|
while (tt) {
|
||||||
|
wo_timer *tnx = tt->next;
|
||||||
|
if (tt->msg) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)tt->msg);
|
||||||
|
free(tt);
|
||||||
|
tt = tnx;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Settle every inbox after phase 1: home-routed frees execute on their
|
||||||
|
* owner vm; payload-carrying strays drop (possibly routing again — the
|
||||||
|
* outer loop runs until everything is quiet). Node memory always freed. */
|
||||||
|
static int eng_settle_inboxes(void) {
|
||||||
|
int moved = 0;
|
||||||
|
for (uint32_t i = 0; i < wo_eng.nshards && i < WO_ENG_MAX_SHARDS; i++) {
|
||||||
|
if (!INBOX_READY[i]) continue;
|
||||||
|
wo_vm *vm = &wo_eng.shards[i];
|
||||||
|
wo_inbox *ib = &INBOX[i];
|
||||||
|
wo_envelope *e = ib->head;
|
||||||
|
ib->head = ib->tail = NULL;
|
||||||
|
while (e) {
|
||||||
|
wo_envelope *nx = e->next;
|
||||||
|
switch (e->kind) {
|
||||||
|
case 2: /* WE are home: the direct drop is the settlement */
|
||||||
|
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)e->payload);
|
||||||
|
break;
|
||||||
|
case 0:
|
||||||
|
case 5:
|
||||||
|
case 7: /* in-flight payloads: drop (may route -> next pass) */
|
||||||
|
if (e->payload)
|
||||||
|
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)e->payload);
|
||||||
|
break;
|
||||||
|
case 1: /* an unadopted actor shell */
|
||||||
|
if (e->actor) {
|
||||||
|
if (e->actor->instance)
|
||||||
|
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)e->actor->instance);
|
||||||
|
free(e->actor->msgs);
|
||||||
|
free(e->actor);
|
||||||
|
}
|
||||||
|
break;
|
||||||
|
default: /* 3/4/6: scalar or engine-side payloads, node-only */
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
free(e);
|
||||||
|
moved++;
|
||||||
|
e = nx;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return moved;
|
||||||
|
}
|
||||||
|
|
||||||
int wo_engine_start(const wo_module *mod, size_t heap_cap, uint32_t nshards) {
|
int wo_engine_start(const wo_module *mod, size_t heap_cap, uint32_t nshards) {
|
||||||
wo_eng.nshards = nshards;
|
wo_eng.nshards = nshards;
|
||||||
eng_heap_cap = heap_cap;
|
eng_heap_cap = heap_cap;
|
||||||
|
|
@ -489,6 +680,13 @@ void wo_engine_stop(void) {
|
||||||
(void)n;
|
(void)n;
|
||||||
}
|
}
|
||||||
for (uint32_t i = 1; i < wo_eng.nshards; i++) pthread_join(ts[i - 1], NULL);
|
for (uint32_t i = 1; i < wo_eng.nshards; i++) pthread_join(ts[i - 1], NULL);
|
||||||
|
/* single-threaded from here: PHASE 1 — real drops while every arena
|
||||||
|
* and the routing fabric are still alive (malloc'd container backings
|
||||||
|
* inside actor state need them; iteration 24's registry map). Settle
|
||||||
|
* passes run until routed frees stop appearing. */
|
||||||
|
for (uint32_t i = 0; i < wo_eng.nshards && i < WO_ENG_MAX_SHARDS; i++)
|
||||||
|
if (wo_eng.shards[i].rt.arena.base) vm_drop_actor_world(&wo_eng.shards[i]);
|
||||||
|
while (eng_settle_inboxes() > 0) {}
|
||||||
/* single-threaded from here. Every arena dies wholesale, so routed
|
/* single-threaded from here. Every arena dies wholesale, so routed
|
||||||
* frees and queued payloads need no per-object drops — DISCARD the
|
* frees and queued payloads need no per-object drops — DISCARD the
|
||||||
* envelopes (freeing the malloc'd nodes/actors) and let the arenas
|
* envelopes (freeing the malloc'd nodes/actors) and let the arenas
|
||||||
|
|
@ -557,19 +755,42 @@ void wo_vm_destroy(wo_vm *vm) {
|
||||||
free(fb);
|
free(fb);
|
||||||
}
|
}
|
||||||
/* actors first — dropping their state and queued messages needs the
|
/* actors first — dropping their state and queued messages needs the
|
||||||
* runtime alive */
|
* runtime alive. BUT: once the engine is in teardown, arenas die
|
||||||
|
* WHOLESALE (the standing doctrine) — a moved-in message's home arena
|
||||||
|
* may belong to an ALREADY-destroyed shard, and even reading its
|
||||||
|
* header is a use-after-free (ASan, chat's drain). Structures are
|
||||||
|
* still freed; payload drops are skipped. */
|
||||||
|
int drops_ok = !eng_teardown;
|
||||||
wo_actor *a = vm->actors;
|
wo_actor *a = vm->actors;
|
||||||
while (a) {
|
while (a) {
|
||||||
wo_actor *nx = a->next_all;
|
wo_actor *nx = a->next_all;
|
||||||
if (a->instance) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)a->instance);
|
if (drops_ok && a->instance)
|
||||||
for (uint32_t i = 0; i < a->mlen; i++) {
|
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)a->instance);
|
||||||
|
for (uint32_t i = 0; drops_ok && i < a->mlen; i++) {
|
||||||
uint64_t m = a->msgs[(a->mhead + i) % a->mcap].payload;
|
uint64_t m = a->msgs[(a->mhead + i) % a->mcap].payload;
|
||||||
if (m) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)m);
|
if (m) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)m);
|
||||||
}
|
}
|
||||||
|
wo_monitor *mo = a->monitors;
|
||||||
|
while (mo) { /* undelivered notices are the runtime's to drop */
|
||||||
|
wo_monitor *mnx = mo->next;
|
||||||
|
if (drops_ok && mo->msg)
|
||||||
|
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)mo->msg);
|
||||||
|
free(mo);
|
||||||
|
mo = mnx;
|
||||||
|
}
|
||||||
free(a->msgs);
|
free(a->msgs);
|
||||||
free(a);
|
free(a);
|
||||||
a = nx;
|
a = nx;
|
||||||
}
|
}
|
||||||
|
wo_timer *tt = vm->timers;
|
||||||
|
vm->timers = NULL;
|
||||||
|
while (tt) { /* unfired timers likewise */
|
||||||
|
wo_timer *tnx = tt->next;
|
||||||
|
if (drops_ok && tt->msg)
|
||||||
|
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)tt->msg);
|
||||||
|
free(tt);
|
||||||
|
tt = tnx;
|
||||||
|
}
|
||||||
vm->actors = NULL;
|
vm->actors = NULL;
|
||||||
wo_io_destroy(vm);
|
wo_io_destroy(vm);
|
||||||
wo_rt_destroy(&vm->rt);
|
wo_rt_destroy(&vm->rt);
|
||||||
|
|
@ -760,6 +981,7 @@ static void actor_die(wo_vm *vm, wo_actor *a, wo_fiber *delivery) {
|
||||||
a->instance = 0;
|
a->instance = 0;
|
||||||
}
|
}
|
||||||
a->active = NULL;
|
a->active = NULL;
|
||||||
|
monitors_fire(vm, a);
|
||||||
}
|
}
|
||||||
|
|
||||||
/* Mailbox nonempty, no delivery fiber: start one on the next message.
|
/* Mailbox nonempty, no delivery fiber: start one on the next message.
|
||||||
|
|
@ -877,6 +1099,57 @@ int wo_vm_actor_send(wo_vm *vm, uint64_t addr, uint64_t msg_val, const char **ms
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* iteration 24 T4/T5: a RUNTIME-sourced delivery (death notice, timer).
|
||||||
|
* No fiber to trap: a full or dead target drops the message with a
|
||||||
|
* stderr line (spec'd disclosure), never silently. Runs on any thread —
|
||||||
|
* cross-shard targets ride the ordinary kind-0 envelope. */
|
||||||
|
static void runtime_notify(wo_vm *vm, wo_actor *target, uint64_t msg_val,
|
||||||
|
const char *what) {
|
||||||
|
if (!target || !msg_val) return;
|
||||||
|
if (target->dead) {
|
||||||
|
actor_drop_payload(vm, msg_val);
|
||||||
|
return; /* send-to-dead: silent by contract */
|
||||||
|
}
|
||||||
|
if (wo_mbox_reserve(target) != 0) {
|
||||||
|
fprintf(stderr, "wovm: %s dropped — the observer's mailbox is full\n", what);
|
||||||
|
actor_drop_payload(vm, msg_val);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (target->home != vm->shard_id) {
|
||||||
|
wo_envelope *e = calloc(1, sizeof *e);
|
||||||
|
if (!e) {
|
||||||
|
wo_mbox_release(target);
|
||||||
|
actor_drop_payload(vm, msg_val);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
e->kind = 0;
|
||||||
|
e->actor = target;
|
||||||
|
e->payload = msg_val;
|
||||||
|
inbox_push_to(target->home, e);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
wo_msg m0 = { msg_val, NULL, 0 };
|
||||||
|
if (actor_push(target, m0) != 0) {
|
||||||
|
wo_mbox_release(target);
|
||||||
|
actor_drop_payload(vm, msg_val);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (!target->active) (void)actor_activate(vm, target);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* iteration 24 T4: the death walk — every registered observer gets its
|
||||||
|
* chosen notice, then the list is gone (an actor dies once). */
|
||||||
|
static void monitors_fire(wo_vm *vm, wo_actor *a) {
|
||||||
|
wo_monitor *m = a->monitors;
|
||||||
|
a->monitors = NULL;
|
||||||
|
while (m) {
|
||||||
|
wo_monitor *nx = m->next;
|
||||||
|
runtime_notify(vm, m->observer, m->msg, "death notice");
|
||||||
|
free(m);
|
||||||
|
m = nx;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/* iteration 24: call — send that waits. First entry enqueues with the
|
/* iteration 24: call — send that waits. First entry enqueues with the
|
||||||
* caller attached and parks (WO_PARK_INBOX, the DB-RPC park); the resume
|
* caller attached and parks (WO_PARK_INBOX, the DB-RPC park); the resume
|
||||||
* RE-EXECUTES this builtin and consumes the scalar reply. No hangs, ever:
|
* RE-EXECUTES this builtin and consumes the scalar reply. No hangs, ever:
|
||||||
|
|
@ -946,6 +1219,105 @@ int wo_vm_actor_call(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
||||||
return WO_SYS_PARKED;
|
return WO_SYS_PARKED;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
int wo_vm_actor_monitor(wo_vm *vm, uint64_t watched, uint64_t observer,
|
||||||
|
uint64_t msg_val, const char **msg) {
|
||||||
|
wo_actor *w = (wo_actor *)(uintptr_t)watched;
|
||||||
|
wo_actor *o = (wo_actor *)(uintptr_t)observer;
|
||||||
|
if (!w || !o) {
|
||||||
|
*msg = "monitor: nil actor address";
|
||||||
|
return WO_T_BOUNDS;
|
||||||
|
}
|
||||||
|
if (!msg_val) {
|
||||||
|
*msg = "monitor: nil notice message";
|
||||||
|
return WO_T_BOUNDS;
|
||||||
|
}
|
||||||
|
/* the registration belongs to the WATCHED actor's home thread */
|
||||||
|
if (w->home != vm->shard_id) {
|
||||||
|
wo_envelope *e = calloc(1, sizeof *e);
|
||||||
|
if (!e) {
|
||||||
|
actor_drop_payload(vm, msg_val);
|
||||||
|
*msg = "out of memory";
|
||||||
|
return WO_T_OOM;
|
||||||
|
}
|
||||||
|
e->kind = 7;
|
||||||
|
e->actor = w;
|
||||||
|
e->payload = msg_val;
|
||||||
|
e->from_fiber = (wo_fiber *)o; /* reused slot: the observer */
|
||||||
|
inbox_push_to(w->home, e);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
if (w->dead) { /* monitoring the dead: the notice fires NOW */
|
||||||
|
runtime_notify(vm, o, msg_val, "death notice");
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
wo_monitor *m = calloc(1, sizeof *m);
|
||||||
|
if (!m) {
|
||||||
|
actor_drop_payload(vm, msg_val);
|
||||||
|
*msg = "out of memory";
|
||||||
|
return WO_T_OOM;
|
||||||
|
}
|
||||||
|
m->observer = o;
|
||||||
|
m->msg = msg_val;
|
||||||
|
m->next = w->monitors;
|
||||||
|
w->monitors = m;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
int wo_vm_timer_after(wo_vm *vm, int64_t ms, uint64_t addr, uint64_t msg_val,
|
||||||
|
const char **msg) {
|
||||||
|
wo_actor *a = (wo_actor *)(uintptr_t)addr;
|
||||||
|
if (!a) {
|
||||||
|
*msg = "time.after: nil actor address";
|
||||||
|
return WO_T_BOUNDS;
|
||||||
|
}
|
||||||
|
if (!msg_val) {
|
||||||
|
*msg = "time.after: nil message";
|
||||||
|
return WO_T_BOUNDS;
|
||||||
|
}
|
||||||
|
if (ms <= 0) { /* no wait to arm: deliver now */
|
||||||
|
runtime_notify(vm, a, msg_val, "timer message");
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
wo_timer *t = calloc(1, sizeof *t);
|
||||||
|
if (!t) {
|
||||||
|
actor_drop_payload(vm, msg_val);
|
||||||
|
*msg = "out of memory";
|
||||||
|
return WO_T_OOM;
|
||||||
|
}
|
||||||
|
struct timespec now;
|
||||||
|
clock_gettime(CLOCK_REALTIME, &now);
|
||||||
|
t->at = (int64_t)now.tv_sec * 1000 + now.tv_nsec / 1000000 + ms;
|
||||||
|
t->target = a;
|
||||||
|
t->msg = msg_val;
|
||||||
|
t->next = vm->timers;
|
||||||
|
vm->timers = t;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
int wo_vm_timers_fire(wo_vm *vm, int64_t now) {
|
||||||
|
int fired = 0;
|
||||||
|
wo_timer **pp = &vm->timers;
|
||||||
|
while (*pp) {
|
||||||
|
wo_timer *t = *pp;
|
||||||
|
if (t->at <= now) {
|
||||||
|
*pp = t->next;
|
||||||
|
runtime_notify(vm, t->target, t->msg, "timer message");
|
||||||
|
free(t);
|
||||||
|
fired++;
|
||||||
|
} else {
|
||||||
|
pp = &t->next;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return fired;
|
||||||
|
}
|
||||||
|
|
||||||
|
int64_t wo_vm_timers_next(wo_vm *vm) {
|
||||||
|
int64_t next = 0;
|
||||||
|
for (wo_timer *t = vm->timers; t; t = t->next)
|
||||||
|
if (next == 0 || t->at < next) next = t->at;
|
||||||
|
return next;
|
||||||
|
}
|
||||||
|
|
||||||
/* The drop-table entry governing instruction [pc]: the last one recorded
|
/* The drop-table entry governing instruction [pc]: the last one recorded
|
||||||
* at or before it. NULL = nothing live there. */
|
* at or before it. NULL = nothing live there. */
|
||||||
static const wo_dropent *vm_dropent(const wo_methodrec *me, uint32_t pc) {
|
static const wo_dropent *vm_dropent(const wo_methodrec *me, uint32_t pc) {
|
||||||
|
|
@ -1227,6 +1599,15 @@ static int vm_run(wo_vm *vm, uint64_t *ret, wo_err *err) {
|
||||||
} \
|
} \
|
||||||
int iorc_ = wo_io_wait(vm); \
|
int iorc_ = wo_io_wait(vm); \
|
||||||
if (iorc_ == WO_IO_STOP) { \
|
if (iorc_ == WO_IO_STOP) { \
|
||||||
|
/* iteration 24: a WORKER on stop keeps DRAINING — its \
|
||||||
|
* serve loop spins adopting the inbox until the primary \
|
||||||
|
* finishes the drain window and sets eng_shutdown, so \
|
||||||
|
* queued shutdown messages (close frames!) still run. \
|
||||||
|
* Only the PRIMARY's stop ends the program. */ \
|
||||||
|
if (!vm->is_primary) { \
|
||||||
|
vm->cur = &vm->f0; \
|
||||||
|
return 2; \
|
||||||
|
} \
|
||||||
fib_reap_all(vm); \
|
fib_reap_all(vm); \
|
||||||
vm->cur = &vm->f0; \
|
vm->cur = &vm->f0; \
|
||||||
return 1; \
|
return 1; \
|
||||||
|
|
@ -1729,8 +2110,31 @@ dispatch:
|
||||||
vm->cur->frames[vm->cur->depth - 1].pc = pc - 1;
|
vm->cur->frames[vm->cur->depth - 1].pc = pc - 1;
|
||||||
vm->cur->ncatch = 0;
|
vm->cur->ncatch = 0;
|
||||||
vm_unwind(vm, 0);
|
vm_unwind(vm, 0);
|
||||||
/* a stop ends the PROGRAM: every fiber — the stopped one,
|
/* iteration 24 (the drain): a STOPPED wait on a NON-main fiber
|
||||||
* queued ones, main wherever it is — unwinds clean */
|
* unwinds that fiber ALONE — the rest of the program (main's
|
||||||
|
* drain code, actors flushing close frames) keeps running.
|
||||||
|
* Main's own STOPPED still ends the program, as ever. */
|
||||||
|
if (vm->cur != &vm->f0) {
|
||||||
|
wo_fiber *dead = vm->cur;
|
||||||
|
if (dead->actor) {
|
||||||
|
wo_actor *da = dead->actor;
|
||||||
|
if (dead->cur_msg) {
|
||||||
|
wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)dead->cur_msg);
|
||||||
|
dead->cur_msg = 0;
|
||||||
|
}
|
||||||
|
call_reply_to(vm, dead->msg_caller, dead->msg_caller_shard,
|
||||||
|
0, WO_T_ACTOR);
|
||||||
|
dead->msg_caller = NULL;
|
||||||
|
da->active = NULL;
|
||||||
|
}
|
||||||
|
vm->nfibers--;
|
||||||
|
fib_retire(vm, dead);
|
||||||
|
NEXT_RUNNABLE();
|
||||||
|
RELOAD();
|
||||||
|
NEXT();
|
||||||
|
}
|
||||||
|
/* main: a stop ends the PROGRAM — every remaining fiber
|
||||||
|
* unwinds clean */
|
||||||
if (vm->cur != &vm->f0) {
|
if (vm->cur != &vm->f0) {
|
||||||
wo_fiber *dead = vm->cur;
|
wo_fiber *dead = vm->cur;
|
||||||
vm->cur = &vm->f0;
|
vm->cur = &vm->f0;
|
||||||
|
|
|
||||||
|
|
@ -120,6 +120,27 @@ typedef struct wo_msg {
|
||||||
* guarantee). Death (iteration 24): a receive trapping uncaught marks
|
* guarantee). Death (iteration 24): a receive trapping uncaught marks
|
||||||
* the actor dead — sends to it drop silently, calls trap, queued
|
* the actor dead — sends to it drop silently, calls trap, queued
|
||||||
* callers are error-unparked; the state and mailbox are released. */
|
* callers are error-unparked; the state and mailbox are released. */
|
||||||
|
/* iteration 24 T4: one death-notice registration. The runtime owns the
|
||||||
|
* moved-in notice message until delivery (or drops it if the observer is
|
||||||
|
* unreachable). The list lives on the WATCHED actor, owned by its home
|
||||||
|
* thread. */
|
||||||
|
typedef struct wo_monitor {
|
||||||
|
struct wo_actor *observer;
|
||||||
|
uint64_t msg;
|
||||||
|
struct wo_monitor *next;
|
||||||
|
} wo_monitor;
|
||||||
|
|
||||||
|
/* iteration 24 T5: one armed one-shot timer — fires as an ordinary
|
||||||
|
* runtime send of the moved message when `at` passes. The list lives on
|
||||||
|
* the ARMING fiber's shard and is scanned by the same deadline machinery
|
||||||
|
* that serves fd-park deadlines. */
|
||||||
|
typedef struct wo_timer {
|
||||||
|
int64_t at; /* wall ms */
|
||||||
|
struct wo_actor *target;
|
||||||
|
uint64_t msg;
|
||||||
|
struct wo_timer *next;
|
||||||
|
} wo_timer;
|
||||||
|
|
||||||
typedef struct wo_actor {
|
typedef struct wo_actor {
|
||||||
uint64_t instance; /* the moved-in state object (runtime-owned) */
|
uint64_t instance; /* the moved-in state object (runtime-owned) */
|
||||||
uint32_t method; /* receive's method index (self + msg = 2 args) */
|
uint32_t method; /* receive's method index (self + msg = 2 args) */
|
||||||
|
|
@ -134,6 +155,7 @@ typedef struct wo_actor {
|
||||||
* overshoot by at most the number of in-flight sends — disclosed. */
|
* overshoot by at most the number of in-flight sends — disclosed. */
|
||||||
uint32_t pending;
|
uint32_t pending;
|
||||||
wo_fiber *active; /* the delivery fiber, NULL when idle */
|
wo_fiber *active; /* the delivery fiber, NULL when idle */
|
||||||
|
wo_monitor *monitors; /* iteration 24 T4: who wants the death notice */
|
||||||
struct wo_actor *next_all; /* the vm's all-actors list */
|
struct wo_actor *next_all; /* the vm's all-actors list */
|
||||||
} wo_actor;
|
} wo_actor;
|
||||||
|
|
||||||
|
|
@ -176,6 +198,9 @@ typedef struct wo_vm {
|
||||||
* freed memory is the UAF this prevents. Steady-state pool size = the
|
* freed memory is the UAF this prevents. Steady-state pool size = the
|
||||||
* peak live fiber count; the pool dies with the vm. */
|
* peak live fiber count; the pool dies with the vm. */
|
||||||
wo_fiber *fib_pool;
|
wo_fiber *fib_pool;
|
||||||
|
/* iteration 24 T5: this shard's armed timers (unsorted list — the
|
||||||
|
* deadline scan is already linear; a wheel is measured-later work) */
|
||||||
|
wo_timer *timers;
|
||||||
/* iteration 35, uring backend: the shard's ONE deadline tick — a
|
/* iteration 35, uring backend: the shard's ONE deadline tick — a
|
||||||
* TIMEOUT op with a sentinel user_data armed for the nearest fd-park
|
* TIMEOUT op with a sentinel user_data armed for the nearest fd-park
|
||||||
* deadline (fd parks keep exactly one POLL op each; expiry wakes them
|
* deadline (fd parks keep exactly one POLL op each; expiry wakes them
|
||||||
|
|
@ -206,6 +231,20 @@ int wo_vm_actor_send(wo_vm *vm, uint64_t addr, uint64_t msg_val, const char **ms
|
||||||
* caller attached and parks (WO_SYS_PARKED); the re-execution consumes the
|
* caller attached and parks (WO_SYS_PARKED); the re-execution consumes the
|
||||||
* scalar reply into R[A] (vm.c owns the protocol, builtin.c dispatches). */
|
* scalar reply into R[A] (vm.c owns the protocol, builtin.c dispatches). */
|
||||||
int wo_vm_actor_call(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg);
|
int wo_vm_actor_call(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg);
|
||||||
|
/* iteration 24 T4: register a death notice — monitor(watched, observer,
|
||||||
|
* msg). The msg MOVES to the runtime; an already-dead watched actor
|
||||||
|
* delivers it immediately. */
|
||||||
|
int wo_vm_actor_monitor(wo_vm *vm, uint64_t watched, uint64_t observer,
|
||||||
|
uint64_t msg_val, const char **msg);
|
||||||
|
/* iteration 24 T5: arm a one-shot timer on THIS shard — time.after(ms,
|
||||||
|
* addr, msg). ms <= 0 delivers now. */
|
||||||
|
int wo_vm_timer_after(wo_vm *vm, int64_t ms, uint64_t addr, uint64_t msg_val,
|
||||||
|
const char **msg);
|
||||||
|
/* iteration 24 T5: fire every timer at or past `now` (park.c's deadline
|
||||||
|
* machinery calls this beside the fd-park sweep). Returns fired count. */
|
||||||
|
int wo_vm_timers_fire(wo_vm *vm, int64_t now);
|
||||||
|
/* The nearest armed timer's deadline, 0 = none (park.c's tick/timeout). */
|
||||||
|
int64_t wo_vm_timers_next(wo_vm *vm);
|
||||||
|
|
||||||
/* ---- the shard engine (arc stage 2) ------------------------------------
|
/* ---- the shard engine (arc stage 2) ------------------------------------
|
||||||
* One pinned thread per shard, each a full wo_vm (own arena, GC, I/O
|
* One pinned thread per shard, each a full wo_vm (own arena, GC, I/O
|
||||||
|
|
@ -231,7 +270,11 @@ typedef struct wo_envelope {
|
||||||
* from_shard/from_fiber = the parked caller),
|
* from_shard/from_fiber = the parked caller),
|
||||||
* 6 = CALL_REPLY (payload = the SCALAR reply, from_fiber =
|
* 6 = CALL_REPLY (payload = the SCALAR reply, from_fiber =
|
||||||
* the caller to unpark; status 0 = ok, WO_T_ACTOR =
|
* the caller to unpark; status 0 = ok, WO_T_ACTOR =
|
||||||
* the callee was/went dead — the caller traps) */
|
* the callee was/went dead — the caller traps),
|
||||||
|
* 7 = MONITOR (iteration 24 T4: actor = the WATCHED one,
|
||||||
|
* from_fiber REUSED as the observer wo_actor*, payload =
|
||||||
|
* the moved notice — registered on the watched actor's
|
||||||
|
* home thread; already-dead delivers the notice now) */
|
||||||
struct wo_actor *actor;
|
struct wo_actor *actor;
|
||||||
uint64_t payload;
|
uint64_t payload;
|
||||||
uint32_t from_shard;
|
uint32_t from_shard;
|
||||||
|
|
|
||||||
|
|
@ -478,8 +478,17 @@ enum {
|
||||||
* return value arrives. R is a SCALAR (v1,
|
* return value arrives. R is a SCALAR (v1,
|
||||||
* compiler-enforced WO-E226). Dead callee =
|
* compiler-enforced WO-E226). Dead callee =
|
||||||
* WO_T_ACTOR, immediately or mid-call. */
|
* WO_T_ACTOR, immediately or mid-call. */
|
||||||
/* ids 89 (monitor) and 90 (time.after) are RESERVED for the rest of
|
WO_B_MONITOR = 89, /* (watched, observer, msg) -> (): the
|
||||||
* the lifecycle slice — do not reuse. */
|
* observer's own M-typed msg is delivered
|
||||||
|
* when watched dies (trap-death); already
|
||||||
|
* dead delivers NOW; msg MOVES. A full
|
||||||
|
* observer's notice is dropped with a
|
||||||
|
* stderr line (no fiber to trap). */
|
||||||
|
WO_B_TIME_AFTER = 90, /* (ms, addr, msg) -> (): one-shot timer —
|
||||||
|
* msg (MOVED) arrives as an ordinary send
|
||||||
|
* after ms; no cancel (the generation-
|
||||||
|
* counter idiom is the documented answer);
|
||||||
|
* ms <= 0 delivers now. */
|
||||||
/* ---- iteration 35: net seams (sysio.c). Deadlines are per-CALL (no
|
/* ---- iteration 35: net seams (sysio.c). Deadlines are per-CALL (no
|
||||||
* hidden fd state); a timeout is an EXPECTED outcome, so it answers
|
* hidden fd state); a timeout is an EXPECTED outcome, so it answers
|
||||||
* nil/false, never a trap. ms <= 0 = no deadline (the old behavior,
|
* nil/false, never a trap. ms <= 0 = no deadline (the old behavior,
|
||||||
|
|
|
||||||
|
|
@ -117,6 +117,397 @@ static void test_roundtrip_replay(void) {
|
||||||
wo_rt_destroy(&rt);
|
wo_rt_destroy(&rt);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* databasev2 4 part A, Task 1: a failed barrier must be DETECTED, and the
|
||||||
|
* caller must be able to tell WHICH operation failed — a pwrite failure and
|
||||||
|
* an fdatasync failure are different operational problems and the diagnostic
|
||||||
|
* has to name the right one. This proves detection only; the fatal exit that
|
||||||
|
* follows it cannot be exercised in-process. */
|
||||||
|
static void test_commit_failure_detected(void) {
|
||||||
|
char path[128];
|
||||||
|
snprintf(path, sizeof path, "%s/commitfail.wal", g_dir);
|
||||||
|
wo_rt rt;
|
||||||
|
T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0);
|
||||||
|
wo_db db;
|
||||||
|
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
|
||||||
|
wo_wal w;
|
||||||
|
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
|
||||||
|
const char *msg = "";
|
||||||
|
|
||||||
|
/* the WAL remembers where it lives — the abort diagnostic is worthless
|
||||||
|
* without it */
|
||||||
|
T_CHECK(w.path != NULL && strstr(w.path, "commitfail.wal") != NULL);
|
||||||
|
|
||||||
|
wo_str *s = wo_str_new(&rt, "abc", 3);
|
||||||
|
uint64_t vals[2] = {7, (uint64_t)(uintptr_t)s};
|
||||||
|
uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL);
|
||||||
|
T_CHECK(id != 0);
|
||||||
|
T_EQ(wo_wal_append_insert(&w, &db, 0, id), 0);
|
||||||
|
T_CHECK(w.len > 0); /* something really is staged */
|
||||||
|
|
||||||
|
/* an unusable descriptor: pwrite reports EBADF. -1 is used rather than
|
||||||
|
* closing the real fd so the close below cannot double-free it. */
|
||||||
|
int real = w.fd;
|
||||||
|
w.fd = -1;
|
||||||
|
T_EQ(wo_wal_commit(&w), WO_WAL_ERR_WRITE);
|
||||||
|
T_CHECK(w.len > 0); /* a failed commit consumes nothing */
|
||||||
|
w.fd = real;
|
||||||
|
|
||||||
|
wo_wal_close(&w);
|
||||||
|
wo_db_destroy(&db);
|
||||||
|
wo_rt_destroy(&rt);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* databasev2 3 Task 1: compaction rewrites the log as one record per LIVE row.
|
||||||
|
* Asserts BOTH halves on purpose: "the file got shorter" is also true of a
|
||||||
|
* truncating bug, so the replay comparison is what actually proves it. */
|
||||||
|
static void test_compact_shortens_and_replays_equal(void) {
|
||||||
|
char path[128];
|
||||||
|
snprintf(path, sizeof path, "%s/compact.wal", g_dir);
|
||||||
|
wo_rt rt;
|
||||||
|
T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0);
|
||||||
|
wo_db db;
|
||||||
|
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
|
||||||
|
wo_wal w;
|
||||||
|
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
|
||||||
|
const char *msg = "";
|
||||||
|
|
||||||
|
uint64_t ids[3];
|
||||||
|
for (int i = 0; i < 3; i++) {
|
||||||
|
wo_str *s = wo_str_new(&rt, "abc", 3);
|
||||||
|
uint64_t vals[2] = {(uint64_t)(i * 10), (uint64_t)(uintptr_t)s};
|
||||||
|
ids[i] = wo_row_insert(&db, 0, vals, &msg, NULL);
|
||||||
|
T_CHECK(ids[i] != 0);
|
||||||
|
T_EQ(wo_wal_append_insert(&w, &db, 0, ids[i]), 0);
|
||||||
|
T_EQ(wo_wal_commit(&w), 0);
|
||||||
|
}
|
||||||
|
/* age it: the SAME row updated repeatedly, so HISTORY grows while the live
|
||||||
|
* set does not — the exact case checkpoint exists for */
|
||||||
|
for (int k = 0; k < 40; k++) {
|
||||||
|
int ek = 0;
|
||||||
|
T_EQ(wo_row_update_field(&db, 0, ids[0], 0, (uint64_t)(500 + k), &msg, &ek), 0);
|
||||||
|
T_EQ(wo_wal_append_update(&w, &db, 0, ids[0]), 0);
|
||||||
|
T_EQ(wo_wal_commit(&w), 0);
|
||||||
|
}
|
||||||
|
uint64_t before_bytes = 0;
|
||||||
|
int64_t before_recs = wo_wal_check(path, &before_bytes);
|
||||||
|
T_CHECK(before_recs == 43); /* 3 inserts + 40 updates, all history */
|
||||||
|
|
||||||
|
T_EQ(wo_wal_compact(&w, &db), 0);
|
||||||
|
|
||||||
|
uint64_t after_bytes = 0;
|
||||||
|
int64_t after_recs = wo_wal_check(path, &after_bytes);
|
||||||
|
T_CHECK(after_recs == 3); /* one record per LIVE row */
|
||||||
|
T_CHECK(after_bytes < before_bytes); /* and the file really shrank */
|
||||||
|
|
||||||
|
/* the WAL stays usable: the descriptor was reopened and the offset reset,
|
||||||
|
* so a further write must land AFTER the compacted records, not over them */
|
||||||
|
wo_str *s4 = wo_str_new(&rt, "xyz", 3);
|
||||||
|
uint64_t v4[2] = {99, (uint64_t)(uintptr_t)s4};
|
||||||
|
uint64_t id4 = wo_row_insert(&db, 0, v4, &msg, NULL);
|
||||||
|
T_CHECK(id4 != 0);
|
||||||
|
T_EQ(wo_wal_append_insert(&w, &db, 0, id4), 0);
|
||||||
|
T_EQ(wo_wal_commit(&w), 0);
|
||||||
|
T_CHECK(wo_wal_check(path, NULL) == 4);
|
||||||
|
wo_wal_close(&w);
|
||||||
|
|
||||||
|
/* the proof: a FRESH store replayed from the compacted log must hold the
|
||||||
|
* same rows, the same ids, and the LAST value each row had */
|
||||||
|
wo_db db2;
|
||||||
|
T_EQ(wo_db_init(&db2, CLASSES, 1, 0, 1), 0);
|
||||||
|
T_EQ(wo_wal_replay(path, &db2), 4);
|
||||||
|
uint64_t out[2];
|
||||||
|
T_EQ(wo_row_read(&db2, &rt, 0, ids[0], out, &msg), 0);
|
||||||
|
T_CHECK(out[0] == 539); /* the 40th update won, not the original 0 */
|
||||||
|
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
|
||||||
|
T_EQ(wo_row_read(&db2, &rt, 0, ids[1], out, &msg), 0);
|
||||||
|
T_CHECK(out[0] == 10);
|
||||||
|
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
|
||||||
|
T_EQ(wo_row_read(&db2, &rt, 0, ids[2], out, &msg), 0);
|
||||||
|
T_CHECK(out[0] == 20);
|
||||||
|
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
|
||||||
|
T_EQ(wo_row_read(&db2, &rt, 0, id4, out, &msg), 0);
|
||||||
|
T_CHECK(out[0] == 99);
|
||||||
|
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
|
||||||
|
|
||||||
|
wo_db_destroy(&db2);
|
||||||
|
wo_db_destroy(&db);
|
||||||
|
wo_rt_destroy(&rt);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* databasev2 3 Task 2: a stale temp file is the one input that could be
|
||||||
|
* mistaken for data — a crash before the rename leaves one behind, full of
|
||||||
|
* well-formed records that are NOT yet authoritative. So the fixture uses
|
||||||
|
* plausible records (a byte copy of a real log), not garbage: garbage would be
|
||||||
|
* rejected by the CRC anyway and would prove nothing. */
|
||||||
|
static void test_stale_compact_temp_is_removed(void) {
|
||||||
|
char path[128], tmp[160];
|
||||||
|
snprintf(path, sizeof path, "%s/stale.wal", g_dir);
|
||||||
|
snprintf(tmp, sizeof tmp, "%s%s", path, WO_WAL_TMP_SUFFIX);
|
||||||
|
wo_rt rt;
|
||||||
|
T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0);
|
||||||
|
wo_db db;
|
||||||
|
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
|
||||||
|
wo_wal w;
|
||||||
|
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
|
||||||
|
const char *msg = "";
|
||||||
|
|
||||||
|
/* two live rows in the REAL log */
|
||||||
|
uint64_t ids[2];
|
||||||
|
for (int i = 0; i < 2; i++) {
|
||||||
|
wo_str *s = wo_str_new(&rt, "abc", 3);
|
||||||
|
uint64_t vals[2] = {(uint64_t)(i + 1), (uint64_t)(uintptr_t)s};
|
||||||
|
ids[i] = wo_row_insert(&db, 0, vals, &msg, NULL);
|
||||||
|
T_EQ(wo_wal_append_insert(&w, &db, 0, ids[i]), 0);
|
||||||
|
T_EQ(wo_wal_commit(&w), 0);
|
||||||
|
}
|
||||||
|
wo_wal_close(&w);
|
||||||
|
|
||||||
|
/* forge a plausible stale temp: a byte copy of the real log */
|
||||||
|
{
|
||||||
|
int src = open(path, O_RDONLY);
|
||||||
|
int dst = open(tmp, O_WRONLY | O_CREAT | O_TRUNC, 0644);
|
||||||
|
T_CHECK(src >= 0 && dst >= 0);
|
||||||
|
char buf[8192];
|
||||||
|
ssize_t n;
|
||||||
|
while ((n = read(src, buf, sizeof buf)) > 0) T_CHECK(write(dst, buf, (size_t)n) == n);
|
||||||
|
close(src);
|
||||||
|
close(dst);
|
||||||
|
T_EQ(access(tmp, F_OK), 0); /* it really is there before we open */
|
||||||
|
}
|
||||||
|
|
||||||
|
wo_wal w2;
|
||||||
|
T_EQ(wo_wal_open(&w2, path, 1 << 16), 0);
|
||||||
|
T_CHECK(access(tmp, F_OK) != 0); /* gone, and never consulted */
|
||||||
|
wo_wal_close(&w2);
|
||||||
|
|
||||||
|
/* and the live log still says exactly what it said */
|
||||||
|
wo_db db2;
|
||||||
|
T_EQ(wo_db_init(&db2, CLASSES, 1, 0, 1), 0);
|
||||||
|
T_EQ(wo_wal_replay(path, &db2), 2);
|
||||||
|
uint64_t out[2];
|
||||||
|
T_EQ(wo_row_read(&db2, &rt, 0, ids[0], out, &msg), 0);
|
||||||
|
T_CHECK(out[0] == 1);
|
||||||
|
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
|
||||||
|
T_EQ(wo_row_read(&db2, &rt, 0, ids[1], out, &msg), 0);
|
||||||
|
T_CHECK(out[0] == 2);
|
||||||
|
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
|
||||||
|
|
||||||
|
wo_db_destroy(&db2);
|
||||||
|
wo_db_destroy(&db);
|
||||||
|
wo_rt_destroy(&rt);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* databasev2 3 Task 3: the trigger, tested as a pure decision. Kept pure
|
||||||
|
* precisely so it CAN be tested — a policy only observable by writing megabytes
|
||||||
|
* and waiting is a policy nobody checks. */
|
||||||
|
static void test_should_compact_policy(void) {
|
||||||
|
/* below the floor, nothing fires however bad the ratio looks */
|
||||||
|
T_EQ(wo_wal_should_compact(1000, 10, 4096, 3), 0);
|
||||||
|
T_EQ(wo_wal_should_compact(4095, 1, 4096, 3), 0);
|
||||||
|
/* past the floor with no prior compaction: run once to learn the size */
|
||||||
|
T_EQ(wo_wal_should_compact(4096, 0, 4096, 3), 1);
|
||||||
|
/* with a known denominator it is a straight ratio test */
|
||||||
|
T_EQ(wo_wal_should_compact(30000, 10000, 4096, 3), 0); /* exactly 3x is not MORE than 3x */
|
||||||
|
T_EQ(wo_wal_should_compact(30001, 10000, 4096, 3), 1);
|
||||||
|
T_EQ(wo_wal_should_compact(19999, 10000, 4096, 2), 0);
|
||||||
|
T_EQ(wo_wal_should_compact(20001, 10000, 4096, 2), 1);
|
||||||
|
/* a zero ratio disables the policy rather than dividing by nothing */
|
||||||
|
T_EQ(wo_wal_should_compact(1u << 30, 10, 4096, 0), 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* databasev2 3 Task 3: the ordering rule, asserted rather than trusted.
|
||||||
|
* Compaction with records staged would write them into a file about to be
|
||||||
|
* replaced, so it must be REFUSED — and refused without touching the log. */
|
||||||
|
static void test_compact_refuses_with_staged_records(void) {
|
||||||
|
char path[128];
|
||||||
|
snprintf(path, sizeof path, "%s/staged.wal", g_dir);
|
||||||
|
wo_rt rt;
|
||||||
|
T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0);
|
||||||
|
wo_db db;
|
||||||
|
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
|
||||||
|
wo_wal w;
|
||||||
|
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
|
||||||
|
const char *msg = "";
|
||||||
|
|
||||||
|
wo_str *s1 = wo_str_new(&rt, "abc", 3);
|
||||||
|
uint64_t v1[2] = {7, (uint64_t)(uintptr_t)s1};
|
||||||
|
uint64_t id1 = wo_row_insert(&db, 0, v1, &msg, NULL);
|
||||||
|
T_EQ(wo_wal_append_insert(&w, &db, 0, id1), 0);
|
||||||
|
T_EQ(wo_wal_commit(&w), 0); /* durable, buffer empty */
|
||||||
|
|
||||||
|
/* now stage WITHOUT committing */
|
||||||
|
wo_str *s2 = wo_str_new(&rt, "xyz", 3);
|
||||||
|
uint64_t v2[2] = {8, (uint64_t)(uintptr_t)s2};
|
||||||
|
uint64_t id2 = wo_row_insert(&db, 0, v2, &msg, NULL);
|
||||||
|
T_EQ(wo_wal_append_insert(&w, &db, 0, id2), 0);
|
||||||
|
T_CHECK(w.len > 0);
|
||||||
|
|
||||||
|
uint64_t before = 0;
|
||||||
|
int64_t recs = wo_wal_check(path, &before);
|
||||||
|
T_EQ(wo_wal_compact(&w, &db), -1); /* refused */
|
||||||
|
T_CHECK(w.len > 0); /* and the staged record is still there */
|
||||||
|
uint64_t after = 0;
|
||||||
|
T_CHECK(wo_wal_check(path, &after) == recs && after == before); /* log untouched */
|
||||||
|
|
||||||
|
/* the staged record still commits normally afterwards */
|
||||||
|
T_EQ(wo_wal_commit(&w), 0);
|
||||||
|
T_CHECK(wo_wal_check(path, NULL) == recs + 1);
|
||||||
|
|
||||||
|
wo_wal_close(&w);
|
||||||
|
wo_db_destroy(&db);
|
||||||
|
wo_rt_destroy(&rt);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* databasev2 3 Task 4: kill -9 DURING compaction.
|
||||||
|
*
|
||||||
|
* The existing battery is insert-only, so its "records >= acks" oracle is
|
||||||
|
* exactly what compaction is allowed to break: collapsing history is the point.
|
||||||
|
* The invariant that survives is the ACKED LIVE SET — every id acked as
|
||||||
|
* inserted and not later acked as deleted must be present with its acked value,
|
||||||
|
* and every id acked as deleted must be absent. Both the pre-compaction and the
|
||||||
|
* post-compaction log satisfy that identically, which is precisely the
|
||||||
|
* "never a mixture" property the design is shaped around.
|
||||||
|
*
|
||||||
|
* The child deletes as it goes so HISTORY accumulates while the live set stays
|
||||||
|
* small — without that, compaction would have nothing to collapse and the test
|
||||||
|
* would prove nothing. */
|
||||||
|
#define CK_DELETED UINT64_MAX
|
||||||
|
|
||||||
|
static void ck_ack(int fd, uint64_t id, uint64_t val) {
|
||||||
|
uint64_t rec[2] = {id, val};
|
||||||
|
if (write(fd, rec, sizeof rec) != (ssize_t)sizeof rec) _exit(0); /* parent gone */
|
||||||
|
}
|
||||||
|
|
||||||
|
static void compact_battery_child(const char *path, int ack_fd) {
|
||||||
|
wo_rt rt;
|
||||||
|
wo_db db;
|
||||||
|
wo_wal w;
|
||||||
|
if (wo_rt_init(&rt, 1 << 20, CLASSES, 1) != 0) _exit(9);
|
||||||
|
if (wo_db_init(&db, CLASSES, 1, 0, 1) != 0) _exit(9);
|
||||||
|
if (wo_wal_open(&w, path, 1 << 20) != 0) _exit(9);
|
||||||
|
const char *msg = "";
|
||||||
|
uint64_t live[512];
|
||||||
|
size_t nlive = 0;
|
||||||
|
for (uint64_t i = 0;; i++) {
|
||||||
|
uint64_t val = i * 7 + 3;
|
||||||
|
wo_str *s = wo_str_new(&rt, "r", 1);
|
||||||
|
uint64_t vals[2] = {val, (uint64_t)(uintptr_t)s};
|
||||||
|
uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL);
|
||||||
|
wo_str_free(&rt, s);
|
||||||
|
if (!id) _exit(9);
|
||||||
|
if (wo_wal_append_insert(&w, &db, 0, id) != 0) _exit(9);
|
||||||
|
if (wo_wal_commit(&w) != 0) _exit(9); /* durable BEFORE the ack */
|
||||||
|
ck_ack(ack_fd, id, val);
|
||||||
|
if (nlive < 512) live[nlive++] = id;
|
||||||
|
|
||||||
|
/* drop the oldest so history grows while the live set does not */
|
||||||
|
if (nlive > 16) {
|
||||||
|
uint64_t victim = live[0];
|
||||||
|
memmove(live, live + 1, (nlive - 1) * sizeof live[0]);
|
||||||
|
nlive--;
|
||||||
|
/* INTENT FIRST, deliberately. An ack after the commit would race:
|
||||||
|
* a kill between them leaves the row legitimately gone on disk
|
||||||
|
* while the last ack still says "inserted", and the parent would
|
||||||
|
* demand a row the engine was right to remove. Announcing intent
|
||||||
|
* makes the row's fate simply UNKNOWN to the parent, which is the
|
||||||
|
* honest thing to assert about it. */
|
||||||
|
ck_ack(ack_fd, victim, CK_DELETED);
|
||||||
|
if (wo_row_remove(&db, 0, victim) != 0) _exit(9);
|
||||||
|
if (wo_wal_append_remove(&w, 0, victim) != 0) _exit(9);
|
||||||
|
if (wo_wal_commit(&w) != 0) _exit(9);
|
||||||
|
}
|
||||||
|
/* compact often, so a kill has a real chance of landing inside one */
|
||||||
|
if (i % 24 == 23) (void)wo_wal_compact(&w, &db);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static void test_compact_crash_battery(void) {
|
||||||
|
int rounds = 40; /* it is a RACE: one green run proves very little */
|
||||||
|
for (int round = 0; round < rounds; round++) {
|
||||||
|
char path[128], tmp[160];
|
||||||
|
snprintf(path, sizeof path, "%s/ckcrash-%d.wal", g_dir, round);
|
||||||
|
snprintf(tmp, sizeof tmp, "%s%s", path, WO_WAL_TMP_SUFFIX);
|
||||||
|
int pipefd[2];
|
||||||
|
T_EQ(pipe(pipefd), 0);
|
||||||
|
pid_t pid = fork();
|
||||||
|
T_CHECK(pid >= 0);
|
||||||
|
if (pid == 0) {
|
||||||
|
close(pipefd[0]);
|
||||||
|
compact_battery_child(path, pipefd[1]);
|
||||||
|
_exit(0);
|
||||||
|
}
|
||||||
|
close(pipefd[1]);
|
||||||
|
/* vary the instant so kills land before, inside and after rewrites */
|
||||||
|
struct timespec ts = {0, (7 + round * 3) * 1000000L};
|
||||||
|
while (nanosleep(&ts, &ts) != 0) {}
|
||||||
|
kill(pid, SIGKILL);
|
||||||
|
int status;
|
||||||
|
waitpid(pid, &status, 0);
|
||||||
|
|
||||||
|
/* replay the acks into the expected live set, in order */
|
||||||
|
uint64_t ids[65536], vals[65536];
|
||||||
|
size_t n = 0;
|
||||||
|
for (;;) {
|
||||||
|
uint64_t rec[2];
|
||||||
|
ssize_t r = read(pipefd[0], rec, sizeof rec);
|
||||||
|
if (r != (ssize_t)sizeof rec) break;
|
||||||
|
if (n < 65536) { ids[n] = rec[0]; vals[n] = rec[1]; n++; }
|
||||||
|
}
|
||||||
|
close(pipefd[0]);
|
||||||
|
T_CHECK(n > 0); /* the child got at least one commit out */
|
||||||
|
|
||||||
|
wo_rt rt;
|
||||||
|
T_EQ(wo_rt_init(&rt, 1 << 22, CLASSES, 1), 0);
|
||||||
|
wo_db db;
|
||||||
|
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
|
||||||
|
int64_t ck_recs = wo_wal_check(path, NULL);
|
||||||
|
int64_t ck_applied = wo_wal_replay(path, &db);
|
||||||
|
T_CHECK(ck_applied >= 0); /* never reported as corruption */
|
||||||
|
|
||||||
|
/* A stale temp may well EXIST after a kill inside compaction — that is
|
||||||
|
* the expected debris. The guarantee is that the next OPEN removes it
|
||||||
|
* and never reads it, so that is what gets asserted here; checking
|
||||||
|
* merely for its absence after a replay would be asserting something
|
||||||
|
* the design never promised (wo_wal_replay does not open the WAL). */
|
||||||
|
{
|
||||||
|
wo_wal probe;
|
||||||
|
T_EQ(wo_wal_open(&probe, path, 1 << 20), 0);
|
||||||
|
T_CHECK(access(tmp, F_OK) != 0);
|
||||||
|
wo_wal_close(&probe);
|
||||||
|
}
|
||||||
|
|
||||||
|
const char *msg = "";
|
||||||
|
int bad = 0, checked = 0;
|
||||||
|
for (size_t k = 0; k < n && !bad; k++) {
|
||||||
|
if (vals[k] == CK_DELETED) continue; /* intent: fate is unknown */
|
||||||
|
/* an id ever announced for deletion may legally be gone */
|
||||||
|
int doomed = 0;
|
||||||
|
for (size_t j = 0; j < n; j++)
|
||||||
|
if (ids[j] == ids[k] && vals[j] == CK_DELETED) { doomed = 1; break; }
|
||||||
|
if (doomed) continue;
|
||||||
|
uint64_t out[2];
|
||||||
|
int rc = wo_row_read(&db, &rt, 0, ids[k], out, &msg);
|
||||||
|
if (0) {
|
||||||
|
} else if (rc != 0 || out[0] != vals[k]) {
|
||||||
|
bad = 1; /* an acked insert is missing or wrong */
|
||||||
|
fprintf(stderr, "CKDIAG round=%d id=%llu rc=%d got=%llu want=%llu ack#%zu/%zu "
|
||||||
|
"log_records=%lld replay_applied=%lld\n",
|
||||||
|
round, (unsigned long long)ids[k], rc,
|
||||||
|
rc == 0 ? (unsigned long long)out[0] : 0ull,
|
||||||
|
(unsigned long long)vals[k], k, n,
|
||||||
|
(long long)ck_recs, (long long)ck_applied);
|
||||||
|
} else {
|
||||||
|
wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]);
|
||||||
|
}
|
||||||
|
checked++;
|
||||||
|
}
|
||||||
|
T_CHECK(checked > 0);
|
||||||
|
T_CHECK(!bad);
|
||||||
|
wo_db_destroy(&db);
|
||||||
|
wo_rt_destroy(&rt);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
static void test_torn_tail(void) {
|
static void test_torn_tail(void) {
|
||||||
char path[128];
|
char path[128];
|
||||||
snprintf(path, sizeof path, "%s/torn.wal", g_dir);
|
snprintf(path, sizeof path, "%s/torn.wal", g_dir);
|
||||||
|
|
@ -534,12 +925,18 @@ int main(void) {
|
||||||
snprintf(g_dir, sizeof g_dir, "/tmp/wo-wal-test-XXXXXX");
|
snprintf(g_dir, sizeof g_dir, "/tmp/wo-wal-test-XXXXXX");
|
||||||
if (!mkdtemp(g_dir)) return 1;
|
if (!mkdtemp(g_dir)) return 1;
|
||||||
test_roundtrip_replay();
|
test_roundtrip_replay();
|
||||||
|
test_commit_failure_detected();
|
||||||
|
test_compact_shortens_and_replays_equal();
|
||||||
|
test_stale_compact_temp_is_removed();
|
||||||
|
test_should_compact_policy();
|
||||||
|
test_compact_refuses_with_staged_records();
|
||||||
test_torn_tail();
|
test_torn_tail();
|
||||||
test_float_bytes_replay();
|
test_float_bytes_replay();
|
||||||
test_offset_capture();
|
test_offset_capture();
|
||||||
test_offset_after_failed_commit();
|
test_offset_after_failed_commit();
|
||||||
test_read_row_at();
|
test_read_row_at();
|
||||||
test_crash_battery();
|
test_crash_battery();
|
||||||
|
test_compact_crash_battery();
|
||||||
/* leave the dir for a failed run's forensics only */
|
/* leave the dir for a failed run's forensics only */
|
||||||
if (!t_fail) {
|
if (!t_fail) {
|
||||||
char cmd[128];
|
char cmd[128];
|
||||||
|
|
|
||||||
433
scripts/chat-accept.sh
Executable file
433
scripts/chat-accept.sh
Executable file
|
|
@ -0,0 +1,433 @@
|
||||||
|
#!/usr/bin/env bash
|
||||||
|
# scripts/chat-accept.sh — iteration 24's gate. The chat sample serves
|
||||||
|
# WebSocket rooms through the framework ([deps], file:// remote); a raw
|
||||||
|
# RFC 6455 python client (stdlib only, INDEPENDENT accept-key check)
|
||||||
|
# proves: the handshake, broadcast + presence + isolation across rooms,
|
||||||
|
# the 1k-clients-one-hot-room soak (fds/RSS accounted), and the SIGTERM
|
||||||
|
# drain (close frames, exit 0) — functional legs on BOTH WO_IO backends
|
||||||
|
# plus an ASan run.
|
||||||
|
set -uo pipefail
|
||||||
|
|
||||||
|
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||||
|
WOC="$ROOT/compiler/_build/default/bin/woc"
|
||||||
|
WOVM="$ROOT/runtime/wovm"
|
||||||
|
ASAN="$ROOT/runtime/build/wovm_asan"
|
||||||
|
PORT0="${CHAT_PORT:-18901}"
|
||||||
|
PORT="$PORT0"
|
||||||
|
SOAK_N="${CHAT_SOAK:-1000}"
|
||||||
|
|
||||||
|
pass=0; fail=0
|
||||||
|
ok() { echo "ok $1"; pass=$((pass + 1)); }
|
||||||
|
bad() { echo "FAIL $1 -- $2"; fail=$((fail + 1)); }
|
||||||
|
|
||||||
|
if [[ ! -x "$WOC" || ! -x "$WOVM" ]]; then
|
||||||
|
echo "chat-accept: build woc and wovm first" >&2; exit 1
|
||||||
|
fi
|
||||||
|
ulimit -n 8192 2>/dev/null || true
|
||||||
|
|
||||||
|
W="$(mktemp -d "${TMPDIR:-/tmp}/chat-accept.XXXXXX")"
|
||||||
|
SRV=""
|
||||||
|
# The example's server log lives at a STABLE path so a developer can
|
||||||
|
# `tail -F /tmp/chat.log` while this runs. It used to go to the per-run temp
|
||||||
|
# dir, which cleanup() deletes on exit — so there was nothing left to read and
|
||||||
|
# nothing to follow live. Truncated once here, then APPENDED by every leg with
|
||||||
|
# a banner, so one file holds the whole run in order.
|
||||||
|
SRVLOG="/tmp/chat.log"
|
||||||
|
: > "$SRVLOG"
|
||||||
|
LEG=0
|
||||||
|
LEGFROM=1
|
||||||
|
echo "server log: $SRVLOG (tail -F \"$SRVLOG\" to follow)"
|
||||||
|
|
||||||
|
cleanup() {
|
||||||
|
# kill EVERY server this run started, not merely the most recent $SRV: a leg
|
||||||
|
# that dies before clearing SRV used to orphan a listener, which then broke
|
||||||
|
# the next run on the same port. $W is unique per run, so matching on it
|
||||||
|
# cannot touch another run's processes.
|
||||||
|
[[ -n "$SRV" ]] && kill -9 "$SRV" 2>/dev/null
|
||||||
|
pkill -9 -f "$W/app/target/chat" 2>/dev/null
|
||||||
|
rm -rf "$W"
|
||||||
|
}
|
||||||
|
trap cleanup EXIT
|
||||||
|
|
||||||
|
cp -r "$ROOT/docs/examples/porch" "$W/fw"
|
||||||
|
git -C "$W/fw" init -q && git -C "$W/fw" add -A
|
||||||
|
git -C "$W/fw" -c user.email=t@t -c user.name=t commit -qm v01 && git -C "$W/fw" tag v0.1.0
|
||||||
|
cp -r "$ROOT/docs/examples/chat" "$W/app"
|
||||||
|
sed -i "s|https://github.com/shoneyj/porch|file://$W/fw|" "$W/app/wo.toml"
|
||||||
|
printf '[build]\nruntime = "%s"\n' "$WOVM" >> "$W/app/wo.toml"
|
||||||
|
|
||||||
|
if "$WOC" "$W/app" >"$W/build.out" 2>&1 && [[ -x "$W/app/target/chat" ]]; then
|
||||||
|
ok "deps chain + build"
|
||||||
|
else
|
||||||
|
bad "build" "$(grep -m1 error "$W/build.out" || head -1 "$W/build.out")"
|
||||||
|
echo "chat-accept: 1 checks, 1 failures"; exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
# the raw client, shared by every leg
|
||||||
|
CLIENT="$W/wsc.py"
|
||||||
|
cat > "$CLIENT" <<'PYEOF'
|
||||||
|
import socket, base64, hashlib, os, time
|
||||||
|
GUID = "258EAFA5-E914-47DA-95CA-C5AB0DC85B11"
|
||||||
|
BUF = {}
|
||||||
|
def connect(port, room, name, timeout=8, rcvbuf=None):
|
||||||
|
# rcvbuf: shrink THIS client's receive buffer so the server's socket fills
|
||||||
|
# quickly — how the WO_MAILBOX leg manufactures a genuinely slow member
|
||||||
|
# without sleeping. Must be set before connect() to take effect.
|
||||||
|
if rcvbuf is None:
|
||||||
|
s = socket.create_connection(("127.0.0.1", port), timeout=timeout)
|
||||||
|
else:
|
||||||
|
s = socket.socket()
|
||||||
|
s.setsockopt(socket.SOL_SOCKET, socket.SO_RCVBUF, rcvbuf)
|
||||||
|
s.settimeout(timeout)
|
||||||
|
s.connect(("127.0.0.1", port))
|
||||||
|
key = base64.b64encode(os.urandom(16)).decode()
|
||||||
|
s.sendall((f"GET /ws?room={room}&name={name} HTTP/1.1\r\nhost: a\r\n"
|
||||||
|
f"upgrade: websocket\r\nconnection: Upgrade\r\n"
|
||||||
|
f"sec-websocket-key: {key}\r\nsec-websocket-version: 13\r\n\r\n").encode())
|
||||||
|
d = b""
|
||||||
|
while b"\r\n\r\n" not in d: d += s.recv(2000)
|
||||||
|
head, _, rest = d.partition(b"\r\n\r\n")
|
||||||
|
BUF[s] = rest # a frame may already ride the same segment
|
||||||
|
head = head.decode()
|
||||||
|
assert " 101 " in head.splitlines()[0], head.splitlines()[0]
|
||||||
|
want = base64.b64encode(hashlib.sha1((key + GUID).encode()).digest()).decode()
|
||||||
|
assert want in head, "accept-key mismatch (independent check)"
|
||||||
|
return s
|
||||||
|
def _take(s, n, timeout):
|
||||||
|
s.settimeout(timeout)
|
||||||
|
b = BUF.get(s, b"")
|
||||||
|
while len(b) < n:
|
||||||
|
c = s.recv(4096)
|
||||||
|
if not c:
|
||||||
|
BUF[s] = b
|
||||||
|
return None
|
||||||
|
b += c
|
||||||
|
BUF[s] = b[n:]
|
||||||
|
return b[:n]
|
||||||
|
def send(s, text):
|
||||||
|
p = text.encode(); mask = os.urandom(4)
|
||||||
|
if len(p) < 126: hdr = bytes([0x81, 0x80 | len(p)])
|
||||||
|
else: hdr = bytes([0x81, 0x80 | 126, len(p) >> 8, len(p) & 255])
|
||||||
|
s.sendall(hdr + mask + bytes(b ^ mask[i % 4] for i, b in enumerate(p)))
|
||||||
|
def recv(s, timeout=5):
|
||||||
|
h = _take(s, 2, timeout)
|
||||||
|
if h is None: return (-2, "") # EOF
|
||||||
|
b0, b1 = h[0], h[1]
|
||||||
|
ln = b1 & 0x7F
|
||||||
|
if ln == 126:
|
||||||
|
e = _take(s, 2, timeout); ln = (e[0] << 8) | e[1]
|
||||||
|
d = _take(s, ln, timeout) if ln else b""
|
||||||
|
return (b0 & 0x0F), (d or b"").decode(errors="replace")
|
||||||
|
PYEOF
|
||||||
|
|
||||||
|
serve() { # serve PORT [env...] — start + wait for THIS server's listener line
|
||||||
|
PORT="$1"; shift
|
||||||
|
LEG=$((LEG + 1))
|
||||||
|
printf '\n===== leg %d — port %s — %s =====\n' "$LEG" "$PORT" "${*:-default env}" >>"$SRVLOG"
|
||||||
|
# readiness is searched only in THIS leg's slice: the log is appended, never
|
||||||
|
# truncated, so a 'listening' line from an earlier leg would lie
|
||||||
|
LEGFROM=$(( $(wc -l < "$SRVLOG") + 1 ))
|
||||||
|
"$@" "$W/app/target/chat" "$PORT" >>"$SRVLOG" 2>&1 &
|
||||||
|
SRV=$!
|
||||||
|
for _ in $(seq 1 80); do
|
||||||
|
tail -n "+$LEGFROM" "$SRVLOG" 2>/dev/null | grep -q listening && return 0
|
||||||
|
sleep 0.1
|
||||||
|
done
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
functional() { # $1 = leg name
|
||||||
|
timeout 30 python3 - "$PORT" <<'PYEOF'
|
||||||
|
import sys; sys.path.insert(0, sys.argv[0].rsplit("/",1)[0])
|
||||||
|
port = int(sys.argv[1])
|
||||||
|
import importlib.util, os
|
||||||
|
spec = importlib.util.spec_from_file_location("wsc", os.environ["WSC"])
|
||||||
|
wsc = importlib.util.module_from_spec(spec); spec.loader.exec_module(wsc)
|
||||||
|
a = wsc.connect(port, "lobby", "alice")
|
||||||
|
assert wsc.recv(a) == (1, "* alice joined")
|
||||||
|
b = wsc.connect(port, "lobby", "bob")
|
||||||
|
assert wsc.recv(a) == (1, "* bob joined")
|
||||||
|
assert wsc.recv(b) == (1, "* bob joined")
|
||||||
|
c = wsc.connect(port, "other", "carol")
|
||||||
|
assert wsc.recv(c) == (1, "* carol joined")
|
||||||
|
wsc.send(a, "hello room")
|
||||||
|
assert wsc.recv(a) == (1, "alice: hello room")
|
||||||
|
assert wsc.recv(b) == (1, "alice: hello room")
|
||||||
|
import socket
|
||||||
|
try:
|
||||||
|
k, t = wsc.recv(c, timeout=0.8); assert False, f"leak into other room: {t}"
|
||||||
|
except socket.timeout: pass
|
||||||
|
b.close()
|
||||||
|
k, t = wsc.recv(a)
|
||||||
|
assert (k, t) == (1, "* bob left"), (k, t)
|
||||||
|
a.close(); c.close()
|
||||||
|
print("functional-ok")
|
||||||
|
PYEOF
|
||||||
|
}
|
||||||
|
|
||||||
|
# ---- 2. functional on both backends ----
|
||||||
|
export WSC="$CLIENT"
|
||||||
|
serve "$((PORT0 + 0))" env WO_IO=uring || bad "serve-uring" "no listener"
|
||||||
|
r="$(functional uring)"; [[ "$r" == *functional-ok* ]] \
|
||||||
|
&& ok "uring: handshake(key verified) + presence + broadcast + isolation + leave" \
|
||||||
|
|| bad "uring-functional" "$r"
|
||||||
|
kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV=""
|
||||||
|
|
||||||
|
serve "$((PORT0 + 1))" env WO_IO=epoll || bad "serve-epoll" "no listener"
|
||||||
|
r="$(functional epoll)"; [[ "$r" == *functional-ok* ]] \
|
||||||
|
&& ok "epoll: the same matrix" || bad "epoll-functional" "$r"
|
||||||
|
kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV=""
|
||||||
|
|
||||||
|
# ---- 3. the soak: N clients, ONE hot room ----
|
||||||
|
serve "$((PORT0 + 2))" || bad "serve-soak" "no listener"
|
||||||
|
fds_before="$(ls /proc/$SRV/fd 2>/dev/null | wc -l)"
|
||||||
|
fds_prev=99999
|
||||||
|
r="$(timeout 180 python3 - "$PORT" "$SOAK_N" <<'PYEOF'
|
||||||
|
import asyncio, sys, os, time, base64, hashlib
|
||||||
|
port, N = int(sys.argv[1]), int(sys.argv[2])
|
||||||
|
GUID = "258EAFA5-E914-47DA-95CA-C5AB0DC85B11"
|
||||||
|
MARK = "the-hot-room-marker"
|
||||||
|
sem = asyncio.Semaphore(100)
|
||||||
|
async def client(i, results):
|
||||||
|
async with sem:
|
||||||
|
r, w = await asyncio.open_connection("127.0.0.1", port)
|
||||||
|
key = base64.b64encode(os.urandom(16)).decode()
|
||||||
|
w.write((f"GET /ws?room=hot&name=c{i} HTTP/1.1\r\nhost: a\r\n"
|
||||||
|
f"upgrade: websocket\r\nconnection: Upgrade\r\n"
|
||||||
|
f"sec-websocket-key: {key}\r\nsec-websocket-version: 13\r\n\r\n").encode())
|
||||||
|
await w.drain()
|
||||||
|
d = b""
|
||||||
|
while b"\r\n\r\n" not in d: d += await r.read(2000)
|
||||||
|
if i == 0:
|
||||||
|
# the sender: wait for the herd, then one marker line
|
||||||
|
await asyncio.sleep(0)
|
||||||
|
results["sender_ready"].set()
|
||||||
|
try:
|
||||||
|
buf = b""
|
||||||
|
deadline = time.time() + 150
|
||||||
|
while time.time() < deadline:
|
||||||
|
try:
|
||||||
|
c = await asyncio.wait_for(r.read(8192), timeout=5)
|
||||||
|
except asyncio.TimeoutError:
|
||||||
|
if results["sent"].is_set(): break
|
||||||
|
continue
|
||||||
|
if not c: break
|
||||||
|
buf += c
|
||||||
|
# scan frames for the marker (server frames are unmasked, small)
|
||||||
|
if MARK.encode() in buf:
|
||||||
|
results["got"] += 1
|
||||||
|
return
|
||||||
|
finally:
|
||||||
|
w.close()
|
||||||
|
async def main():
|
||||||
|
results = {"got": 0, "sender_ready": asyncio.Event(), "sent": asyncio.Event()}
|
||||||
|
conns = []
|
||||||
|
# keep the sender's socket outside the tasks: join first
|
||||||
|
sr, sw = None, None
|
||||||
|
async def sender():
|
||||||
|
nonlocal sr, sw
|
||||||
|
async with sem:
|
||||||
|
sr, sw = await asyncio.open_connection("127.0.0.1", port)
|
||||||
|
key = base64.b64encode(os.urandom(16)).decode()
|
||||||
|
sw.write((f"GET /ws?room=hot&name=sender HTTP/1.1\r\nhost: a\r\n"
|
||||||
|
f"upgrade: websocket\r\nconnection: Upgrade\r\n"
|
||||||
|
f"sec-websocket-key: {key}\r\nsec-websocket-version: 13\r\n\r\n").encode())
|
||||||
|
await sw.drain()
|
||||||
|
d = b""
|
||||||
|
while b"\r\n\r\n" not in d: d += await sr.read(2000)
|
||||||
|
await sender()
|
||||||
|
tasks = [asyncio.create_task(client(i, results)) for i in range(N)]
|
||||||
|
await asyncio.sleep(max(2.0, N / 250)) # let the herd join + drain presence
|
||||||
|
p = MARK.encode(); mask = os.urandom(4)
|
||||||
|
hdr = bytes([0x81, 0x80 | len(p)])
|
||||||
|
sw.write(hdr + mask + bytes(b ^ mask[i % 4] for i, b in enumerate(p)))
|
||||||
|
await sw.drain()
|
||||||
|
results["sent"].set()
|
||||||
|
t0 = time.time()
|
||||||
|
await asyncio.gather(*tasks, return_exceptions=True)
|
||||||
|
el = int((time.time() - t0) * 1000)
|
||||||
|
sw.close()
|
||||||
|
print(f"{results['got']}|{N}|{el}")
|
||||||
|
asyncio.run(main())
|
||||||
|
PYEOF
|
||||||
|
)"
|
||||||
|
got="${r%%|*}"; rest="${r#*|}"; n="${rest%%|*}"; el="${rest#*|}"
|
||||||
|
[[ "$got" == "$n" ]] \
|
||||||
|
&& ok "soak: the marker reached all $got/$n hot-room clients (${el}ms after send)" \
|
||||||
|
|| bad "soak" "$r"
|
||||||
|
# leave-broadcast storms take a moment to settle after 1k closes
|
||||||
|
for _ in $(seq 1 20); do
|
||||||
|
fds_w1="$(ls /proc/$SRV/fd 2>/dev/null | wc -l)"
|
||||||
|
[[ "$fds_w1" -le "$fds_prev" ]] && break
|
||||||
|
fds_prev="$fds_w1"
|
||||||
|
sleep 0.5
|
||||||
|
done
|
||||||
|
# The fd check is for a per-CONNECTION leak, and a fixed tolerance cannot
|
||||||
|
# express that. Shards initialise LAZILY (runtime/src/vm.c: a worker's vm is
|
||||||
|
# not paid for until its first fiber arrives), so the first wave legitimately
|
||||||
|
# adds one io_uring + one eventfd PER SHARD, capped at nproc — on a 20-core
|
||||||
|
# box that is +18, which the old `fds_before + 8` read as a leak. Measured
|
||||||
|
# 2026-08-27: 26 -> 44 after 20 clients, then still 44 after 40 more.
|
||||||
|
#
|
||||||
|
# So assert the invariant itself: a SECOND wave must not raise the count.
|
||||||
|
# Core-count independent, and it catches a slow leak that any fixed
|
||||||
|
# tolerance would hide inside its own slack.
|
||||||
|
timeout 60 python3 - "$PORT" 20 <<'PYEOF' >/dev/null 2>&1
|
||||||
|
import socket, base64, os, sys, time
|
||||||
|
port, n = int(sys.argv[1]), int(sys.argv[2])
|
||||||
|
socks = []
|
||||||
|
for i in range(n):
|
||||||
|
s = socket.create_connection(("127.0.0.1", port), timeout=8)
|
||||||
|
k = base64.b64encode(os.urandom(16)).decode()
|
||||||
|
s.sendall((f"GET /ws?room=fdwave&name=w{i} HTTP/1.1\r\nhost: a\r\n"
|
||||||
|
f"upgrade: websocket\r\nconnection: Upgrade\r\n"
|
||||||
|
f"sec-websocket-key: {k}\r\nsec-websocket-version: 13\r\n\r\n").encode())
|
||||||
|
h = b""
|
||||||
|
while b"\r\n\r\n" not in h:
|
||||||
|
h += s.recv(4096)
|
||||||
|
socks.append(s)
|
||||||
|
time.sleep(0.5)
|
||||||
|
for s in socks:
|
||||||
|
s.close()
|
||||||
|
PYEOF
|
||||||
|
fds_after="$fds_w1"
|
||||||
|
for _ in $(seq 1 20); do
|
||||||
|
fds_after="$(ls /proc/$SRV/fd 2>/dev/null | wc -l)"
|
||||||
|
[[ "$fds_after" -le "$fds_w1" ]] && break
|
||||||
|
sleep 0.5
|
||||||
|
done
|
||||||
|
rss_kb="$(awk '/VmRSS/{print $2}' /proc/$SRV/status 2>/dev/null)"
|
||||||
|
[[ "$fds_after" -le "$fds_w1" ]] \
|
||||||
|
&& ok "no per-connection fd leak (start $fds_before, after $SOAK_N: $fds_w1, after 20 more: $fds_after)" \
|
||||||
|
|| bad "soak-fds" "second wave grew fds: $fds_w1 -> $fds_after (start $fds_before)"
|
||||||
|
[[ -n "$rss_kb" && "$rss_kb" -lt 819200 ]] \
|
||||||
|
&& ok "soak RSS bounded (${rss_kb}KB < 800MB)" || bad "soak-rss" "${rss_kb}KB"
|
||||||
|
|
||||||
|
# ---- 4. drain: SIGTERM with clients connected -> close frames, exit 0 ----
|
||||||
|
# Starts its OWN server. It used to inherit the soak leg's $SRV, which meant
|
||||||
|
# any leg inserted between them silently handed drain an empty pid: its python
|
||||||
|
# died on int(""), the leg reported a bare failure, AND the soak server was
|
||||||
|
# never killed — orphaning a listener that then broke the NEXT run's soak on
|
||||||
|
# the same port. No leg may depend on another leg's server.
|
||||||
|
serve "$((PORT0 + 6))" || bad "serve-drain" "no listener"
|
||||||
|
r="$(timeout 30 python3 - "$PORT" "$SRV" <<'PYEOF'
|
||||||
|
import sys, os, time, signal, socket
|
||||||
|
import importlib.util
|
||||||
|
spec = importlib.util.spec_from_file_location("wsc", os.environ["WSC"])
|
||||||
|
wsc = importlib.util.module_from_spec(spec); spec.loader.exec_module(wsc)
|
||||||
|
port, srv = int(sys.argv[1]), int(sys.argv[2])
|
||||||
|
a = wsc.connect(port, "lobby", "alice"); wsc.recv(a)
|
||||||
|
b = wsc.connect(port, "lobby", "bob"); wsc.recv(a); wsc.recv(b)
|
||||||
|
os.kill(srv, signal.SIGTERM)
|
||||||
|
def drained(s):
|
||||||
|
try:
|
||||||
|
while True:
|
||||||
|
k, _ = wsc.recv(s, timeout=5)
|
||||||
|
if k == 8: return "close-frame"
|
||||||
|
if k == -2: return "eof"
|
||||||
|
except socket.timeout:
|
||||||
|
return "stuck"
|
||||||
|
except (ConnectionResetError, BrokenPipeError):
|
||||||
|
return "reset"
|
||||||
|
print(drained(a) + "|" + drained(b))
|
||||||
|
PYEOF
|
||||||
|
)"
|
||||||
|
[[ "$r" == "close-frame|close-frame" ]] \
|
||||||
|
&& ok "drain: both clients got the close frame" || bad "drain" "$r"
|
||||||
|
stopped=1
|
||||||
|
for _ in $(seq 1 40); do kill -0 "$SRV" 2>/dev/null || { stopped=0; break; }; sleep 0.1; done
|
||||||
|
[[ $stopped -eq 0 ]] && ok "SIGTERM exits 0" || bad "stop" "still running"
|
||||||
|
SRV=""
|
||||||
|
|
||||||
|
# ---- 4b. WO_SHARDS=1: the same matrix on one shard ----
|
||||||
|
# The plan requires `just chat` green at default cores AND on a single shard:
|
||||||
|
# cross-shard placement is where the actor work can hide a bug, so the
|
||||||
|
# one-shard run is the control that says a failure is placement's fault.
|
||||||
|
serve "$((PORT0 + 4))" env WO_SHARDS=1 || bad "serve-shards1" "no listener"
|
||||||
|
r="$(functional shards1)"; [[ "$r" == *functional-ok* ]] \
|
||||||
|
&& ok "WO_SHARDS=1: the same matrix on a single shard" \
|
||||||
|
|| bad "shards1-functional" "$r"
|
||||||
|
kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV=""
|
||||||
|
|
||||||
|
# ---- 4c. WO_MAILBOX=8: the drop-slow-member path FIRES and the room lives ----
|
||||||
|
# The backpressure policy earning its keep. A member that stops reading makes
|
||||||
|
# its writer block on write_dl; with the mailbox capped at 8 the room's
|
||||||
|
# broadcast send traps (WO_T_ACTOR), and the room must CATCH that, drop the
|
||||||
|
# member, and keep serving everyone else. Asserting the room survives is the
|
||||||
|
# point — a room that dies with its slowest member is the bug this policy
|
||||||
|
# exists to prevent.
|
||||||
|
serve "$((PORT0 + 5))" env WO_MAILBOX=8 || bad "serve-mailbox" "no listener"
|
||||||
|
r="$(timeout 90 python3 - "$PORT" <<'PYEOF'
|
||||||
|
import importlib.util, os, socket, sys, time
|
||||||
|
spec = importlib.util.spec_from_file_location("wsc", os.environ["WSC"])
|
||||||
|
wsc = importlib.util.module_from_spec(spec); spec.loader.exec_module(wsc)
|
||||||
|
port = int(sys.argv[1])
|
||||||
|
|
||||||
|
fast = wsc.connect(port, "bp", "fast")
|
||||||
|
wsc.recv(fast) # * fast joined
|
||||||
|
# the slow member: a tiny receive buffer so the server's socket fills fast,
|
||||||
|
# and it never reads a single frame
|
||||||
|
slow = wsc.connect(port, "bp", "slow", rcvbuf=2048)
|
||||||
|
wsc.recv(fast) # * slow joined
|
||||||
|
|
||||||
|
# storm: big frames the slow member never drains
|
||||||
|
blob = "x" * 1024
|
||||||
|
for i in range(400):
|
||||||
|
try:
|
||||||
|
wsc.send(fast, f"{i}-{blob}")
|
||||||
|
except OSError:
|
||||||
|
break
|
||||||
|
# drain what fast owes us so its own mailbox cannot be the thing that fills
|
||||||
|
deadline = time.time() + 20
|
||||||
|
seen = 0
|
||||||
|
while time.time() < deadline:
|
||||||
|
try:
|
||||||
|
k, t = wsc.recv(fast, timeout=0.5)
|
||||||
|
seen += 1
|
||||||
|
except Exception:
|
||||||
|
break
|
||||||
|
|
||||||
|
# the room must still be alive and serving the fast member
|
||||||
|
survivor = wsc.connect(port, "bp", "late")
|
||||||
|
ok_join = False
|
||||||
|
deadline = time.time() + 15
|
||||||
|
while time.time() < deadline:
|
||||||
|
try:
|
||||||
|
k, t = wsc.recv(fast, timeout=1.0)
|
||||||
|
if "late joined" in t:
|
||||||
|
ok_join = True
|
||||||
|
break
|
||||||
|
except Exception:
|
||||||
|
break
|
||||||
|
print("mailbox-ok" if ok_join else f"mailbox-dead seen={seen}")
|
||||||
|
slow.close(); fast.close(); survivor.close()
|
||||||
|
PYEOF
|
||||||
|
)"
|
||||||
|
[[ "$r" == *mailbox-ok* ]] \
|
||||||
|
&& ok "WO_MAILBOX=8: slow member dropped, room survived and kept serving" \
|
||||||
|
|| bad "mailbox-backpressure" "$r"
|
||||||
|
kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV=""
|
||||||
|
|
||||||
|
# ---- 5. the ASan leg: functional matrix, zero leaks ----
|
||||||
|
if [[ -x "$ASAN" ]]; then
|
||||||
|
sed -i "s|runtime = \".*\"|runtime = \"$ASAN\"|" "$W/app/wo.toml"
|
||||||
|
rm -rf "$W/app/target"
|
||||||
|
"$WOC" "$W/app" >/dev/null 2>&1
|
||||||
|
serve "$((PORT0 + 3))" || bad "serve-asan" "no listener"
|
||||||
|
r="$(functional asan)"
|
||||||
|
kill -TERM "$SRV" 2>/dev/null
|
||||||
|
for _ in $(seq 1 60); do kill -0 "$SRV" 2>/dev/null || break; sleep 0.1; done
|
||||||
|
SRV=""
|
||||||
|
if [[ "$r" == *functional-ok* ]] \
|
||||||
|
&& ! tail -n "+$LEGFROM" "$SRVLOG" | grep -q "AddressSanitizer\|LeakSanitizer"; then
|
||||||
|
ok "ASan run clean (functional + drain, zero leaks)"
|
||||||
|
else
|
||||||
|
bad "asan" "$(tail -n "+$LEGFROM" "$SRVLOG" | grep -m1 -E 'ERROR|SUMMARY' || echo "$r")"
|
||||||
|
fi
|
||||||
|
else
|
||||||
|
bad "asan" "runtime/build/wovm_asan missing — make -C runtime wovm-asan"
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo
|
||||||
|
printf 'chat-accept: %d checks, %d failures\n' "$((pass + fail))" "$fail"
|
||||||
|
[[ $fail -eq 0 ]]
|
||||||
|
|
@ -26,6 +26,24 @@ QUICK = "--quick" in sys.argv
|
||||||
WRITE_BASELINE = "--write-baseline" in sys.argv
|
WRITE_BASELINE = "--write-baseline" in sys.argv
|
||||||
|
|
||||||
N = 2000 if QUICK else 20000
|
N = 2000 if QUICK else 20000
|
||||||
|
# databasev2 4: the write-concurrent leg. `mix` writes on one op in ten with
|
||||||
|
# C=4, so group commit had almost nothing to batch there (measured mean batch
|
||||||
|
# 1.01, peak 3) — a property of that workload, not of the mechanism. C is high
|
||||||
|
# on purpose: batching is a function of how many writes are in flight, and
|
||||||
|
# measured mean batch rose 1.13 -> 1.76 -> 5.35 at C = 4 -> 16 -> 64.
|
||||||
|
WMIX_N = 4000 if QUICK else 20000
|
||||||
|
WMIX_C = 32 if QUICK else 64
|
||||||
|
# databasev2 3: the checkpoint leg. Ages a store by UPDATING the same rows, so
|
||||||
|
# history grows while the live set does not — otherwise the leg measures insert
|
||||||
|
# throughput instead of compaction.
|
||||||
|
CKPT_SEED = 2000 if QUICK else 5000
|
||||||
|
CKPT_OPS = 8000 if QUICK else 20000
|
||||||
|
# The stop-the-world budget. 50ms is a stall a serving process can absorb
|
||||||
|
# without a client noticing a timeout; measured at ~13ms for a 2MB live set,
|
||||||
|
# so this leaves real headroom while still failing before a stall becomes
|
||||||
|
# user-visible. Compaction is O(live rows), so this budget is what eventually
|
||||||
|
# forces the incremental design the spec deliberately did not buy in advance.
|
||||||
|
CKPT_PAUSE_BUDGET_US = 50000
|
||||||
MSG_N = 20000 if QUICK else 200000
|
MSG_N = 20000 if QUICK else 200000
|
||||||
WAL_N = 800 if QUICK else 4000
|
WAL_N = 800 if QUICK else 4000
|
||||||
CRASH_REPS = 1 if QUICK else 3
|
CRASH_REPS = 1 if QUICK else 3
|
||||||
|
|
@ -95,6 +113,55 @@ def parse_metrics(lines, into, prefix):
|
||||||
if m:
|
if m:
|
||||||
into[f"{prefix}.msgrate.msgs_sec"] = int(m.group(2))
|
into[f"{prefix}.msgrate.msgs_sec"] = int(m.group(2))
|
||||||
|
|
||||||
|
def wmix_leg(metrics, tag, env, data):
|
||||||
|
"""Every op a durable write, WMIX_C at once — the leg that actually
|
||||||
|
exercises group commit.
|
||||||
|
|
||||||
|
It reuses the store the `all` run just seeded (a fresh process replays it,
|
||||||
|
so `kmod` is there) and asks the runtime for its group-commit counters via
|
||||||
|
WO_WAL_STATS. The counters matter as much as the throughput: if batches are
|
||||||
|
always one the mechanism is inert and any throughput change came from
|
||||||
|
somewhere else, so a payoff would be attributed to the wrong cause."""
|
||||||
|
e = dict(env); e["WO_WAL_STATS"] = "1"
|
||||||
|
rc, lines, _, _ = run(["wmix", str(WMIX_N), str(WMIX_C)], e, 1800)
|
||||||
|
if rc != 0:
|
||||||
|
bad(f"{tag}.wmix", f"rc={rc} tail={lines[-2:]}")
|
||||||
|
return
|
||||||
|
ops = p50 = p99 = None
|
||||||
|
batches = records = peak_batch = peak_staged = None
|
||||||
|
for l in lines:
|
||||||
|
f = l.split()
|
||||||
|
if f and f[0] == "wmix" and len(f) == 5:
|
||||||
|
ops, p50, p99 = int(f[2]), int(f[3]), int(f[4])
|
||||||
|
elif f and f[0] == "walstats":
|
||||||
|
kv = dict(x.split("=", 1) for x in f[1:] if "=" in x)
|
||||||
|
batches = int(kv.get("batches", 0)); records = int(kv.get("records", 0))
|
||||||
|
peak_batch = int(kv.get("peak_batch", 0)); peak_staged = int(kv.get("peak_staged", 0))
|
||||||
|
if ops is None or batches is None:
|
||||||
|
bad(f"{tag}.wmix", "no report or no walstats line")
|
||||||
|
return
|
||||||
|
metrics[f"{tag}.wmix.ops_sec"] = ops
|
||||||
|
metrics[f"{tag}.wmix.p50us"] = p50
|
||||||
|
metrics[f"{tag}.wmix.p99us"] = p99
|
||||||
|
metrics[f"{tag}.wmix.peak_batch"] = peak_batch
|
||||||
|
metrics[f"{tag}.wmix.peak_staged"] = peak_staged
|
||||||
|
mean = round(records / batches, 2) if batches else 0
|
||||||
|
metrics[f"{tag}.wmix.mean_batch"] = mean
|
||||||
|
ok(f"{tag}.wmix: {ops} ops/sec, p50 {p50}us p99 {p99}us; "
|
||||||
|
f"{records} records over {batches} barriers (mean {mean}, peak {peak_batch}), "
|
||||||
|
f"peak staged {peak_staged}B")
|
||||||
|
# The gate that matters. Only the MULTI-shard leg can batch: a worker's
|
||||||
|
# statements marshal to shard 0 and queue, while shard-0 statements run
|
||||||
|
# inline and commit one at a time by design (see db.c).
|
||||||
|
if tag.endswith(".sN"):
|
||||||
|
if mean > 1.0:
|
||||||
|
ok(f"{tag}.wmix batches form (mean {mean} > 1)")
|
||||||
|
else:
|
||||||
|
bad(f"{tag}.wmix-inert",
|
||||||
|
f"mean batch {mean} — group commit is not engaging, so a "
|
||||||
|
f"throughput change would not be attributable to it")
|
||||||
|
|
||||||
|
|
||||||
def campaign():
|
def campaign():
|
||||||
metrics = {}
|
metrics = {}
|
||||||
ncores = os.cpu_count() or 1
|
ncores = os.cpu_count() or 1
|
||||||
|
|
@ -123,6 +190,8 @@ def campaign():
|
||||||
bad(f"{tag}.mix.fds", f"grew {fdg}")
|
bad(f"{tag}.mix.fds", f"grew {fdg}")
|
||||||
else:
|
else:
|
||||||
ok(f"{tag}.mix.fds flat")
|
ok(f"{tag}.mix.fds flat")
|
||||||
|
if flavor == "durable" and data:
|
||||||
|
wmix_leg(metrics, tag, env, data)
|
||||||
if data: shutil.rmtree(data, ignore_errors=True)
|
if data: shutil.rmtree(data, ignore_errors=True)
|
||||||
# msgrate once per shard count, RAM only (no store dependency)
|
# msgrate once per shard count, RAM only (no store dependency)
|
||||||
for shards in (1, ncores):
|
for shards in (1, ncores):
|
||||||
|
|
@ -235,6 +304,49 @@ def tolerance_for(key):
|
||||||
if key.startswith("ceiling."): return 100
|
if key.startswith("ceiling."): return 100
|
||||||
if key.startswith("randread."): return 100
|
if key.startswith("randread."): return 100
|
||||||
if key.startswith("replay."): return 100
|
if key.startswith("replay."): return 100
|
||||||
|
# databasev2 4: batch SHAPE follows arrival timing, so gating it tightly
|
||||||
|
# would gate the scheduler — what must hold is that the mean exceeds one
|
||||||
|
# under contention, which wmix_leg asserts directly against the live run.
|
||||||
|
# wmix's throughput and latency are NOT waived: they are the payoff, and a
|
||||||
|
# blanket waiver here would have left the whole leg ungated.
|
||||||
|
if key.endswith((".wmix.mean_batch", ".wmix.peak_batch", ".wmix.peak_staged")):
|
||||||
|
return 100
|
||||||
|
# databasev2 4: DURABLE multi-shard p99 is an fsync TAIL, and group commit
|
||||||
|
# made it both noisier and legitimately higher. Measured across three full
|
||||||
|
# runs of the same build, durable.sN.mixread.p99 was 1043 / 2318 / 4147 us
|
||||||
|
# and wmix.p99 8758 / 20000 — a 2-4x spread with the box near idle, because
|
||||||
|
# a barrier now blocks the owner shard LONGER (more records per fsync) even
|
||||||
|
# though it blocks LESS OFTEN. That is the trade group commit makes on a
|
||||||
|
# single-threaded owner, and part B (async submission) is what would undo
|
||||||
|
# it. Gating a 2-4x-variable tail at 50% gates the disk, not the engine, so
|
||||||
|
# the FLOOR is the real guard here — and it is not slack: mixread's floor
|
||||||
|
# (4172us) came within 25us of tripping on the worst run.
|
||||||
|
if key.startswith("durable.sN.") and key.endswith(".p99us"):
|
||||||
|
# Widened again 2026-08-29 with more evidence: mixread p99 was measured
|
||||||
|
# at 1043 / 2318 / 4147us and mixwrite at 1623 / 4446us across runs of
|
||||||
|
# the SAME build on a near-idle box — a 3-4x spread. 100% was still
|
||||||
|
# gating the disk. The FLOOR stays the real guard and is not slack:
|
||||||
|
# mixread's came within 25us of tripping on the worst run observed.
|
||||||
|
return 300
|
||||||
|
# databasev2 3: the RECLAIM ratio is structural and gated tightly — it is
|
||||||
|
# the feature's whole claim. Boot time and the pause are wall-clock on a
|
||||||
|
# shared box and are not: waiving them all would have left the leg ungated,
|
||||||
|
# which is the mistake part A's task 4 made and had to undo.
|
||||||
|
if key in ("ckpt.boot_off_ms", "ckpt.boot_on_ms", "ckpt.pause_us_max",
|
||||||
|
"ckpt.compactions", "ckpt.bytes_off", "ckpt.bytes_on"):
|
||||||
|
return 400
|
||||||
|
# compaction BANDWIDTH is the engine's own property, so it is gated for
|
||||||
|
# real — it is what regressed 8x when the dump was fsyncing per flush
|
||||||
|
if key == "ckpt.pause_us_per_mb":
|
||||||
|
return 100
|
||||||
|
# msgrate is actor-to-actor throughput and is scheduling-bound, so its
|
||||||
|
# run-to-run spread is far wider than its old 15%. MEASURED across the 10
|
||||||
|
# full runs recorded on 2026-08-28/29 — several of them predating the
|
||||||
|
# checkpoint work — it ranged 10.7M to 17.9M msgs/sec, a 1.67x spread. A
|
||||||
|
# 15% gate on that gates the scheduler and fails intermittently whatever
|
||||||
|
# the engine does. Pre-existing; found while closing databasev2 3, not
|
||||||
|
# caused by it.
|
||||||
|
if ".msgrate." in key: return 70
|
||||||
if ".mixread." in key or ".mixwrite." in key: return 50
|
if ".mixread." in key or ".mixwrite." in key: return 50
|
||||||
if ".sN." in key: return 50
|
if ".sN." in key: return 50
|
||||||
if ".read." in key or ".query." in key: return 50
|
if ".read." in key or ".query." in key: return 50
|
||||||
|
|
@ -246,7 +358,11 @@ def write_baseline(metrics):
|
||||||
"tolerances come from tolerance_for() in the driver"}}
|
"tolerances come from tolerance_for() in the driver"}}
|
||||||
for k, v in sorted(metrics.items()):
|
for k, v in sorted(metrics.items()):
|
||||||
if k.endswith(("rss_growth_kb", "fd_growth")): continue
|
if k.endswith(("rss_growth_kb", "fd_growth")): continue
|
||||||
higher = k.endswith(("ops_sec", "msgs_sec"))
|
# reclaim_x: MORE reclaimed is better. Recorded as lower-is-better by
|
||||||
|
# the default detector, which would have passed "no reclaim at all" and
|
||||||
|
# failed an improvement — the feature's central claim, gated backwards.
|
||||||
|
higher = k.endswith(("ops_sec", "msgs_sec", "mean_batch", "peak_batch",
|
||||||
|
"reclaim_x"))
|
||||||
floor_div = 8 if k.endswith("msgs_sec") else 4
|
floor_div = 8 if k.endswith("msgs_sec") else 4
|
||||||
# latency floors never sit below 100µs: at post-index µs scale a
|
# latency floors never sit below 100µs: at post-index µs scale a
|
||||||
# 4×1µs "catastrophe line" is noise; the tripwire means "µs became
|
# 4×1µs "catastrophe line" is noise; the tripwire means "µs became
|
||||||
|
|
@ -520,9 +636,12 @@ def randread(metrics):
|
||||||
def wal_used(data_dir):
|
def wal_used(data_dir):
|
||||||
"""Bytes actually written across the store's WAL files.
|
"""Bytes actually written across the store's WAL files.
|
||||||
|
|
||||||
The non-zero prefix, NOT the file size: shard WALs are fallocate'd to
|
The non-zero prefix, NOT the file size: shard WALs are preallocated, so
|
||||||
1 MiB up front, so getsize reports 1048576 for an empty store and proves
|
getsize reports the preallocation (1 MiB) even for an empty store. Same
|
||||||
nothing. Same reason scripts/residency-accept.sh measures it this way."""
|
reason scripts/residency-accept.sh measures it this way.
|
||||||
|
|
||||||
|
databasev2 1 and databasev2 3 each grew their own copy of this helper on
|
||||||
|
separate branches; this is the single one they now share."""
|
||||||
total = 0
|
total = 0
|
||||||
for name in sorted(os.listdir(data_dir)):
|
for name in sorted(os.listdir(data_dir)):
|
||||||
with open(os.path.join(data_dir, name), "rb") as f:
|
with open(os.path.join(data_dir, name), "rb") as f:
|
||||||
|
|
@ -621,6 +740,96 @@ def replay(metrics):
|
||||||
metrics["replay.history_penalty_x"] = round(penalty, 2)
|
metrics["replay.history_penalty_x"] = round(penalty, 2)
|
||||||
ok(f"replay: identical dataset, {penalty:.2f}x the boot cost from history alone "
|
ok(f"replay: identical dataset, {penalty:.2f}x the boot cost from history alone "
|
||||||
f"({ins_ms:.0f} -> {his_ms:.0f} ms) -- what a checkpoint would collapse")
|
f"({ins_ms:.0f} -> {his_ms:.0f} ms) -- what a checkpoint would collapse")
|
||||||
|
def checkpoint_leg(metrics):
|
||||||
|
"""Space reclaimed, boot time, and the stop-the-world PAUSE.
|
||||||
|
|
||||||
|
The same workload runs twice, differing only in whether checkpointing can
|
||||||
|
fire: an enormous floor disables it, a small one lets it. Comparing two runs
|
||||||
|
of one build is what isolates compaction from everything else the workload
|
||||||
|
does.
|
||||||
|
|
||||||
|
Boot is measured with the sample's `boot` mode, which does nothing at all —
|
||||||
|
with WO_DATA set the runtime replays the whole log before main runs, so a
|
||||||
|
mode with no work of its own is the only honest way to price replay."""
|
||||||
|
ncores = os.cpu_count() or 1
|
||||||
|
out = {}
|
||||||
|
for name, knobs in (("off", {"WO_CHECKPOINT_BYTES": "1000000000"}),
|
||||||
|
("on", {"WO_CHECKPOINT_BYTES": "65536", "WO_CHECKPOINT_RATIO": "2"})):
|
||||||
|
data = os.path.join(ROOT, "bench", f"tmp.{os.getpid()}.ckpt.{name}")
|
||||||
|
shutil.rmtree(data, ignore_errors=True); os.makedirs(data, exist_ok=True)
|
||||||
|
env = {"WO_DATA": data, "WO_SHARDS": str(ncores), "WO_WAL_STATS": "1"}
|
||||||
|
env.update(knobs)
|
||||||
|
rc, _, _, _ = run(["seed", str(CKPT_SEED)], env, 1800)
|
||||||
|
if rc != 0:
|
||||||
|
bad(f"ckpt.{name}.seed", f"rc={rc}"); shutil.rmtree(data, ignore_errors=True); return
|
||||||
|
rc, lines, _, _ = run(["wmix", str(CKPT_OPS), "16"], env, 1800)
|
||||||
|
if rc != 0:
|
||||||
|
bad(f"ckpt.{name}.age", f"rc={rc}"); shutil.rmtree(data, ignore_errors=True); return
|
||||||
|
stats = {}
|
||||||
|
for l in lines:
|
||||||
|
f = l.split()
|
||||||
|
if f and f[0] == "walstats":
|
||||||
|
stats = dict(x.split("=", 1) for x in f[1:] if "=" in x)
|
||||||
|
used = wal_used(data)
|
||||||
|
# NOT through run(): it samples RSS on a 250ms poll, so every timing it
|
||||||
|
# produces floors at the poll quantum — boot measured that way reported
|
||||||
|
# 251ms both with and without checkpointing, which is the harness's
|
||||||
|
# clock, not the engine's. Median of 3 because this is wall-clock.
|
||||||
|
benv = dict(os.environ)
|
||||||
|
benv.update({"WO_DATA": data, "WO_SHARDS": str(ncores)})
|
||||||
|
samples = []
|
||||||
|
brc = 0
|
||||||
|
for _ in range(3):
|
||||||
|
t0 = time.monotonic()
|
||||||
|
pr = subprocess.run([BIN, "boot"], stdout=subprocess.DEVNULL,
|
||||||
|
stderr=subprocess.DEVNULL, env=benv, timeout=900)
|
||||||
|
samples.append((time.monotonic() - t0) * 1000.0)
|
||||||
|
brc = pr.returncode or brc
|
||||||
|
boot_ms = sorted(samples)[1]
|
||||||
|
if brc != 0:
|
||||||
|
bad(f"ckpt.{name}.boot", f"rc={brc}"); shutil.rmtree(data, ignore_errors=True); return
|
||||||
|
out[name] = (used, boot_ms, stats)
|
||||||
|
shutil.rmtree(data, ignore_errors=True)
|
||||||
|
|
||||||
|
(off_b, off_boot, _), (on_b, on_boot, st) = out["off"], out["on"]
|
||||||
|
comps = int(st.get("compactions", 0))
|
||||||
|
if comps == 0:
|
||||||
|
bad("ckpt.inert", "no compaction ran — the leg proves nothing about checkpointing")
|
||||||
|
return
|
||||||
|
metrics["ckpt.compactions"] = comps
|
||||||
|
metrics["ckpt.bytes_off"] = off_b
|
||||||
|
metrics["ckpt.bytes_on"] = on_b
|
||||||
|
metrics["ckpt.reclaim_x"] = round(off_b / max(on_b, 1), 2)
|
||||||
|
metrics["ckpt.boot_off_ms"] = int(round(off_boot))
|
||||||
|
metrics["ckpt.boot_on_ms"] = int(round(on_boot))
|
||||||
|
metrics["ckpt.pause_us_max"] = int(st.get("compact_us_max", 0))
|
||||||
|
# The RAW pause scales with the live set, and this workload's live set is
|
||||||
|
# not fixed: wmix's hist_dump inserts a row per latency bucket, so a noisier
|
||||||
|
# box produces more buckets, more rows, and a longer pause. Gating the raw
|
||||||
|
# number against a baseline therefore gates the box. What belongs to the
|
||||||
|
# ENGINE is the rate, so that is what carries a real tolerance; the raw
|
||||||
|
# pause keeps the absolute budget assertion below as its guard.
|
||||||
|
cb = int(st.get("compacted_bytes", 0))
|
||||||
|
if cb > 0 and metrics["ckpt.pause_us_max"] > 0:
|
||||||
|
metrics["ckpt.pause_us_per_mb"] = int(round(
|
||||||
|
metrics["ckpt.pause_us_max"] / (cb / (1024.0 * 1024.0))))
|
||||||
|
ok(f"ckpt: {off_b} -> {on_b} bytes ({metrics['ckpt.reclaim_x']}x reclaimed) over "
|
||||||
|
f"{comps} compactions; boot {off_boot:.0f} -> {on_boot:.0f} ms; "
|
||||||
|
f"stop-the-world pause max {metrics['ckpt.pause_us_max']}us "
|
||||||
|
f"({metrics.get('ckpt.pause_us_per_mb', 0)}us/MB)")
|
||||||
|
# the space claim is the point of the feature, so it is asserted, not just recorded
|
||||||
|
if off_b <= on_b:
|
||||||
|
bad("ckpt.no-reclaim", f"checkpointing did not shrink the log ({off_b} -> {on_b})")
|
||||||
|
else:
|
||||||
|
ok(f"ckpt: the log is smaller with checkpointing on")
|
||||||
|
# THE BUDGET. Stated, not assumed — the spec refused to assume it.
|
||||||
|
if metrics["ckpt.pause_us_max"] > CKPT_PAUSE_BUDGET_US:
|
||||||
|
bad("ckpt.pause-budget",
|
||||||
|
f"stop-the-world pause {metrics['ckpt.pause_us_max']}us exceeds the stated "
|
||||||
|
f"{CKPT_PAUSE_BUDGET_US}us budget — alternatives (incremental copy, "
|
||||||
|
f"fork-and-dump) are bought against THIS number")
|
||||||
|
else:
|
||||||
|
ok(f"ckpt: pause within budget ({metrics['ckpt.pause_us_max']} <= {CKPT_PAUSE_BUDGET_US}us)")
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
|
|
@ -639,6 +848,7 @@ def main():
|
||||||
ceiling(metrics)
|
ceiling(metrics)
|
||||||
randread(metrics)
|
randread(metrics)
|
||||||
replay(metrics)
|
replay(metrics)
|
||||||
|
checkpoint_leg(metrics)
|
||||||
os.makedirs(RESULTS_DIR, exist_ok=True)
|
os.makedirs(RESULTS_DIR, exist_ok=True)
|
||||||
stamp = time.strftime("%Y%m%d-%H%M%S")
|
stamp = time.strftime("%Y%m%d-%H%M%S")
|
||||||
out = os.path.join(RESULTS_DIR, f"run-{stamp}{'-quick' if QUICK else ''}.json")
|
out = os.path.join(RESULTS_DIR, f"run-{stamp}{'-quick' if QUICK else ''}.json")
|
||||||
|
|
|
||||||
|
|
@ -51,6 +51,14 @@ if [[ ! -x "$WOVM" ]]; then
|
||||||
fi
|
fi
|
||||||
|
|
||||||
WORK="$(mktemp -d "${TMPDIR:-/tmp}/lw-accept.XXXXXX")"
|
WORK="$(mktemp -d "${TMPDIR:-/tmp}/lw-accept.XXXXXX")"
|
||||||
|
# stable, tailable log for the example app: the per-run work dir is deleted on
|
||||||
|
# exit, so a developer had nothing to follow. `tail -F /tmp/log-watcher.log`.
|
||||||
|
# Each invocation keeps its own $WORK/*.out (the checks grep those) and is
|
||||||
|
# ALSO teed here, banner-separated, so one file holds the whole run.
|
||||||
|
APPLOG="/tmp/log-watcher.log"
|
||||||
|
: > "$APPLOG"
|
||||||
|
echo "app log: $APPLOG (tail -F \"$APPLOG\" to follow)"
|
||||||
|
|
||||||
# LW_ACCEPT_KEEP=1 leaves the work directory (image, logs, cron.d, the
|
# LW_ACCEPT_KEEP=1 leaves the work directory (image, logs, cron.d, the
|
||||||
# server's own stdout) in place — what you want the moment a check fails.
|
# server's own stdout) in place — what you want the moment a check fails.
|
||||||
cleanup() {
|
cleanup() {
|
||||||
|
|
@ -89,7 +97,8 @@ fi
|
||||||
# watcher to decide the burst is over.
|
# watcher to decide the burst is over.
|
||||||
LOG="$WORK/app.log"
|
LOG="$WORK/app.log"
|
||||||
: >"$LOG"
|
: >"$LOG"
|
||||||
timeout -k 2 12 "$WOVM" "$IMAGE" watch "$LOG" 2 1 >"$WORK/watch.out" 2>&1 &
|
printf '\n===== watch =====\n' >>"$APPLOG"
|
||||||
|
timeout -k 2 12 "$WOVM" "$IMAGE" watch "$LOG" 2 1 > >(tee -a "$APPLOG" >"$WORK/watch.out") 2>&1 &
|
||||||
WATCH_PID=$!
|
WATCH_PID=$!
|
||||||
sleep 2
|
sleep 2
|
||||||
printf 'info service starting\n' >>"$LOG"
|
printf 'info service starting\n' >>"$LOG"
|
||||||
|
|
@ -110,6 +119,7 @@ CRON="$WORK/cron.d"
|
||||||
mkdir -p "$CRON"
|
mkdir -p "$CRON"
|
||||||
printf '* * * * * root /usr/bin/backup.sh > /var/log/backup.log 2>&1\n' >"$CRON/backup"
|
printf '* * * * * root /usr/bin/backup.sh > /var/log/backup.log 2>&1\n' >"$CRON/backup"
|
||||||
timeout 8 "$WOVM" "$IMAGE" run "$CRON" >"$WORK/run.out" 2>&1
|
timeout 8 "$WOVM" "$IMAGE" run "$CRON" >"$WORK/run.out" 2>&1
|
||||||
|
{ printf '\n===== run =====\n'; cat "$WORK/run.out"; } >>"$APPLOG"
|
||||||
if grep -q "^SCHEDULE /var/log/backup.log" "$WORK/run.out"; then
|
if grep -q "^SCHEDULE /var/log/backup.log" "$WORK/run.out"; then
|
||||||
ok "run (parsed and scheduled the cron entry)"
|
ok "run (parsed and scheduled the cron entry)"
|
||||||
else
|
else
|
||||||
|
|
@ -124,7 +134,8 @@ EOF
|
||||||
# -k: `env.stopping()` installs a SIGTERM handler that only sets a flag, and
|
# -k: `env.stopping()` installs a SIGTERM handler that only sets a flag, and
|
||||||
# the serve loop is blocked in accept(), so a plain TERM is swallowed — the
|
# the serve loop is blocked in accept(), so a plain TERM is swallowed — the
|
||||||
# process needs a KILL to actually stop (recorded in docs/00-status.md).
|
# process needs a KILL to actually stop (recorded in docs/00-status.md).
|
||||||
timeout -k 2 20 "$WOVM" "$IMAGE" mcp "$CRON" "$WORK/cfg.json" >"$WORK/mcp.out" 2>&1 &
|
printf '\n===== mcp =====\n' >>"$APPLOG"
|
||||||
|
timeout -k 2 20 "$WOVM" "$IMAGE" mcp "$CRON" "$WORK/cfg.json" > >(tee -a "$APPLOG" >"$WORK/mcp.out") 2>&1 &
|
||||||
SRV_PID=$!
|
SRV_PID=$!
|
||||||
sleep 2
|
sleep 2
|
||||||
|
|
||||||
|
|
@ -256,7 +267,8 @@ if [[ -n "${LW_SOAK:-}" ]]; then
|
||||||
soak_mode() {
|
soak_mode() {
|
||||||
local name="$1" load_fn="$2"
|
local name="$1" load_fn="$2"
|
||||||
shift 2
|
shift 2
|
||||||
"$WOVM" "$IMAGE" "$@" >"$WORK/soak-$name.out" 2>&1 &
|
printf '\n===== soak %s =====\n' "$name" >>"$APPLOG"
|
||||||
|
"$WOVM" "$IMAGE" "$@" > >(tee -a "$APPLOG" >"$WORK/soak-$name.out") 2>&1 &
|
||||||
local pid=$! rss0 fd0 rss1 fd1 drss dfd deadline i
|
local pid=$! rss0 fd0 rss1 fd1 drss dfd deadline i
|
||||||
sleep 3 # first-touch pages and the first work cycle
|
sleep 3 # first-touch pages and the first work cycle
|
||||||
if ! kill -0 "$pid" 2>/dev/null; then
|
if ! kill -0 "$pid" 2>/dev/null; then
|
||||||
|
|
|
||||||
|
|
@ -56,6 +56,11 @@ fi
|
||||||
|
|
||||||
PORT=$((8500 + RANDOM % 400))
|
PORT=$((8500 + RANDOM % 400))
|
||||||
DATA="$W/data"; mkdir -p "$DATA"
|
DATA="$W/data"; mkdir -p "$DATA"
|
||||||
|
# stable, tailable server log — the per-run temp dir is deleted on exit
|
||||||
|
SRVLOG="/tmp/site.log"
|
||||||
|
: > "$SRVLOG"
|
||||||
|
echo "server log: $SRVLOG (tail -F \"$SRVLOG\" to follow)"
|
||||||
|
|
||||||
|
|
||||||
hit() { # path [method] [data] [token] -> "STATUS|BODY" (redirects not followed)
|
hit() { # path [method] [data] [token] -> "STATUS|BODY" (redirects not followed)
|
||||||
python3 - "$PORT" "$1" "${2:-GET}" "${3:-}" "${4:-}" <<'PYEOF'
|
python3 - "$PORT" "$1" "${2:-GET}" "${3:-}" "${4:-}" <<'PYEOF'
|
||||||
|
|
@ -97,7 +102,8 @@ expect() { # name got want_status want_substr
|
||||||
}
|
}
|
||||||
|
|
||||||
serve() {
|
serve() {
|
||||||
SITE_TOKEN=s3cr3t WO_DATA="$DATA" "$W/app/target/site" "$PORT" >>"$W/srv.out" 2>&1 &
|
printf '\n===== serve — port %s =====\n' "$PORT" >>"$SRVLOG"
|
||||||
|
SITE_TOKEN=s3cr3t WO_DATA="$DATA" "$W/app/target/site" "$PORT" >>"$SRVLOG" 2>&1 &
|
||||||
SRV=$!
|
SRV=$!
|
||||||
for _ in $(seq 1 40); do
|
for _ in $(seq 1 40); do
|
||||||
[[ "$(hit /health 2>/dev/null)" == 200* ]] && return 0
|
[[ "$(hit /health 2>/dev/null)" == 200* ]] && return 0
|
||||||
|
|
|
||||||
|
|
@ -83,9 +83,19 @@ else
|
||||||
fi
|
fi
|
||||||
|
|
||||||
DATA="$W/data"; mkdir -p "$DATA"
|
DATA="$W/data"; mkdir -p "$DATA"
|
||||||
WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >"$W/srv.out" 2>&1 &
|
# stable, tailable server log: the per-run temp dir is deleted on exit, so a
|
||||||
|
# developer had nothing to follow. `tail -F /tmp/web-app.log` while this runs.
|
||||||
|
SRVLOG="/tmp/web-app.log"
|
||||||
|
: > "$SRVLOG"
|
||||||
|
echo "server log: $SRVLOG (tail -F \"$SRVLOG\" to follow)"
|
||||||
|
printf '===== boot — port %s =====\n' "$PORT" >>"$SRVLOG"
|
||||||
|
LEGFROM=$(( $(wc -l < "$SRVLOG") + 1 ))
|
||||||
|
WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >>"$SRVLOG" 2>&1 &
|
||||||
SRV=$!
|
SRV=$!
|
||||||
for _ in $(seq 1 40); do grep -q listening "$W/srv.out" 2>/dev/null && break; sleep 0.1; done
|
for _ in $(seq 1 40); do
|
||||||
|
tail -n "+$LEGFROM" "$SRVLOG" 2>/dev/null | grep -q listening && break
|
||||||
|
sleep 0.1
|
||||||
|
done
|
||||||
|
|
||||||
# one tiny HTTP client; python is already a repo test dependency
|
# one tiny HTTP client; python is already a repo test dependency
|
||||||
hit() { # method path [body] [auth: yes|no] [content-type] -> "STATUS|BODY"
|
hit() { # method path [body] [auth: yes|no] [content-type] -> "STATUS|BODY"
|
||||||
|
|
@ -467,7 +477,8 @@ for _ in $(seq 1 30); do kill -0 "$SRV" 2>/dev/null || { stopped=0; break; }; sl
|
||||||
SRV=""
|
SRV=""
|
||||||
|
|
||||||
# ---- 15. restart persistence (WAL replay) ----
|
# ---- 15. restart persistence (WAL replay) ----
|
||||||
WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >>"$W/srv.out" 2>&1 &
|
printf '\n===== restart (WAL replay) — port %s =====\n' "$PORT" >>"$SRVLOG"
|
||||||
|
WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >>"$SRVLOG" 2>&1 &
|
||||||
SRV=$!
|
SRV=$!
|
||||||
sleep 0.5
|
sleep 0.5
|
||||||
expect "product survives a restart (WAL)" "$(hit GET /products)" 200 '"name":"mug"'
|
expect "product survives a restart (WAL)" "$(hit GET /products)" 200 '"name":"mug"'
|
||||||
|
|
|
||||||
3
tests/corpus/run/monitor-death/fixture.out
Normal file
3
tests/corpus/run/monitor-death/fixture.out
Normal file
|
|
@ -0,0 +1,3 @@
|
||||||
|
died: boom
|
||||||
|
died: late
|
||||||
|
done
|
||||||
36
tests/corpus/run/monitor-death/fixture.wo
Normal file
36
tests/corpus/run/monitor-death/fixture.wo
Normal file
|
|
@ -0,0 +1,36 @@
|
||||||
|
use time
|
||||||
|
|
||||||
|
-- iteration 24 T4: actor death is OBSERVABLE. The observer names its own
|
||||||
|
-- notice message; the watched actor trapping uncaught (the runtime's
|
||||||
|
-- stderr line) delivers it. Monitoring an ALREADY dead actor fires
|
||||||
|
-- immediately. WO_SHARDS=1 (the runner) keeps the order deterministic.
|
||||||
|
class Note {
|
||||||
|
who: Text
|
||||||
|
}
|
||||||
|
|
||||||
|
class Watch {
|
||||||
|
pad: Int
|
||||||
|
fn receive(msg: Note) {
|
||||||
|
print("died: ${msg.who}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
class Boom {
|
||||||
|
pad: Int
|
||||||
|
fn receive(msg: Note) {
|
||||||
|
let z = len(msg.who) - len(msg.who);
|
||||||
|
let q = 1 / z;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn main() -> Int {
|
||||||
|
let obs: actor Note = spawn Watch { pad: 0 };
|
||||||
|
let b: actor Note = spawn Boom { pad: 0 };
|
||||||
|
monitor(b, obs, Note { who: "boom" });
|
||||||
|
send(b, Note { who: "x" });
|
||||||
|
time.sleep(100);
|
||||||
|
monitor(b, obs, Note { who: "late" });
|
||||||
|
time.sleep(100);
|
||||||
|
print("done");
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
3
tests/corpus/run/timer-delivery/fixture.out
Normal file
3
tests/corpus/run/timer-delivery/fixture.out
Normal file
|
|
@ -0,0 +1,3 @@
|
||||||
|
tick: now
|
||||||
|
tick: armed
|
||||||
|
done
|
||||||
23
tests/corpus/run/timer-delivery/fixture.wo
Normal file
23
tests/corpus/run/timer-delivery/fixture.wo
Normal file
|
|
@ -0,0 +1,23 @@
|
||||||
|
use time
|
||||||
|
|
||||||
|
-- iteration 24 T5: a timer is a MESSAGE. time.after arms a one-shot on
|
||||||
|
-- this shard; the target receives it like any send. ms <= 0 delivers now.
|
||||||
|
class Tick {
|
||||||
|
tag: Text
|
||||||
|
}
|
||||||
|
|
||||||
|
class Sink {
|
||||||
|
pad: Int
|
||||||
|
fn receive(msg: Tick) {
|
||||||
|
print("tick: ${msg.tag}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn main() -> Int {
|
||||||
|
let a: actor Tick = spawn Sink { pad: 0 };
|
||||||
|
time.after(30, a, Tick { tag: "armed" });
|
||||||
|
time.after(0, a, Tick { tag: "now" });
|
||||||
|
time.sleep(150);
|
||||||
|
print("done");
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
3
tests/corpus/run/timer-generation/fixture.out
Normal file
3
tests/corpus/run/timer-generation/fixture.out
Normal file
|
|
@ -0,0 +1,3 @@
|
||||||
|
stale gen 1 ignored
|
||||||
|
fired gen 2
|
||||||
|
done
|
||||||
28
tests/corpus/run/timer-generation/fixture.wo
Normal file
28
tests/corpus/run/timer-generation/fixture.wo
Normal file
|
|
@ -0,0 +1,28 @@
|
||||||
|
use time
|
||||||
|
|
||||||
|
-- iteration 24 T5: the CANCEL idiom — no cancel builtin, a generation
|
||||||
|
-- counter instead. The actor bumps its generation; a stale timer's
|
||||||
|
-- message names the old one and is recognized and ignored on arrival.
|
||||||
|
class Timer {
|
||||||
|
gen: Int
|
||||||
|
}
|
||||||
|
|
||||||
|
class Gate {
|
||||||
|
gen: Int
|
||||||
|
fn receive(msg: Timer) {
|
||||||
|
if msg.gen == self.gen {
|
||||||
|
print("fired gen ${msg.gen}");
|
||||||
|
} else {
|
||||||
|
print("stale gen ${msg.gen} ignored");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn main() -> Int {
|
||||||
|
let g: actor Timer = spawn Gate { gen: 2 };
|
||||||
|
time.after(30, g, Timer { gen: 1 });
|
||||||
|
time.after(60, g, Timer { gen: 2 });
|
||||||
|
time.sleep(200);
|
||||||
|
print("done");
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
Loading…
Reference in a new issue