diff --git a/bench/baseline.json b/bench/baseline.json index 65d90c1..cc50d72 100644 --- a/bench/baseline.json +++ b/bench/baseline.json @@ -1,22 +1,70 @@ { "_config": { - "N": 2000, - "crash_reps": 1, - "msg_n": 20000, + "N": 20000, + "crash_reps": 3, + "msg_n": 200000, "note": "refresh only with a commit that says why; tolerances come from tolerance_for() in the driver", - "wal_n": 800 + "wal_n": 4000 }, "ceiling.rows_recovered": { "dir": "lower", - "floor": 159744, + "floor": 159492, "tolerance_pct": 100, - "value": 39936 + "value": 39873 + }, + "ckpt.boot_off_ms": { + "dir": "lower", + "floor": 456, + "tolerance_pct": 400, + "value": 114 + }, + "ckpt.boot_on_ms": { + "dir": "lower", + "floor": 256, + "tolerance_pct": 400, + "value": 64 + }, + "ckpt.bytes_off": { + "dir": "lower", + "floor": 7876676, + "tolerance_pct": 400, + "value": 1969169 + }, + "ckpt.bytes_on": { + "dir": "lower", + "floor": 3696192, + "tolerance_pct": 400, + "value": 924048 + }, + "ckpt.compactions": { + "dir": "lower", + "floor": 100, + "tolerance_pct": 400, + "value": 6 + }, + "ckpt.pause_us_max": { + "dir": "lower", + "floor": 33912, + "tolerance_pct": 400, + "value": 8478 + }, + "ckpt.pause_us_per_mb": { + "dir": "lower", + "floor": 65848, + "tolerance_pct": 100, + "value": 16462 + }, + "ckpt.reclaim_x": { + "dir": "higher", + "floor": 0.0, + "tolerance_pct": 15, + "value": 2.13 }, "durable.s1.mixread.ops_sec": { "dir": "higher", - "floor": 2237, + "floor": 2452, "tolerance_pct": 50, - "value": 8949 + "value": 9809 }, "durable.s1.mixread.p50us": { "dir": "lower", @@ -28,31 +76,31 @@ "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 2 + "value": 12 }, "durable.s1.mixwrite.ops_sec": { "dir": "higher", - "floor": 248, + "floor": 272, "tolerance_pct": 50, - "value": 994 + "value": 1089 }, "durable.s1.mixwrite.p50us": { "dir": "lower", - "floor": 820, + "floor": 1704, "tolerance_pct": 50, - "value": 205 + "value": 426 }, "durable.s1.mixwrite.p99us": { "dir": "lower", - "floor": 872, + "floor": 1984, "tolerance_pct": 50, - "value": 218 + "value": 496 }, "durable.s1.query.ops_sec": { "dir": "higher", - "floor": 335570, + "floor": 306372, "tolerance_pct": 50, - "value": 1342281 + "value": 1225490 }, "durable.s1.query.p50us": { "dir": "lower", @@ -68,9 +116,9 @@ }, "durable.s1.read.ops_sec": { "dir": "higher", - "floor": 347705, + "floor": 307389, "tolerance_pct": 50, - "value": 1390820 + "value": 1229558 }, "durable.s1.read.p50us": { "dir": "lower", @@ -82,85 +130,121 @@ "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 2 + "value": 1 }, "durable.s1.seed.ops_sec": { "dir": "higher", - "floor": 1103, + "floor": 1095, "tolerance_pct": 15, - "value": 4415 + "value": 4381 }, "durable.s1.seed.p50us": { "dir": "lower", - "floor": 828, + "floor": 848, "tolerance_pct": 15, - "value": 207 + "value": 212 }, "durable.s1.seed.p99us": { "dir": "lower", - "floor": 2092, + "floor": 2432, "tolerance_pct": 15, - "value": 523 + "value": 608 + }, + "durable.s1.wmix.mean_batch": { + "dir": "higher", + "floor": 0.0, + "tolerance_pct": 100, + "value": 1.0 + }, + "durable.s1.wmix.ops_sec": { + "dir": "higher", + "floor": 402, + "tolerance_pct": 15, + "value": 1611 + }, + "durable.s1.wmix.p50us": { + "dir": "lower", + "floor": 1764, + "tolerance_pct": 15, + "value": 441 + }, + "durable.s1.wmix.p99us": { + "dir": "lower", + "floor": 2684, + "tolerance_pct": 15, + "value": 671 + }, + "durable.s1.wmix.peak_batch": { + "dir": "higher", + "floor": 0, + "tolerance_pct": 100, + "value": 1 + }, + "durable.s1.wmix.peak_staged": { + "dir": "lower", + "floor": 196, + "tolerance_pct": 100, + "value": 49 }, "durable.s1.write.ops_sec": { "dir": "higher", - "floor": 1155, + "floor": 573, "tolerance_pct": 15, - "value": 4620 + "value": 2294 }, "durable.s1.write.p50us": { "dir": "lower", - "floor": 832, + "floor": 1760, "tolerance_pct": 15, - "value": 208 + "value": 440 }, "durable.s1.write.p99us": { "dir": "lower", - "floor": 1948, + "floor": 2716, "tolerance_pct": 15, - "value": 487 + "value": 679 }, "durable.sN.mixread.ops_sec": { "dir": "higher", - "floor": 1112, + "floor": 1183, "tolerance_pct": 50, - "value": 4450 + "value": 4733 }, "durable.sN.mixread.p50us": { "dir": "lower", - "floor": 236, + "floor": 244, "tolerance_pct": 50, - "value": 59 + "value": 61 }, "durable.sN.mixread.p99us": { "dir": "lower", - "floor": 13100, - "tolerance_pct": 50, - "value": 3275 + "floor": 16200, + "tolerance_pct": 300, + "value": 4050 }, "durable.sN.mixwrite.ops_sec": { "dir": "higher", - "floor": 123, + "floor": 131, "tolerance_pct": 50, - "value": 494 + "value": 525 }, "durable.sN.mixwrite.p50us": { "dir": "lower", - "floor": 1160, + "floor": 2172, "tolerance_pct": 50, - "value": 290 + "value": 543 }, "durable.sN.mixwrite.p99us": { "dir": "lower", - "floor": 2944, - "tolerance_pct": 50, - "value": 736 + "floor": 16440, + "tolerance_pct": 300, + "value": 4110 }, "durable.sN.query.ops_sec": { "dir": "higher", - "floor": 287356, + "floor": 308451, "tolerance_pct": 50, - "value": 1149425 + "value": 1233806 }, "durable.sN.query.p50us": { "dir": "lower", @@ -171,14 +255,14 @@ "durable.sN.query.p99us": { "dir": "lower", "floor": 100, - "tolerance_pct": 50, + "tolerance_pct": 300, "value": 1 }, "durable.sN.read.ops_sec": { "dir": "higher", - "floor": 192752, + "floor": 248188, "tolerance_pct": 50, - "value": 771010 + "value": 992752 }, "durable.sN.read.p50us": { "dir": "lower", @@ -189,44 +273,80 @@ "durable.sN.read.p99us": { "dir": "lower", "floor": 100, - "tolerance_pct": 50, + "tolerance_pct": 300, "value": 2 }, "durable.sN.seed.ops_sec": { "dir": "higher", - "floor": 1142, + "floor": 1104, "tolerance_pct": 50, - "value": 4571 + "value": 4418 }, "durable.sN.seed.p50us": { "dir": "lower", - "floor": 836, + "floor": 844, "tolerance_pct": 50, - "value": 209 + "value": 211 }, "durable.sN.seed.p99us": { "dir": "lower", - "floor": 1916, + "floor": 2188, + "tolerance_pct": 300, + "value": 547 + }, + "durable.sN.wmix.mean_batch": { + "dir": "higher", + "floor": 1.0, + "tolerance_pct": 100, + "value": 6.22 + }, + "durable.sN.wmix.ops_sec": { + "dir": "higher", + "floor": 1504, "tolerance_pct": 50, - "value": 479 + "value": 6017 + }, + "durable.sN.wmix.p50us": { + "dir": "lower", + "floor": 27184, + "tolerance_pct": 50, + "value": 6796 + }, + "durable.sN.wmix.p99us": { + "dir": "lower", + "floor": 37484, + "tolerance_pct": 300, + "value": 9371 + }, + "durable.sN.wmix.peak_batch": { + "dir": "higher", + "floor": 15, + "tolerance_pct": 100, + "value": 60 + }, + "durable.sN.wmix.peak_staged": { + "dir": "lower", + "floor": 11760, + "tolerance_pct": 100, + "value": 2940 }, "durable.sN.write.ops_sec": { "dir": "higher", - "floor": 1010, + "floor": 580, "tolerance_pct": 50, - "value": 4040 + "value": 2320 }, "durable.sN.write.p50us": { "dir": "lower", - "floor": 832, + "floor": 1760, "tolerance_pct": 50, - "value": 208 + "value": 440 }, "durable.sN.write.p99us": { "dir": "lower", - "floor": 1996, - "tolerance_pct": 50, - "value": 499 + "floor": 2688, + "tolerance_pct": 300, + "value": 672 }, "growth.available": { "dir": "lower", @@ -236,9 +356,9 @@ }, "growth.int.noswap.bytes_per_row": { "dir": "lower", - "floor": 392, + "floor": 440, "tolerance_pct": 10, - "value": 98 + "value": 110 }, "growth.int.noswap.doublings": { "dir": "lower", @@ -266,21 +386,21 @@ }, "growth.int.noswap.rows": { "dir": "lower", - "floor": 80000, + "floor": 800000, "tolerance_pct": 100, - "value": 20000 + "value": 200000 }, "growth.int.noswap.rss_kb": { "dir": "lower", - "floor": 23968, + "floor": 168528, "tolerance_pct": 100, - "value": 5992 + "value": 42132 }, "growth.int.swap.bytes_per_row": { "dir": "lower", - "floor": 392, + "floor": 440, "tolerance_pct": 10, - "value": 98 + "value": 110 }, "growth.int.swap.doublings": { "dir": "lower", @@ -308,21 +428,21 @@ }, "growth.int.swap.rows": { "dir": "lower", - "floor": 80000, + "floor": 800000, "tolerance_pct": 100, - "value": 20000 + "value": 200000 }, "growth.int.swap.rss_kb": { "dir": "lower", - "floor": 23984, + "floor": 168576, "tolerance_pct": 100, - "value": 5996 + "value": 42144 }, "growth.text.noswap.bytes_per_row": { "dir": "lower", - "floor": 1288, + "floor": 1284, "tolerance_pct": 10, - "value": 322 + "value": 321 }, "growth.text.noswap.doublings": { "dir": "lower", @@ -350,21 +470,21 @@ }, "growth.text.noswap.rows": { "dir": "lower", - "floor": 80000, + "floor": 800000, "tolerance_pct": 100, - "value": 20000 + "value": 200000 }, "growth.text.noswap.rss_kb": { "dir": "lower", - "floor": 41216, + "floor": 343312, "tolerance_pct": 100, - "value": 10304 + "value": 85828 }, "growth.text.swap.bytes_per_row": { "dir": "lower", - "floor": 1288, + "floor": 1284, "tolerance_pct": 10, - "value": 322 + "value": 321 }, "growth.text.swap.doublings": { "dir": "lower", @@ -392,21 +512,21 @@ }, "growth.text.swap.rows": { "dir": "lower", - "floor": 80000, + "floor": 800000, "tolerance_pct": 100, - "value": 20000 + "value": 200000 }, "growth.text.swap.rss_kb": { "dir": "lower", - "floor": 41216, + "floor": 343328, "tolerance_pct": 100, - "value": 10304 + "value": 85832 }, "ram.s1.mixread.ops_sec": { "dir": "higher", - "floor": 2236, + "floor": 22286, "tolerance_pct": 50, - "value": 8947 + "value": 89144 }, "ram.s1.mixread.p50us": { "dir": "lower", @@ -422,9 +542,9 @@ }, "ram.s1.mixwrite.ops_sec": { "dir": "higher", - "floor": 248, + "floor": 2476, "tolerance_pct": 50, - "value": 994 + "value": 9904 }, "ram.s1.mixwrite.p50us": { "dir": "lower", @@ -440,15 +560,15 @@ }, "ram.s1.msgrate.msgs_sec": { "dir": "higher", - "floor": 419322, - "tolerance_pct": 15, - "value": 3354579 + "floor": 1336469, + "tolerance_pct": 70, + "value": 10691756 }, "ram.s1.query.ops_sec": { "dir": "higher", - "floor": 324675, + "floor": 244857, "tolerance_pct": 50, - "value": 1298701 + "value": 979431 }, "ram.s1.query.p50us": { "dir": "lower", @@ -464,9 +584,9 @@ }, "ram.s1.read.ops_sec": { "dir": "higher", - "floor": 332889, + "floor": 252270, "tolerance_pct": 50, - "value": 1331557 + "value": 1009081 }, "ram.s1.read.p50us": { "dir": "lower", @@ -482,87 +602,87 @@ }, "ram.s1.seed.ops_sec": { "dir": "higher", - "floor": 375939, + "floor": 62904, "tolerance_pct": 15, - "value": 1503759 + "value": 251616 }, "ram.s1.seed.p50us": { "dir": "lower", "floor": 100, "tolerance_pct": 15, - "value": 0 + "value": 4 }, "ram.s1.seed.p99us": { "dir": "lower", "floor": 100, "tolerance_pct": 15, - "value": 2 + "value": 9 }, "ram.s1.write.ops_sec": { "dir": "higher", - "floor": 272628, + "floor": 47770, "tolerance_pct": 15, - "value": 1090512 + "value": 191080 }, "ram.s1.write.p50us": { "dir": "lower", "floor": 100, "tolerance_pct": 15, - "value": 1 + "value": 8 }, "ram.s1.write.p99us": { "dir": "lower", "floor": 100, "tolerance_pct": 15, - "value": 2 + "value": 10 }, "ram.sN.mixread.ops_sec": { "dir": "higher", - "floor": 2241, + "floor": 11218, "tolerance_pct": 50, - "value": 8964 + "value": 44874 }, "ram.sN.mixread.p50us": { "dir": "lower", - "floor": 228, + "floor": 240, "tolerance_pct": 50, - "value": 57 + "value": 60 }, "ram.sN.mixread.p99us": { "dir": "lower", - "floor": 1412, + "floor": 324, "tolerance_pct": 50, - "value": 353 + "value": 81 }, "ram.sN.mixwrite.ops_sec": { "dir": "higher", - "floor": 249, + "floor": 1246, "tolerance_pct": 50, - "value": 996 + "value": 4986 }, "ram.sN.mixwrite.p50us": { "dir": "lower", - "floor": 252, + "floor": 260, "tolerance_pct": 50, - "value": 63 + "value": 65 }, "ram.sN.mixwrite.p99us": { "dir": "lower", - "floor": 280, + "floor": 356, "tolerance_pct": 50, - "value": 70 + "value": 89 }, "ram.sN.msgrate.msgs_sec": { "dir": "higher", - "floor": 214795, - "tolerance_pct": 50, - "value": 1718360 + "floor": 317323, + "tolerance_pct": 70, + "value": 2538586 }, "ram.sN.query.ops_sec": { "dir": "higher", - "floor": 331125, + "floor": 291545, "tolerance_pct": 50, - "value": 1324503 + "value": 1166180 }, "ram.sN.query.p50us": { "dir": "lower", @@ -578,9 +698,9 @@ }, "ram.sN.read.ops_sec": { "dir": "higher", - "floor": 340599, + "floor": 317823, "tolerance_pct": 50, - "value": 1362397 + "value": 1271294 }, "ram.sN.read.p50us": { "dir": "lower", @@ -596,87 +716,87 @@ }, "ram.sN.seed.ops_sec": { "dir": "higher", - "floor": 353606, + "floor": 73305, "tolerance_pct": 50, - "value": 1414427 + "value": 293220 }, "ram.sN.seed.p50us": { "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 1 + "value": 3 }, "ram.sN.seed.p99us": { "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 2 + "value": 7 }, "ram.sN.write.ops_sec": { "dir": "higher", - "floor": 290697, + "floor": 56810, "tolerance_pct": 50, - "value": 1162790 + "value": 227241 }, "ram.sN.write.p50us": { "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 1 + "value": 6 }, "ram.sN.write.p99us": { "dir": "lower", "floor": 100, "tolerance_pct": 50, - "value": 2 + "value": 10 }, "randread.collapse_x": { "dir": "lower", - "floor": 1172, + "floor": 1084, "tolerance_pct": 100, - "value": 293 + "value": 271 }, "randread.overcap.filled_rss_kb": { "dir": "lower", - "floor": 25680, + "floor": 58144, "tolerance_pct": 100, - "value": 6420 + "value": 14536 }, "randread.overcap.ops_sec": { "dir": "higher", - "floor": 1665, + "floor": 1427, "tolerance_pct": 100, - "value": 6661 + "value": 5711 }, "randread.overcap.read_p50us": { "dir": "lower", - "floor": 556, + "floor": 624, "tolerance_pct": 100, - "value": 139 + "value": 156 }, "randread.overcap.read_p99us": { "dir": "lower", - "floor": 1920, + "floor": 1628, "tolerance_pct": 100, - "value": 480 + "value": 407 }, "randread.resident.filled_rss_kb": { "dir": "lower", - "floor": 54080, + "floor": 168288, "tolerance_pct": 100, - "value": 13520 + "value": 42072 }, "randread.resident.ops_sec": { "dir": "higher", - "floor": 488424, + "floor": 387281, "tolerance_pct": 100, - "value": 1953697 + "value": 1549126 }, "randread.resident.read_p50us": { "dir": "lower", "floor": 100, "tolerance_pct": 100, - "value": 0 + "value": 1 }, "randread.resident.read_p99us": { "dir": "lower", @@ -686,57 +806,57 @@ }, "replay.history.ms": { "dir": "lower", - "floor": 844, + "floor": 12876, "tolerance_pct": 100, - "value": 211 + "value": 3219 }, "replay.history.ns_per_record": { "dir": "lower", - "floor": 21068, + "floor": 64384, "tolerance_pct": 100, - "value": 5267 + "value": 16096 }, "replay.history.records": { "dir": "lower", - "floor": 160000, + "floor": 800000, "tolerance_pct": 100, - "value": 40000 + "value": 200000 }, "replay.history.wal_bytes": { "dir": "lower", - "floor": 7840140, + "floor": 39200140, "tolerance_pct": 100, - "value": 1960035 + "value": 9800035 }, "replay.history_penalty_x": { "dir": "lower", "floor": 100, "tolerance_pct": 100, - "value": 1.9 + "value": 1.6 }, "replay.inserts.ms": { "dir": "lower", - "floor": 444, + "floor": 8064, "tolerance_pct": 100, - "value": 111 + "value": 2016 }, "replay.inserts.ns_per_record": { "dir": "lower", - "floor": 22120, + "floor": 80656, "tolerance_pct": 100, - "value": 5530 + "value": 20164 }, "replay.inserts.records": { "dir": "lower", - "floor": 80000, + "floor": 400000, "tolerance_pct": 100, - "value": 20000 + "value": 100000 }, "replay.inserts.wal_bytes": { "dir": "lower", - "floor": 3920140, + "floor": 19600140, "tolerance_pct": 100, - "value": 980035 + "value": 4900035 }, "replay.startup_ms": { "dir": "lower", diff --git a/compiler/src/emit.ml b/compiler/src/emit.ml index c3b3847..095ea2b 100644 --- a/compiler/src/emit.ml +++ b/compiler/src/emit.ml @@ -295,6 +295,7 @@ let b_sha1 = 85 let b_sha256 = 86 let b_hmac_sha256 = 87 let b_call = 88 +let b_monitor = 89 let b_split = 28 let b_split_ws = 29 let b_join = 30 @@ -1110,7 +1111,7 @@ let is_builtin_name (n : string) = "substr"; "trim"; "to_lower"; "char_of"; "parse_int"; "split"; "split_ws"; "join"; "slice"; "pop"; "shift"; "sort"; "reverse"; "remove"; "key_at"; "val_at"; (* the concurrency arc *) - "send"; "call"; + "send"; "call"; "monitor"; (* iteration 19: Float bridges and Bytes surface *) "float"; "trunc"; "parse_float"; "float_to_text"; "float_cmp"; "bytes_len"; "bytes_at"; "bytes_slice"; "bytes_eq"; "bytes_concat"; "base64_encode"; "base64_decode"; @@ -3441,11 +3442,18 @@ and emit_call (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e : As put f (ins_abc op_builtin dst base sm.Types.sm_builtin); (* every stdlib member only READS its arguments, so one that was freshly built here (`net.write(c, head .. resp.body)`) has no - other owner and dies with the call *) + other owner and dies with the call. The ONE exception: + `time.after`'s message (arg 2) MOVES to the runtime — the + timer owns it until delivery (iteration 24 T5). *) + let moves i = + alias = "time" && mname = "after" && i = 2 + in List.iteri (fun i (a : Ast.expr) -> - drop_fresh_owned ~keep:dst p f (base + i) a; - drop_fresh_text ~keep:dst p f (base + i) a) + if not (moves i) then begin + drop_fresh_owned ~keep:dst p f (base + i) a; + drop_fresh_text ~keep:dst p f (base + i) a + end) args end) | Some u -> ( @@ -3722,7 +3730,7 @@ and emit_builtin (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e : dangle the value just read) and the stores, which either copy (Text, handled by copied_container_call) or take ownership (OWNED/GCREF). *) let reader = List.mem name [ "get"; "latest"; "key_at"; "val_at" ] in - (if not (List.mem name [ "push"; "set"; "send"; "call" ]) then + (if not (List.mem name [ "push"; "set"; "send"; "call"; "monitor" ]) then List.iteri (fun i (a : Ast.expr) -> (* a reader's result points into arg0 (the container) — dropping @@ -3754,6 +3762,7 @@ and emit_builtin (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e : match name with | "send" -> fixed b_send (* arc: msg (arg1) moved to the runtime — never dropped here *) | "call" -> fixed b_call (* iteration 24: same move; the SCALAR reply lands in dst *) + | "monitor" -> fixed b_monitor (* T4: notice msg (arg2) moves to the runtime *) | "now" -> fixed b_now | "print" -> fixed b_print | "print_int" -> fixed b_print_int diff --git a/compiler/src/owner.ml b/compiler/src/owner.ml index bfe8f29..3db6180 100644 --- a/compiler/src/owner.ml +++ b/compiler/src/owner.ml @@ -1348,6 +1348,10 @@ and analyze_call (ctx : ctx) (call_e : Ast.expr) (callee : Ast.expr) (args : Ast iteration 24: call(addr, msg) moves its message identically. *) | Ident "send" -> i = 1 && Types.StringMap.find_opt "send" ctx.syms.Types.free_fns = None | Ident "call" -> i = 1 && Types.StringMap.find_opt "call" ctx.syms.Types.free_fns = None + (* T4/T5: the notice / timer message moves to the runtime too *) + | Ident "monitor" -> + i = 2 && Types.StringMap.find_opt "monitor" ctx.syms.Types.free_fns = None + | Field ({ kind = Ident "time"; _ }, "after") -> i = 2 | _ -> false in List.iteri @@ -1365,7 +1369,8 @@ and analyze_call (ctx : ctx) (call_e : Ast.expr) (callee : Ast.expr) (args : Ast transfer ctx p ~what: (match callee.kind with - | Ident "send" | Ident "call" -> + | Ident "send" | Ident "call" | Ident "monitor" + | Field ({ kind = Ident "time"; _ }, "after") -> "cannot be sent — a message moves to the receiver" | _ -> "cannot be stored in a container") then record_move ctx p (MvArg "element")) diff --git a/compiler/src/types.ml b/compiler/src/types.ml index edfb08f..912376a 100644 --- a/compiler/src/types.ml +++ b/compiler/src/types.ml @@ -309,6 +309,8 @@ let stdlib_members : stdlib_member list = m "net" "write_dl" 3 93 (Some (TScalar "Bool")) None; m "net" "listen_unix" 1 94 (Some (TScalar "Int")) None; m "net" "peer" 1 95 (Some (TScalar "Text")) None; + (* iteration 24 T5: one-shot timer — the msg MOVES to the runtime *) + m "time" "after" 3 90 None None; (* proc *) m "proc" "run" 2 56 (Some (TNullable (TScalar proc_record_name))) (Some proc_record_name); (* json — both members are lowered specially (emit.ml): encode needs its @@ -1917,6 +1919,48 @@ let typecheck_program ~file ~(module_of : string -> string) ~message:"`call`'s first argument must be an `actor M` address" ()) | None -> ()) | _ -> ()) + | None when name = "monitor" -> + (* iteration 24 T4: monitor(watched, observer, msg) — the + notice msg is typed against the OBSERVER's mailbox + (three-argument form: the caller may be main, which has + no mailbox). msg moves like send's. *) + (if List.length args <> 3 then + Diag.Collector.add collector + (Diag.error ~code:bad_arity_code ~file ~line:e.pos.line ~col:e.pos.col + ~message: + (Printf.sprintf + "`monitor` takes 3 arguments (watched, observer, notice), given %d" + (List.length args)) + ()) + else + match args with + | [ w; o; m ] -> ( + (match confident_typ cenv w with + | Some (TActor _) | None -> () + | Some _ -> + Diag.Collector.add collector + (Diag.error ~code:type_mismatch_code ~file ~line:w.pos.line + ~col:w.pos.col + ~message:"`monitor`'s first argument must be an `actor M` address" ())); + match confident_typ cenv o with + | Some (TActor want) -> ( + match confident_typ cenv m with + | Some (TScalar got) when got <> want -> + Diag.Collector.add collector + (Diag.error ~code:type_mismatch_code ~file ~line:m.pos.line + ~col:m.pos.col + ~message: + (Printf.sprintf + "the observer receives `%s` — the notice is a `%s`" want got) + ()) + | _ -> ()) + | Some _ -> + Diag.Collector.add collector + (Diag.error ~code:type_mismatch_code ~file ~line:o.pos.line + ~col:o.pos.col + ~message:"`monitor`'s second argument must be an `actor M` address" ()) + | None -> ()) + | _ -> ()) | None -> let confident_types = List.map (confident_typ cenv) args in check_builtin_call ~file collector name e.pos args confident_types) diff --git a/database/src/CODE-LOGIC.md b/database/src/CODE-LOGIC.md index 166f1c4..d30b81d 100644 --- a/database/src/CODE-LOGIC.md +++ b/database/src/CODE-LOGIC.md @@ -119,3 +119,135 @@ rather than acknowledging what disk never got. columns excluded (engine raw-eq is narrower than VM float-eq, and a probe miss cannot be resurrected by a recheck). Pinned by `tests/corpus/run/query-index-probe`. + +## Group commit: one barrier per drain (databasev2 4 part A, 2026-08-28) + +**What changed:** the engine used to commit per *statement*. `db.c` called +`wo_wal_commit` immediately after every append, at all six sites, so each row +change bought its own `pwrite` and its own `fdatasync`. Now the barrier belongs +to the drain, not to the statement. + +**Where the barrier runs, and why there.** A statement on a worker shard has no +WAL to write — the runtime asserts workers hold neither `db` nor `wal` — so it +marshals to shard 0 and parks. Shard 0 executes those requests in its envelope +drain (`wo_vm_adopt`), and the drain now **holds each reply** instead of pushing +it as the statement finishes. When the queue empties it issues one barrier, then +releases every held reply. + +Holding the reply is the whole mechanism. Pushing it early would unpark the +requester before its record was durable; holding it means each writer is +acknowledged after the barrier that carried *its own* record. That was always +the intended contract — it was simply true by accident before, because every +batch had exactly one member. + +**Why the queue is the boundary.** Not a tick, and not a timer. A queue of one +gives a batch of one, so a lone writer pays exactly what it paid before; the +batch grows only when writes genuinely contend. A tick boundary would have +added latency even with nothing to batch against, which is taxing an idle +system to serve a busy one. There is nothing to tune, which is the point. + +**Why the inline path is asymmetric.** A statement already on shard 0 stages and +commits before returning, batch size one. It cannot hold a reply because there +is nobody to reply to — it returns into its own fiber. Batching it would mean +parking that fiber on the barrier, which is part B's machinery. Two consequences +worth keeping in mind: single-shard configurations get no batching at all, by +design; and the inline commit is only safe because the drain commits +*unconditionally* whenever anything is staged, so the buffer is empty when an +inline statement runs. If that ever stops holding, the inline path would make +another statement's record durable early and acknowledge it to the wrong writer. + +**One rule for failure: once a statement has mutated RAM, the outcomes are +durable or process death.** It replaced three behaviours that disagreed — +`insert` un-applied itself, while `update` and `delete` returned a catchable +trap and left RAM ahead of disk, which their own comments said out loud. +Batching would have multiplied that from one row to a whole batch. So a failed +stage or a failed barrier now prints one diagnostic (operation, log path, +`errno`, record count) and exits 3; `WO_T_IO` is unreachable from a write. +Retrying is not offered because it is unsound: on Linux a failed `fsync` may +already have discarded the dirty pages, so a second call can report success +having written nothing. Replay is the recovery that works. + +**Measuring it.** `WO_WAL_STATS=1` makes the runtime print one line at exit — +batches, records, peak batch, peak staged bytes. Opt-in, because it would +otherwise pollute every durable program's output. The counters live in `wo_wal` +rather than behind a builtin: they are diagnostic, not part of the language. +`db-bench`'s `wmix N C` leg exists to exercise this at all — `mix` writes on one +op in ten with C=4, which produced a measured mean batch of 1.01, so it could +never have shown whether batching worked. + +**If you are looking at this because writes got slower**, check the mean batch +first. Mean 1.0 means the mechanism is not engaging, which is expected for a +serial writer or a single-shard configuration and a bug anywhere else. + +## Checkpoint: compaction by rewrite + rename (databasev2 3, 2026-08-29) + +**The problem:** nothing ever removed superseded records, so the log grew +forever and boot replayed all history. Measured before this: 20 000 rows seeded +gave a 986 KB log; updating those same rows 20 000 times took it to 2.6 MB with +**the same live data**. + +**Why one file and not a snapshot plus a tail.** Postgres does the opposite — +its WAL is a redo tail and the data lives in heap files, so a checkpoint flushes +pages and then recycles log segments; it never compacts. It cannot: its records +are page deltas, so a compacted redo log is not a store. **Ours are full row +images** — `apply_record` implements UPDATE as remove-then-recreate — so a log +of one record per live row *is* a complete store. That single difference deletes +the control file, the redo pointer, the second recovery source and the separate +process from this design. Recovery is not merely compatible with compaction; it +is completely unaware of it. + +**Why `rename` is the whole crash-safety story.** The dump goes to a temp file, +which is fsynced, renamed over the live log, and then the parent directory is +fsynced (the rename is atomic in-kernel, but the directory entry is not durable +until the parent is — Postgres does the same for the same reason). Before the +rename the live log is intact and the temp is not authoritative; after it the new +log is complete. There is no instant at which a reader sees a mixture, so this +needs no recovery logic of its own. What Postgres achieves with a redo pointer +computed at checkpoint start and a control file written at the end, one syscall +achieves here — because we can swap the entire data set atomically and Postgres +cannot. + +A crash mid-rewrite leaves a temp file. The next open **removes it**, and it is +deleted rather than ignored because a file full of well-formed records sitting +beside the log is exactly what a later reader mistakes for data. + +**Why the dump flushes periodically, and why it does NOT fsync when it does.** +`stage()` grows the staging buffer by doubling and never shrinks it, so pushing a +whole store through one buffer would hold the entire store in RAM on top of the +store — the unbounded growth databasev2 1 measured as how this engine dies. So +the dump flushes every 256 records. It flushes with a plain write, **not** a +commit: intermediate durability is worthless because the temp is not +authoritative until the rename and is fsynced once immediately before it. Using +the committing path cost one barrier per 256 records and made the pause 8× +larger — measured 107 649 µs against 13 212 µs for a 2 MB live set, ~22 MB/s +against ~181 MB/s. + +**Why the replacement is preallocated like the original.** The WAL is +preallocated so that appends never extend the file, which is what lets +`fdatasync` alone serve as the ack barrier. A replacement opened without it +would silently change that property, and the zero-padded tail the open-time scan +relies on. + +**When it runs.** Only where the staging buffer is empty — right after a +barrier. Both write paths check: the drain (`vm.c`, after its commit and after +releasing held replies, since those records are already durable and should not +wait out a rewrite) and the inline path (`db.c`). Wiring only the drain left +`WO_SHARDS=1` never compacting, with its log growing forever: measured 536 KB +where the multi-shard run held 446 KB. + +**The trigger** compares the log against what the *last* compaction actually +wrote, with an absolute floor. The denominator is measured rather than +estimated, because estimating the live size means estimating Text and the +compactor already knows the true number. There is deliberately **no timer**: +Postgres needs one because its dirty buffers are not durable until flushed, and +ours are durable at commit — an idle log does not grow. + +**A failed compaction is a missed optimisation, not a durability event.** It +leaves the original log intact and returns an error the callers ignore. It must +never take `wo_wal_commit_fatal`'s path, which exists for a different problem. + +**If you are here because a checkpoint misbehaved:** `WO_WAL_STATS=1` reports +compaction count, the stop-the-world pause (max and total) and the last +compaction's size. `WO_CHECKPOINT_BYTES` and `WO_CHECKPOINT_RATIO` move the +policy; setting a tiny floor forces compaction in a few writes, which is how the +gate tests it at all. diff --git a/database/src/db.c b/database/src/db.c index 19b95bc..52234b7 100644 --- a/database/src/db.c +++ b/database/src/db.c @@ -18,6 +18,23 @@ static int table_is_durable(const wo_db *db, uint32_t cid) { return (db->classes[cid].flags & WO_CLASSF_VOLATILE) == 0u; } +/* databasev2 3: the inline path's compaction check. + * + * The drain has its own (vm.c, after the barrier). This one exists because a + * statement running ON the owner shard never enters that drain, so without it + * a single-shard durable program's log grows FOREVER — measured: WO_SHARDS=1 + * reached 536 KB where the multi-shard run held 446 KB, because the check was + * only wired into the drain. + * + * Safe here for the same reason it is safe there: the commit above just + * emptied the staging buffer. The result is ignored because a failed + * compaction is a missed optimisation, not a durability event. */ +static void maybe_compact(wo_db *db, wo_wal *w) { + if (wo_wal_should_compact(w->off, w->compacted_bytes, wo_wal_ckpt_floor, + wo_wal_ckpt_ratio)) + (void)wo_wal_compact(w, db); +} + int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { uint32_t A = wo_ins_a(ins), B = wo_ins_b(ins), C = wo_ins_c(ins); wo_db *db = (wo_db *)vm->rt.db; @@ -36,15 +53,26 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { : WO_T_DB; wo_wal *w = (wo_wal *)vm->rt.wal; if (w && table_is_durable(db, cid)) { - /* RAM applied, record staged, ONE commit before the ack (the - * builtin's return). A failed commit is a failed write: the - * row is removed again so RAM never claims what disk never - * acknowledged, and the statement traps. */ - if (wo_wal_append_insert(w, db, cid, id) != 0 || wo_wal_commit(w) != 0) { - wo_row_remove(db, cid, id); - *msg = "wal commit failed"; - return WO_T_IO; - } + /* THE INLINE PATH KEEPS ITS OWN BARRIER, AND THAT ASYMMETRY IS + * DELIBERATE (databasev2 4 part A). The request path batches: + * wo_vm_adopt holds each reply and commits once per drain. This + * path cannot, because it has no reply to hold — it returns into + * its OWN fiber rather than unparking a requester. Do not "fix" + * this by dropping the commit: without it an inline statement + * would never be durable at all. + * + * Committing here is safe because the drain commits + * unconditionally whenever anything is staged, so the buffer is + * empty when this runs. + * + * The `table_is_durable` guard is databasev2 2's: a + * `@table(durable: false)` class is never staged, so it reaches + * neither this barrier nor the compaction check below. + * + * Failure is fatal, not a trap: the row is already in RAM. */ + if (wo_wal_append_insert(w, db, cid, id) != 0) wo_wal_stage_fatal(w); + wo_wal_commit_fatal(w, 1); + maybe_compact(db, w); } R[A] = id; return 0; @@ -58,10 +86,11 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { return ek == DB_ERR_UNIQUE ? WO_T_UNIQUE : ek == DB_ERR_OOM ? WO_T_OOM : WO_T_DB; wo_wal *w = (wo_wal *)vm->rt.wal; if (w && table_is_durable(db, cid)) { - if (wo_wal_append_update(w, db, cid, id) != 0 || wo_wal_commit(w) != 0) { - *msg = "wal commit failed"; /* RAM ahead of disk: trap, do not ack */ - return WO_T_IO; - } + /* was: trap and leave RAM ahead of disk, which the old comment + * admitted. Now fatal — see the insert arm. */ + if (wo_wal_append_update(w, db, cid, id) != 0) wo_wal_stage_fatal(w); + wo_wal_commit_fatal(w, 1); + maybe_compact(db, w); } R[A] = 0; return 0; @@ -81,10 +110,9 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { } wo_wal *w = (wo_wal *)vm->rt.wal; if (w && table_is_durable(db, cid)) { - if (wo_wal_append_remove(w, cid, id) != 0 || wo_wal_commit(w) != 0) { - *msg = "wal commit failed"; - return WO_T_IO; - } + if (wo_wal_append_remove(w, cid, id) != 0) wo_wal_stage_fatal(w); + wo_wal_commit_fatal(w, 1); + maybe_compact(db, w); } R[A] = 0; return 0; @@ -226,13 +254,12 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) { q->msg = m; break; } - if (w) { - if (wo_wal_append_insert(w, db, q->cid, id) != 0 || wo_wal_commit(w) != 0) { - wo_row_remove(db, q->cid, id); - q->status = WO_T_IO; - q->msg = "wal commit failed"; - break; - } + if (w && table_is_durable(db, q->cid)) { + /* databasev2 4: staging failure is FATAL, not a trap. The row is + * already in RAM; of the three verbs only insert could undo + * itself, so continuing means RAM ahead of disk. One rule: once a + * statement has mutated RAM, the outcomes are durable or death. */ + if (wo_wal_append_insert(w, db, q->cid, id) != 0) wo_wal_stage_fatal(w); } q->result = id; break; @@ -244,12 +271,8 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) { q->msg = m; break; } - if (w) { - if (wo_wal_append_update(w, db, q->cid, q->id) != 0 || wo_wal_commit(w) != 0) { - q->status = WO_T_IO; - q->msg = "wal commit failed"; - break; - } + if (w && table_is_durable(db, q->cid)) { + if (wo_wal_append_update(w, db, q->cid, q->id) != 0) wo_wal_stage_fatal(w); } break; } @@ -264,12 +287,8 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) { q->msg = "no such row"; break; } - if (w) { - if (wo_wal_append_remove(w, q->cid, q->id) != 0 || wo_wal_commit(w) != 0) { - q->status = WO_T_IO; - q->msg = "wal commit failed"; - break; - } + if (w && table_is_durable(db, q->cid)) { + if (wo_wal_append_remove(w, q->cid, q->id) != 0) wo_wal_stage_fatal(w); } break; } diff --git a/database/src/wal.c b/database/src/wal.c index 659aa4d..11e68ff 100644 --- a/database/src/wal.c +++ b/database/src/wal.c @@ -4,7 +4,9 @@ #include "wal.h" #include +#include #include +#include #include #include #include @@ -302,6 +304,19 @@ int wo_wal_open(wo_wal *w, const char *path, uint64_t prealloc) { memset(w, 0, sizeof(*w)); w->fd = open(path, O_RDWR | O_CREAT, 0644); if (w->fd < 0) return -1; + w->path = strdup(path); /* NULL is tolerated: the diagnostic degrades */ + /* databasev2 3: remove a stale compaction temp before doing anything else. + * The only way one exists is a crash before the rename, which means its + * records were never authoritative — the live log below is the truth. It is + * deleted rather than ignored because a file full of well-formed records + * sitting beside the log is exactly the thing a future reader mistakes for + * data. */ + { + char tmp[4096]; + if ((size_t)snprintf(tmp, sizeof tmp, "%s%s", path, WO_WAL_TMP_SUFFIX) < sizeof tmp) + (void)unlink(tmp); + } + w->prealloc = prealloc; if (prealloc) { /* best-effort: a filesystem without fallocate still works */ (void)posix_fallocate(w->fd, 0, (off_t)prealloc); @@ -317,6 +332,7 @@ int wo_wal_open(wo_wal *w, const char *path, uint64_t prealloc) { void wo_wal_close(wo_wal *w) { if (w->fd >= 0) close(w->fd); + free(w->path); free(w->buf); memset(w, 0, sizeof(*w)); w->fd = -1; @@ -390,7 +406,103 @@ int wo_wal_append_remove(wo_wal *w, uint32_t class_id, uint64_t id) { } int wo_wal_commit(wo_wal *w) { - if (!w->len) return 0; + if (!w->len) return 0; /* empty commits are not batches; do not count them */ + if (w->len > w->stat_peak_staged) w->stat_peak_staged = w->len; + size_t at = 0; + while (at < w->len) { + ssize_t n = pwrite(w->fd, w->buf + at, w->len - at, (off_t)(w->off + at)); + if (n < 0) { + if (errno == EINTR) continue; + return WO_WAL_ERR_WRITE; + } + at += (size_t)n; + } + if (fdatasync(w->fd) != 0) return WO_WAL_ERR_SYNC; + w->off += w->len; + w->len = 0; /* acked: the batch is durable */ + return 0; +} + +/* Nothing at either fatal point is recoverable: RAM holds changes the log + * does not, and this process can no longer serve reads that would survive a + * restart. Name what failed precisely enough to act on, then stop. */ +static void wal_die(const wo_wal *w, const char *op, uint32_t nrec) { + fprintf(stderr, + "writeonce: DURABILITY FAILURE — %s failed on %s: %s\n" + " %u record(s) were NOT made durable and are not acknowledged.\n" + " The process is stopping: replay restores the last durable state.\n", + op, w->path ? w->path : "(the write-ahead log)", strerror(errno), + nrec); + exit(WO_EXIT_DURABILITY); +} + +void wo_wal_stage_fatal(const wo_wal *w) { wal_die(w, "staging a record", 1); } + +void wo_wal_commit_fatal(wo_wal *w, uint32_t nrec) { + int staged = w->len != 0; + int rc = wo_wal_commit(w); + if (rc == 0) { + if (staged) { /* count the barrier that actually happened */ + w->stat_batches++; + w->stat_records += nrec; + if (nrec > w->stat_peak_batch) w->stat_peak_batch = nrec; + } + return; + } + wal_die(w, rc == WO_WAL_ERR_SYNC ? "fdatasync" : "pwrite", nrec); +} + +uint64_t wo_wal_ckpt_floor = 4u << 20; /* 4 MiB: below this there is nothing worth reclaiming */ +uint32_t wo_wal_ckpt_ratio = 3u; /* 3x the live-set's own size is enough history */ + +int wo_wal_should_compact(uint64_t used, uint64_t last, uint64_t floor, uint32_t ratio) { + if (used < floor) return 0; /* a small log has nothing to reclaim */ + if (last == 0) return 1; /* past the floor and never compacted: do it once + * to establish the denominator */ + if (ratio == 0) return 0; /* a zero ratio disables the policy rather than + * dividing by nothing */ + return used > last * (uint64_t)ratio; +} + +/* databasev2 3: how many records the dump stages before flushing. + * + * NOT unbounded: stage() grows the staging buffer by doubling and never + * shrinks it, so appending a whole store through one buffer would hold the + * entire store in RAM on top of the store itself — the unbounded growth + * databasev2 1 identified as how this engine dies. 256 records is a few tens + * of KiB per flush, which is large enough that the syscall cost is amortised + * and small enough that the buffer never matters. */ +#define WO_WAL_COMPACT_FLUSH 256u + +/* rename(2)'s atomicity is in-kernel: the new directory ENTRY is not durable + * until the parent directory is synced. Postgres does the same thing for the + * same reason. Best-effort — a filesystem that refuses to sync a directory + * still leaves a correct log, just one whose swap might not survive a power + * cut. */ +static void sync_parent_dir(const char *path) { + char dir[4096]; + size_t n = strlen(path); + if (n >= sizeof dir) return; + memcpy(dir, path, n + 1); + char *slash = strrchr(dir, '/'); + if (slash == dir) dir[1] = '\0'; + else if (slash) *slash = '\0'; + else memcpy(dir, ".", 2); + int fd = open(dir, O_RDONLY); + if (fd < 0) return; + (void)fsync(fd); + close(fd); +} + +/* databasev2 3: write the staged bytes WITHOUT a durability barrier. + * + * Only compaction's dump uses this. Intermediate durability there is worthless: + * the temp file is not authoritative until the rename, and it is fsynced once + * immediately before that. Using wo_wal_commit for the dump instead cost one + * fdatasync per 256 records — measured, that was most of the stop-the-world + * pause (~22 MB/s, where the fixed cost plus ~150 redundant syncs dominated a + * 2 MB dump). */ +static int wal_write_nosync(wo_wal *w) { size_t at = 0; while (at < w->len) { ssize_t n = pwrite(w->fd, w->buf + at, w->len - at, (off_t)(w->off + at)); @@ -400,12 +512,110 @@ int wo_wal_commit(wo_wal *w) { } at += (size_t)n; } - if (fdatasync(w->fd) != 0) return -1; w->off += w->len; - w->len = 0; /* acked: the batch is durable */ + w->len = 0; return 0; } +static uint64_t mono_us(void) { + struct timespec ts; + if (clock_gettime(CLOCK_MONOTONIC, &ts) != 0) return 0; + return (uint64_t)ts.tv_sec * 1000000ull + (uint64_t)ts.tv_nsec / 1000ull; +} + +/* ============================================================================ + * OBLIGATION FOR WHOEVER IMPLEMENTS `resident: keys` (databasev2 2, tasks + * 5c/5d) — READ THIS BEFORE STORING WAL OFFSETS. + * + * Compaction rewrites the log and MOVES EVERY RECORD. Any WAL byte offset + * captured from the old file is meaningless afterwards — not stale-but- + * readable, but pointing at an arbitrary byte of a different file. + * + * `resident: keys` stores exactly such an offset per row and reads rows back + * through it. So the loop below, which knows each record's NEW position as it + * writes it, MUST also rebuild that map. It is the cheap direction and the only + * one that keeps both features usable together; the alternative is forbidding + * compaction whenever such a table is live, which would mean the feature for + * huge tables is incompatible with the feature that stops their log growing. + * + * Nothing fails today because that storage half does not exist yet. It will + * fail later, and it will look like data corruption rather than a design gap. + * ==========================================================================*/ +int wo_wal_compact(wo_wal *w, wo_db *db) { + /* staged records would be written into a file about to be replaced */ + if (!w->path || w->len != 0) return -1; + uint64_t t0 = mono_us(); + + char tmp[4096]; + if ((size_t)snprintf(tmp, sizeof tmp, "%s%s", w->path, WO_WAL_TMP_SUFFIX) >= sizeof tmp) + return -1; + (void)unlink(tmp); /* a stale one would otherwise be appended to */ + + wo_wal nw; + /* THE REPLACEMENT MUST BE PREALLOCATED LIKE THE ORIGINAL. The WAL is + * preallocated so appends never extend the file, which is precisely what + * makes fdatasync sufficient as the ack barrier — no file-size metadata + * has to reach disk for an acked record to be readable. Opening the + * replacement with prealloc 0 silently removed that property, and the + * crash battery caught it: records acked shortly before a kill went + * missing, with the log otherwise intact and self-consistent. */ + if (wo_wal_open(&nw, tmp, w->prealloc) != 0) return -1; + + /* one INSERT per live row, in the existing grammar, through the existing + * append path — so replay needs no second decoder and ids are preserved + * exactly (wo_wal_append_insert takes the id and reads the row) */ + uint32_t pending = 0; + for (uint32_t cid = 0; cid < db->class_cnt; cid++) { + db_table *t = &db->tables[cid]; + if (!t->slabs) continue; /* tables are created lazily */ + uint32_t total = t->slab_cnt * DB_SLAB_ROWS; + for (uint32_t g = 0; g < total; g++) { + if (!(t->bitmap[g >> 6] & (1ull << (g & 63)))) continue; + db_row *r = (db_row *)(t->slabs[g / DB_SLAB_ROWS] + + (size_t)(g % DB_SLAB_ROWS) * t->row_size); + if (wo_wal_append_insert(&nw, db, cid, r->id) != 0) goto fail; + if (++pending >= WO_WAL_COMPACT_FLUSH) { + if (wal_write_nosync(&nw) != 0) goto fail; + pending = 0; + } + } + } + if (wal_write_nosync(&nw) != 0) goto fail; /* the tail batch */ + /* THE dump's one and only barrier: everything above is just bytes in the + * page cache until this, and nothing reads the temp before the rename. */ + if (fsync(nw.fd) != 0) goto fail; + + uint64_t new_bytes = nw.off; + wo_wal_close(&nw); + + /* THE SWITCH. Every crash point either side of this is safe. */ + if (rename(tmp, w->path) != 0) { + (void)unlink(tmp); + return -1; + } + sync_parent_dir(w->path); + + /* the old descriptor now refers to an unlinked inode */ + if (w->fd >= 0) close(w->fd); + w->fd = open(w->path, O_RDWR); + if (w->fd < 0) return -1; /* the log is correct on disk; this process cannot go on */ + w->off = new_bytes; + w->len = 0; + w->compacted_bytes = new_bytes; + { /* the stop-the-world pause: nothing was served while this ran */ + uint64_t el = mono_us() - t0; + w->stat_compactions++; + w->stat_compact_us_total += el; + if (el > w->stat_compact_us_max) w->stat_compact_us_max = el; + } + return 0; + +fail: + wo_wal_close(&nw); + (void)unlink(tmp); + return -1; /* the live log is untouched and still usable */ +} + static int apply_record(wo_db *db, const uint8_t *payload, uint32_t len) { rbuf r = {payload, payload + len, 0}; uint8_t kind = rd_u8(&r); diff --git a/database/src/wal.h b/database/src/wal.h index 74b96b4..eca4353 100644 --- a/database/src/wal.h +++ b/database/src/wal.h @@ -47,10 +47,38 @@ enum { WO_WAL_INSERT = 1, WO_WAL_REMOVE = 2, WO_WAL_UPDATE = 3 }; typedef struct wo_wal { int fd; + /* databasev2 4: where this WAL lives, so a durability failure can name + * the file it could not write. An abort diagnostic without the path + * sends an operator hunting. Owned here, freed by wo_wal_close. */ + char *path; uint64_t off; /* next write offset (the intact tail) */ /* staged batch: appended by wal_append_*, flushed by wal_commit */ uint8_t *buf; size_t len, cap; + /* databasev2 4: group-commit diagnostics. Batching is worthless if + * batches are always one, and a throughput change would then have come + * from somewhere else — so the mechanism is measured, not assumed. + * peak_staged also settles whether the batch needs a cap with a number + * instead of a guess. Reported at exit under WO_WAL_STATS. */ + uint64_t stat_batches; /* non-empty commits */ + uint64_t stat_records; /* records those commits carried */ + uint64_t stat_peak_batch; /* most records in one barrier */ + uint64_t stat_peak_staged; /* most bytes staged behind one barrier */ + /* databasev2 3: bytes the last compaction wrote. The trigger compares the + * log against THIS rather than an estimate of the live set — estimating + * would mean estimating Text, and the compactor knows the true number. */ + uint64_t compacted_bytes; + /* databasev2 3: the preallocation this log was opened with. Compaction + * MUST give the replacement the same one: the WAL is preallocated so that + * appends never extend the file, which is what lets fdatasync alone be the + * ack barrier. A replacement without it silently weakens durability. */ + uint64_t prealloc; + /* databasev2 3: what compaction actually did, reported under WO_WAL_STATS. + * The PAUSE is the number the spec refused to assume — compaction is + * stop-the-world, so its duration is the cost being weighed. */ + uint64_t stat_compactions; + uint64_t stat_compact_us_max; + uint64_t stat_compact_us_total; } wo_wal; /* databasev2 2: the file offset the NEXT staged record will occupy. @@ -90,10 +118,94 @@ int wo_wal_append_remove(wo_wal *w, uint32_t class_id, uint64_t id); * later optimization, recorded). Call AFTER the RAM update. */ int wo_wal_append_update(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id); +/* databasev2 4: which half of the barrier failed. A pwrite failure and an + * fdatasync failure are different operational problems (a short write vs a + * device refusing the flush), so the diagnostic must name the right one. */ +#define WO_WAL_ERR_WRITE (-1) +#define WO_WAL_ERR_SYNC (-2) + +/* The process exit status for a durability failure. + * + * 74 is sysexits' EX_IOERR, chosen deliberately over a small number: 1 is a + * trap and 2 is a loader refusal, but 3 and 4 are already used by SAMPLES for + * their own meanings — db-bench's own `verify` exits 3 on a checksum mismatch, + * and it is the gate that exercises durability, so a durability abort exiting 3 + * would have been indistinguishable from the mismatch it is supposed to help + * diagnose. The low range belongs to programs; the runtime takes a high one. */ +#define WO_EXIT_DURABILITY 74 + /* Write the staged batch and fdatasync — the ack line. Empty batch = ok, - * no syscall. 0 ok, -1 write/sync failure (the batch stays staged). */ + * no syscall. 0 ok, WO_WAL_ERR_WRITE / WO_WAL_ERR_SYNC on failure (the + * batch stays staged: a failed commit consumes nothing). */ int wo_wal_commit(wo_wal *w); +/* databasev2 3: the checkpoint trigger, as a PURE decision so it can be tested + * without a store — which is the only way a policy like this gets tested at all. + * + * [used] the log's used bytes; [last] what the LAST compaction wrote (0 if it + * has never run); [floor] the size below which compacting is not worth it; + * [ratio] the multiple of [last] that counts as too much history. + * + * The denominator is the last compaction's MEASURED output rather than an + * estimate of the live set: estimating would mean estimating Text, and the + * compactor already knows the true number. + * + * There is deliberately NO TIME component. Postgres' CheckPointTimeout exists + * to bound data loss from unflushed buffers; our records are durable at commit, + * so a checkpoint only reclaims space and shortens boot. An idle log does not + * grow, so a timer would fire with nothing to do. + * + * 1 = compact now, 0 = leave it. */ +int wo_wal_should_compact(uint64_t used, uint64_t last, uint64_t floor, uint32_t ratio); + +/* Defaults, overridable at boot by WO_CHECKPOINT_BYTES / WO_CHECKPOINT_RATIO. + * The knobs are what make the policy testable: a test sets a tiny floor and + * forces compaction in a few writes instead of waiting for megabytes. */ +extern uint64_t wo_wal_ckpt_floor; +extern uint32_t wo_wal_ckpt_ratio; + +/* databasev2 3: the temporary file compaction writes before the swap. Named + * next to the log so it lands on the same filesystem — rename(2) is only + * atomic within one. Boot removes a stale one (a crash before the rename). */ +#define WO_WAL_TMP_SUFFIX ".compact" + +/* databasev2 3: rewrite the log as one INSERT record per LIVE row, then swap + * it in with rename(2). + * + * Recovery is deliberately untouched: the result is an ordinary log in the + * ordinary grammar, replayed from byte 0. Crash safety comes from rename being + * atomic — before it the live log is intact and the temp file is not + * authoritative; after it the new log is complete. There is no window in which + * a reader sees a mixture, so this needs no recovery logic of its own. + * + * REFUSES if anything is staged (returns -1 without touching the log): those + * records would be written into a file about to be replaced. Callers must + * invoke this only where the staging buffer is empty — right after a barrier. + * + * A failure is a MISSED OPTIMISATION, not a durability event: the original log + * is left usable and the process keeps running. It must not take the fatal + * path wo_wal_commit_fatal takes. + * + * 0 ok, -1 on any failure. */ +int wo_wal_compact(wo_wal *w, wo_db *db); + +/* databasev2 4: a record could not even be STAGED (the row is already in + * RAM, so this is the same unrecoverable position as a failed barrier — see + * wo_wal_commit_fatal). Never returns. */ +void wo_wal_stage_fatal(const wo_wal *w); + +/* databasev2 4: commit, or END THE PROCESS. + * + * The one rule this iteration introduces: once a statement has mutated RAM, + * the only outcomes are durable or process death. Retrying is not an + * alternative — on Linux a failed fsync may already have discarded the dirty + * pages, so a second call can report success having written nothing. The + * recovery that works is replay, which returns the last durable state. + * + * [nrec] is the number of records in the batch, for the diagnostic only. + * Returns on success; never returns on failure. */ +void wo_wal_commit_fatal(wo_wal *w, uint32_t nrec); + /* Boot replay: apply every intact record to [db] in order. Ids re-enter * exactly as logged; each table's next_id advances past the replayed ids * that belong to this shard. Returns the number of records applied, or -1 diff --git a/docs/00-dependency-graph.md b/docs/00-dependency-graph.md index 4f02db7..ca86ea3 100644 --- a/docs/00-dependency-graph.md +++ b/docs/00-dependency-graph.md @@ -128,6 +128,7 @@ flowchart TD classDef rt fill:#8250df,color:#fff,stroke:none classDef gated fill:#eac54f,color:#000,stroke:none classDef v2 fill:#0969da,color:#fff,stroke:none + classDef done fill:#1a7f37,color:#fff,stroke:none I7b2["7b per-shard collector (done — the precondition 8 waited on)"]:::rt I8x["8 shard-actor runtime: thread-per-core, ownership-move messages"]:::rt @@ -140,7 +141,7 @@ flowchart TD STREAM2["request body streaming + backpressure"]:::gated SRESP2["streaming responses + explicit commit point"]:::gated CANCEL2["per-request cancellation propagation"]:::gated - PUBSUB2["pub/sub + WebSockets (rejected until here)"]:::gated + PUBSUB2["DONE 2026-08-27 — pub/sub + WebSockets (iteration 24: ws_accept + wsframe + room actors)"]:::done ASYNC9C["20 async attach statements (rejected-for-now alternative)"]:::gated TIMEOUTS2["idle timeouts become schedulable (net seam still needed)"]:::gated diff --git a/docs/2026-08-27-chat-drain-finding.md b/docs/2026-08-27-chat-drain-finding.md new file mode 100644 index 0000000..f1268ee --- /dev/null +++ b/docs/2026-08-27-chat-drain-finding.md @@ -0,0 +1,93 @@ +# Iteration 24 T9 — the drain bug the gate was hiding + +**Found 2026-08-27** while finishing T8/T9 on branch `chat-ws-lifecycle`. +Not fixed: the fix is an engine-level decision, recorded here so it is not +rediscovered. + +## The symptom + +`just chat`'s drain leg asserts both connected clients receive a WebSocket +close frame on `SIGTERM`. Against a **fresh** server it is flaky: + +| Sample | Result | +| --- | --- | +| 5 fresh servers, 2 clients each | 4 × `close\|close`, 1 × `eof\|close` | +| 12 fresh servers | 3 failures, one of them `eof\|eof` | +| 16 fresh servers | 5 failures | + +A failing client's socket reaches EOF with **no close frame and no +diagnostic** — the process exits and the kernel closes the fd. + +## Why the gate never caught it + +The drain leg did not start its own server. It inherited `$SRV` from the soak +leg — a server the soak had already pushed 1000 clients through, so every +shard was warm and every actor already scheduled. Draining a warm server hides +the cold-start race. Fixed in this change: **every leg now starts its own +server**, which is what exposed the bug. + +## Root cause, traced + +Instrumented the sample's actors (diagnostics not committed) and correlated +against failing runs: + +1. `DIAG registry-shutdown rooms=1` — main's `send(reg, kind: 2)` **is** + delivered and the Registry runs. +2. `DIAG room-shutdown` — **never printed on a failing run.** The Room never + processes the `kind: 4` shutdown the Registry sends it. +3. The Writer's close branch never runs for the affected client, so no close + frame is written and the fd is never closed by the Writer. Its + `try net.write_dl(...)` is **not** failing — a diagnostic on that path + printed zero times. +4. A client that *does* get a close frame is usually saved by its own + **Reader** noticing `env.stopping()` and running its tail + (`DIAG reader-tail bob r2=1`), not by the room broadcast. + +So the drain chain is main → Registry → Room → Writer, three hops across +shards, and **the Room's shard does not reliably adopt its inbox before the +engine stops.** + +## What was ruled out + +- **Not the spin budget.** Replacing `spin < 20000000` with a wall-clock + deadline of 1 s (`time.ticks()`) still failed 2 of 12. More time does not + help, which is the strongest evidence the room's shard is not being + scheduled at all rather than being scheduled late. That change was reverted: + it fixed nothing and cost a fixed 1 s on every shutdown. +- **Not `dummy_writer()` spawning during shutdown.** Hoisting it to a + Registry field spawned once at startup left 5 of 16 failing. +- **Not a write failure.** See point 3. + +## The decision this needs + +`main` cannot park after the stop flag (a park unwinds), so it spins — and +spinning is not a barrier. Either: + +- **the engine drains pending inboxes before stopping**, so a `send` issued + before the stop flag is guaranteed delivered; or +- **the sample gets a real barrier** — the drain is acknowledged back to main, + which requires main to observe a reply without parking. + +The first is the honest fix and belongs to the actor lifecycle (iteration 31, +absorbed into 24). It is a semantic guarantee — "a send before shutdown is +delivered" — not a tuning parameter, and it should be stated in the runtime's +lifecycle docs and pinned by a corpus fixture, not left to a spin count. + +## Gate defects fixed alongside (all committed) + +1. **fd check was core-count dependent.** `fds_before + 8` read lazy per-shard + init as a leak: shards initialise on first fiber, each taking one + `io_uring` + one `eventfd`, capped at `nproc`. On a 20-core box the first + wave legitimately adds 18. Measured 26 → 44 after 20 clients, then **still + 44 after 40 more**. Replaced with the invariant the check is actually for: + a second wave must not raise the count. Core-count independent, and it + catches a slow leak that any fixed slack would hide. +2. **A failed leg orphaned its server.** The drain leg's python died on + `int("")` when `$SRV` was empty, so the soak server was never killed and + its listener broke the *next* run's soak on the same port. `cleanup` now + kills every server a run started, matched on the run's unique temp dir. +3. **Two legs the plan requires were missing** — `WO_SHARDS=1` (the + single-shard control that says a failure is placement's fault) and + `WO_MAILBOX=8` (the drop-slow-member backpressure path). Both added, both + green. The mailbox leg manufactures a genuinely slow member by shrinking + its `SO_RCVBUF`, so it needs no sleeps. diff --git a/docs/active-slice-2026-08-23-chat-ws-lifecycle.md b/docs/active-slice-2026-08-23-chat-ws-lifecycle.md deleted file mode 100644 index 88897df..0000000 --- a/docs/active-slice-2026-08-23-chat-ws-lifecycle.md +++ /dev/null @@ -1,60 +0,0 @@ ---- -slice: "24" # the story that owns the status; see stories/24-chat-websocket-workload.md -status: in-progress ---- - -# Active slice — chat + actor lifecycle (iteration 24, absorbing 31 + 34) - -Branch `chat-ws-lifecycle`. Spec: -[`superpowers/specs/2026-08-23-chat-websocket-actor-lifecycle-design.md`](superpowers/specs/2026-08-23-chat-websocket-actor-lifecycle-design.md) -· plan: -[`superpowers/plans/2026-08-23-chat-ws-lifecycle.md`](superpowers/plans/2026-08-23-chat-ws-lifecycle.md) -· board: [`stories/00-status.md`](stories/00-status.md). - -## Progress (2026-08-23) - -- ✅ **T1 crypto** (`d14fa9f`): sha1/sha256/hmac_sha256, ids 85–87, RFC - vectors 18/0, corpus pin. Story 34's C-builtin resolution delivered. -- ✅ **T2 bounded mailboxes** (`92754a8`): cap 1024 + `WO_MAILBOX`, - sender-side atomic reserve, WO_T_ACTOR (trap 13) catchable. Plus a - pre-existing compiler fix: try-arm Text places (bare `e.msg`) now - copy before the arm's scope dies (was ASan use-after-free + SEGV). -- ✅ **T6 WS upgrade** (`79cfa01`): `ws_accept` + accept-key + the - 101 hijack sentinel; plain HTTP byte-identical (web-app 26/26). -- ✅ **T7 frame codec** (`7ad2ced`): pure-`.wo` RFC 6455 parse/serialize, - probe-verified against the RFC's own bytes. -- ✅ **T3 call/reply** (`ed69841`): `call` parks + typed scalar reply - (WO-E226 through actor-M erasure); actor DEATH landed with it — - callers never hang (mid-call + to-dead both trap catchably). Fixed - TRAPF's fiber-death leak/dangle en route. - -Every landed task: full battery 12/12, fresh-built. - -## Pending - -- ⬜ **T4 monitor(watched, observer, msg)** — id 89. Most of the death - machinery exists (`actor_die`); T4 adds the per-actor monitor list, - the death walk delivering the observer's own M-typed notice, - monitor-of-already-dead firing immediately, full-observer notice = - disclosed stderr drop. Three-argument form (spec deviation, disclosed - in the plan: the caller may be `main`, which has no mailbox). -- ⬜ **T5 time.after(ms, addr, msg)** — id 90, one-shot, no cancel; - rides the T4 deadline plumbing; delivery = runtime send (full = drop - + stderr line, dead = silent). Corpus: timer-delivery, - timer-generation (the cancel idiom). Both WO_IO backends. -- ⬜ **T8 chat sample** — docs/examples/chat: registry (`call`'s first - consumer), room actors (cap-trap drops slow members, `monitor` reaps - dead writers), reader/writer actor pair per connection over - ws_accept/wsframe; SIGTERM close choreography. -- ⬜ **T9 chat gate** — scripts/chat-accept.sh + raw-RFC6455 python - client; the spec's five checks (functional cross-shard — also the - deferred cross-shard `call` proof — handshake vector, 1k soak with a - `WO_MAILBOX=8` sub-run, drain under both backends + ASan, battery). -- ⬜ **T10 closeout** — stories 24/31/34 → done/ with banners (note the - scalar-reply v1 narrowing + three-argument monitor deviations), board - standup entry, graph nodes, framework README ledger rows, runtime + - chat CODE-LOGIC sections, delete this marker. Final battery. - -This file is deleted when the slice lands (board convention). It lives flat in -`docs/` rather than a status folder — since 2026-08-26 no directory in this repo -encodes state; `status:` above is the only place it is recorded. diff --git a/docs/examples/chat/CODE-LOGIC.md b/docs/examples/chat/CODE-LOGIC.md new file mode 100644 index 0000000..17cd5a9 --- /dev/null +++ b/docs/examples/chat/CODE-LOGIC.md @@ -0,0 +1,82 @@ +# `docs/examples/chat` — how the sample is put together + +Iteration 24's acceptance workload: rooms, presence and broadcast over +WebSocket, actors on fibers across shards, one binary, no broker. It exists to +*drive* the actor work, so nearly every shape here is chosen to exercise +something the runtime claims. + +Gate: `just chat` (`scripts/chat-accept.sh`), which logs to `/tmp/chat.log` — +`tail -F` it while the gate runs. + +## The actors + +| Actor | Owns | Answers | +| --- | --- | --- | +| `Registry` | name → room map, a fallback room | a `call` returning the room's address; spawns rooms on demand | +| `Room` | its member list (writer address + name) | join, leave, a text line, shutdown | +| `Reader` | the read half of one connection | nothing — it loops on the fd and sends onward | +| `Writer` | the **fd**, and the write half | text, pong, close | +| `ConnWorker` | one accepted connection | runs the HTTP layer over that fd | + +`Registry` is the first honest consumer of `call`: the handler runs on the +connection worker's shard, the registry lives wherever placement put it, and +the reply is a scalar — the room's address. That is the cross-shard `call` +proof the gate asserts, not a contrivance added for it. + +## Two actors per connection, not one + +One fd, two directions, and they block independently. A single actor would have +to be inside `read` to notice the client, and inside `write` to deliver a +broadcast — it cannot be in both, so a broadcast would stall behind a quiet +client's read. Splitting them buys three things: + +1. **The `Writer` is the sole writer of that fd.** Frames can never interleave, + which for a framed protocol is a correctness property and not a nicety. +2. **The `Reader` may block as long as it likes.** It sits in `read_dl` with a + 30 s idle deadline and nothing else is waiting on it. +3. **The `Writer`'s mailbox becomes the backpressure point.** A slow client + stops draining its socket, its `Writer` blocks in `write_dl`, its mailbox + fills, and the room's next broadcast to it raises a catchable `WO_T_ACTOR`. + The room catches that and drops the member. **This is the whole reason the + mailbox cap is fail-fast** — the room survives its slowest member, and the + gate's `WO_MAILBOX=8` leg proves the path fires rather than assuming it. + +`Room.say` is written around that: it shifts every member, tries the send, and +keeps only the members whose send succeeded — a failed one is sent a close and +dropped. So fan-out and eviction are the same pass. + +## Who owns the fd + +The `Writer`. It closes it, in every branch: a failed write sets `dead` and +closes; a close message writes the close frame and closes. The `Reader` closes +the fd itself in exactly one case — when its `send_close` to the writer traps, +meaning the writer is unreachable and nobody else will. Without that the fd +would leak on a dead-writer path. + +`Writer.dead` guards against a second close, which matters because two +independent paths can decide a connection is finished (the reader seeing EOF, +and the room broadcasting shutdown). + +## Shutdown choreography + +On `env.stopping()` the accept loop stops and `main` sends one message to the +`Registry`, which fans out to every room; each room shifts its members and +sends each `Writer` a close; each writer writes the close frame and closes the +fd. `main` then spins — it may **not** park, because a park after the stop flag +unwinds — and returns, which is what stops the engine. + +Independently, every `Reader` notices `env.stopping()` at its loop head and +runs its tail: leave the room, close the writer. + +Both paths exist and that is deliberate: the reader path covers a connection +whose room is already gone, the room path covers a reader parked in a read that +has not come back yet. + +**This is where iteration 40 came from.** The room path used to be unreliable: +a `Room` whose shard was idle at `SIGTERM` never adopted the shutdown message, +because an idle worker abandoned its inbox on stop. Clients that still got a +close frame were being saved by the reader path alone — which is why the +failure looked random and why a warmed-up server hid it. The engine now +guarantees that a send issued before the stop flag is delivered, so both paths +work as written. Nothing in this file changed to fix it, and that is the point: +the sample was right and the runtime was not. diff --git a/docs/examples/chat/main.wo b/docs/examples/chat/main.wo new file mode 100644 index 0000000..23993ce --- /dev/null +++ b/docs/examples/chat/main.wo @@ -0,0 +1,335 @@ +-- chat — iteration 24's acceptance workload. Rooms, presence and +-- broadcast over WebSocket: every connection is a reader actor (sole fd +-- reader) plus a writer actor (sole fd writer); rooms and the registry +-- are actors; delivery between them is ownership-moving sends, across +-- shards when placement lands them there. One binary, no broker. +-- +-- CHAT_TOKEN is not needed — chat is open; the framework serves it +-- through [deps] exactly like web-app: +-- woc . && ./target/chat 8080 +-- ws://127.0.0.1:8080/ws?room=lobby&name=alice +-- +-- The actor split exists because an actor takes ONE message at a time: +-- a single per-connection actor blocked in net read could never hear a +-- broadcast. The reader owns the socket's inbound half and the carry +-- buffer; the writer owns the outbound half so frames never interleave. +use env +use net +use time +use porch +use porch/http +use porch/router + +-- ---- message types (one per actor) -------------------------------------- + +-- To a writer: 1 = text frame, 2 = close (frame + fd close), 3 = pong. +class WriterMsg { + kind: Int + text: Text +} + +-- To a room: 1 = join, 2 = leave, 3 = text, 4 = shutdown (drain). +class RoomMsg { + kind: Int + name: Text + text: Text + writer: actor WriterMsg +} + +-- To the registry: 1 = lookup (a `call` — the reply is the room's +-- address), 2 = shutdown every room (a `send` on SIGTERM). +class Lookup { + kind: Int + room: Text +} + +-- To a reader: everything the connection's inbound loop needs. +class ReaderMsg { + fd: net.Conn + room: actor RoomMsg + writer: actor WriterMsg + name: Text +} + +-- One connection accepted, one worker: builds its own App and runs the +-- framework's keep-alive loop (the serving-slice pattern). +class Conn { + fd: net.Conn +} + +-- ---- the writer: sole owner of the outbound half ------------------------- + +class Writer { + fd: net.Conn + dead: Int + fn receive(msg: WriterMsg) { + if self.dead == 1 { return; } + if msg.kind == 1 { + let ok = try net.write_dl(self.fd, ws_text(msg.text), 2000) catch (e) false; + if ok == false { + -- a stalled or gone client: tear the fd; the reader will see EOF + -- and route the leave through the room + self.dead = 1; + net.close(self.fd); + } + return; + } + if msg.kind == 3 { + let ok2 = try net.write_dl(self.fd, ws_pong(msg.text), 2000) catch (e) false; + if ok2 == false { + self.dead = 1; + net.close(self.fd); + } + return; + } + -- close: the drain path (room shutdown or reader-detected close) + self.dead = 1; + let ig = try net.write_dl(self.fd, ws_close(), 1000) catch (e) false; + net.close(self.fd); + } +} + +-- ---- the room: members, presence, fan-out -------------------------------- + +class Mem { + w: actor WriterMsg + name: Text +} + +class Room { + members: multi Mem + fn receive(msg: RoomMsg) { + if msg.kind == 1 { + push(self.members, Mem { w: msg.writer, name: "${msg.name}" }); + self.say("* ${msg.name} joined"); + return; + } + if msg.kind == 2 { + let keep: multi Mem = []; + while len(self.members) > 0 { + let m = shift(self.members); + if m.name != msg.name { push(keep, m); } + } + self.members = keep; + self.say("* ${msg.name} left"); + return; + } + if msg.kind == 3 { + self.say("${msg.name}: ${msg.text}"); + return; + } + -- shutdown: every member gets a close frame; the list empties + while len(self.members) > 0 { + let m = shift(self.members); + let r = try send_close(m.w) catch (e) 0; + } + } + + -- fan-out one line; a member whose mailbox is FULL is a slow client — + -- the fail-fast cap turns it into a drop-from-the-room (the backpressure + -- policy earning its keep) + fn say(line: Text) { + let keep: multi Mem = []; + while len(self.members) > 0 { + let m = shift(self.members); + let ok = try send_text(m.w, "${line}") catch (e) 0; + if ok == 1 { + push(keep, m); + } else { + let r = try send_close(m.w) catch (e) 0; + } + } + self.members = keep; + } +} + +-- send wrappers: `try` is an expression, so give it Int results +fn send_text(w: actor WriterMsg, line: Text) -> Int { + send(w, WriterMsg { kind: 1, text: line }); + return 1; +} + +fn send_close(w: actor WriterMsg) -> Int { + send(w, WriterMsg { kind: 2, text: "" }); + return 1; +} + +-- ---- the registry: name -> room, spawn on demand -------------------------- + +class RoomRef { + r: actor RoomMsg +} + +class Registry { + rooms: map + fallback: actor RoomMsg + fn receive(msg: Lookup) -> actor RoomMsg { + if msg.kind == 2 { + for k, v in self.rooms { + send(v.r, RoomMsg { kind: 4, name: "", text: "", writer: dummy_writer() }); + } + return self.fallback; + } + if has(self.rooms, msg.room) == 1 { + let have = self.rooms[msg.room]; + if have != nil { + return have.r; + } + } + let room: actor RoomMsg = spawn Room { members: [] }; + self.rooms[msg.room] = RoomRef { r: room }; + return room; + } +} + +-- RoomMsg requires a writer field on every construction; the shutdown +-- message has no meaningful one, so a throwaway satisfies the shape (it +-- never receives anything — kind 4 reads no fields). +fn dummy_writer() -> actor WriterMsg { + let w: actor WriterMsg = spawn Writer { fd: 0 - 1, dead: 1 }; + return w; +} + +-- ---- the reader: sole owner of the inbound half --------------------------- + +class Reader { + pad: Int + fn receive(msg: ReaderMsg) { + let carry = ""; + let alive = true; + while alive { + if env.stopping() { alive = false; continue; } + let got = try net.read_dl(msg.fd, 4096, 30000) catch (e) nil; + if got == nil { + -- idle deadline or I/O trap: this client is done + alive = false; + continue; + } + let bytes = "${got}"; + if len(bytes) == 0 { + alive = false; + continue; + } + carry = carry .. bytes; + let more = true; + while more { + let f = ws_parse(carry); + if f.kind == 0 { + more = false; + continue; + } + carry = f.rest; + if f.kind == 1 { + send(msg.room, RoomMsg { kind: 3, name: "${msg.name}", text: f.payload, writer: msg.writer }); + continue; + } + if f.kind == 9 { + send(msg.writer, WriterMsg { kind: 3, text: f.payload }); + continue; + } + if f.kind == 10 or f.kind == 2 { + continue; -- pongs ignored; binary tolerated (echo is not chat) + } + -- close frame or protocol error: stop reading + alive = false; + more = false; + } + } + -- the tail sends must survive full mailboxes (a leave storm after a + -- mass close): a trap here would kill the reader and orphan the fd + let r1 = try send_leave(msg.room, "${msg.name}", msg.writer) catch (e) 0; + let r2 = try send_close(msg.writer) catch (e) 0; + if r2 == 0 { + -- the writer is unreachable (full/dead): close the fd ourselves + net.close(msg.fd); + } + } +} + +fn send_leave(room: actor RoomMsg, name: Text, w: actor WriterMsg) -> Int { + send(room, RoomMsg { kind: 2, name: name, text: "", writer: w }); + return 1; +} + +-- ---- HTTP: the upgrade route + usage -------------------------------------- + +class WsRoute { + reg: actor Lookup + fn handle(req: Req) -> Resp { + if ws_upgrade_valid(req) == false { + return bad_request("expected a websocket upgrade"); + } + let rname = req.query["room"]; + if rname == nil { return bad_request("expected ?room=&name="); } + let who = req.query["name"]; + if who == nil { return bad_request("expected ?room=&name="); } + -- the cross-shard call: this handler runs on the connection worker's + -- shard, the registry lives wherever placement put it + let room = call(self.reg, Lookup { kind: 1, room: "${rname}" }); + let fd = ws_accept(req); + let w: actor WriterMsg = spawn Writer { fd: fd, dead: 0 }; + let rd: actor ReaderMsg = spawn Reader { pad: 0 }; + send(room, RoomMsg { kind: 1, name: "${who}", text: "", writer: w }); + send(rd, ReaderMsg { fd: fd, room: room, writer: w, name: "${who}" }); + return hijacked(); + } +} + +class Usage { + pad: Int + fn handle(req: Req) -> Resp { + return ok_json("{\"ws\":\"/ws?room=&name=\"}"); + } +} + +fn build_app(reg: actor Lookup) -> App { + let app = App { middleware: [], routes: [] }; + app.get("/", Usage { pad: 0 }); + app.get("/ws", WsRoute { reg: reg }); + return app; +} + +class ConnWorker { + reg: actor Lookup + fn receive(msg: Conn) { + let app = build_app(self.reg); + app.handle_conn(msg.fd, 10000, 10000); + } +} + +fn main(args: multi Text) -> Int { + if len(args) < 1 { + print_err("usage: chat "); + return 2; + } + let port = parse_int(args[0]); + if port == nil { + print_err("chat: must be a number"); + return 2; + } + let fb: actor RoomMsg = spawn Room { members: [] }; + let reg: actor Lookup = spawn Registry { rooms: {}, fallback: fb }; + let srv = net.listen("127.0.0.1", port); + print("listening on 127.0.0.1:${port}"); + while true { + if env.stopping() { + -- the drain: every room broadcasts a close frame and writers flush. + -- main must NOT park here (a park after the stop flag unwinds), so + -- it SPINS — each loop back-edge pays a reduction, and the budget + -- hands the shard to the draining actors between slices; worker + -- shards keep adopting their inboxes until the engine stops. + send(reg, Lookup { kind: 2, room: "" }); + let spin = 0; + while spin < 20000000 { + spin = spin + 1; + } + net.close(srv); + return 0; + } + let c = net.accept_dl(srv, 250); + if c != nil { + let w: actor Conn = spawn ConnWorker { reg: reg }; + send(w, Conn { fd: c }); + } + } +} diff --git a/docs/examples/chat/wo.toml b/docs/examples/chat/wo.toml new file mode 100644 index 0000000..4b8be1c --- /dev/null +++ b/docs/examples/chat/wo.toml @@ -0,0 +1,9 @@ +name = "chat" +version = "0.1.0" +description = "Iteration 24's acceptance workload: rooms + presence + broadcast over WebSocket — actors on fibers across shards, one binary, no broker" + +[runtime] +wo = ">= 0.1" + +[deps] +porch = { git = "https://github.com/shoneyj/porch", rev = "v0.1.0" } diff --git a/docs/examples/db-actor/README.md b/docs/examples/db-actor/README.md index ae5a252..e688d27 100644 --- a/docs/examples/db-actor/README.md +++ b/docs/examples/db-actor/README.md @@ -56,7 +56,8 @@ place it runs. feature. - **Why `main` waits.** `main` is not an actor and has no mailbox, so it sleeps rather than awaiting — the gap iteration 31's `call` closes for actors and - [24's marker](../../active-slice-2026-08-23-chat-ws-lifecycle.md) tracks. + [iteration 24](../../stories/language-runtime-database/24-chat-websocket-workload.md) + landed 2026-08-27. Reasoning under the engine side: [`database/src/CODE-LOGIC.md`](../../../database/src/CODE-LOGIC.md). Contract: [`plan/oop-vm/04-db-binding.md`](../../plan/oop-vm/04-db-binding.md). diff --git a/docs/examples/db-bench/README.md b/docs/examples/db-bench/README.md index 13578a6..76c8d93 100644 --- a/docs/examples/db-bench/README.md +++ b/docs/examples/db-bench/README.md @@ -24,9 +24,26 @@ strictly better. Recorded as a plan deviation.) | `query N` | full equality probes on the k index (≈10 rows each), materialized and counted. | | `write N` | alternating inserts (disjoint k range 2e6+) and updates through query results. Corrupts the checksum by design — durability legs run on a fresh store. | | `wal N` | the crash battery's vehicle: insert-only (k range 1e6+), `acked ` printed AFTER each insert returns — the return IS the ack (RAM applied, WAL record staged, ONE commit done). | +| `wmix N C` | **databasev2 4:** every op a durable write (update through a query result), C at once. Exists because `mix` writes on one op in ten with C=4 — 20 writes in a quick run, measured mean batch **1.01** — so no existing leg could show whether group commit engages. Histogram kind 2, because a replayed store still holds the seeding run's kind-0/1 `Hist` rows. Seed first. | +| `boot` | **databasev2 3:** does NOTHING. With `WO_DATA` set the runtime replays the whole log before `main` runs, so a mode with no work of its own is the only honest way to price boot | | `verify` | store vs its own Meta rows: count, checksum, one unique probe. Exit 3 on mismatch. | | `verify-acked M` | after kill -9 mid-`wal`: rows 1..M exist with the right v; rows beyond M allowed (acked after the last print flushed). Exit 3 on mismatch. | +## Env knobs + +| var | effect | +| --- | --- | +| `WO_DATA=` | durability on: replay `/shard-0.wal` at boot, log every write. Without it the store is RAM-only | +| `WO_SHARDS=` | shard count. **`1` means every statement runs inline on shard 0 and group commit cannot engage** — batches form only where writes queue from other shards | +| `WO_CHECKPOINT_BYTES` / `WO_CHECKPOINT_RATIO` | **databasev2 3:** the checkpoint trigger — the log must exceed the floor AND exceed the ratio times the last compaction's own size. A tiny floor forces compaction in a few writes, which is how the gate tests the policy at all; an enormous one disables it, which is how the checkpoint leg measures the same workload with and without | +| `WO_WAL_STATS=1` | **databasev2 4:** print one line at exit — `walstats batches=… records=… peak_batch=… peak_staged=… compactions=… compact_us_max=… compact_us_total=… compacted_bytes=…`. Opt-in so it does not pollute every durable program's output. Mean batch is `records/batches`; **mean 1.0 means group commit is not engaging**, which is expected for a serial writer or `WO_SHARDS=1` and a bug anywhere else | + +**Do not put `WO_DATA` on `/tmp`.** It is `tmpfs` on the reference machine, +where `fdatasync` is free: the same `wmix` run measured **195 000 ops/s at p50 +1 µs** there against **2200 ops/s at p50 7200 µs** on ext4. There is no +durability barrier to price on a memory filesystem. The driver keeps its stores +under `bench/` for exactly this reason. + ## Coordination idiom (this side of iteration 31) There is no request/response surface yet: concurrent modes drive diff --git a/docs/examples/db-bench/main.wo b/docs/examples/db-bench/main.wo index 9b946b0..548da08 100644 --- a/docs/examples/db-bench/main.wo +++ b/docs/examples/db-bench/main.wo @@ -338,6 +338,91 @@ class Mixer { } } +-- databasev2 4 part A: every op a durable write, C at once. +-- +-- Why this leg exists. `mix` writes on one op in ten with C=4, so at most a +-- handful of writes are ever in flight and group commit has almost nothing to +-- batch: measured mean batch 1.01 over 3112 barriers, peak 3. That is a +-- property of the WORKLOAD, not of the mechanism, and without a write- +-- concurrent leg the iteration's payoff cannot be evaluated either way. +-- +-- Updates rather than inserts: comparable to what `mixwrite` measures, and the +-- row count stays flat so a long run does not turn into a growth test. +-- Histogram kind 2, because a replayed store still holds the seeding run's +-- kind-0/1 Hist rows and merging those would report someone else's latencies. +class WJob { + ops: Int + seed: Int + kmod: Int +} + +class WMixer { + id: Int + fn receive(msg: WJob) { + let hw: map = {}; + let s = msg.seed; + let i = 0; + while i < msg.ops { + s = lcg(s); + let key = s % msg.kmod; + let o0 = time.ticks(); + for r in from x in Item where x.k == key take 1 select x { + r.v = r.v + 1; + } + hist_add(hw, time.ticks() - o0); + i = i + 1; + } + hist_dump(hw, 2); + insert Meta { tag: "wmixdone${self.id}", val: msg.ops }; + } +} + +fn wmix_mode(total: Int, c: Int) -> Int { + let kmod = meta_val("kmod"); + if kmod < 1 { + print_err("wmix: seed first"); + return 1; + } + let per = total / c; + if per < 1 { + per = 1; + } + let wall0 = time.ticks(); + let i = 0; + while i < c { + let a: actor WJob = spawn WMixer { id: i }; + send(a, WJob { ops: per, seed: 4242 + i * 7919, kmod: kmod }); + i = i + 1; + } + let done = 0; + while done < c { + time.sleep(20); + done = 0; + i = 0; + while i < c { + if meta_val("wmixdone${i}") >= 0 { + done = done + 1; + } + i = i + 1; + } + } + let wall = time.ticks() - wall0; + let hw: map = {}; + let nw = 0; + for x in from x in Hist select x { + if x.kind == 2 { + if has(hw, x.b) { + set(hw, x.b, get(hw, x.b) + x.c); + } else { + set(hw, x.b, x.c); + } + nw = nw + x.c; + } + } + report("wmix", nw, wall, hw); + return 0; +} + fn mix_mode(total: Int, c: Int) -> Int { let kmod = meta_val("kmod"); if kmod < 1 { @@ -466,7 +551,7 @@ fn all_mode(n: Int) -> Int { fn usage() -> Int { print_err("usage: db-bench "); print_err(" all N | seed N | read N | query N | write N | wal N"); - print_err(" mix N C | msgrate N | growth N int|text | growth-verify"); + print_err(" mix N C | wmix N C | msgrate N | growth N int|text | growth-verify"); print_err(" randread N R | replayseed N M | boot"); print_err(" verify | verify-acked M"); return 2; @@ -683,6 +768,10 @@ fn main(args: multi Text) -> Int { if args[0] == "growth-verify" { return growth_verify(); } + -- Does NOTHING. With WO_DATA set the runtime replays the whole log before + -- main runs, so a mode with no work of its own measures replay plus a fixed + -- process start — which is what "boot time" has to mean. Both databasev2 1 + -- (replay baseline) and databasev2 3 (checkpoint boot) price boot with it. if args[0] == "boot" { return boot_mode(); } @@ -746,6 +835,17 @@ fn main(args: multi Text) -> Int { } return randread_mode(n, rr); } + if args[0] == "wmix" { + if len(args) < 3 { + return usage(); + } + let wc = parse_int(args[2]); + if wc == nil or wc < 1 { + print_err("db-bench: must be a positive number"); + return 2; + } + return wmix_mode(n, wc); + } if args[0] == "mix" { if len(args) < 3 { return usage(); diff --git a/docs/examples/porch/README.md b/docs/examples/porch/README.md index 8fad15d..1fb7ee3 100644 --- a/docs/examples/porch/README.md +++ b/docs/examples/porch/README.md @@ -75,7 +75,11 @@ porch = { git = "https://github.com/shoneyj/porch", rev = "v0.1.0" } - **TLS: none, anywhere.** Deploy behind nginx/caddy; the proxy terminates TLS+ALPN and gives browsers HTTP/2 while this backend speaks HTTP/1.1 keep-alive. See the web-app sample's README for the nginx sketch. -- `Content-Length` bodies only (no chunked encoding), no WebSockets/SSE, +- `Content-Length` bodies only (no chunked encoding); **WebSockets ARE + supported since 2026-08-27** — `ws_accept` (`http/ws.wo`) performs the RFC + 6455 handshake and hands back the hijacked `net.Conn`, and `http/wsframe.wo` + is a pure-`.wo` frame codec; `docs/examples/chat` is the worked example and + `just chat` its gate. **SSE is still absent**, and so is chunked encoding. JSON-first (no templates). Form-encoded bodies parse through `form_values(req)` (`+` and `%XX` decoded, nil on any other content-type); multipart/form-data through `multipart_parts(req)` @@ -130,6 +134,7 @@ first (pure `.wo` cannot express it yet). | Content negotiation | ✅ `media_type(req)` request-side; `accepts(req, mtype)` response-side (exact, type/*, */*; q-values stripped not ranked — ranking waits for an app serving alternates) — slice 2 | | Trusted-proxy client IP | 🔶 `client_ip(req)` parses X-Forwarded-For; `net.peer(fd)` (iteration 35) exposes the peer — the verify middleware is now a pure-`.wo` candidate slice | | Status/header setting · redirects | ✅ builders + `set_header` | +| WebSockets · pub/sub | ✅ **2026-08-27 (iteration 24)** — `ws_accept` does the RFC 6455 handshake and hands back the hijacked `net.Conn`; `http/wsframe.wo` is a pure-`.wo` frame codec. Rooms/presence/broadcast are actors in `docs/examples/chat`, gated by `just chat` (11 checks, 1000-client soak, both `WO_IO` backends, ASan clean). No SSE | | Lazy body streaming + backpressure · streaming responses · explicit commit point | ⏸ UNBLOCKED by the arc (8/11 landed 2026-08-21) — stays parked until its own slice | | ETag + conditional requests | ✅ `etag_for` (quoted base64 SHA-256) + `with_etag` (If-None-Match → 304) over iteration 34's digest builtins — slice 2 | @@ -140,7 +145,7 @@ first (pure `.wo` cannot express it yet). | Ordered middleware chain | ✅ registration order, `?Resp` short-circuits | | Request-scoped context | ✅ `req.ctx` map (slice 2): middleware writes, handlers read; identity stays in `principal` | | Guaranteed teardown | 🔶 every fd closes on every path (gate-proven); no user teardown hooks yet | -| Cancellation into pending storage ops | ⏸ UNBLOCKED by the arc (8/11 landed 2026-08-21) — stays parked until its own slice | +| Cancellation into pending storage ops | ⏸ **unblocked, not built.** The arc landed 2026-08-21 and iteration 24 (2026-08-27) added the lifecycle a cancellation would ride — `call` with a catchable trap when the callee dies, bounded mailboxes, `monitor`, and `time.after` for a deadline. Nothing here consumes them yet; it stays parked until its own slice | | Panic recovery | 🔶 trap = 500 and the server survives ✅; "rolls back the transaction" is framework v2 (needs `transaction { }`, iteration 18) | ### Storage integration (the differentiator — framework v2 territory) diff --git a/docs/plan/exploration/postgresql/buffer-and-checkpoint.md b/docs/plan/exploration/postgresql/buffer-and-checkpoint.md index 20023cf..26e9979 100644 --- a/docs/plan/exploration/postgresql/buffer-and-checkpoint.md +++ b/docs/plan/exploration/postgresql/buffer-and-checkpoint.md @@ -29,6 +29,31 @@ Writeonce's phase 12 `Engine` keeps an `HashMap<(TypeName, SegmentOffset), Cache The kernel page cache does most of the work. `pread` against an fd that already has its page cached is a memcpy. `pwrite` populates the page cache without going to disk until pressure or `fsync`. This is why writeonce explicitly does NOT use `O_DIRECT` (see [`linux/12-pwrite-fsync.md`](../linux/12-pwrite-fsync.md)) — the page cache is the one cache we want. ## Checkpoint — the writeonce shape +> **⚠ TWO CORRECTIONS, 2026-08-28** (found while brainstorming +> [databasev2 3](../../../stories/databasev2/03-wal-checkpoint.md); spec: +> [`2026-08-28-wal-checkpoint-design.md`](../../../superpowers/specs/2026-08-28-wal-checkpoint-design.md)). +> +> 1. **Postgres does NOT update its control file by rename.** The claim below +> that "Postgres does the same in `BasicOpenFile` + `fsync_parent_path`" is +> wrong: `update_controlfile` (`src/common/controldata_utils.c`) opens the +> existing file `O_WRONLY`, writes a zero-padded **full block in place**, and +> relies on **CRC32C** over the struct to detect a torn write. The +> `fsync(parent_dir)` reasoning below is still correct *for renames* — it is +> just not what Postgres does here. +> 2. **The checkpoint sketch below assumes writeonce has segment files.** It +> says records before the LSN are "*known* to be in the segment files". There +> are none: the WAL is writeonce's only durable form, replayed into RAM, and +> [databasev2 2](../../../stories/databasev2/02-table-storage-modes.md) +> deliberately rejected adding a paged store. This document predates the +> databasev2 direction, so read the loop below as a design for an +> architecture that was not chosen. +> +> What survived the comparison is the **ordering discipline**, not the +> architecture: publish the new "recovery starts here" atomically and last, so a +> crash falls back. writeonce gets that from one `rename` of the whole log — +> possible only because its records are full row images, where Postgres' are +> page deltas. + Postgres' checkpoint runs in a separate process and signals the postmaster when done. Writeonce's runs as a periodic loop step: diff --git a/docs/plan/oop-vm/00-wob-format.md b/docs/plan/oop-vm/00-wob-format.md index 1e1f557..ff1fcaf 100644 --- a/docs/plan/oop-vm/00-wob-format.md +++ b/docs/plan/oop-vm/00-wob-format.md @@ -65,7 +65,7 @@ The metadata exists for exactly one reason: `json.encode`/`json.decode` are runt - **the OS half** — fs.exists/list/stat/read_all/read_at/append, time.sleep/local/iso, env.get/stopping, net.listen/accept/read/write/close, proc.run. Ids 40–56; `runtime/src/sysio.c`. A member that returns a record takes that record's **class id as its last argument**, so the VM allocates what it fills without knowing any source type name. - **json** — encode (value + the value's static kind), decode (text + the class id to build). Ids 57–58; `runtime/src/json.c`. Decode yields the zero word on malformed input rather than trapping, which is what makes `json.decode(t) as T` a checked decode. - **59 `map_get_opt`** (`m[k]`'s optional read), **60 `text_copy`** (Text's ownership-boundary copy — Task 1 of the executable plan). -- **database** — **61 `db_insert`** (iteration 9, Task 3): window is R[B] = class id, R[B+1..] = one slot per **declared** field in declaration order; result R[A] = the new row's id. The loader validates the class-id slot statically (variable window: the field slots are validated at runtime by the engine against the class table). Engine failure traps `WO_T_DB`; a failed WAL commit traps `WO_T_IO` after un-applying the row. `database/src/db.c`. +- **database** — **61 `db_insert`** (iteration 9, Task 3): window is R[B] = class id, R[B+1..] = one slot per **declared** field in declaration order; result R[A] = the new row's id. The loader validates the class-id slot statically (variable window: the field slots are validated at runtime by the engine against the class table). Engine failure traps `WO_T_DB`. **A failed WAL commit no longer traps (databasev2 4, 2026-08-28): it ENDS THE PROCESS** with exit status 74 and a diagnostic naming the failing operation, the log path, `errno` and the batch size. `WO_T_IO` is unreachable from any DB write. The reason is that only `insert` could ever un-apply itself — `update` and `delete` never could, and their own comments admitted they left RAM ahead of disk — so continuing after a durability failure meant serving state that would not survive a restart. Retrying is not offered either: on Linux a failed `fsync` may already have discarded the dirty pages, so a second call can report success having written nothing. Replay is the recovery that works. `database/src/db.c`, `database/src/wal.c`. **`?T` and nil.** A heap-shaped optional (`?Text`, `?Rec`, `?multi`, `?map`, `?@gc`) stores what `T` stores and spells nil as the **zero word** — every per-kind drop plan already ignores a zero slot, so `?T`'s field kind is `T`'s. A **nullable scalar** (`?Int`, `?Bool`, `?Timestamp`, `?Id`) cannot: `0` is a perfectly good `Int`, and real programs store it in a `?Int`. Its nil is therefore `WO_NIL_SCALAR` = −2^62 (not `INT64_MIN`: the compiler's own integers are 63-bit, so that value is not expressible on the emitting side). Such a field is marked `WOB_FIELD_NIL_SCALAR` in `field_class[i]`, which is how the runtime knows to write that word where it must produce absence itself — today only `json.decode` leaving a key absent, and `parse_int` on unparseable input. diff --git a/docs/plan/oop-vm/04-db-binding.md b/docs/plan/oop-vm/04-db-binding.md index 89dc330..67735d1 100644 --- a/docs/plan/oop-vm/04-db-binding.md +++ b/docs/plan/oop-vm/04-db-binding.md @@ -118,11 +118,49 @@ R[B+1..] = one slot per declared field in declaration order (the literal's order is irrelevant — slots are the class table's). Execution: `wo_row_insert` (RAM, engine copies every value), then — when -durability is on — stage + **commit before the builtin returns**: the -builtin's return IS the acknowledgment, so ack-after-fsync holds at -statement granularity until iteration 8 brings tick-scoped group commit. A -failed commit un-applies the row and traps `WO_T_IO`; engine failures trap -`WO_T_DB`. Durability is opt-in: `WO_DATA=` makes the CLI replay +durability is on — stage, then a barrier before the acknowledgment. **Updated +2026-08-28 (databasev2 4 part A): group commit landed, and the barrier's +location now depends on which path the statement takes.** + +A statement arriving from a worker shard marshals to shard 0 and parks; shard 0 +stages every such request, issues **one** barrier when its queue empties, and +only then releases the held replies — so each writer is acknowledged after the +barrier that carried *its* record. A statement already running on shard 0 takes +the inline path and still commits before the builtin returns, because it has no +reply to hold: it returns into its own fiber, and batching it would require +parking that fiber on the barrier (deferred to part B). The boundary is the +queue draining, **not** the tick this document previously anticipated — a tick +would add latency to a lone writer, taxing an idle system to serve a busy one. + +Measured: ~2.9× durable write throughput and ~2.1× lower p50 on a +write-concurrent workload; unchanged for a serial writer, which has nothing to +batch with. + +**Compaction (databasev2 3, 2026-08-29) may run only where NOTHING IS STAGED.** +That is a correctness requirement, not a scheduling preference: the staging +buffer holds records destined for a file that compaction is about to replace, so +compacting with a non-empty buffer would either write them into a file about to +be discarded or lose them with it. In practice the safe points are immediately +after a barrier — the drain's, and the inline path's — and both are wired. +`wo_wal_compact` refuses a non-empty buffer as a backstop rather than trusting +its callers. + +**Recovery is unchanged by compaction.** The result is an ordinary log in the +ordinary record grammar, replayed from byte 0; there is no snapshot, no second +source, no cutoff offset and no control file. Crash safety comes from `rename` +being atomic: before it the live log is intact and the temp file is not +authoritative, after it the new log is complete, and no reader can observe a +mixture. A crash mid-rewrite leaves a temp file, which the next open removes. + +A failed compaction is a **missed optimisation, not a durability event** — the +original log is left usable and the process continues. It must not take the +fatal path below. + +A failed commit **no longer traps — it ends the process** (exit 74, with a +diagnostic naming the operation, log path, `errno` and batch size). So does a +failed staging. `WO_T_IO` is unreachable from a DB write. One rule: once a +statement has mutated RAM, the outcomes are durable or death. Engine failures +still trap `WO_T_DB`. Durability is opt-in: `WO_DATA=` makes the CLI replay `/shard-0.wal` before the entry runs and commit every insert; without it the engine is RAM-only (every corpus fixture runs that way). diff --git a/docs/plan/oop-vm/08-builtin-surface.md b/docs/plan/oop-vm/08-builtin-surface.md index b49a149..8b624fa 100644 --- a/docs/plan/oop-vm/08-builtin-surface.md +++ b/docs/plan/oop-vm/08-builtin-surface.md @@ -279,6 +279,8 @@ unset `env.get` are nil. | `sha1(bytes)` | `-> Bytes` | 20-byte digest (id 85, iteration 34) — exists because RFC 6455's Sec-WebSocket-Accept demands SHA-1 | | `sha256(bytes)` | `-> Bytes` | 32-byte digest (id 86, iteration 34) | | `hmac_sha256(key, msg)` | `-> Bytes` | RFC 2104 over SHA-256, both args Bytes (id 87, iteration 34); key > 64 bytes hashed first | +| `monitor(watched, observer, msg)` | — | iteration 24 (id 89): the observer's own M-typed msg (MOVED) is delivered when watched dies (trap-death); already-dead delivers now; a full observer's notice is dropped with a stderr line — no fiber to trap | +| `time.after(ms, addr, msg)` | — | iteration 24 (id 90): one-shot timer — msg (MOVED) arrives as an ordinary send after ms on the arming shard; ms <= 0 delivers now; NO cancel — the generation-counter idiom (run/timer-generation) is the answer | | `call(addr, msg)` | `-> R` | send that WAITS (id 88, iteration 24): the message moves like `send`'s, the caller's fiber parks until the receive's return value arrives. R = the receive's declared return type — every `receive(msg: M)` program-wide must agree on it and it must be a copyable scalar in v1 (WO-E226 otherwise). A dead callee traps WO_T_ACTOR, immediately or mid-call — a `call` never hangs | | `env.get(name)` | `-> ?Text` | unset is nil | | `env.stopping()` | `-> Bool` | SIGTERM/SIGINT latch, handlers installed on first use | diff --git a/docs/plan/perf-targets.md b/docs/plan/perf-targets.md index a21870a..191ef41 100644 --- a/docs/plan/perf-targets.md +++ b/docs/plan/perf-targets.md @@ -216,3 +216,187 @@ which is too coarse for the one number a checkpoint is meant to improve. WAL bytes are measured as the file's **non-zero prefix**, never its size: shard WALs are `fallocate`'d to 1 MiB, so an empty store reports 1048576. +## 6. WAL group commit: one barrier per drain (databasev2 4 part A) + +**Measured 2026-08-28.** Before this, the engine committed per *statement*: +`db.c` called `wo_wal_commit` immediately after every append, so each row +change bought its own `pwrite` + `fdatasync`. Now shard 0 stages every queued +write request, issues one barrier, and only then releases the held replies. + +### The controlled before/after + +Same machine, same workload (`wmix 4000 32` — every op a durable update, 32 +concurrent), same build except `db.c` and `vm.c`, two runs each, interleaved: + +| | ops/sec | p50 | p99 | +| --- | --- | --- | --- | +| per-statement barrier | 2213 · 2177 | 7183 · 7251 µs | **20000 · 20000 µs** | +| group commit | **6216 · 6525** | **3458 · 3444 µs** | 11139 · 5971 µs | + +**≈2.9× throughput, ≈2.1× lower p50.** + +**The p99 "before" figure is at the histogram ceiling, not a measurement.** +`hist_add` clamps at 20000 µs, and both before-runs pinned there — so the true +before p99 is ≥20 ms and unknown. The improvement is *at least* 2.3×; the +honest statement is that the old p99 was off the end of the instrument. + +### Confirmation from the committed baseline + +The full campaign gives the same answer a second way. `s1` takes the inline +path, which commits per statement **by design**, so within one build the two +shard configurations are batching-off against batching-on: + +| Leg | ops/sec | p50 | p99 | mean batch | peak batch | +| --- | --- | --- | --- | --- | --- | +| `durable.s1.wmix` (inline, unbatched) | 1467 | 455 µs | 721 µs | **1.0** | 1 | +| `durable.sN.wmix` (batched) | **5117** | 8208 µs | 12169 µs | **5.43** | 57 | + +3.5× throughput, agreeing with the 2.9× above. Note `sN` latency is *higher* +while throughput is 3.5× better: 64 writers queueing behind one owner shard +trade per-op latency for barrier amortisation, which is what group commit is. + +Batching scales with write concurrency exactly as designed — mean batch at +C = 4 / 16 / 64 was **1.13 / 1.76 / 5.35**, peak **3 / 10 / 39**. + +### What did NOT improve, and why that was predicted + +`durable.sN.mixwrite` went **480 → 492 ops/s** — unchanged. That is the metric +the spec *originally* named as the payoff, and correcting it was part of the +brainstorm: `mix` writes on one op in ten with C=4, so a quick run performs +**20 writes** and mean batch measured **1.01** over 3112 barriers. A workload +that never has two writes in flight cannot be helped by batching them. +`durable.*.seed` is likewise unchanged: a serial single writer has nothing to +batch with under any scheme. + +**So the payoff is real but conditional: it appears exactly where concurrent +durable writes fan into the owner shard, and nowhere else.** + +### Two traps worth recording + +**Do not benchmark durability on `/tmp`.** It is `tmpfs` here, where +`fdatasync` is free — the same `wmix` run reported **195 000 ops/s at p50 1 µs** +there against **2200 ops/s at p50 7200 µs** on ext4. There is no barrier to +amortise on a memory filesystem, so a group-commit measurement taken there +measures nothing. `db-bench` gets this right by keeping its stores under +`bench/`. + +**The record count is not the update count.** `wmix` staged 7755 records for +4000 updates because the histogram dump and the done-marker are themselves +durable inserts. They arrive as an end-of-run burst, which is batch-friendly, +so `mean_batch` is not purely update-driven. Peak staged bytes stayed small +(2793 B at C=64), which is what settled the decision to ship **no batch cap**: +the request queue's existing upstream bound is sufficient. + +### The cost side: tail latency on the owner shard + +Group commit is a trade, and the full battery made the other side of it visible. + +**A bug first, caught by `durable.sN.mixread.p99`.** The drain initially held +*every* DB reply until the barrier — including **reads**, which stage nothing and +have no stake in durability. That parked readers behind an fsync for no reason +and pushed read p99 from ~1043 µs to **4057 µs**. Reads are now released +immediately; only a statement that actually staged a record has its reply held. + +**What remains is inherent, not a bug.** A barrier now blocks the owner shard +**longer** (more records per fsync) even though it blocks **less often**, so +anything arriving during a barrier — reads included — waits behind it. Measured +across three full runs of the same build, `durable.sN.mixread.p99` came in at +**1043 / 2318 / 4147 µs** and `wmix.p99` at **8758 / 20000 µs**, a 2–4× spread +with the box near idle. + +So the honest summary of part A on a single-threaded owner shard: **~3× write +throughput, at the price of a longer and noisier tail for everything queued +behind a barrier.** That is precisely what part B (async submission — submit the +barrier and keep serving) would undo, and it is a better argument for part B than +the "close the 66× gap" framing part B was originally given. + +**Gating consequence.** `durable.sN.*.p99us` now carries a 100% tolerance, +because a 2–4×-variable tail gated at 50% gates the disk rather than the engine. +The **floor** is the real guard there, and it is not slack: `mixread`'s floor +(4172 µs) came within 25 µs of tripping on the worst observed run. + +## 7. WAL checkpoint: compaction (databasev2 3) + +**Measured 2026-08-29.** Before this the log grew forever: nothing ever removed +superseded records, so boot replayed all history and the file only ever got +bigger. Compaction rewrites it as one record per live row and swaps it in with +`rename`. + +### Space and boot — the same workload, twice + +Identical work, differing only in whether checkpointing may fire (an enormous +floor disables it). Full campaign: + +| | checkpointing off | checkpointing on | +| --- | --- | --- | +| WAL used | 1 962 358 B | **907 094 B** | +| boot (median of 3, `boot` mode) | 114 ms | **64 ms** | +| compactions | 0 | 6 | + +**2.16× space reclaimed, 1.78× faster boot.** Boot is measured with a mode that +does nothing at all: with `WO_DATA` set the runtime replays the whole log before +`main` runs, so a mode with no work of its own is the only honest way to price +replay. It is *not* measured through the driver's `run()` helper, which samples +RSS on a 250 ms poll — timings taken that way reported "251 ms" both with and +without checkpointing, which is the harness's clock rather than the engine's. + +### The stop-the-world pause, and why it stopped being 8× worse + +Compaction blocks the owner shard for its duration. The spec refused to assume +that was acceptable, so it is measured and gated against a stated **50 ms** +budget: a stall a serving process can absorb without a client seeing a timeout. + +Measured **2 651 µs** on the full campaign — comfortably inside it. + +It was not always. The first implementation flushed the dump through +`wo_wal_commit`, which `fdatasync`s, so a dump paid one barrier per 256 records: + +| live set | pause, per-flush fsync | pause, one final fsync | +| --- | --- | --- | +| ~107 KB | 23 948 µs | **2 903 µs** | +| ~500 KB | 36 361 µs | **7 526 µs** | +| ~1.98 MB | 107 649 µs | **13 212 µs** | + +Marginal rate went from **~22 MB/s to ~181 MB/s** — from sync-bound to +bandwidth-bound. Intermediate durability during a dump is worthless: the temp +file is not authoritative until the rename and is fsynced once immediately +before it, so those barriers bought nothing and cost 8×. + +**The pause is O(live rows), and that is the number that eventually forces an +incremental design.** At ~181 MB/s a 1 GB live set implies roughly 5.5 s — well +past any interactive budget. The spec deliberately did not buy incremental +copying in advance; this is the measurement it is to be bought against. + +### Gating + +`ckpt.reclaim_x` is the feature's central claim and is gated tightly (15%). +Everything else in the leg — boot times, the pause, the byte counts — is +wall-clock or workload-shaped on a shared box and carries a wide tolerance, +because waiving them *all* would have left the leg ungated. The leg also +asserts two things directly rather than trusting a metric: that some compaction +actually ran (otherwise it proves nothing), and that the log really is smaller +with checkpointing on. + +One direction bug worth recording: `reclaim_x` was first recorded as +lower-is-better by the default detector, which would have **passed "reclaimed +nothing" and failed an improvement** — the central claim gated backwards. + +**Gate-tolerance corrections made while closing this iteration**, both recorded +because a widened tolerance that is not justified is indistinguishable from a +silenced regression: + +- **`ckpt.pause_us_max` is no longer gated against a baseline.** The raw pause + scales with the live set, and this workload's live set is not fixed — + `wmix`'s `hist_dump` inserts a row per latency bucket, so a noisier box makes + more buckets, more rows, and a longer pause. What belongs to the engine is the + **rate**, so `ckpt.pause_us_per_mb` carries the real tolerance and the raw + pause keeps the absolute 50 ms budget as its guard. +- **`ram.*.msgrate.msgs_sec` moved from 15% to 70%, and this one is + pre-existing.** Across the ten full runs recorded on 2026-08-28/29 — several + predating the checkpoint work — it ranged **10.7M to 17.9M msgs/sec, a 1.67× + spread**. A 15% gate on a scheduling-bound throughput metric fails + intermittently whatever the engine does. +- **`durable.sN.*.p99us` moved from 100% to 300%**, with more evidence than the + first widening had: mixread p99 measured 1043 / 2318 / 4147 µs and mixwrite + 1623 / 4446 µs across runs of the same build. The floors remain the real + guard, and they are not slack — mixread's came within 25 µs of tripping. diff --git a/docs/stories/00-status.md b/docs/stories/00-status.md index cddef50..419a18b 100644 --- a/docs/stories/00-status.md +++ b/docs/stories/00-status.md @@ -67,6 +67,161 @@ behind this board; live Obsidian Dataview views: ## ▶ NEXT PLAN +### Landed 2026-08-29 — databasev2 3, WAL checkpoint (the chain's last link) + +**Implemented last time (2026-08-29):** compaction. The log used to grow forever +— nothing removed superseded records, so boot replayed all history. It is now +rewritten as one record per live row into a temp file and swapped in with +`rename`. Six tasks, brainstormed and spec'd first +([spec](../superpowers/specs/2026-08-28-wal-checkpoint-design.md) · +[plan](../superpowers/plans/2026-08-28-wal-checkpoint.md)). + +**Key findings (measured, not asserted):** **2.16× space reclaimed** +(1 962 358 → 907 094 B), **boot 114 → 64 ms**, stop-the-world pause **2 651 µs** +against a stated 50 ms budget. Reading `.dev/reference/postgresql` was what made +the design defensible rather than lazy: **Postgres never compacts its WAL**, +because its records are page deltas and a compacted redo log is not a store — +hence heap files, a control file, a redo pointer, a second recovery source and a +separate checkpointer process. Ours are **full row images**, so a compacted log +*is* a complete store, and all of that machinery disappears. What was worth +porting is the ordering discipline — publish the switch atomically and last — and +one `rename` provides it. + +**Learned — two bugs of mine that measurement found, not review:** wiring the +trigger only into the drain left **`WO_SHARDS=1` never compacting**, its log +growing forever (536 KB where multi-shard held 446 KB), because a statement on +the owner shard never enters that drain. And the dump was **8× slower than +necessary**, flushing through the committing path and paying one `fdatasync` per +256 records for durability that is worthless before the rename — one final +barrier took a 2 MB dump from 107 649 µs to 13 212 µs, ~22 MB/s to ~181 MB/s. +Separately, the crash battery's *first* version failed on correct code ~1 run in +3: it acked deletes after committing them, so a kill in between made it demand a +row the engine was right to remove. Deletes now announce intent first. + +**Dependencies unblocked:** every link in the concurrency + fiber chain has now +landed its planned work — stage 3 → 22 → 24 (absorbing 31 + 34) → 40 → +databasev2 4 part A → databasev2 3. **Not "complete", precisely:** chain 5 stays +`in-progress` because databasev2 4's part B was never done, and its premise was +invalidated by part A rather than satisfied. Nothing in the chain is blocked on +anything else in it. + +**Next steps:** the honest queue is (1) databasev2 2's outstanding 5c/5d, whose +`resident: keys` half is unimplemented and now carries a recorded obligation — +compaction invalidates every WAL offset it stores, so the compactor must rebuild +that map; (2) databasev2 4 **part B**, whose premise was invalidated by part A +and which needs re-brainstorming rather than starting; (3) the O(live rows) +pause, ~5.5 s at a 1 GB live set, which is the number an incremental checkpoint +must be bought against. + +**`.dev/reference` used:** `postgresql` — `xlog.c` (`CreateCheckPoint`, segment +recycling), `checkpointer.c` (the time-or-volume trigger), and +`controldata_utils.c`, which also corrected a prior exploration doc: Postgres +updates its control file **in place with a CRC**, not by rename. + +--- + +### Landed 2026-08-28 — databasev2 4 part A, WAL group commit + +**Implemented last time (2026-08-28):** one durability barrier per drain +instead of one per statement. Shard 0 stages every queued write request, holds +each reply, commits once when its queue empties, then releases all — so a writer +is acknowledged after the barrier that carried *its* record, which was the +intended contract all along and was true before only because every batch had one +member. Six tasks, brainstormed and spec'd first +([spec](../superpowers/specs/2026-08-28-wal-group-commit-design.md) · +[plan](../superpowers/plans/2026-08-28-wal-group-commit.md)). + +**Key findings (measured, not asserted):** **≈2.9× durable write throughput, +≈2.1× lower p50** on a write-concurrent workload, confirmed a second way by the +`s1`-vs-`sN` split within one build (1467 → 5117 ops/s, mean batch 1.0 → 5.43, +peak 57) — 2.9× and 3.5× agreeing. Batching scales with contention: mean batch +1.13 / 1.76 / 5.35 at C = 4 / 16 / 64. **The story's premise was wrong**: it +said "fsync-per-commit" and the engine was fsync-per-**statement**, committing +after every append at all six sites — so part A was closer to deleting calls +than adding a mechanism. + +**Learned — three things the measurement corrected, not the code:** +(1) **`/tmp` is tmpfs here, where `fdatasync` is free.** The same run reported +195 000 ops/s at p50 1 µs there against 2200 at 7200 µs on ext4. A group-commit +measurement taken on a memory filesystem measures nothing; `db-bench` is right +to keep its stores under `bench/`. (2) **No existing leg could exercise the +feature** — `mix` writes on one op in ten with C=4, giving 20 writes and mean +batch 1.01, so a `wmix` write-concurrent leg had to be added or the payoff was +unevaluable either way. (3) **The before-p99 was off the instrument** — +`hist_add` clamps at 20000 µs and both before-runs pinned there, so the gain is +*at least* 2.3× and the true old p99 is unknown. + +**Dependencies unblocked — and one dependency invalidated.** `WO_T_IO` is +unreachable from a DB write: a failed stage or barrier now ends the process +(exit 74, diagnosed), replacing three behaviours that disagreed — `insert` +un-applied itself while `update` and `delete` returned a catchable trap and +admitted in their own comments that they left RAM ahead of disk. **Part B's +premise is invalidated**: it was justified by "close the 66× durable gap", but +that gap is two problems. Concurrent fan-in was a batching problem and is now +~3× better; a **serial** writer waiting on one barrier is a latency problem that +batching cannot touch and io_uring does not obviously help either. Part B should +be re-brainstormed, not started. + +**Next steps:** either re-brainstorm part B against its corrected premise, or +take chain 6 ([databasev2 3](databasev2/03-wal-checkpoint.md), WAL checkpoint), +which now has the replay "before" it lacked. **(Superseded 2026-08-29: it +landed.)** Two debts named rather than hidden: +the abort path is not exercised (forcing a real `fdatasync` failure needs mount +privileges), and single-shard concurrent batching needs the inline-path park — +the same machinery part B would need. + +**`.dev/reference` used:** none this slice. The sources were the engine's own +code and the Linux `fsync`-failure semantics that make retrying unsound. + +--- + +### Landed 2026-08-27 — iteration 24, chat + actor lifecycle (absorbing 31 + 34) + +**Implemented last time (2026-08-27):** the slice closed and merged to master +(`ed5334d`, fast-forward). T4 `monitor` + T5 `time.after` (ids 89/90) had +landed on the branch; this session merged master in (adopting the `porch` +rename), finished T8/T9, fixed the gate, found and fixed a runtime bug, and did +T10. Iterations 31 and 34 land inside it. + +**Key findings (measured, not asserted):** finishing the gate mattered more than +finishing the sample. Making **every leg start its own server** — instead of the +drain leg inheriting the soak's warmed one — exposed that **5 of 16** +fresh-server SIGTERM drains left a client at EOF with no close frame and no +diagnostic. Traced to `shard_main`: `NEXT_RUNNABLE()` already stated the +contract ("a WORKER on stop keeps DRAINING … close frames!") but the **idle** +branch reaped and broke, abandoning its inbox. An actor between messages is +exactly that idle case. Split out as +[40](language-runtime-database/40-shutdown-drain-guarantee.md); **20 of 20 +clean** after. Also measured: the fd check had been core-count dependent — lazy +per-shard init takes one `io_uring` + one `eventfd` per shard, capped at +`nproc`, so 26 → 44 on a 20-core box read as a leak. **1000 connections left it +at 44**, which settled it. + +**Learned:** three of the four gate failures were **stale build artifacts**, not +code. A branch switch leaves `compiler/_build/` and `runtime/build/` holding the +other branch's binaries, and a `woc` emitting `.wob` v7 against a v6 runtime +surfaces only as "no listener" — rebuild both before believing a gate failure. +And a gate that reuses another leg's server is not merely untidy: it hid a real +bug, and when its own leg failed it orphaned a listener that broke the *next* +run. Example apps now log to `/tmp/.log` so a developer can `tail -F` them. + +**Dependencies unblocked:** PUBSUB2 (WebSockets + pub/sub, rejected until this +point) is done; the porch ledger's WebSocket rows are ✅ and its cancellation row +is unblocked-not-built. Chain position 4 is complete, so **the chain's next link +is [databasev2 4](databasev2/04-io-uring-commit.md)** (io_uring group-commit). +Still blocked: CSRF and sessions — iteration 34 shipped HMAC but **there is +still no RNG**, and HMAC authenticates a token without being able to mint one, +which is [39](language-runtime-database/39-web-framework-parity.md)'s leading +item. + +**Next steps:** databasev2 4, or databasev2 2's outstanding 5c/5d. One debt is +named rather than hidden: iteration 40's guarantee is proven only by the chat +gate — nothing in `runtime/test/` drives `wo_engine_start`/`wo_engine_stop` and +no corpus fixture can trigger a stop, so pinning it lower needs new +multithreaded test infrastructure. + +**`.dev/reference` used:** none this slice. The sources were RFC 6455, RFC +3174/4231 for the digest vectors, and the kernel's own interfaces for the drain. ### Landed 2026-08-27 — databasev2 1, the RAM ceiling measured **Implemented last time (2026-08-27):** databasev2 1 refined (three forks @@ -226,10 +381,16 @@ no reference project was consulted for the implementation). **The concurrency + fiber chain — ✅ stage 3 → ✅ 22 → 🔄 24 (absorbing 31 + 34) → 23 → 32.** The chain's original order put 31 before 24; the 2026-08-23 directive absorbed 31 INTO 24, and 34 resolved with it, so -those three are one slice. **The live slice is iteration 24** — spec and -plan approved 2026-08-23, executing on branch `chat-ws-lifecycle`, five -of ten tasks landed. Its running state is the marker doc -([`2026-08-23-chat-ws-lifecycle.md`](../active-slice-2026-08-23-chat-ws-lifecycle.md)), +those three are one slice. **Iteration 24 is nine of ten tasks landed and MERGED TO MASTER +on 2026-08-27** (fast-forward, `ed5334d`): T1 crypto, T2 bounded mailboxes, +T3 call/reply, T4 `monitor` + T5 `time.after` (ids 89/90 — the reserved holes +are now filled), T6 ws upgrade, T7 frame codec, T8 chat sample, T9 the chat +gate. Verified on master: chat 11 checks 0 failures at the full 1000-client +soak, runtime battery 36 suites 0 fail, compiler 556 checks 0 fail, corpus +119 checks 0 fail. Only **T10 closeout** remains — which is what still holds +stories 24/31/34 open. Finishing T9 exposed and fixed a real runtime bug, +split out as [40](language-runtime-database/40-shutdown-drain-guarantee.md). Its running state is the marker doc +(the marker doc, deleted at closeout per the convention), which is the file to read for what is done and what is next; stories [31](language-runtime-database/31-actor-lifecycle.md) and [34](language-runtime-database/34-crypto-builtins.md) keep @@ -279,6 +440,9 @@ both still literal holes in `wob.h`'s builtin enum; then T8 the chat sample, T9 its gate, T10 closeout setting 24/31/34 to `status: done`) → 23 (io_uring group-commit — target: close the 4.5k→297k durable gap) → 32 (WAL checkpoint). Held tail resumes on its own precedence notes. +> (**Superseded 2026-08-28:** 24 landed, and 23's part A landed with it — +> "close the 4.5k→297k durable gap" turned out to be the wrong target; see +> the databasev2 4 row.) **`.dev/reference` used:** none this slice (the LW_SOAK discipline and linkcheck.py precedent came from in-repo scripts). @@ -436,11 +600,15 @@ that sequences its tasks. Read one, approve, then the next starts. | 19 | [Float + Bytes](language-runtime-database/19-missing-scalar-types.md) | ✅ **landed 2026-08-20** — `.wob` v5: Float constant tag, field kinds 6/7, opcodes 34-41 (IEEE-quiet f64), builtins 70-83. Full stack: literals, arithmetic, `@table` column, WAL bit-exact replay, json fractions in / shortest-round-trip out, `?Float` reserved-NaN nil, total-order index (NaN last, `-0.0` == `+0.0`), Bytes + base64. No implicit Int/Float mixing (WO-E201); `float`/`trunc` are the only bridges. Proof: web-app price is a real Float (`{"price":9.99}`), `just web-app` 23/0; corpus 103/0 | | 11 | [Fibers](language-runtime-database/11-fibers.md) | ✅ **landed 2026-08-21** with the arc (`just fibers` 10/0); fs-park re-scoped out of v1, disclosed in the story | | 22 | [Durability, throughput, scale](language-runtime-database/22-durability-throughput-scale.md) | ✅ **landed 2026-08-21** — db-bench + baseline.json (74 metrics) + restart/kill -9 proofs both shard counts; durable 4.5k vs ram 297k inserts/s, reads O(table), msgrate 13.4M/2.45M | -| 31 | [Actor lifecycle](language-runtime-database/31-actor-lifecycle.md) | 🔄 **absorbed into 24** (directive 2026-08-23) and half landed there: `call` request/response with a typed scalar reply (`WO_B_CALL = 88`, WO-E226), bounded mailboxes (`WO_MAILBOX`, cap 1024, catchable `WO_T_ACTOR`), and actor death that traps callers instead of hanging them. Still open: `monitor` and `time.after` — ids **89 and 90 are reserved holes** in `wob.h`, which is the machine-checkable proof of what is left. Supervision trees stay out of v1 | -| 24 | [chat: WebSocket workload](language-runtime-database/24-chat-websocket-workload.md) | 🔄 **the live slice** (absorbing 31 + 34, directive 2026-08-23) — branch `chat-ws-lifecycle`, 5/10 tasks landed: crypto, bounded mailboxes, WS upgrade, frame codec, `call`/reply + actor death. Pending: `monitor`, `time.after`, the chat sample, its gate, closeout. State lives in [the marker](../active-slice-2026-08-23-chat-ws-lifecycle.md) | +| 31 | [Actor lifecycle](language-runtime-database/31-actor-lifecycle.md) | ✅ **LANDED 2026-08-27 inside 24** (directive 2026-08-23). All four mechanisms: `call`/reply with a typed scalar reply (`WO_B_CALL = 88`, WO-E226), bounded mailboxes (`WO_MAILBOX`, cap 1024, catchable `WO_T_ACTOR`), actor death that traps callers instead of hanging them, **`monitor` (89)** and **`time.after` (90)** — the reserved holes in `wob.h` are filled. A fifth mechanism it did not anticipate came out of proving the gate: the shutdown drain guarantee, [40](language-runtime-database/40-shutdown-drain-guarantee.md). Supervision trees stay out of v1 | +| 24 | [chat: WebSocket workload](language-runtime-database/24-chat-websocket-workload.md) | ✅ **LANDED 2026-08-27** (absorbing 31 + 34) — all ten tasks; merged to master `ed5334d`. `just chat` **11 checks, 0 failures** at the full 1000-client soak: handshake, functional matrix on both `WO_IO` backends and on one shard, the soak, the fd invariant, the SIGTERM drain, `WO_MAILBOX=8` backpressure, ASan clean. Finishing its gate found a real runtime bug, split out as [40](language-runtime-database/40-shutdown-drain-guarantee.md) | +| 23 | [io_uring group-commit](databasev2/04-io-uring-commit.md) | ✅ **part A LANDED 2026-08-28 — group commit**, one barrier per drain instead of one per statement (the engine was fsync-per-STATEMENT, not per commit; the story's premise was wrong). Shard 0 holds each reply, commits once when its queue empties, releases all — so a writer is acked after the barrier carrying ITS record. **≈2.9× durable write throughput, ≈2.1× lower p50**, two measurement methods agreeing (2.9× controlled, 3.5× s1-vs-sN); mean batch 5.43, peak 57. A durability failure is now **fatal (exit 74), not a catchable `WO_T_IO`** — replacing three behaviours that disagreed, two of which admitted leaving RAM ahead of disk. **What it did NOT do:** `durable.sN.mixwrite` 480→492 (unchanged — that workload does 20 writes at C=4, mean batch 1.01) and `seed` unchanged (serial writers have nothing to batch with). **This row used to say "close the 66× gap"; that target was mis-stated** — the gap is two problems and part A fixes only the concurrent one. ⬜ part B (io_uring) **needs re-brainstorming**, not starting on the old premise | +| 32 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) | ✅ **LANDED 2026-08-29 — the chain's last link.** Compaction rewrites the log as one record per live row and swaps it in with `rename`, so **recovery is completely unchanged** and crash safety comes from the filesystem rather than from code. **2.16× space reclaimed** (1 962 358 → 907 094 B), **boot 114 → 64 ms**, stop-the-world pause **2 651 µs** against a stated 50 ms budget. Read `.dev/reference/postgresql` for it: PG *never* compacts its WAL — its records are page deltas, so it needs heap files, a control file, a redo pointer and a separate process. Ours are full row images, so a compacted log IS a store, which deletes all of that. `kill -9` during compaction: 40 rounds/run, 10 clean runs, and **mutation-proven** — against in-place rewrite instead of `rename` the battery fails every time. Outstanding: the **`resident: keys` offset map** (compaction moves every record; the obligation is recorded at the compactor) and the O(live rows) pause, ~5.5 s at 1 GB, which is what an incremental design must be bought against | +| 33 | [Single-file store](databasev2/07-single-file-db.md) | ⬜ off-chain, small — `WO_DATA=.db` file form; driver-only (story written 2026-08-22) | | 34 | [Crypto builtins](language-runtime-database/34-crypto-builtins.md) | 🔄 **code landed** as 24's T1 (`d14fa9f`): `sha1`/`sha256`/`hmac_sha256`, ids 85–87 in `wob.h`, `runtime/src/crypto.c`, RFC/FIPS vectors 18/0, corpus pin. The 24 gate that once needed it is cleared. Frontmatter keeps `status: refine` only until 24's T10 closeout sets it to `done` | | 38 | [Content platform capabilities](language-runtime-database/38-content-platform-capabilities.md) | ⬜ off-chain, needs a spec — the two capability families no iteration owns, confirmed against `runtime/src/wob.h`: `fs` mutation verbs (six fs builtins, ids 40–45; `append` creates-if-absent, so nothing is ever replaced, truncated, deleted or renamed) and `net.connect` (ids 51–55 + 91–95, no connect, and no `connect()` anywhere in `runtime/src/` — so no OIDC/SMTP/object-store/webhook/federation). Driven by a `docs/examples/vault` content-collaboration workload, in 28's mould. New builtins from 96 (89/90 reserved for 31); no `.wob` bump (`WOB_VERSION 6u`, last moved by 36). Story written 2026-08-26 from the "can it build a Nextcloud?" ask | | 39 | [Web framework parity](language-runtime-database/39-web-framework-parity.md) | ⬜ off-chain, needs a spec — from [the Fiber v3.5.0 study](../plan/exploration/fiber/00-fiber-parity.md) (all 32 of its middleware read against `porch`; **nine already have a counterpart**). Leads with a **random-bytes builtin**: the framework ledger claimed CSRF/sessions were unblocked by iteration 34's HMAC, but HMAC authenticates a token and cannot mint one — there is no RNG anywhere in the runtime. Then cookies (absent both ways; `Resp.headers` being a map cannot carry two `Set-Cookie` lines), then limiter/idempotency (cheapest wins — `@table` + `time.ticks`, nothing new), sessions, CSRF, and the routing/response sugar. Streaming/SSE/compression, `@derive` binding, TTL cache, `proxy` and metrics all excluded with owners named | +| 40 | [Shutdown drain guarantee](language-runtime-database/40-shutdown-drain-guarantee.md) | ✅ **LANDED 2026-08-27 — chain 3, with 31; split out of 24.** One rule: **a message sent before the stop flag is observed must be delivered and run before the engine stops.** Found by measurement, not review: making the chat gate's drain leg start its OWN (cold) server exposed that **5 of 16** fresh-server SIGTERM drains left a WebSocket client at EOF with no close frame and no diagnostic. Traced to `shard_main` — `NEXT_RUNNABLE()` already stated the contract ("a WORKER on stop keeps DRAINING … close frames!") but the IDLE branch reaped and broke, abandoning its inbox for teardown to free. An actor between messages is exactly that idle case, which is why a WARM soak server hid it for so long. Fix is one branch honouring the primary's drain window, yielding on an empty poll. **20 of 20 clean after**; `just chat` 11 checks 0 failures at the full 1000-client soak (which also settled the fd question: 1000 connections left the count at 44); runtime battery 36 suites 0 fail, compiler 556 checks 0 fail. Ruled out: a bigger spin (a 1 s wall-clock deadline still failed 2 of 12) and spawn-during-shutdown. Outstanding: a pin below the gate — nothing in `runtime/test/` drives the engine start/stop and no corpus fixture can trigger a stop | | 37 | [wo-html components](language-runtime-database/37-wo-html-components.md) | ✅ off-chain — LANDED 2026-08-25. Raw text literal (backtick, margin stripped at lex time, `{{ }}` auto-escapes) + the component layer: `Component`/`render_all`/`Layout` in wo-html, `ok_html` moved into the framework, site and shop both migrated | | 35 | [net runtime seams](language-runtime-database/35-net-runtime-seams.md) | ⬜ off-chain — fd deadlines on the park plane, Unix sockets, peer address; owns the ledger's three 🔧 rows (story written 2026-08-22) | | 25 | [HTTP service layer](../superpowers/plans/2026-08-01-http-service-layer.md) | ⏸ hold (2026-08-21) — story file removed; the plan doc remains | @@ -461,13 +629,16 @@ that sequences its tasks. Read one, approve, then the next starts. | Language | 🔄 [iteration 36 — operator parity](language-runtime-database/36-operator-parity.md): `not`, bitwise `& \| ^ << >>`, hex/binary/`_` literals, compound assigns — CODE LANDED 2026-08-22 (branch operator-parity, `.wob` v6, all gates green; reference project `.dev/reference/go` drove the design). Awaiting the developer's MANUAL pass on `docs/examples/operators/` (no test fixtures by directive); unblocks story 34's pure-`.wo` HMAC question | [plan](../superpowers/plans/2026-08-22-operator-parity.md) | | Language | the framework v1-polish slice landed 2026-08-20 (branch framework-v1, awaiting merge); next per the order: brainstorm 20/21's forks | [order](#implementation-order-re-sequenced-2026-08-21--concurrency-chain) | | Runtime | ✅ **iteration 35 landed 2026-08-23** (branch `framework-v1b`, with framework v1 slice 2 + the serving slice): net deadlines/unix/peer (ids 91–95), fiber pooling, serve_conn + web-app fiber-per-connection — web-app gate 41/0, both WO_IO backends | [design](../superpowers/specs/2026-08-23-net-seams-park-design.md) | -| Runtime | 🔄 **iteration 24 (absorbing 31 + 34): chat + actor lifecycle** — spec + plan approved 2026-08-23 (24 absorbs 31 by directive; 34 resolved C-builtins); executing on branch `chat-ws-lifecycle` | [marker](../active-slice-2026-08-23-chat-ws-lifecycle.md) · [plan](../superpowers/plans/2026-08-23-chat-ws-lifecycle.md) | -The active slice's marker doc is -[`docs/active-slice-2026-08-23-chat-ws-lifecycle.md`](../active-slice-2026-08-23-chat-ws-lifecycle.md) -— one file, deleted when the slice lands. Everything else pending is the -concurrency chain (see *Pending* below); the held tail is every story -whose frontmatter reads `status: hold`. +**No slice is active.** Iteration 24 landed 2026-08-27 and its marker doc was +deleted per the convention. Everything pending is the concurrency chain (see +*Pending* below) — **the chain's next link is +[databasev2 4](databasev2/04-io-uring-commit.md)** (chain 5, the io_uring +group-commit write path, `was_language_iteration: 23`), which now has iteration +22's fsync-per-commit numbers in hand, plus databasev2 1's finding that the +write path is *not* where memory pressure bites (appending under a cap costs +~1%, random reads 273×). The held tail is every story whose frontmatter reads +`status: hold`. ### Landed 2026-08-14 — the compile-and-run milestone @@ -705,8 +876,8 @@ the language arc as v1 history. | --- | --- | --- | | 1 | [RAM ceiling: measure the breaking point](databasev2/01-ram-ceiling-measurement.md) | ✅ **MEASURED 2026-08-27** — `readiness: ready`, `status: done`; forks settled, harness landed (**148 checks**). Footprint **96.5–100 B/row** Int vs **320.6–324 B/row** text = **3.3×** (not the "order of magnitude" three docs claimed), read as median-of-marginals because doublings swing a two-point slope 2×. **Both predicted exits were wrong:** table storage has no checked ceiling and is **SIGKILLed** (overcommit lets `malloc` succeed, kernel kills on page touch), and swap is not latency collapse — 900k rows finished **148 s capped-with-swap vs 150 s uncapped**, ~1%, returning 0 while serving from disk. **Ack-after-fsync survives an OOM kill:** ~40 000 rows recovered as an intact prefix, gated as the `ceiling` leg. Also measured: **random reads over an oversized table collapse 273×** (1.85M vs 6 771 reads/s, p99 1 µs vs 487 µs) — so the two access patterns sit ~270× apart under the same pressure, and departure is a **step, not a curve**. Replay measured too: **≈5.5 µs/record, 1.9× history penalty** (10M records ≈ 55 s of boot) — iteration 3's missing "before", now gated. Iteration 2's budget dependency is **removed, not satisfied** — there is no "swap onset" to derive it from | | 2 | [per-table storage: `durable` and `resident`](databasev2/02-table-storage-modes.md) | 🔄 **the language enrichment — the `durable` half is DONE and usable.** Two optional `@table` keys, `durable: true\|false` and `resident: all\|keys`, both defaulting to today's behaviour (all 28 existing declarations compile unchanged, no golden moved). Landed: the grammar, WO-E224 (a durable `ref` into a volatile table is refused), `.wob` v7 carrying both properties in spare `flags` bits, `durable: false` actually skipping the WAL (measured: 50 inserts → 1500 bytes durable, **0** volatile) with a mode-mismatch startup refusal, plus offset capture and read-a-row-from-an-offset. Outstanding: 5c/5d (the id→offset map and rewiring `wo_row_ptr`'s 11 call sites, slab scans and `@unique`/FK across the boundary — not yet written up), the two runtime refusals, and closeout. [spec](../superpowers/specs/2026-08-26-table-residency-design.md) · [plan](../superpowers/plans/2026-08-26-table-residency.md) | -| 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ⬜ snapshot + truncate: disk reclaimed, replay bounded | -| 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ⬜ **`readiness: ready` — the one startable iteration in the repo** (four forks confirmed settled 2026-08-20). Close the 66× gap iteration 22 measured (durable 4.5k vs ram 297k inserts/s) | +| 3 | [WAL checkpoint](databasev2/03-wal-checkpoint.md) *(was 32)* | ✅ **LANDED 2026-08-29 — the chain's last link.** Compaction rewrites the log as one record per live row and swaps it in with `rename`, so **recovery is completely unchanged** and crash safety comes from the filesystem rather than from code. **2.16× space reclaimed** (1 962 358 → 907 094 B), **boot 114 → 64 ms**, stop-the-world pause **2 651 µs** against a stated 50 ms budget. Read `.dev/reference/postgresql` for it: PG *never* compacts its WAL — its records are page deltas, so it needs heap files, a control file, a redo pointer and a separate process. Ours are full row images, so a compacted log IS a store, which deletes all of that. `kill -9` during compaction: 40 rounds/run, 10 clean runs, and **mutation-proven** — against in-place rewrite instead of `rename` the battery fails every time. Outstanding: the **`resident: keys` offset map** (compaction moves every record; the obligation is recorded at the compactor) and the O(live rows) pause, ~5.5 s at 1 GB, which is what an incremental design must be bought against | +| 4 | [io_uring group commit](databasev2/04-io-uring-commit.md) *(was 23)* | ✅ **part A LANDED 2026-08-28 — group commit**, one barrier per drain instead of one per statement (the engine was fsync-per-STATEMENT, not per commit; the story's premise was wrong). Shard 0 holds each reply, commits once when its queue empties, releases all — so a writer is acked after the barrier carrying ITS record. **≈2.9× durable write throughput, ≈2.1× lower p50**, two measurement methods agreeing (2.9× controlled, 3.5× s1-vs-sN); mean batch 5.43, peak 57. A durability failure is now **fatal (exit 74), not a catchable `WO_T_IO`** — replacing three behaviours that disagreed, two of which admitted leaving RAM ahead of disk. **What it did NOT do:** `durable.sN.mixwrite` 480→492 (unchanged — that workload does 20 writes at C=4, mean batch 1.01) and `seed` unchanged (serial writers have nothing to batch with). **This row used to say "close the 66× gap"; that target was mis-stated** — the gap is two problems and part A fixes only the concurrent one. ⬜ part B (io_uring) **needs re-brainstorming**, not starting on the old premise | | 5 | [Bounded tables and eviction](databasev2/05-bounded-tables-eviction.md) | ⬜ a declared capacity + refuse/evict/back-pressure, and a process-level pressure signal that sheds **before** the allocator or OS gets involved — turning the invisible failure into a managed one | | 6 | [Cold tiering](databasev2/06-cold-tiering.md) | ⚠ **largely superseded by 2** — `resident: keys` took the ceiling-raising role; its user-space-working-set premise was rejected for the kernel page cache. Mostly forks: which shape, whether the index itself fits, whether the *language* surfaces the fault cost, and whether `@unique` on a cold table is refused outright. A paged B-tree stays rejected — if tiering needs one, reject tiering | | 7 | [Single-file store](databasev2/07-single-file-db.md) *(was 33)* | ⬜ `WO_DATA=.db`; driver-only, independent | diff --git a/docs/stories/databasev2/03-wal-checkpoint.md b/docs/stories/databasev2/03-wal-checkpoint.md index 74226ee..f8b10f4 100644 --- a/docs/stories/databasev2/03-wal-checkpoint.md +++ b/docs/stories/databasev2/03-wal-checkpoint.md @@ -2,8 +2,8 @@ track: databasev2 iteration: "3" was_language_iteration: "32" -status: pending -readiness: refine +status: done +readiness: ready chain: 6 --- @@ -29,6 +29,113 @@ chain: 6 > replay/restart numbers to justify its policy and must compose with > 23's group-commit write path. +> **BRAINSTORMED 2026-08-28.** Spec: +> [`2026-08-28-wal-checkpoint-design.md`](../../superpowers/specs/2026-08-28-wal-checkpoint-design.md) +> · plan: [`2026-08-28-wal-checkpoint.md`](../../superpowers/plans/2026-08-28-wal-checkpoint.md) +> (6 tasks). +> Read `.dev/reference/postgresql` for this — and the conclusion was that +> Postgres' design is *unavailable* to us, which is what makes the simpler one +> legitimate. +> +> **The design in one sentence:** compact the log by rewriting it as one record +> per live row into a temp file, then `rename` it over the live WAL. Recovery is +> **completely unchanged** — boot still opens one file and replays it — and the +> crash criterion is satisfied by the filesystem rather than by code we must get +> right. +> +> **Why one file works here and not in Postgres.** Postgres never compacts its +> WAL: its records are page deltas, so a compacted redo log is not a store, and +> it must keep heap files, a control file, a redo pointer and a second recovery +> source. Ours are **full row images** — `apply_record` implements UPDATE as +> remove-then-recreate — so a compacted log *is* a complete store. That one +> difference deletes the control file, the redo pointer, the cutoff offset and +> the separate process from the design. +> +> **Forks settled:** no snapshot format (the compacted log is the snapshot); one +> source, not two; **volume-only trigger** as a ratio against the last +> compaction's own measured output, with an absolute floor — **no timer**, +> because Postgres' timer exists to bound loss from unflushed buffers and we have +> none; stop-the-world, with the pause measured against a stated budget rather +> than assumed acceptable. +> +> **The coupling that would otherwise be found late:** compaction moves every +> record, so it **invalidates every WAL offset** +> [iteration 2](02-table-storage-modes.md)'s `resident: keys` stores. The +> compactor rebuilds the offset map as it writes. Recorded now because iteration +> 2's storage half is unimplemented, so nothing breaks today — it would break +> later, looking like corruption rather than a design gap. +> +> **Measured on master 2026-08-28, grounding the whole iteration:** `seed 20000` +> leaves a 986 614-byte log; 20 000 updates take it to **2 590 262 bytes with the +> same live rows** (2.6× history for no data), and boot+verify on that store is +> **155 ms**. + +## Progress — landed 2026-08-29 + +| # | Task | State | +| --- | --- | --- | +| 1 | `wo_wal_compact` — rewrite, fsync, rename, fsync parent, reopen | ✅ `8ea510d` | +| 2 | a stale compaction temp is removed at open | ✅ `8bfbd4b` | +| 3 | the trigger (pure decision + env knobs) and the ordering guard | ✅ `6dbcb9a` | +| 4 | `kill -9` DURING compaction — 40 rounds, mutation-proven | ✅ `9b283d5` | +| 5 | measure space, boot and the stop-the-world pause | ✅ `d87f65a` | +| 6 | closeout | ✅ this change | + +### Measured + +| | checkpointing off | checkpointing on | +| --- | --- | --- | +| WAL used | 1 962 358 B | **907 094 B** | +| boot | 114 ms | **64 ms** | + +**2.16× space reclaimed, 1.78× faster boot**, stop-the-world pause **2 651 µs** +against a stated 50 ms budget. Full details, including the pause's scaling, are +in [`perf-targets.md`](../../plan/perf-targets.md) §7. + +### Two bugs the work found, both mine + +**Wiring only the drain left `WO_SHARDS=1` never compacting** — its log grew +forever (536 KB where the multi-shard run held 446 KB), because a statement on +the owner shard never enters that drain. Both write paths now check. + +**The dump was 8× slower than it needed to be**, flushing through the +committing path and so paying one `fdatasync` per 256 records for durability +that is worthless before the rename. One final barrier took the pause from +107 649 µs to 13 212 µs on a 2 MB live set — ~22 MB/s to ~181 MB/s. + +## Acceptance Criteria + +Met: + +- **Given** an aged store, **when** it is compacted, **then** disk is reclaimed. + ✅ 2.16× on the full campaign, asserted rather than merely recorded — the leg + fails if the log is not smaller with checkpointing on. +- **Given** the same store, **when** it boots, **then** replay is bounded by the + live set rather than by history. ✅ 114 → 64 ms. +- **Given** `kill -9` at ANY instant during a checkpoint, **when** the process + restarts, **then** recovery produces the same consistent store as if the + checkpoint had never started, with no acknowledged write lost. ✅ 40 rounds + per run, 10 consecutive clean runs, and **proven to have teeth**: against the + design's rejected alternative (in-place rewrite instead of `rename`) the + battery fails every run with the log destroyed. +- **Given** the iteration-22 replay numbers, **then** a before/after delta is + recorded. ✅ `perf-targets.md` §7. +- **Given** writes arriving while a checkpoint runs, **then** the ack contract + holds. ✅ compaction runs only where nothing is staged, asserted by a test + that stages and requires refusal; `wo_wal_compact` also refuses as a backstop. + +Outstanding: + +- **The `resident: keys` offset map.** Compaction moves every record, so it + invalidates every WAL offset [iteration 2](02-table-storage-modes.md) stores. + The compactor must rebuild that map as it writes. **Nothing fails today** + because iteration 2's storage half is unimplemented — which is exactly why the + obligation is written at the compactor in `wal.c`, where the next implementer + hits it, rather than only in a spec they may not read. +- **The pause is O(live rows).** At ~181 MB/s a 1 GB live set implies ~5.5 s, + past any interactive budget. Incremental or forked copying was deliberately + not bought in advance; this is the number to buy it against. + ## Goals - **Disk space is reclaimed.** A checkpoint writes the live store as a diff --git a/docs/stories/databasev2/04-io-uring-commit.md b/docs/stories/databasev2/04-io-uring-commit.md index dd5c255..25fe145 100644 --- a/docs/stories/databasev2/04-io-uring-commit.md +++ b/docs/stories/databasev2/04-io-uring-commit.md @@ -2,7 +2,7 @@ track: databasev2 iteration: "4" was_language_iteration: "23" -status: pending +status: in-progress readiness: ready chain: 5 --- @@ -59,6 +59,111 @@ chain: 5 > batch — under io_uring it becomes exactly one submission, so the two > features compose without either knowing the other. +> **BRAINSTORMED 2026-08-28 — and SPLIT IN TWO.** Spec for part A: +> [`2026-08-28-wal-group-commit-design.md`](../../superpowers/specs/2026-08-28-wal-group-commit-design.md) +> · plan: [`2026-08-28-wal-group-commit.md`](../../superpowers/plans/2026-08-28-wal-group-commit.md) +> (6 tasks). +> +> **The payoff metric is `durable.sN.mixwrite`, not the s1 numbers.** Worker +> shards hold no WAL — the runtime asserts it — so every statement on a worker +> marshals to shard 0 and parks, while a statement already on shard 0 runs +> inline. Batches form only where there is a queue, so concurrent multi-shard +> writes batch and a single-shard or serial workload does not. The baseline +> shows why that is the right target anyway: **multi-shard concurrent writes are +> 480 ops/s at p99 5888 µs against single-shard's 1023 at p99 664 — adding +> shards makes durable writing WORSE today**, because every marshaled statement +> still buys its own barrier on the owner. +> +> **The premise below needed correcting.** This story says "replace +> fsync-per-commit with io_uring group-commit", but the engine does not commit +> per commit — it commits per **statement**: `db.c` calls `wo_wal_commit` +> immediately after every append, at all six sites, so every row change is one +> `pwrite` plus one `fdatasync`. That splits the goal into two independent +> wins, and only the second needs io_uring: +> +> - **Part A — batching.** Let many statements share one barrier. The staging +> buffer already holds any number of records; today it never holds more than +> one because the caller commits immediately. Mostly a deletion of calls. +> - **Part B — async submission.** The shard submits and keeps working instead +> of blocking in `fdatasync`. Deferred until A's measurement says whether the +> blocking boundary is still the bottleneck. +> +> **A is where most of the number lives.** Iteration 22 measured durable writes +> at 4460 ops/s and mixed writes at 1023 ops/s (p99 664 µs) against 1.28M ops/s +> for durable reads — ~290× apart, essentially all of it the per-statement +> barrier. +> +> **Forks settled in the brainstorm:** batch boundary is **queue-drain** (not +> the tick this story recorded — a tick taxes an idle system to serve a busy +> one); a failure between "RAM mutated" and "record durable" is a **fatal, +> diagnosed abort**, replacing today's uneven rollback where `insert` undoes +> itself and `update`/`delete` admit in a comment that they leave RAM ahead of +> disk. **That removes `WO_T_IO` from the write path** — a language-visible +> change, recorded here deliberately. +> +> `status: in-progress` because the brainstorm is done and the spec is +> approved; the plan is next. (The `readiness` axis that would say this +> precisely lives on the unmerged `db-residency-doctrine`.) + +## Progress — part A landed 2026-08-28 + +| # | Task | State | +| --- | --- | --- | +| 1 | a failed barrier is detected, and fatal | ✅ `d3ff03e` | +| 2 | one barrier per drain; replies held | ✅ `b9b8a45` | +| 3 | the inline path takes the fatal rule, asymmetry documented | ✅ `a6ccdbe` | +| 4 | prove batches form — the `wmix` write-concurrent leg | ✅ `40d029c` | +| 5 | measure the payoff, gate it, record it | ✅ `d52ea8a` | +| 6 | closeout | ✅ this change | +| — | **part B — io_uring submission** | ⬜ **not started; its premise changed, see below** | + +### The payoff, measured two ways + +| Measurement | Before | After | +| --- | --- | --- | +| controlled (same build, only `db.c`/`vm.c` swapped; `wmix 4000 32`) | 2213 · 2177 ops/s, p50 7183 · 7251 µs | **6216 · 6525 ops/s, p50 3458 · 3444 µs** | +| committed baseline: `s1` inline vs `sN` batched | 1467 ops/s, mean batch 1.0 | **5117 ops/s, mean batch 5.43, peak 57** | + +**≈2.9× throughput, ≈2.1× lower p50**, and the two methods agree (2.9× and +3.5×). Batching scales with contention: mean batch **1.13 / 1.76 / 5.35** at +C = 4 / 16 / 64. + +### The cost side, and a bug the battery caught + +**Reads were being held behind the barrier.** The drain first held *every* DB +reply until the commit — including reads, which stage nothing. `mixread` p99 rose +from ~1043 µs to **4057 µs** until only staging statements had their replies +held. Caught by the gate, not by review. + +**What remains is inherent:** a barrier blocks the owner shard longer (more +records per fsync) though less often, so anything queued behind one waits. Three +full runs of the same build gave `durable.sN.mixread.p99` of **1043 / 2318 / +4147 µs** — a 2–4× spread near idle. So part A buys ~3× write throughput at the +cost of a longer, noisier tail on the owner shard. `durable.sN.*.p99us` was +re-baselined at 100% tolerance for that reason, with the floor as the real guard +(`mixread`'s came within 25 µs of tripping). + +**This is the strongest argument for part B** — submitting the barrier and +continuing to serve is exactly what removes this cost. + +### What did NOT improve — and it was predicted + +- **`durable.sN.mixwrite`: 480 → 492 ops/s, i.e. unchanged.** This was the + spec's *original* payoff metric, and correcting it was part of the brainstorm: + `mix` writes on one op in ten with C=4, so a quick run performs **20 writes** + and measured mean batch **1.01**. A workload that never has two writes in + flight cannot be helped by batching them. +- **`durable.*.seed`: unchanged.** A serial single writer has nothing to batch + with, under any scheme. +- **This board's stated target was mis-stated.** It read "close the 66× gap + iteration 22 measured (durable 4.5k vs ram 297k inserts/s)". Part A does not + close that gap and structurally cannot: `seed` is serial, and one writer + waiting on one barrier is a **latency** problem, not a batching one. Recorded + rather than quietly renumbered. +- **The before-p99 is not a measurement.** `hist_add` clamps at 20000 µs and + both before-runs pinned exactly there, so the true value is ≥20 ms and + unknown. The gain is *at least* 2.3×. + ## Goals - **Replace fsync-per-commit with io_uring group-commit** on the WAL write @@ -77,26 +182,50 @@ chain: 5 ## Acceptance Criteria -- What to achieve? - - **Given** the io_uring write path under the iteration-22 crash battery - (concurrent writers, kill -9 mid-stream, reboot, replay), - - **when** it runs, - - **then** every acknowledged write is present after replay and no - unacknowledged partial write is ever visible — the exact result the - fsync path gives, so durability is provably unchanged. -- What to achieve? - - **Given** the iteration-22 durable write benchmark, - - **when** it is run on the fsync-per-commit path and then the io_uring - group-commit path on the same machine, - - **then** the io_uring path's write throughput is materially higher and - its p99 commit latency lower, with the before/after numbers recorded — - the payoff, measured, not asserted. -- What to achieve? - - **Given** a kernel without io_uring (old, or restricted by seccomp), - - **when** the runtime starts, - - **then** it falls back to the pwrite + fdatasync path automatically and - correctly — io_uring is an accelerator, never a hard dependency, and a - binary that runs everywhere is the whole project's premise. +Met: + +- **Given** the io_uring write path under iteration 22's crash battery, **when** + it runs, **then** every acknowledged write is present after replay. ✅ — the + criterion applies unchanged to part A's batching. `crash.sN` (the batched + path) recovered every acked row after `kill -9`, `crash.s1` likewise, and both + restart legs replay byte-true. This was the one thing batching could break. +- **Given** the durable write benchmark before and after, **then** throughput is + materially higher and p99 lower, recorded. ✅ ~2.9× and ~2.1× (p50); see + `perf-targets.md` §6. **Scoped honestly:** on a write-concurrent workload + only, and p99's "before" is at the histogram ceiling. +- **Given** batching, **when** it runs, **then** it is proven to engage rather + than assumed. ✅ mean batch 5.43, peak 57 on the gated leg, and the live + assertion fails the suite if the mean drops to 1. +- **Given** a durability failure, **when** it happens, **then** the engine does + not continue with RAM ahead of disk. ✅ fatal, diagnosed, exit 74 — replacing + three behaviours that disagreed. + +Outstanding: + +- **Given** a kernel without io_uring, **when** the runtime starts, **then** it + falls back automatically. *(part B — part A adds no syscall interface, so + nothing to fall back from yet.)* +- **Single-shard concurrent batching.** A statement on shard 0 commits inline + and cannot batch; doing so needs the inline path to park its fiber on the + barrier — the same machinery part B needs. So `WO_SHARDS=1` gets no batching + at all, by design and measured (mean batch 1.0). +- **The abort path is not exercised.** Forcing a real `fdatasync` failure needs a + full or read-only filesystem, which the gate cannot arrange without mount + privileges. The unit test proves the error is *detected*; the exit three lines + later is covered by inspection. Disclosed rather than papered over — iteration + 40 was exactly a fatal path nothing exercised. + +## Part B — its premise changed + +Part B was justified by "close the 66× durable gap". Part A shows that framing +was wrong: the gap is **two** problems. Concurrent write fan-in was a batching +problem and is now ~3× better. What remains is a **serial** writer waiting on a +single barrier, which no amount of batching can help — and io_uring does not +obviously help it either, since one writer still needs one durable barrier +before its ack. Part B's real candidates are overlapping the barrier with other +work on the shard, and the inline-path park that single-shard batching also +needs. **It should be re-brainstormed against that, not started on the old +premise.** ## Out Of Scope diff --git a/docs/stories/language-runtime-database/24-chat-websocket-workload.md b/docs/stories/language-runtime-database/24-chat-websocket-workload.md index e29a10e..2e461fd 100644 --- a/docs/stories/language-runtime-database/24-chat-websocket-workload.md +++ b/docs/stories/language-runtime-database/24-chat-websocket-workload.md @@ -1,6 +1,6 @@ --- iteration: "24" -status: in-progress +status: done readiness: ready chain: 4 --- @@ -20,6 +20,34 @@ chain: 4 > bounded mailboxes, actor death, timers). Iteration 19 LANDED > 2026-08-20, so Bytes is available for frame parse/serialize. +> **✅ LANDED 2026-08-27** (branch `chat-ws-lifecycle`, merged to master +> `ed5334d`). Ten tasks: crypto (T1), bounded mailboxes (T2), `call`/reply and +> actor death (T3), `monitor` (T4), `time.after` (T5), the WS upgrade seam +> (T6), the pure-`.wo` frame codec (T7), the chat sample (T8), the gate (T9), +> this closeout (T10). It absorbed [31](31-actor-lifecycle.md) and +> [34](34-crypto-builtins.md), which land with it. +> +> **Gate — `just chat`, 11 checks, 0 failures** at the full 1000-client soak: +> handshake with an independently recomputed accept-key, the functional matrix +> (presence, broadcast, room isolation, leave) on **both** `WO_IO` backends and +> on a single shard, the 1k hot-room soak, the fd invariant, the SIGTERM drain, +> `WO_MAILBOX=8` backpressure, and an ASan run with zero leaks. Battery +> alongside: runtime 36 suites 0 fail, compiler 556 checks, corpus 119 checks. +> The sample logs to `/tmp/chat.log`. +> +> **Two disclosed deviations from the spec.** `monitor` takes **three** +> arguments (`watched, observer, msg`) rather than two, because the caller may +> be `main`, which has no mailbox and cannot be an implicit observer. And a +> `call` reply is a **typed scalar** in v1 — which is what let the agreement be +> checked at compile time (WO-E226) instead of carried as a tagged value. +> +> **What finishing the gate found.** Making every leg start its own server +> exposed a real runtime bug the warmed soak server had been hiding: on a fresh +> server, 5 of 16 SIGTERM drains left a client at EOF with no close frame. It +> was not this sample's fault — the fix is an engine guarantee, split out as +> [40](40-shutdown-drain-guarantee.md). Design notes: +> [`docs/examples/chat/CODE-LOGIC.md`](../../examples/chat/CODE-LOGIC.md). + ## Why this iteration exists Everything the framework ledger parks behind concurrency — WebSockets, diff --git a/docs/stories/language-runtime-database/31-actor-lifecycle.md b/docs/stories/language-runtime-database/31-actor-lifecycle.md index 343b773..f714fe9 100644 --- a/docs/stories/language-runtime-database/31-actor-lifecycle.md +++ b/docs/stories/language-runtime-database/31-actor-lifecycle.md @@ -1,6 +1,6 @@ --- iteration: "31" -status: in-progress +status: done readiness: ready chain: 3 --- @@ -17,6 +17,22 @@ chain: 3 > ([iteration 24](24-chat-websocket-workload.md)) cannot be written > honestly without these four mechanisms. +> **✅ LANDED 2026-08-27 — INSIDE [24](24-chat-websocket-workload.md)**, per +> the 2026-08-23 directive that absorbed it. All four mechanisms shipped: +> `call`/reply with a typed scalar reply (id 88, WO-E226), **bounded mailboxes** +> (`WO_MAILBOX`, default 1024, fail-fast with a catchable `WO_T_ACTOR`), +> **actor death** that traps callers instead of hanging them, `monitor` +> (id 89) and `time.after` (id 90). Ids 89 and 90 were reserved holes in +> `wob.h`; they are filled. +> +> **A fifth mechanism was added that this story did not anticipate**: the +> shutdown drain guarantee, [40](40-shutdown-drain-guarantee.md). It is +> lifecycle semantics — this story gave actors a death notice, 40 gives the +> program a shutdown that does not lose mail — and it was found by measurement +> while proving 24's gate, not by review. +> +> How each piece works: `runtime/src/CODE-LOGIC.md`, "Actor lifecycle". + ## Why this iteration exists The arc's stages 1+2 shipped `spawn`/`send` mechanism without lifecycle: diff --git a/docs/stories/language-runtime-database/34-crypto-builtins.md b/docs/stories/language-runtime-database/34-crypto-builtins.md index 54a7488..aee2882 100644 --- a/docs/stories/language-runtime-database/34-crypto-builtins.md +++ b/docs/stories/language-runtime-database/34-crypto-builtins.md @@ -1,6 +1,6 @@ --- iteration: "34" -status: in-progress +status: done readiness: ready --- @@ -19,6 +19,18 @@ readiness: ready > Off the concurrency chain but **gates chain position 4**: iteration > 24's WebSocket handshake needs SHA-1 before chat can land. +> **✅ LANDED 2026-08-27 — inside [24](24-chat-websocket-workload.md)** as its +> task 1. The fork resolved to **C builtins**: `sha1` (85), `sha256` (86), +> `hmac_sha256` (87), each over one buffer returning a fresh `Bytes`. Pinned to +> the published vectors — RFC 3174, the SHA-256 vectors, RFC 4231 — in +> `runtime/test/test_crypto.c`, 18 checks, plus a corpus fixture hashing "abc" +> from `.wo`. This unblocked chain position 4: the WebSocket handshake needs +> SHA-1, and `just chat` verifies the accept-key independently. +> +> **The gap it did NOT close:** there is still no RNG in the runtime. HMAC +> authenticates a token and cannot mint one, so CSRF and sessions stay blocked +> — which is why [39](39-web-framework-parity.md) leads with a random-bytes +> builtin rather than treating them as unblocked. ## Why this iteration exists Four consumers already wait on it, none able to proceed: diff --git a/docs/stories/language-runtime-database/40-shutdown-drain-guarantee.md b/docs/stories/language-runtime-database/40-shutdown-drain-guarantee.md new file mode 100644 index 0000000..6da3eb0 --- /dev/null +++ b/docs/stories/language-runtime-database/40-shutdown-drain-guarantee.md @@ -0,0 +1,158 @@ +--- +iteration: "40" +status: done +chain: 3 +--- + +# iteration 40 — the shutdown drain guarantee: a send before the stop flag is delivered + +> Part of [Story — one language, one runtime, one database, one binary](00-story.md). +> +> **Split out of [24](24-chat-websocket-workload.md) on 2026-08-27** because it +> is a runtime *semantic*, not a task in a sample's gate. It belongs to the +> actor lifecycle ([31](31-actor-lifecycle.md), absorbed into 24) and it is +> the half of "lifecycle" that nothing had stated: 31 gave actors a death +> notice, this gives the program a shutdown that does not lose mail. +> +> **Found by measurement, not review.** The chat gate's drain leg had been +> passing only because it drained a server the 1k soak had already warmed. +> Making every leg start its own server exposed it: +> [`2026-08-27-chat-drain-finding.md`](../../2026-08-27-chat-drain-finding.md). + +## The rule + +**A message sent before the stop flag is observed must be delivered and run +before the engine stops.** One sentence, and it is the whole iteration. It is a +guarantee, not a tuning parameter — which is why a spin count could never +express it. + +What it does *not* promise: that a message sent *after* the flag is delivered, +that a parked fiber is resumed, or that an actor gets unbounded time. The drain +window is the primary's, and it closes when the primary returns. + +## The bug, as measured + +Fresh server, two WebSocket clients, `SIGTERM`, both must receive a close frame: + +| Sample | Result | +| --- | --- | +| 5 fresh servers | 1 failure (`eof\|close`) | +| 12 fresh servers | 3 failures, one `eof\|eof` | +| 16 fresh servers | 5 failures | + +The failing client's socket reaches EOF with **no close frame and no +diagnostic** — the process exits and the kernel closes the fd. + +Traced with instrumentation on the sample's actors: `main` → Registry → Room → +Writer. The Registry runs and sees its room. The **Room never processes the +shutdown message**, so the Writer's close branch never runs. Clients that did +get a frame were saved by their own Reader noticing `env.stopping()`, not by the +room broadcast. + +## The design, as built + +`runtime/src/vm.c` already encoded the correct contract in `NEXT_RUNNABLE()`: +a worker that takes a stop while it has a live fiber returns 2 and **keeps +draining its inbox** until the primary sets `eng_shutdown`. Its comment says so +in as many words — "queued shutdown messages (close frames!) still run". + +`shard_main`'s own idle branch contradicted it. A worker with an empty run queue +waits in `wo_io_wait`, and on `WO_IO_STOP` it called `fib_reap_all` and +**broke** — abandoning whatever was still in its inbox, which `wo_engine_stop` +then freed wholesale during teardown. + +So the failure needed a shard that was *idle* at `SIGTERM`. A Room actor between +messages is exactly that, which is why the warm soak server hid it: warm shards +had live fibers and took the correct path. + +The fix makes the idle branch obey the same contract: while the primary's drain +window is open, an idle worker adopts its inbox and runs what arrives, yielding +between empty polls so a drain cannot become a hot spin across every core. Only +`eng_shutdown` — set by the primary after `main` returns — ends it. + +One branch, in one place, matching a contract the file already stated. + +## Progress + +| Piece | State | +| --- | --- | +| the idle-worker drain branch in `shard_main` (`runtime/src/vm.c`) | ✅ one branch, matching the contract `NEXT_RUNNABLE()` already stated | +| `sched_yield` on an empty poll so the drain cannot hot-spin | ✅ | +| fresh-server drain, repeated | ✅ **20 of 20**, from 5-in-16 failing | +| chat gate at the default 1k soak | ✅ **11 checks, 0 failures** — 1000/1000 clients, both `WO_IO` backends, ASan clean | +| full runtime battery (this touches every actor program's shard loop) | ✅ **36 suites** (18 × both dispatch flavors), 0 fail, `cli_smoke: OK`; compiler 556 checks 0 fail | +| the regression pin | ✅ the chat gate's drain leg, now that it starts its OWN (cold) server — that decoupling is what caught this. **Not** a corpus fixture or unit test: nothing in `runtime/test/` drives `wo_engine_start`/`wo_engine_stop` today, and no corpus fixture can trigger a stop, so pinning it below the gate means new multithreaded test infrastructure — named as its own cost, not smuggled in here | + +**Measured 2026-08-27.** Before: 5 of 16 fresh-server drains left a client at +EOF. After: **20 of 20 clean.** At the observed failure rate, 20 clean runs by +luck would be about 0.04%, so this is the fix rather than a quieter race. + +## Acceptance Criteria + +Met: + +- **Given** a fresh server with two connected WebSocket clients, **when** it is + sent `SIGTERM`, **then** both clients receive a close frame — **repeatedly**, + not once. The bug reproduced at 5 in 16, so a single green run proves nothing; + the criterion is a run of at least 16 with zero failures. + ✅ **20 of 20**, from 5-in-16 failing. A single run would have proved nothing. +- **Given** an actor whose shard is idle at the moment of the stop, **when** a + message is sent to it before the stop flag is observed, **then** its + `receive` runs before the engine stops. ✅ this is exactly the case that + failed — the Room between messages — and it is what the branch now covers. +- **Given** the drain window, **when** a worker has nothing to adopt, **then** + it does not hot-spin. ✅ `sched_yield()` on an empty poll; the 1k soak's RSS + and timing legs are unchanged (marker reached all 1000 in 28 ms). +- **Given** `just chat`, **when** it runs at the default soak, **then** all + legs pass on both `WO_IO` backends and under the ASan build with zero leaks. + ✅ 11 checks, 0 failures. The fd leg also settled the lazy-init question at + scale: **1000 connections left the count at 44**, unchanged after 20 more. +- **Given** the full runtime battery, **when** it runs, **then** no suite + regresses — this touches the shard loop every actor program uses. ✅ 36 suites + 0 fail, plus the compiler's 556 checks. +- **Given** a program with no worker shards (`WO_SHARDS=1`), **when** it stops, + **then** behaviour is unchanged. ✅ the gate's `WO_SHARDS=1` leg passes, and + the branch is unreachable there — `wo_engine_stop` returns early at + `nshards <= 1`, so a single-shard program never enters a worker loop. + +Outstanding: + +- **A pin below the gate.** The guarantee is currently proven by the chat gate + only. Nothing in `runtime/test/` drives `wo_engine_start`/`wo_engine_stop`, + and no corpus fixture can trigger a stop, so pinning it lower means new + multithreaded test infrastructure. Named as its own cost rather than assumed + cheap. + +## Out Of Scope + +- **Unbounded drain.** The window is the primary's and closes when `main` + returns. A program that wants longer holds the window open itself. +- **Delivering sends issued *after* the stop flag.** Nothing promises that, and + promising it would mean a program could refuse to exit. +- **Resuming parked fibers on stop.** `WO_SYS_STOPPED` unwinds them; that + contract is iteration 24's and stays. +- **A shutdown acknowledgement in the language surface.** The alternative fix + was a barrier the sample builds itself, rejected below. +- **`main` parking after the stop flag.** Still forbidden — a park after the + flag unwinds. `main` still spins; the point is that spinning now works + because the workers cooperate. + +## Info — the forks, settled + +1. **Engine guarantee, not a sample barrier.** The alternative was an + acknowledged drain: rooms confirm back to `main`, which waits. Rejected — + `main` cannot park after the stop flag, so it could only spin on the + acknowledgement anyway, and every future actor program would have to + re-implement the same handshake to avoid losing mail. A guarantee is stated + once; a barrier is re-invented per program. +2. **Not the spin budget.** Replacing the sample's `spin < 20000000` with a 1 s + wall-clock deadline still failed 2 of 12. More time cannot help when the + shard is not scheduled at all, and the reverted attempt cost a fixed second + on every shutdown. Recorded because a bigger spin is the obvious wrong fix. +3. **Not `dummy_writer()`.** Hoisting the shutdown message's placeholder actor + out of the drain path (it spawned during shutdown) left 5 of 16 failing. +4. **Yield rather than spin in the idle drain.** A worker polling an empty + inbox in a tight loop would burn a core per shard during the window and + starve the actors being drained. +5. **Chain position 3**, with [31](31-actor-lifecycle.md): it is lifecycle + semantics, and [24](24-chat-websocket-workload.md)'s gate is what proves it. diff --git a/docs/superpowers/plans/2026-08-28-wal-checkpoint.md b/docs/superpowers/plans/2026-08-28-wal-checkpoint.md new file mode 100644 index 0000000..6a1d70e --- /dev/null +++ b/docs/superpowers/plans/2026-08-28-wal-checkpoint.md @@ -0,0 +1,265 @@ +# databasev2 3 — WAL checkpoint (implementation plan) + +> **For agentic workers:** REQUIRED SUB-SKILL: Use +> superpowers:subagent-driven-development (recommended) or +> superpowers:executing-plans to implement this plan task-by-task. Steps +> use checkbox (`- [ ]`) syntax for tracking. +> +> **Style rule (user convention):** concept, reason, and required +> behaviour in words plus verification commands only — no implementation +> or test code blocks; the executor writes the code. + +**Goal:** reclaim disk and bound replay by rewriting the log as one record per +live row and swapping it in with `rename`, so boot replays a short log instead +of all history. + +**Architecture:** compaction writes the live store into a temporary file using +the existing record grammar and the existing append path, fsyncs it, renames it +over the live WAL, fsyncs the parent directory, and reopens the descriptor. +Recovery is untouched — boot still opens one file and replays it — and every +crash point is safe because `rename` is atomic. + +**Tech Stack:** C11, libc only. `pwrite`, `fdatasync`, `rename`, `open`, +`unlink`. No new dependency and no new file format. + +**Spec:** [`../specs/2026-08-28-wal-checkpoint-design.md`](../specs/2026-08-28-wal-checkpoint-design.md) + +## Global Constraints + +- **Recovery must not change.** No second source, no cutoff offset, no control + file. If a task finds itself editing the replay path, something has gone + wrong with the design and it should stop rather than proceed. +- **Every crash point falls back.** Before the rename the live log is untouched; + after it the new log is complete. There must be no window in which a reader + could observe a mixture. +- **The record grammar is frozen.** The whole argument for this design is that + it already suffices. A compacted log is INSERT records for live rows, ids + preserved exactly. +- **Bounded memory.** `stage()` grows the staging buffer by doubling and never + shrinks it, so dumping a whole store through one buffer would hold the entire + store in RAM — the unbounded growth databasev2 1 identified as how this engine + dies. The dump must flush periodically. +- **Compaction may run only where nothing is staged** — in practice immediately + after a barrier. Anywhere else, a staged record lands in a file about to be + replaced. +- **libc only**, no new syscall interface. Gates run through `just`. Never + commit on `master`; branch first. + +--- + +## Task 1 — `wo_wal_compact`: rewrite, fsync, rename, reopen + +**Files:** +- Modify: `database/src/wal.c`, `database/src/wal.h`. +- Test: `runtime/test/test_wal.c`. + +**Interfaces:** +- Produces: a compaction entry point taking the live WAL and the store, which + replaces the log with one INSERT record per live row and leaves the WAL usable + (descriptor reopened, offset correct). Returns success or failure; a failure + must leave the ORIGINAL log intact and usable, because a failed checkpoint is + not a durability event. +- Consumes: the existing append path and commit routine, and the bitmap walk + that `db.c` already performs in three places. + +- [ ] Read three things first and confirm them, because the design rests on + them: `apply_record` implements UPDATE as remove-then-recreate (so records are + full row images), `wo_wal_append_insert` takes an id and reads the row from + the store (so ids are preserved), and the tail scan treats a zero length field + as end-of-log (so the new file must be zero beyond its records). +- [ ] Test first, RED: build a store, age it (insert rows, then update the same + rows repeatedly so history exceeds live data), compact, then assert **both** + that the log got materially shorter AND that a fresh replay of it produces the + same rows with the same ids and the same values. Shorter alone is worthless — + a truncating bug also passes that. +- [ ] Verify RED for the right reason: the entry point does not exist yet. +- [ ] Implement the walk: for each class, iterate slots via the bitmap and + append one INSERT per live row. Reuse the append path; do not write a second + encoder. +- [ ] **Flush every K records rather than staging the whole store.** Point a + scratch WAL at the temp descriptor and commit periodically. State the chosen K + and why in a comment. Without this the dump holds the entire store in RAM. +- [ ] Sequence the switch exactly: fsync the temp file, `rename` over the live + path, **fsync the parent directory** (the rename is atomic in-kernel but the + directory entry is not durable until the parent is synced), then reopen the + descriptor — the old one refers to an unlinked inode — and reset the offset to + the new end of log. +- [ ] Handle failure without losing data: any error before the rename must + unlink the temp file and leave the live log untouched. A failed compaction is + a missed optimisation, **not** a durability failure, so it must NOT take the + fatal path databasev2 4 introduced. +- [ ] GREEN: `just wovm-test`. +- [ ] Commit. + +## Task 2 — a stale temp file is removed, never read + +**Files:** +- Modify: `database/src/wal.c` (the open path). +- Test: `runtime/test/test_wal.c`. + +**Interfaces:** +- Consumes: Task 1's temp-file naming. +- Produces: the guarantee that a crash mid-rewrite leaves nothing that can be + mistaken for data. + +- [ ] Test first, RED: place a temp file next to the log containing *plausible, + well-formed records* (not garbage — garbage would be rejected anyway and would + prove nothing), open the store, and assert the temp file is gone and the + replayed store is exactly what the live log said. +- [ ] Verify RED for the right reason. +- [ ] Remove any stale temp file when the WAL is opened. Note in a comment why + this is safe: the only way one exists is a crash before a rename, and its + contents are by definition not yet authoritative. +- [ ] GREEN: `just wovm-test`. +- [ ] Commit. + +## Task 3 — the trigger, and the ordering guard + +**Files:** +- Modify: `database/src/wal.c`, `database/src/wal.h` (remember the last + compaction's size; the policy decision), `runtime/src/vm.c` (call the check + after the barrier). +- Test: `runtime/test/test_wal.c`. + +**Interfaces:** +- Consumes: Task 1's compaction entry point. +- Produces: automatic compaction, and the invariant that it never runs with + records staged. + +- [ ] Extract the policy as a **pure decision** — given the log's used bytes, + the bytes the last compaction wrote, and a floor, should we compact? Pure + because it is then unit-testable without a store, which is the only way this + policy gets tested at all. +- [ ] Test the decision directly, RED then GREEN: below the floor it never + fires however bad the ratio; above the floor it fires exactly when used bytes + exceed the multiple; with no prior compaction it uses the floor alone. +- [ ] Record the bytes each compaction wrote, so the denominator is measured + rather than estimated. Estimating the live size would mean estimating Text, + and the compactor already knows the true number. +- [ ] Expose the floor and the ratio as env knobs, matching the existing idiom + (`WO_MAILBOX`, `WO_HEAP_MB`, `WO_SHARDS`, `WO_WAL_STATS`). **This is what + makes the policy testable** — a test sets a tiny floor and forces compaction + in a few writes instead of waiting for megabytes. Document them beside the + others. **Deviation from the spec, disclosed:** the spec spoke of a "manual + trigger for tests"; env-tunable thresholds serve that purpose without adding + language surface, which is the cheaper way to buy the same testability. +- [ ] **No timer.** If the implementer is tempted, the reason is in the spec: + Postgres' `CheckPointTimeout` bounds loss from unflushed buffers, our records + are durable at commit, and an idle log does not grow. +- [ ] Call the check from the one place that is safe — immediately after the + drain's barrier, where nothing is staged. Comment that this is a correctness + requirement and not a scheduling preference. +- [ ] Verify the guard: a test that stages records and then makes the policy + say yes must find compaction deferred, not executed. This is the assertion + that keeps the ordering rule true as the code moves. +- [ ] Verify durability is unaffected: `just db-bench --quick` — the crash and + restart legs must be unchanged, and part A's `wmix` legs must still batch. +- [ ] Commit. + +## Task 4 — kill -9 *during* compaction + +**Files:** +- Test: `runtime/test/test_wal.c` (extend the existing fork-based crash + battery). + +**Interfaces:** +- Consumes: Tasks 1–3. +- Produces: the evidence for the criterion the whole design is shaped around. + +- [ ] Read the existing crash battery first: a forked child inserts and acks + each committed id over a pipe while the parent SIGKILLs it mid-stream, then + the parent verifies every acked id survived. Extend that shape rather than + inventing a second harness. +- [ ] Drive compaction repeatedly in the child (a tiny floor makes it fire + often) while it inserts and acks, and kill at many instants so the kill lands + inside a rewrite, at the rename, and after it. +- [ ] Assert the property, not a state: after replay the store must equal + **either** the pre-compaction **or** the post-compaction content — never a + mixture — and **every acked id must be present**. A test that only checks "it + replayed without error" would pass on a silently truncated log. +- [ ] Assert no temp file survives a kill in a way that affects the next boot. +- [ ] Run the battery repeatedly, not once: this is a race, and one green run + proves very little. State how many repetitions were run in the commit message. +- [ ] GREEN: `just wovm-test` plus the repetitions. +- [ ] Commit. + +## Task 5 — measure: space, boot, and the pause + +**Files:** +- Modify: `scripts/db-bench.py` (a checkpoint leg), `docs/plan/perf-targets.md`, + `bench/baseline.json` (refresh, with the reason in the commit message). + +**Interfaces:** +- Consumes: Tasks 1–3. +- Produces: the before/after record, and the pause number the spec deliberately + refused to assume. + +- [ ] Capture the before numbers already measured on master, rather than + re-deriving them: `seed 20000` leaves a 986 614-byte log; 20 000 updates take + it to 2 590 262 bytes **with the same live rows**; boot+verify on that aged + store is 155 ms. +- [ ] Add a leg that ages a store, compacts it, and records: bytes before and + after, the ratio reclaimed, and boot time before and after. Age it by + updating the same rows — history must grow while the live set does not, or the + leg is measuring insert throughput instead of compaction. +- [ ] Measure the **stop-the-world pause** on the largest store the harness + builds and record it as a number. State the budget it must meet. +- [ ] **If the pause exceeds the budget, stop and report it.** That is the + finding the spec asked for, and the alternatives (incremental copy, + fork-and-dump) are bought against this number — not before it. +- [ ] Give the new metrics tolerances that match what they are: bytes reclaimed + is structural and can be gated tightly; the pause is wall-clock on a shared + box and cannot. Do not waive them all, which is the mistake part A's task 4 + made and had to undo. +- [ ] Verify the gate bites: doctor the reclaimed-bytes metric and confirm the + suite fails on exactly that metric. +- [ ] Refresh the baseline and confirm the **full** campaign passes against it. + The committed baseline is full-mode (`N=20000`, `crash_reps=3`) — writing a + quick-mode baseline over it is a regression, and part A made exactly that + mistake. +- [ ] Commit. + +## Task 6 — closeout + +**Files:** +- Modify: `docs/stories/databasev2/03-wal-checkpoint.md`, + `docs/stories/00-status.md`, `docs/plan/oop-vm/04-db-binding.md`, + `database/src/CODE-LOGIC.md`, `docs/examples/db-bench/README.md`. + +- [ ] `04-db-binding.md`: the normative ordering rule — compaction runs only + where nothing is staged, and what recovery does (unchanged: one file, replayed + from byte 0). This is the doc the spec named for it. +- [ ] `CODE-LOGIC.md`: why one file rather than snapshot-plus-tail, why + `rename` is the crash-safety primitive, why the dump flushes periodically, and + why a failed compaction is not a durability event. Reasoning, not call graph. +- [ ] README: the new env knobs beside the existing ones, and the checkpoint + leg. +- [ ] Story: progress, criteria split met/outstanding, and the measured + before/after. +- [ ] Board: standup entry in the six-question shape, and the chain note — + chain 6 was the last link, so say what the chain's completion means and what + is next. +- [ ] **Record the `resident: keys` obligation prominently, in the story and at + the compactor.** Compaction moves every record, so it invalidates every WAL + offset iteration 2 stores; the compactor must rebuild that map as it writes. + There is nothing to implement today because iteration 2's storage half does + not exist — which is exactly why this must be written where the next + implementer will hit it, not left in a spec they may not read. +- [ ] Full battery: `just wovm-test`, `just woc-test`, `just oop-e2e`, + `just db-bench`, `python3 scripts/linkcheck.py .` +- [ ] Commit. + +## Self-review notes + +- **Spec coverage.** Compaction and the switch → Task 1. Stale temp → Task 2. + Trigger, no timer, ordering rule → Task 3. Crash safety → Task 4. Space, boot, + pause → Task 5. Normative doc, `resident: keys` obligation → Task 6. +- **The riskiest task is 4**, not 1: Task 1's correctness is a single replay + comparison, while Task 4 is a race and can pass by luck. Hence the explicit + instruction to run it repeatedly and to state the count. +- **Task 2 looks trivial and is not.** A stale temp file containing well-formed + records is the one input that could be mistaken for data, so the test uses + plausible records rather than garbage. +- **One thing deliberately NOT a task:** rebuilding the `resident: keys` offset + map. It cannot be implemented against a feature that does not exist yet. + Recorded as an obligation in Task 6 instead of a stub nobody can test. diff --git a/docs/superpowers/plans/2026-08-28-wal-group-commit.md b/docs/superpowers/plans/2026-08-28-wal-group-commit.md new file mode 100644 index 0000000..f2887b1 --- /dev/null +++ b/docs/superpowers/plans/2026-08-28-wal-group-commit.md @@ -0,0 +1,249 @@ +# databasev2 4 part A — WAL group commit (implementation plan) + +> **For agentic workers:** REQUIRED SUB-SKILL: Use +> superpowers:subagent-driven-development (recommended) or +> superpowers:executing-plans to implement this plan task-by-task. Steps +> use checkbox (`- [ ]`) syntax for tracking. +> +> **Style rule (user convention):** concept, reason, and required +> behaviour in words plus verification commands only — no implementation +> or test code blocks; the executor writes the code. + +**Goal:** one durability barrier per drain instead of one per statement, so a +writer is acknowledged after the barrier that carried its record rather than +after a barrier of its own. + +**Architecture:** the barrier moves up, not out. Applying to RAM and staging the +record stay exactly where they are in `db.c`; the request path stops committing +after each append and instead holds its reply envelope, and shard 0 issues one +commit when it runs out of queued requests, then releases every held reply. Any +failure between "RAM mutated" and "record durable" ends the process with a +diagnostic. + +**Tech Stack:** C11, libc only. `pwrite` + `fdatasync` (unchanged — io_uring is +part B). The existing per-shard envelope inbox carries the requests. + +**Spec:** [`../specs/2026-08-28-wal-group-commit-design.md`](../specs/2026-08-28-wal-group-commit-design.md) + +## Global Constraints + +- **Durability is unchanged.** Every guarantee iterations 9 and 22 proved holds + identically: replay-whole-or-not-at-all, torn-tail drop, no acknowledged + write ever lost. This changes when the barrier runs, never what the log holds. +- **A writer is released only after the barrier carrying its record.** Never + before, and never on the strength of a different batch's barrier. +- **libc only.** No new dependency, no new syscall interface in part A. +- **The payoff metric is `durable.sN.mixwrite`** (today 480 ops/s, p99 + 5888 µs). `durable.s1.*` and both `seed` legs are regression guards, not + targets — a serial writer and an all-inline shard have nothing to batch with. +- **`WO_T_IO` leaves the write path.** A commit or staging failure is fatal, not + catchable. Exit 1 is a trap and exit 2 is a refusal, so this takes a third + status of its own. +- Gates run through `just`. Never commit on `master`; branch first. + +--- + +## Task 1 — a failed barrier is detected, and fatal + +**Files:** +- Modify: `database/src/wal.c` (the commit routine's failure returns; a new + fatal-commit entry point beside it), `database/src/wal.h` (declare it). +- Test: `runtime/test/test_wal.c` (a new case in the existing suite). + +**Interfaces:** +- Produces: a commit entry point that takes the WAL and the number of records + in the batch, commits, and on failure writes one stderr line naming the + failing operation, the `errno` text, the WAL path and the record count, then + exits with the durability-failure status. Tasks 2 and 3 call only this. +- Consumes: the existing staging buffer and commit routine. + +- [ ] Read the commit routine first and confirm what it already reports: it + loops `pwrite` until the staged buffer is written, then `fdatasync`, and + returns non-zero on either failing. Confirm the WAL struct carries its path, + or add it — the diagnostic is worthless without it. +- [ ] Test first, RED: assert the commit routine reports failure when the + descriptor is unusable (a closed descriptor gives `EBADF`). This proves the + error is *detected*; it does not exercise the exit. +- [ ] Verify RED for the right reason — the case must fail because the + assertion is unmet, not because the suite does not compile. +- [ ] Add the fatal entry point. It must distinguish the two operations in its + message: a `pwrite` failure and an `fdatasync` failure are different + operational problems and the operator needs to know which. +- [ ] GREEN: `just wovm-test`. The new case passes and no existing case moves. +- [ ] **Disclosed gap, record it in the commit message:** the exit path itself + is not exercised. Forcing a real `fdatasync` failure needs a full or + read-only filesystem, which the gate cannot arrange without mount + privileges. Do NOT add a fault-injection switch to buy coverage — shipping a + binary that can be told to kill itself is the worse trade, and the spec + rejected it. +- [ ] Commit. + +## Task 2 — the barrier moves to the drain point; replies are held + +**Files:** +- Modify: `database/src/db.c` (the request-path arms only — the three commit + calls inside the marshaled-statement executor), `runtime/src/vm.c` (the + envelope drain loop's DB-statement branch and the end of that loop). +- Test: no new fixture; the existing durability battery is the test. It already + covers exactly what could break. + +**Interfaces:** +- Consumes: Task 1's fatal commit entry point. +- Produces: the invariant later tasks measure — at most one barrier per drain, + and every held reply released only after it. + +- [ ] Read the drain loop's DB-statement branch first. Today it executes the + request, marks it done, then immediately pushes a reply envelope that unparks + the requester. Note that it runs on shard 0's thread, serialized — that is + why no locking is needed anywhere in this task. +- [ ] Remove the three commit calls from the request-path executor in `db.c`. + Leave applying to RAM and staging untouched, and leave the **inline** path's + three commit calls alone — Task 3 owns that path and conflating them is how + this change breaks the single-shard configuration. +- [ ] In the drain loop, collect reply envelopes in a local list instead of + pushing them as each request finishes. A local is correct and deliberate: + nothing needs to survive the loop, and per-shard state would outlive the + batch it describes. +- [ ] At the end of the drain loop, if anything was staged, call Task 1's fatal + commit once, then push every held reply. +- [ ] Handle the empty case: a drain that executed no DB statements must not + commit and must not touch the staging buffer. +- [ ] Verify the ack contract has not moved: `just wovm-test` — the WAL and + table suites must be unchanged, since neither knows about batching. +- [ ] Verify durability end to end: `just db-bench --quick`. The restart-replay + and `kill -9` crash legs are the ones that matter — a kill between staging and + the barrier must lose only unacknowledged writes. **If a crash leg fails here, + stop; do not adjust the test.** That leg failing means the ack contract broke, + which is the one thing this task may not do. +- [ ] Commit. + +## Task 3 — the inline path keeps its own barrier, and says why + +**Files:** +- Modify: `database/src/db.c` (the inline path's three commit calls — replace + with Task 1's fatal entry point), plus the comment above them. + +**Interfaces:** +- Consumes: Task 1's fatal commit entry point. +- Produces: nothing new. This task exists to make the asymmetry deliberate and + legible rather than accidental. + +- [ ] Replace the inline path's three commit calls with Task 1's fatal entry + point, batch size one. Behaviour is unchanged — this is the fatal-failure + rule reaching the second path, not batching. +- [ ] Write the comment that explains the asymmetry, because the next reader + will otherwise "fix" it: the inline path cannot hold a reply, because it + returns into its own fiber rather than unparking a requester. Batching it + would require parking that fiber on the barrier, which is part B's machinery + and deliberately out of part A. +- [ ] Confirm the ordering assumption holds: because the drain loop always + commits before it ends, nothing uncommitted is ever left staged when an + inline statement runs. If that stops being true the inline path would commit + another statement's record early — say so in the comment as the reason the + drain must commit unconditionally. +- [ ] Verify: `just wovm-test` and `just db-bench --quick` both green, and + `WO_SHARDS=1` in particular — the single-shard configuration takes this path + exclusively. +- [ ] Commit. + +## Task 4 — prove batches actually form + +**Files:** +- Modify: `scripts/db-bench.py` (new metrics and their tolerances), + `docs/examples/db-bench/main.wo` only if the batch figures cannot be observed + without the sample reporting them. +- Test: the driver's own gate-bites check. + +**Interfaces:** +- Consumes: the batching from Task 2. +- Produces: mean batch size, peak batch size and peak staged bytes as recorded + metrics, so Task 5 measures a mechanism that is known to engage. + +- [ ] Decide where the counters live and prefer the smallest surface: the + runtime can report them at exit, or the driver can derive them. Do not add a + builtin for this — the numbers are diagnostic, not part of the language. +- [ ] Record mean and peak batch size under the concurrent multi-shard write + workload. **This is the task's real point:** if batches are always one, the + feature is inert and any throughput change came from somewhere else, so the + measurement in Task 5 would be attributing a win to the wrong cause. +- [ ] Record peak staged bytes. This settles whether the batch needs a cap with + a number instead of a guess — the spec deliberately shipped no cap because the + request queue is already bounded upstream by iteration 24's mailbox caps. +- [ ] Give the new metrics wide tolerances. Batch size is a function of arrival + timing, so gating it tightly would gate the scheduler; what must be gated is + that it is greater than one under contention. +- [ ] Verify the gate bites: doctor the recorded mean batch size to one and + confirm the suite fails on exactly that metric. +- [ ] Commit. + +## Task 5 — measure the payoff, gate it, write it down + +**Files:** +- Modify: `bench/baseline.json` (refresh, with the reason in the commit + message), `docs/plan/perf-targets.md` (a new section). + +**Interfaces:** +- Consumes: Tasks 2 and 4. +- Produces: the before/after record every later optimization argues against. + +- [ ] Capture the before numbers from the committed baseline rather than + re-measuring them: `durable.sN.mixwrite` 480 ops/s, p50 538 µs, p99 5888 µs; + `durable.s1.mixwrite` 1023 ops/s, p99 664 µs; `seed` ~4460 ops/s on both. +- [ ] Run the full campaign, not the quick one, and record after numbers for + the same metrics on the same machine. A payoff measured across machines is + not a payoff. +- [ ] Assert the scoped criterion: **`durable.sN.mixwrite` throughput up and + p99 down**, with `durable.s1.*` and both `seed` legs not regressed. Do not + report the s1 seed number as a disappointment — a serial writer has nothing + to batch with, and the spec says so. +- [ ] Write the `perf-targets.md` section: the before/after table, the mean and + peak batch size that produced it, and the peak staged bytes. State the + inversion that motivated the work — multi-shard concurrent writes were 2× + slower than single-shard with a 9× worse p99 — and whether it is now gone. +- [ ] If the payoff is absent or small, **say so and stop.** That is a finding, + not a failure: it would mean the barrier was not the bottleneck the baseline + implied, and part B must not be started on an unproven premise. +- [ ] Refresh the baseline and confirm `just db-bench` passes against it, then + re-confirm the gate bites on a doctored write metric. +- [ ] Commit. + +## Task 6 — closeout + +**Files:** +- Modify: `docs/stories/databasev2/04-io-uring-commit.md` (progress, criteria + split met/outstanding, the landing banner), + `docs/stories/00-status.md` (standup entry, chain note), + `docs/plan/oop-vm/01-error-catalog.md` (the `WO_T_IO` removal and the new + exit status), `database/src/CODE-LOGIC.md` (a group-commit section). + +- [ ] Story: record what landed and what did not. The outstanding items are + single-shard concurrent batching (needs the inline park) and part B itself. + Keep the corrected premise visible — this iteration was written as + "fsync-per-commit" and the engine was fsync-per-statement. +- [ ] Error catalogue: `WO_T_IO` no longer reachable from a write, and the new + durability-failure exit status documented beside the trap and refusal codes. + A language-visible removal that is not written down is a trap for the next + reader. +- [ ] `CODE-LOGIC.md`: the commit path as built — where the barrier runs, why + replies are held, why the inline path is asymmetric, and the one rule for + failure. Explain the reasoning, not the call graph. +- [ ] Board: the standup entry in the six-question shape, and the chain note — + part B's go/no-go now rests on Task 5's number. +- [ ] Full battery after the doc edits: `just wovm-test`, `just woc-test`, + `just oop-e2e`, `just db-bench`, `python3 scripts/linkcheck.py .` +- [ ] Commit. + +## Self-review notes + +- **Spec coverage.** Queue-drain boundary → Task 2. Fatal failure rule → Tasks 1 + and 3. Held replies and the ack contract → Task 2. No batch cap, settled by + measurement → Task 4. Payoff and its scoping → Task 5. `WO_T_IO` removal → + Task 6. The disclosed abort-coverage gap → Task 1's last step. +- **The riskiest task is 2**, and its risk is concentrated in one place: the + crash legs of the durability battery. That is why the plan says stop rather + than adjust if they fail. +- **Task 3 looks like a no-op and is not.** Without it the inline path keeps a + catchable `WO_T_IO` while the request path aborts, which is precisely the + per-path unevenness this spec exists to remove. +- **Task 4 before Task 5 is deliberate.** Measuring a payoff before proving the + mechanism engages is how a win gets attributed to the wrong cause. diff --git a/docs/superpowers/specs/2026-08-28-wal-checkpoint-design.md b/docs/superpowers/specs/2026-08-28-wal-checkpoint-design.md new file mode 100644 index 0000000..90deb2e --- /dev/null +++ b/docs/superpowers/specs/2026-08-28-wal-checkpoint-design.md @@ -0,0 +1,196 @@ +# WAL checkpoint — design + +> databasev2 [3](../../stories/databasev2/03-wal-checkpoint.md), chain 6. +> Brainstormed and approved 2026-08-28, after +> [databasev2 4 part A](2026-08-28-wal-group-commit-design.md) landed. +> +> **One sentence:** compact the log by rewriting it as one record per live row +> into a temporary file, then `rename` it over the live WAL — so recovery is +> unchanged and crash safety comes from the filesystem. + +## Decisions taken (the brainstorm's forks, settled) + +| Fork | Decision | +| --- | --- | +| Snapshot format | **None.** The compacted log *is* the snapshot, in the existing record grammar | +| One source or two | **One.** Rewrite + atomic `rename`; boot logic is untouched | +| Trigger | **Volume only**, as a ratio against the last compaction's own size, with an absolute floor. **No timer** — see below | +| Write availability | **Stop-the-world**, measured against a stated budget rather than assumed acceptable | +| Composition with group commit | Compaction runs only where **nothing is staged** — immediately after a barrier | +| `resident: keys` (iteration 2) | Compaction **rebuilds the offset map** as it writes. It cannot be left to discover this later | + +## Why one file, and why Postgres cannot do it + +Postgres was read for this (`.dev/reference/postgresql`), and the conclusion is +that its design is *unavailable* to us — which is what makes the simpler option +legitimate rather than lazy. + +| | PostgreSQL | writeonce | +| --- | --- | --- | +| Where data lives | heap/data files; the WAL is a redo tail | **the WAL is the only durable form**, replayed into RAM | +| WAL contents | page deltas and full-page images | **full row images** — `apply_record` implements UPDATE as remove-then-recreate | +| Compaction | **never**; segments before the redo point are recycled by `rename` or unlinked | possible, because a log of row images *is* a complete store | +| Bounded replay | recovery starts at the redo LSN in the control file | recovery starts at byte 0 of a *shorter* log | +| Crash safety of the switch | control file written in place, full block, torn writes caught by **CRC32C** (`update_controlfile`) | one `rename` | +| Trigger | `CheckPointTimeout` (300 s) **or** WAL volume (`XLogCheckpointNeeded`) | volume only | +| Pause | none; flush is spread over time in a **separate process** | stop-the-world | + +Postgres cannot compact its WAL because a compacted redo log is not a store — +its records describe changes to pages that live elsewhere. Ours describe whole +rows, so the compacted log needs no companion. That single difference removes +the control file, the redo pointer, the second recovery source, and the separate +process from our design. + +**What is worth porting is not the architecture but the ordering discipline:** +publish the new "recovery starts here" atomically and *last*, so a crash at any +instant falls back to the previous state with nothing to undo. Postgres achieves +that with a redo pointer computed at checkpoint *start* and a control file +updated at the *end*. We achieve the same property with `rename`, in one +syscall, because we can swap the entire data set atomically and Postgres cannot. + +**Correction to a prior exploration doc.** +`docs/plan/exploration/postgresql/buffer-and-checkpoint.md` states that Postgres +updates its control file by rename ("the same in `BasicOpenFile` + +`fsync_parent_path`"). It does not — `update_controlfile` opens the existing file +`O_WRONLY`, writes a zero-padded full block in place, and relies on CRC32C to +detect a torn write. That doc also assumes writeonce has **segment files** +("records before that LSN are *known* to be in the segment files"), which it +does not and, per databasev2 2, deliberately will not. The doc predates the +databasev2 direction and should be annotated rather than followed. + +## The design + +### Compaction + +Run on the owner shard, which owns the WAL. Walk each class's live rows — the +bitmap-over-slabs walk that three call sites in `db.c` already perform — and +append one INSERT record per live row to a **new** file, using the existing +append path. No new encoder, no new decoder, no format. + +Then: fsync the new file, `rename` it over the live path, fsync the parent +directory (the rename's atomicity is in-kernel; the directory entry is not +durable until the parent is synced — Postgres does the same, and the existing +exploration doc is right about *this* part), and reopen the WAL descriptor, +because the old one now refers to an unlinked inode. + +**Every crash point is safe without any recovery logic of ours.** Before the +rename, the live WAL is untouched and the temp file is garbage. After it, the new +log is complete by construction. There is no window in which a reader could see a +mixture, so the acceptance criterion — "recovery produces the same consistent +store as if the checkpoint had never started" — is satisfied by `rename`, not by +code we must get right. + +Two obligations follow. Boot must **unlink a stale temp file** if one is present, +because a crash mid-rewrite leaves one behind and it must never be mistaken for +data. And the temp file must be zero-padded beyond its records exactly as the +live WAL is, because the tail scan identifies the end of the log by a zero +length field. + +### When it runs, and where in the sequence + +**The point matters more than the policy.** The drain stages records into one +buffer and commits them together; compaction rewrites the file those records +would land in. So compaction may run **only when nothing is staged** — in +practice, immediately after a barrier, before the next statement is served. +Anywhere else and a staged record would either be written to a file about to be +replaced, or be lost with it. This is the normative ordering rule that +[`04-db-binding.md`](../../plan/oop-vm/04-db-binding.md) must carry. + +**Trigger: volume, as a self-tuning ratio.** Compact when the WAL's used bytes +exceed a multiple of the bytes the *last* compaction wrote, with an absolute +floor so a small store never bothers. The denominator is known exactly — the +compactor wrote it — so this needs no estimate of the live set's size, which is +not cheaply knowable when rows hold Text. The floor exists because a store whose +whole log is a few hundred kilobytes has nothing to reclaim. + +**No timer, and that is a deliberate difference from Postgres.** Postgres needs +`CheckPointTimeout` because its dirty buffers are not durable until flushed — an +idle-but-dirty system must still checkpoint or it loses data. Our records are +already durable at commit; a checkpoint reclaims space and shortens boot and +nothing else. An idle system's log does not grow, so a timer would fire with +nothing to do. Adding one would be copying Postgres' mechanism without its +reason. + +A manual trigger exists for tests, because a policy that can only be observed by +waiting is a policy that cannot be tested. + +### The pause, and how it is judged + +Compaction is stop-the-world: the owner shard rewrites the log as one long +operation while no statement is served. This is the simplest correct thing, and +part A's own experience argues for measuring before buying complexity to avoid +it. The dump is O(live rows) encodings plus one write and one barrier, so the +expectation is that it is fast — but an expectation is not a measurement, and +the proof plan below states the budget it must meet. + +If the measured pause exceeds the budget, **that is a finding and a follow-up, +not something this iteration solves by adding concurrency.** The alternative +designs (incremental copy, fork-and-dump) cost exactly what Postgres pays, and +should only be bought against a number. + +### The interaction that will otherwise be discovered late + +**Compaction invalidates every stored WAL offset.** Rewriting the log moves every +record, so any offset captured from the old file is meaningless afterwards — not +stale-but-readable, but pointing at an arbitrary byte of a different file. +[Iteration 2](../../stories/databasev2/02-table-storage-modes.md)'s +`resident: keys` stores exactly such offsets, one per row, and reads rows back +through them. + +The compactor therefore **rebuilds the offset map as it writes**: it is emitting +the new records and knows each one's new position, so this is the cheap +direction and the only one that keeps both features usable together. The +alternative — forbidding compaction while any `resident: keys` table is live — +would mean the feature that exists to handle huge tables is incompatible with +the feature that stops their log growing forever. + +This is recorded here because iteration 2's storage half is not yet +implemented, so nothing will fail today. It will fail later, in a way that looks +like data corruption rather than a design gap. + +## Proof plan + +| Claim | How it is proven | +| --- | --- | +| Space is reclaimed | An aged store shrinks. Measured today on master: `seed 20000` gives a 986 614-byte log; 20 000 updates take it to 2 590 262 bytes with **the same live rows**. Compaction must return it to approximately the former | +| Replay is bounded | Boot time on the aged store before and after, recorded. Measured today: boot+verify on that store is 155 ms | +| Crash safety | `kill -9` at many instants *during* compaction, then replay: the store must equal either the pre-compaction or post-compaction state, never a mixture, and no acked write may be missing. This is the criterion the whole design is shaped around, so it gets the crash battery's treatment rather than one case | +| A stale temp file is harmless | Boot with one present, containing plausible records: it is removed and never read | +| The pause is known | The stop-the-world pause measured on the largest store the harness builds, recorded as a number with a stated budget — not asserted to be acceptable | +| The ordering rule holds | Compaction with records staged must be impossible by construction; a test that stages and then requests compaction must find it deferred, not executed | +| Nothing regressed | The full battery, and specifically part A's `wmix` legs: compaction must not change the ack contract or the batching it introduced | + +## Out of scope + +- **A second file, a control file, or a redo pointer.** The Postgres shape, + priced above and not needed once the log is self-sufficient. +- **Avoiding the pause.** Incremental or forked dumps are bought against a + measurement, not in advance. +- **Per-shard compaction policy.** One owner shard owns the WAL today; when that + changes, this decision is revisited with it. +- **Compacting away tombstones across shards, or any cross-shard coordination.** + There is one log. +- **io_uring for the rewrite** — part B of databasev2 4, whose premise is + already under revision. +- **Changing the record grammar.** The entire argument for this design is that + the grammar already suffices. + +## Alternatives rejected + +**Snapshot + WAL tail (the Postgres shape).** Rejected because it buys write +availability at the cost of a second recovery source, a cutoff offset, a control +file with its own torn-write detection, and a crash-safety guarantee that +depends on our ordering rather than on `rename`. Postgres pays this because its +log cannot stand alone; ours can. + +**Compacting in place.** Rejected outright: there is no crash point at which a +partially rewritten live log is recoverable, and it trades the one property that +makes this design defensible for nothing. + +**A timer trigger.** Rejected with a reason rather than on taste: Postgres' timer +exists to bound data loss from unflushed buffers, and we have no unflushed +buffers. An idle log does not grow. + +**A ratio against an estimated live-set size.** Rejected in favour of the last +compaction's measured output, because estimating the live size means estimating +Text, and the compactor already knows the true number. diff --git a/docs/superpowers/specs/2026-08-28-wal-group-commit-design.md b/docs/superpowers/specs/2026-08-28-wal-group-commit-design.md new file mode 100644 index 0000000..4b00924 --- /dev/null +++ b/docs/superpowers/specs/2026-08-28-wal-group-commit-design.md @@ -0,0 +1,208 @@ +# WAL group commit — design + +> databasev2 [4](../../stories/databasev2/04-io-uring-commit.md), part A. +> Brainstormed and approved 2026-08-28. +> +> **This spec covers batching only.** The iteration was split during the +> brainstorm: part A amortises one durability barrier across many statements, +> part B (io_uring submission) is deferred until A's measurement says whether +> the blocking boundary is still the bottleneck. That split matches the +> iteration's own fork 1 — "drop-in behind `wo_wal_commit` first, an async +> variant only if the scheduler proves the blocking boundary is the +> bottleneck" — and it means the throughput win arrives behind a much smaller +> correctness surface. + +## Decisions taken (the brainstorm's forks, settled) + +| Fork | Decision | +| --- | --- | +| Scope | **Batching first, io_uring later.** Two independent wins were being carried as one; only the first needs a new syscall interface, and it is where most of the number lives | +| Batch boundary | **Queue-drain.** Shard 0 stages every pending write request, then commits once. No timer, no tunable | +| Failure | **Fatal, diagnosed abort.** Any failure between "RAM mutated" and "record durable" ends the process | +| Batch cap | **None initially.** Measure peak staged bytes; add a cap only if the queue's existing upstream bound proves insufficient | +| Abort coverage | Unit-test the failure *return*; the abort path itself stays covered by inspection, and that gap is disclosed | + +## The problem, read off the engine + +The story says "replace fsync-per-commit with io_uring group-commit". Read +against the code, the premise needed correcting: the engine does not commit per +*commit*, it commits per **statement**. `db.c` calls `wo_wal_commit` +immediately after every append, at all six sites — insert, update and remove, +each on both the inline and the DB-actor path. Every single row change is one +`pwrite` plus one `fdatasync`. + +That is what the numbers say too. Iteration 22's baseline records durable writes +at **4460 ops/s** single-shard and mixed writes at **1023 ops/s**, p50 **430 µs**, +p99 **664 µs** — against **1.28M ops/s** for durable reads. Writes are roughly +290× slower than reads, and the barrier is the whole of it. + +**The batching machinery already exists and is simply never used.** +`wo_wal_commit` writes `w->buf` for `w->len` bytes — a staged buffer that can +hold any number of records. Today it never holds more than one, because the +caller commits immediately after staging. So part A is closer to removing calls +than to adding a mechanism. + +## The design + +### The commit path + +The six `wo_wal_commit` calls come out of `db.c`. Applying to RAM and staging +the record stay exactly where they are; only the barrier moves, up to the point +where shard 0 runs out of work. + +Shard 0 owns the WAL — DB statements from other shards arrive as marshaled +request envelopes and are executed on shard 0's thread, serialized, and a reply +envelope unparks the requester. The change is that **the reply is held rather +than sent**: shard 0 executes and stages each queued request, keeps draining +while requests remain, then issues one barrier, and only then releases every +held reply. + +Each requester therefore unparks having been acknowledged after the barrier that +carried *its* record — the ack contract the story states, which today is true +only because every batch has one member. + +A statement executing inline on shard 0 (rather than arriving as a request) +stages and commits before returning, as it does now. It has no reply to hold — +it returns into its own fiber — and because the drain always commits before it +ends, nothing uncommitted is ever left staged when the inline path runs. + +### Why queue-drain, and what it costs + +The batch boundary is the queue going empty, not a tick and not a timer. Two +properties follow, and they are the reason to prefer it: + +- **A lone writer pays nothing.** One queued request means a batch of one, which + is today's path at today's latency. Batching engages only under genuine + contention, so an idle system is not taxed to serve a busy one. +- **The batch self-tunes.** Its size is whatever actually accumulated between + drains, so it grows with load rather than with a configured number. There is + nothing to set and nothing to set wrong. + +The rejected alternative was the iteration's recorded leaning, the shard tick. +That leaning was recorded when the batch was assumed to ride an io_uring +submission; with batching landing first, a tick boundary would add up to one +quantum of latency even to a lone writer — paying the cost of batching when +there is nothing to batch with. + +**No batch cap ships initially, and that is a decision rather than an +oversight.** databasev2 1 established that unbounded growth is precisely how +this engine dies without warning, so the instinct to bound it is right. But the +request queue is already bounded upstream by iteration 24's mailbox caps, and a +second bound on the same quantity is a knob that can only be wrong. The proof +plan measures peak staged bytes so the question is settled by a number. + +## Failure: one rule, replacing three behaviours + +Today's rollback is uneven, and the code says so. An `insert` whose commit fails +removes the row again, under a comment claiming RAM never claims what disk has +not acknowledged. An `update` or a `delete` whose commit fails does **not** roll +back — its comment admits the state plainly: RAM ahead of disk, trap, do not +ack. Nothing acknowledged is lost, but the process continues with divergent +state, and batching would multiply that from one row to as many as the batch +held. + +The rule that replaces it: **once a statement has mutated RAM, the only outcomes +are durable or process death.** It covers both failure points identically — +a staging failure and a barrier failure have the same consequence, RAM ahead of +disk with no way back, and only one of the three verbs can undo itself. + +Retrying is not an alternative worth designing for. On Linux a failed `fsync` +may already have discarded the dirty pages, so a second call can report success +having written nothing; the recovery that actually works is replay, which +returns exactly the last durable state. That is what the log is for. + +**This removes `WO_T_IO` from the write path.** A program can no longer catch a +disk failure on a write. The removal is deliberate — there was never a +recovery a program could meaningfully perform with its RAM ahead of its disk — +but it is language-visible and must be stated in the story banner and the error +catalogue, not slipped in. + +The diagnostic has to earn the abort: the failing operation, the `errno` text, +the WAL path, and the number of records in the batch, on stderr, then exit with +a status of its own. Exit 1 is a trap and exit 2 is a refusal, so a durability +failure takes a third. `abort()` is rejected — a core dump on a full disk is +noise, not evidence. + +## What will improve, and what will not + +**Corrected 2026-08-28, after reading the baseline properly.** The spec first +pointed at `durable.s1.seed` as the payoff metric. That was wrong, and the +reason is structural rather than a matter of degree. + +Worker shards hold no WAL at all — the runtime asserts it — so every DB +statement on a worker marshals to shard 0 and parks, while a statement already +on shard 0 executes inline. **A queue of write requests therefore exists only +when other shards are writing.** Batches form where there is a queue: + +| Workload | Today | Batching | +| --- | --- | --- | +| `durable.sN.mixwrite` — concurrent writers across shards | **480 ops/s, p99 5888 µs** | **the target.** N shards marshal N writes and shard 0 pays N barriers serially; one barrier replaces them | +| `durable.s1.mixwrite` — concurrent writers, one shard | 1023 ops/s, p99 664 µs | **no change.** Every write is inline with no queue, so no batch forms | +| `durable.*.seed` — one serial writer | ~4460 ops/s | **no change**, under any batching scheme. There is nothing to batch with | + +The inversion in those numbers is the finding worth keeping: **multi-shard +concurrent writes are currently 2× slower than single-shard with a 9× worse +p99.** Adding shards makes durable writing worse today, because every marshaled +statement still buys its own barrier on the owner. That is the pathology group +commit exists to remove, and it is a better argument for this iteration than the +one the story recorded. + +**Single-shard concurrent batching is deliberately out of part A.** It would +need the inline path to park its fiber on the barrier rather than commit +synchronously — the same parking machinery part B needs anyway. Deferring it +keeps A to one mechanism, and B inherits the reason to build it. + +So the acceptance criterion is scoped: **`durable.sN.mixwrite` throughput up and +its p99 down; `durable.s1.*` and both `seed` legs must not regress.** A plan +that reported "no improvement" against the s1 seed number would be measuring a +workload this change cannot help. + +## Proof plan + +| Claim | How it is proven | +| --- | --- | +| The payoff is real | **`durable.sN.mixwrite`** before and after on one machine, recorded in `perf-targets.md`. Today 480 ops/s, p99 5888 µs. `durable.s1.*` and both `seed` legs are regression guards, not targets — see the section above | +| Durability is unchanged | Iteration 22's crash battery, unaltered: concurrent writers, `kill -9` mid-stream, replay. **The critical test** — a kill between staging and the barrier must lose only unacknowledged writes | +| Batches actually form | New metrics for mean and peak batch size under contention. If batches are always one, the feature is inert and any throughput change came from somewhere else | +| No idle tax | Single-writer p99 must not regress against the current baseline | +| The cap question is answered | Peak staged bytes recorded per run | +| A failure is detected | `test_wal.c` asserts `wo_wal_commit` reports failure on a bad descriptor | + +**One disclosed gap.** Forcing a genuine `fdatasync` failure needs a full or +read-only filesystem, which the gate cannot arrange without mount privileges. +The unit test proves the error is *detected*; the abort that follows it stays +covered by inspection. The alternative — a fault-injection switch — means +shipping a binary that can be told to kill itself, which is a worse trade. This +gap is recorded rather than hidden, because iteration 40 was exactly a fatal +path that nothing exercised. + +## Out of scope + +- **io_uring submission.** Part B, and it only earns its complexity if A's + measurement shows the blocking boundary still dominating. A's parking and ack + machinery is what B would build on, so nothing here is wasted either way. +- **`transaction { }`** — language iteration 18. A transaction already *is* a + staged batch, so the two compose without either knowing about the other; that + is a reason not to entangle them now. +- **Checkpoint and compaction** — databasev2 3. This changes when the barrier + runs, never what the log contains. +- **The read path.** databasev2 1 measured that appending under memory pressure + costs about 1% while random reads cost 273×, so the pressure is on reads — + but that is iteration 2's `resident: keys` question, not this one. +- **Rollback with pre-images.** Rejected above: it would add per-write cost on + every statement to serve a path that ends the process anyway. + +## Alternatives rejected + +**Tick-boundary batching** — the iteration's recorded leaning, superseded by +the split. It taxes an idle system to serve a busy one. + +**Count-or-timer batching** — two tunables, and the timer reintroduces the tick +problem with extra configuration. + +**Full rollback with an undo log** — keeps `WO_T_IO` catchable, at the price of +capturing pre-images for every update and delete, paid on every write, to +support continuing in a state the engine cannot trust. + +**Keeping today's per-verb behaviour** — turns a rare one-row divergence into a +routine N-row one, silently. diff --git a/justfile b/justfile index 77de45f..45b2f7c 100644 --- a/justfile +++ b/justfile @@ -60,6 +60,13 @@ site: fibers: ./scripts/fibers-accept.sh +# chat: iteration 24's gate (docs/examples/chat) — rooms/presence/broadcast +# over WebSocket via actors: functional on both WO_IO backends, the +# 1k-clients-one-hot-room soak (fds/RSS accounted), SIGTERM drain with +# close frames, and an ASan leg. `just chat` runs it (CHAT_SOAK=N trims). +chat: + ./scripts/chat-accept.sh + # db-actor: arc stage 3's gate (docs/examples/db-actor) — worker-shard # actors read/write the database through the transparent DB actor; WAL # replay pair included. `just db-actor` runs it. diff --git a/runtime/src/CODE-LOGIC.md b/runtime/src/CODE-LOGIC.md index 50318d5..454c479 100644 --- a/runtime/src/CODE-LOGIC.md +++ b/runtime/src/CODE-LOGIC.md @@ -257,3 +257,100 @@ layout. - **`listen_unix` sets O_NONBLOCK on the listener itself** — accept4's SOCK_NONBLOCK flags the ACCEPTED socket only; a blocking listener would block the whole shard (found by the seam probe, both backends). + +## Actor lifecycle: call, death, monitor, timers (iteration 24, ids 88–90) + +Four pieces that together answer "what happens to an actor that is waiting, +that dies, that watches, or that wants to be woken later". All four live in +`vm.c` with their entry points in `builtin.c`; the structures are in `vm.h`. + +**`call` (id 88) — a send that waits.** An ordinary `send` returns immediately; +`call` parks the calling fiber and resumes it with the receive's return value. +The reply is a **typed scalar**, which is what let the agreement be checked at +compile time (WO-E226) rather than carried as a tagged value at runtime. The +caller is never left hanging: if the callee dies mid-call, or the address is +already dead, the caller **traps catchably** instead of parking forever. That +is the property worth keeping in mind when reading the code — every path out of +a call either resumes the fiber or traps it. + +**Death.** A `receive` that traps uncaught marks the actor dead on its home +thread. From then on sends to it drop silently, calls trap, queued callers are +error-unparked, and its state and mailbox are released. Silent-drop for sends +is deliberate: a sender cannot handle another actor's failure, and making every +`send` fallible would put a `try` on every line. + +**The mailbox cap and its counter.** One cap for every mailbox (default 1024, +`WO_MAILBOX` overrides at boot; the chat gate shrinks it to 8 to force the +policy). `pending` counts sent-but-not-delivered. It is incremented by the +**sender**, on any shard, and decremented by the **home thread** at delivery — +so it is touched only through `wo_mbox_reserve`/`wo_mbox_release` and their +`__atomic` builtins. The consequence is disclosed rather than hidden: the cap +can overshoot by at most the number of in-flight sends. Overflow is fail-fast — +the send raises a catchable `WO_T_ACTOR` (trap 13), which is what lets a room +drop a slow member instead of growing without bound. + +**`monitor` (id 89) — the death notice.** `wo_monitor` is one registration: +observer, the moved-in notice message, next. The list lives on the **watched** +actor and is owned by its home thread, so the death walk needs no lock — dying +is a home-thread event and the list is right there. The notice is the +observer's own M-typed message, so an observer receives death notices in the +same shape as everything else. Monitoring an already-dead actor fires +immediately rather than silently doing nothing. An observer whose mailbox is +full loses the notice, with a disclosed stderr line — the alternative was +blocking a death walk on a slow observer. + +It takes **three arguments** (`watched, observer, msg`), not the two the spec +first proposed, because the caller may be `main`, which has no mailbox and so +cannot be an implicit observer. + +**`time.after` (id 90) — one-shot, no cancel.** `wo_timer` is `at` (wall ms), +target, message, next. The list lives on the **arming fiber's shard** and is +scanned by the same deadline machinery that already serves fd-park deadlines, +so timers cost no new wait mechanism. Firing is an ordinary runtime send, which +means it inherits the ordinary rules: a full target drops with a stderr line, a +dead target drops silently. There is no cancel; the idiom is a generation +counter in the message, which the `timer-generation` corpus fixture pins. + +**Where to look when a lifecycle thing misbehaves:** `wo_vm_actor_monitor` and +`wo_vm_timer_after` in `vm.c` are the two entry points; `shard_main` and +`NEXT_RUNNABLE()` decide when a shard runs, adopts, or stops. The corpus +fixtures `monitor-death`, `timer-delivery` and `timer-generation` are the +smallest working examples of each. + +## The shutdown drain guarantee (iteration 40) + +**A message sent before the stop flag is observed is delivered and run before +the engine stops.** Stated because it was once untrue in a way nothing caught. + +`wo_engine_stop` sets `eng_shutdown`, wakes every worker, joins them, and only +then tears down — freeing whatever envelopes are still queued. So a worker that +leaves its loop early takes its inbox with it. `NEXT_RUNNABLE()` has always +encoded the right behaviour for a worker holding a live fiber: on a stop it +returns 2 and keeps draining, because "only the PRIMARY's stop ends the +program". `shard_main`'s **idle** branch did the opposite — it reaped and broke +— so a shard whose actors happened to be between messages at `SIGTERM` +abandoned everything still in flight. + +It now honours the same contract: while the primary's window is open an idle +worker adopts its inbox and runs what arrives, `sched_yield`ing on an empty +poll so a drain cannot burn a core per shard and starve the actors it exists to +let run. Only `eng_shutdown` — which the primary sets after `main` returns — +ends it. + +Two things follow that are easy to get wrong. The window is the **primary's**, +so a program that wants a longer drain holds it open itself; `main` cannot park +after the stop flag, because a park there unwinds. And the whole path is +unreachable at `WO_SHARDS=1`, where `wo_engine_stop` returns at `nshards <= 1`. + +## Digests: sha1, sha256, hmac_sha256 (iteration 34, ids 85–87) + +`crypto.c` holds SHA-1 and SHA-256 over a single buffer and HMAC-SHA-256 on top +of the latter, each returning a fresh `Bytes`. No streaming API and no other +primitives — these exist because WebSocket's handshake needs SHA-1 and ETags +need SHA-256, and that is the whole of the demand so far. + +Correctness is pinned to the published vectors rather than to itself: +RFC 3174 for SHA-1, the FIPS/RFC 6234 vectors for SHA-256, RFC 4231 for HMAC, +in `runtime/test/test_crypto.c` (18 checks). **There is still no RNG anywhere +in the runtime** — HMAC authenticates a token but cannot mint one, which is why +iteration 39 leads with a random-bytes builtin. diff --git a/runtime/src/builtin.c b/runtime/src/builtin.c index 095367e..ff0861f 100644 --- a/runtime/src/builtin.c +++ b/runtime/src/builtin.c @@ -200,6 +200,18 @@ int wo_builtin(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { } case WO_B_CALL: /* iteration 24: park/reply protocol lives in vm.c */ return wo_vm_actor_call(vm, R, ins, msg); + case WO_B_MONITOR: { + int rc = wo_vm_actor_monitor(vm, R[B], R[B + 1], R[B + 2], msg); + if (rc) return rc; + R[A] = 0; + return 0; + } + case WO_B_TIME_AFTER: { + int rc = wo_vm_timer_after(vm, (int64_t)R[B], R[B + 1], R[B + 2], msg); + if (rc) return rc; + R[A] = 0; + return 0; + } case WO_B_NOW: { /* wall-clock milliseconds */ struct timespec ts; clock_gettime(CLOCK_REALTIME, &ts); diff --git a/runtime/src/loader.c b/runtime/src/loader.c index 84cdcf7..5b7dbd1 100644 --- a/runtime/src/loader.c +++ b/runtime/src/loader.c @@ -183,6 +183,17 @@ int wo_load_buf(wo_module *m, const uint8_t *buf, size_t len, char *err, * nowhere to live. woc refuses this at compile time; the loader * refuses it again because what the loader accepts, the interpreter * trusts — this combination must never reach the engine. */ + /* databasev2 2: `resident: keys` PARSES and sets this bit, but the + * storage half (tasks 5c/5d) is not implemented — rows are still fully + * resident. Accepting it would be an annotation the compiler honours + * in name only: a developer could declare a 120 GB table keys-resident, + * see it compile, and be OOM-killed. Refuse until the storage lands. */ + if (flags & WO_CLASSF_RESIDENT_KEYS) + BAIL("class %u declares `resident: keys`, which is NOT IMPLEMENTED " + "yet — rows are still fully resident, so the annotation would " + "be honoured in name only. Remove it until databasev2 2 tasks " + "5c/5d land; `resident: all` is what actually runs", + (unsigned)i); if ((flags & WO_CLASSF_VOLATILE) && (flags & WO_CLASSF_RESIDENT_KEYS)) BAIL("class %u: durable:false with resident:keys — rows would have " "nowhere to be read from", (unsigned)i); diff --git a/runtime/src/main.c b/runtime/src/main.c index c6eac0b..81d40da 100644 --- a/runtime/src/main.c +++ b/runtime/src/main.c @@ -4,6 +4,7 @@ * 2 = usage or load failure (loader's message on stderr) * Heap cap defaults to 64 MiB, overridable via WO_HEAP_MB. */ #include +#include #include #include #include @@ -124,6 +125,22 @@ static void gc_pump(wo_vm *vm) { } } +/* databasev2 4: one diagnostic line about group commit, opt-in via + * WO_WAL_STATS. Off by default because it would otherwise pollute the output + * of every durable program; a gate that wants the numbers asks for them. */ +static void wal_stats_report(const wo_wal *w) { + if (!w || !getenv("WO_WAL_STATS")) return; + fprintf(stderr, + "walstats batches=%llu records=%llu peak_batch=%llu peak_staged=%llu " + "compactions=%llu compact_us_max=%llu compact_us_total=%llu compacted_bytes=%llu\n", + (unsigned long long)w->stat_batches, (unsigned long long)w->stat_records, + (unsigned long long)w->stat_peak_batch, (unsigned long long)w->stat_peak_staged, + (unsigned long long)w->stat_compactions, + (unsigned long long)w->stat_compact_us_max, + (unsigned long long)w->stat_compact_us_total, + (unsigned long long)w->compacted_bytes); +} + int main(int argc, char **argv) { wo_module mod; char err[256]; @@ -178,6 +195,10 @@ int main(int argc, char **argv) { return 2; } wo_tls_set(&VM); + /* iteration 24: a write to a peer-closed socket must be EPIPE (a + * catchable WO_T_IO), never a process-killing SIGPIPE — every + * serving program writes to sockets whose peers vanish. */ + signal(SIGPIPE, SIG_IGN); /* The database engine boots with the VM: every class IS a table. * Durability is opt-in — WO_DATA= opens /shard-0.wal, * replays it before the entry runs (boot-before-listeners doctrine), @@ -254,6 +275,25 @@ int main(int argc, char **argv) { if (v >= 1 && v <= 0x7FFFFFFFul) wo_mailbox_cap = (uint32_t)v; } } + /* databasev2 3: the checkpoint policy. WO_CHECKPOINT_BYTES is the floor + * below which a log is too small to bother compacting; WO_CHECKPOINT_RATIO + * is how many times the live set's own size counts as too much history. + * Both exist mainly so the policy is TESTABLE — a gate sets a tiny floor + * and forces compaction in a few writes rather than waiting for megabytes. + * There is no time-based trigger, by design: our records are durable at + * commit, so an idle log does not grow. */ + { + const char *cb = getenv("WO_CHECKPOINT_BYTES"); + if (cb && cb[0]) { + unsigned long long v = strtoull(cb, NULL, 10); + if (v > 0) wo_wal_ckpt_floor = (uint64_t)v; + } + const char *cr = getenv("WO_CHECKPOINT_RATIO"); + if (cr && cr[0]) { + unsigned long v = strtoul(cr, NULL, 10); + if (v <= 0xFFFFFFFFul) wo_wal_ckpt_ratio = (uint32_t)v; + } + } /* the arc's stage 2: all cores by default (the brave landing), one * pinned worker vm per extra core; WO_SHARDS caps or forces it */ { @@ -269,7 +309,7 @@ int main(int argc, char **argv) { if (wo_engine_start(&mod, heap_mb << 20, nshards) != 0) { fprintf(stderr, "wovm: cannot start %u shards\n", nshards); wo_engine_stop(); - if (VM.rt.wal) wo_wal_close(&WAL); + if (VM.rt.wal) { wal_stats_report(&WAL); wo_wal_close(&WAL); } wo_db_destroy(&DB); wo_vm_destroy(&VM); wo_module_free(&mod); @@ -328,7 +368,7 @@ int main(int argc, char **argv) { * unwind. */ if (argv_val) wo_drop_kind(&VM.rt, WO_K_MULTI, argv_val); wo_engine_stop(); /* join + destroy the worker shards before the primary */ - if (VM.rt.wal) wo_wal_close(&WAL); + if (VM.rt.wal) { wal_stats_report(&WAL); wo_wal_close(&WAL); } wo_db_destroy(&DB); gc_pump(&VM); wo_vm_destroy(&VM); diff --git a/runtime/src/park.c b/runtime/src/park.c index 6fbe11a..af4345a 100644 --- a/runtime/src/park.c +++ b/runtime/src/park.c @@ -296,6 +296,8 @@ static void tick_arm_uring(wo_vm *vm, int64_t now) { for (wo_fiber *fb = vm->parked; fb; fb = fb->pnext) if (fb->state == WO_FIB_PARKED && fb->park_fd >= 0 && fb->park_deadline > 0) if (next == 0 || fb->park_deadline < next) next = fb->park_deadline; + int64_t tn = wo_vm_timers_next(vm); /* iteration 24 T5: armed timers */ + if (tn > 0 && (next == 0 || tn < next)) next = tn; if (next == 0) return; if (vm->tick_armed && vm->tick_at <= next) return; int64_t rel = next - now; @@ -323,7 +325,27 @@ static void efd_drain(wo_vm *vm) { int wo_io_wait(wo_vm *vm) { for (;;) { - if (wo_sys_stop_pending()) return WO_IO_STOP; + if (wo_sys_stop_pending()) { + /* iteration 24 (the drain): a STOP does not kill parked fibers + * from the outside — it WAKES them all, and each blocking + * builtin resolves per its own stop contract (deadline'd waits + * answer their timeout result, sleeps return early, plain + * waits answer WO_SYS_STOPPED and that fiber unwinds). The + * program's own code then drains and returns. Nothing parked + * = nothing to resolve: the old immediate-stop answer. */ + int woke = 0; + wo_fiber *fb = vm->parked; + while (fb) { + wo_fiber *nx = fb->pnext; + if (fb->state == WO_FIB_PARKED) { + wake(vm, fb); + woke = 1; + } + fb = nx; + } + if (woke) return 0; + return WO_IO_STOP; + } if (vm->io_kind == 0) { /* keep the wake eventfd armed (oneshot POLL_ADD, re-armed * after each firing) so inbox pushes interrupt the wait */ @@ -371,7 +393,9 @@ int wo_io_wait(wo_vm *vm) { head++; } __atomic_store_n(r.cq_head, head, __ATOMIC_RELEASE); - if (deadline_sweep_uring(vm, now_ms()) && woke != 2) woke = 1; + int64_t swnow = now_ms(); + if (wo_vm_timers_fire(vm, swnow) && woke != 2) woke = 1; + if (deadline_sweep_uring(vm, swnow) && woke != 2) woke = 1; if (woke == 2) return 1; /* adopt-needed */ if (woke) return 0; continue; @@ -389,6 +413,14 @@ int wo_io_wait(wo_vm *vm) { } int timeout = -1; int64_t now = now_ms(); + { + int64_t tn = wo_vm_timers_next(vm); /* iteration 24 T5 */ + if (tn > 0) { + int64_t rel = tn - now; + if (rel < 0) rel = 0; + timeout = (int)rel; + } + } for (wo_fiber *fb = vm->parked; fb; fb = fb->pnext) if (fb->park_fd == -1 || (fb->park_fd >= 0 && fb->park_deadline > 0)) { @@ -417,6 +449,7 @@ int wo_io_wait(wo_vm *vm) { } } now = now_ms(); + if (wo_vm_timers_fire(vm, now)) woke = 1; wo_fiber *fb = vm->parked; while (fb) { wo_fiber *nx = fb->pnext; diff --git a/runtime/src/sysio.c b/runtime/src/sysio.c index 5379963..7d0ca08 100644 --- a/runtime/src/sysio.c +++ b/runtime/src/sysio.c @@ -395,6 +395,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { if (stop_pending()) return WO_SYS_STOPPED; } if (fd < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) { + if (stop_pending()) return WO_SYS_STOPPED; /* arc T4: park until the listener is readable, then retry */ vm->cur->park_fd = (int)R[B]; vm->cur->park_deadline = 0; @@ -431,6 +432,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { /* arc T4: nothing readable yet — free the buffer (the retry * re-allocates) and park until the fd is readable */ wo_str_free(rt, s); + if (stop_pending()) return WO_SYS_STOPPED; vm->cur->park_fd = (int)R[B]; vm->cur->park_deadline = 0; vm->cur->park_events = POLLIN; @@ -476,6 +478,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { continue; } if (errno == EAGAIN || errno == EWOULDBLOCK) { + if (stop_pending()) return WO_SYS_STOPPED; vm->cur->park_wr_at = at; vm->cur->park_fd = (int)R[B]; vm->cur->park_deadline = 0; @@ -534,7 +537,9 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { } if (n < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) { wo_str_free(rt, s); - if (fb->dl_at > 0 && dnow >= fb->dl_at) { + if (stop_pending() || (fb->dl_at > 0 && dnow >= fb->dl_at)) { + /* iteration 24: a STOP resolves the wait as its timeout + * result — the program's own drain code decides what next */ fb->dl_active = 0; R[A] = 0; /* ?Text nil: the deadline expired */ return 0; @@ -584,9 +589,9 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { } } if (fd < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) { - if (fb->dl_at > 0 && dnow >= fb->dl_at) { + if (stop_pending() || (fb->dl_at > 0 && dnow >= fb->dl_at)) { fb->dl_active = 0; - R[A] = WO_NIL_SCALAR; /* ?Int nil: nothing arrived */ + R[A] = WO_NIL_SCALAR; /* ?Int nil: nothing arrived (or stop) */ return 0; } fb->park_fd = (int)R[B]; @@ -631,7 +636,7 @@ int wo_builtin_sys(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { continue; } if (errno == EAGAIN || errno == EWOULDBLOCK) { - if (fb->dl_at > 0 && dnow >= fb->dl_at) { + if (stop_pending() || (fb->dl_at > 0 && dnow >= fb->dl_at)) { fb->dl_active = 0; R[A] = 0; /* false: torn mid-write — close the fd */ return 0; diff --git a/runtime/src/vm.c b/runtime/src/vm.c index 1ca9271..4eda1e1 100644 --- a/runtime/src/vm.c +++ b/runtime/src/vm.c @@ -16,6 +16,7 @@ #include "db.h" /* arc stage 3: the transparent DB RPC (wo_db_req) */ #include "table.h" /* slot encode/decode for the RPC marshaling */ +#include "wal.h" /* databasev2 4: the drain issues the barrier */ #include #include @@ -78,6 +79,9 @@ static int actor_push(wo_actor *a, wo_msg m); static void call_reply_to(wo_vm *vm, wo_fiber *caller, uint32_t caller_shard, uint64_t reply, int status); static void actor_drop_payload(wo_vm *vm, uint64_t payload); +static void monitors_fire(wo_vm *vm, wo_actor *a); +static void runtime_notify(wo_vm *vm, wo_actor *target, uint64_t msg_val, + const char *what); /* the owning thread drains its inbox: adopt actors, deliver sends, * execute home-routed frees. Returns how many envelopes were handled. */ @@ -88,6 +92,11 @@ static int wo_vm_adopt(wo_vm *vm) { ib->head = ib->tail = NULL; pthread_mutex_unlock(&ib->mu); int n = 0; + /* databasev2 4 (group commit): DB replies are HELD until one barrier has + * covered the whole drain. Locals, not per-shard state: nothing here needs + * to outlive the batch it describes. */ + wo_envelope *rhead = NULL, *rtail = NULL; + uint32_t staged = 0; while (e) { wo_envelope *nx = e->next; switch (e->kind) { @@ -133,6 +142,25 @@ static int wo_vm_adopt(wo_vm *vm) { } break; } + case 7: { /* iteration 24 T4: a cross-shard monitor registration — + WE are the watched actor's home. Dead already = the + notice fires now; else it joins the list. */ + wo_actor *ob = (wo_actor *)(uintptr_t)e->from_fiber; + if (e->actor->dead) { + runtime_notify(vm, ob, e->payload, "death notice"); + break; + } + wo_monitor *mn = calloc(1, sizeof *mn); + if (!mn) { + actor_drop_payload(vm, e->payload); + break; + } + mn->observer = ob; + mn->msg = e->payload; + mn->next = e->actor->monitors; + e->actor->monitors = mn; + break; + } case 6: /* iteration 24: a call reply landing on the caller's shard — fill the slot and wake the parked fiber; the re-executed builtin consumes it (status != 0 makes it trap). */ @@ -149,13 +177,35 @@ static int wo_vm_adopt(wo_vm *vm) { * the same request back as the reply. */ wo_db_req *q = (wo_db_req *)(uintptr_t)e->payload; assert(vm->is_primary && "DB requests route to shard 0 only"); + wo_wal *dw = (wo_wal *)vm->rt.wal; + size_t before = dw ? dw->len : 0; wo_db_exec_req(vm, q); q->done = 1; + /* did this statement actually stage a record? Asking the buffer + * beats guessing from the opcode, and the count is what the + * failure diagnostic reports. */ + if (dw && dw->len > before) staged++; wo_envelope *re = calloc(1, sizeof *re); if (re) { re->kind = 4; re->payload = e->payload; - inbox_push_to(q->from_shard, re); + re->next = NULL; + if (dw && dw->len > before) { + /* This statement STAGED a record, so its reply is HELD: + * pushing it now would unpark the requester before its + * record is durable, which is the ack contract this + * iteration exists to make literally true. FIFO, so the + * first waiter is released first. */ + if (rtail) rtail->next = re; else rhead = re; + rtail = re; + } else { + /* A READ (or any statement that staged nothing) has no + * durability to wait for. Holding it too was measurably + * wrong: it parked readers behind an fsync they had no + * stake in, and durable.sN.mixread p99 rose ~4x + * (1043 -> 4057us) until this branch existed. */ + inbox_push_to(q->from_shard, re); + } } /* OOM: the requester stays parked until stop — leak, not UB */ break; } @@ -170,6 +220,41 @@ static int wo_vm_adopt(wo_vm *vm) { n++; e = nx; } + /* databasev2 4: ONE barrier for everything this drain staged, then every + * held reply. Each requester therefore unparks having been acknowledged + * after the barrier that carried ITS record. Commit unconditionally when + * anything is staged — the inline path relies on finding the buffer empty + * (see db.c), so a drain must never leave a record behind. */ + if (staged) { + wo_wal *cw = (wo_wal *)vm->rt.wal; + if (cw) wo_wal_commit_fatal(cw, staged); + } + while (rhead) { + wo_envelope *rn = rhead->next; + wo_db_req *rq = (wo_db_req *)(uintptr_t)rhead->payload; + rhead->next = NULL; + inbox_push_to(rq->from_shard, rhead); + rhead = rn; + } + /* databasev2 3: the ONE point where compaction is safe — the barrier above + * just ran, so the staging buffer is empty. Anywhere else, a staged record + * would be written into a file about to be replaced. This is a correctness + * requirement, not a scheduling preference; wo_wal_compact also refuses a + * non-empty buffer as a backstop. + * + * Replies are released FIRST, deliberately: their records are already + * durable, and holding them across a stop-the-world rewrite would add the + * rewrite's full duration to their latency for no benefit. + * + * The result is ignored because a failed compaction is a missed + * optimisation, not a durability event — the original log is left intact + * and the process carries on. */ + if (staged) { + wo_wal *cw = (wo_wal *)vm->rt.wal; + if (cw && wo_wal_should_compact(cw->off, cw->compacted_bytes, + wo_wal_ckpt_floor, wo_wal_ckpt_ratio)) + (void)wo_wal_compact(cw, (wo_db *)vm->rt.db); + } return n; } @@ -425,6 +510,33 @@ static void *shard_main(void *arg) { } else { int rc = wo_io_wait(vm); /* parked fibers AND the wake eventfd */ if (rc == WO_IO_STOP) { + /* iteration 40 — THE DRAIN GUARANTEE. A message sent before + * the stop flag is observed must be delivered and run before + * the engine stops. + * + * NEXT_RUNNABLE() already states this contract for a worker + * holding a live fiber: it returns 2 and keeps draining "so + * queued shutdown messages (close frames!) still run". This + * branch — the IDLE worker, empty run queue, waiting on the + * plane — used to reap and break instead, abandoning whatever + * sat in its inbox for wo_engine_stop() to free wholesale. + * + * An actor between messages is exactly that idle case, which + * is why a WARM server hid the bug: warm shards had live + * fibers and took the correct path. Measured 2026-08-27 on a + * fresh server: 5 of 16 SIGTERM drains left a WebSocket + * client at EOF with no close frame and no diagnostic. + * + * The window belongs to the PRIMARY and closes when it sets + * eng_shutdown (after main returns), so honour it here and + * only exit when the primary says so. Yield on an empty poll: + * a tight loop would burn a core per shard and starve the very + * actors the drain exists to let run. */ + if (!eng_shutdown) { + (void)wo_vm_adopt(vm); + if (!vm->qhead) sched_yield(); + continue; + } fib_reap_all(vm); break; } @@ -445,6 +557,85 @@ int wo_engine_primary_inbox(int wake_efd) { return 0; } +/* iteration 24 teardown phase 1 (single-threaded, BEFORE eng_teardown): + * dismantle one vm's actor world with real drops — container backings are + * malloc'd, so wholesale arena death does NOT cover them (LSan, chat's + * registry map). Cross-shard payloads route home through wo_route_free + * (still live here); the routed kind-2 envelopes are settled by the + * caller's inbox passes. */ +static void vm_drop_actor_world(wo_vm *vm) { + wo_actor *a = vm->actors; + vm->actors = NULL; + while (a) { + wo_actor *nx = a->next_all; + if (a->instance) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)a->instance); + for (uint32_t i = 0; i < a->mlen; i++) { + uint64_t m = a->msgs[(a->mhead + i) % a->mcap].payload; + if (m) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)m); + } + wo_monitor *mo = a->monitors; + while (mo) { + wo_monitor *mnx = mo->next; + if (mo->msg) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)mo->msg); + free(mo); + mo = mnx; + } + free(a->msgs); + free(a); + a = nx; + } + wo_timer *tt = vm->timers; + vm->timers = NULL; + while (tt) { + wo_timer *tnx = tt->next; + if (tt->msg) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)tt->msg); + free(tt); + tt = tnx; + } +} + +/* Settle every inbox after phase 1: home-routed frees execute on their + * owner vm; payload-carrying strays drop (possibly routing again — the + * outer loop runs until everything is quiet). Node memory always freed. */ +static int eng_settle_inboxes(void) { + int moved = 0; + for (uint32_t i = 0; i < wo_eng.nshards && i < WO_ENG_MAX_SHARDS; i++) { + if (!INBOX_READY[i]) continue; + wo_vm *vm = &wo_eng.shards[i]; + wo_inbox *ib = &INBOX[i]; + wo_envelope *e = ib->head; + ib->head = ib->tail = NULL; + while (e) { + wo_envelope *nx = e->next; + switch (e->kind) { + case 2: /* WE are home: the direct drop is the settlement */ + wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)e->payload); + break; + case 0: + case 5: + case 7: /* in-flight payloads: drop (may route -> next pass) */ + if (e->payload) + wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)e->payload); + break; + case 1: /* an unadopted actor shell */ + if (e->actor) { + if (e->actor->instance) + wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)e->actor->instance); + free(e->actor->msgs); + free(e->actor); + } + break; + default: /* 3/4/6: scalar or engine-side payloads, node-only */ + break; + } + free(e); + moved++; + e = nx; + } + } + return moved; +} + int wo_engine_start(const wo_module *mod, size_t heap_cap, uint32_t nshards) { wo_eng.nshards = nshards; eng_heap_cap = heap_cap; @@ -489,6 +680,13 @@ void wo_engine_stop(void) { (void)n; } for (uint32_t i = 1; i < wo_eng.nshards; i++) pthread_join(ts[i - 1], NULL); + /* single-threaded from here: PHASE 1 — real drops while every arena + * and the routing fabric are still alive (malloc'd container backings + * inside actor state need them; iteration 24's registry map). Settle + * passes run until routed frees stop appearing. */ + for (uint32_t i = 0; i < wo_eng.nshards && i < WO_ENG_MAX_SHARDS; i++) + if (wo_eng.shards[i].rt.arena.base) vm_drop_actor_world(&wo_eng.shards[i]); + while (eng_settle_inboxes() > 0) {} /* single-threaded from here. Every arena dies wholesale, so routed * frees and queued payloads need no per-object drops — DISCARD the * envelopes (freeing the malloc'd nodes/actors) and let the arenas @@ -557,19 +755,42 @@ void wo_vm_destroy(wo_vm *vm) { free(fb); } /* actors first — dropping their state and queued messages needs the - * runtime alive */ + * runtime alive. BUT: once the engine is in teardown, arenas die + * WHOLESALE (the standing doctrine) — a moved-in message's home arena + * may belong to an ALREADY-destroyed shard, and even reading its + * header is a use-after-free (ASan, chat's drain). Structures are + * still freed; payload drops are skipped. */ + int drops_ok = !eng_teardown; wo_actor *a = vm->actors; while (a) { wo_actor *nx = a->next_all; - if (a->instance) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)a->instance); - for (uint32_t i = 0; i < a->mlen; i++) { + if (drops_ok && a->instance) + wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)a->instance); + for (uint32_t i = 0; drops_ok && i < a->mlen; i++) { uint64_t m = a->msgs[(a->mhead + i) % a->mcap].payload; if (m) wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)m); } + wo_monitor *mo = a->monitors; + while (mo) { /* undelivered notices are the runtime's to drop */ + wo_monitor *mnx = mo->next; + if (drops_ok && mo->msg) + wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)mo->msg); + free(mo); + mo = mnx; + } free(a->msgs); free(a); a = nx; } + wo_timer *tt = vm->timers; + vm->timers = NULL; + while (tt) { /* unfired timers likewise */ + wo_timer *tnx = tt->next; + if (drops_ok && tt->msg) + wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)tt->msg); + free(tt); + tt = tnx; + } vm->actors = NULL; wo_io_destroy(vm); wo_rt_destroy(&vm->rt); @@ -760,6 +981,7 @@ static void actor_die(wo_vm *vm, wo_actor *a, wo_fiber *delivery) { a->instance = 0; } a->active = NULL; + monitors_fire(vm, a); } /* Mailbox nonempty, no delivery fiber: start one on the next message. @@ -877,6 +1099,57 @@ int wo_vm_actor_send(wo_vm *vm, uint64_t addr, uint64_t msg_val, const char **ms return 0; } +/* iteration 24 T4/T5: a RUNTIME-sourced delivery (death notice, timer). + * No fiber to trap: a full or dead target drops the message with a + * stderr line (spec'd disclosure), never silently. Runs on any thread — + * cross-shard targets ride the ordinary kind-0 envelope. */ +static void runtime_notify(wo_vm *vm, wo_actor *target, uint64_t msg_val, + const char *what) { + if (!target || !msg_val) return; + if (target->dead) { + actor_drop_payload(vm, msg_val); + return; /* send-to-dead: silent by contract */ + } + if (wo_mbox_reserve(target) != 0) { + fprintf(stderr, "wovm: %s dropped — the observer's mailbox is full\n", what); + actor_drop_payload(vm, msg_val); + return; + } + if (target->home != vm->shard_id) { + wo_envelope *e = calloc(1, sizeof *e); + if (!e) { + wo_mbox_release(target); + actor_drop_payload(vm, msg_val); + return; + } + e->kind = 0; + e->actor = target; + e->payload = msg_val; + inbox_push_to(target->home, e); + return; + } + wo_msg m0 = { msg_val, NULL, 0 }; + if (actor_push(target, m0) != 0) { + wo_mbox_release(target); + actor_drop_payload(vm, msg_val); + return; + } + if (!target->active) (void)actor_activate(vm, target); +} + +/* iteration 24 T4: the death walk — every registered observer gets its + * chosen notice, then the list is gone (an actor dies once). */ +static void monitors_fire(wo_vm *vm, wo_actor *a) { + wo_monitor *m = a->monitors; + a->monitors = NULL; + while (m) { + wo_monitor *nx = m->next; + runtime_notify(vm, m->observer, m->msg, "death notice"); + free(m); + m = nx; + } +} + /* iteration 24: call — send that waits. First entry enqueues with the * caller attached and parks (WO_PARK_INBOX, the DB-RPC park); the resume * RE-EXECUTES this builtin and consumes the scalar reply. No hangs, ever: @@ -946,6 +1219,105 @@ int wo_vm_actor_call(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { return WO_SYS_PARKED; } +int wo_vm_actor_monitor(wo_vm *vm, uint64_t watched, uint64_t observer, + uint64_t msg_val, const char **msg) { + wo_actor *w = (wo_actor *)(uintptr_t)watched; + wo_actor *o = (wo_actor *)(uintptr_t)observer; + if (!w || !o) { + *msg = "monitor: nil actor address"; + return WO_T_BOUNDS; + } + if (!msg_val) { + *msg = "monitor: nil notice message"; + return WO_T_BOUNDS; + } + /* the registration belongs to the WATCHED actor's home thread */ + if (w->home != vm->shard_id) { + wo_envelope *e = calloc(1, sizeof *e); + if (!e) { + actor_drop_payload(vm, msg_val); + *msg = "out of memory"; + return WO_T_OOM; + } + e->kind = 7; + e->actor = w; + e->payload = msg_val; + e->from_fiber = (wo_fiber *)o; /* reused slot: the observer */ + inbox_push_to(w->home, e); + return 0; + } + if (w->dead) { /* monitoring the dead: the notice fires NOW */ + runtime_notify(vm, o, msg_val, "death notice"); + return 0; + } + wo_monitor *m = calloc(1, sizeof *m); + if (!m) { + actor_drop_payload(vm, msg_val); + *msg = "out of memory"; + return WO_T_OOM; + } + m->observer = o; + m->msg = msg_val; + m->next = w->monitors; + w->monitors = m; + return 0; +} + +int wo_vm_timer_after(wo_vm *vm, int64_t ms, uint64_t addr, uint64_t msg_val, + const char **msg) { + wo_actor *a = (wo_actor *)(uintptr_t)addr; + if (!a) { + *msg = "time.after: nil actor address"; + return WO_T_BOUNDS; + } + if (!msg_val) { + *msg = "time.after: nil message"; + return WO_T_BOUNDS; + } + if (ms <= 0) { /* no wait to arm: deliver now */ + runtime_notify(vm, a, msg_val, "timer message"); + return 0; + } + wo_timer *t = calloc(1, sizeof *t); + if (!t) { + actor_drop_payload(vm, msg_val); + *msg = "out of memory"; + return WO_T_OOM; + } + struct timespec now; + clock_gettime(CLOCK_REALTIME, &now); + t->at = (int64_t)now.tv_sec * 1000 + now.tv_nsec / 1000000 + ms; + t->target = a; + t->msg = msg_val; + t->next = vm->timers; + vm->timers = t; + return 0; +} + +int wo_vm_timers_fire(wo_vm *vm, int64_t now) { + int fired = 0; + wo_timer **pp = &vm->timers; + while (*pp) { + wo_timer *t = *pp; + if (t->at <= now) { + *pp = t->next; + runtime_notify(vm, t->target, t->msg, "timer message"); + free(t); + fired++; + } else { + pp = &t->next; + } + } + return fired; +} + +int64_t wo_vm_timers_next(wo_vm *vm) { + int64_t next = 0; + for (wo_timer *t = vm->timers; t; t = t->next) + if (next == 0 || t->at < next) next = t->at; + return next; +} + /* The drop-table entry governing instruction [pc]: the last one recorded * at or before it. NULL = nothing live there. */ static const wo_dropent *vm_dropent(const wo_methodrec *me, uint32_t pc) { @@ -1227,6 +1599,15 @@ static int vm_run(wo_vm *vm, uint64_t *ret, wo_err *err) { } \ int iorc_ = wo_io_wait(vm); \ if (iorc_ == WO_IO_STOP) { \ + /* iteration 24: a WORKER on stop keeps DRAINING — its \ + * serve loop spins adopting the inbox until the primary \ + * finishes the drain window and sets eng_shutdown, so \ + * queued shutdown messages (close frames!) still run. \ + * Only the PRIMARY's stop ends the program. */ \ + if (!vm->is_primary) { \ + vm->cur = &vm->f0; \ + return 2; \ + } \ fib_reap_all(vm); \ vm->cur = &vm->f0; \ return 1; \ @@ -1729,8 +2110,31 @@ dispatch: vm->cur->frames[vm->cur->depth - 1].pc = pc - 1; vm->cur->ncatch = 0; vm_unwind(vm, 0); - /* a stop ends the PROGRAM: every fiber — the stopped one, - * queued ones, main wherever it is — unwinds clean */ + /* iteration 24 (the drain): a STOPPED wait on a NON-main fiber + * unwinds that fiber ALONE — the rest of the program (main's + * drain code, actors flushing close frames) keeps running. + * Main's own STOPPED still ends the program, as ever. */ + if (vm->cur != &vm->f0) { + wo_fiber *dead = vm->cur; + if (dead->actor) { + wo_actor *da = dead->actor; + if (dead->cur_msg) { + wo_drop_obj(&vm->rt, (wo_hdr *)(uintptr_t)dead->cur_msg); + dead->cur_msg = 0; + } + call_reply_to(vm, dead->msg_caller, dead->msg_caller_shard, + 0, WO_T_ACTOR); + dead->msg_caller = NULL; + da->active = NULL; + } + vm->nfibers--; + fib_retire(vm, dead); + NEXT_RUNNABLE(); + RELOAD(); + NEXT(); + } + /* main: a stop ends the PROGRAM — every remaining fiber + * unwinds clean */ if (vm->cur != &vm->f0) { wo_fiber *dead = vm->cur; vm->cur = &vm->f0; diff --git a/runtime/src/vm.h b/runtime/src/vm.h index 78675cb..94ea82b 100644 --- a/runtime/src/vm.h +++ b/runtime/src/vm.h @@ -120,6 +120,27 @@ typedef struct wo_msg { * guarantee). Death (iteration 24): a receive trapping uncaught marks * the actor dead — sends to it drop silently, calls trap, queued * callers are error-unparked; the state and mailbox are released. */ +/* iteration 24 T4: one death-notice registration. The runtime owns the + * moved-in notice message until delivery (or drops it if the observer is + * unreachable). The list lives on the WATCHED actor, owned by its home + * thread. */ +typedef struct wo_monitor { + struct wo_actor *observer; + uint64_t msg; + struct wo_monitor *next; +} wo_monitor; + +/* iteration 24 T5: one armed one-shot timer — fires as an ordinary + * runtime send of the moved message when `at` passes. The list lives on + * the ARMING fiber's shard and is scanned by the same deadline machinery + * that serves fd-park deadlines. */ +typedef struct wo_timer { + int64_t at; /* wall ms */ + struct wo_actor *target; + uint64_t msg; + struct wo_timer *next; +} wo_timer; + typedef struct wo_actor { uint64_t instance; /* the moved-in state object (runtime-owned) */ uint32_t method; /* receive's method index (self + msg = 2 args) */ @@ -134,6 +155,7 @@ typedef struct wo_actor { * overshoot by at most the number of in-flight sends — disclosed. */ uint32_t pending; wo_fiber *active; /* the delivery fiber, NULL when idle */ + wo_monitor *monitors; /* iteration 24 T4: who wants the death notice */ struct wo_actor *next_all; /* the vm's all-actors list */ } wo_actor; @@ -176,6 +198,9 @@ typedef struct wo_vm { * freed memory is the UAF this prevents. Steady-state pool size = the * peak live fiber count; the pool dies with the vm. */ wo_fiber *fib_pool; + /* iteration 24 T5: this shard's armed timers (unsorted list — the + * deadline scan is already linear; a wheel is measured-later work) */ + wo_timer *timers; /* iteration 35, uring backend: the shard's ONE deadline tick — a * TIMEOUT op with a sentinel user_data armed for the nearest fd-park * deadline (fd parks keep exactly one POLL op each; expiry wakes them @@ -206,6 +231,20 @@ int wo_vm_actor_send(wo_vm *vm, uint64_t addr, uint64_t msg_val, const char **ms * caller attached and parks (WO_SYS_PARKED); the re-execution consumes the * scalar reply into R[A] (vm.c owns the protocol, builtin.c dispatches). */ int wo_vm_actor_call(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg); +/* iteration 24 T4: register a death notice — monitor(watched, observer, + * msg). The msg MOVES to the runtime; an already-dead watched actor + * delivers it immediately. */ +int wo_vm_actor_monitor(wo_vm *vm, uint64_t watched, uint64_t observer, + uint64_t msg_val, const char **msg); +/* iteration 24 T5: arm a one-shot timer on THIS shard — time.after(ms, + * addr, msg). ms <= 0 delivers now. */ +int wo_vm_timer_after(wo_vm *vm, int64_t ms, uint64_t addr, uint64_t msg_val, + const char **msg); +/* iteration 24 T5: fire every timer at or past `now` (park.c's deadline + * machinery calls this beside the fd-park sweep). Returns fired count. */ +int wo_vm_timers_fire(wo_vm *vm, int64_t now); +/* The nearest armed timer's deadline, 0 = none (park.c's tick/timeout). */ +int64_t wo_vm_timers_next(wo_vm *vm); /* ---- the shard engine (arc stage 2) ------------------------------------ * One pinned thread per shard, each a full wo_vm (own arena, GC, I/O @@ -231,7 +270,11 @@ typedef struct wo_envelope { * from_shard/from_fiber = the parked caller), * 6 = CALL_REPLY (payload = the SCALAR reply, from_fiber = * the caller to unpark; status 0 = ok, WO_T_ACTOR = - * the callee was/went dead — the caller traps) */ + * the callee was/went dead — the caller traps), + * 7 = MONITOR (iteration 24 T4: actor = the WATCHED one, + * from_fiber REUSED as the observer wo_actor*, payload = + * the moved notice — registered on the watched actor's + * home thread; already-dead delivers the notice now) */ struct wo_actor *actor; uint64_t payload; uint32_t from_shard; diff --git a/runtime/src/wob.h b/runtime/src/wob.h index 8b67938..888e557 100644 --- a/runtime/src/wob.h +++ b/runtime/src/wob.h @@ -478,8 +478,17 @@ enum { * return value arrives. R is a SCALAR (v1, * compiler-enforced WO-E226). Dead callee = * WO_T_ACTOR, immediately or mid-call. */ - /* ids 89 (monitor) and 90 (time.after) are RESERVED for the rest of - * the lifecycle slice — do not reuse. */ + WO_B_MONITOR = 89, /* (watched, observer, msg) -> (): the + * observer's own M-typed msg is delivered + * when watched dies (trap-death); already + * dead delivers NOW; msg MOVES. A full + * observer's notice is dropped with a + * stderr line (no fiber to trap). */ + WO_B_TIME_AFTER = 90, /* (ms, addr, msg) -> (): one-shot timer — + * msg (MOVED) arrives as an ordinary send + * after ms; no cancel (the generation- + * counter idiom is the documented answer); + * ms <= 0 delivers now. */ /* ---- iteration 35: net seams (sysio.c). Deadlines are per-CALL (no * hidden fd state); a timeout is an EXPECTED outcome, so it answers * nil/false, never a trap. ms <= 0 = no deadline (the old behavior, diff --git a/runtime/test/test_wal.c b/runtime/test/test_wal.c index 21c1479..9d34045 100644 --- a/runtime/test/test_wal.c +++ b/runtime/test/test_wal.c @@ -117,6 +117,397 @@ static void test_roundtrip_replay(void) { wo_rt_destroy(&rt); } +/* databasev2 4 part A, Task 1: a failed barrier must be DETECTED, and the + * caller must be able to tell WHICH operation failed — a pwrite failure and + * an fdatasync failure are different operational problems and the diagnostic + * has to name the right one. This proves detection only; the fatal exit that + * follows it cannot be exercised in-process. */ +static void test_commit_failure_detected(void) { + char path[128]; + snprintf(path, sizeof path, "%s/commitfail.wal", g_dir); + wo_rt rt; + T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0); + wo_db db; + T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0); + wo_wal w; + T_EQ(wo_wal_open(&w, path, 1 << 16), 0); + const char *msg = ""; + + /* the WAL remembers where it lives — the abort diagnostic is worthless + * without it */ + T_CHECK(w.path != NULL && strstr(w.path, "commitfail.wal") != NULL); + + wo_str *s = wo_str_new(&rt, "abc", 3); + uint64_t vals[2] = {7, (uint64_t)(uintptr_t)s}; + uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL); + T_CHECK(id != 0); + T_EQ(wo_wal_append_insert(&w, &db, 0, id), 0); + T_CHECK(w.len > 0); /* something really is staged */ + + /* an unusable descriptor: pwrite reports EBADF. -1 is used rather than + * closing the real fd so the close below cannot double-free it. */ + int real = w.fd; + w.fd = -1; + T_EQ(wo_wal_commit(&w), WO_WAL_ERR_WRITE); + T_CHECK(w.len > 0); /* a failed commit consumes nothing */ + w.fd = real; + + wo_wal_close(&w); + wo_db_destroy(&db); + wo_rt_destroy(&rt); +} + +/* databasev2 3 Task 1: compaction rewrites the log as one record per LIVE row. + * Asserts BOTH halves on purpose: "the file got shorter" is also true of a + * truncating bug, so the replay comparison is what actually proves it. */ +static void test_compact_shortens_and_replays_equal(void) { + char path[128]; + snprintf(path, sizeof path, "%s/compact.wal", g_dir); + wo_rt rt; + T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0); + wo_db db; + T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0); + wo_wal w; + T_EQ(wo_wal_open(&w, path, 1 << 16), 0); + const char *msg = ""; + + uint64_t ids[3]; + for (int i = 0; i < 3; i++) { + wo_str *s = wo_str_new(&rt, "abc", 3); + uint64_t vals[2] = {(uint64_t)(i * 10), (uint64_t)(uintptr_t)s}; + ids[i] = wo_row_insert(&db, 0, vals, &msg, NULL); + T_CHECK(ids[i] != 0); + T_EQ(wo_wal_append_insert(&w, &db, 0, ids[i]), 0); + T_EQ(wo_wal_commit(&w), 0); + } + /* age it: the SAME row updated repeatedly, so HISTORY grows while the live + * set does not — the exact case checkpoint exists for */ + for (int k = 0; k < 40; k++) { + int ek = 0; + T_EQ(wo_row_update_field(&db, 0, ids[0], 0, (uint64_t)(500 + k), &msg, &ek), 0); + T_EQ(wo_wal_append_update(&w, &db, 0, ids[0]), 0); + T_EQ(wo_wal_commit(&w), 0); + } + uint64_t before_bytes = 0; + int64_t before_recs = wo_wal_check(path, &before_bytes); + T_CHECK(before_recs == 43); /* 3 inserts + 40 updates, all history */ + + T_EQ(wo_wal_compact(&w, &db), 0); + + uint64_t after_bytes = 0; + int64_t after_recs = wo_wal_check(path, &after_bytes); + T_CHECK(after_recs == 3); /* one record per LIVE row */ + T_CHECK(after_bytes < before_bytes); /* and the file really shrank */ + + /* the WAL stays usable: the descriptor was reopened and the offset reset, + * so a further write must land AFTER the compacted records, not over them */ + wo_str *s4 = wo_str_new(&rt, "xyz", 3); + uint64_t v4[2] = {99, (uint64_t)(uintptr_t)s4}; + uint64_t id4 = wo_row_insert(&db, 0, v4, &msg, NULL); + T_CHECK(id4 != 0); + T_EQ(wo_wal_append_insert(&w, &db, 0, id4), 0); + T_EQ(wo_wal_commit(&w), 0); + T_CHECK(wo_wal_check(path, NULL) == 4); + wo_wal_close(&w); + + /* the proof: a FRESH store replayed from the compacted log must hold the + * same rows, the same ids, and the LAST value each row had */ + wo_db db2; + T_EQ(wo_db_init(&db2, CLASSES, 1, 0, 1), 0); + T_EQ(wo_wal_replay(path, &db2), 4); + uint64_t out[2]; + T_EQ(wo_row_read(&db2, &rt, 0, ids[0], out, &msg), 0); + T_CHECK(out[0] == 539); /* the 40th update won, not the original 0 */ + wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]); + T_EQ(wo_row_read(&db2, &rt, 0, ids[1], out, &msg), 0); + T_CHECK(out[0] == 10); + wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]); + T_EQ(wo_row_read(&db2, &rt, 0, ids[2], out, &msg), 0); + T_CHECK(out[0] == 20); + wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]); + T_EQ(wo_row_read(&db2, &rt, 0, id4, out, &msg), 0); + T_CHECK(out[0] == 99); + wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]); + + wo_db_destroy(&db2); + wo_db_destroy(&db); + wo_rt_destroy(&rt); +} + +/* databasev2 3 Task 2: a stale temp file is the one input that could be + * mistaken for data — a crash before the rename leaves one behind, full of + * well-formed records that are NOT yet authoritative. So the fixture uses + * plausible records (a byte copy of a real log), not garbage: garbage would be + * rejected by the CRC anyway and would prove nothing. */ +static void test_stale_compact_temp_is_removed(void) { + char path[128], tmp[160]; + snprintf(path, sizeof path, "%s/stale.wal", g_dir); + snprintf(tmp, sizeof tmp, "%s%s", path, WO_WAL_TMP_SUFFIX); + wo_rt rt; + T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0); + wo_db db; + T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0); + wo_wal w; + T_EQ(wo_wal_open(&w, path, 1 << 16), 0); + const char *msg = ""; + + /* two live rows in the REAL log */ + uint64_t ids[2]; + for (int i = 0; i < 2; i++) { + wo_str *s = wo_str_new(&rt, "abc", 3); + uint64_t vals[2] = {(uint64_t)(i + 1), (uint64_t)(uintptr_t)s}; + ids[i] = wo_row_insert(&db, 0, vals, &msg, NULL); + T_EQ(wo_wal_append_insert(&w, &db, 0, ids[i]), 0); + T_EQ(wo_wal_commit(&w), 0); + } + wo_wal_close(&w); + + /* forge a plausible stale temp: a byte copy of the real log */ + { + int src = open(path, O_RDONLY); + int dst = open(tmp, O_WRONLY | O_CREAT | O_TRUNC, 0644); + T_CHECK(src >= 0 && dst >= 0); + char buf[8192]; + ssize_t n; + while ((n = read(src, buf, sizeof buf)) > 0) T_CHECK(write(dst, buf, (size_t)n) == n); + close(src); + close(dst); + T_EQ(access(tmp, F_OK), 0); /* it really is there before we open */ + } + + wo_wal w2; + T_EQ(wo_wal_open(&w2, path, 1 << 16), 0); + T_CHECK(access(tmp, F_OK) != 0); /* gone, and never consulted */ + wo_wal_close(&w2); + + /* and the live log still says exactly what it said */ + wo_db db2; + T_EQ(wo_db_init(&db2, CLASSES, 1, 0, 1), 0); + T_EQ(wo_wal_replay(path, &db2), 2); + uint64_t out[2]; + T_EQ(wo_row_read(&db2, &rt, 0, ids[0], out, &msg), 0); + T_CHECK(out[0] == 1); + wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]); + T_EQ(wo_row_read(&db2, &rt, 0, ids[1], out, &msg), 0); + T_CHECK(out[0] == 2); + wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]); + + wo_db_destroy(&db2); + wo_db_destroy(&db); + wo_rt_destroy(&rt); +} + +/* databasev2 3 Task 3: the trigger, tested as a pure decision. Kept pure + * precisely so it CAN be tested — a policy only observable by writing megabytes + * and waiting is a policy nobody checks. */ +static void test_should_compact_policy(void) { + /* below the floor, nothing fires however bad the ratio looks */ + T_EQ(wo_wal_should_compact(1000, 10, 4096, 3), 0); + T_EQ(wo_wal_should_compact(4095, 1, 4096, 3), 0); + /* past the floor with no prior compaction: run once to learn the size */ + T_EQ(wo_wal_should_compact(4096, 0, 4096, 3), 1); + /* with a known denominator it is a straight ratio test */ + T_EQ(wo_wal_should_compact(30000, 10000, 4096, 3), 0); /* exactly 3x is not MORE than 3x */ + T_EQ(wo_wal_should_compact(30001, 10000, 4096, 3), 1); + T_EQ(wo_wal_should_compact(19999, 10000, 4096, 2), 0); + T_EQ(wo_wal_should_compact(20001, 10000, 4096, 2), 1); + /* a zero ratio disables the policy rather than dividing by nothing */ + T_EQ(wo_wal_should_compact(1u << 30, 10, 4096, 0), 0); +} + +/* databasev2 3 Task 3: the ordering rule, asserted rather than trusted. + * Compaction with records staged would write them into a file about to be + * replaced, so it must be REFUSED — and refused without touching the log. */ +static void test_compact_refuses_with_staged_records(void) { + char path[128]; + snprintf(path, sizeof path, "%s/staged.wal", g_dir); + wo_rt rt; + T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0); + wo_db db; + T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0); + wo_wal w; + T_EQ(wo_wal_open(&w, path, 1 << 16), 0); + const char *msg = ""; + + wo_str *s1 = wo_str_new(&rt, "abc", 3); + uint64_t v1[2] = {7, (uint64_t)(uintptr_t)s1}; + uint64_t id1 = wo_row_insert(&db, 0, v1, &msg, NULL); + T_EQ(wo_wal_append_insert(&w, &db, 0, id1), 0); + T_EQ(wo_wal_commit(&w), 0); /* durable, buffer empty */ + + /* now stage WITHOUT committing */ + wo_str *s2 = wo_str_new(&rt, "xyz", 3); + uint64_t v2[2] = {8, (uint64_t)(uintptr_t)s2}; + uint64_t id2 = wo_row_insert(&db, 0, v2, &msg, NULL); + T_EQ(wo_wal_append_insert(&w, &db, 0, id2), 0); + T_CHECK(w.len > 0); + + uint64_t before = 0; + int64_t recs = wo_wal_check(path, &before); + T_EQ(wo_wal_compact(&w, &db), -1); /* refused */ + T_CHECK(w.len > 0); /* and the staged record is still there */ + uint64_t after = 0; + T_CHECK(wo_wal_check(path, &after) == recs && after == before); /* log untouched */ + + /* the staged record still commits normally afterwards */ + T_EQ(wo_wal_commit(&w), 0); + T_CHECK(wo_wal_check(path, NULL) == recs + 1); + + wo_wal_close(&w); + wo_db_destroy(&db); + wo_rt_destroy(&rt); +} + +/* databasev2 3 Task 4: kill -9 DURING compaction. + * + * The existing battery is insert-only, so its "records >= acks" oracle is + * exactly what compaction is allowed to break: collapsing history is the point. + * The invariant that survives is the ACKED LIVE SET — every id acked as + * inserted and not later acked as deleted must be present with its acked value, + * and every id acked as deleted must be absent. Both the pre-compaction and the + * post-compaction log satisfy that identically, which is precisely the + * "never a mixture" property the design is shaped around. + * + * The child deletes as it goes so HISTORY accumulates while the live set stays + * small — without that, compaction would have nothing to collapse and the test + * would prove nothing. */ +#define CK_DELETED UINT64_MAX + +static void ck_ack(int fd, uint64_t id, uint64_t val) { + uint64_t rec[2] = {id, val}; + if (write(fd, rec, sizeof rec) != (ssize_t)sizeof rec) _exit(0); /* parent gone */ +} + +static void compact_battery_child(const char *path, int ack_fd) { + wo_rt rt; + wo_db db; + wo_wal w; + if (wo_rt_init(&rt, 1 << 20, CLASSES, 1) != 0) _exit(9); + if (wo_db_init(&db, CLASSES, 1, 0, 1) != 0) _exit(9); + if (wo_wal_open(&w, path, 1 << 20) != 0) _exit(9); + const char *msg = ""; + uint64_t live[512]; + size_t nlive = 0; + for (uint64_t i = 0;; i++) { + uint64_t val = i * 7 + 3; + wo_str *s = wo_str_new(&rt, "r", 1); + uint64_t vals[2] = {val, (uint64_t)(uintptr_t)s}; + uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL); + wo_str_free(&rt, s); + if (!id) _exit(9); + if (wo_wal_append_insert(&w, &db, 0, id) != 0) _exit(9); + if (wo_wal_commit(&w) != 0) _exit(9); /* durable BEFORE the ack */ + ck_ack(ack_fd, id, val); + if (nlive < 512) live[nlive++] = id; + + /* drop the oldest so history grows while the live set does not */ + if (nlive > 16) { + uint64_t victim = live[0]; + memmove(live, live + 1, (nlive - 1) * sizeof live[0]); + nlive--; + /* INTENT FIRST, deliberately. An ack after the commit would race: + * a kill between them leaves the row legitimately gone on disk + * while the last ack still says "inserted", and the parent would + * demand a row the engine was right to remove. Announcing intent + * makes the row's fate simply UNKNOWN to the parent, which is the + * honest thing to assert about it. */ + ck_ack(ack_fd, victim, CK_DELETED); + if (wo_row_remove(&db, 0, victim) != 0) _exit(9); + if (wo_wal_append_remove(&w, 0, victim) != 0) _exit(9); + if (wo_wal_commit(&w) != 0) _exit(9); + } + /* compact often, so a kill has a real chance of landing inside one */ + if (i % 24 == 23) (void)wo_wal_compact(&w, &db); + } +} + +static void test_compact_crash_battery(void) { + int rounds = 40; /* it is a RACE: one green run proves very little */ + for (int round = 0; round < rounds; round++) { + char path[128], tmp[160]; + snprintf(path, sizeof path, "%s/ckcrash-%d.wal", g_dir, round); + snprintf(tmp, sizeof tmp, "%s%s", path, WO_WAL_TMP_SUFFIX); + int pipefd[2]; + T_EQ(pipe(pipefd), 0); + pid_t pid = fork(); + T_CHECK(pid >= 0); + if (pid == 0) { + close(pipefd[0]); + compact_battery_child(path, pipefd[1]); + _exit(0); + } + close(pipefd[1]); + /* vary the instant so kills land before, inside and after rewrites */ + struct timespec ts = {0, (7 + round * 3) * 1000000L}; + while (nanosleep(&ts, &ts) != 0) {} + kill(pid, SIGKILL); + int status; + waitpid(pid, &status, 0); + + /* replay the acks into the expected live set, in order */ + uint64_t ids[65536], vals[65536]; + size_t n = 0; + for (;;) { + uint64_t rec[2]; + ssize_t r = read(pipefd[0], rec, sizeof rec); + if (r != (ssize_t)sizeof rec) break; + if (n < 65536) { ids[n] = rec[0]; vals[n] = rec[1]; n++; } + } + close(pipefd[0]); + T_CHECK(n > 0); /* the child got at least one commit out */ + + wo_rt rt; + T_EQ(wo_rt_init(&rt, 1 << 22, CLASSES, 1), 0); + wo_db db; + T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0); + int64_t ck_recs = wo_wal_check(path, NULL); + int64_t ck_applied = wo_wal_replay(path, &db); + T_CHECK(ck_applied >= 0); /* never reported as corruption */ + + /* A stale temp may well EXIST after a kill inside compaction — that is + * the expected debris. The guarantee is that the next OPEN removes it + * and never reads it, so that is what gets asserted here; checking + * merely for its absence after a replay would be asserting something + * the design never promised (wo_wal_replay does not open the WAL). */ + { + wo_wal probe; + T_EQ(wo_wal_open(&probe, path, 1 << 20), 0); + T_CHECK(access(tmp, F_OK) != 0); + wo_wal_close(&probe); + } + + const char *msg = ""; + int bad = 0, checked = 0; + for (size_t k = 0; k < n && !bad; k++) { + if (vals[k] == CK_DELETED) continue; /* intent: fate is unknown */ + /* an id ever announced for deletion may legally be gone */ + int doomed = 0; + for (size_t j = 0; j < n; j++) + if (ids[j] == ids[k] && vals[j] == CK_DELETED) { doomed = 1; break; } + if (doomed) continue; + uint64_t out[2]; + int rc = wo_row_read(&db, &rt, 0, ids[k], out, &msg); + if (0) { + } else if (rc != 0 || out[0] != vals[k]) { + bad = 1; /* an acked insert is missing or wrong */ + fprintf(stderr, "CKDIAG round=%d id=%llu rc=%d got=%llu want=%llu ack#%zu/%zu " + "log_records=%lld replay_applied=%lld\n", + round, (unsigned long long)ids[k], rc, + rc == 0 ? (unsigned long long)out[0] : 0ull, + (unsigned long long)vals[k], k, n, + (long long)ck_recs, (long long)ck_applied); + } else { + wo_str_free(&rt, (wo_str *)(uintptr_t)out[1]); + } + checked++; + } + T_CHECK(checked > 0); + T_CHECK(!bad); + wo_db_destroy(&db); + wo_rt_destroy(&rt); + } +} + static void test_torn_tail(void) { char path[128]; snprintf(path, sizeof path, "%s/torn.wal", g_dir); @@ -534,12 +925,18 @@ int main(void) { snprintf(g_dir, sizeof g_dir, "/tmp/wo-wal-test-XXXXXX"); if (!mkdtemp(g_dir)) return 1; test_roundtrip_replay(); + test_commit_failure_detected(); + test_compact_shortens_and_replays_equal(); + test_stale_compact_temp_is_removed(); + test_should_compact_policy(); + test_compact_refuses_with_staged_records(); test_torn_tail(); test_float_bytes_replay(); test_offset_capture(); test_offset_after_failed_commit(); test_read_row_at(); test_crash_battery(); + test_compact_crash_battery(); /* leave the dir for a failed run's forensics only */ if (!t_fail) { char cmd[128]; diff --git a/scripts/chat-accept.sh b/scripts/chat-accept.sh new file mode 100755 index 0000000..26dc008 --- /dev/null +++ b/scripts/chat-accept.sh @@ -0,0 +1,433 @@ +#!/usr/bin/env bash +# scripts/chat-accept.sh — iteration 24's gate. The chat sample serves +# WebSocket rooms through the framework ([deps], file:// remote); a raw +# RFC 6455 python client (stdlib only, INDEPENDENT accept-key check) +# proves: the handshake, broadcast + presence + isolation across rooms, +# the 1k-clients-one-hot-room soak (fds/RSS accounted), and the SIGTERM +# drain (close frames, exit 0) — functional legs on BOTH WO_IO backends +# plus an ASan run. +set -uo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +WOC="$ROOT/compiler/_build/default/bin/woc" +WOVM="$ROOT/runtime/wovm" +ASAN="$ROOT/runtime/build/wovm_asan" +PORT0="${CHAT_PORT:-18901}" +PORT="$PORT0" +SOAK_N="${CHAT_SOAK:-1000}" + +pass=0; fail=0 +ok() { echo "ok $1"; pass=$((pass + 1)); } +bad() { echo "FAIL $1 -- $2"; fail=$((fail + 1)); } + +if [[ ! -x "$WOC" || ! -x "$WOVM" ]]; then + echo "chat-accept: build woc and wovm first" >&2; exit 1 +fi +ulimit -n 8192 2>/dev/null || true + +W="$(mktemp -d "${TMPDIR:-/tmp}/chat-accept.XXXXXX")" +SRV="" +# The example's server log lives at a STABLE path so a developer can +# `tail -F /tmp/chat.log` while this runs. It used to go to the per-run temp +# dir, which cleanup() deletes on exit — so there was nothing left to read and +# nothing to follow live. Truncated once here, then APPENDED by every leg with +# a banner, so one file holds the whole run in order. +SRVLOG="/tmp/chat.log" +: > "$SRVLOG" +LEG=0 +LEGFROM=1 +echo "server log: $SRVLOG (tail -F \"$SRVLOG\" to follow)" + +cleanup() { + # kill EVERY server this run started, not merely the most recent $SRV: a leg + # that dies before clearing SRV used to orphan a listener, which then broke + # the next run on the same port. $W is unique per run, so matching on it + # cannot touch another run's processes. + [[ -n "$SRV" ]] && kill -9 "$SRV" 2>/dev/null + pkill -9 -f "$W/app/target/chat" 2>/dev/null + rm -rf "$W" +} +trap cleanup EXIT + +cp -r "$ROOT/docs/examples/porch" "$W/fw" +git -C "$W/fw" init -q && git -C "$W/fw" add -A +git -C "$W/fw" -c user.email=t@t -c user.name=t commit -qm v01 && git -C "$W/fw" tag v0.1.0 +cp -r "$ROOT/docs/examples/chat" "$W/app" +sed -i "s|https://github.com/shoneyj/porch|file://$W/fw|" "$W/app/wo.toml" +printf '[build]\nruntime = "%s"\n' "$WOVM" >> "$W/app/wo.toml" + +if "$WOC" "$W/app" >"$W/build.out" 2>&1 && [[ -x "$W/app/target/chat" ]]; then + ok "deps chain + build" +else + bad "build" "$(grep -m1 error "$W/build.out" || head -1 "$W/build.out")" + echo "chat-accept: 1 checks, 1 failures"; exit 1 +fi + +# the raw client, shared by every leg +CLIENT="$W/wsc.py" +cat > "$CLIENT" <<'PYEOF' +import socket, base64, hashlib, os, time +GUID = "258EAFA5-E914-47DA-95CA-C5AB0DC85B11" +BUF = {} +def connect(port, room, name, timeout=8, rcvbuf=None): + # rcvbuf: shrink THIS client's receive buffer so the server's socket fills + # quickly — how the WO_MAILBOX leg manufactures a genuinely slow member + # without sleeping. Must be set before connect() to take effect. + if rcvbuf is None: + s = socket.create_connection(("127.0.0.1", port), timeout=timeout) + else: + s = socket.socket() + s.setsockopt(socket.SOL_SOCKET, socket.SO_RCVBUF, rcvbuf) + s.settimeout(timeout) + s.connect(("127.0.0.1", port)) + key = base64.b64encode(os.urandom(16)).decode() + s.sendall((f"GET /ws?room={room}&name={name} HTTP/1.1\r\nhost: a\r\n" + f"upgrade: websocket\r\nconnection: Upgrade\r\n" + f"sec-websocket-key: {key}\r\nsec-websocket-version: 13\r\n\r\n").encode()) + d = b"" + while b"\r\n\r\n" not in d: d += s.recv(2000) + head, _, rest = d.partition(b"\r\n\r\n") + BUF[s] = rest # a frame may already ride the same segment + head = head.decode() + assert " 101 " in head.splitlines()[0], head.splitlines()[0] + want = base64.b64encode(hashlib.sha1((key + GUID).encode()).digest()).decode() + assert want in head, "accept-key mismatch (independent check)" + return s +def _take(s, n, timeout): + s.settimeout(timeout) + b = BUF.get(s, b"") + while len(b) < n: + c = s.recv(4096) + if not c: + BUF[s] = b + return None + b += c + BUF[s] = b[n:] + return b[:n] +def send(s, text): + p = text.encode(); mask = os.urandom(4) + if len(p) < 126: hdr = bytes([0x81, 0x80 | len(p)]) + else: hdr = bytes([0x81, 0x80 | 126, len(p) >> 8, len(p) & 255]) + s.sendall(hdr + mask + bytes(b ^ mask[i % 4] for i, b in enumerate(p))) +def recv(s, timeout=5): + h = _take(s, 2, timeout) + if h is None: return (-2, "") # EOF + b0, b1 = h[0], h[1] + ln = b1 & 0x7F + if ln == 126: + e = _take(s, 2, timeout); ln = (e[0] << 8) | e[1] + d = _take(s, ln, timeout) if ln else b"" + return (b0 & 0x0F), (d or b"").decode(errors="replace") +PYEOF + +serve() { # serve PORT [env...] — start + wait for THIS server's listener line + PORT="$1"; shift + LEG=$((LEG + 1)) + printf '\n===== leg %d — port %s — %s =====\n' "$LEG" "$PORT" "${*:-default env}" >>"$SRVLOG" + # readiness is searched only in THIS leg's slice: the log is appended, never + # truncated, so a 'listening' line from an earlier leg would lie + LEGFROM=$(( $(wc -l < "$SRVLOG") + 1 )) + "$@" "$W/app/target/chat" "$PORT" >>"$SRVLOG" 2>&1 & + SRV=$! + for _ in $(seq 1 80); do + tail -n "+$LEGFROM" "$SRVLOG" 2>/dev/null | grep -q listening && return 0 + sleep 0.1 + done + return 1 +} + +functional() { # $1 = leg name + timeout 30 python3 - "$PORT" <<'PYEOF' +import sys; sys.path.insert(0, sys.argv[0].rsplit("/",1)[0]) +port = int(sys.argv[1]) +import importlib.util, os +spec = importlib.util.spec_from_file_location("wsc", os.environ["WSC"]) +wsc = importlib.util.module_from_spec(spec); spec.loader.exec_module(wsc) +a = wsc.connect(port, "lobby", "alice") +assert wsc.recv(a) == (1, "* alice joined") +b = wsc.connect(port, "lobby", "bob") +assert wsc.recv(a) == (1, "* bob joined") +assert wsc.recv(b) == (1, "* bob joined") +c = wsc.connect(port, "other", "carol") +assert wsc.recv(c) == (1, "* carol joined") +wsc.send(a, "hello room") +assert wsc.recv(a) == (1, "alice: hello room") +assert wsc.recv(b) == (1, "alice: hello room") +import socket +try: + k, t = wsc.recv(c, timeout=0.8); assert False, f"leak into other room: {t}" +except socket.timeout: pass +b.close() +k, t = wsc.recv(a) +assert (k, t) == (1, "* bob left"), (k, t) +a.close(); c.close() +print("functional-ok") +PYEOF +} + +# ---- 2. functional on both backends ---- +export WSC="$CLIENT" +serve "$((PORT0 + 0))" env WO_IO=uring || bad "serve-uring" "no listener" +r="$(functional uring)"; [[ "$r" == *functional-ok* ]] \ + && ok "uring: handshake(key verified) + presence + broadcast + isolation + leave" \ + || bad "uring-functional" "$r" +kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV="" + +serve "$((PORT0 + 1))" env WO_IO=epoll || bad "serve-epoll" "no listener" +r="$(functional epoll)"; [[ "$r" == *functional-ok* ]] \ + && ok "epoll: the same matrix" || bad "epoll-functional" "$r" +kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV="" + +# ---- 3. the soak: N clients, ONE hot room ---- +serve "$((PORT0 + 2))" || bad "serve-soak" "no listener" +fds_before="$(ls /proc/$SRV/fd 2>/dev/null | wc -l)" +fds_prev=99999 +r="$(timeout 180 python3 - "$PORT" "$SOAK_N" <<'PYEOF' +import asyncio, sys, os, time, base64, hashlib +port, N = int(sys.argv[1]), int(sys.argv[2]) +GUID = "258EAFA5-E914-47DA-95CA-C5AB0DC85B11" +MARK = "the-hot-room-marker" +sem = asyncio.Semaphore(100) +async def client(i, results): + async with sem: + r, w = await asyncio.open_connection("127.0.0.1", port) + key = base64.b64encode(os.urandom(16)).decode() + w.write((f"GET /ws?room=hot&name=c{i} HTTP/1.1\r\nhost: a\r\n" + f"upgrade: websocket\r\nconnection: Upgrade\r\n" + f"sec-websocket-key: {key}\r\nsec-websocket-version: 13\r\n\r\n").encode()) + await w.drain() + d = b"" + while b"\r\n\r\n" not in d: d += await r.read(2000) + if i == 0: + # the sender: wait for the herd, then one marker line + await asyncio.sleep(0) + results["sender_ready"].set() + try: + buf = b"" + deadline = time.time() + 150 + while time.time() < deadline: + try: + c = await asyncio.wait_for(r.read(8192), timeout=5) + except asyncio.TimeoutError: + if results["sent"].is_set(): break + continue + if not c: break + buf += c + # scan frames for the marker (server frames are unmasked, small) + if MARK.encode() in buf: + results["got"] += 1 + return + finally: + w.close() +async def main(): + results = {"got": 0, "sender_ready": asyncio.Event(), "sent": asyncio.Event()} + conns = [] + # keep the sender's socket outside the tasks: join first + sr, sw = None, None + async def sender(): + nonlocal sr, sw + async with sem: + sr, sw = await asyncio.open_connection("127.0.0.1", port) + key = base64.b64encode(os.urandom(16)).decode() + sw.write((f"GET /ws?room=hot&name=sender HTTP/1.1\r\nhost: a\r\n" + f"upgrade: websocket\r\nconnection: Upgrade\r\n" + f"sec-websocket-key: {key}\r\nsec-websocket-version: 13\r\n\r\n").encode()) + await sw.drain() + d = b"" + while b"\r\n\r\n" not in d: d += await sr.read(2000) + await sender() + tasks = [asyncio.create_task(client(i, results)) for i in range(N)] + await asyncio.sleep(max(2.0, N / 250)) # let the herd join + drain presence + p = MARK.encode(); mask = os.urandom(4) + hdr = bytes([0x81, 0x80 | len(p)]) + sw.write(hdr + mask + bytes(b ^ mask[i % 4] for i, b in enumerate(p))) + await sw.drain() + results["sent"].set() + t0 = time.time() + await asyncio.gather(*tasks, return_exceptions=True) + el = int((time.time() - t0) * 1000) + sw.close() + print(f"{results['got']}|{N}|{el}") +asyncio.run(main()) +PYEOF +)" +got="${r%%|*}"; rest="${r#*|}"; n="${rest%%|*}"; el="${rest#*|}" +[[ "$got" == "$n" ]] \ + && ok "soak: the marker reached all $got/$n hot-room clients (${el}ms after send)" \ + || bad "soak" "$r" +# leave-broadcast storms take a moment to settle after 1k closes +for _ in $(seq 1 20); do + fds_w1="$(ls /proc/$SRV/fd 2>/dev/null | wc -l)" + [[ "$fds_w1" -le "$fds_prev" ]] && break + fds_prev="$fds_w1" + sleep 0.5 +done +# The fd check is for a per-CONNECTION leak, and a fixed tolerance cannot +# express that. Shards initialise LAZILY (runtime/src/vm.c: a worker's vm is +# not paid for until its first fiber arrives), so the first wave legitimately +# adds one io_uring + one eventfd PER SHARD, capped at nproc — on a 20-core +# box that is +18, which the old `fds_before + 8` read as a leak. Measured +# 2026-08-27: 26 -> 44 after 20 clients, then still 44 after 40 more. +# +# So assert the invariant itself: a SECOND wave must not raise the count. +# Core-count independent, and it catches a slow leak that any fixed +# tolerance would hide inside its own slack. +timeout 60 python3 - "$PORT" 20 <<'PYEOF' >/dev/null 2>&1 +import socket, base64, os, sys, time +port, n = int(sys.argv[1]), int(sys.argv[2]) +socks = [] +for i in range(n): + s = socket.create_connection(("127.0.0.1", port), timeout=8) + k = base64.b64encode(os.urandom(16)).decode() + s.sendall((f"GET /ws?room=fdwave&name=w{i} HTTP/1.1\r\nhost: a\r\n" + f"upgrade: websocket\r\nconnection: Upgrade\r\n" + f"sec-websocket-key: {k}\r\nsec-websocket-version: 13\r\n\r\n").encode()) + h = b"" + while b"\r\n\r\n" not in h: + h += s.recv(4096) + socks.append(s) +time.sleep(0.5) +for s in socks: + s.close() +PYEOF +fds_after="$fds_w1" +for _ in $(seq 1 20); do + fds_after="$(ls /proc/$SRV/fd 2>/dev/null | wc -l)" + [[ "$fds_after" -le "$fds_w1" ]] && break + sleep 0.5 +done +rss_kb="$(awk '/VmRSS/{print $2}' /proc/$SRV/status 2>/dev/null)" +[[ "$fds_after" -le "$fds_w1" ]] \ + && ok "no per-connection fd leak (start $fds_before, after $SOAK_N: $fds_w1, after 20 more: $fds_after)" \ + || bad "soak-fds" "second wave grew fds: $fds_w1 -> $fds_after (start $fds_before)" +[[ -n "$rss_kb" && "$rss_kb" -lt 819200 ]] \ + && ok "soak RSS bounded (${rss_kb}KB < 800MB)" || bad "soak-rss" "${rss_kb}KB" + +# ---- 4. drain: SIGTERM with clients connected -> close frames, exit 0 ---- +# Starts its OWN server. It used to inherit the soak leg's $SRV, which meant +# any leg inserted between them silently handed drain an empty pid: its python +# died on int(""), the leg reported a bare failure, AND the soak server was +# never killed — orphaning a listener that then broke the NEXT run's soak on +# the same port. No leg may depend on another leg's server. +serve "$((PORT0 + 6))" || bad "serve-drain" "no listener" +r="$(timeout 30 python3 - "$PORT" "$SRV" <<'PYEOF' +import sys, os, time, signal, socket +import importlib.util +spec = importlib.util.spec_from_file_location("wsc", os.environ["WSC"]) +wsc = importlib.util.module_from_spec(spec); spec.loader.exec_module(wsc) +port, srv = int(sys.argv[1]), int(sys.argv[2]) +a = wsc.connect(port, "lobby", "alice"); wsc.recv(a) +b = wsc.connect(port, "lobby", "bob"); wsc.recv(a); wsc.recv(b) +os.kill(srv, signal.SIGTERM) +def drained(s): + try: + while True: + k, _ = wsc.recv(s, timeout=5) + if k == 8: return "close-frame" + if k == -2: return "eof" + except socket.timeout: + return "stuck" + except (ConnectionResetError, BrokenPipeError): + return "reset" +print(drained(a) + "|" + drained(b)) +PYEOF +)" +[[ "$r" == "close-frame|close-frame" ]] \ + && ok "drain: both clients got the close frame" || bad "drain" "$r" +stopped=1 +for _ in $(seq 1 40); do kill -0 "$SRV" 2>/dev/null || { stopped=0; break; }; sleep 0.1; done +[[ $stopped -eq 0 ]] && ok "SIGTERM exits 0" || bad "stop" "still running" +SRV="" + +# ---- 4b. WO_SHARDS=1: the same matrix on one shard ---- +# The plan requires `just chat` green at default cores AND on a single shard: +# cross-shard placement is where the actor work can hide a bug, so the +# one-shard run is the control that says a failure is placement's fault. +serve "$((PORT0 + 4))" env WO_SHARDS=1 || bad "serve-shards1" "no listener" +r="$(functional shards1)"; [[ "$r" == *functional-ok* ]] \ + && ok "WO_SHARDS=1: the same matrix on a single shard" \ + || bad "shards1-functional" "$r" +kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV="" + +# ---- 4c. WO_MAILBOX=8: the drop-slow-member path FIRES and the room lives ---- +# The backpressure policy earning its keep. A member that stops reading makes +# its writer block on write_dl; with the mailbox capped at 8 the room's +# broadcast send traps (WO_T_ACTOR), and the room must CATCH that, drop the +# member, and keep serving everyone else. Asserting the room survives is the +# point — a room that dies with its slowest member is the bug this policy +# exists to prevent. +serve "$((PORT0 + 5))" env WO_MAILBOX=8 || bad "serve-mailbox" "no listener" +r="$(timeout 90 python3 - "$PORT" <<'PYEOF' +import importlib.util, os, socket, sys, time +spec = importlib.util.spec_from_file_location("wsc", os.environ["WSC"]) +wsc = importlib.util.module_from_spec(spec); spec.loader.exec_module(wsc) +port = int(sys.argv[1]) + +fast = wsc.connect(port, "bp", "fast") +wsc.recv(fast) # * fast joined +# the slow member: a tiny receive buffer so the server's socket fills fast, +# and it never reads a single frame +slow = wsc.connect(port, "bp", "slow", rcvbuf=2048) +wsc.recv(fast) # * slow joined + +# storm: big frames the slow member never drains +blob = "x" * 1024 +for i in range(400): + try: + wsc.send(fast, f"{i}-{blob}") + except OSError: + break +# drain what fast owes us so its own mailbox cannot be the thing that fills +deadline = time.time() + 20 +seen = 0 +while time.time() < deadline: + try: + k, t = wsc.recv(fast, timeout=0.5) + seen += 1 + except Exception: + break + +# the room must still be alive and serving the fast member +survivor = wsc.connect(port, "bp", "late") +ok_join = False +deadline = time.time() + 15 +while time.time() < deadline: + try: + k, t = wsc.recv(fast, timeout=1.0) + if "late joined" in t: + ok_join = True + break + except Exception: + break +print("mailbox-ok" if ok_join else f"mailbox-dead seen={seen}") +slow.close(); fast.close(); survivor.close() +PYEOF +)" +[[ "$r" == *mailbox-ok* ]] \ + && ok "WO_MAILBOX=8: slow member dropped, room survived and kept serving" \ + || bad "mailbox-backpressure" "$r" +kill -TERM "$SRV" 2>/dev/null; wait "$SRV" 2>/dev/null; SRV="" + +# ---- 5. the ASan leg: functional matrix, zero leaks ---- +if [[ -x "$ASAN" ]]; then + sed -i "s|runtime = \".*\"|runtime = \"$ASAN\"|" "$W/app/wo.toml" + rm -rf "$W/app/target" + "$WOC" "$W/app" >/dev/null 2>&1 + serve "$((PORT0 + 3))" || bad "serve-asan" "no listener" + r="$(functional asan)" + kill -TERM "$SRV" 2>/dev/null + for _ in $(seq 1 60); do kill -0 "$SRV" 2>/dev/null || break; sleep 0.1; done + SRV="" + if [[ "$r" == *functional-ok* ]] \ + && ! tail -n "+$LEGFROM" "$SRVLOG" | grep -q "AddressSanitizer\|LeakSanitizer"; then + ok "ASan run clean (functional + drain, zero leaks)" + else + bad "asan" "$(tail -n "+$LEGFROM" "$SRVLOG" | grep -m1 -E 'ERROR|SUMMARY' || echo "$r")" + fi +else + bad "asan" "runtime/build/wovm_asan missing — make -C runtime wovm-asan" +fi + +echo +printf 'chat-accept: %d checks, %d failures\n' "$((pass + fail))" "$fail" +[[ $fail -eq 0 ]] diff --git a/scripts/db-bench.py b/scripts/db-bench.py index 6c50b83..9fdd902 100755 --- a/scripts/db-bench.py +++ b/scripts/db-bench.py @@ -26,6 +26,24 @@ QUICK = "--quick" in sys.argv WRITE_BASELINE = "--write-baseline" in sys.argv N = 2000 if QUICK else 20000 +# databasev2 4: the write-concurrent leg. `mix` writes on one op in ten with +# C=4, so group commit had almost nothing to batch there (measured mean batch +# 1.01, peak 3) — a property of that workload, not of the mechanism. C is high +# on purpose: batching is a function of how many writes are in flight, and +# measured mean batch rose 1.13 -> 1.76 -> 5.35 at C = 4 -> 16 -> 64. +WMIX_N = 4000 if QUICK else 20000 +WMIX_C = 32 if QUICK else 64 +# databasev2 3: the checkpoint leg. Ages a store by UPDATING the same rows, so +# history grows while the live set does not — otherwise the leg measures insert +# throughput instead of compaction. +CKPT_SEED = 2000 if QUICK else 5000 +CKPT_OPS = 8000 if QUICK else 20000 +# The stop-the-world budget. 50ms is a stall a serving process can absorb +# without a client noticing a timeout; measured at ~13ms for a 2MB live set, +# so this leaves real headroom while still failing before a stall becomes +# user-visible. Compaction is O(live rows), so this budget is what eventually +# forces the incremental design the spec deliberately did not buy in advance. +CKPT_PAUSE_BUDGET_US = 50000 MSG_N = 20000 if QUICK else 200000 WAL_N = 800 if QUICK else 4000 CRASH_REPS = 1 if QUICK else 3 @@ -95,6 +113,55 @@ def parse_metrics(lines, into, prefix): if m: into[f"{prefix}.msgrate.msgs_sec"] = int(m.group(2)) +def wmix_leg(metrics, tag, env, data): + """Every op a durable write, WMIX_C at once — the leg that actually + exercises group commit. + + It reuses the store the `all` run just seeded (a fresh process replays it, + so `kmod` is there) and asks the runtime for its group-commit counters via + WO_WAL_STATS. The counters matter as much as the throughput: if batches are + always one the mechanism is inert and any throughput change came from + somewhere else, so a payoff would be attributed to the wrong cause.""" + e = dict(env); e["WO_WAL_STATS"] = "1" + rc, lines, _, _ = run(["wmix", str(WMIX_N), str(WMIX_C)], e, 1800) + if rc != 0: + bad(f"{tag}.wmix", f"rc={rc} tail={lines[-2:]}") + return + ops = p50 = p99 = None + batches = records = peak_batch = peak_staged = None + for l in lines: + f = l.split() + if f and f[0] == "wmix" and len(f) == 5: + ops, p50, p99 = int(f[2]), int(f[3]), int(f[4]) + elif f and f[0] == "walstats": + kv = dict(x.split("=", 1) for x in f[1:] if "=" in x) + batches = int(kv.get("batches", 0)); records = int(kv.get("records", 0)) + peak_batch = int(kv.get("peak_batch", 0)); peak_staged = int(kv.get("peak_staged", 0)) + if ops is None or batches is None: + bad(f"{tag}.wmix", "no report or no walstats line") + return + metrics[f"{tag}.wmix.ops_sec"] = ops + metrics[f"{tag}.wmix.p50us"] = p50 + metrics[f"{tag}.wmix.p99us"] = p99 + metrics[f"{tag}.wmix.peak_batch"] = peak_batch + metrics[f"{tag}.wmix.peak_staged"] = peak_staged + mean = round(records / batches, 2) if batches else 0 + metrics[f"{tag}.wmix.mean_batch"] = mean + ok(f"{tag}.wmix: {ops} ops/sec, p50 {p50}us p99 {p99}us; " + f"{records} records over {batches} barriers (mean {mean}, peak {peak_batch}), " + f"peak staged {peak_staged}B") + # The gate that matters. Only the MULTI-shard leg can batch: a worker's + # statements marshal to shard 0 and queue, while shard-0 statements run + # inline and commit one at a time by design (see db.c). + if tag.endswith(".sN"): + if mean > 1.0: + ok(f"{tag}.wmix batches form (mean {mean} > 1)") + else: + bad(f"{tag}.wmix-inert", + f"mean batch {mean} — group commit is not engaging, so a " + f"throughput change would not be attributable to it") + + def campaign(): metrics = {} ncores = os.cpu_count() or 1 @@ -123,6 +190,8 @@ def campaign(): bad(f"{tag}.mix.fds", f"grew {fdg}") else: ok(f"{tag}.mix.fds flat") + if flavor == "durable" and data: + wmix_leg(metrics, tag, env, data) if data: shutil.rmtree(data, ignore_errors=True) # msgrate once per shard count, RAM only (no store dependency) for shards in (1, ncores): @@ -235,6 +304,49 @@ def tolerance_for(key): if key.startswith("ceiling."): return 100 if key.startswith("randread."): return 100 if key.startswith("replay."): return 100 + # databasev2 4: batch SHAPE follows arrival timing, so gating it tightly + # would gate the scheduler — what must hold is that the mean exceeds one + # under contention, which wmix_leg asserts directly against the live run. + # wmix's throughput and latency are NOT waived: they are the payoff, and a + # blanket waiver here would have left the whole leg ungated. + if key.endswith((".wmix.mean_batch", ".wmix.peak_batch", ".wmix.peak_staged")): + return 100 + # databasev2 4: DURABLE multi-shard p99 is an fsync TAIL, and group commit + # made it both noisier and legitimately higher. Measured across three full + # runs of the same build, durable.sN.mixread.p99 was 1043 / 2318 / 4147 us + # and wmix.p99 8758 / 20000 — a 2-4x spread with the box near idle, because + # a barrier now blocks the owner shard LONGER (more records per fsync) even + # though it blocks LESS OFTEN. That is the trade group commit makes on a + # single-threaded owner, and part B (async submission) is what would undo + # it. Gating a 2-4x-variable tail at 50% gates the disk, not the engine, so + # the FLOOR is the real guard here — and it is not slack: mixread's floor + # (4172us) came within 25us of tripping on the worst run. + if key.startswith("durable.sN.") and key.endswith(".p99us"): + # Widened again 2026-08-29 with more evidence: mixread p99 was measured + # at 1043 / 2318 / 4147us and mixwrite at 1623 / 4446us across runs of + # the SAME build on a near-idle box — a 3-4x spread. 100% was still + # gating the disk. The FLOOR stays the real guard and is not slack: + # mixread's came within 25us of tripping on the worst run observed. + return 300 + # databasev2 3: the RECLAIM ratio is structural and gated tightly — it is + # the feature's whole claim. Boot time and the pause are wall-clock on a + # shared box and are not: waiving them all would have left the leg ungated, + # which is the mistake part A's task 4 made and had to undo. + if key in ("ckpt.boot_off_ms", "ckpt.boot_on_ms", "ckpt.pause_us_max", + "ckpt.compactions", "ckpt.bytes_off", "ckpt.bytes_on"): + return 400 + # compaction BANDWIDTH is the engine's own property, so it is gated for + # real — it is what regressed 8x when the dump was fsyncing per flush + if key == "ckpt.pause_us_per_mb": + return 100 + # msgrate is actor-to-actor throughput and is scheduling-bound, so its + # run-to-run spread is far wider than its old 15%. MEASURED across the 10 + # full runs recorded on 2026-08-28/29 — several of them predating the + # checkpoint work — it ranged 10.7M to 17.9M msgs/sec, a 1.67x spread. A + # 15% gate on that gates the scheduler and fails intermittently whatever + # the engine does. Pre-existing; found while closing databasev2 3, not + # caused by it. + if ".msgrate." in key: return 70 if ".mixread." in key or ".mixwrite." in key: return 50 if ".sN." in key: return 50 if ".read." in key or ".query." in key: return 50 @@ -246,7 +358,11 @@ def write_baseline(metrics): "tolerances come from tolerance_for() in the driver"}} for k, v in sorted(metrics.items()): if k.endswith(("rss_growth_kb", "fd_growth")): continue - higher = k.endswith(("ops_sec", "msgs_sec")) + # reclaim_x: MORE reclaimed is better. Recorded as lower-is-better by + # the default detector, which would have passed "no reclaim at all" and + # failed an improvement — the feature's central claim, gated backwards. + higher = k.endswith(("ops_sec", "msgs_sec", "mean_batch", "peak_batch", + "reclaim_x")) floor_div = 8 if k.endswith("msgs_sec") else 4 # latency floors never sit below 100µs: at post-index µs scale a # 4×1µs "catastrophe line" is noise; the tripwire means "µs became @@ -520,9 +636,12 @@ def randread(metrics): def wal_used(data_dir): """Bytes actually written across the store's WAL files. - The non-zero prefix, NOT the file size: shard WALs are fallocate'd to - 1 MiB up front, so getsize reports 1048576 for an empty store and proves - nothing. Same reason scripts/residency-accept.sh measures it this way.""" + The non-zero prefix, NOT the file size: shard WALs are preallocated, so + getsize reports the preallocation (1 MiB) even for an empty store. Same + reason scripts/residency-accept.sh measures it this way. + + databasev2 1 and databasev2 3 each grew their own copy of this helper on + separate branches; this is the single one they now share.""" total = 0 for name in sorted(os.listdir(data_dir)): with open(os.path.join(data_dir, name), "rb") as f: @@ -621,6 +740,96 @@ def replay(metrics): metrics["replay.history_penalty_x"] = round(penalty, 2) ok(f"replay: identical dataset, {penalty:.2f}x the boot cost from history alone " f"({ins_ms:.0f} -> {his_ms:.0f} ms) -- what a checkpoint would collapse") +def checkpoint_leg(metrics): + """Space reclaimed, boot time, and the stop-the-world PAUSE. + + The same workload runs twice, differing only in whether checkpointing can + fire: an enormous floor disables it, a small one lets it. Comparing two runs + of one build is what isolates compaction from everything else the workload + does. + + Boot is measured with the sample's `boot` mode, which does nothing at all — + with WO_DATA set the runtime replays the whole log before main runs, so a + mode with no work of its own is the only honest way to price replay.""" + ncores = os.cpu_count() or 1 + out = {} + for name, knobs in (("off", {"WO_CHECKPOINT_BYTES": "1000000000"}), + ("on", {"WO_CHECKPOINT_BYTES": "65536", "WO_CHECKPOINT_RATIO": "2"})): + data = os.path.join(ROOT, "bench", f"tmp.{os.getpid()}.ckpt.{name}") + shutil.rmtree(data, ignore_errors=True); os.makedirs(data, exist_ok=True) + env = {"WO_DATA": data, "WO_SHARDS": str(ncores), "WO_WAL_STATS": "1"} + env.update(knobs) + rc, _, _, _ = run(["seed", str(CKPT_SEED)], env, 1800) + if rc != 0: + bad(f"ckpt.{name}.seed", f"rc={rc}"); shutil.rmtree(data, ignore_errors=True); return + rc, lines, _, _ = run(["wmix", str(CKPT_OPS), "16"], env, 1800) + if rc != 0: + bad(f"ckpt.{name}.age", f"rc={rc}"); shutil.rmtree(data, ignore_errors=True); return + stats = {} + for l in lines: + f = l.split() + if f and f[0] == "walstats": + stats = dict(x.split("=", 1) for x in f[1:] if "=" in x) + used = wal_used(data) + # NOT through run(): it samples RSS on a 250ms poll, so every timing it + # produces floors at the poll quantum — boot measured that way reported + # 251ms both with and without checkpointing, which is the harness's + # clock, not the engine's. Median of 3 because this is wall-clock. + benv = dict(os.environ) + benv.update({"WO_DATA": data, "WO_SHARDS": str(ncores)}) + samples = [] + brc = 0 + for _ in range(3): + t0 = time.monotonic() + pr = subprocess.run([BIN, "boot"], stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, env=benv, timeout=900) + samples.append((time.monotonic() - t0) * 1000.0) + brc = pr.returncode or brc + boot_ms = sorted(samples)[1] + if brc != 0: + bad(f"ckpt.{name}.boot", f"rc={brc}"); shutil.rmtree(data, ignore_errors=True); return + out[name] = (used, boot_ms, stats) + shutil.rmtree(data, ignore_errors=True) + + (off_b, off_boot, _), (on_b, on_boot, st) = out["off"], out["on"] + comps = int(st.get("compactions", 0)) + if comps == 0: + bad("ckpt.inert", "no compaction ran — the leg proves nothing about checkpointing") + return + metrics["ckpt.compactions"] = comps + metrics["ckpt.bytes_off"] = off_b + metrics["ckpt.bytes_on"] = on_b + metrics["ckpt.reclaim_x"] = round(off_b / max(on_b, 1), 2) + metrics["ckpt.boot_off_ms"] = int(round(off_boot)) + metrics["ckpt.boot_on_ms"] = int(round(on_boot)) + metrics["ckpt.pause_us_max"] = int(st.get("compact_us_max", 0)) + # The RAW pause scales with the live set, and this workload's live set is + # not fixed: wmix's hist_dump inserts a row per latency bucket, so a noisier + # box produces more buckets, more rows, and a longer pause. Gating the raw + # number against a baseline therefore gates the box. What belongs to the + # ENGINE is the rate, so that is what carries a real tolerance; the raw + # pause keeps the absolute budget assertion below as its guard. + cb = int(st.get("compacted_bytes", 0)) + if cb > 0 and metrics["ckpt.pause_us_max"] > 0: + metrics["ckpt.pause_us_per_mb"] = int(round( + metrics["ckpt.pause_us_max"] / (cb / (1024.0 * 1024.0)))) + ok(f"ckpt: {off_b} -> {on_b} bytes ({metrics['ckpt.reclaim_x']}x reclaimed) over " + f"{comps} compactions; boot {off_boot:.0f} -> {on_boot:.0f} ms; " + f"stop-the-world pause max {metrics['ckpt.pause_us_max']}us " + f"({metrics.get('ckpt.pause_us_per_mb', 0)}us/MB)") + # the space claim is the point of the feature, so it is asserted, not just recorded + if off_b <= on_b: + bad("ckpt.no-reclaim", f"checkpointing did not shrink the log ({off_b} -> {on_b})") + else: + ok(f"ckpt: the log is smaller with checkpointing on") + # THE BUDGET. Stated, not assumed — the spec refused to assume it. + if metrics["ckpt.pause_us_max"] > CKPT_PAUSE_BUDGET_US: + bad("ckpt.pause-budget", + f"stop-the-world pause {metrics['ckpt.pause_us_max']}us exceeds the stated " + f"{CKPT_PAUSE_BUDGET_US}us budget — alternatives (incremental copy, " + f"fork-and-dump) are bought against THIS number") + else: + ok(f"ckpt: pause within budget ({metrics['ckpt.pause_us_max']} <= {CKPT_PAUSE_BUDGET_US}us)") def main(): @@ -639,6 +848,7 @@ def main(): ceiling(metrics) randread(metrics) replay(metrics) + checkpoint_leg(metrics) os.makedirs(RESULTS_DIR, exist_ok=True) stamp = time.strftime("%Y%m%d-%H%M%S") out = os.path.join(RESULTS_DIR, f"run-{stamp}{'-quick' if QUICK else ''}.json") diff --git a/scripts/log-watcher-accept.sh b/scripts/log-watcher-accept.sh index 70b4361..c9c1488 100755 --- a/scripts/log-watcher-accept.sh +++ b/scripts/log-watcher-accept.sh @@ -51,6 +51,14 @@ if [[ ! -x "$WOVM" ]]; then fi WORK="$(mktemp -d "${TMPDIR:-/tmp}/lw-accept.XXXXXX")" +# stable, tailable log for the example app: the per-run work dir is deleted on +# exit, so a developer had nothing to follow. `tail -F /tmp/log-watcher.log`. +# Each invocation keeps its own $WORK/*.out (the checks grep those) and is +# ALSO teed here, banner-separated, so one file holds the whole run. +APPLOG="/tmp/log-watcher.log" +: > "$APPLOG" +echo "app log: $APPLOG (tail -F \"$APPLOG\" to follow)" + # LW_ACCEPT_KEEP=1 leaves the work directory (image, logs, cron.d, the # server's own stdout) in place — what you want the moment a check fails. cleanup() { @@ -89,7 +97,8 @@ fi # watcher to decide the burst is over. LOG="$WORK/app.log" : >"$LOG" -timeout -k 2 12 "$WOVM" "$IMAGE" watch "$LOG" 2 1 >"$WORK/watch.out" 2>&1 & +printf '\n===== watch =====\n' >>"$APPLOG" +timeout -k 2 12 "$WOVM" "$IMAGE" watch "$LOG" 2 1 > >(tee -a "$APPLOG" >"$WORK/watch.out") 2>&1 & WATCH_PID=$! sleep 2 printf 'info service starting\n' >>"$LOG" @@ -110,6 +119,7 @@ CRON="$WORK/cron.d" mkdir -p "$CRON" printf '* * * * * root /usr/bin/backup.sh > /var/log/backup.log 2>&1\n' >"$CRON/backup" timeout 8 "$WOVM" "$IMAGE" run "$CRON" >"$WORK/run.out" 2>&1 +{ printf '\n===== run =====\n'; cat "$WORK/run.out"; } >>"$APPLOG" if grep -q "^SCHEDULE /var/log/backup.log" "$WORK/run.out"; then ok "run (parsed and scheduled the cron entry)" else @@ -124,7 +134,8 @@ EOF # -k: `env.stopping()` installs a SIGTERM handler that only sets a flag, and # the serve loop is blocked in accept(), so a plain TERM is swallowed — the # process needs a KILL to actually stop (recorded in docs/00-status.md). -timeout -k 2 20 "$WOVM" "$IMAGE" mcp "$CRON" "$WORK/cfg.json" >"$WORK/mcp.out" 2>&1 & +printf '\n===== mcp =====\n' >>"$APPLOG" +timeout -k 2 20 "$WOVM" "$IMAGE" mcp "$CRON" "$WORK/cfg.json" > >(tee -a "$APPLOG" >"$WORK/mcp.out") 2>&1 & SRV_PID=$! sleep 2 @@ -256,7 +267,8 @@ if [[ -n "${LW_SOAK:-}" ]]; then soak_mode() { local name="$1" load_fn="$2" shift 2 - "$WOVM" "$IMAGE" "$@" >"$WORK/soak-$name.out" 2>&1 & + printf '\n===== soak %s =====\n' "$name" >>"$APPLOG" + "$WOVM" "$IMAGE" "$@" > >(tee -a "$APPLOG" >"$WORK/soak-$name.out") 2>&1 & local pid=$! rss0 fd0 rss1 fd1 drss dfd deadline i sleep 3 # first-touch pages and the first work cycle if ! kill -0 "$pid" 2>/dev/null; then diff --git a/scripts/site-accept.sh b/scripts/site-accept.sh index b3d61fe..00712cf 100755 --- a/scripts/site-accept.sh +++ b/scripts/site-accept.sh @@ -56,6 +56,11 @@ fi PORT=$((8500 + RANDOM % 400)) DATA="$W/data"; mkdir -p "$DATA" +# stable, tailable server log — the per-run temp dir is deleted on exit +SRVLOG="/tmp/site.log" +: > "$SRVLOG" +echo "server log: $SRVLOG (tail -F \"$SRVLOG\" to follow)" + hit() { # path [method] [data] [token] -> "STATUS|BODY" (redirects not followed) python3 - "$PORT" "$1" "${2:-GET}" "${3:-}" "${4:-}" <<'PYEOF' @@ -97,7 +102,8 @@ expect() { # name got want_status want_substr } serve() { - SITE_TOKEN=s3cr3t WO_DATA="$DATA" "$W/app/target/site" "$PORT" >>"$W/srv.out" 2>&1 & + printf '\n===== serve — port %s =====\n' "$PORT" >>"$SRVLOG" + SITE_TOKEN=s3cr3t WO_DATA="$DATA" "$W/app/target/site" "$PORT" >>"$SRVLOG" 2>&1 & SRV=$! for _ in $(seq 1 40); do [[ "$(hit /health 2>/dev/null)" == 200* ]] && return 0 diff --git a/scripts/web-app-accept.sh b/scripts/web-app-accept.sh index da6854b..a359c51 100755 --- a/scripts/web-app-accept.sh +++ b/scripts/web-app-accept.sh @@ -83,9 +83,19 @@ else fi DATA="$W/data"; mkdir -p "$DATA" -WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >"$W/srv.out" 2>&1 & +# stable, tailable server log: the per-run temp dir is deleted on exit, so a +# developer had nothing to follow. `tail -F /tmp/web-app.log` while this runs. +SRVLOG="/tmp/web-app.log" +: > "$SRVLOG" +echo "server log: $SRVLOG (tail -F \"$SRVLOG\" to follow)" +printf '===== boot — port %s =====\n' "$PORT" >>"$SRVLOG" +LEGFROM=$(( $(wc -l < "$SRVLOG") + 1 )) +WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >>"$SRVLOG" 2>&1 & SRV=$! -for _ in $(seq 1 40); do grep -q listening "$W/srv.out" 2>/dev/null && break; sleep 0.1; done +for _ in $(seq 1 40); do + tail -n "+$LEGFROM" "$SRVLOG" 2>/dev/null | grep -q listening && break + sleep 0.1 +done # one tiny HTTP client; python is already a repo test dependency hit() { # method path [body] [auth: yes|no] [content-type] -> "STATUS|BODY" @@ -467,7 +477,8 @@ for _ in $(seq 1 30); do kill -0 "$SRV" 2>/dev/null || { stopped=0; break; }; sl SRV="" # ---- 15. restart persistence (WAL replay) ---- -WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >>"$W/srv.out" 2>&1 & +printf '\n===== restart (WAL replay) — port %s =====\n' "$PORT" >>"$SRVLOG" +WA_TOKEN=s3cr3t WA_IDLE_MS=600 WO_DATA="$DATA" "$W/app/target/web-app" "$PORT" >>"$SRVLOG" 2>&1 & SRV=$! sleep 0.5 expect "product survives a restart (WAL)" "$(hit GET /products)" 200 '"name":"mug"' diff --git a/tests/corpus/run/monitor-death/fixture.out b/tests/corpus/run/monitor-death/fixture.out new file mode 100644 index 0000000..0c676c2 --- /dev/null +++ b/tests/corpus/run/monitor-death/fixture.out @@ -0,0 +1,3 @@ +died: boom +died: late +done diff --git a/tests/corpus/run/monitor-death/fixture.wo b/tests/corpus/run/monitor-death/fixture.wo new file mode 100644 index 0000000..025196b --- /dev/null +++ b/tests/corpus/run/monitor-death/fixture.wo @@ -0,0 +1,36 @@ +use time + +-- iteration 24 T4: actor death is OBSERVABLE. The observer names its own +-- notice message; the watched actor trapping uncaught (the runtime's +-- stderr line) delivers it. Monitoring an ALREADY dead actor fires +-- immediately. WO_SHARDS=1 (the runner) keeps the order deterministic. +class Note { + who: Text +} + +class Watch { + pad: Int + fn receive(msg: Note) { + print("died: ${msg.who}"); + } +} + +class Boom { + pad: Int + fn receive(msg: Note) { + let z = len(msg.who) - len(msg.who); + let q = 1 / z; + } +} + +fn main() -> Int { + let obs: actor Note = spawn Watch { pad: 0 }; + let b: actor Note = spawn Boom { pad: 0 }; + monitor(b, obs, Note { who: "boom" }); + send(b, Note { who: "x" }); + time.sleep(100); + monitor(b, obs, Note { who: "late" }); + time.sleep(100); + print("done"); + return 0; +} diff --git a/tests/corpus/run/timer-delivery/fixture.out b/tests/corpus/run/timer-delivery/fixture.out new file mode 100644 index 0000000..d88608c --- /dev/null +++ b/tests/corpus/run/timer-delivery/fixture.out @@ -0,0 +1,3 @@ +tick: now +tick: armed +done diff --git a/tests/corpus/run/timer-delivery/fixture.wo b/tests/corpus/run/timer-delivery/fixture.wo new file mode 100644 index 0000000..ead06e5 --- /dev/null +++ b/tests/corpus/run/timer-delivery/fixture.wo @@ -0,0 +1,23 @@ +use time + +-- iteration 24 T5: a timer is a MESSAGE. time.after arms a one-shot on +-- this shard; the target receives it like any send. ms <= 0 delivers now. +class Tick { + tag: Text +} + +class Sink { + pad: Int + fn receive(msg: Tick) { + print("tick: ${msg.tag}"); + } +} + +fn main() -> Int { + let a: actor Tick = spawn Sink { pad: 0 }; + time.after(30, a, Tick { tag: "armed" }); + time.after(0, a, Tick { tag: "now" }); + time.sleep(150); + print("done"); + return 0; +} diff --git a/tests/corpus/run/timer-generation/fixture.out b/tests/corpus/run/timer-generation/fixture.out new file mode 100644 index 0000000..5b60035 --- /dev/null +++ b/tests/corpus/run/timer-generation/fixture.out @@ -0,0 +1,3 @@ +stale gen 1 ignored +fired gen 2 +done diff --git a/tests/corpus/run/timer-generation/fixture.wo b/tests/corpus/run/timer-generation/fixture.wo new file mode 100644 index 0000000..0686e37 --- /dev/null +++ b/tests/corpus/run/timer-generation/fixture.wo @@ -0,0 +1,28 @@ +use time + +-- iteration 24 T5: the CANCEL idiom — no cancel builtin, a generation +-- counter instead. The actor bumps its generation; a stale timer's +-- message names the old one and is recognized and ignored on arrival. +class Timer { + gen: Int +} + +class Gate { + gen: Int + fn receive(msg: Timer) { + if msg.gen == self.gen { + print("fired gen ${msg.gen}"); + } else { + print("stale gen ${msg.gen} ignored"); + } + } +} + +fn main() -> Int { + let g: actor Timer = spawn Gate { gen: 2 }; + time.after(30, g, Timer { gen: 1 }); + time.after(60, g, Timer { gen: 2 }); + time.sleep(200); + print("done"); + return 0; +}