From 3a30b4ac3be3df21b02de20994cf3c74418b6815 Mon Sep 17 00:00:00 2001 From: "shoney.arickathil" Date: Fri, 21 Aug 2026 16:54:13 +0200 Subject: [PATCH] =?UTF-8?q?feat(db-bench):=20first=20baseline=20=E2=80=94?= =?UTF-8?q?=20the=20measurement=20contract=20(T5)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - bench/baseline.json: 74 metrics from the first full campaign; tolerances tuned by a two-run repeatability check (mix* 50%, read/query 35%, rest 15% — rationale in _config) - gate bites: --check mode; doctored copy fails, both real runs 74/0 - headline: durable seed 4.5k/s vs ram 297k/s (23's case); reads O(table) at ~1.5k/s; mixread 1280 vs 21 ops/s single-vs-multi (the arc's price); msgrate 13.4M vs 2.45M (mutex-inbox number) - arc delta recorded in story 8; findings in sample README Co-Authored-By: Claude Fable 5 --- bench/baseline.json | 453 ++++++++++++++++++ docs/examples/db-bench/README.md | 21 + .../done/08-shard-actor-runtime.md | 13 + docs/superpowers/plans/2026-08-21-db-bench.md | 25 +- scripts/db-bench.py | 8 + 5 files changed, 511 insertions(+), 9 deletions(-) create mode 100644 bench/baseline.json diff --git a/bench/baseline.json b/bench/baseline.json new file mode 100644 index 0000000..9273cc8 --- /dev/null +++ b/bench/baseline.json @@ -0,0 +1,453 @@ +{ + "_config": { + "N": 20000, + "crash_reps": 3, + "msg_n": 200000, + "note": "refresh only with a commit that says why; tolerances widened 2026-08-21 from the two-run repeatability check: mix* 50% (scheduling-dependent small counts), read/query 35% (machine jitter), everything else 15%", + "wal_n": 4000 + }, + "durable.s1.mixread.ops_sec": { + "dir": "higher", + "floor": 187, + "tolerance_pct": 50, + "value": 751 + }, + "durable.s1.mixread.p50us": { + "dir": "lower", + "floor": 11844, + "tolerance_pct": 50, + "value": 2961 + }, + "durable.s1.mixread.p99us": { + "dir": "lower", + "floor": 19044, + "tolerance_pct": 50, + "value": 4761 + }, + "durable.s1.mixwrite.ops_sec": { + "dir": "higher", + "floor": 20, + "tolerance_pct": 50, + "value": 83 + }, + "durable.s1.mixwrite.p50us": { + "dir": "lower", + "floor": 19408, + "tolerance_pct": 50, + "value": 4852 + }, + "durable.s1.mixwrite.p99us": { + "dir": "lower", + "floor": 80000, + "tolerance_pct": 50, + "value": 20000 + }, + "durable.s1.query.ops_sec": { + "dir": "higher", + "floor": 335, + "tolerance_pct": 35, + "value": 1341 + }, + "durable.s1.query.p50us": { + "dir": "lower", + "floor": 2996, + "tolerance_pct": 35, + "value": 749 + }, + "durable.s1.query.p99us": { + "dir": "lower", + "floor": 3204, + "tolerance_pct": 35, + "value": 801 + }, + "durable.s1.read.ops_sec": { + "dir": "higher", + "floor": 336, + "tolerance_pct": 35, + "value": 1346 + }, + "durable.s1.read.p50us": { + "dir": "lower", + "floor": 3032, + "tolerance_pct": 35, + "value": 758 + }, + "durable.s1.read.p99us": { + "dir": "lower", + "floor": 3324, + "tolerance_pct": 35, + "value": 831 + }, + "durable.s1.seed.ops_sec": { + "dir": "higher", + "floor": 1123, + "tolerance_pct": 15, + "value": 4492 + }, + "durable.s1.seed.p50us": { + "dir": "lower", + "floor": 840, + "tolerance_pct": 15, + "value": 210 + }, + "durable.s1.seed.p99us": { + "dir": "lower", + "floor": 2188, + "tolerance_pct": 15, + "value": 547 + }, + "durable.s1.write.ops_sec": { + "dir": "higher", + "floor": 307, + "tolerance_pct": 15, + "value": 1230 + }, + "durable.s1.write.p50us": { + "dir": "lower", + "floor": 3036, + "tolerance_pct": 15, + "value": 759 + }, + "durable.s1.write.p99us": { + "dir": "lower", + "floor": 7232, + "tolerance_pct": 15, + "value": 1808 + }, + "durable.sN.mixread.ops_sec": { + "dir": "higher", + "floor": 5, + "tolerance_pct": 50, + "value": 20 + }, + "durable.sN.mixread.p50us": { + "dir": "lower", + "floor": 80000, + "tolerance_pct": 50, + "value": 20000 + }, + "durable.sN.mixread.p99us": { + "dir": "lower", + "floor": 80000, + "tolerance_pct": 50, + "value": 20000 + }, + "durable.sN.mixwrite.ops_sec": { + "dir": "higher", + "floor": 0, + "tolerance_pct": 50, + "value": 2 + }, + "durable.sN.mixwrite.p50us": { + "dir": "lower", + "floor": 80000, + "tolerance_pct": 50, + "value": 20000 + }, + "durable.sN.mixwrite.p99us": { + "dir": "lower", + "floor": 80000, + "tolerance_pct": 50, + "value": 20000 + }, + "durable.sN.query.ops_sec": { + "dir": "higher", + "floor": 328, + "tolerance_pct": 35, + "value": 1312 + }, + "durable.sN.query.p50us": { + "dir": "lower", + "floor": 3028, + "tolerance_pct": 35, + "value": 757 + }, + "durable.sN.query.p99us": { + "dir": "lower", + "floor": 3392, + "tolerance_pct": 35, + "value": 848 + }, + "durable.sN.read.ops_sec": { + "dir": "higher", + "floor": 350, + "tolerance_pct": 35, + "value": 1403 + }, + "durable.sN.read.p50us": { + "dir": "lower", + "floor": 3000, + "tolerance_pct": 35, + "value": 750 + }, + "durable.sN.read.p99us": { + "dir": "lower", + "floor": 3420, + "tolerance_pct": 35, + "value": 855 + }, + "durable.sN.seed.ops_sec": { + "dir": "higher", + "floor": 1119, + "tolerance_pct": 15, + "value": 4478 + }, + "durable.sN.seed.p50us": { + "dir": "lower", + "floor": 836, + "tolerance_pct": 15, + "value": 209 + }, + "durable.sN.seed.p99us": { + "dir": "lower", + "floor": 2372, + "tolerance_pct": 15, + "value": 593 + }, + "durable.sN.write.ops_sec": { + "dir": "higher", + "floor": 316, + "tolerance_pct": 15, + "value": 1265 + }, + "durable.sN.write.p50us": { + "dir": "lower", + "floor": 3064, + "tolerance_pct": 15, + "value": 766 + }, + "durable.sN.write.p99us": { + "dir": "lower", + "floor": 6616, + "tolerance_pct": 15, + "value": 1654 + }, + "ram.s1.mixread.ops_sec": { + "dir": "higher", + "floor": 320, + "tolerance_pct": 50, + "value": 1280 + }, + "ram.s1.mixread.p50us": { + "dir": "lower", + "floor": 10904, + "tolerance_pct": 50, + "value": 2726 + }, + "ram.s1.mixread.p99us": { + "dir": "lower", + "floor": 12472, + "tolerance_pct": 50, + "value": 3118 + }, + "ram.s1.mixwrite.ops_sec": { + "dir": "higher", + "floor": 35, + "tolerance_pct": 50, + "value": 142 + }, + "ram.s1.mixwrite.p50us": { + "dir": "lower", + "floor": 11216, + "tolerance_pct": 50, + "value": 2804 + }, + "ram.s1.mixwrite.p99us": { + "dir": "lower", + "floor": 16368, + "tolerance_pct": 50, + "value": 4092 + }, + "ram.s1.msgrate.msgs_sec": { + "dir": "higher", + "floor": 3356155, + "tolerance_pct": 15, + "value": 13424620 + }, + "ram.s1.query.ops_sec": { + "dir": "higher", + "floor": 378, + "tolerance_pct": 35, + "value": 1512 + }, + "ram.s1.query.p50us": { + "dir": "lower", + "floor": 2536, + "tolerance_pct": 35, + "value": 634 + }, + "ram.s1.query.p99us": { + "dir": "lower", + "floor": 3196, + "tolerance_pct": 35, + "value": 799 + }, + "ram.s1.read.ops_sec": { + "dir": "higher", + "floor": 407, + "tolerance_pct": 35, + "value": 1630 + }, + "ram.s1.read.p50us": { + "dir": "lower", + "floor": 2396, + "tolerance_pct": 35, + "value": 599 + }, + "ram.s1.read.p99us": { + "dir": "lower", + "floor": 3124, + "tolerance_pct": 35, + "value": 781 + }, + "ram.s1.seed.ops_sec": { + "dir": "higher", + "floor": 64354, + "tolerance_pct": 15, + "value": 257416 + }, + "ram.s1.seed.p50us": { + "dir": "lower", + "floor": 16, + "tolerance_pct": 15, + "value": 4 + }, + "ram.s1.seed.p99us": { + "dir": "lower", + "floor": 32, + "tolerance_pct": 15, + "value": 8 + }, + "ram.s1.write.ops_sec": { + "dir": "higher", + "floor": 718, + "tolerance_pct": 15, + "value": 2872 + }, + "ram.s1.write.p50us": { + "dir": "lower", + "floor": 2320, + "tolerance_pct": 15, + "value": 580 + }, + "ram.s1.write.p99us": { + "dir": "lower", + "floor": 3444, + "tolerance_pct": 15, + "value": 861 + }, + "ram.sN.mixread.ops_sec": { + "dir": "higher", + "floor": 5, + "tolerance_pct": 50, + "value": 21 + }, + "ram.sN.mixread.p50us": { + "dir": "lower", + "floor": 80000, + "tolerance_pct": 50, + "value": 20000 + }, + "ram.sN.mixread.p99us": { + "dir": "lower", + "floor": 80000, + "tolerance_pct": 50, + "value": 20000 + }, + "ram.sN.mixwrite.ops_sec": { + "dir": "higher", + "floor": 0, + "tolerance_pct": 50, + "value": 2 + }, + "ram.sN.mixwrite.p50us": { + "dir": "lower", + "floor": 80000, + "tolerance_pct": 50, + "value": 20000 + }, + "ram.sN.mixwrite.p99us": { + "dir": "lower", + "floor": 80000, + "tolerance_pct": 50, + "value": 20000 + }, + "ram.sN.msgrate.msgs_sec": { + "dir": "higher", + "floor": 611486, + "tolerance_pct": 15, + "value": 2445944 + }, + "ram.sN.query.ops_sec": { + "dir": "higher", + "floor": 333, + "tolerance_pct": 35, + "value": 1333 + }, + "ram.sN.query.p50us": { + "dir": "lower", + "floor": 2968, + "tolerance_pct": 35, + "value": 742 + }, + "ram.sN.query.p99us": { + "dir": "lower", + "floor": 3392, + "tolerance_pct": 35, + "value": 848 + }, + "ram.sN.read.ops_sec": { + "dir": "higher", + "floor": 372, + "tolerance_pct": 35, + "value": 1488 + }, + "ram.sN.read.p50us": { + "dir": "lower", + "floor": 2504, + "tolerance_pct": 35, + "value": 626 + }, + "ram.sN.read.p99us": { + "dir": "lower", + "floor": 3228, + "tolerance_pct": 35, + "value": 807 + }, + "ram.sN.seed.ops_sec": { + "dir": "higher", + "floor": 74404, + "tolerance_pct": 15, + "value": 297619 + }, + "ram.sN.seed.p50us": { + "dir": "lower", + "floor": 12, + "tolerance_pct": 15, + "value": 3 + }, + "ram.sN.seed.p99us": { + "dir": "lower", + "floor": 28, + "tolerance_pct": 15, + "value": 7 + }, + "ram.sN.write.ops_sec": { + "dir": "higher", + "floor": 634, + "tolerance_pct": 15, + "value": 2537 + }, + "ram.sN.write.p50us": { + "dir": "lower", + "floor": 2408, + "tolerance_pct": 15, + "value": 602 + }, + "ram.sN.write.p99us": { + "dir": "lower", + "floor": 3552, + "tolerance_pct": 15, + "value": 888 + } +} \ No newline at end of file diff --git a/docs/examples/db-bench/README.md b/docs/examples/db-bench/README.md index 3b3dd5c..316c693 100644 --- a/docs/examples/db-bench/README.md +++ b/docs/examples/db-bench/README.md @@ -40,3 +40,24 @@ compiler classifies the elements OWNED while table refs are scalar ids. Query-built multis are runtime-typed and safe. Worked around here (single ref local, bucket-major seeding); the compiler fix is its own slice. + +## Reference-machine numbers (first campaign, 2026-08-21) + +`bench/baseline.json` is the contract; headline readings: + +- ram seed 257–298k inserts/s; **durable seed ≈4.5k/s** (fsync-per-commit + ≈220µs each — the gap iteration 23 exists to close). +- reads ≈1.5k/s at p50 ≈600µs on a 20k-row store: point lookups are + O(table) — the probe walks every slab; index selection never reaches + the lookup path. THE read-path finding. +- mixread 1,280 ops/s single-shard vs **21 ops/s** multi-shard: RPC + round-trip × O(table) probes × owner serialization — the arc's honest + price until reads index properly. +- msgrate 13.4M msgs/s same-heap vs 2.45M cross-shard (the mutex-inbox + number, stage-2 deviation 4). + +## The gate must bite (proven 2026-08-21) + +`scripts/db-bench.py --check ` evaluates a recorded run: +the real results pass 74/0; a doctored copy (one ops/sec halved) FAILS +on exactly that metric. Re-run the smoke after any gate change. diff --git a/docs/stories/language-runtime-database/done/08-shard-actor-runtime.md b/docs/stories/language-runtime-database/done/08-shard-actor-runtime.md index 438edeb..7334e55 100644 --- a/docs/stories/language-runtime-database/done/08-shard-actor-runtime.md +++ b/docs/stories/language-runtime-database/done/08-shard-actor-runtime.md @@ -141,6 +141,19 @@ chain: 1 ## Info +**The arc's measured delta** (iteration 22's first campaign, +2026-08-21, `bench/baseline.json` — N=20000, 4 mix actors, real-disk +WO_DATA; single-shard column = the local path, multi-shard = the +stage-3 RPC): + +| metric | WO_SHARDS=1 | default cores | reading | +| --- | --- | --- | --- | +| seed inserts/s (ram) | 257,416 | 297,619 | local either way (primary seeds); parity ✓ | +| seed inserts/s (durable) | 4,492 | 4,478 | fsync-per-commit ≈220µs dominates — the 57× ram gap is iteration 23's case | +| read ops/s (ram, primary) | 1,630 | 1,488 | point lookups are O(table): the probe walks every slab — the read-path finding | +| mixread ops/s (actors) | 1,280 | **21** | the RPC price × O(table) probes × owner serialization — the arc's honest cost until reads index properly | +| msgrate msgs/s | 13,424,620 | 2,445,944 | same-heap vs mutex-inbox: 5.5× — deviation 4's number; rings stay unearned until this is the bottleneck | + **Guarantee contract** (stage-3 refinement 2026-08-21; moved here from the slice's marker doc when it landed): diff --git a/docs/superpowers/plans/2026-08-21-db-bench.md b/docs/superpowers/plans/2026-08-21-db-bench.md index dd4c050..e55443b 100644 --- a/docs/superpowers/plans/2026-08-21-db-bench.md +++ b/docs/superpowers/plans/2026-08-21-db-bench.md @@ -183,15 +183,22 @@ there). Story: (the arc's recorded delta: single- vs multi-shard columns), `docs/examples/db-bench/README.md` (the reference-machine numbers). -- [ ] Full campaign on this machine; inspect the tail; commit the - baseline with a message that says it IS the first contract. -- [ ] Gate-bites smoke (spec acceptance): doctor a copy of the results - (halve one ops/sec) and run the gate against it — must FAIL; then the - real results — must PASS. Record the procedure in the README. -- [ ] Copy the arc delta + msgrate numbers into story 8's record and - the mutex-inbox note (arc plan deviation 4 references it). -- [ ] Verify: `just db-bench` exit 0 twice in a row (repeatability); - battery. Commit. +- [x] Full campaign run twice (20 checks/0 fail each incl. restart + proofs and 3x kill -9 batteries per shard count); baseline committed + as the first contract. DEVIATION: the two-run repeatability check + found ~25% jitter on read/query latencies and scheduling-dependent + spread on mix* — per-metric tolerances tuned (mix* 50%, read/query + 35%, rest 15%, rationale in the baseline's _config note); both runs + pass the tuned contract 74/0. +- [x] Gate-bites proven: `--check` on a doctored copy (one ops/sec + halved) FAILS on exactly that metric; both real runs PASS. Procedure + in the README (a `--check` gate-only mode was added for this). +- [x] Arc delta + msgrate recorded: story 8 carries the measured table + (durable seed 4.5k/s vs ram 297k/s = 23's case; mixread 1280 vs 21 + ops/s = the RPC x O(table)-probe price; msgrate 13.4M vs 2.45M = + deviation 4's mutex-inbox number). Headline findings in the sample + README: point lookups are O(table) — the read-path finding. +- [x] Verified: two full campaigns green; battery green. Commit. ## Task 6 — closeout diff --git a/scripts/db-bench.py b/scripts/db-bench.py index aa323a7..8438b6c 100755 --- a/scripts/db-bench.py +++ b/scripts/db-bench.py @@ -219,6 +219,14 @@ def write_baseline(metrics): ok(f"baseline written ({len(base) - 1} metrics)") def main(): + # --check : gate-only evaluation of a recorded run — the + # gate-bites smoke doctors a copy and this mode must FAIL on it + if "--check" in sys.argv: + f = sys.argv[sys.argv.index("--check") + 1] + gate(json.load(open(f))) + print() + print(f"db-bench --check: {len(passed) + len(failed)} checks, {len(failed)} failures") + sys.exit(1 if failed else 0) build() metrics = campaign() durability(metrics)