From f66aa680a953b003c9e9a9d887b1b6a3e401ab7a Mon Sep 17 00:00:00 2001 From: "shoney.arickathil" Date: Fri, 21 Aug 2026 16:18:26 +0200 Subject: [PATCH] =?UTF-8?q?feat(db-bench):=20sample=20=E2=80=94=20serial?= =?UTF-8?q?=20modes,=20histogram=20stats=20(T2)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - seed/read/query/write/wal/verify/verify-acked + all (one-process campaign: RAM store dies with the process) - per-op time.ticks, 1us-bucket histogram percentiles (reservoir deviation: no element-write/sort in language; better tail anyway) - Meta expectation rows ride the same WAL verify checks - finding: hand-built multi SEGVs on drop (elems classed OWNED, refs are scalar ids) — worked around, recorded - finding: reads ~1.6k/s p50 595us vs 287k/s inserts — probe walks all slabs; the number 22 exists to surface Co-Authored-By: Claude Fable 5 --- docs/examples/db-bench/README.md | 42 ++++ docs/examples/db-bench/main.wo | 359 +++++++++++++++++++++++++++++++ docs/examples/db-bench/types.wo | 24 +++ docs/examples/db-bench/wo.toml | 6 + 4 files changed, 431 insertions(+) create mode 100644 docs/examples/db-bench/README.md create mode 100644 docs/examples/db-bench/main.wo create mode 100644 docs/examples/db-bench/types.wo create mode 100644 docs/examples/db-bench/wo.toml diff --git a/docs/examples/db-bench/README.md b/docs/examples/db-bench/README.md new file mode 100644 index 0000000..3b3dd5c --- /dev/null +++ b/docs/examples/db-bench/README.md @@ -0,0 +1,42 @@ +# db-bench — iteration 22's load generator + +The measurement backbone (spec: +`docs/superpowers/specs/2026-08-21-db-bench-design.md`). Pure `.wo`; +every measured mode prints one machine-parsable line per operation +class: + + + +Timing is per-operation via `time.ticks` (CLOCK_MONOTONIC µs). +Percentiles come from a 1µs-bucket histogram clamped at 20000µs — exact +to the microsecond below the clamp; a p99 AT 20000 means "clamp or +worse". (A histogram, not the spec's reservoir: the language has no +container element-write or sort, and the histogram's tail fidelity is +strictly better. Recorded as a plan deviation.) + +## Modes + +| mode | what it prices | +| --- | --- | +| `all N` | the throughput campaign in ONE process: seed N, read N/2, query N/10, write N/2. Without `WO_DATA` the store is RAM and dies with the process, so the measured modes must share the seeding run. | +| `seed N` | timed inserts: one parent per 100 children (FK probe each insert, unique-index maintenance per parent), k non-unique (10 rows/key), deterministic v. Writes `Meta` expectation rows. | +| `read N` | indexed take-1 point lookups, LCG-spread keys. | +| `query N` | full equality probes on the k index (≈10 rows each), materialized and counted. | +| `write N` | alternating inserts (disjoint k range 2e6+) and updates through query results. Corrupts the checksum by design — durability legs run on a fresh store. | +| `wal N` | the crash battery's vehicle: insert-only (k range 1e6+), `acked ` printed AFTER each insert returns — the return IS the ack (RAM applied, WAL record staged, ONE commit done). | +| `verify` | store vs its own Meta rows: count, checksum, one unique probe. Exit 3 on mismatch. | +| `verify-acked M` | after kill -9 mid-`wal`: rows 1..M exist with the right v; rows beyond M allowed (acked after the last print flushed). Exit 3 on mismatch. | + +## Coordination idiom (this side of iteration 31) + +There is no request/response surface yet: concurrent modes drive +completion the db-actor way — actors write rows, main polls the store +until the expected count, then settles. Retired when 31 lands. + +## Standing finding (2026-08-21, first run) + +A hand-built `multi Bucket` of insert results SEGVs on drop: the +compiler classifies the elements OWNED while table refs are scalar ids. +Query-built multis are runtime-typed and safe. Worked around here +(single ref local, bucket-major seeding); the compiler fix is its own +slice. diff --git a/docs/examples/db-bench/main.wo b/docs/examples/db-bench/main.wo new file mode 100644 index 0000000..ab506fb --- /dev/null +++ b/docs/examples/db-bench/main.wo @@ -0,0 +1,359 @@ +use time + +-- db-bench — iteration 22's load generator. Every measured mode prints +-- one machine-parsable line per operation class: +-- +-- +-- +-- Timing is per-operation via time.ticks (CLOCK_MONOTONIC µs); +-- percentiles come from a 1µs-bucket histogram clamped at HIST_CLAMP — +-- exact to the microsecond below the clamp, and the clamp bucket keeps +-- the tail honest (a p99 AT the clamp means "clamp or worse"). +-- The wal mode prints a running `acked ` line after every insert +-- RETURNS (the return IS the ack): the crash battery kills this mode +-- mid-run and verify-acked proves every acknowledged row survived. + +-- ---- deterministic helpers ---- + +fn lcg(seed: Int) -> Int { + let x = seed * 1103515245 + 12345; + if x < 0 { + x = 0 - x; + } + return x; +} + +fn item_v(i: Int) -> Int { + return (i * 37) % 1000; +} + +-- ---- the histogram (percentiles without a sort) ---- + +fn hist_add(mut h: map, us: Int) { + let b = us; + if b < 0 { + b = 0; + } + if b > 20000 { + b = 20000; + } + if has(h, b) { + set(h, b, get(h, b) + 1); + } else { + set(h, b, 1); + } +} + +fn hist_pct(h: map, total: Int, pct: Int) -> Int { + let target = total * pct / 100; + if target < 1 { + target = 1; + } + let seen = 0; + let b = 0; + while b <= 20000 { + if has(h, b) { + seen = seen + get(h, b); + if seen >= target { + return b; + } + } + b = b + 1; + } + return 20000; +} + +fn report(op: Text, n: Int, total_us: Int, h: map) { + let us = total_us; + if us < 1 { + us = 1; + } + let rate = n * 1000000 / us; + print("${op} ${n} ${rate} ${hist_pct(h, n, 50)} ${hist_pct(h, n, 99)}"); +} + +-- ---- modes ---- + +-- seed N: N children, one parent per 100, k = i % (N/10) (10 rows per +-- key), v deterministic. Meta rows record the expectations verify reads. +fn seed(n: Int) -> Int { + let h: map = {}; + let kmod = n / 10; + if kmod < 1 { + kmod = 1; + } + -- bucket-major: one parent, then its 100 children, using a single ref + -- local. (A hand-built `multi Bucket` of insert results SEGVs on drop — + -- the compiler classifies the elements OWNED while table refs are + -- scalar ids; recorded as a standing finding, not this iteration's fix. + -- Query-built multis are runtime-typed and safe.) + let vsum = 0; + let t0 = time.ticks(); + let i = 1; + let b = 0; + while i <= n { + let bref = insert Bucket { tag: "b${b}" }; + b = b + 1; + let j = 0; + while j < 100 and i <= n { + let o0 = time.ticks(); + insert Item { k: i % kmod, v: item_v(i), bucket: bref }; + hist_add(h, time.ticks() - o0); + vsum = vsum + item_v(i); + i = i + 1; + j = j + 1; + } + } + let t1 = time.ticks(); + insert Meta { tag: "count", val: n }; + insert Meta { tag: "vsum", val: vsum }; + insert Meta { tag: "kmod", val: kmod }; + report("seed", n, t1 - t0, h); + return 0; +} + +fn meta_val(tag: Text) -> Int { + let ms = from m in Meta where m.tag == tag take 1 select m; + if len(ms) == 0 { + return -1; + } + return ms[0].val; +} + +-- read N: indexed take-1 point lookups (the point-read this surface +-- offers), keys spread by LCG over the seeded key range. +fn read_mode(n: Int) -> Int { + let kmod = meta_val("kmod"); + if kmod < 1 { + print_err("read: seed first"); + return 1; + } + let h: map = {}; + let sink = 0; + let s = 42; + let t0 = time.ticks(); + let i = 0; + while i < n { + s = lcg(s); + let key = s % kmod; + let o0 = time.ticks(); + let xs = from x in Item where x.k == key take 1 select x; + if len(xs) > 0 { + sink = sink + xs[0].v; + } + hist_add(h, time.ticks() - o0); + i = i + 1; + } + let t1 = time.ticks(); + report("read", n, t1 - t0, h); + if sink < 0 { + print("impossible ${sink}"); + } + return 0; +} + +-- query N: full equality probes on the k index (≈10 rows per key), +-- each materialized and counted. +fn query_mode(n: Int) -> Int { + let kmod = meta_val("kmod"); + if kmod < 1 { + print_err("query: seed first"); + return 1; + } + let h: map = {}; + let rows = 0; + let s = 7; + let t0 = time.ticks(); + let i = 0; + while i < n { + s = lcg(s); + let key = s % kmod; + let o0 = time.ticks(); + for x in from x in Item where x.k == key select x { + rows = rows + 1; + } + hist_add(h, time.ticks() - o0); + i = i + 1; + } + let t1 = time.ticks(); + report("query", n, t1 - t0, h); + print("query rows ${rows}"); + return 0; +} + +-- write N: alternating inserts (disjoint k range 2e6+) and updates +-- through a query result. Corrupts vsum by design — the durability legs +-- run on their own fresh store. +fn write_mode(n: Int) -> Int { + let kmod = meta_val("kmod"); + if kmod < 1 { + print_err("write: seed first"); + return 1; + } + let bs = from b in Bucket where b.tag == "b0" take 1 select b; + if len(bs) == 0 { + print_err("write: no buckets"); + return 1; + } + let h: map = {}; + let s = 99; + let t0 = time.ticks(); + let i = 0; + while i < n { + let o0 = time.ticks(); + if i % 2 == 0 { + insert Item { k: 2000000 + i, v: item_v(i), bucket: bs[0] }; + } else { + s = lcg(s); + let key = s % kmod; + let xs = from x in Item where x.k == key take 1 select x; + if len(xs) > 0 { + xs[0].v = xs[0].v + 1; + } + } + hist_add(h, time.ticks() - o0); + i = i + 1; + } + let t1 = time.ticks(); + report("write", n, t1 - t0, h); + return 0; +} + +-- wal N: the crash battery's vehicle — insert-only, disjoint k range +-- (1e6+), `acked ` printed AFTER each insert returns (the return is +-- the ack: RAM applied, record staged, ONE commit done). +fn wal_mode(n: Int) -> Int { + let bs = from b in Bucket where b.tag == "b0" take 1 select b; + if len(bs) == 0 { + push(bs, insert Bucket { tag: "b0" }); + } + let i = 1; + while i <= n { + insert Item { k: 1000000 + i, v: item_v(i), bucket: bs[0] }; + print("acked ${i}"); + i = i + 1; + } + return 0; +} + +-- verify: the store against its own Meta expectations — count, checksum, +-- one unique-index probe. Exit 3 on any mismatch. +fn verify() -> Int { + let want_n = meta_val("count"); + let want_sum = meta_val("vsum"); + if want_n < 0 or want_sum < 0 { + print_err("verify: no meta (seed first)"); + return 3; + } + let got_n = 0; + let got_sum = 0; + for x in from x in Item select x { + if x.k < 1000000 { + got_n = got_n + 1; + got_sum = got_sum + x.v; + } + } + if got_n != want_n or got_sum != want_sum { + print_err("verify: count ${got_n}/${want_n} sum ${got_sum}/${want_sum}"); + return 3; + } + let bs = from b in Bucket where b.tag == "b0" take 1 select b; + if len(bs) == 0 { + print_err("verify: unique probe b0 missing"); + return 3; + } + print("verify ok ${got_n} rows sum ${got_sum}"); + return 0; +} + +-- verify-acked M: after a kill -9 mid-wal — rows 1..M (k = 1e6+i) must +-- exist with the right v; rows beyond M are allowed (acked after the +-- last print landed). Exit 3 on any missing/wrong row. +fn verify_acked(m: Int) -> Int { + let i = 1; + while i <= m { + let key = 1000000 + i; + let xs = from x in Item where x.k == key take 1 select x; + if len(xs) == 0 { + print_err("verify-acked: row ${i} missing"); + return 3; + } + if xs[0].v != item_v(i) { + print_err("verify-acked: row ${i} v ${xs[0].v} != ${item_v(i)}"); + return 3; + } + i = i + 1; + } + print("verify-acked ok ${m} rows"); + return 0; +} + +-- all N: the throughput campaign in ONE process — without WO_DATA the +-- store is RAM and dies with the process, so seed and the measured +-- modes must share a run; under WO_DATA the same mode prices the +-- durable flavor. Restart/crash legs use the separate modes. +fn all_mode(n: Int) -> Int { + let rc = seed(n); + if rc != 0 { + return rc; + } + rc = read_mode(n / 2); + if rc != 0 { + return rc; + } + rc = query_mode(n / 10); + if rc != 0 { + return rc; + } + rc = write_mode(n / 2); + if rc != 0 { + return rc; + } + return 0; +} + +fn usage() -> Int { + print_err("usage: db-bench "); + print_err(" all N | seed N | read N | query N | write N | wal N"); + print_err(" verify | verify-acked M"); + return 2; +} + +fn main(args: multi Text) -> Int { + if len(args) < 1 { + return usage(); + } + if args[0] == "verify" { + return verify(); + } + if len(args) < 2 { + return usage(); + } + let n = parse_int(args[1]); + if n == nil or n < 1 { + print_err("db-bench: must be a positive number"); + return 2; + } + if args[0] == "all" { + return all_mode(n); + } + if args[0] == "seed" { + return seed(n); + } + if args[0] == "read" { + return read_mode(n); + } + if args[0] == "query" { + return query_mode(n); + } + if args[0] == "write" { + return write_mode(n); + } + if args[0] == "wal" { + return wal_mode(n); + } + if args[0] == "verify-acked" { + return verify_acked(n); + } + return usage(); +} diff --git a/docs/examples/db-bench/types.wo b/docs/examples/db-bench/types.wo new file mode 100644 index 0000000..04da482 --- /dev/null +++ b/docs/examples/db-bench/types.wo @@ -0,0 +1,24 @@ +-- The bench store: the employee shape reduced to what pricing needs — +-- a parent with a @unique Text column (unique-index maintenance), a +-- child with a ref parent (FK probe on insert) and two indexed columns +-- (the equality-probe path). Meta rows carry seed expectations so +-- `verify` checks the store against facts that survived the same WAL. + +@table(name: "buckets", index: [tag]) +class Bucket { + tag: Text @unique +} + +@table(name: "items", index: [k], index: [bucket]) +class Item { + k: Int -- probe key; seeded non-unique (10 rows per key), + -- wal/write modes use disjoint high ranges + v: Int -- payload column, checksummed by verify + bucket: ref Bucket -- FK: primary-index probe on every insert +} + +@table(name: "meta", index: [tag]) +class Meta { + tag: Text @unique + val: Int +} diff --git a/docs/examples/db-bench/wo.toml b/docs/examples/db-bench/wo.toml new file mode 100644 index 0000000..ee587bf --- /dev/null +++ b/docs/examples/db-bench/wo.toml @@ -0,0 +1,6 @@ +name = "db-bench" +version = "0.1.0" +description = "iteration 22: the measurement backbone — timed DB workloads, single- and multi-shard" + +[runtime] +wo = ">= 0.1"