- hreadall/hreadkeys: the same resident A/B as wread_*, but ~2 KB per row so a GB of data is reachable in a few hundred thousand inserts - the insert path is fsync-bound at roughly 2 000 rows/s, so row COUNT is the expensive axis and row SIZE is nearly free — 20k rows already produce 38 MB - NOT RUN: the GB-scale measurement was called off. These modes are committed working and typechecking so the leg can be run later without rebuilding it, not because a result exists Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> (cherry picked from commit abc276ac39dd45a1052b6ae24d25aedb9eea3e1e)
1072 lines
30 KiB
Text
1072 lines
30 KiB
Text
use fs
|
|
use time
|
|
|
|
-- db-bench — iteration 22's load generator. Every measured mode prints
|
|
-- one machine-parsable line per operation class:
|
|
--
|
|
-- <op> <count> <ops/sec> <p50us> <p99us>
|
|
--
|
|
-- Timing is per-operation via time.ticks (CLOCK_MONOTONIC µs);
|
|
-- percentiles come from a 1µs-bucket histogram clamped at HIST_CLAMP —
|
|
-- exact to the microsecond below the clamp, and the clamp bucket keeps
|
|
-- the tail honest (a p99 AT the clamp means "clamp or worse").
|
|
-- The wal mode prints a running `acked <n>` line after every insert
|
|
-- RETURNS (the return IS the ack): the crash battery kills this mode
|
|
-- mid-run and verify-acked proves every acknowledged row survived.
|
|
|
|
-- ---- deterministic helpers ----
|
|
|
|
fn lcg(seed: Int) -> Int {
|
|
let x = seed * 1103515245 + 12345;
|
|
if x < 0 {
|
|
x = 0 - x;
|
|
}
|
|
return x;
|
|
}
|
|
|
|
fn item_v(i: Int) -> Int {
|
|
return (i * 37) % 1000;
|
|
}
|
|
|
|
-- ---- the histogram (percentiles without a sort) ----
|
|
|
|
fn hist_add(mut h: map<Int, Int>, us: Int) {
|
|
let b = us;
|
|
if b < 0 {
|
|
b = 0;
|
|
}
|
|
if b > 20000 {
|
|
b = 20000;
|
|
}
|
|
if has(h, b) {
|
|
set(h, b, get(h, b) + 1);
|
|
} else {
|
|
set(h, b, 1);
|
|
}
|
|
}
|
|
|
|
fn hist_pct(h: map<Int, Int>, total: Int, pct: Int) -> Int {
|
|
let target = total * pct / 100;
|
|
if target < 1 {
|
|
target = 1;
|
|
}
|
|
let seen = 0;
|
|
let b = 0;
|
|
while b <= 20000 {
|
|
if has(h, b) {
|
|
seen = seen + get(h, b);
|
|
if seen >= target {
|
|
return b;
|
|
}
|
|
}
|
|
b = b + 1;
|
|
}
|
|
return 20000;
|
|
}
|
|
|
|
fn report(op: Text, n: Int, total_us: Int, h: map<Int, Int>) {
|
|
let us = total_us;
|
|
if us < 1 {
|
|
us = 1;
|
|
}
|
|
let rate = n * 1000000 / us;
|
|
print("${op} ${n} ${rate} ${hist_pct(h, n, 50)} ${hist_pct(h, n, 99)}");
|
|
}
|
|
|
|
-- ---- modes ----
|
|
|
|
-- seed N: N children, one parent per 100, k = i % (N/10) (10 rows per
|
|
-- key), v deterministic. Meta rows record the expectations verify reads.
|
|
fn seed(n: Int) -> Int {
|
|
let h: map<Int, Int> = {};
|
|
let kmod = n / 10;
|
|
if kmod < 1 {
|
|
kmod = 1;
|
|
}
|
|
-- bucket-major: one parent, then its 100 children, using a single ref
|
|
-- local. (A hand-built `multi Bucket` of insert results SEGVs on drop —
|
|
-- the compiler classifies the elements OWNED while table refs are
|
|
-- scalar ids; recorded as a standing finding, not this iteration's fix.
|
|
-- Query-built multis are runtime-typed and safe.)
|
|
let vsum = 0;
|
|
let t0 = time.ticks();
|
|
let i = 1;
|
|
let b = 0;
|
|
while i <= n {
|
|
let bref = insert Bucket { tag: "b${b}" };
|
|
b = b + 1;
|
|
let j = 0;
|
|
while j < 100 and i <= n {
|
|
let o0 = time.ticks();
|
|
insert Item { k: i % kmod, v: item_v(i), bucket: bref };
|
|
hist_add(h, time.ticks() - o0);
|
|
vsum = vsum + item_v(i);
|
|
i = i + 1;
|
|
j = j + 1;
|
|
}
|
|
}
|
|
let t1 = time.ticks();
|
|
insert Meta { tag: "count", val: n };
|
|
insert Meta { tag: "vsum", val: vsum };
|
|
insert Meta { tag: "kmod", val: kmod };
|
|
report("seed", n, t1 - t0, h);
|
|
return 0;
|
|
}
|
|
|
|
fn meta_val(tag: Text) -> Int {
|
|
let ms = from m in Meta where m.tag == tag take 1 select m;
|
|
if len(ms) == 0 {
|
|
return -1;
|
|
}
|
|
return ms[0].val;
|
|
}
|
|
|
|
-- read N: indexed take-1 point lookups (the point-read this surface
|
|
-- offers), keys spread by LCG over the seeded key range.
|
|
fn read_mode(n: Int) -> Int {
|
|
let kmod = meta_val("kmod");
|
|
if kmod < 1 {
|
|
print_err("read: seed first");
|
|
return 1;
|
|
}
|
|
let h: map<Int, Int> = {};
|
|
let sink = 0;
|
|
let s = 42;
|
|
let t0 = time.ticks();
|
|
let i = 0;
|
|
while i < n {
|
|
s = lcg(s);
|
|
let key = s % kmod;
|
|
let o0 = time.ticks();
|
|
let xs = from x in Item where x.k == key take 1 select x;
|
|
if len(xs) > 0 {
|
|
sink = sink + xs[0].v;
|
|
}
|
|
hist_add(h, time.ticks() - o0);
|
|
i = i + 1;
|
|
}
|
|
let t1 = time.ticks();
|
|
report("read", n, t1 - t0, h);
|
|
if sink < 0 {
|
|
print("impossible ${sink}");
|
|
}
|
|
return 0;
|
|
}
|
|
|
|
-- query N: full equality probes on the k index (≈10 rows per key),
|
|
-- each materialized and counted.
|
|
fn query_mode(n: Int) -> Int {
|
|
let kmod = meta_val("kmod");
|
|
if kmod < 1 {
|
|
print_err("query: seed first");
|
|
return 1;
|
|
}
|
|
let h: map<Int, Int> = {};
|
|
let rows = 0;
|
|
let s = 7;
|
|
let t0 = time.ticks();
|
|
let i = 0;
|
|
while i < n {
|
|
s = lcg(s);
|
|
let key = s % kmod;
|
|
let o0 = time.ticks();
|
|
for x in from x in Item where x.k == key select x {
|
|
rows = rows + 1;
|
|
}
|
|
hist_add(h, time.ticks() - o0);
|
|
i = i + 1;
|
|
}
|
|
let t1 = time.ticks();
|
|
report("query", n, t1 - t0, h);
|
|
print("query rows ${rows}");
|
|
return 0;
|
|
}
|
|
|
|
-- write N: alternating inserts (disjoint k range 2e6+) and updates
|
|
-- through a query result. Corrupts vsum by design — the durability legs
|
|
-- run on their own fresh store.
|
|
fn write_mode(n: Int) -> Int {
|
|
let kmod = meta_val("kmod");
|
|
if kmod < 1 {
|
|
print_err("write: seed first");
|
|
return 1;
|
|
}
|
|
let bs = from b in Bucket where b.tag == "b0" take 1 select b;
|
|
if len(bs) == 0 {
|
|
print_err("write: no buckets");
|
|
return 1;
|
|
}
|
|
let h: map<Int, Int> = {};
|
|
let s = 99;
|
|
let t0 = time.ticks();
|
|
let i = 0;
|
|
while i < n {
|
|
let o0 = time.ticks();
|
|
if i % 2 == 0 {
|
|
insert Item { k: 2000000 + i, v: item_v(i), bucket: bs[0] };
|
|
} else {
|
|
s = lcg(s);
|
|
let key = s % kmod;
|
|
let xs = from x in Item where x.k == key take 1 select x;
|
|
if len(xs) > 0 {
|
|
xs[0].v = xs[0].v + 1;
|
|
}
|
|
}
|
|
hist_add(h, time.ticks() - o0);
|
|
i = i + 1;
|
|
}
|
|
let t1 = time.ticks();
|
|
report("write", n, t1 - t0, h);
|
|
return 0;
|
|
}
|
|
|
|
-- wal N: the crash battery's vehicle — insert-only, disjoint k range
|
|
-- (1e6+), `acked <i>` printed AFTER each insert returns (the return is
|
|
-- the ack: RAM applied, record staged, ONE commit done).
|
|
fn wal_mode(n: Int) -> Int {
|
|
let bs = from b in Bucket where b.tag == "b0" take 1 select b;
|
|
if len(bs) == 0 {
|
|
push(bs, insert Bucket { tag: "b0" });
|
|
}
|
|
let i = 1;
|
|
while i <= n {
|
|
insert Item { k: 1000000 + i, v: item_v(i), bucket: bs[0] };
|
|
print("acked ${i}");
|
|
i = i + 1;
|
|
}
|
|
return 0;
|
|
}
|
|
|
|
-- verify: the store against its own Meta expectations — count, checksum,
|
|
-- one unique-index probe. Exit 3 on any mismatch.
|
|
fn verify() -> Int {
|
|
let want_n = meta_val("count");
|
|
let want_sum = meta_val("vsum");
|
|
if want_n < 0 or want_sum < 0 {
|
|
print_err("verify: no meta (seed first)");
|
|
return 3;
|
|
}
|
|
let got_n = 0;
|
|
let got_sum = 0;
|
|
for x in from x in Item select x {
|
|
if x.k < 1000000 {
|
|
got_n = got_n + 1;
|
|
got_sum = got_sum + x.v;
|
|
}
|
|
}
|
|
if got_n != want_n or got_sum != want_sum {
|
|
print_err("verify: count ${got_n}/${want_n} sum ${got_sum}/${want_sum}");
|
|
return 3;
|
|
}
|
|
let bs = from b in Bucket where b.tag == "b0" take 1 select b;
|
|
if len(bs) == 0 {
|
|
print_err("verify: unique probe b0 missing");
|
|
return 3;
|
|
}
|
|
print("verify ok ${got_n} rows sum ${got_sum}");
|
|
return 0;
|
|
}
|
|
|
|
-- verify-acked M: after a kill -9 mid-wal — rows 1..M (k = 1e6+i) must
|
|
-- exist with the right v; rows beyond M are allowed (acked after the
|
|
-- last print landed). Exit 3 on any missing/wrong row.
|
|
fn verify_acked(m: Int) -> Int {
|
|
let i = 1;
|
|
while i <= m {
|
|
let key = 1000000 + i;
|
|
let xs = from x in Item where x.k == key take 1 select x;
|
|
if len(xs) == 0 {
|
|
print_err("verify-acked: row ${i} missing");
|
|
return 3;
|
|
}
|
|
if xs[0].v != item_v(i) {
|
|
print_err("verify-acked: row ${i} v ${xs[0].v} != ${item_v(i)}");
|
|
return 3;
|
|
}
|
|
i = i + 1;
|
|
}
|
|
print("verify-acked ok ${m} rows");
|
|
return 0;
|
|
}
|
|
|
|
-- ---- the concurrent modes (mix, msgrate) ----
|
|
|
|
fn hist_dump(h: map<Int, Int>, kind: Int) {
|
|
let b = 0;
|
|
while b <= 20000 {
|
|
if has(h, b) {
|
|
insert Hist { kind: kind, b: b, c: get(h, b) };
|
|
}
|
|
b = b + 1;
|
|
}
|
|
}
|
|
|
|
-- One mixer = one actor: 90/10 read/write over the seeded store. On a
|
|
-- worker shard every statement below rides the stage-3 DB RPC — the
|
|
-- code must not know or care (transparency is the point). Done signal:
|
|
-- a Meta row main polls for (the coordination idiom this side of 31).
|
|
class Mixer {
|
|
id: Int
|
|
fn receive(msg: MixJob) {
|
|
let hr: map<Int, Int> = {};
|
|
let hw: map<Int, Int> = {};
|
|
let s = msg.seed;
|
|
let sink = 0;
|
|
let i = 0;
|
|
while i < msg.ops {
|
|
s = lcg(s);
|
|
let key = s % msg.kmod;
|
|
let o0 = time.ticks();
|
|
if i % 10 == 9 {
|
|
let xs = from x in Item where x.k == key take 1 select x;
|
|
if len(xs) > 0 {
|
|
xs[0].v = xs[0].v + 1;
|
|
}
|
|
hist_add(hw, time.ticks() - o0);
|
|
} else {
|
|
let xs = from x in Item where x.k == key take 1 select x;
|
|
if len(xs) > 0 {
|
|
sink = sink + xs[0].v;
|
|
}
|
|
hist_add(hr, time.ticks() - o0);
|
|
}
|
|
i = i + 1;
|
|
}
|
|
hist_dump(hr, 0);
|
|
hist_dump(hw, 1);
|
|
insert Meta { tag: "mixdone${self.id}", val: sink };
|
|
}
|
|
}
|
|
|
|
-- databasev2 4 part A: every op a durable write, C at once.
|
|
--
|
|
-- Why this leg exists. `mix` writes on one op in ten with C=4, so at most a
|
|
-- handful of writes are ever in flight and group commit has almost nothing to
|
|
-- batch: measured mean batch 1.01 over 3112 barriers, peak 3. That is a
|
|
-- property of the WORKLOAD, not of the mechanism, and without a write-
|
|
-- concurrent leg the iteration's payoff cannot be evaluated either way.
|
|
--
|
|
-- Updates rather than inserts: comparable to what `mixwrite` measures, and the
|
|
-- row count stays flat so a long run does not turn into a growth test.
|
|
-- Histogram kind 2, because a replayed store still holds the seeding run's
|
|
-- kind-0/1 Hist rows and merging those would report someone else's latencies.
|
|
class WJob {
|
|
ops: Int
|
|
seed: Int
|
|
kmod: Int
|
|
}
|
|
|
|
class WMixer {
|
|
id: Int
|
|
fn receive(msg: WJob) {
|
|
let hw: map<Int, Int> = {};
|
|
let s = msg.seed;
|
|
let i = 0;
|
|
while i < msg.ops {
|
|
s = lcg(s);
|
|
let key = s % msg.kmod;
|
|
let o0 = time.ticks();
|
|
for r in from x in Item where x.k == key take 1 select x {
|
|
r.v = r.v + 1;
|
|
}
|
|
hist_add(hw, time.ticks() - o0);
|
|
i = i + 1;
|
|
}
|
|
hist_dump(hw, 2);
|
|
insert Meta { tag: "wmixdone${self.id}", val: msg.ops };
|
|
}
|
|
}
|
|
|
|
fn wmix_mode(total: Int, c: Int) -> Int {
|
|
let kmod = meta_val("kmod");
|
|
if kmod < 1 {
|
|
print_err("wmix: seed first");
|
|
return 1;
|
|
}
|
|
let per = total / c;
|
|
if per < 1 {
|
|
per = 1;
|
|
}
|
|
let wall0 = time.ticks();
|
|
let i = 0;
|
|
while i < c {
|
|
let a: actor WJob = spawn WMixer { id: i };
|
|
send(a, WJob { ops: per, seed: 4242 + i * 7919, kmod: kmod });
|
|
i = i + 1;
|
|
}
|
|
let done = 0;
|
|
while done < c {
|
|
time.sleep(20);
|
|
done = 0;
|
|
i = 0;
|
|
while i < c {
|
|
if meta_val("wmixdone${i}") >= 0 {
|
|
done = done + 1;
|
|
}
|
|
i = i + 1;
|
|
}
|
|
}
|
|
let wall = time.ticks() - wall0;
|
|
let hw: map<Int, Int> = {};
|
|
let nw = 0;
|
|
for x in from x in Hist select x {
|
|
if x.kind == 2 {
|
|
if has(hw, x.b) {
|
|
set(hw, x.b, get(hw, x.b) + x.c);
|
|
} else {
|
|
set(hw, x.b, x.c);
|
|
}
|
|
nw = nw + x.c;
|
|
}
|
|
}
|
|
report("wmix", nw, wall, hw);
|
|
return 0;
|
|
}
|
|
|
|
fn mix_mode(total: Int, c: Int) -> Int {
|
|
let kmod = meta_val("kmod");
|
|
if kmod < 1 {
|
|
print_err("mix: seed first");
|
|
return 1;
|
|
}
|
|
let per = total / c;
|
|
if per < 1 {
|
|
per = 1;
|
|
}
|
|
let wall0 = time.ticks();
|
|
let i = 0;
|
|
while i < c {
|
|
let a: actor MixJob = spawn Mixer { id: i };
|
|
send(a, MixJob { ops: per, seed: 1000 + i * 7919, kmod: kmod });
|
|
i = i + 1;
|
|
}
|
|
-- poll until every mixer's done row exists
|
|
let done = 0;
|
|
while done < c {
|
|
time.sleep(20);
|
|
done = 0;
|
|
i = 0;
|
|
while i < c {
|
|
if meta_val("mixdone${i}") >= 0 {
|
|
done = done + 1;
|
|
}
|
|
i = i + 1;
|
|
}
|
|
}
|
|
let wall = time.ticks() - wall0;
|
|
-- merge the dumped histograms; wall time is shared by both classes
|
|
let hr: map<Int, Int> = {};
|
|
let hw: map<Int, Int> = {};
|
|
let nr = 0;
|
|
let nw = 0;
|
|
for x in from x in Hist select x {
|
|
if x.kind == 0 {
|
|
if has(hr, x.b) {
|
|
set(hr, x.b, get(hr, x.b) + x.c);
|
|
} else {
|
|
set(hr, x.b, x.c);
|
|
}
|
|
nr = nr + x.c;
|
|
} else {
|
|
if has(hw, x.b) {
|
|
set(hw, x.b, get(hw, x.b) + x.c);
|
|
} else {
|
|
set(hw, x.b, x.c);
|
|
}
|
|
nw = nw + x.c;
|
|
}
|
|
}
|
|
report("mixread", nr, wall, hr);
|
|
report("mixwrite", nw, wall, hw);
|
|
return 0;
|
|
}
|
|
|
|
-- msgrate: one-way flood — main sends N messages at a sink actor; the
|
|
-- sink counts and writes the done row at N. Spawn TWO sinks and flood
|
|
-- the second: round-robin placement puts it off the primary whenever
|
|
-- more than one shard exists, so the multi-shard number prices the
|
|
-- mutex inbox (stage-2 deviation 4's number); single-shard prices the
|
|
-- same-heap path.
|
|
class Sink {
|
|
got: Int
|
|
fn receive(msg: Flood) {
|
|
self.got = self.got + 1;
|
|
if self.got == msg.n {
|
|
insert Meta { tag: "flooddone", val: self.got };
|
|
}
|
|
}
|
|
}
|
|
|
|
fn msgrate_mode(n: Int) -> Int {
|
|
let first: actor Flood = spawn Sink { got: 0 };
|
|
let a: actor Flood = spawn Sink { got: 0 };
|
|
if first == a {
|
|
print_err("msgrate: impossible");
|
|
}
|
|
let t0 = time.ticks();
|
|
let i = 0;
|
|
while i < n {
|
|
send(a, Flood { n: n });
|
|
i = i + 1;
|
|
}
|
|
while meta_val("flooddone") < 0 {
|
|
time.sleep(5);
|
|
}
|
|
let us = time.ticks() - t0;
|
|
if us < 1 {
|
|
us = 1;
|
|
}
|
|
print("msgrate ${n} ${n * 1000000 / us}");
|
|
return 0;
|
|
}
|
|
|
|
-- all N: the throughput campaign in ONE process — without WO_DATA the
|
|
-- store is RAM and dies with the process, so seed and the measured
|
|
-- modes must share a run; under WO_DATA the same mode prices the
|
|
-- durable flavor. Restart/crash legs use the separate modes.
|
|
fn all_mode(n: Int) -> Int {
|
|
let rc = seed(n);
|
|
if rc != 0 {
|
|
return rc;
|
|
}
|
|
rc = read_mode(n / 2);
|
|
if rc != 0 {
|
|
return rc;
|
|
}
|
|
rc = query_mode(n / 10);
|
|
if rc != 0 {
|
|
return rc;
|
|
}
|
|
rc = write_mode(n / 2);
|
|
if rc != 0 {
|
|
return rc;
|
|
}
|
|
-- mix at N/10: every point lookup is O(table) today (the probe walks
|
|
-- all slabs — a headline finding, not a bug to hide), so a read-heavy
|
|
-- mix over a seeded store is quadratic in N. The campaign driver
|
|
-- chooses absolute sizes; this keeps `all` finishing in minutes.
|
|
return mix_mode(n / 10, 4);
|
|
}
|
|
|
|
fn usage() -> Int {
|
|
print_err("usage: db-bench <mode>");
|
|
print_err(" all N | seed N | read N | query N | write N | wal N");
|
|
print_err(" mix N C | wmix N C | msgrate N | growth N int|text | growth-verify");
|
|
print_err(" randread N R | replayseed N M | boot");
|
|
print_err(" verify | verify-acked M");
|
|
return 2;
|
|
}
|
|
|
|
-- databasev2 1: the process's own resident size, in KiB. Read here rather
|
|
-- than sampled by the driver because the driver polls /proc every 250 ms and
|
|
-- would miss the value AT a decile boundary; per-row footprint is the headline
|
|
-- number of this iteration and deserves an exact reading, not a nearby one.
|
|
-- Absence is nil by stdlib convention, so a kernel without VmRSS reports 0
|
|
-- and the driver treats the leg as unavailable rather than as zero growth.
|
|
fn self_rss_kb() -> Int {
|
|
let st = try fs.read_all("/proc/self/status", 16384) catch (e) "";
|
|
let i = index_of(st, "VmRSS:");
|
|
if i < 0 {
|
|
return 0;
|
|
}
|
|
let rest = substr(st, i + 6, 24);
|
|
let n = 0;
|
|
let j = 0;
|
|
while j < len(rest) {
|
|
let c = byte_at(rest, j);
|
|
if c >= 48 and c <= 57 {
|
|
n = n * 10 + (c - 48);
|
|
} else {
|
|
if n > 0 {
|
|
return n;
|
|
}
|
|
}
|
|
j = j + 1;
|
|
}
|
|
return n;
|
|
}
|
|
|
|
-- databasev2 1: growth N SHAPE — insert N rows of one reference shape,
|
|
-- sampling read latency as the table grows so the driver can plot the CURVE
|
|
-- rather than two endpoints. Reports one metric line per decile so the point
|
|
-- at which p99 leaves its baseline is a MEASURED sample, not an estimate.
|
|
--
|
|
-- SHAPE is "int" (Item: two Ints plus a ref, all inline slot words) or "text"
|
|
-- (Wide: three Text columns, each a separate db_text allocation on top of the
|
|
-- slab slot). Per-row footprint differs by an order of magnitude between them,
|
|
-- which is exactly why the driver reports the two separately and never a single
|
|
-- "bytes per row".
|
|
--
|
|
-- The memory CAP is the driver's job (systemd-run --user --scope), not this
|
|
-- program's: the sample just grows and reports, so the same binary serves the
|
|
-- swap-off and swap-on legs unchanged.
|
|
-- after the process is OOM-killed mid-insert, the durable prefix must be
|
|
-- intact: rows 1..M all present with the right v and no holes. M is whatever
|
|
-- survived -- the claim under test is the SHAPE of the survivor, not its size,
|
|
-- because a SIGKILL can land between any two inserts.
|
|
-- databasev2 1: randread N R -- fill N rows, then read R of them by key in a
|
|
-- Weyl-sequence order that spreads across the WHOLE range. Under a cap smaller
|
|
-- than the table most of those reads must fault a page back in.
|
|
--
|
|
-- This is the leg the swap measurement was MISSING. `growth` inserts, and
|
|
-- inserting is append-mostly: cold pages are written once and never re-read, so
|
|
-- swap cost it ~1% (148s vs 150s uncapped). Random reads over an oversized
|
|
-- table are the opposite access pattern -- and they are exactly what
|
|
-- databasev2 2's `resident: keys` creates, since it reads rows back from a log
|
|
-- larger than RAM. No RNG in the language and none needed: i*2654435761 mod n
|
|
-- is a Weyl sequence, deterministic and spread, so the two legs read the SAME
|
|
-- key order and only residency differs.
|
|
-- databasev2 1, for iteration 3: the replay "before".
|
|
--
|
|
-- `boot` does NOTHING. That is the point: with WO_DATA set the runtime replays
|
|
-- the whole WAL before main runs, so the process's wall time IS the replay cost
|
|
-- plus a fixed startup. Any mode that touches rows would mix its own work into
|
|
-- the number.
|
|
fn boot_mode() -> Int {
|
|
print("booted");
|
|
return 0;
|
|
}
|
|
|
|
-- Build a store with N live rows and N+M total WAL records: M updates on top of
|
|
-- N inserts. The live dataset is IDENTICAL for any M -- only the history grows.
|
|
-- That is iteration 3's whole case: with no checkpoint, boot replays HISTORY,
|
|
-- not data, so a long-lived row that has been updated a thousand times costs a
|
|
-- thousand records at every boot forever.
|
|
fn replayseed_mode(n: Int, m: Int) -> Int {
|
|
let bref = insert Bucket { tag: "replay" };
|
|
let i = 1;
|
|
while i <= n {
|
|
insert Item { k: i, v: item_v(i), bucket: bref };
|
|
i = i + 1;
|
|
}
|
|
let j = 0;
|
|
while j < m {
|
|
let key = 1 + (j * 2654435761) % n;
|
|
for r in from x in Item where x.k == key take 1 select x {
|
|
r.v = r.v + 1;
|
|
}
|
|
j = j + 1;
|
|
}
|
|
print("replayseeded ${n} ${m}");
|
|
return 0;
|
|
}
|
|
|
|
-- databasev2 2 task 7: the residency A/B, one mode per table so the two runs
|
|
-- differ ONLY in the annotation. Same fill, same Weyl key order, same read
|
|
-- count as randread_mode above — the comparison is against that leg's own
|
|
-- 273x swap figure, measured on the same box under the same cap.
|
|
-- The WIDE half of the residency A/B. Same structure as kread_all/kread_keys
|
|
-- but with Text columns, which is the only shape where dropping a payload
|
|
-- frees anything: an Int's value is its inline slot word, a Text's is a
|
|
-- separate allocation.
|
|
-- databasev2 2 task 7, the GB-scale leg. Same A/B as wread_*, but each row
|
|
-- carries ~2 KB of Text so a realistic data volume is reachable in a few
|
|
-- hundred thousand inserts rather than millions — the insert path is
|
|
-- fsync-bound at roughly 2 000 rows/s, so row COUNT is the expensive axis and
|
|
-- row SIZE is nearly free.
|
|
--
|
|
-- This is the shape the mode actually claims: data far larger than the cap,
|
|
-- with only the id map, the indexes and the (never-released) slabs resident.
|
|
fn heavy_pad() -> Text {
|
|
let p = "0123456789abcdef0123456789abcdef";
|
|
let out = "";
|
|
let i = 0;
|
|
while i < 20 { out = out .. p; i = i + 1; }
|
|
return out;
|
|
}
|
|
|
|
fn hread_all(n: Int, r: Int) -> Int {
|
|
let pad = heavy_pad();
|
|
let i = 1;
|
|
while i <= n {
|
|
insert WideA { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}" };
|
|
i = i + 1;
|
|
}
|
|
print("hreadfilled ${n} ${self_rss_kb()}");
|
|
let h: map<Int, Int> = {};
|
|
let hits = 0;
|
|
let t0 = time.ticks();
|
|
let j = 0;
|
|
while j < r {
|
|
let key = 1 + (j * 2654435761) % n;
|
|
let o0 = time.ticks();
|
|
for row in from x in WideA where x.k == key take 1 select x {
|
|
if len(row.a) > 0 { hits = hits + 1; }
|
|
}
|
|
hist_add(h, time.ticks() - o0);
|
|
j = j + 1;
|
|
}
|
|
let el = time.ticks() - t0;
|
|
report("hreadall", r, el, h);
|
|
print("hreadallrss ${self_rss_kb()} ${hits}");
|
|
return 0;
|
|
}
|
|
|
|
fn hread_keys(n: Int, r: Int) -> Int {
|
|
let pad = heavy_pad();
|
|
let i = 1;
|
|
while i <= n {
|
|
insert WideK { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}" };
|
|
i = i + 1;
|
|
}
|
|
print("hreadfilled ${n} ${self_rss_kb()}");
|
|
let h: map<Int, Int> = {};
|
|
let hits = 0;
|
|
let t0 = time.ticks();
|
|
let j = 0;
|
|
while j < r {
|
|
let key = 1 + (j * 2654435761) % n;
|
|
let o0 = time.ticks();
|
|
for row in from x in WideK where x.k == key take 1 select x {
|
|
if len(row.a) > 0 { hits = hits + 1; }
|
|
}
|
|
hist_add(h, time.ticks() - o0);
|
|
j = j + 1;
|
|
}
|
|
let el = time.ticks() - t0;
|
|
report("hreadkeys", r, el, h);
|
|
print("hreadkeysrss ${self_rss_kb()} ${hits}");
|
|
return 0;
|
|
}
|
|
|
|
fn wread_all(n: Int, r: Int) -> Int {
|
|
let pad = "0123456789abcdef0123456789abcdef";
|
|
let i = 1;
|
|
while i <= n {
|
|
insert WideA { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" };
|
|
i = i + 1;
|
|
}
|
|
print("wreadfilled ${n} ${self_rss_kb()}");
|
|
let h: map<Int, Int> = {};
|
|
let hits = 0;
|
|
let t0 = time.ticks();
|
|
let j = 0;
|
|
while j < r {
|
|
let key = 1 + (j * 2654435761) % n;
|
|
let o0 = time.ticks();
|
|
for row in from x in WideA where x.k == key take 1 select x {
|
|
if len(row.a) > 0 { hits = hits + 1; }
|
|
}
|
|
hist_add(h, time.ticks() - o0);
|
|
j = j + 1;
|
|
}
|
|
let el = time.ticks() - t0;
|
|
report("wreadall", r, el, h);
|
|
print("wreadallrss ${self_rss_kb()} ${hits}");
|
|
return 0;
|
|
}
|
|
|
|
fn wread_keys(n: Int, r: Int) -> Int {
|
|
let pad = "0123456789abcdef0123456789abcdef";
|
|
let i = 1;
|
|
while i <= n {
|
|
insert WideK { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" };
|
|
i = i + 1;
|
|
}
|
|
print("wreadfilled ${n} ${self_rss_kb()}");
|
|
let h: map<Int, Int> = {};
|
|
let hits = 0;
|
|
let t0 = time.ticks();
|
|
let j = 0;
|
|
while j < r {
|
|
let key = 1 + (j * 2654435761) % n;
|
|
let o0 = time.ticks();
|
|
for row in from x in WideK where x.k == key take 1 select x {
|
|
if len(row.a) > 0 { hits = hits + 1; }
|
|
}
|
|
hist_add(h, time.ticks() - o0);
|
|
j = j + 1;
|
|
}
|
|
let el = time.ticks() - t0;
|
|
report("wreadkeys", r, el, h);
|
|
print("wreadkeysrss ${self_rss_kb()} ${hits}");
|
|
return 0;
|
|
}
|
|
|
|
fn kread_all(n: Int, r: Int) -> Int {
|
|
let i = 1;
|
|
while i <= n {
|
|
insert RowA { k: i, v: item_v(i) };
|
|
i = i + 1;
|
|
}
|
|
print("kreadfilled ${n} ${self_rss_kb()}");
|
|
let h: map<Int, Int> = {};
|
|
let hits = 0;
|
|
let t0 = time.ticks();
|
|
let j = 0;
|
|
while j < r {
|
|
let key = 1 + (j * 2654435761) % n;
|
|
let o0 = time.ticks();
|
|
for row in from x in RowA where x.k == key take 1 select x {
|
|
if row.v == item_v(key) { hits = hits + 1; }
|
|
}
|
|
hist_add(h, time.ticks() - o0);
|
|
j = j + 1;
|
|
}
|
|
let el = time.ticks() - t0;
|
|
report("kreadall", r, el, h);
|
|
print("kreadallrss ${self_rss_kb()} ${hits}");
|
|
return 0;
|
|
}
|
|
|
|
fn kread_keys(n: Int, r: Int) -> Int {
|
|
let i = 1;
|
|
while i <= n {
|
|
insert RowK { k: i, v: item_v(i) };
|
|
i = i + 1;
|
|
}
|
|
print("kreadfilled ${n} ${self_rss_kb()}");
|
|
let h: map<Int, Int> = {};
|
|
let hits = 0;
|
|
let t0 = time.ticks();
|
|
let j = 0;
|
|
while j < r {
|
|
let key = 1 + (j * 2654435761) % n;
|
|
let o0 = time.ticks();
|
|
for row in from x in RowK where x.k == key take 1 select x {
|
|
if row.v == item_v(key) { hits = hits + 1; }
|
|
}
|
|
hist_add(h, time.ticks() - o0);
|
|
j = j + 1;
|
|
}
|
|
let el = time.ticks() - t0;
|
|
report("kreadkeys", r, el, h);
|
|
print("kreadkeysrss ${self_rss_kb()} ${hits}");
|
|
return 0;
|
|
}
|
|
|
|
fn randread_mode(n: Int, r: Int) -> Int {
|
|
let bref = insert Bucket { tag: "randread" };
|
|
let i = 1;
|
|
while i <= n {
|
|
insert Item { k: i, v: item_v(i), bucket: bref };
|
|
i = i + 1;
|
|
}
|
|
print("randreadfilled ${n} ${self_rss_kb()}");
|
|
let h: map<Int, Int> = {};
|
|
let hits = 0;
|
|
let t0 = time.ticks();
|
|
let j = 0;
|
|
while j < r {
|
|
let key = 1 + (j * 2654435761) % n;
|
|
let o0 = time.ticks();
|
|
for row in from x in Item where x.k == key take 1 select x {
|
|
if row.v == item_v(key) {
|
|
hits = hits + 1;
|
|
}
|
|
}
|
|
hist_add(h, time.ticks() - o0);
|
|
j = j + 1;
|
|
}
|
|
let el = time.ticks() - t0;
|
|
report("randread", r, el, h);
|
|
-- hits proves the reads RESOLVED; a collapse measured over misses is noise
|
|
print("randreadrss ${self_rss_kb()} ${hits}");
|
|
return 0;
|
|
}
|
|
|
|
fn growth_verify() -> Int {
|
|
let seen: map<Int, Int> = {};
|
|
let maxk = 0;
|
|
for r in from x in Item select x {
|
|
set(seen, r.k, r.v);
|
|
if r.k > maxk {
|
|
maxk = r.k;
|
|
}
|
|
}
|
|
let i = 1;
|
|
while i <= maxk {
|
|
if has(seen, i) == false {
|
|
print_err("growth-verify: hole at ${i} below max ${maxk}");
|
|
return 3;
|
|
}
|
|
if get(seen, i) != item_v(i) {
|
|
print_err("growth-verify: row ${i} v ${get(seen, i)} != ${item_v(i)}");
|
|
return 3;
|
|
}
|
|
i = i + 1;
|
|
}
|
|
print("growthverify ${maxk}");
|
|
return 0;
|
|
}
|
|
|
|
fn growth_mode(n: Int, shape: Text) -> Int {
|
|
let wide = shape == "text";
|
|
if wide == false and shape != "int" {
|
|
print_err("db-bench: growth SHAPE must be `int` or `text`");
|
|
return 2;
|
|
}
|
|
let step = n / 10;
|
|
if step < 1 {
|
|
step = 1;
|
|
}
|
|
let bref = insert Bucket { tag: "growth" };
|
|
let pad = "0123456789abcdef0123456789abcdef";
|
|
let i = 1;
|
|
while i <= n {
|
|
if wide {
|
|
insert Wide { k: i, a: "a${i}${pad}", b: "b${i}${pad}", note: "n${i}${pad}${pad}" };
|
|
} else {
|
|
insert Item { k: i, v: item_v(i), bucket: bref };
|
|
}
|
|
-- at each decile, sample the read path against what is resident NOW
|
|
if i % step == 0 {
|
|
let h: map<Int, Int> = {};
|
|
let probes = 200;
|
|
let pt0 = time.ticks();
|
|
let j = 0;
|
|
while j < probes {
|
|
let key = 1 + (j * step) % i;
|
|
let o0 = time.ticks();
|
|
if wide {
|
|
for r in from x in Wide where x.k == key take 1 select x {
|
|
hist_add(h, time.ticks() - o0);
|
|
}
|
|
} else {
|
|
for r in from x in Item where x.k == key take 1 select x {
|
|
hist_add(h, time.ticks() - o0);
|
|
}
|
|
}
|
|
j = j + 1;
|
|
}
|
|
let pel = time.ticks() - pt0;
|
|
-- op name carries the decile so the driver keys each sample distinctly
|
|
report("growth${i / step}", probes, pel, h);
|
|
-- rows and resident KiB at this decile: the driver divides to get the
|
|
-- per-row footprint for THIS shape
|
|
print("growthrss ${i / step} ${i} ${self_rss_kb()}");
|
|
}
|
|
i = i + 1;
|
|
}
|
|
print("growthdone ${n}");
|
|
return 0;
|
|
}
|
|
|
|
fn main(args: multi Text) -> Int {
|
|
if len(args) < 1 {
|
|
return usage();
|
|
}
|
|
if args[0] == "verify" {
|
|
return verify();
|
|
}
|
|
if args[0] == "growth-verify" {
|
|
return growth_verify();
|
|
}
|
|
-- Does NOTHING. With WO_DATA set the runtime replays the whole log before
|
|
-- main runs, so a mode with no work of its own measures replay plus a fixed
|
|
-- process start — which is what "boot time" has to mean. Both databasev2 1
|
|
-- (replay baseline) and databasev2 3 (checkpoint boot) price boot with it.
|
|
if args[0] == "boot" {
|
|
return boot_mode();
|
|
}
|
|
if len(args) < 2 {
|
|
return usage();
|
|
}
|
|
let n = parse_int(args[1]);
|
|
if n == nil or n < 1 {
|
|
print_err("db-bench: <n> must be a positive number");
|
|
return 2;
|
|
}
|
|
if args[0] == "all" {
|
|
return all_mode(n);
|
|
}
|
|
if args[0] == "seed" {
|
|
return seed(n);
|
|
}
|
|
if args[0] == "read" {
|
|
return read_mode(n);
|
|
}
|
|
if args[0] == "query" {
|
|
return query_mode(n);
|
|
}
|
|
if args[0] == "write" {
|
|
return write_mode(n);
|
|
}
|
|
if args[0] == "wal" {
|
|
return wal_mode(n);
|
|
}
|
|
if args[0] == "verify-acked" {
|
|
return verify_acked(n);
|
|
}
|
|
if args[0] == "msgrate" {
|
|
return msgrate_mode(n);
|
|
}
|
|
if args[0] == "growth" {
|
|
if len(args) < 3 {
|
|
return usage();
|
|
}
|
|
return growth_mode(n, args[2]);
|
|
}
|
|
if args[0] == "replayseed" {
|
|
if len(args) < 3 {
|
|
return usage();
|
|
}
|
|
let mm = parse_int(args[2]);
|
|
if mm == nil or mm < 0 {
|
|
print_err("db-bench: <m> must be zero or more");
|
|
return 2;
|
|
}
|
|
return replayseed_mode(n, mm);
|
|
}
|
|
if args[0] == "hreadall" or args[0] == "hreadkeys" {
|
|
if len(args) < 3 { print_err("usage: hreadall|hreadkeys N R"); return 2; }
|
|
let hn = parse_int(args[1]);
|
|
let hr = parse_int(args[2]);
|
|
if hn == nil or hr == nil { print_err("db-bench: N and R must be positive"); return 2; }
|
|
if args[0] == "hreadall" { return hread_all(hn, hr); }
|
|
return hread_keys(hn, hr);
|
|
}
|
|
if args[0] == "wreadall" or args[0] == "wreadkeys" {
|
|
if len(args) < 3 { print_err("usage: wreadall|wreadkeys N R"); return 2; }
|
|
let wn = parse_int(args[1]);
|
|
let wr = parse_int(args[2]);
|
|
if wn == nil or wr == nil { print_err("db-bench: N and R must be positive"); return 2; }
|
|
if args[0] == "wreadall" { return wread_all(wn, wr); }
|
|
return wread_keys(wn, wr);
|
|
}
|
|
if args[0] == "kreadall" or args[0] == "kreadkeys" {
|
|
if len(args) < 3 { print_err("usage: kreadall|kreadkeys N R"); return 2; }
|
|
let n = parse_int(args[1]);
|
|
let rr = parse_int(args[2]);
|
|
if n == nil or rr == nil {
|
|
print_err("db-bench: N and R must be positive numbers");
|
|
return 2;
|
|
}
|
|
if args[0] == "kreadall" { return kread_all(n, rr); }
|
|
return kread_keys(n, rr);
|
|
}
|
|
if args[0] == "randread" {
|
|
if len(args) < 3 {
|
|
return usage();
|
|
}
|
|
let rr = parse_int(args[2]);
|
|
if rr == nil or rr < 1 {
|
|
print_err("db-bench: <r> must be a positive number");
|
|
return 2;
|
|
}
|
|
return randread_mode(n, rr);
|
|
}
|
|
if args[0] == "wmix" {
|
|
if len(args) < 3 {
|
|
return usage();
|
|
}
|
|
let wc = parse_int(args[2]);
|
|
if wc == nil or wc < 1 {
|
|
print_err("db-bench: <c> must be a positive number");
|
|
return 2;
|
|
}
|
|
return wmix_mode(n, wc);
|
|
}
|
|
if args[0] == "mix" {
|
|
if len(args) < 3 {
|
|
return usage();
|
|
}
|
|
let c = parse_int(args[2]);
|
|
if c == nil or c < 1 {
|
|
print_err("db-bench: <c> must be a positive number");
|
|
return 2;
|
|
}
|
|
return mix_mode(n, c);
|
|
}
|
|
return usage();
|
|
}
|