diff --git a/database/src/table.c b/database/src/table.c index a999674..da2fbdd 100644 --- a/database/src/table.c +++ b/database/src/table.c @@ -774,7 +774,8 @@ db_row *wo_row_ptr(wo_db *db, uint32_t class_id, uint64_t id) { * direction, invisible for scalars (decode is identity there) and silent * wrong-bytes for Text/Bytes, which is exactly what stayed unexercised. */ static db_row *keys_fold_into(wo_db *db, uint32_t class_id, uint64_t id, - uint64_t off, uint8_t *buf, const char **msg) { + uint64_t off, uint8_t *buf, uint32_t *hops_out, + const char **msg) { const wo_classdesc *c = &db->classes[class_id]; db_row *r = (db_row *)buf; uint32_t got_cid = 0; @@ -783,7 +784,8 @@ static db_row *keys_fold_into(wo_db *db, uint32_t class_id, uint64_t id, * read — a row's current offset may point at a delta, not a base row. * Folds straight into r->slots: field_cnt uint64_t slots is exactly * what out_vals expects, and what a db_row already provides. */ - if (wo_wal_fold_row_at((wo_wal *)db->rt->wal, db, off, &got_cid, &got_id, r->slots, msg) != 0) + if (wo_wal_fold_row_at((wo_wal *)db->rt->wal, db, off, &got_cid, &got_id, r->slots, + hops_out, msg) != 0) return NULL; if (got_cid != class_id || got_id != id) { /* the offset pointed at someone else's record — a compaction that @@ -844,7 +846,7 @@ db_row *wo_row_borrow(wo_db *db, uint32_t class_id, uint64_t id, const char **ms t->scratch = nb; t->scratch_cap = t->row_size; } - db_row *r = keys_fold_into(db, class_id, id, o1 - 1, t->scratch, msg); + db_row *r = keys_fold_into(db, class_id, id, o1 - 1, t->scratch, &t->scratch_hops, msg); if (!r) return NULL; t->scratch_busy = 1; return r; @@ -1234,7 +1236,7 @@ static int row_apply_field_keys(wo_db *db, uint32_t class_id, uint64_t id, if (!cand_off1) continue; /* stale bucket entry: no row, no clash */ const char *obmsg = ""; db_row *other = - keys_fold_into(db, class_id, b->ids[i], cand_off1 - 1, cand_buf, &obmsg); + keys_fold_into(db, class_id, b->ids[i], cand_off1 - 1, cand_buf, NULL, &obmsg); int clash = other && idx_cols_equal(c, ix, r, other); /* keys_fold_into decoded fresh ENGINE values for EVERY field, same as a real borrow — nobody else owns them, so drop them @@ -1264,7 +1266,30 @@ static int row_apply_field_keys(wo_db *db, uint32_t class_id, uint64_t id, (void)idx_add_row(db, t, r); /* cannot violate uniqueness: the shadow check above already cleared it */ - if (wo_wal_append_delta(w, db, class_id, id, field, back_off, nv) != 0) + /* databasev2 11: FLATTEN ON UPDATE. + * + * `r` now holds the complete post-update row, because maintaining the + * indexes above required folding it — so writing a full-row image costs no + * extra read, only the bytes. Past WO_DELTA_MAX_HOPS we spend those bytes + * and terminate the chain instead of lengthening it. + * + * Why this lives here rather than in the checkpoint: compaction bounds + * chain length in principle, but its trigger is a byte ratio over the whole + * log and cannot see that ONE row has a long chain. A single hot row — + * this feature's own motivating workload, a popular SKU whose stock moves + * on every order — grows without ever moving that ratio. PostgreSQL solves + * the same shape the same way: heap_page_prune_opt collapses a HOT chain + * opportunistically, on a page the process already holds, rather than + * waiting for the background sweep. + * + * A full-row record is written as WO_WAL_INSERT because that is what a + * chain's base must be — it has to replay into a database where nothing + * precedes it. Replay, compaction and the fold all already handle that + * shape; none of them needs to know this happened. */ + int flattened = (t->scratch_hops >= WO_DELTA_MAX_HOPS); + int arc = flattened ? wo_wal_append_row_image(w, db, class_id, id, r) + : wo_wal_append_delta(w, db, class_id, id, field, back_off, nv); + if (arc != 0) wo_wal_stage_fatal(w); /* RAM already moved; see the insert arm */ db_val_free(c->kinds[field], old_eng); /* old value done: r now holds nv */ diff --git a/database/src/table.h b/database/src/table.h index cf5675b..8f562a8 100644 --- a/database/src/table.h +++ b/database/src/table.h @@ -130,6 +130,11 @@ typedef struct db_table { uint8_t *scratch; size_t scratch_cap; int scratch_busy; + /* databasev2 11: how many DELTA records the last borrow's fold crossed. + * The fold reports it for free, and the update path uses it to decide when + * a chain is long enough to be worth terminating with a full-row record. + * Meaningful only while scratch_busy is set. */ + uint32_t scratch_hops; } db_table; typedef struct wo_db { @@ -194,6 +199,22 @@ int wo_row_remove(wo_db *db, uint32_t class_id, uint64_t id); * 0 ok, -1 unknown class/row. */ int wo_row_drop_payload(wo_db *db, uint32_t class_id, uint64_t id, uint64_t wal_off); int wo_row_set_offset(wo_db *db, uint32_t class_id, uint64_t id, uint64_t wal_off); + +/* databasev2 11: how many DELTA records a keys-resident row's chain may carry + * before an update terminates it with a full-row image instead of lengthening + * it. A BOUND, not a tuning knob — PostgreSQL ships `fillfactor` and + * autovacuum's base threshold as documented constants that are rarely touched, + * and this is the same kind of number. Anything in the low tens caps the + * pathology; being wrong by a factor of two costs one row-sized write per K + * updates, which is not a correctness failure in either direction. + * + * It deliberately does NOT scale with table size. PostgreSQL scales autovacuum + * by reltuples because it thresholds a table-level aggregate whose harm is + * proportional; a chain is a per-ROW property with additive cost — reading one + * row costs 1 + depth reads whether the table holds a hundred rows or ten + * million, and replay is the sum over every row's chain. Scaling this up with + * table size would make the largest databases boot worst. */ +#define WO_DELTA_MAX_HOPS 16u uint64_t wo_row_offset1(const wo_db *db, uint32_t class_id, uint64_t id); /* databasev2 2 (5d): iterate the live row IDS of a table, whichever backing it diff --git a/database/src/wal.c b/database/src/wal.c index c59751f..c3921ff 100644 --- a/database/src/wal.c +++ b/database/src/wal.c @@ -498,6 +498,25 @@ int wo_wal_append_update(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id) { return rc; } +/* databasev2 11: the chain-terminating write. Same encode as append_insert, + * but from a row the caller already holds — a keys-resident row whose payload + * has been dropped has no slab image for wo_row_ptr to find, and the update + * path is the one caller that legitimately has the full, post-update values in + * hand because it folded them to maintain indexes. */ +int wo_wal_append_row_image(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id, + const db_row *r) { + if (!r) return -1; + wbuf p = {0}; + wput_u8(&p, WO_WAL_INSERT); + wput_u32(&p, class_id); + wput_u64(&p, id); + const wo_classdesc *c = &db->classes[class_id]; + for (uint32_t i = 0; i < c->field_cnt; i++) enc_val(&p, db->classes, c->kinds[i], r->slots[i]); + int rc = stage(w, &p); + free(p.b); + return rc; +} + int wo_wal_append_delta(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id, uint32_t field_idx, uint64_t back_off, uint64_t value) { wbuf p = {0}; @@ -586,7 +605,28 @@ int wo_wal_should_compact(uint64_t used, uint64_t last, uint64_t floor, uint32_t * to establish the denominator */ if (ratio == 0) return 0; /* a zero ratio disables the policy rather than * dividing by nothing */ - return used > last * (uint64_t)ratio; + + /* databasev2 11: an ABSOLUTE garbage term, and a ceiling on the + * proportional one. + * + * NOTE THE VOCABULARY TRAP this fixes. `floor` above SUPPRESSES compaction + * on a small log — the opposite of what the same word means in PostgreSQL, + * where autovacuum's `vac_base_thresh` (default 50) TRIGGERS cleanup on a + * small absolute problem that the proportional term would hide. We had the + * proportion and the suppressor and neither of the two guards that keep a + * size-based policy honest: + * + * - without a triggering term, garbage that is large in bytes but small + * relative to a big live set is never reclaimed; + * - without a ceiling, a very large live set defers compaction forever, + * which is what autovacuum_vacuum_max_threshold exists to stop. */ + uint64_t garbage = used > last ? used - last : 0; + if (garbage >= WO_CKPT_ABS_BYTES) return 1; + + uint64_t trigger = last * (uint64_t)ratio; + uint64_t ceiling = last + WO_CKPT_MAX_GARBAGE; + if (trigger > ceiling) trigger = ceiling; + return used > trigger; } /* databasev2 3: how many records the dump stages before flushing. @@ -731,7 +771,7 @@ static int stage_flattened_row(wo_wal *nw, wo_wal *ow, wo_db *db, uint64_t off, uint32_t got_cid = 0; uint64_t got_id = 0; const char *fmsg = ""; - if (wo_wal_fold_row_at(ow, db, off, &got_cid, &got_id, vals, &fmsg) != 0 || + if (wo_wal_fold_row_at(ow, db, off, &got_cid, &got_id, vals, NULL, &fmsg) != 0 || got_cid != class_id || got_id != id) { for (uint32_t i = 0; i < c->field_cnt; i++) wo_db_val_free(db, c->kinds[i], vals[i]); free(vals); @@ -922,7 +962,7 @@ static int apply_delta(wo_db *db, uint32_t cid, uint64_t id, rbuf *r) { uint64_t got_id = 0; const char *fmsg = ""; if (wo_wal_fold_row_at((wo_wal *)db->rt->wal, db, back_off, &got_cid, &got_id, vals, - &fmsg) != 0 || + NULL, &fmsg) != 0 || got_cid != cid || got_id != id) { for (uint32_t i = 0; i < field_cnt; i++) wo_db_val_free(db, c->kinds[i], vals[i]); free(vals); @@ -1083,7 +1123,10 @@ int wo_wal_read_row_at(wo_wal *w, wo_db *db, wo_rt *rt, uint64_t off, /* keys-resident delta updates, Task 2: THE fold. See wal.h. Reads, replay, * and compaction all call this one function — never a second copy. */ int wo_wal_fold_row_at(wo_wal *w, wo_db *db, uint64_t off, uint32_t *class_out, - uint64_t *id_out, uint64_t *out_vals, const char **msg) { + uint64_t *id_out, uint64_t *out_vals, uint32_t *hops_out, + const char **msg) { + uint32_t hops = 0; /* databasev2 11: DELTA records crossed */ + if (hops_out) *hops_out = 0; uint32_t cid = 0; uint64_t id = 0; uint32_t field_cnt = 0; @@ -1180,6 +1223,8 @@ int wo_wal_fold_row_at(wo_wal *w, wo_db *db, uint64_t off, uint32_t *class_out, wo_db_val_free(db, c->kinds[field_idx], v); } free(payload); + hops++; + if (hops_out) *hops_out = hops; cur = back_off; continue; } diff --git a/database/src/wal.h b/database/src/wal.h index ea6e850..aee893f 100644 --- a/database/src/wal.h +++ b/database/src/wal.h @@ -217,6 +217,20 @@ void wo_db_flush_drops(wo_db *db, wo_wal *w); * grow, so a timer would fire with nothing to do. * * 1 = compact now, 0 = leave it. */ +/* databasev2 11: the two terms a size-based policy needs beside its ratio. + * + * WO_CKPT_ABS_BYTES is the TRIGGERING threshold — PostgreSQL's + * `autovacuum_vacuum_threshold`, not our `floor`, which suppresses instead. + * Past this much reclaimable garbage, compact regardless of proportion, so + * garbage that is large absolutely but small against a big live set still gets + * reclaimed. + * + * WO_CKPT_MAX_GARBAGE caps the proportional term, mirroring + * `autovacuum_vacuum_max_threshold`, so a very large live set cannot defer + * compaction indefinitely. */ +#define WO_CKPT_ABS_BYTES (64u * 1024u * 1024u) +#define WO_CKPT_MAX_GARBAGE (256u * 1024u * 1024u) + int wo_wal_should_compact(uint64_t used, uint64_t last, uint64_t floor, uint32_t ratio); /* Defaults, overridable at boot by WO_CHECKPOINT_BYTES / WO_CHECKPOINT_RATIO. @@ -342,8 +356,25 @@ int wo_wal_read_row_at(wo_wal *w, wo_db *db, wo_rt *rt, uint64_t off, * * 0 ok, -1 no intact/malformed/corrupt record anywhere in the chain (or a * REMOVE tombstone reached mid-chain), -2 out of memory (*msg set). */ +/* databasev2 11: `hops_out` (may be NULL) reports how many DELTA records the + * walk crossed before reaching the full-row record that terminates the chain — + * 0 for a row that has never been updated. The walk already visits each hop, so + * this costs nothing, and it is the signal the update path uses to decide when + * to flatten. It is this design's equivalent of PostgreSQL's `pd_prune_xid`: a + * cheap "is work worth doing" hint obtained from something already being done. */ int wo_wal_fold_row_at(wo_wal *w, wo_db *db, uint64_t off, uint32_t *class_out, - uint64_t *id_out, uint64_t *out_vals, const char **msg); + uint64_t *id_out, uint64_t *out_vals, uint32_t *hops_out, + const char **msg); + +/* databasev2 11: append a FULL-ROW image taken from a caller-supplied row, + * rather than one looked up by id. wo_wal_append_insert sources its values via + * wo_row_ptr, which is NULL for a keys-resident row whose payload has been + * dropped; the update path holds a materialised row and needs to log it as a + * chain-terminating record. Written as WO_WAL_INSERT because that is what a + * chain's base must be: it has to replay into a database where nothing + * precedes it. */ +int wo_wal_append_row_image(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id, + const db_row *r); /* Offline verification (no engine): scan [path], count intact records. * *intact_bytes (optional) = where the intact prefix ends. -1 = open diff --git a/docs/stories/databasev2/11-bounded-delta-chains.md b/docs/stories/databasev2/11-bounded-delta-chains.md index 3dc1694..a060fbe 100644 --- a/docs/stories/databasev2/11-bounded-delta-chains.md +++ b/docs/stories/databasev2/11-bounded-delta-chains.md @@ -1,7 +1,7 @@ --- track: databasev2 iteration: "11" -status: pending +status: in-progress readiness: ready --- @@ -59,9 +59,41 @@ Read from PostgreSQL's source at `.dev/reference/postgresql`, not recalled: operational constraint, not an economic comparison. That ruled out the byte-ratio shape here too. +## Progress + +| Part | State | +| --- | --- | +| Tier 1 — the fold reports hop count | ✅ `wo_wal_fold_row_at` takes `hops_out`; the walk already visited each hop, so it costs nothing | +| Tier 1 — the update branches on depth | ✅ `row_apply_field_keys` writes a full-row image past `WO_DELTA_MAX_HOPS` (16) instead of a delta | +| Tier 1 — the chain-terminating write | ✅ `wo_wal_append_row_image`, encoded as `WO_WAL_INSERT` so replay, compaction and the fold need no change | +| Tier 2 — absolute garbage term | ✅ `WO_CKPT_ABS_BYTES` (64 MiB) triggers regardless of proportion | +| Tier 2 — proportional ceiling | ✅ `WO_CKPT_MAX_GARBAGE` (256 MiB) caps the ratio term | +| **Tests** | ⏸ **DELIBERATELY HELD** — see below | + +**Verified by construction, not by test.** Both update entry points converge on +`row_apply_field_keys` (`table.c:1039` and `:1319`), so one branch covers both. +The re-point is transparent to flattening because `db.c` captures +`wo_wal_next_offset(w)` *before* calling into `table.c` — it targets wherever +the next record lands, delta or full row alike. And a fold that reaches a +flattened record terminates there, so the next update sees depth 0. + +**What holding the tests costs, stated plainly.** The existing suite passes +(36 suites, 0 failures) but that proves only that threading `hops_out` through +the fold, `keys_fold_into` and their callers broke nothing — which is the change +most likely to break something silently, so it is worth having. It does **not** +exercise either new behaviour: + +- No existing test builds a chain 16 deep, so the flatten branch is almost + certainly never executed by the suite. +- Existing checkpoint tests use logs far below 64 MiB, so the two new + compaction terms never fire either. + +A green run here means "did not break what existed", not "works". + ## Acceptance Criteria -Outstanding — none met; this iteration has not started. +Outstanding — none verified, because the tests are held. The logic for every +one of them is implemented; nothing is proven. - **Given** a row updated K times, **when** updated once more, **then** the record its offset names is a full row and its chain length is zero.