feat(db2-delta): wire the request path, defer re-point to the barrier

- db.c: guard both WO_B_DB_UPDATE_FIELD arms on keys-resident tables —
  wo_wal_append_update read a NULL wo_row_ptr there; a live crash, fixed
- ruling override on Task 3: row_apply_field_keys no longer commits or
  moves the id map — stages the delta, does the index swap (RAM apply,
  unconditional past the shadow-check; a stage failure past that point
  is now fatal, like insert). Commit/re-point move to the caller,
  mirroring insert. table.c's WAL commit removal is this ruling, not a
  regression
- offset passed back via caller-side wo_wal_next_offset(), insert's
  koff pattern
- inline arm commits then re-points; request arm records
  wo_wal_pend_repoint (own list/name — a drop and a re-point differ),
  flushed by wo_db_flush_drops after the barrier
- back_off checks the pending re-point before the durable offset, else
  a second update in one drain skips the first delta; verified failing
  this way, passing after
- wal.c: fixed a stale comment — keys-resident updates CAN reach
  wo_wal_append_update's caller now, they just never call it
- test_wal.c: 2 tests updated for the new contract; new test drives 2
  same-row updates via wo_row_update_field_slot in one uncommitted
  "drain", checks the value and delta 2's on-disk back-pointer

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
(cherry picked from commit 4d13bcebfe51936ba8c608dd9e791c784e5b8983)
This commit is contained in:
shoney.arickathil 2026-08-30 09:26:36 +02:00
parent d492d1fefb
commit 3525435eb5
5 changed files with 238 additions and 47 deletions

View file

@ -90,14 +90,28 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
uint64_t id = R[B + 1];
uint32_t field = (uint32_t)R[B + 2];
int ek = 0;
wo_wal *w = (wo_wal *)vm->rt.wal;
int keys_res = wo_table_is_keys_resident(db, cid);
/* databasev2 2 (5c) / Task 4: the delta's own offset, taken BEFORE
* the call the same way the insert arm takes koff — table.c stages
* the delta at exactly this position and nothing else stages bytes
* on `w` in between. */
uint64_t roff = (w && keys_res) ? wo_wal_next_offset(w) : 0;
if (wo_row_update_field(db, cid, id, field, R[B + 3], msg, &ek) != 0)
return ek == DB_ERR_UNIQUE ? WO_T_UNIQUE : ek == DB_ERR_OOM ? WO_T_OOM : WO_T_DB;
wo_wal *w = (wo_wal *)vm->rt.wal;
if (w && table_is_durable(db, cid)) {
/* was: trap and leave RAM ahead of disk, which the old comment
* admitted. Now fatal — see the insert arm. */
if (wo_wal_append_update(w, db, cid, id) != 0) wo_wal_stage_fatal(w);
wo_wal_commit_fatal(w, 1);
if (keys_res) {
/* the delta is already staged (table.c); this is the
* inline path's OWN barrier, same as insert, then the map
* moves — commit before re-point, always. */
wo_wal_commit_fatal(w, 1);
(void)wo_row_set_offset(db, cid, id, roff);
} else {
/* was: trap and leave RAM ahead of disk, which the old
* comment admitted. Now fatal — see the insert arm. */
if (wo_wal_append_update(w, db, cid, id) != 0) wo_wal_stage_fatal(w);
wo_wal_commit_fatal(w, 1);
}
maybe_compact(db, w);
}
R[A] = 0;
@ -288,13 +302,24 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) {
}
case WO_B_DB_UPDATE_FIELD: {
int ek = 0;
int keys_res = wo_table_is_keys_resident(db, q->cid);
/* Task 4: see the inline arm — the delta's own offset, captured
* BEFORE the call the same way insert's koff is. */
uint64_t roff = (w && keys_res) ? wo_wal_next_offset(w) : 0;
if (wo_row_update_field_slot(db, q->cid, q->id, q->field, q->slots[0], &m, &ek) != 0) {
q->status = ek == DB_ERR_UNIQUE ? WO_T_UNIQUE : ek == DB_ERR_OOM ? WO_T_OOM : WO_T_DB;
q->msg = m;
break;
}
if (w && table_is_durable(db, q->cid)) {
if (wo_wal_append_update(w, db, q->cid, q->id) != 0) wo_wal_stage_fatal(w);
if (keys_res) {
/* recorded, not performed: this batch's barrier runs in the
* drain (vm.c), and only then does the map move — mirrors
* the insert arm's wo_wal_pend_drop exactly. */
(void)wo_wal_pend_repoint(w, q->cid, q->id, roff);
} else {
if (wo_wal_append_update(w, db, q->cid, q->id) != 0) wo_wal_stage_fatal(w);
}
}
break;
}

View file

@ -1134,23 +1134,37 @@ static int row_apply_field_slot(wo_db *db, db_table *t, const wo_classdesc *c,
/* keys-resident counterpart of row_apply_field_slot. There is no slab slot
* to swap — the row lives in the log — so the shape is read-modify-APPEND:
* borrow (folds), append a delta with the row's current offset as the
* back-pointer, commit, THEN move the id map to the new record. [nv] is
* already engine-encoded (same convention as row_apply_field_slot);
* consumed on every path.
* back-pointer. [nv] is already engine-encoded (same convention as
* row_apply_field_slot); consumed on every path.
*
* The borrow's materialised row holds VM values (wo_row_release drops every
* slot through the runtime), so [nv] is decoded to a VM value up front and
* that is what ever lands in r->slots[field] — the engine encoding is used
* only for the WAL record and freed once logged.
* only for the WAL record and freed once staged.
*
* Task 4 (keys-resident delta updates) ruling: this function stages the
* delta but does NOT commit and does NOT move the id map — mirroring
* insert, where table.c applies RAM and db.c owns staging/commit and the
* post-barrier map move (wo_wal_pend_drop / wo_db_flush_drops for insert;
* wo_wal_pend_repoint / wo_db_flush_drops for this). The caller re-points
* using the offset it captured via wo_wal_next_offset() BEFORE calling in
* here — insert's own `koff` pattern — since nothing between that capture
* and the wo_wal_append_delta call below stages any other bytes on [w].
*
* Ordering: the unique shadow-check (against a shadow of the row, mirroring
* row_apply_field_slot's promise that a rejected update leaves the row
* untouched) and the WAL append+commit both happen BEFORE either the index
* or the id map move — so a failure at any point up to and including the
* commit leaves the live row (offset AND index) exactly as it was. Only a
* successful, durable commit is followed by the index swap and the offset
* repoint, which — being pure RAM bookkeeping a replay rebuilds from the log
* regardless — cannot itself meaningfully "fail" once reached. */
* untouched) is the only SOFT-trap gate and runs first, before anything
* moves. Once it passes, the index swap is RAM apply and happens
* unconditionally, mirroring wo_row_insert's doctrine order (RAM, then
* log) — from that point a failure to even STAGE the delta is fatal,
* exactly like insert's own append, because RAM has already moved and
* there is no undo.
*
* back_off checks a PENDING re-point first (wo_wal_repoint_offset1) before
* falling back to the durable wo_row_offset1: a second update to this same
* row, staged behind the same barrier as a first, must chain to the
* first's delta — the id map won't move until the barrier, but the delta
* itself is already staged and its offset already fixed. */
static int row_apply_field_keys(wo_db *db, uint32_t class_id, uint64_t id,
uint32_t field, uint64_t nv, const char **msg,
int *err_kind) {
@ -1164,7 +1178,9 @@ static int row_apply_field_keys(wo_db *db, uint32_t class_id, uint64_t id,
return -1;
}
/* successful borrow proves db->rt and db->rt->wal are both set */
uint64_t back_off = wo_row_offset1(db, class_id, id) - 1;
wo_wal *w = (wo_wal *)db->rt->wal;
uint64_t pending1 = wo_wal_repoint_offset1(w, class_id, id);
uint64_t back_off = (pending1 ? pending1 : wo_row_offset1(db, class_id, id)) - 1;
int ok = 1;
uint64_t nv_vm = wo_val_decode_vm(db, db->rt, c->kinds[field], nv, &ok, msg);
@ -1235,34 +1251,21 @@ static int row_apply_field_keys(wo_db *db, uint32_t class_id, uint64_t id,
}
}
free(cand_buf);
r->slots[field] = old_vm; /* restored: still the OLD row until committed */
r->slots[field] = old_vm; /* restored: still the OLD row until applied */
wo_wal *w = (wo_wal *)db->rt->wal;
uint64_t new_off = wo_wal_next_offset(w);
if (wo_wal_append_delta(w, db, class_id, id, field, back_off, nv) != 0) {
wo_row_release(db, class_id, r);
wo_drop_kind(db->rt, c->kinds[field], nv_vm);
db_val_free(c->kinds[field], nv);
if (err_kind) *err_kind = DB_ERR_OOM;
*msg = "out of memory appending delta";
return -1;
}
if (wo_wal_commit(w) != 0) {
wo_row_release(db, class_id, r);
wo_drop_kind(db->rt, c->kinds[field], nv_vm);
db_val_free(c->kinds[field], nv);
*msg = "wal commit failed";
return -1;
}
db_val_free(c->kinds[field], nv); /* durable now; the engine copy served the log */
/* commit: the borrowed row is the OLD row — out of every index under the
OLD value, then in again under the NEW one — then the id map moves */
/* RAM apply (Task 4 ruling): the borrowed row is the OLD row — out of
every index under the OLD value, then in again under the NEW one.
Unconditional from here: a failure below is fatal, not a trap. */
idx_remove_row(db, t, r);
r->slots[field] = nv_vm;
(void)idx_add_row(db, t, r); /* cannot violate uniqueness: the shadow
check above already cleared it */
wo_row_set_offset(db, class_id, id, new_off);
if (wo_wal_append_delta(w, db, class_id, id, field, back_off, nv) != 0)
wo_wal_stage_fatal(w); /* RAM already moved; see the insert arm */
db_val_free(c->kinds[field], nv); /* staged now; the engine copy served
the log record */
wo_drop_kind(db->rt, c->kinds[field], old_vm);
wo_row_release(db, class_id, r);
if (err_kind) *err_kind = DB_ERR_NONE;

View file

@ -345,14 +345,48 @@ int wo_wal_pend_drop(wo_wal *w, uint32_t cid, uint64_t id, uint64_t off) {
return 0;
}
/* Task 4 (keys-resident delta updates): pend_drop's counterpart for a
* re-point, own list, same reason (see the `repoint` field in wal.h). */
int wo_wal_pend_repoint(wo_wal *w, uint32_t cid, uint64_t id, uint64_t off) {
if (w->repoint_len == w->repoint_cap) {
size_t nc = w->repoint_cap ? w->repoint_cap * 2 : 16;
struct wo_wal_pend *np = realloc(w->repoint, nc * sizeof *np);
if (!np) return -1; /* the map stays where it was: see the header */
w->repoint = np;
w->repoint_cap = nc;
}
w->repoint[w->repoint_len].cid = cid;
w->repoint[w->repoint_len].id = id;
w->repoint[w->repoint_len].off = off;
w->repoint_len++;
return 0;
}
uint64_t wo_wal_repoint_offset1(const wo_wal *w, uint32_t cid, uint64_t id) {
/* backward: the LATEST entry for (cid, id) is the one still current if
this row was updated more than once behind the same barrier */
for (size_t i = w->repoint_len; i > 0; i--) {
if (w->repoint[i - 1].cid == cid && w->repoint[i - 1].id == id)
return w->repoint[i - 1].off + 1;
}
return 0;
}
void wo_db_flush_drops(wo_db *db, wo_wal *w) {
for (size_t i = 0; i < w->pend_len; i++)
(void)wo_row_drop_payload(db, w->pend[i].cid, w->pend[i].id, w->pend[i].off);
w->pend_len = 0;
/* Task 4: the request path's deferred update re-points, held behind the
same barrier as inserts' drops for the same reason — before the
commit above, these offsets pread zeros. */
for (size_t i = 0; i < w->repoint_len; i++)
(void)wo_row_set_offset(db, w->repoint[i].cid, w->repoint[i].id, w->repoint[i].off);
w->repoint_len = 0;
}
void wo_wal_close(wo_wal *w) {
free(w->pend);
free(w->repoint);
if (w->fd >= 0) close(w->fd);
free(w->path);
free(w->buf);
@ -410,9 +444,12 @@ int wo_wal_append_insert(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id) {
}
int wo_wal_append_update(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id) {
/* wo_row_ptr is right here for the same reason as append_insert, and the
* keys case cannot arrive at all: wo_row_update_field{,_slot} refuse a
* keys-resident table before any log record is staged. */
/* wo_row_ptr is right here for the same reason as append_insert. A
* keys-resident update CAN succeed now (Task 3's row_apply_field_keys),
* but this function never runs for one: db.c routes a keys-resident
* update to a staged DELTA record instead (row_apply_field_keys), and
* never re-logs the whole row. This function stays reachable only for
* resident: all, guarded at both db.c call sites. */
db_row *r = wo_row_ptr(db, class_id, id);
if (!r) return -1;
wbuf p = {0};

View file

@ -88,6 +88,16 @@ typedef struct wo_wal {
* dies with it, which is correct: nothing was dropped and nothing lost. */
struct wo_wal_pend { uint32_t cid; uint64_t id; uint64_t off; } *pend;
size_t pend_len, pend_cap;
/* Task 4 (keys-resident delta updates): rows whose id-map entry must
* move to a NEW offset once the delta staged there is durable. Same
* three fields as `pend` above, deliberately its OWN list: a drop
* discards a payload and a re-point moves a live row's chain head — two
* different meanings a shared list would force a future reader to guess
* between. Same lifetime discipline as `pend`: recorded before the
* barrier, applied after it, and lost with the process if it dies
* first — which is correct, since nothing was re-pointed either. */
struct wo_wal_pend *repoint;
size_t repoint_len, repoint_cap;
uint64_t stat_compactions;
uint64_t stat_compact_us_max;
uint64_t stat_compact_us_total;
@ -162,8 +172,25 @@ int wo_wal_commit(wo_wal *w);
* 0 ok, -1 out of memory (the row simply stays resident, which is safe). */
int wo_wal_pend_drop(wo_wal *w, uint32_t cid, uint64_t id, uint64_t off);
/* databasev2 2 (5c): perform every pending drop. Call ONLY after a commit has
* succeeded — that is what makes the recorded offsets readable. */
/* Task 4 (keys-resident delta updates): note a keys-resident row's id-map
* entry that must move to [off] once the delta staged there is durable —
* the update-arm counterpart of wo_wal_pend_drop, on its own list (see the
* `repoint` field). 0 ok, -1 out of memory (the map simply stays where it
* was; a durable delta with a stale map is exactly what replay reconciles,
* so this is safe, just deferred further than intended). */
int wo_wal_pend_repoint(wo_wal *w, uint32_t cid, uint64_t id, uint64_t off);
/* Task 4: the most recent PENDING re-point recorded for (cid, id), not yet
* flushed to the id map — needed so a second update to the same row, staged
* behind the SAME barrier as the first, computes its back-pointer against
* the first's delta instead of the row's last DURABLE offset (which would
* skip it). Off + 1, 0 = none pending (the caller falls back to
* wo_row_offset1). Does NOT consult the durable map itself. */
uint64_t wo_wal_repoint_offset1(const wo_wal *w, uint32_t cid, uint64_t id);
/* databasev2 2 (5c): perform every pending drop, THEN every pending
* re-point (Task 4). Call ONLY after a commit has succeeded — that is what
* makes the recorded offsets readable. */
void wo_db_flush_drops(wo_db *db, wo_wal *w);
/* databasev2 3: the checkpoint trigger, as a PURE decision so it can be tested

View file

@ -859,7 +859,13 @@ static void test_fold_refuses_forward_pointing_delta(void) {
/* Task 3 (keys-resident delta updates): the plain case, through the real
* API — wo_row_update_field, not a hand-rolled append+commit+set_offset like
* the fold tests above. Before this task it refused outright with "update on
* a `resident: keys` table is not implemented". */
* a `resident: keys` table is not implemented".
*
* Task 4 ruling: wo_row_update_field now only STAGES the delta and applies
* the index swap — it does not commit and does not move the id map (that
* mirrors insert, whose koff/commit/pend_drop live in the CALLER). So this
* test now does the caller's half itself, exactly as db.c's inline arm
* does: capture the offset before calling in, commit, then re-point. */
static void test_keys_resident_update_field(void) {
char path[128];
snprintf(path, sizeof path, "%s/keysupd.wal", g_dir);
@ -882,8 +888,11 @@ static void test_keys_resident_update_field(void) {
T_EQ(wo_row_drop_payload(&db, 0, id, off), 0);
int ek = 0;
uint64_t roff = wo_wal_next_offset(&w);
T_EQ(wo_row_update_field(&db, 0, id, 0, 999, &msg, &ek), 0);
T_EQ(ek, DB_ERR_NONE);
T_EQ(wo_wal_commit(&w), 0);
T_EQ(wo_row_set_offset(&db, 0, id, roff), 0);
db_row *r = wo_row_borrow(&db, 0, id, &msg);
T_CHECK(r != NULL);
@ -918,7 +927,12 @@ static const wo_classdesc KEYS_IDX_CLASSES[] = {
/* Task 3, the test that matters: updating an INDEXED column on a
* keys-resident row must move the row in the index too, not just in the
* log — queried through wo_idx_probe, the row is found by its NEW value and
* gone from its OLD one. */
* gone from its OLD one.
*
* Task 4 ruling: wo_idx_probe's bucket hit is verified by folding the row
* from the log (table.c's idx_cols_equal path), so the probes below must
* run AFTER the caller's commit + re-point — mirroring db.c's inline arm —
* not straight after wo_row_update_field, which now only stages. */
static void test_keys_resident_update_indexed(void) {
char path[128];
snprintf(path, sizeof path, "%s/keysidx.wal", g_dir);
@ -955,8 +969,11 @@ static void test_keys_resident_update_indexed(void) {
free(ids);
int ek = 0;
uint64_t roff = wo_wal_next_offset(&w);
T_EQ(wo_row_update_field(&db, 0, a, 0, 150, &msg, &ek), 0);
T_EQ(ek, DB_ERR_NONE);
T_EQ(wo_wal_commit(&w), 0);
T_EQ(wo_row_set_offset(&db, 0, a, roff), 0);
/* found by the NEW value */
T_EQ(wo_idx_probe(&db, 0, 0, 150, NULL, 0, &ids, &cnt), 1);
@ -1051,6 +1068,87 @@ static void test_keys_resident_update_unique_violation_refused(void) {
wo_rt_destroy(&rt);
}
/* Task 4 (keys-resident delta updates): the request path's group-commit
* shape, the test the brief asked for. Two updates to the SAME row through
* wo_row_update_field_slot (the request-path entry point) with NEITHER
* wo_wal_commit NOR the re-point called in between — exactly two requests
* landing in the SAME drain before its one barrier. The re-point for each
* is only RECORDED (wo_wal_pend_repoint), mirroring db.c's request arm;
* the barrier commits once, then wo_db_flush_drops applies both.
*
* The failure this catches: a back_off read straight off the (still stale,
* pre-barrier) durable map would have the second delta name the FIRST
* request's insert-time offset instead of the first delta — skipping it.
* Checked two ways: the final value must reflect BOTH updates in order,
* and delta 2's back-pointer, read straight off disk, must equal delta 1's
* own offset, not the base insert's. */
static void test_keys_resident_two_updates_one_drain(void) {
char path[128];
snprintf(path, sizeof path, "%s/keys2upd.wal", g_dir);
wo_rt rt;
T_EQ(wo_rt_init(&rt, 1 << 20, KEYS_CLASSES, 1), 0);
wo_db db;
T_EQ(wo_db_init(&db, KEYS_CLASSES, 1, 0, 1), 0);
wo_wal w;
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
db.rt = &rt; rt.wal = &w; rt.db = &db;
const char *msg = "";
wo_str *s = wo_str_new(&rt, "sku", 3);
uint64_t vals[2] = {111, (uint64_t)(uintptr_t)s};
uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL);
T_CHECK(id != 0);
uint64_t base_off = wo_wal_next_offset(&w);
T_EQ(wo_wal_append_insert(&w, &db, 0, id), 0);
T_EQ(wo_wal_commit(&w), 0);
T_EQ(wo_row_drop_payload(&db, 0, id, base_off), 0);
/* "request" 1: field 0, 111 -> 222 — staged, NOT committed, the map
NOT moved (only recorded as pending) */
int ek = 0;
uint64_t roff1 = wo_wal_next_offset(&w);
T_EQ(wo_row_update_field_slot(&db, 0, id, 0, 222, &msg, &ek), 0);
T_EQ(ek, DB_ERR_NONE);
T_EQ(wo_wal_pend_repoint(&w, 0, id, roff1), 0);
/* "request" 2, SAME drain: field 0, 222 -> 333. The id map still names
the base insert (the re-point above is only PENDING) — back_off must
come from the pending list, not wo_row_offset1, or this chains to
the wrong predecessor. */
uint64_t roff2 = wo_wal_next_offset(&w);
T_EQ(wo_row_update_field_slot(&db, 0, id, 0, 333, &msg, &ek), 0);
T_EQ(ek, DB_ERR_NONE);
T_EQ(wo_wal_pend_repoint(&w, 0, id, roff2), 0);
/* the drain's barrier: ONE commit for both staged deltas, then both
pending re-points applied — db.c/vm.c's exact shape */
T_EQ(wo_wal_commit(&w), 0);
wo_db_flush_drops(&db, &w);
/* both updates visible, in order */
db_row *r = wo_row_borrow(&db, 0, id, &msg);
T_CHECK(r != NULL && r->slots[0] == 333);
wo_str *back = (wo_str *)(uintptr_t)r->slots[1];
T_CHECK(back != NULL && back->len == 3 && memcmp(back->data, "sku", 3) == 0);
wo_row_release(&db, 0, r);
/* the chain itself: delta 2's back-pointer names delta 1's OWN offset,
not the base insert's — payload layout established by
test_delta_record (kind|class|id|field_idx|back_off|value, 33 bytes
for a scalar field) */
uint8_t body[33];
T_EQ((int)pread(w.fd, body, 33, (off_t)(roff2 + 8)), 33);
T_EQ(body[0], WO_WAL_DELTA);
uint64_t back_off;
memcpy(&back_off, body + 17, 8);
T_EQ(back_off, roff1);
T_CHECK(back_off != base_off); /* the skip this test exists to catch */
wo_wal_close(&w);
wo_db_destroy(&db);
wo_rt_destroy(&rt);
}
/* databasev2 2 (5c): boot. A keys-resident store must come back from replay
* with its rows readable FROM THE LOG — the map rebuilt to offsets, not slabs.
* This is the half the round-trip test cannot cover: it runs in a fresh db,
@ -1702,6 +1800,7 @@ int main(void) {
test_keys_resident_update_field();
test_keys_resident_update_indexed();
test_keys_resident_update_unique_violation_refused();
test_keys_resident_two_updates_one_drain();
test_keys_resident_replay();
test_keys_resident_survives_compaction();
test_keys_resident_delete();