diff --git a/database/src/db.c b/database/src/db.c index f46aaff..6101fa5 100644 --- a/database/src/db.c +++ b/database/src/db.c @@ -90,14 +90,28 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { uint64_t id = R[B + 1]; uint32_t field = (uint32_t)R[B + 2]; int ek = 0; + wo_wal *w = (wo_wal *)vm->rt.wal; + int keys_res = wo_table_is_keys_resident(db, cid); + /* databasev2 2 (5c) / Task 4: the delta's own offset, taken BEFORE + * the call the same way the insert arm takes koff — table.c stages + * the delta at exactly this position and nothing else stages bytes + * on `w` in between. */ + uint64_t roff = (w && keys_res) ? wo_wal_next_offset(w) : 0; if (wo_row_update_field(db, cid, id, field, R[B + 3], msg, &ek) != 0) return ek == DB_ERR_UNIQUE ? WO_T_UNIQUE : ek == DB_ERR_OOM ? WO_T_OOM : WO_T_DB; - wo_wal *w = (wo_wal *)vm->rt.wal; if (w && table_is_durable(db, cid)) { - /* was: trap and leave RAM ahead of disk, which the old comment - * admitted. Now fatal — see the insert arm. */ - if (wo_wal_append_update(w, db, cid, id) != 0) wo_wal_stage_fatal(w); - wo_wal_commit_fatal(w, 1); + if (keys_res) { + /* the delta is already staged (table.c); this is the + * inline path's OWN barrier, same as insert, then the map + * moves — commit before re-point, always. */ + wo_wal_commit_fatal(w, 1); + (void)wo_row_set_offset(db, cid, id, roff); + } else { + /* was: trap and leave RAM ahead of disk, which the old + * comment admitted. Now fatal — see the insert arm. */ + if (wo_wal_append_update(w, db, cid, id) != 0) wo_wal_stage_fatal(w); + wo_wal_commit_fatal(w, 1); + } maybe_compact(db, w); } R[A] = 0; @@ -288,13 +302,24 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) { } case WO_B_DB_UPDATE_FIELD: { int ek = 0; + int keys_res = wo_table_is_keys_resident(db, q->cid); + /* Task 4: see the inline arm — the delta's own offset, captured + * BEFORE the call the same way insert's koff is. */ + uint64_t roff = (w && keys_res) ? wo_wal_next_offset(w) : 0; if (wo_row_update_field_slot(db, q->cid, q->id, q->field, q->slots[0], &m, &ek) != 0) { q->status = ek == DB_ERR_UNIQUE ? WO_T_UNIQUE : ek == DB_ERR_OOM ? WO_T_OOM : WO_T_DB; q->msg = m; break; } if (w && table_is_durable(db, q->cid)) { - if (wo_wal_append_update(w, db, q->cid, q->id) != 0) wo_wal_stage_fatal(w); + if (keys_res) { + /* recorded, not performed: this batch's barrier runs in the + * drain (vm.c), and only then does the map move — mirrors + * the insert arm's wo_wal_pend_drop exactly. */ + (void)wo_wal_pend_repoint(w, q->cid, q->id, roff); + } else { + if (wo_wal_append_update(w, db, q->cid, q->id) != 0) wo_wal_stage_fatal(w); + } } break; } diff --git a/database/src/table.c b/database/src/table.c index bcd1408..a098263 100644 --- a/database/src/table.c +++ b/database/src/table.c @@ -1134,23 +1134,37 @@ static int row_apply_field_slot(wo_db *db, db_table *t, const wo_classdesc *c, /* keys-resident counterpart of row_apply_field_slot. There is no slab slot * to swap — the row lives in the log — so the shape is read-modify-APPEND: * borrow (folds), append a delta with the row's current offset as the - * back-pointer, commit, THEN move the id map to the new record. [nv] is - * already engine-encoded (same convention as row_apply_field_slot); - * consumed on every path. + * back-pointer. [nv] is already engine-encoded (same convention as + * row_apply_field_slot); consumed on every path. * * The borrow's materialised row holds VM values (wo_row_release drops every * slot through the runtime), so [nv] is decoded to a VM value up front and * that is what ever lands in r->slots[field] — the engine encoding is used - * only for the WAL record and freed once logged. + * only for the WAL record and freed once staged. + * + * Task 4 (keys-resident delta updates) ruling: this function stages the + * delta but does NOT commit and does NOT move the id map — mirroring + * insert, where table.c applies RAM and db.c owns staging/commit and the + * post-barrier map move (wo_wal_pend_drop / wo_db_flush_drops for insert; + * wo_wal_pend_repoint / wo_db_flush_drops for this). The caller re-points + * using the offset it captured via wo_wal_next_offset() BEFORE calling in + * here — insert's own `koff` pattern — since nothing between that capture + * and the wo_wal_append_delta call below stages any other bytes on [w]. * * Ordering: the unique shadow-check (against a shadow of the row, mirroring * row_apply_field_slot's promise that a rejected update leaves the row - * untouched) and the WAL append+commit both happen BEFORE either the index - * or the id map move — so a failure at any point up to and including the - * commit leaves the live row (offset AND index) exactly as it was. Only a - * successful, durable commit is followed by the index swap and the offset - * repoint, which — being pure RAM bookkeeping a replay rebuilds from the log - * regardless — cannot itself meaningfully "fail" once reached. */ + * untouched) is the only SOFT-trap gate and runs first, before anything + * moves. Once it passes, the index swap is RAM apply and happens + * unconditionally, mirroring wo_row_insert's doctrine order (RAM, then + * log) — from that point a failure to even STAGE the delta is fatal, + * exactly like insert's own append, because RAM has already moved and + * there is no undo. + * + * back_off checks a PENDING re-point first (wo_wal_repoint_offset1) before + * falling back to the durable wo_row_offset1: a second update to this same + * row, staged behind the same barrier as a first, must chain to the + * first's delta — the id map won't move until the barrier, but the delta + * itself is already staged and its offset already fixed. */ static int row_apply_field_keys(wo_db *db, uint32_t class_id, uint64_t id, uint32_t field, uint64_t nv, const char **msg, int *err_kind) { @@ -1164,7 +1178,9 @@ static int row_apply_field_keys(wo_db *db, uint32_t class_id, uint64_t id, return -1; } /* successful borrow proves db->rt and db->rt->wal are both set */ - uint64_t back_off = wo_row_offset1(db, class_id, id) - 1; + wo_wal *w = (wo_wal *)db->rt->wal; + uint64_t pending1 = wo_wal_repoint_offset1(w, class_id, id); + uint64_t back_off = (pending1 ? pending1 : wo_row_offset1(db, class_id, id)) - 1; int ok = 1; uint64_t nv_vm = wo_val_decode_vm(db, db->rt, c->kinds[field], nv, &ok, msg); @@ -1235,34 +1251,21 @@ static int row_apply_field_keys(wo_db *db, uint32_t class_id, uint64_t id, } } free(cand_buf); - r->slots[field] = old_vm; /* restored: still the OLD row until committed */ + r->slots[field] = old_vm; /* restored: still the OLD row until applied */ - wo_wal *w = (wo_wal *)db->rt->wal; - uint64_t new_off = wo_wal_next_offset(w); - if (wo_wal_append_delta(w, db, class_id, id, field, back_off, nv) != 0) { - wo_row_release(db, class_id, r); - wo_drop_kind(db->rt, c->kinds[field], nv_vm); - db_val_free(c->kinds[field], nv); - if (err_kind) *err_kind = DB_ERR_OOM; - *msg = "out of memory appending delta"; - return -1; - } - if (wo_wal_commit(w) != 0) { - wo_row_release(db, class_id, r); - wo_drop_kind(db->rt, c->kinds[field], nv_vm); - db_val_free(c->kinds[field], nv); - *msg = "wal commit failed"; - return -1; - } - db_val_free(c->kinds[field], nv); /* durable now; the engine copy served the log */ - - /* commit: the borrowed row is the OLD row — out of every index under the - OLD value, then in again under the NEW one — then the id map moves */ + /* RAM apply (Task 4 ruling): the borrowed row is the OLD row — out of + every index under the OLD value, then in again under the NEW one. + Unconditional from here: a failure below is fatal, not a trap. */ idx_remove_row(db, t, r); r->slots[field] = nv_vm; (void)idx_add_row(db, t, r); /* cannot violate uniqueness: the shadow check above already cleared it */ - wo_row_set_offset(db, class_id, id, new_off); + + if (wo_wal_append_delta(w, db, class_id, id, field, back_off, nv) != 0) + wo_wal_stage_fatal(w); /* RAM already moved; see the insert arm */ + db_val_free(c->kinds[field], nv); /* staged now; the engine copy served + the log record */ + wo_drop_kind(db->rt, c->kinds[field], old_vm); wo_row_release(db, class_id, r); if (err_kind) *err_kind = DB_ERR_NONE; diff --git a/database/src/wal.c b/database/src/wal.c index 6abaa07..59bca47 100644 --- a/database/src/wal.c +++ b/database/src/wal.c @@ -345,14 +345,48 @@ int wo_wal_pend_drop(wo_wal *w, uint32_t cid, uint64_t id, uint64_t off) { return 0; } +/* Task 4 (keys-resident delta updates): pend_drop's counterpart for a + * re-point, own list, same reason (see the `repoint` field in wal.h). */ +int wo_wal_pend_repoint(wo_wal *w, uint32_t cid, uint64_t id, uint64_t off) { + if (w->repoint_len == w->repoint_cap) { + size_t nc = w->repoint_cap ? w->repoint_cap * 2 : 16; + struct wo_wal_pend *np = realloc(w->repoint, nc * sizeof *np); + if (!np) return -1; /* the map stays where it was: see the header */ + w->repoint = np; + w->repoint_cap = nc; + } + w->repoint[w->repoint_len].cid = cid; + w->repoint[w->repoint_len].id = id; + w->repoint[w->repoint_len].off = off; + w->repoint_len++; + return 0; +} + +uint64_t wo_wal_repoint_offset1(const wo_wal *w, uint32_t cid, uint64_t id) { + /* backward: the LATEST entry for (cid, id) is the one still current if + this row was updated more than once behind the same barrier */ + for (size_t i = w->repoint_len; i > 0; i--) { + if (w->repoint[i - 1].cid == cid && w->repoint[i - 1].id == id) + return w->repoint[i - 1].off + 1; + } + return 0; +} + void wo_db_flush_drops(wo_db *db, wo_wal *w) { for (size_t i = 0; i < w->pend_len; i++) (void)wo_row_drop_payload(db, w->pend[i].cid, w->pend[i].id, w->pend[i].off); w->pend_len = 0; + /* Task 4: the request path's deferred update re-points, held behind the + same barrier as inserts' drops for the same reason — before the + commit above, these offsets pread zeros. */ + for (size_t i = 0; i < w->repoint_len; i++) + (void)wo_row_set_offset(db, w->repoint[i].cid, w->repoint[i].id, w->repoint[i].off); + w->repoint_len = 0; } void wo_wal_close(wo_wal *w) { free(w->pend); + free(w->repoint); if (w->fd >= 0) close(w->fd); free(w->path); free(w->buf); @@ -410,9 +444,12 @@ int wo_wal_append_insert(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id) { } int wo_wal_append_update(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id) { - /* wo_row_ptr is right here for the same reason as append_insert, and the - * keys case cannot arrive at all: wo_row_update_field{,_slot} refuse a - * keys-resident table before any log record is staged. */ + /* wo_row_ptr is right here for the same reason as append_insert. A + * keys-resident update CAN succeed now (Task 3's row_apply_field_keys), + * but this function never runs for one: db.c routes a keys-resident + * update to a staged DELTA record instead (row_apply_field_keys), and + * never re-logs the whole row. This function stays reachable only for + * resident: all, guarded at both db.c call sites. */ db_row *r = wo_row_ptr(db, class_id, id); if (!r) return -1; wbuf p = {0}; diff --git a/database/src/wal.h b/database/src/wal.h index 0d7e694..840b136 100644 --- a/database/src/wal.h +++ b/database/src/wal.h @@ -88,6 +88,16 @@ typedef struct wo_wal { * dies with it, which is correct: nothing was dropped and nothing lost. */ struct wo_wal_pend { uint32_t cid; uint64_t id; uint64_t off; } *pend; size_t pend_len, pend_cap; + /* Task 4 (keys-resident delta updates): rows whose id-map entry must + * move to a NEW offset once the delta staged there is durable. Same + * three fields as `pend` above, deliberately its OWN list: a drop + * discards a payload and a re-point moves a live row's chain head — two + * different meanings a shared list would force a future reader to guess + * between. Same lifetime discipline as `pend`: recorded before the + * barrier, applied after it, and lost with the process if it dies + * first — which is correct, since nothing was re-pointed either. */ + struct wo_wal_pend *repoint; + size_t repoint_len, repoint_cap; uint64_t stat_compactions; uint64_t stat_compact_us_max; uint64_t stat_compact_us_total; @@ -162,8 +172,25 @@ int wo_wal_commit(wo_wal *w); * 0 ok, -1 out of memory (the row simply stays resident, which is safe). */ int wo_wal_pend_drop(wo_wal *w, uint32_t cid, uint64_t id, uint64_t off); -/* databasev2 2 (5c): perform every pending drop. Call ONLY after a commit has - * succeeded — that is what makes the recorded offsets readable. */ +/* Task 4 (keys-resident delta updates): note a keys-resident row's id-map + * entry that must move to [off] once the delta staged there is durable — + * the update-arm counterpart of wo_wal_pend_drop, on its own list (see the + * `repoint` field). 0 ok, -1 out of memory (the map simply stays where it + * was; a durable delta with a stale map is exactly what replay reconciles, + * so this is safe, just deferred further than intended). */ +int wo_wal_pend_repoint(wo_wal *w, uint32_t cid, uint64_t id, uint64_t off); + +/* Task 4: the most recent PENDING re-point recorded for (cid, id), not yet + * flushed to the id map — needed so a second update to the same row, staged + * behind the SAME barrier as the first, computes its back-pointer against + * the first's delta instead of the row's last DURABLE offset (which would + * skip it). Off + 1, 0 = none pending (the caller falls back to + * wo_row_offset1). Does NOT consult the durable map itself. */ +uint64_t wo_wal_repoint_offset1(const wo_wal *w, uint32_t cid, uint64_t id); + +/* databasev2 2 (5c): perform every pending drop, THEN every pending + * re-point (Task 4). Call ONLY after a commit has succeeded — that is what + * makes the recorded offsets readable. */ void wo_db_flush_drops(wo_db *db, wo_wal *w); /* databasev2 3: the checkpoint trigger, as a PURE decision so it can be tested diff --git a/runtime/test/test_wal.c b/runtime/test/test_wal.c index a05a5df..a23441b 100644 --- a/runtime/test/test_wal.c +++ b/runtime/test/test_wal.c @@ -859,7 +859,13 @@ static void test_fold_refuses_forward_pointing_delta(void) { /* Task 3 (keys-resident delta updates): the plain case, through the real * API — wo_row_update_field, not a hand-rolled append+commit+set_offset like * the fold tests above. Before this task it refused outright with "update on - * a `resident: keys` table is not implemented". */ + * a `resident: keys` table is not implemented". + * + * Task 4 ruling: wo_row_update_field now only STAGES the delta and applies + * the index swap — it does not commit and does not move the id map (that + * mirrors insert, whose koff/commit/pend_drop live in the CALLER). So this + * test now does the caller's half itself, exactly as db.c's inline arm + * does: capture the offset before calling in, commit, then re-point. */ static void test_keys_resident_update_field(void) { char path[128]; snprintf(path, sizeof path, "%s/keysupd.wal", g_dir); @@ -882,8 +888,11 @@ static void test_keys_resident_update_field(void) { T_EQ(wo_row_drop_payload(&db, 0, id, off), 0); int ek = 0; + uint64_t roff = wo_wal_next_offset(&w); T_EQ(wo_row_update_field(&db, 0, id, 0, 999, &msg, &ek), 0); T_EQ(ek, DB_ERR_NONE); + T_EQ(wo_wal_commit(&w), 0); + T_EQ(wo_row_set_offset(&db, 0, id, roff), 0); db_row *r = wo_row_borrow(&db, 0, id, &msg); T_CHECK(r != NULL); @@ -918,7 +927,12 @@ static const wo_classdesc KEYS_IDX_CLASSES[] = { /* Task 3, the test that matters: updating an INDEXED column on a * keys-resident row must move the row in the index too, not just in the * log — queried through wo_idx_probe, the row is found by its NEW value and - * gone from its OLD one. */ + * gone from its OLD one. + * + * Task 4 ruling: wo_idx_probe's bucket hit is verified by folding the row + * from the log (table.c's idx_cols_equal path), so the probes below must + * run AFTER the caller's commit + re-point — mirroring db.c's inline arm — + * not straight after wo_row_update_field, which now only stages. */ static void test_keys_resident_update_indexed(void) { char path[128]; snprintf(path, sizeof path, "%s/keysidx.wal", g_dir); @@ -955,8 +969,11 @@ static void test_keys_resident_update_indexed(void) { free(ids); int ek = 0; + uint64_t roff = wo_wal_next_offset(&w); T_EQ(wo_row_update_field(&db, 0, a, 0, 150, &msg, &ek), 0); T_EQ(ek, DB_ERR_NONE); + T_EQ(wo_wal_commit(&w), 0); + T_EQ(wo_row_set_offset(&db, 0, a, roff), 0); /* found by the NEW value */ T_EQ(wo_idx_probe(&db, 0, 0, 150, NULL, 0, &ids, &cnt), 1); @@ -1051,6 +1068,87 @@ static void test_keys_resident_update_unique_violation_refused(void) { wo_rt_destroy(&rt); } +/* Task 4 (keys-resident delta updates): the request path's group-commit + * shape, the test the brief asked for. Two updates to the SAME row through + * wo_row_update_field_slot (the request-path entry point) with NEITHER + * wo_wal_commit NOR the re-point called in between — exactly two requests + * landing in the SAME drain before its one barrier. The re-point for each + * is only RECORDED (wo_wal_pend_repoint), mirroring db.c's request arm; + * the barrier commits once, then wo_db_flush_drops applies both. + * + * The failure this catches: a back_off read straight off the (still stale, + * pre-barrier) durable map would have the second delta name the FIRST + * request's insert-time offset instead of the first delta — skipping it. + * Checked two ways: the final value must reflect BOTH updates in order, + * and delta 2's back-pointer, read straight off disk, must equal delta 1's + * own offset, not the base insert's. */ +static void test_keys_resident_two_updates_one_drain(void) { + char path[128]; + snprintf(path, sizeof path, "%s/keys2upd.wal", g_dir); + wo_rt rt; + T_EQ(wo_rt_init(&rt, 1 << 20, KEYS_CLASSES, 1), 0); + wo_db db; + T_EQ(wo_db_init(&db, KEYS_CLASSES, 1, 0, 1), 0); + wo_wal w; + T_EQ(wo_wal_open(&w, path, 1 << 16), 0); + db.rt = &rt; rt.wal = &w; rt.db = &db; + const char *msg = ""; + + wo_str *s = wo_str_new(&rt, "sku", 3); + uint64_t vals[2] = {111, (uint64_t)(uintptr_t)s}; + uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL); + T_CHECK(id != 0); + uint64_t base_off = wo_wal_next_offset(&w); + T_EQ(wo_wal_append_insert(&w, &db, 0, id), 0); + T_EQ(wo_wal_commit(&w), 0); + T_EQ(wo_row_drop_payload(&db, 0, id, base_off), 0); + + /* "request" 1: field 0, 111 -> 222 — staged, NOT committed, the map + NOT moved (only recorded as pending) */ + int ek = 0; + uint64_t roff1 = wo_wal_next_offset(&w); + T_EQ(wo_row_update_field_slot(&db, 0, id, 0, 222, &msg, &ek), 0); + T_EQ(ek, DB_ERR_NONE); + T_EQ(wo_wal_pend_repoint(&w, 0, id, roff1), 0); + + /* "request" 2, SAME drain: field 0, 222 -> 333. The id map still names + the base insert (the re-point above is only PENDING) — back_off must + come from the pending list, not wo_row_offset1, or this chains to + the wrong predecessor. */ + uint64_t roff2 = wo_wal_next_offset(&w); + T_EQ(wo_row_update_field_slot(&db, 0, id, 0, 333, &msg, &ek), 0); + T_EQ(ek, DB_ERR_NONE); + T_EQ(wo_wal_pend_repoint(&w, 0, id, roff2), 0); + + /* the drain's barrier: ONE commit for both staged deltas, then both + pending re-points applied — db.c/vm.c's exact shape */ + T_EQ(wo_wal_commit(&w), 0); + wo_db_flush_drops(&db, &w); + + /* both updates visible, in order */ + db_row *r = wo_row_borrow(&db, 0, id, &msg); + T_CHECK(r != NULL && r->slots[0] == 333); + wo_str *back = (wo_str *)(uintptr_t)r->slots[1]; + T_CHECK(back != NULL && back->len == 3 && memcmp(back->data, "sku", 3) == 0); + wo_row_release(&db, 0, r); + + /* the chain itself: delta 2's back-pointer names delta 1's OWN offset, + not the base insert's — payload layout established by + test_delta_record (kind|class|id|field_idx|back_off|value, 33 bytes + for a scalar field) */ + uint8_t body[33]; + T_EQ((int)pread(w.fd, body, 33, (off_t)(roff2 + 8)), 33); + T_EQ(body[0], WO_WAL_DELTA); + uint64_t back_off; + memcpy(&back_off, body + 17, 8); + T_EQ(back_off, roff1); + T_CHECK(back_off != base_off); /* the skip this test exists to catch */ + + wo_wal_close(&w); + wo_db_destroy(&db); + wo_rt_destroy(&rt); +} + /* databasev2 2 (5c): boot. A keys-resident store must come back from replay * with its rows readable FROM THE LOG — the map rebuilt to offsets, not slabs. * This is the half the round-trip test cannot cover: it runs in a fresh db, @@ -1702,6 +1800,7 @@ int main(void) { test_keys_resident_update_field(); test_keys_resident_update_indexed(); test_keys_resident_update_unique_violation_refused(); + test_keys_resident_two_updates_one_drain(); test_keys_resident_replay(); test_keys_resident_survives_compaction(); test_keys_resident_delete();