diff --git a/database/src/db.c b/database/src/db.c index 52234b7..f35ed6b 100644 --- a/database/src/db.c +++ b/database/src/db.c @@ -70,8 +70,16 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { * neither this barrier nor the compaction check below. * * Failure is fatal, not a trap: the row is already in RAM. */ + /* databasev2 2 (5c): the offset this record WILL occupy. Taken + * BEFORE the append, recorded as pending, and acted on only after + * the commit below — a keys-resident payload dropped any earlier + * would leave an offset whose bytes are still in the staging + * buffer. */ + uint64_t koff = wo_wal_next_offset(w); if (wo_wal_append_insert(w, db, cid, id) != 0) wo_wal_stage_fatal(w); + if (wo_table_is_keys_resident(db, cid)) (void)wo_wal_pend_drop(w, cid, id, koff); wo_wal_commit_fatal(w, 1); + wo_db_flush_drops(db, w); maybe_compact(db, w); } R[A] = id; @@ -259,7 +267,12 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) { * already in RAM; of the three verbs only insert could undo * itself, so continuing means RAM ahead of disk. One rule: once a * statement has mutated RAM, the outcomes are durable or death. */ + uint64_t koff = wo_wal_next_offset(w); if (wo_wal_append_insert(w, db, q->cid, id) != 0) wo_wal_stage_fatal(w); + /* recorded, not performed: this batch's barrier runs in the drain + * (vm.c), and only then are these offsets readable */ + if (wo_table_is_keys_resident(db, q->cid)) + (void)wo_wal_pend_drop(w, q->cid, id, koff); } q->result = id; break; diff --git a/database/src/wal.c b/database/src/wal.c index 11e68ff..6bc699f 100644 --- a/database/src/wal.c +++ b/database/src/wal.c @@ -330,7 +330,29 @@ int wo_wal_open(wo_wal *w, const char *path, uint64_t prealloc) { return 0; } +int wo_wal_pend_drop(wo_wal *w, uint32_t cid, uint64_t id, uint64_t off) { + if (w->pend_len == w->pend_cap) { + size_t nc = w->pend_cap ? w->pend_cap * 2 : 16; + struct wo_wal_pend *np = realloc(w->pend, nc * sizeof *np); + if (!np) return -1; /* the row stays resident: safe, just not dropped */ + w->pend = np; + w->pend_cap = nc; + } + w->pend[w->pend_len].cid = cid; + w->pend[w->pend_len].id = id; + w->pend[w->pend_len].off = off; + w->pend_len++; + return 0; +} + +void wo_db_flush_drops(wo_db *db, wo_wal *w) { + for (size_t i = 0; i < w->pend_len; i++) + (void)wo_row_drop_payload(db, w->pend[i].cid, w->pend[i].id, w->pend[i].off); + w->pend_len = 0; +} + void wo_wal_close(wo_wal *w) { + free(w->pend); if (w->fd >= 0) close(w->fd); free(w->path); free(w->buf); @@ -753,6 +775,16 @@ int64_t wo_wal_replay_ex(const char *path, wo_db *db, uint32_t *volatile_cid) { | ((uint32_t)payload[3] << 16) | ((uint32_t)payload[4] << 24) : 0u; + /* databasev2 2 (5c): a keys-resident row must end boot pointing at the + * LOG, not at a slab. The record is applied normally (so indexes and + * uniqueness are built exactly as for any other table) and then its + * payload is dropped, leaving the id map holding THIS record's offset. + * For an update the later record wins, because each apply overwrites + * the map in order — which is the same rule replay already follows. */ + uint8_t rec_kind = len >= 1u ? payload[0] : 0u; + uint64_t rec_id = 0; + if (len >= 13u) + for (int b = 0; b < 8; b++) rec_id |= (uint64_t)payload[5 + b] << (8 * b); int rc = apply_record(db, payload, len); free(payload); if (rc != 0) { @@ -760,6 +792,9 @@ int64_t wo_wal_replay_ex(const char *path, wo_db *db, uint32_t *volatile_cid) { if (rc == -2 && volatile_cid) *volatile_cid = rec_cid; return rc == -2 ? -2 : -1; } + if ((rec_kind == WO_WAL_INSERT || rec_kind == WO_WAL_UPDATE) && rec_id && + wo_table_is_keys_resident(db, rec_cid)) + (void)wo_row_drop_payload(db, rec_cid, rec_id, off); off += 8u + len + 4u; applied++; } diff --git a/database/src/wal.h b/database/src/wal.h index eca4353..9f1c3a8 100644 --- a/database/src/wal.h +++ b/database/src/wal.h @@ -76,6 +76,15 @@ typedef struct wo_wal { /* databasev2 3: what compaction actually did, reported under WO_WAL_STATS. * The PAUSE is the number the spec refused to assume — compaction is * stop-the-world, so its duration is the cost being weighed. */ + /* databasev2 2 (5c): rows whose payload may be dropped ONCE the barrier + * they are staged behind succeeds. A keys-resident row cannot be dropped + * at append time: with group commit the record is still in the staging + * buffer, so its offset would pread zeros. Recorded here and performed by + * wo_db_flush_drops after the commit — the same shape as the drain's held + * replies, and for the same reason. If the process dies first the list + * dies with it, which is correct: nothing was dropped and nothing lost. */ + struct wo_wal_pend { uint32_t cid; uint64_t id; uint64_t off; } *pend; + size_t pend_len, pend_cap; uint64_t stat_compactions; uint64_t stat_compact_us_max; uint64_t stat_compact_us_total; @@ -139,6 +148,14 @@ int wo_wal_append_update(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id); * batch stays staged: a failed commit consumes nothing). */ int wo_wal_commit(wo_wal *w); +/* databasev2 2 (5c): note a payload that may be dropped after the next commit. + * 0 ok, -1 out of memory (the row simply stays resident, which is safe). */ +int wo_wal_pend_drop(wo_wal *w, uint32_t cid, uint64_t id, uint64_t off); + +/* databasev2 2 (5c): perform every pending drop. Call ONLY after a commit has + * succeeded — that is what makes the recorded offsets readable. */ +void wo_db_flush_drops(wo_db *db, wo_wal *w); + /* databasev2 3: the checkpoint trigger, as a PURE decision so it can be tested * without a store — which is the only way a policy like this gets tested at all. * diff --git a/runtime/src/vm.c b/runtime/src/vm.c index 4eda1e1..8d537a3 100644 --- a/runtime/src/vm.c +++ b/runtime/src/vm.c @@ -227,7 +227,13 @@ static int wo_vm_adopt(wo_vm *vm) { * (see db.c), so a drain must never leave a record behind. */ if (staged) { wo_wal *cw = (wo_wal *)vm->rt.wal; - if (cw) wo_wal_commit_fatal(cw, staged); + if (cw) { + wo_wal_commit_fatal(cw, staged); + /* databasev2 2 (5c): NOW the batch's offsets are readable, so any + * keys-resident payload staged behind this barrier can be dropped. + * Before the barrier those offsets pread zeros. */ + wo_db_flush_drops((wo_db *)vm->rt.db, cw); + } } while (rhead) { wo_envelope *rn = rhead->next; diff --git a/runtime/test/test_wal.c b/runtime/test/test_wal.c index 4f1b9f8..087e098 100644 --- a/runtime/test/test_wal.c +++ b/runtime/test/test_wal.c @@ -574,6 +574,58 @@ static void test_keys_resident_round_trip(void) { wo_rt_destroy(&rt); } +/* databasev2 2 (5c): boot. A keys-resident store must come back from replay + * with its rows readable FROM THE LOG — the map rebuilt to offsets, not slabs. + * This is the half the round-trip test cannot cover: it runs in a fresh db, + * exactly as a restart would. */ +static void test_keys_resident_replay(void) { + char path[128]; + snprintf(path, sizeof path, "%s/keysboot.wal", g_dir); + wo_rt rt; + T_EQ(wo_rt_init(&rt, 1 << 20, KEYS_CLASSES, 1), 0); + wo_db db; + T_EQ(wo_db_init(&db, KEYS_CLASSES, 1, 0, 1), 0); + wo_wal w; + T_EQ(wo_wal_open(&w, path, 1 << 16), 0); + db.rt = &rt; rt.wal = &w; rt.db = &db; + const char *msg = ""; + + uint64_t ids[3]; + for (int i = 0; i < 3; i++) { + wo_str *sv = wo_str_new(&rt, "abc", 3); + uint64_t vals[2] = {(uint64_t)(i * 11 + 1), (uint64_t)(uintptr_t)sv}; + ids[i] = wo_row_insert(&db, 0, vals, &msg, NULL); + T_CHECK(ids[i] != 0); + uint64_t off = wo_wal_next_offset(&w); + T_EQ(wo_wal_append_insert(&w, &db, 0, ids[i]), 0); + T_EQ(wo_wal_commit(&w), 0); + T_EQ(wo_row_drop_payload(&db, 0, ids[i], off), 0); + } + wo_wal_close(&w); + wo_db_destroy(&db); + + /* a fresh process would do exactly this */ + wo_db db2; + T_EQ(wo_db_init(&db2, KEYS_CLASSES, 1, 0, 1), 0); + T_EQ(wo_wal_replay(path, &db2), 3); + wo_wal w2; + T_EQ(wo_wal_open(&w2, path, 1 << 16), 0); + db2.rt = &rt; rt.wal = &w2; rt.db = &db2; + + T_CHECK(db2.tables[0].count == 3); /* live, though nothing is in a slab */ + for (int i = 0; i < 3; i++) { + db_row *r = wo_row_borrow(&db2, 0, ids[i], &msg); + T_CHECK(r != NULL); + T_CHECK(r->slots[0] == (uint64_t)(i * 11 + 1)); + wo_str *back = (wo_str *)(uintptr_t)r->slots[1]; + T_CHECK(back != NULL && back->len == 3 && memcmp(back->data, "abc", 3) == 0); + wo_row_release(&db2, 0, r); + } + wo_wal_close(&w2); + wo_db_destroy(&db2); + wo_rt_destroy(&rt); +} + static void test_torn_tail(void) { char path[128]; snprintf(path, sizeof path, "%s/torn.wal", g_dir); @@ -994,6 +1046,7 @@ int main(void) { test_commit_failure_detected(); test_compact_shortens_and_replays_equal(); test_keys_resident_round_trip(); + test_keys_resident_replay(); test_stale_compact_temp_is_removed(); test_should_compact_policy(); test_compact_refuses_with_staged_records();