feat(db2-keys): inserts and boot — payload dropped after the barrier
databasev2 2, task 5c. The write and boot halves. Still not exposed: the loader refuses `resident: keys` until 5d rewires the readers. GROUP COMMIT FORCED THE DESIGN. A keys-resident payload can only be dropped once its record is durable, but databasev2 4 deferred the barrier to the drain — so at append time the bytes are still in the staging buffer and the recorded offset would pread ZEROS. Dropping at append would have produced rows that read as garbage, intermittently, only under multi-shard load. So the drop is recorded, not performed: - wo_wal gains a pending-drop list, the same shape as the drain's held replies and for the same reason - both write paths take the offset BEFORE the append (wo_wal_next_offset) and record it; the inline path flushes right after its own commit, the request path's flush runs in the drain immediately after the barrier - if the process dies before the barrier the list dies with it, which is correct: nothing was dropped and nothing was lost - an out-of-memory pend is ignored on purpose — the row simply stays resident, which is safe Boot: replay now leaves a keys-resident table pointing at the LOG. Each record is applied normally, so indexes and uniqueness are built exactly as for any other table, and the payload is then dropped with THAT record's offset. For an update the later record wins, because each apply overwrites the map in order — the rule replay already follows. Tests: the round trip (insert, commit, drop, read back with Text intact) and now BOOT — a fresh wo_db replays the store and every row materialises from the log, count intact, nothing in a slab. Verified: just wovm-test — 36 suites 0 fail, test_wal 4301 pass, cli_smoke OK. REMAINING (5d), and precise: every reader still goes through wo_row_ptr, which for a keys table would index a freed slot. The scans in db.c walk the BITMAP, and a keys table's bitmap is empty by construction — so a query over one would today return no rows at all. That, FK restrict, and the @unique shadow are 5d, and the loader refusal stays until they land. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> (cherry picked from commit 08abd09bf88918f2582e74713dc7903beb8aaeb8)
This commit is contained in:
parent
7cc80405dd
commit
e16d4896f8
5 changed files with 125 additions and 1 deletions
|
|
@ -70,8 +70,16 @@ int wo_builtin_db(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
|
|||
* neither this barrier nor the compaction check below.
|
||||
*
|
||||
* Failure is fatal, not a trap: the row is already in RAM. */
|
||||
/* databasev2 2 (5c): the offset this record WILL occupy. Taken
|
||||
* BEFORE the append, recorded as pending, and acted on only after
|
||||
* the commit below — a keys-resident payload dropped any earlier
|
||||
* would leave an offset whose bytes are still in the staging
|
||||
* buffer. */
|
||||
uint64_t koff = wo_wal_next_offset(w);
|
||||
if (wo_wal_append_insert(w, db, cid, id) != 0) wo_wal_stage_fatal(w);
|
||||
if (wo_table_is_keys_resident(db, cid)) (void)wo_wal_pend_drop(w, cid, id, koff);
|
||||
wo_wal_commit_fatal(w, 1);
|
||||
wo_db_flush_drops(db, w);
|
||||
maybe_compact(db, w);
|
||||
}
|
||||
R[A] = id;
|
||||
|
|
@ -259,7 +267,12 @@ void wo_db_exec_req(wo_vm *vm, wo_db_req *q) {
|
|||
* already in RAM; of the three verbs only insert could undo
|
||||
* itself, so continuing means RAM ahead of disk. One rule: once a
|
||||
* statement has mutated RAM, the outcomes are durable or death. */
|
||||
uint64_t koff = wo_wal_next_offset(w);
|
||||
if (wo_wal_append_insert(w, db, q->cid, id) != 0) wo_wal_stage_fatal(w);
|
||||
/* recorded, not performed: this batch's barrier runs in the drain
|
||||
* (vm.c), and only then are these offsets readable */
|
||||
if (wo_table_is_keys_resident(db, q->cid))
|
||||
(void)wo_wal_pend_drop(w, q->cid, id, koff);
|
||||
}
|
||||
q->result = id;
|
||||
break;
|
||||
|
|
|
|||
|
|
@ -330,7 +330,29 @@ int wo_wal_open(wo_wal *w, const char *path, uint64_t prealloc) {
|
|||
return 0;
|
||||
}
|
||||
|
||||
int wo_wal_pend_drop(wo_wal *w, uint32_t cid, uint64_t id, uint64_t off) {
|
||||
if (w->pend_len == w->pend_cap) {
|
||||
size_t nc = w->pend_cap ? w->pend_cap * 2 : 16;
|
||||
struct wo_wal_pend *np = realloc(w->pend, nc * sizeof *np);
|
||||
if (!np) return -1; /* the row stays resident: safe, just not dropped */
|
||||
w->pend = np;
|
||||
w->pend_cap = nc;
|
||||
}
|
||||
w->pend[w->pend_len].cid = cid;
|
||||
w->pend[w->pend_len].id = id;
|
||||
w->pend[w->pend_len].off = off;
|
||||
w->pend_len++;
|
||||
return 0;
|
||||
}
|
||||
|
||||
void wo_db_flush_drops(wo_db *db, wo_wal *w) {
|
||||
for (size_t i = 0; i < w->pend_len; i++)
|
||||
(void)wo_row_drop_payload(db, w->pend[i].cid, w->pend[i].id, w->pend[i].off);
|
||||
w->pend_len = 0;
|
||||
}
|
||||
|
||||
void wo_wal_close(wo_wal *w) {
|
||||
free(w->pend);
|
||||
if (w->fd >= 0) close(w->fd);
|
||||
free(w->path);
|
||||
free(w->buf);
|
||||
|
|
@ -753,6 +775,16 @@ int64_t wo_wal_replay_ex(const char *path, wo_db *db, uint32_t *volatile_cid) {
|
|||
| ((uint32_t)payload[3] << 16)
|
||||
| ((uint32_t)payload[4] << 24)
|
||||
: 0u;
|
||||
/* databasev2 2 (5c): a keys-resident row must end boot pointing at the
|
||||
* LOG, not at a slab. The record is applied normally (so indexes and
|
||||
* uniqueness are built exactly as for any other table) and then its
|
||||
* payload is dropped, leaving the id map holding THIS record's offset.
|
||||
* For an update the later record wins, because each apply overwrites
|
||||
* the map in order — which is the same rule replay already follows. */
|
||||
uint8_t rec_kind = len >= 1u ? payload[0] : 0u;
|
||||
uint64_t rec_id = 0;
|
||||
if (len >= 13u)
|
||||
for (int b = 0; b < 8; b++) rec_id |= (uint64_t)payload[5 + b] << (8 * b);
|
||||
int rc = apply_record(db, payload, len);
|
||||
free(payload);
|
||||
if (rc != 0) {
|
||||
|
|
@ -760,6 +792,9 @@ int64_t wo_wal_replay_ex(const char *path, wo_db *db, uint32_t *volatile_cid) {
|
|||
if (rc == -2 && volatile_cid) *volatile_cid = rec_cid;
|
||||
return rc == -2 ? -2 : -1;
|
||||
}
|
||||
if ((rec_kind == WO_WAL_INSERT || rec_kind == WO_WAL_UPDATE) && rec_id &&
|
||||
wo_table_is_keys_resident(db, rec_cid))
|
||||
(void)wo_row_drop_payload(db, rec_cid, rec_id, off);
|
||||
off += 8u + len + 4u;
|
||||
applied++;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -76,6 +76,15 @@ typedef struct wo_wal {
|
|||
/* databasev2 3: what compaction actually did, reported under WO_WAL_STATS.
|
||||
* The PAUSE is the number the spec refused to assume — compaction is
|
||||
* stop-the-world, so its duration is the cost being weighed. */
|
||||
/* databasev2 2 (5c): rows whose payload may be dropped ONCE the barrier
|
||||
* they are staged behind succeeds. A keys-resident row cannot be dropped
|
||||
* at append time: with group commit the record is still in the staging
|
||||
* buffer, so its offset would pread zeros. Recorded here and performed by
|
||||
* wo_db_flush_drops after the commit — the same shape as the drain's held
|
||||
* replies, and for the same reason. If the process dies first the list
|
||||
* dies with it, which is correct: nothing was dropped and nothing lost. */
|
||||
struct wo_wal_pend { uint32_t cid; uint64_t id; uint64_t off; } *pend;
|
||||
size_t pend_len, pend_cap;
|
||||
uint64_t stat_compactions;
|
||||
uint64_t stat_compact_us_max;
|
||||
uint64_t stat_compact_us_total;
|
||||
|
|
@ -139,6 +148,14 @@ int wo_wal_append_update(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id);
|
|||
* batch stays staged: a failed commit consumes nothing). */
|
||||
int wo_wal_commit(wo_wal *w);
|
||||
|
||||
/* databasev2 2 (5c): note a payload that may be dropped after the next commit.
|
||||
* 0 ok, -1 out of memory (the row simply stays resident, which is safe). */
|
||||
int wo_wal_pend_drop(wo_wal *w, uint32_t cid, uint64_t id, uint64_t off);
|
||||
|
||||
/* databasev2 2 (5c): perform every pending drop. Call ONLY after a commit has
|
||||
* succeeded — that is what makes the recorded offsets readable. */
|
||||
void wo_db_flush_drops(wo_db *db, wo_wal *w);
|
||||
|
||||
/* databasev2 3: the checkpoint trigger, as a PURE decision so it can be tested
|
||||
* without a store — which is the only way a policy like this gets tested at all.
|
||||
*
|
||||
|
|
|
|||
|
|
@ -227,7 +227,13 @@ static int wo_vm_adopt(wo_vm *vm) {
|
|||
* (see db.c), so a drain must never leave a record behind. */
|
||||
if (staged) {
|
||||
wo_wal *cw = (wo_wal *)vm->rt.wal;
|
||||
if (cw) wo_wal_commit_fatal(cw, staged);
|
||||
if (cw) {
|
||||
wo_wal_commit_fatal(cw, staged);
|
||||
/* databasev2 2 (5c): NOW the batch's offsets are readable, so any
|
||||
* keys-resident payload staged behind this barrier can be dropped.
|
||||
* Before the barrier those offsets pread zeros. */
|
||||
wo_db_flush_drops((wo_db *)vm->rt.db, cw);
|
||||
}
|
||||
}
|
||||
while (rhead) {
|
||||
wo_envelope *rn = rhead->next;
|
||||
|
|
|
|||
|
|
@ -574,6 +574,58 @@ static void test_keys_resident_round_trip(void) {
|
|||
wo_rt_destroy(&rt);
|
||||
}
|
||||
|
||||
/* databasev2 2 (5c): boot. A keys-resident store must come back from replay
|
||||
* with its rows readable FROM THE LOG — the map rebuilt to offsets, not slabs.
|
||||
* This is the half the round-trip test cannot cover: it runs in a fresh db,
|
||||
* exactly as a restart would. */
|
||||
static void test_keys_resident_replay(void) {
|
||||
char path[128];
|
||||
snprintf(path, sizeof path, "%s/keysboot.wal", g_dir);
|
||||
wo_rt rt;
|
||||
T_EQ(wo_rt_init(&rt, 1 << 20, KEYS_CLASSES, 1), 0);
|
||||
wo_db db;
|
||||
T_EQ(wo_db_init(&db, KEYS_CLASSES, 1, 0, 1), 0);
|
||||
wo_wal w;
|
||||
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
|
||||
db.rt = &rt; rt.wal = &w; rt.db = &db;
|
||||
const char *msg = "";
|
||||
|
||||
uint64_t ids[3];
|
||||
for (int i = 0; i < 3; i++) {
|
||||
wo_str *sv = wo_str_new(&rt, "abc", 3);
|
||||
uint64_t vals[2] = {(uint64_t)(i * 11 + 1), (uint64_t)(uintptr_t)sv};
|
||||
ids[i] = wo_row_insert(&db, 0, vals, &msg, NULL);
|
||||
T_CHECK(ids[i] != 0);
|
||||
uint64_t off = wo_wal_next_offset(&w);
|
||||
T_EQ(wo_wal_append_insert(&w, &db, 0, ids[i]), 0);
|
||||
T_EQ(wo_wal_commit(&w), 0);
|
||||
T_EQ(wo_row_drop_payload(&db, 0, ids[i], off), 0);
|
||||
}
|
||||
wo_wal_close(&w);
|
||||
wo_db_destroy(&db);
|
||||
|
||||
/* a fresh process would do exactly this */
|
||||
wo_db db2;
|
||||
T_EQ(wo_db_init(&db2, KEYS_CLASSES, 1, 0, 1), 0);
|
||||
T_EQ(wo_wal_replay(path, &db2), 3);
|
||||
wo_wal w2;
|
||||
T_EQ(wo_wal_open(&w2, path, 1 << 16), 0);
|
||||
db2.rt = &rt; rt.wal = &w2; rt.db = &db2;
|
||||
|
||||
T_CHECK(db2.tables[0].count == 3); /* live, though nothing is in a slab */
|
||||
for (int i = 0; i < 3; i++) {
|
||||
db_row *r = wo_row_borrow(&db2, 0, ids[i], &msg);
|
||||
T_CHECK(r != NULL);
|
||||
T_CHECK(r->slots[0] == (uint64_t)(i * 11 + 1));
|
||||
wo_str *back = (wo_str *)(uintptr_t)r->slots[1];
|
||||
T_CHECK(back != NULL && back->len == 3 && memcmp(back->data, "abc", 3) == 0);
|
||||
wo_row_release(&db2, 0, r);
|
||||
}
|
||||
wo_wal_close(&w2);
|
||||
wo_db_destroy(&db2);
|
||||
wo_rt_destroy(&rt);
|
||||
}
|
||||
|
||||
static void test_torn_tail(void) {
|
||||
char path[128];
|
||||
snprintf(path, sizeof path, "%s/torn.wal", g_dir);
|
||||
|
|
@ -994,6 +1046,7 @@ int main(void) {
|
|||
test_commit_failure_detected();
|
||||
test_compact_shortens_and_replays_equal();
|
||||
test_keys_resident_round_trip();
|
||||
test_keys_resident_replay();
|
||||
test_stale_compact_temp_is_removed();
|
||||
test_should_compact_policy();
|
||||
test_compact_refuses_with_staged_records();
|
||||
|
|
|
|||
Loading…
Reference in a new issue