feat(db2-keys): storage — drop the payload, read it back from the log

databasev2 2, task 5c step 2. The storage half the accessor was left waiting
for. Not yet wired into insert (that and 5d remain), and the loader still
refuses `resident: keys`, so nothing is exposed to a program yet.

- wo_db gains an `rt` back-pointer, set in main.c beside VM.rt.db. wo_rt
  already carries `db` and `wal` as opaque handles, so this closes the loop
  and a borrow can reach the log WITHOUT threading a wal pointer through
  eleven call sites — which is the whole reason 5c is one accessor
- wo_row_drop_payload: the operation the plan recorded as MISSING. Frees the
  slot and its engine-owned values, then re-points the id map at the record's
  log offset (off + 1, reusing the same 0-is-empty trick as slot + 1). It
  deliberately does NOT touch the secondary indexes (they store row ids, so
  they stay correct), does NOT decrement count (the row is still live, only
  its backing moved), and does NOT remove the id (that is how it is found)
- wo_row_borrow materialises for a keys table: reads the offset from the id
  map, calls 5b's wo_wal_read_row_at into the per-table scratch, and checks
  the record actually holds the expected class and id — a compaction that
  moved records without rebuilding the map lands exactly there, which is the
  obligation recorded at wo_wal_compact
- fully-resident tables keep today's path and pay one predicate

A REAL BUG, exposed the first time the path was used: wo_row_release freed the
materialised values with the ENGINE's allocator. They are VM values —
wo_wal_read_row_at is the out-gate and always copies — so ASan reported a
bad-free immediately. It now drops them through the runtime. That stub was
written in 5c step 1 for a path that did not exist yet.

Recorded while implementing: wo_wal_next_offset's contract says to trust an
offset "only after the matching commit returns 0". Group commit (databasev2 4)
defers that barrier to the drain, so db.c can no longer check inline — but part
A also made a failed commit FATAL, so no execution can record an offset whose
record never became durable. Same guarantee, different mechanism.

Test: a heap-valued row is inserted, committed, has its payload dropped, and is
read back out of the log with its Text intact; count is unchanged (still live);
and a second borrow succeeds, which fails if release did not clear the scratch.

Verified: just wovm-test — 36 suites 0 fail, test_wal 4273 pass, cli_smoke OK.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
(cherry picked from commit 125bd09218d616b2a16b140de770d3f38b45f0ac)
This commit is contained in:
shoney.arickathil 2026-08-29 20:47:52 +02:00
parent 02b4b13a52
commit 7cc80405dd
4 changed files with 197 additions and 9 deletions

View file

@ -4,6 +4,8 @@
#include <string.h>
#include "cont.h"
#include "gc.h" /* databasev2 2 (5c): VM-side drops for materialised rows */
#include "wal.h" /* databasev2 2 (5c): a keys-resident borrow reads the log */
/* ---- engine-owned value encode / free / decode ------------------------- */
@ -729,14 +731,57 @@ db_row *wo_row_ptr(wo_db *db, uint32_t class_id, uint64_t id) {
}
db_row *wo_row_borrow(wo_db *db, uint32_t class_id, uint64_t id, const char **msg) {
(void)msg;
/* databasev2 2: only the resident backing exists so far. When
* `resident: keys` storage lands, this is where hvals is read as an
* OFFSET (it is already a uint64 holding slot+1, so offset+1 fits the
* same field) and 5b's wo_wal_read_row_at fills the table's scratch.
* Keeping the seam here, unused, is what makes that a local change
* instead of another sweep of every reader. */
return wo_row_ptr(db, class_id, id);
/* Fully-resident tables: exactly today's lookup, and releasing is a no-op.
* The hot path pays one predicate. */
if (!wo_table_is_keys_resident(db, class_id)) return wo_row_ptr(db, class_id, id);
/* Keys-resident: the id map holds the record's LOG OFFSET (off + 1), not a
* slot, so the row is materialised into the table's scratch. */
db_table *t = &db->tables[class_id];
if (!t->row_size) return NULL;
uint64_t o1 = hget(t, id);
if (!o1) return NULL;
if (!db->rt || !db->rt->wal) {
/* a keys-resident table cannot exist without a log to read from; the
* loader refuses the annotation outright, so this is a defensive arm */
if (msg) *msg = "resident: keys table without a write-ahead log";
return NULL;
}
if (t->scratch_busy) {
/* One scratch per TABLE, so two live borrows on the same table would
* hand back the same buffer. The unique shadow borrows one candidate
* at a time, which is why per-table is enough — but say so rather than
* corrupting the first borrow silently. */
if (msg) *msg = "nested borrow on one table";
return NULL;
}
if (t->scratch_cap < t->row_size) {
uint8_t *nb = realloc(t->scratch, t->row_size);
if (!nb) {
if (msg) *msg = "out of memory";
return NULL;
}
t->scratch = nb;
t->scratch_cap = t->row_size;
}
db_row *r = (db_row *)t->scratch;
uint32_t got_cid = 0;
uint64_t got_id = 0;
if (wo_wal_read_row_at((wo_wal *)db->rt->wal, db, db->rt, o1 - 1, &got_cid, &got_id,
r->slots, msg) != 0)
return NULL;
if (got_cid != class_id || got_id != id) {
/* the offset pointed at someone else's record — a compaction that
* moved records without rebuilding this map would land here, which is
* exactly the obligation recorded at wo_wal_compact */
if (msg) *msg = "log offset does not hold the expected row";
return NULL;
}
r->id = id;
r->class_id = class_id;
r->flags = 0;
t->scratch_busy = 1;
return r;
}
void wo_row_release(wo_db *db, uint32_t class_id, db_row *r) {
@ -744,7 +789,13 @@ void wo_row_release(wo_db *db, uint32_t class_id, db_row *r) {
db_table *t = &db->tables[class_id];
if (!t->scratch_busy || (uint8_t *)r != t->scratch) return; /* slab-backed */
const wo_classdesc *c = &db->classes[class_id];
for (uint32_t i = 0; i < c->field_cnt; i++) wo_db_val_free(db, c->kinds[i], r->slots[i]);
/* These are VM values, not engine values. wo_wal_read_row_at is the
* out-gate — it always COPIES, producing fresh runtime allocations — so
* they must be dropped through the runtime. Freeing them with the engine's
* allocator (as this did while the materialising path was still a stub)
* is a bad-free the moment a keys-resident row is actually read back. */
if (db->rt)
for (uint32_t i = 0; i < c->field_cnt; i++) wo_drop_kind(db->rt, c->kinds[i], r->slots[i]);
t->scratch_busy = 0;
}
@ -1035,6 +1086,38 @@ int wo_row_has_referrers(wo_db *db, uint32_t class_id, uint64_t id) {
return 0;
}
int wo_table_is_keys_resident(const wo_db *db, uint32_t class_id) {
if (class_id >= db->class_cnt) return 0;
return (db->classes[class_id].flags & WO_CLASSF_RESIDENT_KEYS) != 0u;
}
int wo_row_drop_payload(wo_db *db, uint32_t class_id, uint64_t id, uint64_t wal_off) {
if (class_id >= db->class_cnt) return -1;
db_table *t = &db->tables[class_id];
if (!t->row_size) return -1;
uint64_t s1 = hget(t, id);
if (!s1) return -1;
uint32_t g = (uint32_t)(s1 - 1);
db_row *r = slot_row(t, g);
/* the values are engine-owned; the log holds their bytes now */
const wo_classdesc *c = &db->classes[class_id];
for (uint32_t i = 0; i < c->field_cnt; i++) db_val_free(c->kinds[i], r->slots[i]);
t->bitmap[g >> 6] &= ~(1ull << (g & 63));
/* the id STAYS, now pointing at the log rather than at a slab. No
* idx_remove_row and no count change: the row is live, only its backing
* moved. */
if (hput(t, id, wal_off + 1) != 0) return -1;
if (t->free_cnt == t->free_cap) {
uint32_t ncap = t->free_cap ? t->free_cap * 2 : 16;
uint32_t *nf = realloc(t->free_slots, (size_t)ncap * 4);
if (!nf) return 0; /* slot simply not recycled; the bitmap still frees it */
t->free_slots = nf;
t->free_cap = ncap;
}
t->free_slots[t->free_cnt++] = g;
return 0;
}
int wo_row_remove(wo_db *db, uint32_t class_id, uint64_t id) {
if (class_id >= db->class_cnt) return -1;
db_table *t = &db->tables[class_id];

View file

@ -133,6 +133,14 @@ typedef struct db_table {
} db_table;
typedef struct wo_db {
/* databasev2 2 (5c): the runtime this store belongs to, so a borrow can
* reach the WAL. wo_rt already carries `db` and `wal` as opaque handles,
* so this closes the loop without threading a wal pointer through
* wo_row_borrow's eleven call sites — which is the whole reason 5c is one
* accessor rather than eleven rewrites. NULL in test binaries and with
* durability off; a `resident: keys` table cannot exist in either case,
* because it has no log to read rows back from. */
wo_rt *rt;
const wo_classdesc *classes;
uint32_t class_cnt;
uint32_t shard, nshards; /* S of N; ids interleave S+1, S+1+N, … */
@ -160,6 +168,35 @@ int wo_row_read(wo_db *db, wo_rt *rt, uint32_t class_id, uint64_t id,
* it. 0 ok, -1 no such row. */
int wo_row_remove(wo_db *db, uint32_t class_id, uint64_t id);
/* databasev2 2 (5c): drop a row's PAYLOAD while keeping it live.
*
* The operation the plan recorded as missing. For a `resident: keys` table the
* row's bytes live in the log, not in a slab: this frees the slot and its
* engine-owned values, then re-points the id map at [wal_off] (stored as
* off + 1, reusing the same 0-is-empty trick the slot encoding uses — a table
* is wholly `all` or wholly `keys`, so the interpretation is per-table and
* never ambiguous).
*
* What it deliberately does NOT do, and why:
* - it does not touch the secondary indexes. They store row IDS, not slots
* (see db_ibucket), so they are already indirect through the id map and
* stay correct across this.
* - it does not decrement `count`. The row is still LIVE; only its backing
* moved.
* - it does not remove the id. The id is how the row is found afterwards.
*
* [wal_off] must be the offset of a record whose commit succeeded. Since
* databasev2 4 made a failed commit fatal, no execution can reach here with an
* offset that never became durable — which is what wo_wal_next_offset's
* contract asks for, now guaranteed by process death rather than by an inline
* check the deferred barrier no longer allows.
*
* 0 ok, -1 unknown class/row. */
int wo_row_drop_payload(wo_db *db, uint32_t class_id, uint64_t id, uint64_t wal_off);
/* databasev2 2 (5c): is this table's row data in the log rather than in slabs? */
int wo_table_is_keys_resident(const wo_db *db, uint32_t class_id);
/* iteration 9b FK restrict: 1 if some row in some class holds a non-nullable
* `ref` to [class_id] equal to [id] — i.e. deleting this row would dangle a
* reference. The compiler records a ref field's target class in the class

View file

@ -217,6 +217,7 @@ int main(int argc, char **argv) {
return 2;
}
VM.rt.db = &DB;
DB.rt = &VM.rt; /* databasev2 2 (5c): the loop a borrow reads the WAL through */
const char *data_dir = getenv("WO_DATA");
if (data_dir && data_dir[0]) {
char wal_path[512];

View file

@ -508,6 +508,72 @@ static void test_compact_crash_battery(void) {
}
}
/* databasev2 2 (5c): the keys-resident round trip. A row is inserted, its
* record committed, its PAYLOAD DROPPED from the slab, and then read back out
* of the log by offset — including its heap-valued column, which is the case
* that would silently return garbage if the materialisation were wrong. */
static const uint8_t keys_kinds[] = {WO_K_SCALAR, WO_K_TEXT};
static const wo_classdesc KEYS_CLASSES[] = {
{.name = 0, .flags = WO_CLASSF_RESIDENT_KEYS, .field_cnt = 2, .kinds = keys_kinds},
};
static void test_keys_resident_round_trip(void) {
char path[128];
snprintf(path, sizeof path, "%s/keysres.wal", g_dir);
wo_rt rt;
T_EQ(wo_rt_init(&rt, 1 << 20, KEYS_CLASSES, 1), 0);
wo_db db;
T_EQ(wo_db_init(&db, KEYS_CLASSES, 1, 0, 1), 0);
wo_wal w;
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
db.rt = &rt; /* the loop a borrow reads the WAL through */
rt.wal = &w;
rt.db = &db;
const char *msg = "";
T_CHECK(wo_table_is_keys_resident(&db, 0) == 1);
wo_str *s = wo_str_new(&rt, "hello", 5);
uint64_t vals[2] = {4242, (uint64_t)(uintptr_t)s};
uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL);
T_CHECK(id != 0);
/* the offset this record WILL occupy — valid because the commit below
* succeeds; a failed commit is fatal since databasev2 4 */
uint64_t off = wo_wal_next_offset(&w);
T_EQ(wo_wal_append_insert(&w, &db, 0, id), 0);
T_EQ(wo_wal_commit(&w), 0);
/* while still resident, the row reads out of the slab */
db_row *res = wo_row_borrow(&db, 0, id, &msg);
T_CHECK(res != NULL && res->slots[0] == 4242);
wo_row_release(&db, 0, res);
/* drop the payload: slot freed, id kept, indexes untouched, still live */
uint64_t before = db.tables[0].count;
T_EQ(wo_row_drop_payload(&db, 0, id, off), 0);
T_CHECK(db.tables[0].count == before); /* still LIVE, only unbacked */
/* and now it comes back out of the LOG */
db_row *r = wo_row_borrow(&db, 0, id, &msg);
T_CHECK(r != NULL);
T_CHECK(r->id == id);
T_CHECK(r->slots[0] == 4242);
wo_str *back = (wo_str *)(uintptr_t)r->slots[1];
T_CHECK(back != NULL && back->len == 5 && memcmp(back->data, "hello", 5) == 0);
wo_row_release(&db, 0, r);
/* the scratch is reusable: a second borrow must succeed, which it cannot
* if release failed to clear the busy flag */
db_row *again = wo_row_borrow(&db, 0, id, &msg);
T_CHECK(again != NULL && again->slots[0] == 4242);
wo_row_release(&db, 0, again);
wo_wal_close(&w);
wo_db_destroy(&db);
wo_rt_destroy(&rt);
}
static void test_torn_tail(void) {
char path[128];
snprintf(path, sizeof path, "%s/torn.wal", g_dir);
@ -927,6 +993,7 @@ int main(void) {
test_roundtrip_replay();
test_commit_failure_detected();
test_compact_shortens_and_replays_equal();
test_keys_resident_round_trip();
test_stale_compact_temp_is_removed();
test_should_compact_policy();
test_compact_refuses_with_staged_records();