diff --git a/database/src/table.c b/database/src/table.c index 183353f..46b5215 100644 --- a/database/src/table.c +++ b/database/src/table.c @@ -4,6 +4,8 @@ #include #include "cont.h" +#include "gc.h" /* databasev2 2 (5c): VM-side drops for materialised rows */ +#include "wal.h" /* databasev2 2 (5c): a keys-resident borrow reads the log */ /* ---- engine-owned value encode / free / decode ------------------------- */ @@ -729,14 +731,57 @@ db_row *wo_row_ptr(wo_db *db, uint32_t class_id, uint64_t id) { } db_row *wo_row_borrow(wo_db *db, uint32_t class_id, uint64_t id, const char **msg) { - (void)msg; - /* databasev2 2: only the resident backing exists so far. When - * `resident: keys` storage lands, this is where hvals is read as an - * OFFSET (it is already a uint64 holding slot+1, so offset+1 fits the - * same field) and 5b's wo_wal_read_row_at fills the table's scratch. - * Keeping the seam here, unused, is what makes that a local change - * instead of another sweep of every reader. */ - return wo_row_ptr(db, class_id, id); + /* Fully-resident tables: exactly today's lookup, and releasing is a no-op. + * The hot path pays one predicate. */ + if (!wo_table_is_keys_resident(db, class_id)) return wo_row_ptr(db, class_id, id); + + /* Keys-resident: the id map holds the record's LOG OFFSET (off + 1), not a + * slot, so the row is materialised into the table's scratch. */ + db_table *t = &db->tables[class_id]; + if (!t->row_size) return NULL; + uint64_t o1 = hget(t, id); + if (!o1) return NULL; + if (!db->rt || !db->rt->wal) { + /* a keys-resident table cannot exist without a log to read from; the + * loader refuses the annotation outright, so this is a defensive arm */ + if (msg) *msg = "resident: keys table without a write-ahead log"; + return NULL; + } + if (t->scratch_busy) { + /* One scratch per TABLE, so two live borrows on the same table would + * hand back the same buffer. The unique shadow borrows one candidate + * at a time, which is why per-table is enough — but say so rather than + * corrupting the first borrow silently. */ + if (msg) *msg = "nested borrow on one table"; + return NULL; + } + if (t->scratch_cap < t->row_size) { + uint8_t *nb = realloc(t->scratch, t->row_size); + if (!nb) { + if (msg) *msg = "out of memory"; + return NULL; + } + t->scratch = nb; + t->scratch_cap = t->row_size; + } + db_row *r = (db_row *)t->scratch; + uint32_t got_cid = 0; + uint64_t got_id = 0; + if (wo_wal_read_row_at((wo_wal *)db->rt->wal, db, db->rt, o1 - 1, &got_cid, &got_id, + r->slots, msg) != 0) + return NULL; + if (got_cid != class_id || got_id != id) { + /* the offset pointed at someone else's record — a compaction that + * moved records without rebuilding this map would land here, which is + * exactly the obligation recorded at wo_wal_compact */ + if (msg) *msg = "log offset does not hold the expected row"; + return NULL; + } + r->id = id; + r->class_id = class_id; + r->flags = 0; + t->scratch_busy = 1; + return r; } void wo_row_release(wo_db *db, uint32_t class_id, db_row *r) { @@ -744,7 +789,13 @@ void wo_row_release(wo_db *db, uint32_t class_id, db_row *r) { db_table *t = &db->tables[class_id]; if (!t->scratch_busy || (uint8_t *)r != t->scratch) return; /* slab-backed */ const wo_classdesc *c = &db->classes[class_id]; - for (uint32_t i = 0; i < c->field_cnt; i++) wo_db_val_free(db, c->kinds[i], r->slots[i]); + /* These are VM values, not engine values. wo_wal_read_row_at is the + * out-gate — it always COPIES, producing fresh runtime allocations — so + * they must be dropped through the runtime. Freeing them with the engine's + * allocator (as this did while the materialising path was still a stub) + * is a bad-free the moment a keys-resident row is actually read back. */ + if (db->rt) + for (uint32_t i = 0; i < c->field_cnt; i++) wo_drop_kind(db->rt, c->kinds[i], r->slots[i]); t->scratch_busy = 0; } @@ -1035,6 +1086,38 @@ int wo_row_has_referrers(wo_db *db, uint32_t class_id, uint64_t id) { return 0; } +int wo_table_is_keys_resident(const wo_db *db, uint32_t class_id) { + if (class_id >= db->class_cnt) return 0; + return (db->classes[class_id].flags & WO_CLASSF_RESIDENT_KEYS) != 0u; +} + +int wo_row_drop_payload(wo_db *db, uint32_t class_id, uint64_t id, uint64_t wal_off) { + if (class_id >= db->class_cnt) return -1; + db_table *t = &db->tables[class_id]; + if (!t->row_size) return -1; + uint64_t s1 = hget(t, id); + if (!s1) return -1; + uint32_t g = (uint32_t)(s1 - 1); + db_row *r = slot_row(t, g); + /* the values are engine-owned; the log holds their bytes now */ + const wo_classdesc *c = &db->classes[class_id]; + for (uint32_t i = 0; i < c->field_cnt; i++) db_val_free(c->kinds[i], r->slots[i]); + t->bitmap[g >> 6] &= ~(1ull << (g & 63)); + /* the id STAYS, now pointing at the log rather than at a slab. No + * idx_remove_row and no count change: the row is live, only its backing + * moved. */ + if (hput(t, id, wal_off + 1) != 0) return -1; + if (t->free_cnt == t->free_cap) { + uint32_t ncap = t->free_cap ? t->free_cap * 2 : 16; + uint32_t *nf = realloc(t->free_slots, (size_t)ncap * 4); + if (!nf) return 0; /* slot simply not recycled; the bitmap still frees it */ + t->free_slots = nf; + t->free_cap = ncap; + } + t->free_slots[t->free_cnt++] = g; + return 0; +} + int wo_row_remove(wo_db *db, uint32_t class_id, uint64_t id) { if (class_id >= db->class_cnt) return -1; db_table *t = &db->tables[class_id]; diff --git a/database/src/table.h b/database/src/table.h index bbde56d..e8b3cc9 100644 --- a/database/src/table.h +++ b/database/src/table.h @@ -133,6 +133,14 @@ typedef struct db_table { } db_table; typedef struct wo_db { + /* databasev2 2 (5c): the runtime this store belongs to, so a borrow can + * reach the WAL. wo_rt already carries `db` and `wal` as opaque handles, + * so this closes the loop without threading a wal pointer through + * wo_row_borrow's eleven call sites — which is the whole reason 5c is one + * accessor rather than eleven rewrites. NULL in test binaries and with + * durability off; a `resident: keys` table cannot exist in either case, + * because it has no log to read rows back from. */ + wo_rt *rt; const wo_classdesc *classes; uint32_t class_cnt; uint32_t shard, nshards; /* S of N; ids interleave S+1, S+1+N, … */ @@ -160,6 +168,35 @@ int wo_row_read(wo_db *db, wo_rt *rt, uint32_t class_id, uint64_t id, * it. 0 ok, -1 no such row. */ int wo_row_remove(wo_db *db, uint32_t class_id, uint64_t id); +/* databasev2 2 (5c): drop a row's PAYLOAD while keeping it live. + * + * The operation the plan recorded as missing. For a `resident: keys` table the + * row's bytes live in the log, not in a slab: this frees the slot and its + * engine-owned values, then re-points the id map at [wal_off] (stored as + * off + 1, reusing the same 0-is-empty trick the slot encoding uses — a table + * is wholly `all` or wholly `keys`, so the interpretation is per-table and + * never ambiguous). + * + * What it deliberately does NOT do, and why: + * - it does not touch the secondary indexes. They store row IDS, not slots + * (see db_ibucket), so they are already indirect through the id map and + * stay correct across this. + * - it does not decrement `count`. The row is still LIVE; only its backing + * moved. + * - it does not remove the id. The id is how the row is found afterwards. + * + * [wal_off] must be the offset of a record whose commit succeeded. Since + * databasev2 4 made a failed commit fatal, no execution can reach here with an + * offset that never became durable — which is what wo_wal_next_offset's + * contract asks for, now guaranteed by process death rather than by an inline + * check the deferred barrier no longer allows. + * + * 0 ok, -1 unknown class/row. */ +int wo_row_drop_payload(wo_db *db, uint32_t class_id, uint64_t id, uint64_t wal_off); + +/* databasev2 2 (5c): is this table's row data in the log rather than in slabs? */ +int wo_table_is_keys_resident(const wo_db *db, uint32_t class_id); + /* iteration 9b FK restrict: 1 if some row in some class holds a non-nullable * `ref` to [class_id] equal to [id] — i.e. deleting this row would dangle a * reference. The compiler records a ref field's target class in the class diff --git a/runtime/src/main.c b/runtime/src/main.c index 81d40da..a2bd783 100644 --- a/runtime/src/main.c +++ b/runtime/src/main.c @@ -217,6 +217,7 @@ int main(int argc, char **argv) { return 2; } VM.rt.db = &DB; + DB.rt = &VM.rt; /* databasev2 2 (5c): the loop a borrow reads the WAL through */ const char *data_dir = getenv("WO_DATA"); if (data_dir && data_dir[0]) { char wal_path[512]; diff --git a/runtime/test/test_wal.c b/runtime/test/test_wal.c index 9d34045..4f1b9f8 100644 --- a/runtime/test/test_wal.c +++ b/runtime/test/test_wal.c @@ -508,6 +508,72 @@ static void test_compact_crash_battery(void) { } } +/* databasev2 2 (5c): the keys-resident round trip. A row is inserted, its + * record committed, its PAYLOAD DROPPED from the slab, and then read back out + * of the log by offset — including its heap-valued column, which is the case + * that would silently return garbage if the materialisation were wrong. */ +static const uint8_t keys_kinds[] = {WO_K_SCALAR, WO_K_TEXT}; +static const wo_classdesc KEYS_CLASSES[] = { + {.name = 0, .flags = WO_CLASSF_RESIDENT_KEYS, .field_cnt = 2, .kinds = keys_kinds}, +}; + +static void test_keys_resident_round_trip(void) { + char path[128]; + snprintf(path, sizeof path, "%s/keysres.wal", g_dir); + wo_rt rt; + T_EQ(wo_rt_init(&rt, 1 << 20, KEYS_CLASSES, 1), 0); + wo_db db; + T_EQ(wo_db_init(&db, KEYS_CLASSES, 1, 0, 1), 0); + wo_wal w; + T_EQ(wo_wal_open(&w, path, 1 << 16), 0); + db.rt = &rt; /* the loop a borrow reads the WAL through */ + rt.wal = &w; + rt.db = &db; + const char *msg = ""; + + T_CHECK(wo_table_is_keys_resident(&db, 0) == 1); + + wo_str *s = wo_str_new(&rt, "hello", 5); + uint64_t vals[2] = {4242, (uint64_t)(uintptr_t)s}; + uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL); + T_CHECK(id != 0); + + /* the offset this record WILL occupy — valid because the commit below + * succeeds; a failed commit is fatal since databasev2 4 */ + uint64_t off = wo_wal_next_offset(&w); + T_EQ(wo_wal_append_insert(&w, &db, 0, id), 0); + T_EQ(wo_wal_commit(&w), 0); + + /* while still resident, the row reads out of the slab */ + db_row *res = wo_row_borrow(&db, 0, id, &msg); + T_CHECK(res != NULL && res->slots[0] == 4242); + wo_row_release(&db, 0, res); + + /* drop the payload: slot freed, id kept, indexes untouched, still live */ + uint64_t before = db.tables[0].count; + T_EQ(wo_row_drop_payload(&db, 0, id, off), 0); + T_CHECK(db.tables[0].count == before); /* still LIVE, only unbacked */ + + /* and now it comes back out of the LOG */ + db_row *r = wo_row_borrow(&db, 0, id, &msg); + T_CHECK(r != NULL); + T_CHECK(r->id == id); + T_CHECK(r->slots[0] == 4242); + wo_str *back = (wo_str *)(uintptr_t)r->slots[1]; + T_CHECK(back != NULL && back->len == 5 && memcmp(back->data, "hello", 5) == 0); + wo_row_release(&db, 0, r); + + /* the scratch is reusable: a second borrow must succeed, which it cannot + * if release failed to clear the busy flag */ + db_row *again = wo_row_borrow(&db, 0, id, &msg); + T_CHECK(again != NULL && again->slots[0] == 4242); + wo_row_release(&db, 0, again); + + wo_wal_close(&w); + wo_db_destroy(&db); + wo_rt_destroy(&rt); +} + static void test_torn_tail(void) { char path[128]; snprintf(path, sizeof path, "%s/torn.wal", g_dir); @@ -927,6 +993,7 @@ int main(void) { test_roundtrip_replay(); test_commit_failure_detected(); test_compact_shortens_and_replays_equal(); + test_keys_resident_round_trip(); test_stale_compact_temp_is_removed(); test_should_compact_policy(); test_compact_refuses_with_staged_records();