diff --git a/database/src/CODE-LOGIC.md b/database/src/CODE-LOGIC.md new file mode 100644 index 0000000..1836786 --- /dev/null +++ b/database/src/CODE-LOGIC.md @@ -0,0 +1,40 @@ +# database/src — how the engine hangs together + +The database engine is its own top-level directory, statically linked into +every `wovm` and every runtime test binary (`runtime/Makefile`'s `DBSRC`). +One binary, unchanged. Format doc: `docs/plan/oop-vm/04-db-binding.md`. +Memory-safety doctrine: the 9b design's section 6. + +## table.c — rows (iteration 9, Task 1) + +``` +VM values ──copy──▶ row slots (engine-owned malloc) ──copy──▶ fresh VM values + wo_row_insert wo_row_read +``` + +- **No VM pointer ever enters a slab; no slab pointer ever leaves.** Encode + copies per kind (Texts to `db_text`, owned objects flattened recursively to + `db_rec`, containers element-wise); decode allocates fresh VM values from + the caller's `wo_rt`. The GCREF kind is refused at encode — the compiler + should have made that impossible (the GC bulkhead), the engine refuses it + anyway. +- **Rows never move.** Slabs of 256 are malloc'd and kept for the table's + life; the free-slot list recycles removed slots before any slab grows; + the id hash maps id → slot. Ids are never reused (per-table counter, + shard-interleaved `S+1, S+1+N, …`), which is also what makes the hash's + tombstone sentinel safe. +- **Choke points**: `wo_row_insert` / `wo_row_remove` carry the `INDEX HOOK` + comments where Task 4's secondary indexes attach and Task 2's WAL stages + its record. Nothing else may mutate storage. +- One deliberate file-static: `g_classes` for recursive frees (`db_val_free` + has no context parameter). One process, one class table; revisit at + iteration 8 (shards share the same immutable table). + +## Verifying a change + +- `make -C runtime test` — `test_table` is this directory's suite (round + trips across kinds, nil encodings, shard interleave, slab growth, slot + reuse, misuse), ASan+UBSan like every runtime test. +- `just oop-e2e`, `just log-watcher` — regression that linking the engine + into wovm changed nothing observable (it is dead code until Task 3 wires + the first builtin). diff --git a/database/src/table.c b/database/src/table.c new file mode 100644 index 0000000..caef7eb --- /dev/null +++ b/database/src/table.c @@ -0,0 +1,454 @@ +#include "table.h" + +#include +#include + +#include "cont.h" + +/* ---- engine-owned value encode / free / decode ------------------------- */ + +/* Free one encoded slot value of [kind]. Recursion mirrors encoding. */ +static void db_val_free(uint8_t kind, uint64_t v); + +static void db_rec_free(db_rec *r, const wo_classdesc *classes) { + const wo_classdesc *c = &classes[r->class_id]; + for (uint32_t i = 0; i < c->field_cnt; i++) db_val_free(c->kinds[i], r->slots[i]); + free(r); +} + +/* db_val_free needs the class table for nested records; a file-static is + * the honest signature here — one engine per process today (N=1), and the + * pointer is set once at init. Revisit when iteration 8 brings N>1 shards + * (each shard's wo_db shares the same immutable class table anyway). */ +static const wo_classdesc *g_classes; + +static void db_val_free(uint8_t kind, uint64_t v) { + if (!v) return; + switch (kind) { + case WO_K_SCALAR: return; + case WO_K_TEXT: free((db_text *)(uintptr_t)v); return; + case WO_K_OWNED: db_rec_free((db_rec *)(uintptr_t)v, g_classes); return; + case WO_K_MULTI: { + db_multi *m = (db_multi *)(uintptr_t)v; + for (uint32_t i = 0; i < m->len; i++) db_val_free(m->elem_kind, m->items[i]); + free(m); + return; + } + case WO_K_MAP: { + db_map *m = (db_map *)(uintptr_t)v; + for (uint32_t i = 0; i < m->len; i++) { + db_val_free(m->key_kind, m->kv[2 * i]); + db_val_free(m->val_kind, m->kv[2 * i + 1]); + } + free(m); + return; + } + default: return; /* GCREF never stored */ + } +} + +/* Encode one VM value into an engine-owned slot value. 0-with-*ok=0 means + * failure (OOM or a GCREF); a genuine nil encodes as 0 with *ok=1. */ +static uint64_t db_val_encode(const wo_classdesc *classes, uint8_t kind, uint64_t v, + int *ok, const char **msg) { + *ok = 1; + switch (kind) { + case WO_K_SCALAR: return v; + case WO_K_TEXT: { + if (!v) return 0; + const wo_str *s = (const wo_str *)(uintptr_t)v; + db_text *t = malloc(sizeof(db_text) + s->len); + if (!t) goto oom; + t->len = s->len; + memcpy(t->bytes, s->data, s->len); + return (uint64_t)(uintptr_t)t; + } + case WO_K_OWNED: { + if (!v) return 0; + const wo_hdr *o = (const wo_hdr *)(uintptr_t)v; + const wo_classdesc *c = &classes[o->class_id]; + db_rec *r = malloc(sizeof(db_rec) + (size_t)c->field_cnt * 8u); + if (!r) goto oom; + r->class_id = o->class_id; + r->_pad = 0; + const uint64_t *f = (const uint64_t *)(const void *)(o + 1); + for (uint32_t i = 0; i < c->field_cnt; i++) { + r->slots[i] = db_val_encode(classes, c->kinds[i], f[i], ok, msg); + if (!*ok) { /* free what we built so far, then fail upward */ + for (uint32_t j = 0; j < i; j++) db_val_free(c->kinds[j], r->slots[j]); + free(r); + return 0; + } + } + return (uint64_t)(uintptr_t)r; + } + case WO_K_MULTI: { + if (!v) return 0; + const wo_multi *m = (const wo_multi *)(uintptr_t)v; + db_multi *d = malloc(sizeof(db_multi) + (size_t)m->len * 8u); + if (!d) goto oom; + d->elem_kind = m->elem_kind; + d->len = m->len; + for (uint32_t i = 0; i < m->len; i++) { + d->items[i] = db_val_encode(classes, m->elem_kind, m->items[i], ok, msg); + if (!*ok) { + for (uint32_t j = 0; j < i; j++) db_val_free(d->elem_kind, d->items[j]); + free(d); + return 0; + } + } + return (uint64_t)(uintptr_t)d; + } + case WO_K_MAP: { + if (!v) return 0; + const wo_map *m = (const wo_map *)(uintptr_t)v; + db_map *d = malloc(sizeof(db_map) + (size_t)m->len * 16u); + if (!d) goto oom; + d->key_kind = m->key_kind; + d->val_kind = m->val_kind; + d->len = m->len; + for (uint32_t i = 0; i < m->len; i++) { + d->kv[2 * i] = db_val_encode(classes, m->key_kind, m->keys[i], ok, msg); + uint64_t dv = 0; + if (*ok) dv = db_val_encode(classes, m->val_kind, m->vals[i], ok, msg); + d->kv[2 * i + 1] = dv; + if (!*ok) { + for (uint32_t j = 0; j <= i; j++) { + db_val_free(d->key_kind, d->kv[2 * j]); + db_val_free(d->val_kind, d->kv[2 * j + 1]); + } + free(d); + return 0; + } + } + return (uint64_t)(uintptr_t)d; + } + default: + *ok = 0; + *msg = "a garbage-collected value cannot be stored in a table field"; + return 0; + } +oom: + *ok = 0; + *msg = "out of memory encoding a row"; + return 0; +} + +/* Decode one engine slot back into a fresh VM value (the out-gate: always + * a copy). 0-with-*ok=0 = OOM; nil decodes as 0 with *ok=1. */ +static uint64_t db_val_decode(wo_rt *rt, uint8_t kind, uint64_t v, int *ok, + const char **msg) { + *ok = 1; + switch (kind) { + case WO_K_SCALAR: return v; + case WO_K_TEXT: { + if (!v) return 0; + const db_text *t = (const db_text *)(uintptr_t)v; + wo_str *s = wo_str_new(rt, t->bytes, t->len); + if (!s) goto oom; + return (uint64_t)(uintptr_t)s; + } + case WO_K_OWNED: { + if (!v) return 0; + const db_rec *r = (const db_rec *)(uintptr_t)v; + wo_hdr *o = wo_obj_new(rt, r->class_id); + if (!o) goto oom; + const wo_classdesc *c = &rt->classes[r->class_id]; + uint64_t *f = wo_fields(o); + for (uint32_t i = 0; i < c->field_cnt; i++) { + f[i] = db_val_decode(rt, c->kinds[i], r->slots[i], ok, msg); + if (!*ok) return 0; /* partial object: rt teardown reclaims (test scope) */ + } + return (uint64_t)(uintptr_t)o; + } + case WO_K_MULTI: { + if (!v) return 0; + const db_multi *d = (const db_multi *)(uintptr_t)v; + wo_multi *m = wo_multi_new(rt, d->elem_kind); + if (!m) goto oom; + for (uint32_t i = 0; i < d->len; i++) { + uint64_t ev = db_val_decode(rt, d->elem_kind, d->items[i], ok, msg); + if (!*ok || wo_multi_push(m, ev) != 0) goto oom; + } + return (uint64_t)(uintptr_t)m; + } + case WO_K_MAP: { + if (!v) return 0; + const db_map *d = (const db_map *)(uintptr_t)v; + wo_map *m = wo_map_new(rt, d->key_kind, d->val_kind); + if (!m) goto oom; + for (uint32_t i = 0; i < d->len; i++) { + uint64_t kv = db_val_decode(rt, d->key_kind, d->kv[2 * i], ok, msg); + uint64_t vv = 0; + if (*ok) vv = db_val_decode(rt, d->val_kind, d->kv[2 * i + 1], ok, msg); + uint64_t old; + if (!*ok || wo_map_set(m, kv, vv, &old) < 0) goto oom; + } + return (uint64_t)(uintptr_t)m; + } + default: return 0; /* GCREF never stored, so never decoded */ + } +oom: + *ok = 0; + *msg = "out of memory decoding a row"; + return 0; +} + +/* ---- id hash (open addressing, pow2, id -> global slot + 1) ----------- */ + +static uint64_t hmix(uint64_t x) { /* splitmix64 finalizer */ + x += 0x9e3779b97f4a7c15ull; + x = (x ^ (x >> 30)) * 0xbf58476d1ce4e5b9ull; + x = (x ^ (x >> 27)) * 0x94d049bb133111ebull; + return x ^ (x >> 31); +} + +/* Ids are never 0 (0 spells "empty bucket") and never reused, so all-ones + * can never collide with a live id — it marks a deleted bucket that probes + * walk straight past. */ +#define H_DELETED ((uint64_t)-1) + +static int hgrow(db_table *t) { + size_t ncap = t->hcap ? t->hcap * 2 : 64; + uint64_t *nk = calloc(ncap, 8), *nv = calloc(ncap, 8); + if (!nk || !nv) { + free(nk); + free(nv); + return -1; + } + for (size_t i = 0; i < t->hcap; i++) { + if (!t->hkeys[i] || t->hkeys[i] == H_DELETED) continue; + size_t j = hmix(t->hkeys[i]) & (ncap - 1); + while (nk[j]) j = (j + 1) & (ncap - 1); + nk[j] = t->hkeys[i]; + nv[j] = t->hvals[i]; + } + free(t->hkeys); + free(t->hvals); + t->hkeys = nk; + t->hvals = nv; + t->hcap = ncap; + return 0; +} + +static int hput(db_table *t, uint64_t id, uint64_t slot1) { + if (t->hlen * 10 >= t->hcap * 7 && hgrow(t) != 0) return -1; + size_t j = hmix(id) & (t->hcap - 1); + while (t->hkeys[j] && t->hkeys[j] != id) j = (j + 1) & (t->hcap - 1); + if (!t->hkeys[j]) t->hlen++; + t->hkeys[j] = id; + t->hvals[j] = slot1; + return 0; +} + +static uint64_t hget(const db_table *t, uint64_t id) { + if (!t->hcap) return 0; + size_t j = hmix(id) & (t->hcap - 1); + while (t->hkeys[j]) { + if (t->hkeys[j] == id) return t->hvals[j]; + j = (j + 1) & (t->hcap - 1); + } + return 0; +} + +static void hdel(db_table *t, uint64_t id) { + if (!t->hcap) return; + size_t j = hmix(id) & (t->hcap - 1); + while (t->hkeys[j]) { + if (t->hkeys[j] == id) { + t->hkeys[j] = H_DELETED; + t->hvals[j] = 0; + return; + } + j = (j + 1) & (t->hcap - 1); + } +} + +/* ---- tables and rows ---------------------------------------------------- */ + +int wo_db_init(wo_db *db, const wo_classdesc *classes, uint32_t class_cnt, + uint32_t shard, uint32_t nshards) { + if (!nshards || shard >= nshards) return -1; + memset(db, 0, sizeof(*db)); + db->classes = classes; + db->class_cnt = class_cnt; + db->shard = shard; + db->nshards = nshards; + db->tables = calloc(class_cnt ? class_cnt : 1, sizeof(db_table)); + if (!db->tables) return -1; + g_classes = classes; + return 0; +} + +static void table_destroy(wo_db *db, db_table *t) { + /* free every live row's engine-owned values, then the slabs */ + const wo_classdesc *c = &db->classes[t->class_id]; + for (uint32_t s = 0; s < t->slab_cnt; s++) { + for (uint32_t i = 0; i < DB_SLAB_ROWS; i++) { + uint32_t g = s * DB_SLAB_ROWS + i; + if (!(t->bitmap[g >> 6] & (1ull << (g & 63)))) continue; + db_row *r = (db_row *)(t->slabs[s] + (size_t)i * t->row_size); + for (uint32_t f = 0; f < c->field_cnt; f++) + db_val_free(c->kinds[f], r->slots[f]); + } + free(t->slabs[s]); + } + free(t->slabs); + free(t->bitmap); + free(t->free_slots); + free(t->hkeys); + free(t->hvals); +} + +void wo_db_destroy(wo_db *db) { + if (!db->tables) return; + for (uint32_t i = 0; i < db->class_cnt; i++) + if (db->tables[i].slab_cnt || db->tables[i].hkeys) table_destroy(db, &db->tables[i]); + free(db->tables); + db->tables = NULL; +} + +static db_table *table_of(wo_db *db, uint32_t class_id) { + if (class_id >= db->class_cnt) return NULL; + db_table *t = &db->tables[class_id]; + if (!t->row_size) { /* lazy init on first touch */ + t->class_id = class_id; + t->row_size = sizeof(db_row) + (size_t)db->classes[class_id].field_cnt * 8u; + t->next_id = db->shard + 1; /* S+1, then += N: interleaved, local-only */ + } + return t; +} + +static db_row *slot_row(db_table *t, uint32_t g) { + return (db_row *)(t->slabs[g / DB_SLAB_ROWS] + (size_t)(g % DB_SLAB_ROWS) * t->row_size); +} + +/* Pick the slot a new row lands in: recycled first, else the next free bit, + * else grow a slab. Returns the global slot or UINT32_MAX on OOM. */ +static uint32_t slot_alloc(db_table *t) { + if (t->free_cnt) return t->free_slots[--t->free_cnt]; + uint32_t total = t->slab_cnt * DB_SLAB_ROWS; + for (uint32_t g = 0; g < total; g++) /* cheap at slab granularity: only + reached when free list is empty, and the bitmap scan is bounded by + one word test per 64 slots */ + if (!(t->bitmap[g >> 6] & (1ull << (g & 63)))) return g; + /* grow */ + if (t->slab_cnt == t->slab_cap) { + uint32_t ncap = t->slab_cap ? t->slab_cap * 2 : 4; + uint8_t **ns = realloc(t->slabs, (size_t)ncap * sizeof(uint8_t *)); + if (!ns) return UINT32_MAX; + t->slabs = ns; + t->slab_cap = ncap; + } + uint8_t *slab = malloc((size_t)DB_SLAB_ROWS * t->row_size); + if (!slab) return UINT32_MAX; + size_t nwords = ((size_t)(t->slab_cnt + 1) * DB_SLAB_ROWS + 63) / 64; + uint64_t *nb = realloc(t->bitmap, nwords * 8); + if (!nb) { + free(slab); + return UINT32_MAX; + } + memset(nb + ((size_t)t->slab_cnt * DB_SLAB_ROWS) / 64, 0, + (nwords - ((size_t)t->slab_cnt * DB_SLAB_ROWS) / 64) * 8); + t->bitmap = nb; + t->slabs[t->slab_cnt] = slab; + return t->slab_cnt++ * DB_SLAB_ROWS; +} + +uint64_t wo_row_insert(wo_db *db, uint32_t class_id, const uint64_t *vals, + const char **msg) { + db_table *t = table_of(db, class_id); + if (!t) { + *msg = "no such class"; + return 0; + } + const wo_classdesc *c = &db->classes[class_id]; + uint32_t g = slot_alloc(t); + if (g == UINT32_MAX) { + *msg = "out of memory growing a table"; + return 0; + } + db_row *r = slot_row(t, g); + r->class_id = class_id; + r->flags = 0; + int ok = 1; + uint32_t i = 0; + for (; i < c->field_cnt; i++) { + r->slots[i] = db_val_encode(db->classes, c->kinds[i], vals[i], &ok, msg); + if (!ok) break; + } + if (!ok) { + for (uint32_t j = 0; j < i; j++) db_val_free(c->kinds[j], r->slots[j]); + /* slot never became live: recycle it (bitmap bit was never set) */ + if (t->free_cnt == t->free_cap) { + uint32_t ncap = t->free_cap ? t->free_cap * 2 : 16; + uint32_t *nf = realloc(t->free_slots, (size_t)ncap * 4); + if (nf) { + t->free_slots = nf; + t->free_cap = ncap; + } + } + if (t->free_cnt < t->free_cap) t->free_slots[t->free_cnt++] = g; + return 0; + } + r->id = t->next_id; + t->next_id += db->nshards; + if (hput(t, r->id, (uint64_t)g + 1) != 0) { + for (uint32_t j = 0; j < c->field_cnt; j++) db_val_free(c->kinds[j], r->slots[j]); + *msg = "out of memory indexing a row"; + return 0; + } + t->bitmap[g >> 6] |= 1ull << (g & 63); + t->count++; + /* INDEX HOOK (Task 4): secondary indexes update here, inside the choke + point, never anywhere else. */ + return r->id; +} + +db_row *wo_row_ptr(wo_db *db, uint32_t class_id, uint64_t id) { + if (class_id >= db->class_cnt) return NULL; + db_table *t = &db->tables[class_id]; + if (!t->row_size) return NULL; + uint64_t s1 = hget(t, id); + if (!s1) return NULL; + return slot_row(t, (uint32_t)(s1 - 1)); +} + +int wo_row_read(wo_db *db, wo_rt *rt, uint32_t class_id, uint64_t id, + uint64_t *out_vals, const char **msg) { + db_row *r = wo_row_ptr(db, class_id, id); + if (!r) return -1; + const wo_classdesc *c = &db->classes[class_id]; + int ok = 1; + for (uint32_t i = 0; i < c->field_cnt; i++) { + out_vals[i] = db_val_decode(rt, c->kinds[i], r->slots[i], &ok, msg); + if (!ok) return -2; + } + return 0; +} + +int wo_row_remove(wo_db *db, uint32_t class_id, uint64_t id) { + if (class_id >= db->class_cnt) return -1; + db_table *t = &db->tables[class_id]; + if (!t->row_size) return -1; + uint64_t s1 = hget(t, id); + if (!s1) return -1; + uint32_t g = (uint32_t)(s1 - 1); + db_row *r = slot_row(t, g); + /* INDEX HOOK (Task 4): secondary indexes remove here, before the row's + values die. */ + const wo_classdesc *c = &db->classes[class_id]; + for (uint32_t i = 0; i < c->field_cnt; i++) db_val_free(c->kinds[i], r->slots[i]); + t->bitmap[g >> 6] &= ~(1ull << (g & 63)); + hdel(t, id); + t->count--; + if (t->free_cnt == t->free_cap) { + uint32_t ncap = t->free_cap ? t->free_cap * 2 : 16; + uint32_t *nf = realloc(t->free_slots, (size_t)ncap * 4); + if (!nf) return 0; /* slot simply not recycled; bitmap still frees it */ + t->free_slots = nf; + t->free_cap = ncap; + } + t->free_slots[t->free_cnt++] = g; + return 0; +} diff --git a/database/src/table.h b/database/src/table.h new file mode 100644 index 0000000..2b98120 --- /dev/null +++ b/database/src/table.h @@ -0,0 +1,132 @@ +/* table.h — class-shaped row storage (iteration 9, Task 1). + * + * The engine and the VM heap are two memory worlds crossed only by copy + * (the 9b design's section 6): a row stores NO VM pointer. Every field + * lands in one 8-byte slot, kind-driven: + * + * SCALAR the 8 bytes themselves (WO_NIL_SCALAR spells a ?scalar's nil) + * TEXT engine-owned db_text* (0 = nil) + * OWNED engine-owned db_rec* — the object flattened by value, + * recursively, through these same rules (0 = nil) + * MULTI engine-owned db_multi* — elements encoded element-wise + * MAP engine-owned db_map* — keys and values encoded pair-wise + * GCREF never stored: the compiler rejects it (the GC bulkhead); + * the engine refuses it defensively as an encode error + * + * `ref T` is a SCALAR at this layer — the target row's id, an ordinary + * number the compiler produced; the engine learns nothing about it until + * the FK checks (9b plan, Task 3). + * + * Row layout: a 16-byte header (id, class, flags) then field_cnt 8-byte + * slots — deliberately the VM object layout's shape, so encode/decode walk + * the same class-table kinds the VM walks. Rows live in per-class SLABS + * (fixed-count, malloc'd, never moved: a row's address is stable for its + * lifetime, which is what lets 9b hand out loop-scoped row views). A + * per-table bitmap tracks occupancy; removed slots go on a free list and + * are reused before any slab grows. The id->row map is an open-addressing + * hash owned by the table. + * + * Id discipline (the c-runtime plan's shipped behavior): per table, per + * shard, ids interleave — shard S of N allocates S+1, S+1+N, S+1+2N, … — + * so creation is coordination-free and a row's owner shard is (id-1) % N. + * Milestone runs at N=1 (iteration 8 not yet landed); everything here is + * N-parametric and degenerates cleanly. + * + * CHOKE POINT DOCTRINE: wo_row_insert / wo_row_remove are the only paths + * that touch storage. Task 4's secondary indexes hook exactly these two + * functions; anything else mutating a slab is a defect by definition. + */ +#ifndef WO_TABLE_H +#define WO_TABLE_H + +#include "obj.h" /* wo_rt, wo_classdesc, kinds, wo_str, containers */ + +/* ---- engine-owned value shapes (all malloc'd, all reachable only from + * row slots, all freed through db_val_free) ---- */ + +typedef struct db_text { + uint32_t len; + char bytes[]; /* len bytes, no NUL */ +} db_text; + +typedef struct db_rec { /* an owned object flattened by value */ + uint32_t class_id; /* index into the SAME class table the VM uses */ + uint32_t _pad; + uint64_t slots[]; /* field_cnt slots, encoded by these rules */ +} db_rec; + +typedef struct db_multi { + uint8_t elem_kind; + uint32_t len; + uint64_t items[]; +} db_multi; + +typedef struct db_map { + uint8_t key_kind, val_kind; + uint32_t len; + uint64_t kv[]; /* len pairs: k0 v0 k1 v1 … */ +} db_map; + +/* ---- rows and tables ---- */ + +typedef struct db_row { + uint64_t id; + uint32_t class_id; + uint32_t flags; /* reserved (0) */ + uint64_t slots[]; +} db_row; + +#define DB_SLAB_ROWS 256u + +typedef struct db_table { + uint32_t class_id; + size_t row_size; /* 16 + field_cnt * 8 */ + /* slabs of DB_SLAB_ROWS rows each; addresses stable forever */ + uint8_t **slabs; + uint32_t slab_cnt, slab_cap; + uint64_t *bitmap; /* one bit per slot, slab-major */ + /* removed slots, reused LIFO before any slab grows */ + uint32_t *free_slots; + uint32_t free_cnt, free_cap; + uint64_t next_id; /* next id THIS shard hands out for this table */ + uint64_t count; /* live rows */ + /* id -> (global slot + 1); 0 = empty. Open addressing, pow2. */ + uint64_t *hkeys; + uint64_t *hvals; + size_t hcap, hlen; +} db_table; + +typedef struct wo_db { + const wo_classdesc *classes; + uint32_t class_cnt; + uint32_t shard, nshards; /* S of N; ids interleave S+1, S+1+N, … */ + db_table *tables; /* class_cnt entries, created lazily on first insert */ +} wo_db; + +/* 0 ok, -1 alloc failure. nshards >= 1, shard < nshards. */ +int wo_db_init(wo_db *db, const wo_classdesc *classes, uint32_t class_cnt, + uint32_t shard, uint32_t nshards); +void wo_db_destroy(wo_db *db); + +/* Insert: encode field_cnt VM values (register words, kinds from the class + * table) into a fresh row. Returns the new id, or 0 with *msg set (OOM, or + * a GCREF field — which the compiler should have refused upstream). */ +uint64_t wo_row_insert(wo_db *db, uint32_t class_id, const uint64_t *vals, + const char **msg); + +/* Read: decode the row's fields into VM values freshly allocated from + * [rt] — always copies, never a pointer into the slab (the out-gate). + * 0 ok, -1 no such row, -2 OOM (*msg set). */ +int wo_row_read(wo_db *db, wo_rt *rt, uint32_t class_id, uint64_t id, + uint64_t *out_vals, const char **msg); + +/* Remove: free the row's engine-owned field values, clear the slot, recycle + * it. 0 ok, -1 no such row. */ +int wo_row_remove(wo_db *db, uint32_t class_id, uint64_t id); + +/* Borrowed row pointer for engine-internal callers (the WAL writes a row's + * encoded bytes; indexes read key slots). NULL = no such row. NEVER handed + * to the VM. */ +db_row *wo_row_ptr(wo_db *db, uint32_t class_id, uint64_t id); + +#endif /* WO_TABLE_H */ diff --git a/docs/plan/oop-vm/04-db-binding.md b/docs/plan/oop-vm/04-db-binding.md new file mode 100644 index 0000000..f038545 --- /dev/null +++ b/docs/plan/oop-vm/04-db-binding.md @@ -0,0 +1,79 @@ +# DB binding — row format, id discipline, WAL layout, query subset + +> Normative companion to the engine plan +> ([`2026-08-01-db-engine-binding.md`](../../superpowers/plans/2026-08-01-db-engine-binding.md)), +> the way `00-wob-format.md` is normative for the image. Grows with the +> plan's tasks; this revision covers **Task 1 (row storage)**. Memory-safety +> doctrine lives in the 9b design's section 6 (the copy bulkhead) — this doc +> is the *format*. + +## Two memory worlds, one crossing rule + +Rows store **no VM pointer**, ever. Values cross from VM heap to row storage +by copy on insert, and back by copy on read (`wo_row_read` allocates fresh VM +values from the shard's runtime). The engine's own allocations are plain +malloc — never the VM arena, so table growth cannot eat the program's heap +cap, and a heap-exhausted program can still read its data. + +## Row format + +``` +row := header slots +header := id u64 | class_id u32 | flags u32 (16 bytes) +slots := field_cnt × u64, declaration order (the VM object shape) +``` + +One 8-byte slot per field, kind-driven — the same kind bytes the `.wob` +class table carries, walked the same way the VM walks them: + +| kind | slot holds | engine-owned shape | +| --- | --- | --- | +| `SCALAR` | the 8 bytes themselves | — (`WO_NIL_SCALAR` spells a `?scalar` nil) | +| `TEXT` | pointer, 0 = nil | `db_text { len u32; bytes[] }` | +| `OWNED` | pointer, 0 = nil | `db_rec { class_id u32; slots[] }` — flattened by value, recursively through these same rules | +| `MULTI` | pointer, 0 = nil | `db_multi { elem_kind u8; len u32; items[] }`, elements encoded element-wise | +| `MAP` | pointer, 0 = nil | `db_map { key_kind, val_kind u8; len u32; kv pairs }` | +| `GCREF` | **never stored** | compile error upstream (the GC bulkhead); the engine refuses it defensively as an encode error | + +`ref T` is a `SCALAR` at this layer — the target row's id. The engine learns +what it references only when the FK checks land (9b plan, Task 3). + +## Storage + +Per shard, per class, created lazily on first insert: + +- **Slabs** of 256 rows (`DB_SLAB_ROWS`), malloc'd, **never moved or freed + while the table lives** — a row's address is stable for its lifetime, + which is the property 9b's loop-scoped row views stand on. +- An **occupancy bitmap** (one bit per slot, slab-major) and a LIFO + **free-slot list**: removal recycles the slot; a recycled slot is always + used before a new slab grows. Ids are never reused; slots are. +- The **primary index**: an open-addressing hash, id → slot, splitmix64 + finalizer, power-of-two capacity, 0.7 load, tombstoned deletes (ids are + never 0 and never reused, so the all-ones sentinel cannot collide). + +## Id discipline + +Per table, per shard: shard S of N allocates `S+1, S+1+N, S+1+2N, …` — the +c-runtime plan's shipped interleave. Creation is coordination-free; a row's +owner shard is `(id-1) % N`. Milestone 1 runs at N=1 and everything +degenerates to `1, 2, 3, …`. Id 0 does not exist (it is the hash's "empty" +and the `?ref`'s nil). + +## Choke points + +`wo_row_insert` and `wo_row_remove` are the only functions that mutate a +table. Task 4's secondary indexes hook exactly these two sites (marked +`INDEX HOOK` in `database/src/table.c`); the WAL (Task 2) stages its record +beside the same calls. Anything else touching a slab is a defect by +definition — the doctrine the Rust engine learned and this engine enforces. + +## Still to come in this document + +- **Task 2**: WAL record framing (`length | crc | payload | commit-mark`), + payload encoding for typed rows, group-commit ordering, replay rules, + torn-tail handling, the `wal-check` oracle. +- **Task 3**: the `insert` statement's builtin ids (appended to + `00-wob-format.md`'s builtin table) and execution contract. +- **Task 4**: secondary-index format, `@unique` trap code. +- **Task 5**: the select subset and its builtins. diff --git a/docs/superpowers/plans/2026-08-01-db-engine-binding.md b/docs/superpowers/plans/2026-08-01-db-engine-binding.md index ffa86c5..97acc6d 100644 --- a/docs/superpowers/plans/2026-08-01-db-engine-binding.md +++ b/docs/superpowers/plans/2026-08-01-db-engine-binding.md @@ -1,6 +1,6 @@ # DB Engine Binding Implementation Plan -> **Status: ⬜ pending** (story iteration 9) — class-shaped tables, typed WAL + recovery, `insert`/`select` execution. Story iteration 9b (`@table` relations + language-integrated query) follows it and needs a spec brainstormed first. Board: [00-status.md](../../00-status.md) +> **Status: 🔄 in progress — Task 1 done 2026-08-15** (story iteration 9) — class-shaped tables, typed WAL + recovery, `insert`/`select` execution. Story iteration 9b (`@table` relations + language-integrated query) follows it and needs a spec brainstormed first. Board: [00-status.md](../../00-status.md) > **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. > @@ -48,9 +48,20 @@ sanitizers included. `database/` gets its own CODE-LOGIC.md as code lands. **Concept & reason:** generalize phase B. Per shard, per class: a slab of fixed-size row slots sized from the class's field count (16-byte row header — id, class, flags — plus the same 8-byte slots the VM object layout uses, so a row and an object share their field encoding; text and container fields store engine-owned copies, not VM pointers). An allocation bitmap per slab; slab growth by arena extension. Id allocation interleaved per shard for coordination-free global uniqueness (shipped phase-A behavior). Row create/read/remove go through one API that Task 4's indexes hook — the doctrine choke point. The binding doc pins the row format, the field-encoding rules (what happens to each of the six kinds when a value crosses from VM heap to row storage — scalars copy, texts copy, owned objects flatten by value, `@gc` references are a compile error in stored fields already, `ref` is an id, containers copy element-wise), and the query subset promised by Task 5. -- [ ] Failing tests: create/read/remove round-trips across kinds; id interleave across shards; slab growth; removal reuses slots. -- [ ] Implement; ASan green. Write the binding doc. -- [ ] Record commit draft: `feat(runtime): class-shaped row storage — per-shard per-class slabs from .wob class table, VM-compatible field encoding, interleaved id allocation, single choke-point row API; docs/plan/oop-vm/04-db-binding.md.` +- [x] Tests first: round-trips across every kind (Text/owned-nested/multi/map + copies proven by mutating the originals), nil encodings incl. + `WO_NIL_SCALAR`, id interleave at N=3, slab growth past three slabs + with stable addresses, removal reuses the slot while never reusing the + id, misuse (unknown class, double remove). `test_table` 827/0 under + ASan+UBSan. +- [x] Implemented in `database/src/table.{c,h}` (the 2026-08-15 directory + decision), linked into every wovm and test binary via the Makefile's + `DBSRC`. Binding doc written (`docs/plan/oop-vm/04-db-binding.md`: + row format, encoding table, id discipline, choke points). All prior + gates stay green with the engine linked (oop-e2e 71/0, log-watcher + 7/0) — it is dead code until Task 3 wires the first builtin. +- [x] Committed locally (2026-08-15). N=1 today: iteration 8 has not landed, + so everything is N-parametric and tested at N=3 through the API. ### Task 2: Typed WAL + boot replay diff --git a/runtime/Makefile b/runtime/Makefile index 1535fbb..bb5744c 100644 --- a/runtime/Makefile +++ b/runtime/Makefile @@ -8,8 +8,13 @@ wo-rt: wo-rt.c # ---- wovm VM core (src/) + unit tests (test/) ---- # Each test/test_*.c builds into its own ASan+UBSan binary linked against # every src/*.c except main.c; `make test` runs them all. -VMSRC := $(filter-out src/main.c,$(wildcard src/*.c)) -VMHDR := $(wildcard src/*.h) $(wildcard test/*.h) +# the database engine lives in its own top-level directory (iteration 9; +# user decision 2026-08-15) and is statically linked into every wovm and +# every test binary — one binary, unchanged +DBSRC := $(wildcard ../database/src/*.c) +DBHDR := $(wildcard ../database/src/*.h) +VMSRC := $(filter-out src/main.c,$(wildcard src/*.c)) $(DBSRC) +VMHDR := $(wildcard src/*.h) $(wildcard test/*.h) $(DBHDR) TESTS := $(wildcard test/test_*.c) TESTBIN := $(TESTS:test/%.c=build/%) TCFLAGS := -std=c11 -Wall -Wextra -Werror -g -O1 \ @@ -21,7 +26,7 @@ build: TESTHELP := $(wildcard test/wob_build.c) build/%: test/%.c $(VMSRC) $(TESTHELP) $(VMHDR) | build - $(CC) $(TCFLAGS) -Isrc -Itest -o $@ $< $(TESTHELP) $(VMSRC) + $(CC) $(TCFLAGS) -Isrc -Itest -I../database/src -o $@ $< $(TESTHELP) $(VMSRC) test: $(TESTBIN) @for t in $(TESTBIN); do echo "== $$t"; ./$$t || exit 1; done @@ -31,20 +36,20 @@ test: $(TESTBIN) ISOBIN := $(TESTS:test/%.c=build/iso_%) build/iso_%: test/%.c $(VMSRC) $(TESTHELP) $(VMHDR) | build - $(CC) $(TCFLAGS) -DWO_ISO_C -Isrc -Itest -o $@ $< $(TESTHELP) $(VMSRC) + $(CC) $(TCFLAGS) -DWO_ISO_C -Isrc -Itest -I../database/src -o $@ $< $(TESTHELP) $(VMSRC) test-iso: $(ISOBIN) @for t in $(ISOBIN); do echo "== $$t"; ./$$t || exit 1; done # the wovm binary (plain optimized build; the test suite is the ASan gate) wovm: src/main.c $(VMSRC) $(VMHDR) - $(CC) $(CFLAGS) -Isrc -o $@ src/main.c $(VMSRC) + $(CC) $(CFLAGS) -Isrc -I../database/src -o $@ src/main.c $(VMSRC) # ASan+UBSan wovm, same flags as the unit tests, for corpus fixtures that # need a sanitizer to prove a free actually happened (gc/ cycle fixtures) — # tasks 3/4 hand-built this each time because it didn't exist yet build/wovm_asan: src/main.c $(VMSRC) $(VMHDR) | build - $(CC) $(TCFLAGS) -Isrc -o $@ src/main.c $(VMSRC) + $(CC) $(TCFLAGS) -Isrc -I../database/src -o $@ src/main.c $(VMSRC) wovm-asan: build/wovm_asan diff --git a/runtime/test/test_table.c b/runtime/test/test_table.c new file mode 100644 index 0000000..aa54ad4 --- /dev/null +++ b/runtime/test/test_table.c @@ -0,0 +1,182 @@ +/* test_table — iteration 9 Task 1: class-shaped row storage. + * Round-trips across kinds, nil encodings, id interleave across shards, + * slab growth past one slab, slot reuse after removal, and the out-gate + * invariant (a read hands back FRESH VM values, never slab pointers). */ +#include + +#include "cont.h" +#include "gc.h" +#include "obj.h" +#include "t.h" +#include "table.h" + +/* class 0: Addr { city: Text } + * class 1: Emp { name: Text, salary: Int(scalar), addr: OWNED Addr, + * tags: multi Text, meta: map } + * class 2: Tiny { n: scalar } (slab-growth workhorse) */ +static const uint8_t addr_kinds[] = {WO_K_TEXT}; +static const uint8_t emp_kinds[] = {WO_K_TEXT, WO_K_SCALAR, WO_K_OWNED, WO_K_MULTI, + WO_K_MAP}; +static const uint8_t tiny_kinds[] = {WO_K_SCALAR}; +static const wo_classdesc CLASSES[] = { + {.name = 0, .flags = 0, .field_cnt = 1, .kinds = addr_kinds}, + {.name = 0, .flags = 0, .field_cnt = 5, .kinds = emp_kinds}, + {.name = 0, .flags = 0, .field_cnt = 1, .kinds = tiny_kinds}, +}; + +static void test_roundtrip_all_kinds(void) { + wo_rt rt; + T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 3), 0); + wo_db db; + T_EQ(wo_db_init(&db, CLASSES, 3, 0, 1), 0); + const char *msg = ""; + + /* build the VM-side value: Emp{"Asha", 9200000, Addr{"Pune"}, ["a","b"], {"k": 7}} */ + wo_str *name = wo_str_new(&rt, "Asha", 4); + wo_hdr *addr = wo_obj_new(&rt, 0); + wo_fields(addr)[0] = (uint64_t)(uintptr_t)wo_str_new(&rt, "Pune", 4); + wo_multi *tags = wo_multi_new(&rt, WO_K_TEXT); + wo_multi_push(tags, (uint64_t)(uintptr_t)wo_str_new(&rt, "a", 1)); + wo_multi_push(tags, (uint64_t)(uintptr_t)wo_str_new(&rt, "b", 1)); + wo_map *meta = wo_map_new(&rt, WO_K_TEXT, WO_K_SCALAR); + uint64_t old; + wo_map_set(meta, (uint64_t)(uintptr_t)wo_str_new(&rt, "k", 1), 7, &old); + + uint64_t vals[5] = {(uint64_t)(uintptr_t)name, 9200000, + (uint64_t)(uintptr_t)addr, (uint64_t)(uintptr_t)tags, + (uint64_t)(uintptr_t)meta}; + uint64_t id = wo_row_insert(&db, 1, vals, &msg); + T_EQ(id, 1); /* shard 0 of 1: first id is 1 */ + + /* the row stored COPIES: mutate the VM originals, then read back */ + name->data[0] = 'X'; + ((wo_str *)(uintptr_t)wo_fields(addr)[0])->data[0] = 'X'; + + uint64_t out[5] = {0}; + T_EQ(wo_row_read(&db, &rt, 1, id, out, &msg), 0); + wo_str *rname = (wo_str *)(uintptr_t)out[0]; + T_EQ(rname->len, 4); + T_CHECK(memcmp(rname->data, "Asha", 4) == 0); /* not "Xsha" */ + T_CHECK(rname != name); /* fresh allocation */ + T_EQ(out[1], 9200000); + wo_hdr *raddr = (wo_hdr *)(uintptr_t)out[2]; + T_CHECK(raddr != addr); + wo_str *rcity = (wo_str *)(uintptr_t)wo_fields(raddr)[0]; + T_CHECK(memcmp(rcity->data, "Pune", 4) == 0); /* not "Xune" */ + wo_multi *rtags = (wo_multi *)(uintptr_t)out[3]; + T_EQ(rtags->len, 2); + T_CHECK(memcmp(((wo_str *)(uintptr_t)rtags->items[1])->data, "b", 1) == 0); + wo_map *rmeta = (wo_map *)(uintptr_t)out[4]; + uint64_t got = 0; + wo_str *k = wo_str_new(&rt, "k", 1); + T_EQ(wo_map_get(rmeta, (uint64_t)(uintptr_t)k, &got), 0); + T_EQ(got, 7); + + /* nil TEXT / nil OWNED / WO_NIL_SCALAR round-trip */ + uint64_t nilvals[5] = {0, WO_NIL_SCALAR, 0, 0, 0}; + uint64_t id2 = wo_row_insert(&db, 1, nilvals, &msg); + T_EQ(id2, 2); + uint64_t out2[5] = {(uint64_t)-1, 0, (uint64_t)-1, (uint64_t)-1, (uint64_t)-1}; + T_EQ(wo_row_read(&db, &rt, 1, id2, out2, &msg), 0); + T_EQ(out2[0], 0); + T_EQ(out2[1], WO_NIL_SCALAR); + T_EQ(out2[2], 0); + T_EQ(out2[3], 0); + + /* the VM-side values are containers with malloc'd backing arrays: + real drops, not arena teardown, are what frees them */ + wo_drop_obj(&rt, (wo_hdr *)name); + wo_drop_obj(&rt, addr); + wo_drop_obj(&rt, (wo_hdr *)tags); + wo_drop_obj(&rt, (wo_hdr *)meta); + wo_drop_obj(&rt, (wo_hdr *)k); + for (int i = 0; i < 5; i++) + if (i != 1 && out[i]) wo_drop_obj(&rt, (wo_hdr *)(uintptr_t)out[i]); + wo_db_destroy(&db); + wo_rt_destroy(&rt); +} + +static void test_id_interleave_across_shards(void) { + const char *msg = ""; + wo_db a, b, c; + T_EQ(wo_db_init(&a, CLASSES, 3, 0, 3), 0); + T_EQ(wo_db_init(&b, CLASSES, 3, 1, 3), 0); + T_EQ(wo_db_init(&c, CLASSES, 3, 2, 3), 0); + uint64_t v[1] = {42}; + T_EQ(wo_row_insert(&a, 2, v, &msg), 1); /* shard 0: 1, 4, 7 */ + T_EQ(wo_row_insert(&a, 2, v, &msg), 4); + T_EQ(wo_row_insert(&b, 2, v, &msg), 2); /* shard 1: 2, 5 */ + T_EQ(wo_row_insert(&b, 2, v, &msg), 5); + T_EQ(wo_row_insert(&c, 2, v, &msg), 3); /* shard 2: 3, 6 */ + T_EQ(wo_row_insert(&c, 2, v, &msg), 6); + /* owner-shard discipline: (id-1) % N names the shard */ + T_EQ((4 - 1) % 3, 0); + T_EQ((5 - 1) % 3, 1); + T_EQ((6 - 1) % 3, 2); + /* shard/nshards misuse refused */ + wo_db bad; + T_EQ(wo_db_init(&bad, CLASSES, 3, 3, 3), -1); + T_EQ(wo_db_init(&bad, CLASSES, 3, 0, 0), -1); + wo_db_destroy(&a); + wo_db_destroy(&b); + wo_db_destroy(&c); +} + +static void test_slab_growth_and_reuse(void) { + wo_rt rt; + T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 3), 0); + const char *msg = ""; + wo_db db; + T_EQ(wo_db_init(&db, CLASSES, 3, 0, 1), 0); + /* three slabs' worth of Tiny rows */ + enum { N = 3 * DB_SLAB_ROWS + 5 }; + uint64_t ids[N]; + for (uint32_t i = 0; i < N; i++) { + uint64_t v[1] = {i}; + ids[i] = wo_row_insert(&db, 2, v, &msg); + T_CHECK(ids[i] == i + 1); + } + T_EQ(db.tables[2].slab_cnt, 4); + T_EQ(db.tables[2].count, N); + /* every row readable after growth (addresses were never moved) */ + uint64_t out[1]; + T_EQ(wo_row_read(&db, &rt, 2, ids[0], out, &msg), 0); + T_EQ(out[0], 0); + T_EQ(wo_row_read(&db, &rt, 2, ids[N - 1], out, &msg), 0); + T_EQ(out[0], N - 1); + + /* remove a middle row: its slot is reused BEFORE any new slab grows */ + db_row *victim = wo_row_ptr(&db, 2, ids[100]); + T_CHECK(victim != NULL); + T_EQ(wo_row_remove(&db, 2, ids[100]), 0); + T_EQ(wo_row_read(&db, &rt, 2, ids[100], out, &msg), -1); /* gone */ + T_EQ(wo_row_remove(&db, 2, ids[100]), -1); /* twice = miss */ + uint64_t v[1] = {777}; + uint64_t fresh = wo_row_insert(&db, 2, v, &msg); + T_CHECK(fresh > (uint64_t)N); /* ids never reused ... */ + db_row *fresh_row = wo_row_ptr(&db, 2, fresh); + T_CHECK(fresh_row == victim); /* ... but the SLOT is */ + T_EQ(db.tables[2].slab_cnt, 4); + wo_db_destroy(&db); + wo_rt_destroy(&rt); +} + +static void test_misuse(void) { + const char *msg = ""; + wo_db db; + T_EQ(wo_db_init(&db, CLASSES, 3, 0, 1), 0); + uint64_t v[1] = {1}; + T_EQ(wo_row_insert(&db, 99, v, &msg), 0); /* unknown class */ + T_CHECK(wo_row_ptr(&db, 99, 1) == NULL); + T_CHECK(wo_row_ptr(&db, 2, 1) == NULL); /* table never touched */ + T_EQ(wo_row_remove(&db, 2, 1), -1); + wo_db_destroy(&db); +} + +int main(void) { + test_roundtrip_all_kinds(); + test_id_interleave_across_shards(); + test_slab_growth_and_reuse(); + test_misuse(); + return t_report("test_table"); +}