test(db2-chain): cover flattening, and drop a ceiling no input could reach
- flattened row image is WO_WAL_UPDATE, not WO_WAL_INSERT: the row's original INSERT is already in a live log, so a second one for the same id is a duplicate replay refuses as corruption. INSERT is right only for compaction, which builds a fresh log - remove WO_CKPT_MAX_GARBAGE: with the absolute term at 64 MiB, garbage large enough to reach a 256 MiB ceiling has already tripped it, so the branch was unreachable. Postgres needs both constants because it thresholds on tuples with its pair at opposite ends; this thresholds on bytes, where one constant does both jobs - test_delta_chain_flattens_at_k: chain depth stays <= WO_DELTA_MAX_HOPS across 2K+2 updates, and a reset is observed - test_delta_chain_flatten_replays: a flattened chain replays correctly - test_keys_resident_indexed_across_flatten: a delta on an indexed column composes with flattening, checked at every step across the bound and after restart. Found no product defect - test_should_compact_absolute_and_ceiling: pins the absolute term, the boundary just under it, and the small-log case the ratio still governs - test_wal 5700 pass / 0 fail; wovm-test and woc-test green Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> (cherry picked from commit f93b5d9db753305c297e868d977670e6d703684c)
This commit is contained in:
parent
2cad84b7c6
commit
68dd3d88b8
5 changed files with 387 additions and 49 deletions
|
|
@ -502,12 +502,25 @@ int wo_wal_append_update(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id) {
|
|||
* but from a row the caller already holds — a keys-resident row whose payload
|
||||
* has been dropped has no slab image for wo_row_ptr to find, and the update
|
||||
* path is the one caller that legitimately has the full, post-update values in
|
||||
* hand because it folded them to maintain indexes. */
|
||||
* hand because it folded them to maintain indexes.
|
||||
*
|
||||
* WO_WAL_UPDATE, not WO_WAL_INSERT, and the distinction is not cosmetic.
|
||||
* Compaction's own flattening (stage_flattened_row) writes INSERT because it
|
||||
* builds a FRESH log in which each row appears exactly once. This function
|
||||
* appends into a LIVE log that already carries the row's original insert, so
|
||||
* an INSERT here is a duplicate id, and replay correctly refuses a duplicate as
|
||||
* corruption — caught by test_delta_chain_flatten_replays, which is the only
|
||||
* way this could have been caught: the record reads back perfectly in-process
|
||||
* and only fails on the next boot.
|
||||
*
|
||||
* UPDATE is exactly right anyway: replay applies it as remove-then-recreate,
|
||||
* which is what replacing a row wholesale means, and the fold terminates on
|
||||
* either full-row kind. */
|
||||
int wo_wal_append_row_image(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id,
|
||||
const db_row *r) {
|
||||
if (!r) return -1;
|
||||
wbuf p = {0};
|
||||
wput_u8(&p, WO_WAL_INSERT);
|
||||
wput_u8(&p, WO_WAL_UPDATE);
|
||||
wput_u32(&p, class_id);
|
||||
wput_u64(&p, id);
|
||||
const wo_classdesc *c = &db->classes[class_id];
|
||||
|
|
@ -618,15 +631,19 @@ int wo_wal_should_compact(uint64_t used, uint64_t last, uint64_t floor, uint32_t
|
|||
*
|
||||
* - without a triggering term, garbage that is large in bytes but small
|
||||
* relative to a big live set is never reclaimed;
|
||||
* - without a ceiling, a very large live set defers compaction forever,
|
||||
* which is what autovacuum_vacuum_max_threshold exists to stop. */
|
||||
* - a very large live set would otherwise defer compaction forever,
|
||||
* which is what autovacuum_vacuum_max_threshold exists to stop.
|
||||
*
|
||||
* ONE term does both jobs here, and a separate ceiling was tried and
|
||||
* removed as dead code. PostgreSQL needs two because its threshold counts
|
||||
* TUPLES and its two constants sit at opposite ends (base 50, max 1e8).
|
||||
* Ours counts BYTES, so "compact once garbage exceeds X" already caps
|
||||
* deferral: any ceiling above X is unreachable, and any ceiling below it
|
||||
* would be the trigger. Caught by trying to write a test that exercised
|
||||
* the ceiling and finding none could. */
|
||||
uint64_t garbage = used > last ? used - last : 0;
|
||||
if (garbage >= WO_CKPT_ABS_BYTES) return 1;
|
||||
|
||||
uint64_t trigger = last * (uint64_t)ratio;
|
||||
uint64_t ceiling = last + WO_CKPT_MAX_GARBAGE;
|
||||
if (trigger > ceiling) trigger = ceiling;
|
||||
return used > trigger;
|
||||
return used > last * (uint64_t)ratio;
|
||||
}
|
||||
|
||||
/* databasev2 3: how many records the dump stages before flushing.
|
||||
|
|
|
|||
|
|
@ -224,12 +224,13 @@ void wo_db_flush_drops(wo_db *db, wo_wal *w);
|
|||
* Past this much reclaimable garbage, compact regardless of proportion, so
|
||||
* garbage that is large absolutely but small against a big live set still gets
|
||||
* reclaimed.
|
||||
*
|
||||
* WO_CKPT_MAX_GARBAGE caps the proportional term, mirroring
|
||||
* `autovacuum_vacuum_max_threshold`, so a very large live set cannot defer
|
||||
* compaction indefinitely. */
|
||||
* It also does the job PostgreSQL splits into a second constant
|
||||
* (`autovacuum_vacuum_max_threshold`): capping how long a very large live set
|
||||
* can defer compaction. A separate ceiling was implemented and then removed as
|
||||
* unreachable — postgres needs two constants because it counts TUPLES with its
|
||||
* pair at opposite ends (50 and 1e8); this counts BYTES, so any ceiling above
|
||||
* this value can never fire and any below it would simply be the trigger. */
|
||||
#define WO_CKPT_ABS_BYTES (64u * 1024u * 1024u)
|
||||
#define WO_CKPT_MAX_GARBAGE (256u * 1024u * 1024u)
|
||||
|
||||
int wo_wal_should_compact(uint64_t used, uint64_t last, uint64_t floor, uint32_t ratio);
|
||||
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
---
|
||||
track: databasev2
|
||||
iteration: "11"
|
||||
status: in-progress
|
||||
status: complete
|
||||
readiness: ready
|
||||
---
|
||||
|
||||
|
|
@ -65,52 +65,76 @@ Read from PostgreSQL's source at `.dev/reference/postgresql`, not recalled:
|
|||
| --- | --- |
|
||||
| Tier 1 — the fold reports hop count | ✅ `wo_wal_fold_row_at` takes `hops_out`; the walk already visited each hop, so it costs nothing |
|
||||
| Tier 1 — the update branches on depth | ✅ `row_apply_field_keys` writes a full-row image past `WO_DELTA_MAX_HOPS` (16) instead of a delta |
|
||||
| Tier 1 — the chain-terminating write | ✅ `wo_wal_append_row_image`, encoded as `WO_WAL_INSERT` so replay, compaction and the fold need no change |
|
||||
| Tier 1 — the chain-terminating write | ✅ `wo_wal_append_row_image`, encoded as `WO_WAL_UPDATE` — see the correction below |
|
||||
| Tier 2 — absolute garbage term | ✅ `WO_CKPT_ABS_BYTES` (64 MiB) triggers regardless of proportion |
|
||||
| Tier 2 — proportional ceiling | ✅ `WO_CKPT_MAX_GARBAGE` (256 MiB) caps the ratio term |
|
||||
| **Tests** | ⏸ **DELIBERATELY HELD** — see below |
|
||||
| Tier 2 — proportional ceiling | ✅ **removed as dead code** — the absolute term already does this job |
|
||||
| **Tests** | ✅ four tests; `test_wal` 5700 pass / 0 fail, `just wovm-test` and `just woc-test` green |
|
||||
|
||||
**Verified by construction, not by test.** Both update entry points converge on
|
||||
`row_apply_field_keys` (`table.c:1039` and `:1319`), so one branch covers both.
|
||||
The re-point is transparent to flattening because `db.c` captures
|
||||
`wo_wal_next_offset(w)` *before* calling into `table.c` — it targets wherever
|
||||
the next record lands, delta or full row alike. And a fold that reaches a
|
||||
flattened record terminates there, so the next update sees depth 0.
|
||||
**Two corrections the tests forced, both worth recording.**
|
||||
|
||||
**What holding the tests costs, stated plainly.** The existing suite passes
|
||||
(36 suites, 0 failures) but that proves only that threading `hops_out` through
|
||||
the fold, `keys_fold_into` and their callers broke nothing — which is the change
|
||||
most likely to break something silently, so it is worth having. It does **not**
|
||||
exercise either new behaviour:
|
||||
*The flattened record is a `WO_WAL_UPDATE`, not an `INSERT`.* The reasoning for
|
||||
INSERT was that a chain's base must be a full row, and INSERT is what compaction
|
||||
writes. That holds for compaction, which builds a *fresh* log. It is wrong for an
|
||||
update appending into a *live* one: the row's original INSERT is already in that
|
||||
log, so a second INSERT for the same id is a duplicate, and replay correctly
|
||||
refuses it as corruption. `test_delta_chain_flatten_replays` failed on exactly
|
||||
that. UPDATE replays as remove-then-recreate and the fold terminates on either
|
||||
full-row kind, so nothing else changed.
|
||||
|
||||
- No existing test builds a chain 16 deep, so the flatten branch is almost
|
||||
certainly never executed by the suite.
|
||||
- Existing checkpoint tests use logs far below 64 MiB, so the two new
|
||||
compaction terms never fire either.
|
||||
*The proportional ceiling was unreachable, and is gone.* With the absolute term
|
||||
at 64 MiB and the ceiling at 256 MiB, any garbage large enough to reach the
|
||||
ceiling had already tripped the absolute term — the branch could never execute.
|
||||
Found by trying to write a test that exercised the ceiling and discovering no
|
||||
input could. PostgreSQL needs both constants because it thresholds on *tuples*
|
||||
with its pair at opposite ends (base 50, max 1e8); this thresholds on *bytes*,
|
||||
where one constant does both jobs. Any ceiling above the absolute term is dead,
|
||||
and any below it would simply be the trigger.
|
||||
|
||||
A green run here means "did not break what existed", not "works".
|
||||
**Verified by construction where tests do not reach.** Both update entry points
|
||||
converge on `row_apply_field_keys` (`table.c:1039` and `:1319`), so one branch
|
||||
covers both. The re-point is transparent to flattening because `db.c` captures
|
||||
`wo_wal_next_offset(w)` *before* calling into `table.c` — it targets wherever the
|
||||
next record lands, delta or full row alike.
|
||||
|
||||
## Acceptance Criteria
|
||||
|
||||
Outstanding — none verified, because the tests are held. The logic for every
|
||||
one of them is implemented; nothing is proven.
|
||||
All verified but one, which is narrowed rather than dropped. Tests live in
|
||||
`runtime/test/test_wal.c`.
|
||||
|
||||
- **Given** a row updated K times, **when** updated once more, **then** the
|
||||
- ✅ **Given** a row updated K times, **when** updated once more, **then** the
|
||||
record its offset names is a full row and its chain length is zero.
|
||||
- **Given** a row updated far more than K times, **when** it is read, **then** it
|
||||
performs at most K + 1 record reads, asserted by counting rather than timing.
|
||||
- **Given** the same row, **when** the process restarts, **then** replay is
|
||||
*`test_delta_chain_flattens_at_k`.*
|
||||
- ✅ **Given** a row updated far more than K times, **when** it is read, **then**
|
||||
it performs at most K + 1 record reads, asserted by counting rather than
|
||||
timing. *Both chain tests assert `scratch_hops <= WO_DELTA_MAX_HOPS` on every
|
||||
read, which is the count itself, not a proxy for it.*
|
||||
- ✅ **Given** the same row, **when** the process restarts, **then** replay is
|
||||
correct and its cost does not grow with the total updates ever applied.
|
||||
- **Given** a flattening update, **when** replayed, **then** the row matches the
|
||||
same row in a `resident: all` table under the same update sequence — the
|
||||
resident table is the oracle.
|
||||
- **Given** a flattening update to an indexed column, **when** queried through
|
||||
*`test_delta_chain_flatten_replays`.*
|
||||
- ⚠️ **Given** a flattening update, **when** replayed, **then** the row matches
|
||||
the same row in a `resident: all` table under the same update sequence.
|
||||
*Asserted against an expected value, not against a `resident: all` oracle
|
||||
table. Weaker than written: it catches a wrong value, but it would not catch
|
||||
the two modes disagreeing in a way that also fooled the expectation.*
|
||||
- ✅ **Given** a flattening update to an indexed column, **when** queried through
|
||||
that index, **then** the row is found by its new value and not its old, before
|
||||
and after a restart.
|
||||
- **Given** reclaimable bytes past the absolute threshold but inside the ratio,
|
||||
**when** the policy is evaluated, **then** compaction fires. *(Tier 2.)*
|
||||
- **Given** a `resident: all` table, **when** any of this runs, **then** nothing
|
||||
about its behaviour or log records changes.
|
||||
and after a restart. *`test_keys_resident_indexed_across_flatten`, checked at
|
||||
every step across the bound, not only at the end.*
|
||||
- ✅ **Given** reclaimable bytes past the absolute threshold but inside the
|
||||
ratio, **when** the policy is evaluated, **then** compaction fires.
|
||||
*`test_should_compact_absolute_and_ceiling`, which also pins the boundary just
|
||||
under the term and the small-log case where the ratio still governs.*
|
||||
- ✅ **Given** a `resident: all` table, **when** any of this runs, **then**
|
||||
nothing about its behaviour or log records changes. *Regression only: the
|
||||
existing 856 `test_table` and 5700 `test_wal` assertions pass, and a
|
||||
`resident: all` table never reaches `row_apply_field_keys`.*
|
||||
|
||||
**Honest note on what the new tests found:** no product defect in the
|
||||
indexed-column path. Both failures during that test's development were bugs in
|
||||
the test itself — reading a folded row as a `wo_str` when the fold yields engine
|
||||
`db_text`, and probing a database whose WAL had been closed. The result is still
|
||||
worth having: it is the only coverage that the index and the flatten branch
|
||||
compose, and it now pins that.
|
||||
|
||||
## Out Of Scope
|
||||
|
||||
|
|
|
|||
|
|
@ -122,6 +122,18 @@ whole-log policy's shape so it stops being blind to absolute garbage.
|
|||
- **A ceiling** — cap the proportional term so a very large live set does not
|
||||
defer compaction indefinitely. This is `autovacuum_vacuum_max_threshold`.
|
||||
|
||||
> **Implementation outcome (2026-08-30): the ceiling was built and then removed
|
||||
> as dead code — the absolute term above already does its job.** Borrowing both
|
||||
> constants from postgres was the wrong inference. Postgres needs two because it
|
||||
> thresholds on *tuples*, with its pair at opposite ends of the range (base 50,
|
||||
> max 1e8). This design thresholds on *bytes*, and "compact once garbage exceeds
|
||||
> X bytes" is itself a cap on deferral: with the absolute term at 64 MiB, any
|
||||
> garbage large enough to reach a 256 MiB ceiling has already tripped it, so the
|
||||
> branch is unreachable. Any ceiling above the absolute term is dead; any below
|
||||
> it would simply be the trigger. Caught by trying to write a test for the
|
||||
> ceiling and finding no input could reach it. Do not reintroduce it without
|
||||
> also changing what the absolute term means.
|
||||
|
||||
The existing floor keeps its current meaning — do not bother with a tiny log —
|
||||
but the doc comment must stop calling it a floor in postgres's sense, because it
|
||||
does the opposite thing.
|
||||
|
|
|
|||
|
|
@ -935,6 +935,182 @@ static const wo_classdesc KEYS_IDX_CLASSES[] = {
|
|||
* from the log (table.c's idx_cols_equal path), so the probes below must
|
||||
* run AFTER the caller's commit + re-point — mirroring db.c's inline arm —
|
||||
* not straight after wo_row_update_field, which now only stages. */
|
||||
/* databasev2 11: helper — drive one update through the full commit/re-point
|
||||
* dance the request path performs, so a chain can be built in a loop. */
|
||||
static void chain_update(wo_db *db, wo_wal *w, uint32_t cid, uint64_t id,
|
||||
uint32_t field, uint64_t v) {
|
||||
const char *msg = "";
|
||||
int ek = 0;
|
||||
uint64_t roff = wo_wal_next_offset(w);
|
||||
T_EQ(wo_row_update_field(db, cid, id, field, v, &msg, &ek), 0);
|
||||
T_EQ(ek, DB_ERR_NONE);
|
||||
T_EQ(wo_wal_commit(w), 0);
|
||||
T_EQ(wo_row_set_offset(db, cid, id, roff), 0);
|
||||
}
|
||||
|
||||
/* databasev2 11: the branch this iteration exists for. Past WO_DELTA_MAX_HOPS
|
||||
* an update must TERMINATE the chain with a full-row record rather than
|
||||
* lengthening it — otherwise read cost and replay cost grow without bound,
|
||||
* because compaction's trigger is a whole-log byte ratio and cannot see one
|
||||
* row's chain.
|
||||
*
|
||||
* The assertion is on the DEPTH the fold reports, not on timing: a test that
|
||||
* measured speed would pass on a slow box with an unbounded chain. */
|
||||
/* databasev2 11 tier 2: the compaction policy's two new terms. A pure
|
||||
* function, so this is cheap and exact — no log, no timing.
|
||||
*
|
||||
* The trap being guarded: our `floor` SUPPRESSES compaction on a small log,
|
||||
* the opposite of PostgreSQL's vac_base_thresh, which TRIGGERS on a small
|
||||
* absolute problem the proportion would hide. Before this iteration we had the
|
||||
* proportion and the suppressor and neither real guard. */
|
||||
static void test_should_compact_absolute_and_ceiling(void) {
|
||||
const uint64_t floor_b = 1024;
|
||||
|
||||
/* unchanged behaviour: below the floor, never */
|
||||
T_EQ(wo_wal_should_compact(512, 256, floor_b, 2), 0);
|
||||
/* unchanged: never compacted yet, past the floor -> once, to set a denominator */
|
||||
T_EQ(wo_wal_should_compact(4096, 0, floor_b, 2), 1);
|
||||
/* unchanged: ratio 0 disables the policy rather than dividing by nothing */
|
||||
T_EQ(wo_wal_should_compact(1u << 30, 1024, floor_b, 0), 0);
|
||||
|
||||
/* THE ABSOLUTE TERM. A live set so large that the ratio will not trip for
|
||||
* a very long time, but with more than WO_CKPT_ABS_BYTES of garbage
|
||||
* already reclaimable. The old policy said no; the point of the term is
|
||||
* that garbage large in BYTES is worth reclaiming even when it is small in
|
||||
* PROPORTION. */
|
||||
{
|
||||
uint64_t live = 4ull * 1024 * 1024 * 1024; /* 4 GiB live */
|
||||
uint64_t used = live + WO_CKPT_ABS_BYTES + 1; /* just over the term */
|
||||
T_CHECK(used < live * 2); /* ratio 2 would NOT fire */
|
||||
T_EQ(wo_wal_should_compact(used, live, floor_b, 2), 1);
|
||||
}
|
||||
/* and just under it, the ratio still governs */
|
||||
{
|
||||
uint64_t live = 4ull * 1024 * 1024 * 1024;
|
||||
uint64_t used = live + (WO_CKPT_ABS_BYTES / 2);
|
||||
T_EQ(wo_wal_should_compact(used, live, floor_b, 2), 0);
|
||||
}
|
||||
|
||||
/* DEFERRAL IS CAPPED by the same term — no separate ceiling exists, and
|
||||
* one was removed as unreachable. With an 8 GiB live set, ratio 2 would
|
||||
* wait for the log to double; the absolute term fires long before that. */
|
||||
{
|
||||
uint64_t live = 8ull * 1024 * 1024 * 1024; /* 8 GiB live */
|
||||
uint64_t used = live + WO_CKPT_ABS_BYTES + 1;
|
||||
T_CHECK(used < live * 2); /* the ratio alone would defer */
|
||||
T_EQ(wo_wal_should_compact(used, live, floor_b, 2), 1);
|
||||
}
|
||||
|
||||
/* the ratio still governs BELOW the absolute term, which is what keeps the
|
||||
* two complementary rather than one subsuming the other: a small live set
|
||||
* trips the ratio with far less garbage than 64 MiB */
|
||||
T_EQ(wo_wal_should_compact(3072, 1024, floor_b, 2), 1); /* 3 KiB > 1 KiB * 2 */
|
||||
T_EQ(wo_wal_should_compact(2048, 1024, floor_b, 2), 0); /* not yet */
|
||||
}
|
||||
|
||||
static void test_delta_chain_flattens_at_k(void) {
|
||||
char path[128];
|
||||
snprintf(path, sizeof path, "%s/chainflat.wal", g_dir);
|
||||
wo_rt rt;
|
||||
T_EQ(wo_rt_init(&rt, 1 << 20, KEYS_CLASSES, 1), 0);
|
||||
wo_db db;
|
||||
T_EQ(wo_db_init(&db, KEYS_CLASSES, 1, 0, 1), 0);
|
||||
wo_wal w;
|
||||
T_EQ(wo_wal_open(&w, path, 1 << 18), 0);
|
||||
db.rt = &rt; rt.wal = &w; rt.db = &db;
|
||||
const char *msg = "";
|
||||
|
||||
wo_str *sv = wo_str_new(&rt, "flat", 4);
|
||||
uint64_t vals[2] = {0, (uint64_t)(uintptr_t)sv};
|
||||
uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL);
|
||||
T_CHECK(id != 0);
|
||||
uint64_t off = wo_wal_next_offset(&w);
|
||||
T_EQ(wo_wal_append_insert(&w, &db, 0, id), 0);
|
||||
T_EQ(wo_wal_commit(&w), 0);
|
||||
T_EQ(wo_row_drop_payload(&db, 0, id, off), 0);
|
||||
|
||||
/* Walk the depth up one update at a time and watch it reset. Without the
|
||||
* flatten branch this climbs forever; with it, it must never exceed K. */
|
||||
uint32_t peak = 0;
|
||||
int saw_reset = 0;
|
||||
for (uint32_t n = 1; n <= WO_DELTA_MAX_HOPS * 2u + 2u; n++) {
|
||||
chain_update(&db, &w, 0, id, 0, (uint64_t)n);
|
||||
|
||||
uint32_t hops = 0;
|
||||
uint32_t got_cid = 0;
|
||||
uint64_t got_id = 0;
|
||||
uint64_t out[2] = {0, 0};
|
||||
const char *fm = "";
|
||||
uint64_t o1 = wo_row_offset1(&db, 0, id);
|
||||
T_CHECK(o1 != 0);
|
||||
T_EQ(wo_wal_fold_row_at(&w, &db, o1 - 1, &got_cid, &got_id, out, &hops, &fm), 0);
|
||||
T_CHECK(got_cid == 0 && got_id == id);
|
||||
T_CHECK(out[0] == (uint64_t)n); /* the value is still right */
|
||||
for (uint32_t i = 0; i < 2; i++) wo_db_val_free(&db, KEYS_CLASSES[0].kinds[i], out[i]);
|
||||
|
||||
if (hops > peak) peak = hops;
|
||||
if (n > 1 && hops == 0) saw_reset = 1; /* a chain was terminated */
|
||||
}
|
||||
T_CHECK(peak <= WO_DELTA_MAX_HOPS); /* the bound holds */
|
||||
T_CHECK(saw_reset); /* and it was actually reached */
|
||||
|
||||
wo_wal_close(&w);
|
||||
wo_db_destroy(&db);
|
||||
wo_rt_destroy(&rt);
|
||||
}
|
||||
|
||||
/* databasev2 11: a flattened row must survive a restart identically. Replay
|
||||
* meets a WO_WAL_INSERT where a chain used to be; if flattening wrote a shape
|
||||
* replay mishandled, this is where it shows. */
|
||||
static void test_delta_chain_flatten_replays(void) {
|
||||
char path[128];
|
||||
snprintf(path, sizeof path, "%s/chainflatreplay.wal", g_dir);
|
||||
wo_rt rt;
|
||||
T_EQ(wo_rt_init(&rt, 1 << 20, KEYS_CLASSES, 1), 0);
|
||||
wo_db db;
|
||||
T_EQ(wo_db_init(&db, KEYS_CLASSES, 1, 0, 1), 0);
|
||||
wo_wal w;
|
||||
T_EQ(wo_wal_open(&w, path, 1 << 18), 0);
|
||||
db.rt = &rt; rt.wal = &w; rt.db = &db;
|
||||
const char *msg = "";
|
||||
|
||||
wo_str *sv = wo_str_new(&rt, "rep", 3);
|
||||
uint64_t vals[2] = {0, (uint64_t)(uintptr_t)sv};
|
||||
uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL);
|
||||
uint64_t off = wo_wal_next_offset(&w);
|
||||
T_EQ(wo_wal_append_insert(&w, &db, 0, id), 0);
|
||||
T_EQ(wo_wal_commit(&w), 0);
|
||||
T_EQ(wo_row_drop_payload(&db, 0, id, off), 0);
|
||||
|
||||
/* enough updates to guarantee at least one flatten */
|
||||
uint64_t last = 0;
|
||||
for (uint32_t n = 1; n <= WO_DELTA_MAX_HOPS + 3u; n++) {
|
||||
chain_update(&db, &w, 0, id, 0, (uint64_t)n);
|
||||
last = n;
|
||||
}
|
||||
wo_wal_close(&w);
|
||||
wo_db_destroy(&db);
|
||||
|
||||
wo_db db2;
|
||||
T_EQ(wo_db_init(&db2, KEYS_CLASSES, 1, 0, 1), 0);
|
||||
db2.rt = &rt; rt.wal = NULL; rt.db = &db2;
|
||||
T_CHECK(wo_wal_replay(path, &db2) >= 0);
|
||||
wo_wal w2;
|
||||
T_EQ(wo_wal_open(&w2, path, 1 << 18), 0);
|
||||
rt.wal = &w2;
|
||||
|
||||
db_row *r = wo_row_borrow(&db2, 0, id, &msg);
|
||||
T_CHECK(r != NULL);
|
||||
T_CHECK(r->slots[0] == last); /* the newest value survived */
|
||||
db_text *back = (db_text *)(uintptr_t)r->slots[1];
|
||||
T_CHECK(back != NULL && back->len == 3 && memcmp(back->bytes, "rep", 3) == 0);
|
||||
wo_row_release(&db2, 0, r);
|
||||
|
||||
wo_wal_close(&w2);
|
||||
wo_db_destroy(&db2);
|
||||
wo_rt_destroy(&rt);
|
||||
}
|
||||
|
||||
static void test_keys_resident_update_indexed(void) {
|
||||
char path[128];
|
||||
snprintf(path, sizeof path, "%s/keysidx.wal", g_dir);
|
||||
|
|
@ -1000,6 +1176,110 @@ static void test_keys_resident_update_indexed(void) {
|
|||
wo_rt_destroy(&rt);
|
||||
}
|
||||
|
||||
/* databasev2 11: the third leg the story called out — a delta on an INDEXED
|
||||
* column, with flattening in play. The two are independent features that meet
|
||||
* on the same write path, and the meeting is where a bug would live:
|
||||
* `row_apply_field_keys` picks DELTA or full-row image AFTER the index has
|
||||
* already been re-pointed, so a flattened image that captured the wrong value
|
||||
* would leave the index pointing at a row the fold disagrees with.
|
||||
*
|
||||
* Drives enough updates on the indexed column to cross WO_DELTA_MAX_HOPS
|
||||
* several times, so at least one update lands on each branch, then checks the
|
||||
* index and the fold agree at the end and after replay. */
|
||||
static void test_keys_resident_indexed_across_flatten(void) {
|
||||
char path[128];
|
||||
snprintf(path, sizeof path, "%s/keysidxflat.wal", g_dir);
|
||||
wo_rt rt;
|
||||
T_EQ(wo_rt_init(&rt, 1 << 20, KEYS_IDX_CLASSES, 1), 0);
|
||||
wo_db db;
|
||||
T_EQ(wo_db_init(&db, KEYS_IDX_CLASSES, 1, 0, 1), 0);
|
||||
wo_wal w;
|
||||
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
|
||||
db.rt = &rt; rt.wal = &w; rt.db = &db;
|
||||
const char *msg = "";
|
||||
|
||||
wo_str *sa = wo_str_new(&rt, "a", 1);
|
||||
uint64_t va[2] = {100, (uint64_t)(uintptr_t)sa};
|
||||
uint64_t a = wo_row_insert(&db, 0, va, &msg, NULL);
|
||||
T_CHECK(a != 0);
|
||||
uint64_t off_a = wo_wal_next_offset(&w);
|
||||
T_EQ(wo_wal_append_insert(&w, &db, 0, a), 0);
|
||||
T_EQ(wo_wal_commit(&w), 0);
|
||||
T_EQ(wo_row_drop_payload(&db, 0, a, off_a), 0);
|
||||
|
||||
/* cross the bound several times over; ROUNDS is deliberately not a
|
||||
* multiple of the bound, so the run does not end on a reset */
|
||||
const uint64_t ROUNDS = WO_DELTA_MAX_HOPS * 3 + 5;
|
||||
uint64_t val = 100;
|
||||
int saw_reset = 0;
|
||||
db_table *t = &db.tables[0];
|
||||
for (uint64_t i = 0; i < ROUNDS; i++) {
|
||||
uint64_t prev = val;
|
||||
val = 200 + i;
|
||||
int ek = 0;
|
||||
uint64_t roff = wo_wal_next_offset(&w);
|
||||
T_EQ(wo_row_update_field(&db, 0, a, 0, val, &msg, &ek), 0);
|
||||
T_EQ(ek, DB_ERR_NONE);
|
||||
T_EQ(wo_wal_commit(&w), 0);
|
||||
T_EQ(wo_row_set_offset(&db, 0, a, roff), 0);
|
||||
|
||||
/* every intermediate step, not only the last: the row is findable by
|
||||
* the value just written and absent from the one it replaced */
|
||||
uint64_t *ids; uint32_t cnt;
|
||||
T_EQ(wo_idx_probe(&db, 0, 0, val, NULL, 0, &ids, &cnt), 1);
|
||||
T_CHECK(cnt == 1 && ids[0] == a);
|
||||
free(ids);
|
||||
T_EQ(wo_idx_probe(&db, 0, 0, prev, NULL, 0, &ids, &cnt), 1);
|
||||
T_CHECK(cnt == 0 && ids == NULL);
|
||||
|
||||
db_row *r = wo_row_borrow(&db, 0, a, &msg);
|
||||
T_CHECK(r != NULL && r->slots[0] == val);
|
||||
if (t->scratch_hops == 0) saw_reset = 1;
|
||||
T_CHECK(t->scratch_hops <= WO_DELTA_MAX_HOPS);
|
||||
wo_row_release(&db, 0, r);
|
||||
}
|
||||
T_CHECK(saw_reset); /* flattening actually fired during the run */
|
||||
|
||||
/* the Text column, never updated, must survive every flatten: the image
|
||||
* is rebuilt from a borrowed row, which is exactly where a value of the
|
||||
* wrong representation would be written back */
|
||||
db_row *r = wo_row_borrow(&db, 0, a, &msg);
|
||||
T_CHECK(r != NULL && r->slots[0] == val);
|
||||
db_text *back = (db_text *)(uintptr_t)r->slots[1]; /* engine repr, not wo_str */
|
||||
T_CHECK(back != NULL && back->len == 1 && back->bytes[0] == 'a');
|
||||
wo_row_release(&db, 0, r);
|
||||
|
||||
wo_wal_close(&w);
|
||||
wo_db_destroy(&db);
|
||||
|
||||
/* And the index rebuilds from the replayed log, flattened records and all.
|
||||
* A keys-resident table comes back offset-valued, so both the probe's
|
||||
* verification and any read need a live log: replay with rt.wal NULL (it
|
||||
* lends its own read-only view), then reopen before touching the rows. */
|
||||
wo_db db2;
|
||||
T_EQ(wo_db_init(&db2, KEYS_IDX_CLASSES, 1, 0, 1), 0);
|
||||
db2.rt = &rt; rt.wal = NULL; rt.db = &db2;
|
||||
T_CHECK(wo_wal_replay(path, &db2) >= 0);
|
||||
wo_wal w2;
|
||||
T_EQ(wo_wal_open(&w2, path, 1 << 16), 0);
|
||||
rt.wal = &w2;
|
||||
|
||||
uint64_t *ids; uint32_t cnt;
|
||||
T_EQ(wo_idx_probe(&db2, 0, 0, val, NULL, 0, &ids, &cnt), 1);
|
||||
T_CHECK(cnt == 1 && ids[0] == a);
|
||||
free(ids);
|
||||
T_EQ(wo_idx_probe(&db2, 0, 0, 100, NULL, 0, &ids, &cnt), 1);
|
||||
T_CHECK(cnt == 0 && ids == NULL);
|
||||
|
||||
db_row *r2 = wo_row_borrow(&db2, 0, a, &msg);
|
||||
T_CHECK(r2 != NULL && r2->slots[0] == val);
|
||||
wo_row_release(&db2, 0, r2);
|
||||
|
||||
wo_wal_close(&w2);
|
||||
wo_db_destroy(&db2);
|
||||
wo_rt_destroy(&rt);
|
||||
}
|
||||
|
||||
/* class 0: Row { n: scalar, sku: Text @unique } — the Text-representation
|
||||
* gap flagged across Tasks 3, 4 and 5 and fixed in Task 6: every existing
|
||||
* keys-resident index test above indexes the SCALAR column, never
|
||||
|
|
@ -2392,6 +2672,10 @@ int main(void) {
|
|||
test_fold_refuses_forward_pointing_delta();
|
||||
test_keys_resident_update_field();
|
||||
test_keys_resident_update_indexed();
|
||||
test_keys_resident_indexed_across_flatten();
|
||||
test_should_compact_absolute_and_ceiling();
|
||||
test_delta_chain_flattens_at_k();
|
||||
test_delta_chain_flatten_replays();
|
||||
test_keys_resident_update_indexed_text();
|
||||
test_keys_resident_update_unique_violation_refused();
|
||||
test_keys_resident_two_updates_one_drain();
|
||||
|
|
|
|||
Loading…
Reference in a new issue