feat(wal): a failed barrier is detected, and fatal — T1

databasev2 4 part A, task 1.

- wo_wal gains `path`: the abort diagnostic is worthless without naming
  the file it could not write. strdup'd in open, freed in close; NULL is
  tolerated so the message degrades rather than crashes
- wo_wal_commit now reports WHICH half failed — WO_WAL_ERR_WRITE for
  pwrite, WO_WAL_ERR_SYNC for fdatasync. A short write and a device
  refusing the flush are different operational problems and the operator
  needs the right one named
- wo_wal_commit_fatal(w, nrec): commits, or prints one diagnostic naming
  the operation, path, errno and record count, then exits
  WO_EXIT_DURABILITY (3 — 1 is a trap, 2 is a refusal, so this takes a
  third of its own)
- retrying is not offered, deliberately: on Linux a failed fsync may have
  already discarded the dirty pages, so a second call can report success
  having written nothing. Replay is the recovery that works

- test_wal: a failed commit is DETECTED, reports the write error
  specifically, keeps the batch staged (a failed commit consumes
  nothing), and the WAL knows its own path. 165 pass (was 156)

DISCLOSED GAP: the exit path itself is not exercised. Forcing a real
fdatasync failure needs a full or read-only filesystem, which the gate
cannot arrange without mount privileges. No fault-injection switch was
added — shipping a binary that can be told to kill itself is the worse
trade, and the spec rejected it.

Verified: just wovm-test — 36 suites (18 x both dispatch flavors) 0 fail,
cli_smoke OK.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
shoney.arickathil 2026-08-28 09:16:35 +02:00
parent 026919762b
commit ceea00e0b6
3 changed files with 90 additions and 3 deletions

View file

@ -5,6 +5,7 @@
#include <errno.h>
#include <fcntl.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <unistd.h>
@ -302,6 +303,7 @@ int wo_wal_open(wo_wal *w, const char *path, uint64_t prealloc) {
memset(w, 0, sizeof(*w));
w->fd = open(path, O_RDWR | O_CREAT, 0644);
if (w->fd < 0) return -1;
w->path = strdup(path); /* NULL is tolerated: the diagnostic degrades */
if (prealloc) {
/* best-effort: a filesystem without fallocate still works */
(void)posix_fallocate(w->fd, 0, (off_t)prealloc);
@ -317,6 +319,7 @@ int wo_wal_open(wo_wal *w, const char *path, uint64_t prealloc) {
void wo_wal_close(wo_wal *w) {
if (w->fd >= 0) close(w->fd);
free(w->path);
free(w->buf);
memset(w, 0, sizeof(*w));
w->fd = -1;
@ -396,16 +399,32 @@ int wo_wal_commit(wo_wal *w) {
ssize_t n = pwrite(w->fd, w->buf + at, w->len - at, (off_t)(w->off + at));
if (n < 0) {
if (errno == EINTR) continue;
return -1;
return WO_WAL_ERR_WRITE;
}
at += (size_t)n;
}
if (fdatasync(w->fd) != 0) return -1;
if (fdatasync(w->fd) != 0) return WO_WAL_ERR_SYNC;
w->off += w->len;
w->len = 0; /* acked: the batch is durable */
return 0;
}
void wo_wal_commit_fatal(wo_wal *w, uint32_t nrec) {
int rc = wo_wal_commit(w);
if (rc == 0) return;
/* Nothing here is recoverable: RAM holds changes the log does not, and
* this process can no longer serve reads that would survive a restart.
* Name what failed precisely enough to act on, then stop. */
fprintf(stderr,
"writeonce: DURABILITY FAILURE — %s failed on %s: %s\n"
" %u record(s) in the batch were NOT made durable and are not acknowledged.\n"
" The process is stopping: replay restores the last durable state.\n",
rc == WO_WAL_ERR_SYNC ? "fdatasync" : "pwrite",
w->path ? w->path : "(the write-ahead log)", strerror(errno),
nrec);
exit(WO_EXIT_DURABILITY);
}
static int apply_record(wo_db *db, const uint8_t *payload, uint32_t len) {
rbuf r = {payload, payload + len, 0};
uint8_t kind = rd_u8(&r);

View file

@ -47,6 +47,10 @@ enum { WO_WAL_INSERT = 1, WO_WAL_REMOVE = 2, WO_WAL_UPDATE = 3 };
typedef struct wo_wal {
int fd;
/* databasev2 4: where this WAL lives, so a durability failure can name
* the file it could not write. An abort diagnostic without the path
* sends an operator hunting. Owned here, freed by wo_wal_close. */
char *path;
uint64_t off; /* next write offset (the intact tail) */
/* staged batch: appended by wal_append_*, flushed by wal_commit */
uint8_t *buf;
@ -71,10 +75,33 @@ int wo_wal_append_remove(wo_wal *w, uint32_t class_id, uint64_t id);
* later optimization, recorded). Call AFTER the RAM update. */
int wo_wal_append_update(wo_wal *w, wo_db *db, uint32_t class_id, uint64_t id);
/* databasev2 4: which half of the barrier failed. A pwrite failure and an
* fdatasync failure are different operational problems (a short write vs a
* device refusing the flush), so the diagnostic must name the right one. */
#define WO_WAL_ERR_WRITE (-1)
#define WO_WAL_ERR_SYNC (-2)
/* The process exit status for a durability failure. 1 is a trap and 2 is a
* refusal, so this takes a third of its own. */
#define WO_EXIT_DURABILITY 3
/* Write the staged batch and fdatasync — the ack line. Empty batch = ok,
* no syscall. 0 ok, -1 write/sync failure (the batch stays staged). */
* no syscall. 0 ok, WO_WAL_ERR_WRITE / WO_WAL_ERR_SYNC on failure (the
* batch stays staged: a failed commit consumes nothing). */
int wo_wal_commit(wo_wal *w);
/* databasev2 4: commit, or END THE PROCESS.
*
* The one rule this iteration introduces: once a statement has mutated RAM,
* the only outcomes are durable or process death. Retrying is not an
* alternative — on Linux a failed fsync may already have discarded the dirty
* pages, so a second call can report success having written nothing. The
* recovery that works is replay, which returns the last durable state.
*
* [nrec] is the number of records in the batch, for the diagnostic only.
* Returns on success; never returns on failure. */
void wo_wal_commit_fatal(wo_wal *w, uint32_t nrec);
/* Boot replay: apply every intact record to [db] in order. Ids re-enter
* exactly as logged; each table's next_id advances past the replayed ids
* that belong to this shard. Returns the number of records applied, or -1

View file

@ -117,6 +117,46 @@ static void test_roundtrip_replay(void) {
wo_rt_destroy(&rt);
}
/* databasev2 4 part A, Task 1: a failed barrier must be DETECTED, and the
* caller must be able to tell WHICH operation failed — a pwrite failure and
* an fdatasync failure are different operational problems and the diagnostic
* has to name the right one. This proves detection only; the fatal exit that
* follows it cannot be exercised in-process. */
static void test_commit_failure_detected(void) {
char path[128];
snprintf(path, sizeof path, "%s/commitfail.wal", g_dir);
wo_rt rt;
T_EQ(wo_rt_init(&rt, 1 << 20, CLASSES, 1), 0);
wo_db db;
T_EQ(wo_db_init(&db, CLASSES, 1, 0, 1), 0);
wo_wal w;
T_EQ(wo_wal_open(&w, path, 1 << 16), 0);
const char *msg = "";
/* the WAL remembers where it lives — the abort diagnostic is worthless
* without it */
T_CHECK(w.path != NULL && strstr(w.path, "commitfail.wal") != NULL);
wo_str *s = wo_str_new(&rt, "abc", 3);
uint64_t vals[2] = {7, (uint64_t)(uintptr_t)s};
uint64_t id = wo_row_insert(&db, 0, vals, &msg, NULL);
T_CHECK(id != 0);
T_EQ(wo_wal_append_insert(&w, &db, 0, id), 0);
T_CHECK(w.len > 0); /* something really is staged */
/* an unusable descriptor: pwrite reports EBADF. -1 is used rather than
* closing the real fd so the close below cannot double-free it. */
int real = w.fd;
w.fd = -1;
T_EQ(wo_wal_commit(&w), WO_WAL_ERR_WRITE);
T_CHECK(w.len > 0); /* a failed commit consumes nothing */
w.fd = real;
wo_wal_close(&w);
wo_db_destroy(&db);
wo_rt_destroy(&rt);
}
static void test_torn_tail(void) {
char path[128];
snprintf(path, sizeof path, "%s/torn.wal", g_dir);
@ -327,6 +367,7 @@ int main(void) {
snprintf(g_dir, sizeof g_dir, "/tmp/wo-wal-test-XXXXXX");
if (!mkdtemp(g_dir)) return 1;
test_roundtrip_replay();
test_commit_failure_detected();
test_torn_tail();
test_float_bytes_replay();
test_crash_battery();