From b9ce271b3deb1fd8225f5cd5dc9a000bfd5bd9fc Mon Sep 17 00:00:00 2001 From: "shoney.arickathil" Date: Sun, 30 Aug 2026 05:15:52 +0200 Subject: [PATCH] fix(lang41): an unadopted shard must not impersonate shard 0 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - root cause: a worker's runtime is initialised lazily on first fiber adoption, and rt.shard_id is stamped only there — but INBOX_READY[i] is set at thread creation. A shard that never adopts is still settled at shutdown, carrying rt.shard_id 0 from the memset - it then impersonated shard 0: wo_drop_obj saw 0 == 0 for anything the primary allocated, took the "we are home" branch instead of routing, and called class_free against rt->classes, which lazy init never filled. &rt->classes[class_id] off a NULL base is the faulting read - fix: stamp the runtime's real identity at thread creation. An uninitialised shard owns nothing, so its true id makes every payload correctly foreign and routes it to an owner that can free it - ASan could not name this: the arena is one hand-managed malloc block, so intra-arena reuse is invisible and it surfaces as a bare SEGV - pinned by tests/regress/lang-41, driven from db-actor-accept. Needs multiple shards (the corpus runner pins WO_SHARDS=1) and the ASan build. SEGVs twice per run unfixed, clean fixed - the HANG is a separate defect and is NOT fixed: with this in place the harness stops losing whole sections, but idempotent-stop-2 still fires ~1 run in 6. The story records where to look Co-Authored-By: Claude Opus 5 (1M context) (cherry picked from commit 9dca0b4b4727b976d326b29cb4c6522b62d48a73) --- .../41-actor-arena-crash.md | 140 ++++++++++++++++++ runtime/src/vm.c | 14 ++ scripts/db-actor-accept.sh | 34 +++++ tests/regress/lang-41/shard-settle-crash.wo | 61 ++++++++ 4 files changed, 249 insertions(+) create mode 100644 docs/stories/language-runtime-database/41-actor-arena-crash.md create mode 100644 tests/regress/lang-41/shard-settle-crash.wo diff --git a/docs/stories/language-runtime-database/41-actor-arena-crash.md b/docs/stories/language-runtime-database/41-actor-arena-crash.md new file mode 100644 index 0000000..84c95c7 --- /dev/null +++ b/docs/stories/language-runtime-database/41-actor-arena-crash.md @@ -0,0 +1,140 @@ +--- +track: language-runtime-database +iteration: "41" +status: in-progress +readiness: refine +--- + +# 41 — the actor arena crash: a SIGSEGV under concurrent parked callers + +> Found 2026-08-30 while implementing [porch 1](../porch/01-store-backed-middleware.md). +> It blocks [porch 9](../porch/09-idempotent-replay.md) outright and it is not a +> porch bug — it is in the C runtime, and it threatens **any** actor code that +> allocates heavily inside `receive`. + +## Status 2026-08-30 — the SIGSEGV is FIXED. The hang is not. + +**They were two defects, not one.** This file used to say "two shapes, almost +certainly one cause". That guess was wrong, and it was disproven by fixing one +and watching the other survive. + +### Fixed: the SIGSEGV — an unadopted shard impersonating shard 0 + +Root cause, reproduced minimally and pinned: + +A worker shard's runtime is initialised **lazily**, when it adopts its first +fiber, and `rt.shard_id` is stamped only there (`vm.c`, worker late-init). But +`INBOX_READY[i]` is set at **thread creation**, long before. So a shard that +never adopts a fiber still gets settled at shutdown by `eng_settle_inboxes`, +which — unlike its sibling loop over `vm_drop_actor_world` — was guarded only on +`INBOX_READY`, not on the runtime actually being initialised. + +Such a shard carried `rt.shard_id == 0` from the `memset`, so it +**impersonated shard 0**. `wo_drop_obj` compares `o->shard_id` against +`rt->shard_id`, saw `0 == 0` for anything the primary had allocated, took the +"we are home" branch instead of routing, and called `class_free` against +`rt->classes` — which lazy init had never filled. `&rt->classes[class_id]` off a +NULL base is the faulting read; ASan reported `0x148`, the offset of the class +id it was carrying. + +The fix stamps the runtime's real identity at thread creation. An +uninitialised shard has allocated nothing and therefore owns nothing, so its +true id makes every payload correctly foreign and routes it to the owner that +can free it. + +**Why ASan never caught this in the existing suite:** the arena is one +hand-managed `malloc` block, so intra-arena reuse is invisible to the +sanitizer. The failure surfaces as a bare SEGV, never as a use-after-free +report — which is also why the original investigation could only localise it +rather than name it. + +Pinned by `tests/regress/lang-41/shard-settle-crash.wo`, driven from +`scripts/db-actor-accept.sh`. It needs **multiple shards** (the corpus runner +pins `WO_SHARDS=1`, which is why it does not live there) and the ASan build. It +SEGVs twice per run against the unfixed runtime and is clean with the fix. + +### Still open: the hang + +With the SIGSEGV fixed, the reproduction harness stopped losing whole sections +to crashes — the runs that used to fail six checks with `000` status codes are +gone. What remains is a single check, `idempotent-stop-2`: after SIGTERM the +process is still alive. Measured at roughly **1 run in 6** with the fix in +place, against ~1 in 4 before. + +So the hang is its own defect and needs its own investigation. It was not +caught in this pass: a `gdb` attach needs the hung process held open, and eight +scripted attempts to catch one in the act did not land inside the time budget. + +**Where to look first.** `wo_engine_stop` sets `eng_shutdown`, wakes each +worker's eventfd, then `pthread_join`s. Two candidates worth eliminating before +anything else: a worker blocked in `wo_io_wait` with a fiber parked on a `call` +whose reply will never arrive, and the primary spinning in +`while (eng_settle_inboxes() > 0) {}` if two shards can route the same payload +to each other indefinitely. The second is cheap to rule out with a counter. + +## The original symptom + +Two shapes, believed at the time to share one cause: + +- **A hang.** Under concurrent `call()`-parked callers doing real per-request + table I/O, `main()` returns cleanly and then the OS process fails to exit. + Roughly one run in five, non-deterministic. +- **A SIGSEGV.** The aggressive variant — same workload with forced + `Connection: close` — crashes outright. + +A `gdb` backtrace puts it inside **`wo_arena_alloc` / `wo_str_new` / `vm_run`**. + +## What is already known, and how it was measured + +This is the part worth keeping: the evidence was gathered by accident and would +be expensive to reproduce from scratch. + +- **It tracks allocation inside `receive`, not `call` itself.** Two middlewares + drive the same actor pool through the same `call`/park machinery. The one + whose actor arm has ~5 allocation sites and moves a whole `Req` plus a + `Handler` through the mailbox crashes; the one with ~2 sites moving four + scalars does not. Over ten consecutive gate runs, **every** failure belonged + to the allocation-heavy path and **none** to the light one. +- **It scales with sequential insert+delete volume against one key.** N=4 and + N=5 crashed 1 time in 3 and 3 times in 3; N=1–3 stayed clean across 12+ + trials. That points at slot recycling or arena reuse rather than at anything + request-shaped. +- **A freshly restarted server makes it much rarer**, which is consistent with + state accumulated in the arena rather than a single bad allocation. + +## The reproduction harness + +`archive/porch-idempotency` is a working reproduction, not a description. +Sections 18a–18h and 19 of `scripts/web-app-accept.sh` on that tag drive it. +Recover with `git checkout -b archive/porch-idempotency`. + +The two most reliable triggers there are the concurrent-duplicate leg (two +parallel clients through one actor against a slow handler) and the ephemeral-row +leg (N sequential insert+delete cycles on one key). + +## Two smaller runtime defects found alongside it + +Independent of the crash, both worked around rather than fixed, both worth +fixing while someone is in this code: + +1. **`try EXPR catch (e) nil` cannot distinguish a literal `Int 0` reply from a + trap.** Any code whose valid reply includes 0 silently treats success as + failure. Worked around in the archived code by never packing a zero outcome. +2. **A `Text`/map value read off `json.decode(...) as T` is corrupted once + embedded in a struct that crosses a function-return boundary.** Worked around + by forcing fresh text with `.. ""` on every field copied out of a decoded + record. This one is a data-corruption class defect and deserves its own + minimal fixture. + +## Why this outranks the porch work behind it + +A crash in `wo_arena_alloc` under concurrent actors is not a niche failure. The +actor model is the concurrency story for this runtime, and allocation inside +`receive` is the normal thing for an actor to do — porch merely happened to do +enough of it to find this. Anything built on actors is exposed until it is +fixed. + +## Out of scope + +Fixing the porch feature that found it. That is [porch 9](../porch/09-idempotent-replay.md), +and it is already written; it only needs this to land first. diff --git a/runtime/src/vm.c b/runtime/src/vm.c index aa9c0a5..3b45882 100644 --- a/runtime/src/vm.c +++ b/runtime/src/vm.c @@ -658,6 +658,20 @@ int wo_engine_start(const wo_module *mod, size_t heap_cap, uint32_t nshards) { memset(sv, 0, sizeof *sv); sv->mod = mod; sv->shard_id = i; + /* language 41: stamp the RUNTIME's identity here too, not only at + * lazy init. A worker's rt is initialised when it adopts its first + * fiber (T6), but INBOX_READY[i] is set right below — at thread + * creation. So a shard that never adopts still gets settled at + * shutdown, with rt.shard_id left 0 by the memset above. It then + * IMPERSONATES shard 0: wo_drop_obj compares o->shard_id against + * rt->shard_id, sees 0 == 0 for any payload the primary allocated, + * concludes "we are home" instead of routing, and calls class_free + * against rt->classes — which lazy init never filled, so it is NULL. + * That is the SIGSEGV: &rt->classes[class_id] off a null base. + * An uninitialised shard has allocated nothing and therefore owns + * nothing, so carrying its real id makes every payload correctly + * foreign and routes it to the owner that can actually free it. */ + sv->rt.shard_id = (uint16_t)i; sv->is_primary = 0; sv->wake_efd = eventfd(0, EFD_NONBLOCK); if (sv->wake_efd < 0) return -1; diff --git a/scripts/db-actor-accept.sh b/scripts/db-actor-accept.sh index ae8ed9c..0cf9193 100755 --- a/scripts/db-actor-accept.sh +++ b/scripts/db-actor-accept.sh @@ -67,6 +67,40 @@ else bad "WAL replay" "r1=$r1 r2=$r2" fi +# ---- language 41: an unadopted shard must not impersonate shard 0 --------- +# A worker shard's runtime is initialised lazily, when it adopts its first +# fiber -- but INBOX_READY is set at thread creation. A shard that never +# adopts therefore still gets settled at shutdown, and before the fix its +# rt.shard_id was left 0 by the memset. It then impersonated shard 0: +# wo_drop_obj saw 0 == 0 for anything the primary allocated, took the "we +# are home" branch instead of routing, and called class_free against a +# class table lazy init never filled -- &rt->classes[id] off a NULL base. +# +# Needs MULTIPLE SHARDS (the corpus runner pins WO_SHARDS=1, which is why +# this lives here) and the ASan build, because the arena is one hand-managed +# malloc block: intra-arena reuse is invisible to ASan, so the failure +# surfaces as a bare SEGV rather than a use-after-free report. +L41_W="$(mktemp -d)" +trap 'rm -rf "$L41_W"' EXIT +L41_WOC="$ROOT/compiler/_build/default/bin/woc" +L41_VM="$ROOT/runtime/build/wovm_asan" +L41_SRC="$ROOT/tests/regress/lang-41/shard-settle-crash.wo" +if [[ ! -x "$L41_VM" ]]; then + bad "lang-41 shard settle" "build it first: make -C runtime wovm-asan" +elif "$L41_WOC" --emit "$L41_SRC" -o "$L41_W/l41.wob" >/dev/null 2>&1; then + mkdir -p "$L41_W/l41data" + l41_out="$(WO_DATA="$L41_W/l41data" WO_SHARDS=4 timeout 60 "$L41_VM" "$L41_W/l41.wob" 2>&1)" + if grep -q 'SEGV\|AddressSanitizer' <<<"$l41_out"; then + bad "lang-41 shard settle" "$(grep -m1 'ERROR' <<<"$l41_out")" + elif grep -q 'dispatched' <<<"$l41_out"; then + ok "lang-41: an unadopted shard routes instead of impersonating shard 0" + else + bad "lang-41 shard settle" "no output: $(head -c 120 <<<"$l41_out")" + fi +else + bad "lang-41 shard settle" "fixture did not compile" +fi + echo echo "db-actor-accept: $((pass + fail)) checks, $fail failures" [[ $fail -eq 0 ]] || exit 1 diff --git a/tests/regress/lang-41/shard-settle-crash.wo b/tests/regress/lang-41/shard-settle-crash.wo new file mode 100644 index 0000000..eb7d349 --- /dev/null +++ b/tests/regress/lang-41/shard-settle-crash.wo @@ -0,0 +1,61 @@ +use json + +@table(name: "rows", index: [k]) +class Row { + k: Text + v: Text +} + +class Ask { n: Int } +class Go { n: Int } + +typedef Stored = { + status: Int, + body: Text +} + +-- The shared actor: allocates inside receive, churns insert+delete on ONE key. +class Churn { + fn receive(msg: Ask) -> Int { + let i = 0; + while i < 5 { + let enc = json.encode(Stored { status: 200 + i, body: "payload-${msg.n}-${i}" }); + insert Row { k: "one", v: enc }; + let hits = from r in Row where r.k == "one" take 1 select r; + if len(hits) > 0 { + let back = json.decode(hits[0].v) as Stored; + if back != nil { i = i + back.status - back.status; } + delete hits[0]; + } + i = i + 1; + } + return msg.n; + } +} + +-- Clients: each parks on `call` into the SAME churn actor, concurrently. +class Client { + target: actor Ask + fn receive(msg: Go) -> Int { + let j = 0; + while j < 8 { + let r = call(self.target, Ask { n: msg.n * 100 + j }); + j = j + 1; + } + return 0; + } +} + +fn main() -> Int { + let churn: actor Ask = spawn Churn {}; + let c1: actor Go = spawn Client { target: churn }; + let c2: actor Go = spawn Client { target: churn }; + let c3: actor Go = spawn Client { target: churn }; + let c4: actor Go = spawn Client { target: churn }; + send(c1, Go { n: 1 }); + send(c2, Go { n: 2 }); + send(c3, Go { n: 3 }); + send(c4, Go { n: 4 }); + print("dispatched"); + return 0; +}