feat: nil literal + systems-stdlib text/container builtins

log-watcher diagnostics 272 -> 129 (parse errors 7, stdlib-module calls 49,
lowering gaps 72). corpus 71/0, woc runtest 565/0, wovm unit gates green.

- nil (haxe-parity Task 6's literal half): `nil` keyword, Ast.NilLit, lowered
  to the zero word for every `?T` — the representation the format doc already
  fixes ("a nullable field stores exactly what T stores and spells nil as 0"),
  so no boxing, no unbox on read, and every drop plan already skips it.
  Contextual on its destination in both type derivers, like `[]`/`{}`
- 23 new builtins (wob.h ids 16..38, loader arities, builtin.c): len, byte_at,
  print_err, starts_with, ends_with, index_of, last_index_of, substr, trim,
  to_lower, char_of, parse_int, split, split_ws, join, slice, pop, shift,
  sort, reverse, remove, key_at, val_at
  - fresh-Text/fresh-multi results allocate in the VM; `split`/`split_ws` fix
    their element kind (Text), `slice` copies its source's — and COPIES Text
    elements so a slice and its source never both own one value
  - pop/shift hand the element's ownership to the caller; remove drops the
    map's own key and value; key_at/val_at expose slot-ordered enumeration
    (what `for k, v in m` will lower onto)
  - parse_int is optional-shaped: unparseable is 0, `?Int`'s own nil
- obj.c/obj.h: wo_str_alloc (uninitialized Text of known length) so `join`
  builds its result in one allocation instead of one per element
- types.ml/emit.ml: builtin signatures, argument-shape requirements and
  return types for all 23 — the return table is also what classifies a `let`
  holding a fresh Text or multi as owned, so an omission there is a leak

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
shoney.arickathil 2026-08-14 16:43:38 +02:00
parent 2bb39d6b6b
commit e5c5952cdc
13 changed files with 578 additions and 9 deletions

View file

@ -197,6 +197,12 @@ and expr_kind =
| IntLit of int
| StrLit of string
| BoolLit of bool
(* haxe-parity Task 6: `nil`, the absent value of a `?T`. One
representation for every T: the zero word — "a nullable field stores
exactly what T stores and spells nil as 0", docs/plan/oop-vm/
08-builtin-surface.md. Nothing to allocate, nothing to unbox, and
every per-kind drop plan already ignores a zero slot. *)
| NilLit
| Ident of string
| Field of expr * string
| Index of expr * expr

View file

@ -65,6 +65,7 @@ let kind_label (k : Token.kind) : string =
| Token.KwTypedef -> "KW_TYPEDEF"
| Token.KwTry -> "KW_TRY"
| Token.KwCatch -> "KW_CATCH"
| Token.KwNil -> "KW_NIL"
| Token.LBrace -> "LBRACE"
| Token.RBrace -> "RBRACE"
| Token.LParen -> "LPAREN"
@ -224,6 +225,7 @@ let rec expr_str (e : Ast.expr) : string =
| Ast.Interp inner -> Printf.sprintf "INTERP(%s)" (expr_str inner)
| Ast.ListLit items -> Printf.sprintf "[%s]" (String.concat ", " (List.map expr_str items))
| Ast.MapLit -> "{}"
| Ast.NilLit -> "nil"
(* Like SWITCH above: a one-line summary, not a full unparse of the
catch arm's statements. *)
| Ast.Try { body; ename; handler } ->

View file

@ -230,6 +230,31 @@ let b_variant_tag = 14
source-callable name. *)
let b_err_fill = 15
(* systems stdlib (runtime/src/wob.h WO_B_LEN..WO_B_MAP_VAL_AT) *)
let b_len = 16
let b_byte_at = 17
let b_print_err = 18
let b_starts_with = 19
let b_ends_with = 20
let b_index_of = 21
let b_last_index_of = 22
let b_substr = 23
let b_trim = 24
let b_to_lower = 25
let b_char_of = 26
let b_parse_int = 27
let b_split = 28
let b_split_ws = 29
let b_join = 30
let b_slice = 31
let b_pop = 32
let b_shift = 33
let b_sort = 34
let b_reverse = 35
let b_map_remove = 36
let b_map_key_at = 37
let b_map_val_at = 38
let ins_abc op a b c = op lor (a lsl 8) lor (b lsl 16) lor (c lsl 24)
let ins_abx op a bx = op lor (a lsl 8) lor (bx lsl 16)
let ins_asbx op a sbx = ins_abx op a (sbx + 32768)
@ -831,12 +856,34 @@ let builtin_ret (name : string) (argty : Ast.field_ty option) : Ast.field_ty opt
match argty with
| Some t -> ( match unwrap t with Multi e -> Some (Scalar e) | Map (_, v) -> Some (Scalar v) | _ -> None)
| None -> None)
(* systems stdlib — kept in sync with Types.builtin_confident_ret (both
tables, same contract, different type languages). This is also what
classifies a `let` holding a fresh Text or a fresh `multi` as owned, so
a missing entry here is a leak, not just a lost type. *)
| "len" | "byte_at" | "index_of" | "last_index_of" -> Some (Scalar "Int")
| "print_err" | "sort" | "reverse" -> Some (Scalar "Int")
| "starts_with" | "ends_with" | "remove" -> Some (Scalar "Bool")
| "substr" | "trim" | "to_lower" | "char_of" | "join" -> Some (Scalar "Text")
| "parse_int" -> Some (Nullable (Scalar "Int"))
| "split" | "split_ws" -> Some (Multi "Text")
| "slice" -> (
match argty with Some t -> ( match unwrap t with Multi e -> Some (Multi e) | _ -> None) | None -> None)
| "pop" | "shift" -> (
match argty with Some t -> ( match unwrap t with Multi e -> Some (Scalar e) | _ -> None) | None -> None)
| "key_at" -> (
match argty with Some t -> ( match unwrap t with Map (k, _) -> Some (Scalar k) | _ -> None) | None -> None)
| "val_at" -> (
match argty with Some t -> ( match unwrap t with Map (_, v) -> Some (Scalar v) | _ -> None) | None -> None)
| _ -> None
let is_builtin_name (n : string) =
List.mem n
[ "now"; "print"; "print_int"; "words"; "multi_new"; "push"; "get"; "count"; "latest";
"map_new"; "set"; "has"; "int_to_text" ]
"map_new"; "set"; "has"; "int_to_text";
(* systems stdlib *)
"len"; "byte_at"; "print_err"; "starts_with"; "ends_with"; "index_of"; "last_index_of";
"substr"; "trim"; "to_lower"; "char_of"; "parse_int"; "split"; "split_ws"; "join"; "slice";
"pop"; "shift"; "sort"; "reverse"; "remove"; "key_at"; "val_at" ]
(* ---- unions and variants (haxe-parity Task 4) ------------------------
@ -878,6 +925,8 @@ let rec ty_of_expr (p : pctx) (f : fstate) (e : Ast.expr) : Ast.field_ty option
| ListLit (first :: _) -> (
match ty_of_expr p f first with Some (Scalar n) -> Some (Multi n) | _ -> None)
| ListLit [] | MapLit -> None
(* haxe-parity Task 6: contextual on its destination (see owner.ml). *)
| NilLit -> None
(* haxe-parity Task 5: a `try` yields its try arm's type — types.ml has
already required the catch arm to agree. *)
| Try t -> ty_of_expr p f t.body
@ -1237,6 +1286,9 @@ let rec emit_expr (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e
| IntLit n -> put f (ins_abx op_loadk dst (check_bx p f e.pos "constant" (const_int p n)))
| BoolLit b -> put f (ins_abx op_loadk dst (check_bx p f e.pos "constant" (const_int p (if b then 1 else 0))))
| StrLit s -> put f (ins_abx op_loadk dst (check_bx p f e.pos "constant" (const_text p s)))
(* haxe-parity Task 6: `nil` is the zero word, whatever `?T` it stands
in for (docs/plan/oop-vm/08-builtin-surface.md's `?T` section). *)
| NilLit -> put f (ins_abx op_loadk dst (const_int p 0))
(* Container literals lower to exactly what `multi_new()`/`map_new()`
lower to — the element kinds are the destination's, never guessed
(docs/plan/oop-vm/08-builtin-surface.md) — plus one `multi_push` per
@ -2376,8 +2428,18 @@ and emit_builtin (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e :
else if
id = b_print || id = b_print_int || id = b_words || id = b_count || id = b_latest
|| id = b_int_to_text
(* systems stdlib, one argument *)
|| id = b_len || id = b_print_err || id = b_trim || id = b_to_lower || id = b_char_of
|| id = b_parse_int || id = b_split_ws || id = b_pop || id = b_shift || id = b_sort
|| id = b_reverse
then 1
else if id = b_multi_push || id = b_multi_get || id = b_map_get || id = b_map_has then 2
else if
id = b_multi_push || id = b_multi_get || id = b_map_get || id = b_map_has
(* systems stdlib, two arguments *)
|| id = b_byte_at || id = b_starts_with || id = b_ends_with || id = b_index_of
|| id = b_last_index_of || id = b_split || id = b_join || id = b_map_remove
|| id = b_map_key_at || id = b_map_val_at
then 2
else 3
in
let container_id first_arg on_multi on_map =
@ -2410,6 +2472,33 @@ and emit_builtin (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e :
| "count" -> fixed b_count
| "latest" -> fixed b_latest
| "int_to_text" -> fixed b_int_to_text
(* systems stdlib: every one of these resolves to a single id — no
container-kind branching, no destination immediate (the ones that
return a fresh container fix their own element kind: `split`/`split_ws`
are always `multi Text`, `slice` copies its source's kind). *)
| "len" -> fixed b_len
| "byte_at" -> fixed b_byte_at
| "print_err" -> fixed b_print_err
| "starts_with" -> fixed b_starts_with
| "ends_with" -> fixed b_ends_with
| "index_of" -> fixed b_index_of
| "last_index_of" -> fixed b_last_index_of
| "substr" -> fixed b_substr
| "trim" -> fixed b_trim
| "to_lower" -> fixed b_to_lower
| "char_of" -> fixed b_char_of
| "parse_int" -> fixed b_parse_int
| "split" -> fixed b_split
| "split_ws" -> fixed b_split_ws
| "join" -> fixed b_join
| "slice" -> fixed b_slice
| "pop" -> fixed b_pop
| "shift" -> fixed b_shift
| "sort" -> fixed b_sort
| "reverse" -> fixed b_reverse
| "remove" -> fixed b_map_remove
| "key_at" -> fixed b_map_key_at
| "val_at" -> fixed b_map_val_at
| "multi_new" | "map_new" ->
let is_map = name = "map_new" in
if args <> [] then bad (Printf.sprintf "builtin `%s` takes no arguments" name)

View file

@ -138,6 +138,7 @@ let keyword_kind = function
| "typedef" -> Some Token.KwTypedef
| "try" -> Some Token.KwTry
| "catch" -> Some Token.KwCatch
| "nil" -> Some Token.KwNil
| "INSERT" -> Some Token.KwInsert
| "SELECT" -> Some Token.KwSelect
| _ -> None

View file

@ -511,6 +511,9 @@ let rec expr_ty (ctx : ctx) (e : Ast.expr) : Ast.field_ty option =
| ListLit (first :: _) -> (
match expr_ty ctx first with Some (Scalar n) -> Some (Multi n) | _ -> None)
| ListLit [] | MapLit -> None
(* haxe-parity Task 6: `nil` is the zero word — contextual on its
destination, and never something this frame owns. *)
| NilLit -> None
(* haxe-parity Task 5: a `try` yields its try arm's type (types.ml has
already required the catch arm to agree). *)
| Try t -> expr_ty ctx t.body
@ -1065,7 +1068,7 @@ let rec read_expr (ctx : ctx) (e : Ast.expr) : unit =
it; it is a runtime-semantics gap to close with `push`, not a
literal-specific one. *)
| ListLit items -> List.iter (read_expr ctx) items
| MapLit -> ()
| MapLit | NilLit -> ()
| Try t -> analyze_try ctx e t.body t.ename t.handler
| DbStub _ ->
(* trap-capable: the frame needs its drop map here *)

View file

@ -1026,6 +1026,11 @@ and parse_primary (st : state) : Ast.expr =
let id = fresh_id st in
ignore (advance st);
{ Ast.id; pos; kind = Ast.BoolLit false }
| Token.KwNil ->
let pos = peek_pos st in
let id = fresh_id st in
ignore (advance st);
{ Ast.id; pos; kind = Ast.NilLit }
| Token.LParen ->
ignore (advance st);
(* Parens make the enclosed expression unambiguous again, so a
@ -1757,7 +1762,7 @@ let rec subst_expr (consts : Ast.expr StringMap.t) (bound : StringSet.t) (e : As
{ e with Ast.kind = Ast.Ctor (cn, List.map (fun (n, v) -> (n, subst_expr consts bound v)) fields) }
| Ast.Interp inner -> { e with Ast.kind = Ast.Interp (subst_expr consts bound inner) }
| Ast.ListLit items -> { e with Ast.kind = Ast.ListLit (List.map (subst_expr consts bound) items) }
| Ast.MapLit -> e
| Ast.MapLit | Ast.NilLit -> e
| Ast.Try { body; ename; handler } ->
{ e with
Ast.kind =

View file

@ -89,6 +89,10 @@ type kind =
followed). *)
| KwTry
| KwCatch
(* haxe-parity Task 6: the `?T` absent value. A keyword, not an
identifier — `nil` appears in the corpus and the driving workload
only ever as this literal. *)
| KwNil
(* haxe-parity Task 4: `typedef Name = { ... }` structural records. A
real keyword (grepped the corpus/sample first, same discipline as
every keyword above — `typedef` appears only as this declaration's

View file

@ -621,6 +621,33 @@ let builtin_signatures : (string * int * builtin_arg_req list) list =
("set", 3, [ ReqMap; ReqAny; ReqAny ]);
("has", 2, [ ReqMap; ReqAny ]);
("int_to_text", 1, [ ReqInt ]);
(* systems stdlib (docs/plan/oop-vm/08-builtin-surface.md): the text and
container vocabulary the driving workload writes. `len` resolves on a
text OR either container, so its argument is unchecked here the same
way `get`'s key is. *)
("len", 1, [ ReqAny ]);
("byte_at", 2, [ ReqText; ReqInt ]);
("print_err", 1, [ ReqText ]);
("starts_with", 2, [ ReqText; ReqText ]);
("ends_with", 2, [ ReqText; ReqText ]);
("index_of", 2, [ ReqText; ReqText ]);
("last_index_of", 2, [ ReqText; ReqText ]);
("substr", 3, [ ReqText; ReqInt; ReqInt ]);
("trim", 1, [ ReqText ]);
("to_lower", 1, [ ReqText ]);
("char_of", 1, [ ReqInt ]);
("parse_int", 1, [ ReqText ]);
("split", 2, [ ReqText; ReqText ]);
("split_ws", 1, [ ReqText ]);
("join", 2, [ ReqMulti; ReqText ]);
("slice", 3, [ ReqMulti; ReqInt; ReqInt ]);
("pop", 1, [ ReqMulti ]);
("shift", 1, [ ReqMulti ]);
("sort", 1, [ ReqMulti ]);
("reverse", 1, [ ReqMulti ]);
("remove", 2, [ ReqMap; ReqAny ]);
("key_at", 2, [ ReqMap; ReqInt ]);
("val_at", 2, [ ReqMap; ReqInt ]);
]
let rec unwrap_nullable (t : typ) : typ =
@ -692,6 +719,21 @@ let builtin_confident_ret (name : string) (arg0 : typ option) : typ option =
| "has" -> Some (TScalar "Bool")
| "latest" -> ( match arg0 with Some (TMulti e) -> Some e | _ -> None)
| "get" -> ( match arg0 with Some (TMulti e) -> Some e | Some (TMap (_, v)) -> Some v | _ -> None)
(* systems stdlib. Every fresh-Text and fresh-`multi` result is an owned
value, so these entries are what make a `let` holding one get its
drop (owner.ml classifies through this table too). *)
| "len" | "byte_at" | "index_of" | "last_index_of" -> Some (TScalar "Int")
| "print_err" | "sort" | "reverse" -> Some (TScalar "Int")
| "starts_with" | "ends_with" | "remove" -> Some (TScalar "Bool")
| "substr" | "trim" | "to_lower" | "char_of" | "join" -> Some (TScalar "Text")
(* `parse_int` is optional-shaped: an unparseable text is 0, which is how
a `?Int` spells nil (08-builtin-surface.md's `?T` section). *)
| "parse_int" -> Some (TNullable (TScalar "Int"))
| "split" | "split_ws" -> Some (TMulti (TScalar "Text"))
| "slice" -> ( match arg0 with Some (TMulti e) -> Some (TMulti e) | _ -> None)
| "pop" | "shift" -> ( match arg0 with Some (TMulti e) -> Some e | _ -> None)
| "key_at" -> ( match arg0 with Some (TMap (k, _)) -> Some k | _ -> None)
| "val_at" -> ( match arg0 with Some (TMap (_, v)) -> Some v | _ -> None)
| _ -> None
(* `use_edge`/`uses_of_program`/`path_str` -- relocated here (hotfix)
@ -871,6 +913,10 @@ let typecheck_program ~file ~(module_of : string -> string)
| ListLit (first :: _) -> (
match confident_typ cenv first with Some t -> Some (TMulti t) | None -> None)
| ListLit [] | MapLit -> None
(* haxe-parity Task 6: `nil` is contextual on its destination, the same
as an empty container literal — nothing about the literal itself
says which `?T` it is the absent value of. *)
| NilLit -> None
(* haxe-parity Task 5: a `try` expression's type is its try arm's — the
handler is checked to agree (typecheck_expr below), so either arm
would answer, and the try arm is the one that always has a value. *)
@ -1200,6 +1246,11 @@ let typecheck_program ~file ~(module_of : string -> string)
| t :: _ -> { typ = TMulti t; is_nil = false }
| [] -> { typ = TScalar "Int"; is_nil = false })
| MapLit -> { typ = TScalar "Int"; is_nil = false }
(* `is_nil` is what marks the literal: the `typ` is the same
placeholder every contextual value here reports, and the flag is
what lets a comparison or a binding treat it as the absent value of
whatever `?T` it meets. *)
| NilLit -> { typ = TScalar "Int"; is_nil = true }
| Try { body; ename; handler } ->
let body_res = typecheck_expr env cenv body in
(* The catch arm sees exactly one new name: the error record. *)
@ -1862,7 +1913,7 @@ and walk_expr (bound : StringSet.t) (visit : StringSet.t -> expr -> unit) (e : e
| Ctor (_, fields) -> List.iter (fun (_, v) -> walk_expr bound visit v) fields
| Interp inner -> walk_expr bound visit inner
| ListLit items -> List.iter (walk_expr bound visit) items
| MapLit -> ()
| MapLit | NilLit -> ()
| Try { body; ename; handler } ->
walk_expr bound visit body;
walk_block (StringSet.add ename bound) visit handler

View file

@ -23,6 +23,43 @@ static void *native_check(uint64_t v, uint32_t cls, const char **msg) {
return o;
}
/* ---- systems-stdlib helpers ------------------------------------------
* Text is bytes with an explicit length and no NUL, so every scan below is
* length-driven; "whitespace" is the same four bytes WO_B_WORDS already
* treats as separators. */
static int ws_byte(char c) { return c == ' ' || c == '\t' || c == '\n' || c == '\r'; }
/* First (dir > 0) or last (dir < 0) byte offset where [needle] occurs in
* [hay], or -1. An empty needle is found at 0 / at hay->len. */
static int64_t str_find(const wo_str *hay, const wo_str *needle, int dir) {
if (needle->len > hay->len) return -1;
uint32_t span = hay->len - needle->len;
if (dir > 0) {
for (uint32_t i = 0; i <= span; i++)
if (!memcmp(hay->data + i, needle->data, needle->len)) return (int64_t)i;
} else {
for (uint32_t i = span + 1; i > 0; i--)
if (!memcmp(hay->data + (i - 1), needle->data, needle->len))
return (int64_t)(i - 1);
}
return -1;
}
/* Element compare for WO_B_SORT: Text elements by content (memcmp over the
* shared prefix, then length), everything else as signed integers. */
static int elem_cmp(uint8_t kind, uint64_t a, uint64_t b) {
if (kind == WO_K_TEXT) {
const wo_str *x = (const wo_str *)(uintptr_t)a, *y = (const wo_str *)(uintptr_t)b;
if (!x || !y) return (x ? 1 : 0) - (y ? 1 : 0);
uint32_t n = x->len < y->len ? x->len : y->len;
int c = n ? memcmp(x->data, y->data, n) : 0;
if (c) return c;
return x->len == y->len ? 0 : (x->len < y->len ? -1 : 1);
}
int64_t ia = (int64_t)a, ib = (int64_t)b;
return ia == ib ? 0 : (ia < ib ? -1 : 1);
}
int wo_builtin(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
wo_rt *rt = &vm->rt;
uint8_t A = wo_ins_a(ins), B = wo_ins_b(ins), C = wo_ins_c(ins);
@ -216,6 +253,335 @@ int wo_builtin(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) {
R[A] = R[B];
return 0;
}
/* ---- systems stdlib: text ---------------------------------------- */
case WO_B_LEN: { /* one name for "how many": bytes, elements, entries */
if (!R[B]) {
*msg = "null receiver";
return WO_T_BOUNDS;
}
wo_hdr *o = (wo_hdr *)(uintptr_t)R[B];
if (o->class_id == WO_CLS_STR) R[A] = ((wo_str *)o)->len;
else if (o->class_id == WO_CLS_MULTI) R[A] = ((wo_multi *)o)->len;
else if (o->class_id == WO_CLS_MAP) R[A] = ((wo_map *)o)->len;
else {
*msg = "`len` needs a text, a multi, or a map";
return WO_T_BOUNDS;
}
return 0;
}
case WO_B_BYTE_AT: {
wo_str *s = native_check(R[B], WO_CLS_STR, msg);
if (!s) return WO_T_BOUNDS;
uint64_t i = R[B + 1];
if (i >= s->len) {
*msg = "byte index out of range";
return WO_T_BOUNDS;
}
R[A] = (uint64_t)(uint8_t)s->data[i];
return 0;
}
case WO_B_PRINT_ERR: {
wo_str *s = native_check(R[B], WO_CLS_STR, msg);
if (!s) return WO_T_BOUNDS;
fwrite(s->data, 1, s->len, stderr);
fputc('\n', stderr);
R[A] = 0;
return 0;
}
case WO_B_STARTS_WITH:
case WO_B_ENDS_WITH: {
wo_str *s = native_check(R[B], WO_CLS_STR, msg);
if (!s) return WO_T_BOUNDS;
wo_str *fix = native_check(R[B + 1], WO_CLS_STR, msg);
if (!fix) return WO_T_BOUNDS;
if (fix->len > s->len) R[A] = 0;
else {
const char *at = C == WO_B_STARTS_WITH ? s->data : s->data + (s->len - fix->len);
R[A] = memcmp(at, fix->data, fix->len) ? 0 : 1;
}
return 0;
}
case WO_B_INDEX_OF:
case WO_B_LAST_INDEX_OF: {
wo_str *s = native_check(R[B], WO_CLS_STR, msg);
if (!s) return WO_T_BOUNDS;
wo_str *n = native_check(R[B + 1], WO_CLS_STR, msg);
if (!n) return WO_T_BOUNDS;
R[A] = (uint64_t)str_find(s, n, C == WO_B_INDEX_OF ? 1 : -1);
return 0;
}
case WO_B_SUBSTR: { /* clamped, never trapping: a start past the end or a
* length past the end yields the empty/short text */
wo_str *s = native_check(R[B], WO_CLS_STR, msg);
if (!s) return WO_T_BOUNDS;
int64_t start = (int64_t)R[B + 1], want = (int64_t)R[B + 2];
if (start < 0) start = 0;
if (start > (int64_t)s->len) start = s->len;
if (want < 0) want = 0;
if (start + want > (int64_t)s->len) want = (int64_t)s->len - start;
wo_str *out = wo_str_new(rt, s->data + start, (uint32_t)want);
if (!out) {
*msg = "out of memory";
return WO_T_OOM;
}
R[A] = (uint64_t)(uintptr_t)out;
return 0;
}
case WO_B_TRIM: {
wo_str *s = native_check(R[B], WO_CLS_STR, msg);
if (!s) return WO_T_BOUNDS;
uint32_t lo = 0, hi = s->len;
while (lo < hi && ws_byte(s->data[lo])) lo++;
while (hi > lo && ws_byte(s->data[hi - 1])) hi--;
wo_str *out = wo_str_new(rt, s->data + lo, hi - lo);
if (!out) {
*msg = "out of memory";
return WO_T_OOM;
}
R[A] = (uint64_t)(uintptr_t)out;
return 0;
}
case WO_B_TO_LOWER: {
wo_str *s = native_check(R[B], WO_CLS_STR, msg);
if (!s) return WO_T_BOUNDS;
wo_str *out = wo_str_new(rt, s->data, s->len);
if (!out) {
*msg = "out of memory";
return WO_T_OOM;
}
for (uint32_t i = 0; i < out->len; i++)
if (out->data[i] >= 'A' && out->data[i] <= 'Z') out->data[i] += 32;
R[A] = (uint64_t)(uintptr_t)out;
return 0;
}
case WO_B_CHAR_OF: {
char c = (char)(uint8_t)R[B];
wo_str *out = wo_str_new(rt, &c, 1);
if (!out) {
*msg = "out of memory";
return WO_T_OOM;
}
R[A] = (uint64_t)(uintptr_t)out;
return 0;
}
case WO_B_PARSE_INT: { /* optional-shaped: unparseable is 0, which is
* exactly how a `?Int` spells nil */
wo_str *s = native_check(R[B], WO_CLS_STR, msg);
if (!s) return WO_T_BOUNDS;
uint32_t i = 0;
int neg = 0;
while (i < s->len && ws_byte(s->data[i])) i++;
if (i < s->len && (s->data[i] == '-' || s->data[i] == '+')) neg = s->data[i++] == '-';
int64_t acc = 0;
int digits = 0;
while (i < s->len && s->data[i] >= '0' && s->data[i] <= '9') {
acc = acc * 10 + (s->data[i++] - '0');
digits++;
}
R[A] = digits ? (uint64_t)(neg ? -acc : acc) : 0;
return 0;
}
case WO_B_SPLIT:
case WO_B_SPLIT_WS: {
wo_str *s = native_check(R[B], WO_CLS_STR, msg);
if (!s) return WO_T_BOUNDS;
wo_str *sep = NULL;
if (C == WO_B_SPLIT) {
sep = native_check(R[B + 1], WO_CLS_STR, msg);
if (!sep) return WO_T_BOUNDS;
if (sep->len == 0) {
*msg = "`split` needs a non-empty separator";
return WO_T_BOUNDS;
}
}
wo_multi *out = wo_multi_new(rt, WO_K_TEXT);
if (!out) {
*msg = "out of memory";
return WO_T_OOM;
}
uint32_t i = 0;
while (i <= s->len) {
uint32_t start = i, end;
if (C == WO_B_SPLIT) {
end = s->len;
for (uint32_t j = i; j + sep->len <= s->len; j++)
if (!memcmp(s->data + j, sep->data, sep->len)) {
end = j;
break;
}
i = end + sep->len;
} else {
while (start < s->len && ws_byte(s->data[start])) start++;
if (start >= s->len) break;
end = start;
while (end < s->len && !ws_byte(s->data[end])) end++;
i = end;
}
wo_str *part = wo_str_new(rt, s->data + start, end - start);
if (!part || wo_multi_push(out, (uint64_t)(uintptr_t)part) != 0) {
if (part) wo_str_free(rt, part);
wo_drop_obj(rt, &out->h);
*msg = "out of memory";
return WO_T_OOM;
}
if (C == WO_B_SPLIT && end == s->len) break;
}
R[A] = (uint64_t)(uintptr_t)out;
return 0;
}
case WO_B_JOIN: {
wo_multi *m = native_check(R[B], WO_CLS_MULTI, msg);
if (!m) return WO_T_BOUNDS;
wo_str *sep = native_check(R[B + 1], WO_CLS_STR, msg);
if (!sep) return WO_T_BOUNDS;
if (m->elem_kind != WO_K_TEXT) {
*msg = "`join` needs a `multi Text`";
return WO_T_BOUNDS;
}
uint32_t total = m->len ? (m->len - 1) * sep->len : 0;
for (uint32_t i = 0; i < m->len; i++) {
const wo_str *e = (const wo_str *)(uintptr_t)m->items[i];
if (e) total += e->len;
}
wo_str *out = wo_str_alloc(rt, total);
if (!out) {
*msg = "out of memory";
return WO_T_OOM;
}
uint32_t at = 0;
for (uint32_t i = 0; i < m->len; i++) {
if (i && sep->len) {
memcpy(out->data + at, sep->data, sep->len);
at += sep->len;
}
const wo_str *e = (const wo_str *)(uintptr_t)m->items[i];
if (e && e->len) {
memcpy(out->data + at, e->data, e->len);
at += e->len;
}
}
R[A] = (uint64_t)(uintptr_t)out;
return 0;
}
/* ---- systems stdlib: containers ---------------------------------- */
case WO_B_SLICE: { /* [from, to) — Text elements are COPIED so the slice
* and its source never both own one value */
wo_multi *m = native_check(R[B], WO_CLS_MULTI, msg);
if (!m) return WO_T_BOUNDS;
if (m->elem_kind != WO_K_TEXT && m->elem_kind != WO_K_SCALAR) {
*msg = "`slice` needs a `multi Text` or a `multi` of scalars";
return WO_T_BOUNDS;
}
int64_t from = (int64_t)R[B + 1], to = (int64_t)R[B + 2];
if (from < 0) from = 0;
if (to > (int64_t)m->len) to = m->len;
wo_multi *out = wo_multi_new(rt, m->elem_kind);
if (!out) {
*msg = "out of memory";
return WO_T_OOM;
}
for (int64_t i = from; i < to; i++) {
uint64_t v = m->items[i];
if (m->elem_kind == WO_K_TEXT && v) {
const wo_str *e = (const wo_str *)(uintptr_t)v;
wo_str *cp = wo_str_new(rt, e->data, e->len);
if (!cp) {
wo_drop_obj(rt, &out->h);
*msg = "out of memory";
return WO_T_OOM;
}
v = (uint64_t)(uintptr_t)cp;
}
if (wo_multi_push(out, v) != 0) {
wo_drop_obj(rt, &out->h);
*msg = "out of memory";
return WO_T_OOM;
}
}
R[A] = (uint64_t)(uintptr_t)out;
return 0;
}
case WO_B_POP:
case WO_B_SHIFT: { /* the element LEAVES the container: its ownership goes
* to the caller's register, so nothing is dropped here */
wo_multi *m = native_check(R[B], WO_CLS_MULTI, msg);
if (!m) return WO_T_BOUNDS;
if (m->len == 0) {
*msg = C == WO_B_POP ? "`pop` on an empty multi" : "`shift` on an empty multi";
return WO_T_BOUNDS;
}
if (C == WO_B_POP) R[A] = m->items[--m->len];
else {
R[A] = m->items[0];
memmove(m->items, m->items + 1, (size_t)(m->len - 1) * sizeof(uint64_t));
m->len--;
}
return 0;
}
case WO_B_SORT: { /* insertion sort: in place, stable, and the workload's
* lists are short (a directory's file names) */
wo_multi *m = native_check(R[B], WO_CLS_MULTI, msg);
if (!m) return WO_T_BOUNDS;
for (uint32_t i = 1; i < m->len; i++) {
uint64_t v = m->items[i];
uint32_t j = i;
while (j > 0 && elem_cmp(m->elem_kind, m->items[j - 1], v) > 0) {
m->items[j] = m->items[j - 1];
j--;
}
m->items[j] = v;
}
R[A] = 0;
return 0;
}
case WO_B_REVERSE: {
wo_multi *m = native_check(R[B], WO_CLS_MULTI, msg);
if (!m) return WO_T_BOUNDS;
for (uint32_t i = 0, j = m->len; i + 1 < j; i++, j--) {
uint64_t t = m->items[i];
m->items[i] = m->items[j - 1];
m->items[j - 1] = t;
}
R[A] = 0;
return 0;
}
case WO_B_MAP_REMOVE: { /* the map owned the key and the value, so both die
* here (their kinds are the map's own drop plan) */
wo_map *m = native_check(R[B], WO_CLS_MAP, msg);
if (!m) return WO_T_BOUNDS;
uint64_t k = R[B + 1];
for (uint32_t i = 0; i < m->len; i++) {
int hit = m->key_kind == WO_K_TEXT
? (m->keys[i] && k &&
wo_str_eq((const wo_str *)(uintptr_t)m->keys[i],
(const wo_str *)(uintptr_t)k))
: m->keys[i] == k;
if (!hit) continue;
wo_drop_kind(rt, m->key_kind, m->keys[i]);
wo_drop_kind(rt, m->val_kind, m->vals[i]);
memmove(m->keys + i, m->keys + i + 1, (size_t)(m->len - i - 1) * sizeof(uint64_t));
memmove(m->vals + i, m->vals + i + 1, (size_t)(m->len - i - 1) * sizeof(uint64_t));
m->len--;
R[A] = 1;
return 0;
}
R[A] = 0;
return 0;
}
case WO_B_MAP_KEY_AT:
case WO_B_MAP_VAL_AT: { /* slot-indexed enumeration — what `for k, v in m`
* lowers onto (the parallel arrays are insertion
* ordered, cont.h) */
wo_map *m = native_check(R[B], WO_CLS_MAP, msg);
if (!m) return WO_T_BOUNDS;
uint64_t i = R[B + 1];
if (i >= m->len) {
*msg = "map slot out of range";
return WO_T_BOUNDS;
}
R[A] = C == WO_B_MAP_KEY_AT ? m->keys[i] : m->vals[i];
return 0;
}
default: /* unreachable: loader validated the id */
*msg = "unknown builtin";
return WO_T_EXPLICIT;

View file

@ -42,6 +42,15 @@ static const uint8_t b_arity[WO_B_MAX + 1] = {
[WO_B_MAP_NEW] = 0, [WO_B_MAP_SET] = 3, [WO_B_MAP_GET] = 2,
[WO_B_MAP_HAS] = 2, [WO_B_INT_TO_TEXT] = 1,
[WO_B_VARIANT_TAG] = 1, [WO_B_ERR_FILL] = 1,
/* systems stdlib */
[WO_B_LEN] = 1, [WO_B_BYTE_AT] = 2, [WO_B_PRINT_ERR] = 1,
[WO_B_STARTS_WITH] = 2, [WO_B_ENDS_WITH] = 2, [WO_B_INDEX_OF] = 2,
[WO_B_LAST_INDEX_OF] = 2, [WO_B_SUBSTR] = 3, [WO_B_TRIM] = 1,
[WO_B_TO_LOWER] = 1, [WO_B_CHAR_OF] = 1, [WO_B_PARSE_INT] = 1,
[WO_B_SPLIT] = 2, [WO_B_SPLIT_WS] = 1, [WO_B_JOIN] = 2,
[WO_B_SLICE] = 3, [WO_B_POP] = 1, [WO_B_SHIFT] = 1,
[WO_B_SORT] = 1, [WO_B_REVERSE] = 1, [WO_B_MAP_REMOVE] = 2,
[WO_B_MAP_KEY_AT] = 2, [WO_B_MAP_VAL_AT] = 2,
};
static int vtab_cmp(const void *a, const void *b) {

View file

@ -76,7 +76,7 @@ wo_hdr *wo_obj_new(wo_rt *rt, uint32_t class_id) {
return o;
}
static wo_str *str_alloc(wo_rt *rt, uint32_t len) {
wo_str *wo_str_alloc(wo_rt *rt, uint32_t len) {
wo_str *s = wo_arena_alloc(&rt->arena, sizeof(wo_str) + len);
if (!s) return NULL;
memset(&s->h, 0, sizeof(s->h));
@ -86,14 +86,14 @@ static wo_str *str_alloc(wo_rt *rt, uint32_t len) {
}
wo_str *wo_str_new(wo_rt *rt, const char *bytes, uint32_t len) {
wo_str *s = str_alloc(rt, len);
wo_str *s = wo_str_alloc(rt, len);
if (!s) return NULL;
memcpy(s->data, bytes, len);
return s;
}
wo_str *wo_str_concat(wo_rt *rt, const wo_str *a, const wo_str *b) {
wo_str *s = str_alloc(rt, a->len + b->len);
wo_str *s = wo_str_alloc(rt, a->len + b->len);
if (!s) return NULL;
memcpy(s->data, a->data, a->len);
memcpy(s->data + a->len, b->data, b->len);

View file

@ -56,6 +56,10 @@ typedef struct wo_str {
} wo_str;
wo_str *wo_str_new(wo_rt *rt, const char *bytes, uint32_t len); /* NULL=OOM */
/* A Text of [len] UNINITIALIZED bytes for a caller that writes them itself
* (the systems-stdlib `join`, which knows the total length up front and
* would otherwise need one allocation per element). NULL = OOM. */
wo_str *wo_str_alloc(wo_rt *rt, uint32_t len);
wo_str *wo_str_concat(wo_rt *rt, const wo_str *a, const wo_str *b);
int wo_str_eq(const wo_str *a, const wo_str *b); /* content equality */
void wo_str_free(wo_rt *rt, wo_str *s); /* no-op on WO_F_CONST */

View file

@ -176,8 +176,37 @@ enum {
* a 4-field class object, WO_T_OOM if either Text cannot be
* allocated. */
WO_B_ERR_FILL = 15,
/* ---- systems stdlib: text and container surface (the driving
* workload's own vocabulary — docs/plan/oop-vm/08-builtin-surface.md
* lists the source spelling of each). Every one that returns a fresh
* Text or a fresh `multi` allocates it here, so a `?T` result spells
* absence as 0 like every other nullable. ---- */
WO_B_LEN = 16, /* (text|multi|map) -> i64 length */
WO_B_BYTE_AT = 17, /* (text, i) -> i64 byte; out of range traps BOUNDS */
WO_B_PRINT_ERR = 18, /* (text) -> stderr, newline-terminated */
WO_B_STARTS_WITH = 19, /* (text, prefix) -> 1/0 */
WO_B_ENDS_WITH = 20, /* (text, suffix) -> 1/0 */
WO_B_INDEX_OF = 21, /* (text, needle) -> first byte offset, -1 = absent */
WO_B_LAST_INDEX_OF = 22, /* (text, needle) -> last byte offset, -1 = absent */
WO_B_SUBSTR = 23, /* (text, start, len) -> fresh Text, clamped */
WO_B_TRIM = 24, /* (text) -> fresh Text without leading/trailing space */
WO_B_TO_LOWER = 25, /* (text) -> fresh Text, ASCII-lowercased */
WO_B_CHAR_OF = 26, /* (i64) -> fresh one-byte Text */
WO_B_PARSE_INT = 27, /* (text) -> i64; unparseable is 0, `?Int`'s own nil */
WO_B_SPLIT = 28, /* (text, sep) -> fresh multi Text */
WO_B_SPLIT_WS = 29, /* (text) -> fresh multi Text, whitespace-separated */
WO_B_JOIN = 30, /* (multi Text, sep) -> fresh Text */
WO_B_SLICE = 31, /* (multi, from, to) -> fresh multi; Text elements are
* COPIED, so the two containers never share a value */
WO_B_POP = 32, /* (multi) -> last element, removed; empty traps BOUNDS */
WO_B_SHIFT = 33, /* (multi) -> first element, removed; empty traps BOUNDS */
WO_B_SORT = 34, /* (multi) -> 0; in place, Text by content else by value */
WO_B_REVERSE = 35, /* (multi) -> 0; in place */
WO_B_MAP_REMOVE = 36, /* (map, key) -> 1/0; drops the removed key and value */
WO_B_MAP_KEY_AT = 37, /* (map, i) -> key at slot i (insertion order) */
WO_B_MAP_VAL_AT = 38, /* (map, i) -> value at slot i */
};
#define WO_B_MAX 15u
#define WO_B_MAX 38u
/* ---- instruction encode/decode: op:8 A:8 then B:8 C:8 or Bx:16 ---- */
static inline uint32_t wo_ins_abc(uint8_t op, uint8_t a, uint8_t b, uint8_t c) {