From e5c5952cdcce3673e15f44ac5a77a0099eb24469 Mon Sep 17 00:00:00 2001 From: "shoney.arickathil" Date: Fri, 14 Aug 2026 16:43:38 +0200 Subject: [PATCH] feat: nil literal + systems-stdlib text/container builtins MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit log-watcher diagnostics 272 -> 129 (parse errors 7, stdlib-module calls 49, lowering gaps 72). corpus 71/0, woc runtest 565/0, wovm unit gates green. - nil (haxe-parity Task 6's literal half): `nil` keyword, Ast.NilLit, lowered to the zero word for every `?T` — the representation the format doc already fixes ("a nullable field stores exactly what T stores and spells nil as 0"), so no boxing, no unbox on read, and every drop plan already skips it. Contextual on its destination in both type derivers, like `[]`/`{}` - 23 new builtins (wob.h ids 16..38, loader arities, builtin.c): len, byte_at, print_err, starts_with, ends_with, index_of, last_index_of, substr, trim, to_lower, char_of, parse_int, split, split_ws, join, slice, pop, shift, sort, reverse, remove, key_at, val_at - fresh-Text/fresh-multi results allocate in the VM; `split`/`split_ws` fix their element kind (Text), `slice` copies its source's — and COPIES Text elements so a slice and its source never both own one value - pop/shift hand the element's ownership to the caller; remove drops the map's own key and value; key_at/val_at expose slot-ordered enumeration (what `for k, v in m` will lower onto) - parse_int is optional-shaped: unparseable is 0, `?Int`'s own nil - obj.c/obj.h: wo_str_alloc (uninitialized Text of known length) so `join` builds its result in one allocation instead of one per element - types.ml/emit.ml: builtin signatures, argument-shape requirements and return types for all 23 — the return table is also what classifies a `let` holding a fresh Text or multi as owned, so an omission there is a leak Co-Authored-By: Claude Opus 5 (1M context) --- compiler/src/ast.ml | 6 + compiler/src/dump.ml | 2 + compiler/src/emit.ml | 93 ++++++++++- compiler/src/lexer.ml | 1 + compiler/src/owner.ml | 5 +- compiler/src/parser.ml | 7 +- compiler/src/token.ml | 4 + compiler/src/types.ml | 53 +++++- runtime/src/builtin.c | 366 +++++++++++++++++++++++++++++++++++++++++ runtime/src/loader.c | 9 + runtime/src/obj.c | 6 +- runtime/src/obj.h | 4 + runtime/src/wob.h | 31 +++- 13 files changed, 578 insertions(+), 9 deletions(-) diff --git a/compiler/src/ast.ml b/compiler/src/ast.ml index da4b8ff..8564ad6 100644 --- a/compiler/src/ast.ml +++ b/compiler/src/ast.ml @@ -197,6 +197,12 @@ and expr_kind = | IntLit of int | StrLit of string | BoolLit of bool + (* haxe-parity Task 6: `nil`, the absent value of a `?T`. One + representation for every T: the zero word — "a nullable field stores + exactly what T stores and spells nil as 0", docs/plan/oop-vm/ + 08-builtin-surface.md. Nothing to allocate, nothing to unbox, and + every per-kind drop plan already ignores a zero slot. *) + | NilLit | Ident of string | Field of expr * string | Index of expr * expr diff --git a/compiler/src/dump.ml b/compiler/src/dump.ml index fd81e8a..5012ded 100644 --- a/compiler/src/dump.ml +++ b/compiler/src/dump.ml @@ -65,6 +65,7 @@ let kind_label (k : Token.kind) : string = | Token.KwTypedef -> "KW_TYPEDEF" | Token.KwTry -> "KW_TRY" | Token.KwCatch -> "KW_CATCH" + | Token.KwNil -> "KW_NIL" | Token.LBrace -> "LBRACE" | Token.RBrace -> "RBRACE" | Token.LParen -> "LPAREN" @@ -224,6 +225,7 @@ let rec expr_str (e : Ast.expr) : string = | Ast.Interp inner -> Printf.sprintf "INTERP(%s)" (expr_str inner) | Ast.ListLit items -> Printf.sprintf "[%s]" (String.concat ", " (List.map expr_str items)) | Ast.MapLit -> "{}" + | Ast.NilLit -> "nil" (* Like SWITCH above: a one-line summary, not a full unparse of the catch arm's statements. *) | Ast.Try { body; ename; handler } -> diff --git a/compiler/src/emit.ml b/compiler/src/emit.ml index ea9646d..cd7df77 100644 --- a/compiler/src/emit.ml +++ b/compiler/src/emit.ml @@ -230,6 +230,31 @@ let b_variant_tag = 14 source-callable name. *) let b_err_fill = 15 +(* systems stdlib (runtime/src/wob.h WO_B_LEN..WO_B_MAP_VAL_AT) *) +let b_len = 16 +let b_byte_at = 17 +let b_print_err = 18 +let b_starts_with = 19 +let b_ends_with = 20 +let b_index_of = 21 +let b_last_index_of = 22 +let b_substr = 23 +let b_trim = 24 +let b_to_lower = 25 +let b_char_of = 26 +let b_parse_int = 27 +let b_split = 28 +let b_split_ws = 29 +let b_join = 30 +let b_slice = 31 +let b_pop = 32 +let b_shift = 33 +let b_sort = 34 +let b_reverse = 35 +let b_map_remove = 36 +let b_map_key_at = 37 +let b_map_val_at = 38 + let ins_abc op a b c = op lor (a lsl 8) lor (b lsl 16) lor (c lsl 24) let ins_abx op a bx = op lor (a lsl 8) lor (bx lsl 16) let ins_asbx op a sbx = ins_abx op a (sbx + 32768) @@ -831,12 +856,34 @@ let builtin_ret (name : string) (argty : Ast.field_ty option) : Ast.field_ty opt match argty with | Some t -> ( match unwrap t with Multi e -> Some (Scalar e) | Map (_, v) -> Some (Scalar v) | _ -> None) | None -> None) + (* systems stdlib — kept in sync with Types.builtin_confident_ret (both + tables, same contract, different type languages). This is also what + classifies a `let` holding a fresh Text or a fresh `multi` as owned, so + a missing entry here is a leak, not just a lost type. *) + | "len" | "byte_at" | "index_of" | "last_index_of" -> Some (Scalar "Int") + | "print_err" | "sort" | "reverse" -> Some (Scalar "Int") + | "starts_with" | "ends_with" | "remove" -> Some (Scalar "Bool") + | "substr" | "trim" | "to_lower" | "char_of" | "join" -> Some (Scalar "Text") + | "parse_int" -> Some (Nullable (Scalar "Int")) + | "split" | "split_ws" -> Some (Multi "Text") + | "slice" -> ( + match argty with Some t -> ( match unwrap t with Multi e -> Some (Multi e) | _ -> None) | None -> None) + | "pop" | "shift" -> ( + match argty with Some t -> ( match unwrap t with Multi e -> Some (Scalar e) | _ -> None) | None -> None) + | "key_at" -> ( + match argty with Some t -> ( match unwrap t with Map (k, _) -> Some (Scalar k) | _ -> None) | None -> None) + | "val_at" -> ( + match argty with Some t -> ( match unwrap t with Map (_, v) -> Some (Scalar v) | _ -> None) | None -> None) | _ -> None let is_builtin_name (n : string) = List.mem n [ "now"; "print"; "print_int"; "words"; "multi_new"; "push"; "get"; "count"; "latest"; - "map_new"; "set"; "has"; "int_to_text" ] + "map_new"; "set"; "has"; "int_to_text"; + (* systems stdlib *) + "len"; "byte_at"; "print_err"; "starts_with"; "ends_with"; "index_of"; "last_index_of"; + "substr"; "trim"; "to_lower"; "char_of"; "parse_int"; "split"; "split_ws"; "join"; "slice"; + "pop"; "shift"; "sort"; "reverse"; "remove"; "key_at"; "val_at" ] (* ---- unions and variants (haxe-parity Task 4) ------------------------ @@ -878,6 +925,8 @@ let rec ty_of_expr (p : pctx) (f : fstate) (e : Ast.expr) : Ast.field_ty option | ListLit (first :: _) -> ( match ty_of_expr p f first with Some (Scalar n) -> Some (Multi n) | _ -> None) | ListLit [] | MapLit -> None + (* haxe-parity Task 6: contextual on its destination (see owner.ml). *) + | NilLit -> None (* haxe-parity Task 5: a `try` yields its try arm's type — types.ml has already required the catch arm to agree. *) | Try t -> ty_of_expr p f t.body @@ -1237,6 +1286,9 @@ let rec emit_expr (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e | IntLit n -> put f (ins_abx op_loadk dst (check_bx p f e.pos "constant" (const_int p n))) | BoolLit b -> put f (ins_abx op_loadk dst (check_bx p f e.pos "constant" (const_int p (if b then 1 else 0)))) | StrLit s -> put f (ins_abx op_loadk dst (check_bx p f e.pos "constant" (const_text p s))) + (* haxe-parity Task 6: `nil` is the zero word, whatever `?T` it stands + in for (docs/plan/oop-vm/08-builtin-surface.md's `?T` section). *) + | NilLit -> put f (ins_abx op_loadk dst (const_int p 0)) (* Container literals lower to exactly what `multi_new()`/`map_new()` lower to — the element kinds are the destination's, never guessed (docs/plan/oop-vm/08-builtin-surface.md) — plus one `multi_push` per @@ -2376,8 +2428,18 @@ and emit_builtin (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e : else if id = b_print || id = b_print_int || id = b_words || id = b_count || id = b_latest || id = b_int_to_text + (* systems stdlib, one argument *) + || id = b_len || id = b_print_err || id = b_trim || id = b_to_lower || id = b_char_of + || id = b_parse_int || id = b_split_ws || id = b_pop || id = b_shift || id = b_sort + || id = b_reverse then 1 - else if id = b_multi_push || id = b_multi_get || id = b_map_get || id = b_map_has then 2 + else if + id = b_multi_push || id = b_multi_get || id = b_map_get || id = b_map_has + (* systems stdlib, two arguments *) + || id = b_byte_at || id = b_starts_with || id = b_ends_with || id = b_index_of + || id = b_last_index_of || id = b_split || id = b_join || id = b_map_remove + || id = b_map_key_at || id = b_map_val_at + then 2 else 3 in let container_id first_arg on_multi on_map = @@ -2410,6 +2472,33 @@ and emit_builtin (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e : | "count" -> fixed b_count | "latest" -> fixed b_latest | "int_to_text" -> fixed b_int_to_text + (* systems stdlib: every one of these resolves to a single id — no + container-kind branching, no destination immediate (the ones that + return a fresh container fix their own element kind: `split`/`split_ws` + are always `multi Text`, `slice` copies its source's kind). *) + | "len" -> fixed b_len + | "byte_at" -> fixed b_byte_at + | "print_err" -> fixed b_print_err + | "starts_with" -> fixed b_starts_with + | "ends_with" -> fixed b_ends_with + | "index_of" -> fixed b_index_of + | "last_index_of" -> fixed b_last_index_of + | "substr" -> fixed b_substr + | "trim" -> fixed b_trim + | "to_lower" -> fixed b_to_lower + | "char_of" -> fixed b_char_of + | "parse_int" -> fixed b_parse_int + | "split" -> fixed b_split + | "split_ws" -> fixed b_split_ws + | "join" -> fixed b_join + | "slice" -> fixed b_slice + | "pop" -> fixed b_pop + | "shift" -> fixed b_shift + | "sort" -> fixed b_sort + | "reverse" -> fixed b_reverse + | "remove" -> fixed b_map_remove + | "key_at" -> fixed b_map_key_at + | "val_at" -> fixed b_map_val_at | "multi_new" | "map_new" -> let is_map = name = "map_new" in if args <> [] then bad (Printf.sprintf "builtin `%s` takes no arguments" name) diff --git a/compiler/src/lexer.ml b/compiler/src/lexer.ml index 336914f..0278fcb 100644 --- a/compiler/src/lexer.ml +++ b/compiler/src/lexer.ml @@ -138,6 +138,7 @@ let keyword_kind = function | "typedef" -> Some Token.KwTypedef | "try" -> Some Token.KwTry | "catch" -> Some Token.KwCatch + | "nil" -> Some Token.KwNil | "INSERT" -> Some Token.KwInsert | "SELECT" -> Some Token.KwSelect | _ -> None diff --git a/compiler/src/owner.ml b/compiler/src/owner.ml index c6e93b3..aaf80a6 100644 --- a/compiler/src/owner.ml +++ b/compiler/src/owner.ml @@ -511,6 +511,9 @@ let rec expr_ty (ctx : ctx) (e : Ast.expr) : Ast.field_ty option = | ListLit (first :: _) -> ( match expr_ty ctx first with Some (Scalar n) -> Some (Multi n) | _ -> None) | ListLit [] | MapLit -> None + (* haxe-parity Task 6: `nil` is the zero word — contextual on its + destination, and never something this frame owns. *) + | NilLit -> None (* haxe-parity Task 5: a `try` yields its try arm's type (types.ml has already required the catch arm to agree). *) | Try t -> expr_ty ctx t.body @@ -1065,7 +1068,7 @@ let rec read_expr (ctx : ctx) (e : Ast.expr) : unit = it; it is a runtime-semantics gap to close with `push`, not a literal-specific one. *) | ListLit items -> List.iter (read_expr ctx) items - | MapLit -> () + | MapLit | NilLit -> () | Try t -> analyze_try ctx e t.body t.ename t.handler | DbStub _ -> (* trap-capable: the frame needs its drop map here *) diff --git a/compiler/src/parser.ml b/compiler/src/parser.ml index a4ab3ed..25ea599 100644 --- a/compiler/src/parser.ml +++ b/compiler/src/parser.ml @@ -1026,6 +1026,11 @@ and parse_primary (st : state) : Ast.expr = let id = fresh_id st in ignore (advance st); { Ast.id; pos; kind = Ast.BoolLit false } + | Token.KwNil -> + let pos = peek_pos st in + let id = fresh_id st in + ignore (advance st); + { Ast.id; pos; kind = Ast.NilLit } | Token.LParen -> ignore (advance st); (* Parens make the enclosed expression unambiguous again, so a @@ -1757,7 +1762,7 @@ let rec subst_expr (consts : Ast.expr StringMap.t) (bound : StringSet.t) (e : As { e with Ast.kind = Ast.Ctor (cn, List.map (fun (n, v) -> (n, subst_expr consts bound v)) fields) } | Ast.Interp inner -> { e with Ast.kind = Ast.Interp (subst_expr consts bound inner) } | Ast.ListLit items -> { e with Ast.kind = Ast.ListLit (List.map (subst_expr consts bound) items) } - | Ast.MapLit -> e + | Ast.MapLit | Ast.NilLit -> e | Ast.Try { body; ename; handler } -> { e with Ast.kind = diff --git a/compiler/src/token.ml b/compiler/src/token.ml index 712d019..8ebdc8d 100644 --- a/compiler/src/token.ml +++ b/compiler/src/token.ml @@ -89,6 +89,10 @@ type kind = followed). *) | KwTry | KwCatch + (* haxe-parity Task 6: the `?T` absent value. A keyword, not an + identifier — `nil` appears in the corpus and the driving workload + only ever as this literal. *) + | KwNil (* haxe-parity Task 4: `typedef Name = { ... }` structural records. A real keyword (grepped the corpus/sample first, same discipline as every keyword above — `typedef` appears only as this declaration's diff --git a/compiler/src/types.ml b/compiler/src/types.ml index a5776d5..f665ff2 100644 --- a/compiler/src/types.ml +++ b/compiler/src/types.ml @@ -621,6 +621,33 @@ let builtin_signatures : (string * int * builtin_arg_req list) list = ("set", 3, [ ReqMap; ReqAny; ReqAny ]); ("has", 2, [ ReqMap; ReqAny ]); ("int_to_text", 1, [ ReqInt ]); + (* systems stdlib (docs/plan/oop-vm/08-builtin-surface.md): the text and + container vocabulary the driving workload writes. `len` resolves on a + text OR either container, so its argument is unchecked here the same + way `get`'s key is. *) + ("len", 1, [ ReqAny ]); + ("byte_at", 2, [ ReqText; ReqInt ]); + ("print_err", 1, [ ReqText ]); + ("starts_with", 2, [ ReqText; ReqText ]); + ("ends_with", 2, [ ReqText; ReqText ]); + ("index_of", 2, [ ReqText; ReqText ]); + ("last_index_of", 2, [ ReqText; ReqText ]); + ("substr", 3, [ ReqText; ReqInt; ReqInt ]); + ("trim", 1, [ ReqText ]); + ("to_lower", 1, [ ReqText ]); + ("char_of", 1, [ ReqInt ]); + ("parse_int", 1, [ ReqText ]); + ("split", 2, [ ReqText; ReqText ]); + ("split_ws", 1, [ ReqText ]); + ("join", 2, [ ReqMulti; ReqText ]); + ("slice", 3, [ ReqMulti; ReqInt; ReqInt ]); + ("pop", 1, [ ReqMulti ]); + ("shift", 1, [ ReqMulti ]); + ("sort", 1, [ ReqMulti ]); + ("reverse", 1, [ ReqMulti ]); + ("remove", 2, [ ReqMap; ReqAny ]); + ("key_at", 2, [ ReqMap; ReqInt ]); + ("val_at", 2, [ ReqMap; ReqInt ]); ] let rec unwrap_nullable (t : typ) : typ = @@ -692,6 +719,21 @@ let builtin_confident_ret (name : string) (arg0 : typ option) : typ option = | "has" -> Some (TScalar "Bool") | "latest" -> ( match arg0 with Some (TMulti e) -> Some e | _ -> None) | "get" -> ( match arg0 with Some (TMulti e) -> Some e | Some (TMap (_, v)) -> Some v | _ -> None) + (* systems stdlib. Every fresh-Text and fresh-`multi` result is an owned + value, so these entries are what make a `let` holding one get its + drop (owner.ml classifies through this table too). *) + | "len" | "byte_at" | "index_of" | "last_index_of" -> Some (TScalar "Int") + | "print_err" | "sort" | "reverse" -> Some (TScalar "Int") + | "starts_with" | "ends_with" | "remove" -> Some (TScalar "Bool") + | "substr" | "trim" | "to_lower" | "char_of" | "join" -> Some (TScalar "Text") + (* `parse_int` is optional-shaped: an unparseable text is 0, which is how + a `?Int` spells nil (08-builtin-surface.md's `?T` section). *) + | "parse_int" -> Some (TNullable (TScalar "Int")) + | "split" | "split_ws" -> Some (TMulti (TScalar "Text")) + | "slice" -> ( match arg0 with Some (TMulti e) -> Some (TMulti e) | _ -> None) + | "pop" | "shift" -> ( match arg0 with Some (TMulti e) -> Some e | _ -> None) + | "key_at" -> ( match arg0 with Some (TMap (k, _)) -> Some k | _ -> None) + | "val_at" -> ( match arg0 with Some (TMap (_, v)) -> Some v | _ -> None) | _ -> None (* `use_edge`/`uses_of_program`/`path_str` -- relocated here (hotfix) @@ -871,6 +913,10 @@ let typecheck_program ~file ~(module_of : string -> string) | ListLit (first :: _) -> ( match confident_typ cenv first with Some t -> Some (TMulti t) | None -> None) | ListLit [] | MapLit -> None + (* haxe-parity Task 6: `nil` is contextual on its destination, the same + as an empty container literal — nothing about the literal itself + says which `?T` it is the absent value of. *) + | NilLit -> None (* haxe-parity Task 5: a `try` expression's type is its try arm's — the handler is checked to agree (typecheck_expr below), so either arm would answer, and the try arm is the one that always has a value. *) @@ -1200,6 +1246,11 @@ let typecheck_program ~file ~(module_of : string -> string) | t :: _ -> { typ = TMulti t; is_nil = false } | [] -> { typ = TScalar "Int"; is_nil = false }) | MapLit -> { typ = TScalar "Int"; is_nil = false } + (* `is_nil` is what marks the literal: the `typ` is the same + placeholder every contextual value here reports, and the flag is + what lets a comparison or a binding treat it as the absent value of + whatever `?T` it meets. *) + | NilLit -> { typ = TScalar "Int"; is_nil = true } | Try { body; ename; handler } -> let body_res = typecheck_expr env cenv body in (* The catch arm sees exactly one new name: the error record. *) @@ -1862,7 +1913,7 @@ and walk_expr (bound : StringSet.t) (visit : StringSet.t -> expr -> unit) (e : e | Ctor (_, fields) -> List.iter (fun (_, v) -> walk_expr bound visit v) fields | Interp inner -> walk_expr bound visit inner | ListLit items -> List.iter (walk_expr bound visit) items - | MapLit -> () + | MapLit | NilLit -> () | Try { body; ename; handler } -> walk_expr bound visit body; walk_block (StringSet.add ename bound) visit handler diff --git a/runtime/src/builtin.c b/runtime/src/builtin.c index fa9e8af..a184216 100644 --- a/runtime/src/builtin.c +++ b/runtime/src/builtin.c @@ -23,6 +23,43 @@ static void *native_check(uint64_t v, uint32_t cls, const char **msg) { return o; } +/* ---- systems-stdlib helpers ------------------------------------------ + * Text is bytes with an explicit length and no NUL, so every scan below is + * length-driven; "whitespace" is the same four bytes WO_B_WORDS already + * treats as separators. */ +static int ws_byte(char c) { return c == ' ' || c == '\t' || c == '\n' || c == '\r'; } + +/* First (dir > 0) or last (dir < 0) byte offset where [needle] occurs in + * [hay], or -1. An empty needle is found at 0 / at hay->len. */ +static int64_t str_find(const wo_str *hay, const wo_str *needle, int dir) { + if (needle->len > hay->len) return -1; + uint32_t span = hay->len - needle->len; + if (dir > 0) { + for (uint32_t i = 0; i <= span; i++) + if (!memcmp(hay->data + i, needle->data, needle->len)) return (int64_t)i; + } else { + for (uint32_t i = span + 1; i > 0; i--) + if (!memcmp(hay->data + (i - 1), needle->data, needle->len)) + return (int64_t)(i - 1); + } + return -1; +} + +/* Element compare for WO_B_SORT: Text elements by content (memcmp over the + * shared prefix, then length), everything else as signed integers. */ +static int elem_cmp(uint8_t kind, uint64_t a, uint64_t b) { + if (kind == WO_K_TEXT) { + const wo_str *x = (const wo_str *)(uintptr_t)a, *y = (const wo_str *)(uintptr_t)b; + if (!x || !y) return (x ? 1 : 0) - (y ? 1 : 0); + uint32_t n = x->len < y->len ? x->len : y->len; + int c = n ? memcmp(x->data, y->data, n) : 0; + if (c) return c; + return x->len == y->len ? 0 : (x->len < y->len ? -1 : 1); + } + int64_t ia = (int64_t)a, ib = (int64_t)b; + return ia == ib ? 0 : (ia < ib ? -1 : 1); +} + int wo_builtin(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { wo_rt *rt = &vm->rt; uint8_t A = wo_ins_a(ins), B = wo_ins_b(ins), C = wo_ins_c(ins); @@ -216,6 +253,335 @@ int wo_builtin(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { R[A] = R[B]; return 0; } + /* ---- systems stdlib: text ---------------------------------------- */ + case WO_B_LEN: { /* one name for "how many": bytes, elements, entries */ + if (!R[B]) { + *msg = "null receiver"; + return WO_T_BOUNDS; + } + wo_hdr *o = (wo_hdr *)(uintptr_t)R[B]; + if (o->class_id == WO_CLS_STR) R[A] = ((wo_str *)o)->len; + else if (o->class_id == WO_CLS_MULTI) R[A] = ((wo_multi *)o)->len; + else if (o->class_id == WO_CLS_MAP) R[A] = ((wo_map *)o)->len; + else { + *msg = "`len` needs a text, a multi, or a map"; + return WO_T_BOUNDS; + } + return 0; + } + case WO_B_BYTE_AT: { + wo_str *s = native_check(R[B], WO_CLS_STR, msg); + if (!s) return WO_T_BOUNDS; + uint64_t i = R[B + 1]; + if (i >= s->len) { + *msg = "byte index out of range"; + return WO_T_BOUNDS; + } + R[A] = (uint64_t)(uint8_t)s->data[i]; + return 0; + } + case WO_B_PRINT_ERR: { + wo_str *s = native_check(R[B], WO_CLS_STR, msg); + if (!s) return WO_T_BOUNDS; + fwrite(s->data, 1, s->len, stderr); + fputc('\n', stderr); + R[A] = 0; + return 0; + } + case WO_B_STARTS_WITH: + case WO_B_ENDS_WITH: { + wo_str *s = native_check(R[B], WO_CLS_STR, msg); + if (!s) return WO_T_BOUNDS; + wo_str *fix = native_check(R[B + 1], WO_CLS_STR, msg); + if (!fix) return WO_T_BOUNDS; + if (fix->len > s->len) R[A] = 0; + else { + const char *at = C == WO_B_STARTS_WITH ? s->data : s->data + (s->len - fix->len); + R[A] = memcmp(at, fix->data, fix->len) ? 0 : 1; + } + return 0; + } + case WO_B_INDEX_OF: + case WO_B_LAST_INDEX_OF: { + wo_str *s = native_check(R[B], WO_CLS_STR, msg); + if (!s) return WO_T_BOUNDS; + wo_str *n = native_check(R[B + 1], WO_CLS_STR, msg); + if (!n) return WO_T_BOUNDS; + R[A] = (uint64_t)str_find(s, n, C == WO_B_INDEX_OF ? 1 : -1); + return 0; + } + case WO_B_SUBSTR: { /* clamped, never trapping: a start past the end or a + * length past the end yields the empty/short text */ + wo_str *s = native_check(R[B], WO_CLS_STR, msg); + if (!s) return WO_T_BOUNDS; + int64_t start = (int64_t)R[B + 1], want = (int64_t)R[B + 2]; + if (start < 0) start = 0; + if (start > (int64_t)s->len) start = s->len; + if (want < 0) want = 0; + if (start + want > (int64_t)s->len) want = (int64_t)s->len - start; + wo_str *out = wo_str_new(rt, s->data + start, (uint32_t)want); + if (!out) { + *msg = "out of memory"; + return WO_T_OOM; + } + R[A] = (uint64_t)(uintptr_t)out; + return 0; + } + case WO_B_TRIM: { + wo_str *s = native_check(R[B], WO_CLS_STR, msg); + if (!s) return WO_T_BOUNDS; + uint32_t lo = 0, hi = s->len; + while (lo < hi && ws_byte(s->data[lo])) lo++; + while (hi > lo && ws_byte(s->data[hi - 1])) hi--; + wo_str *out = wo_str_new(rt, s->data + lo, hi - lo); + if (!out) { + *msg = "out of memory"; + return WO_T_OOM; + } + R[A] = (uint64_t)(uintptr_t)out; + return 0; + } + case WO_B_TO_LOWER: { + wo_str *s = native_check(R[B], WO_CLS_STR, msg); + if (!s) return WO_T_BOUNDS; + wo_str *out = wo_str_new(rt, s->data, s->len); + if (!out) { + *msg = "out of memory"; + return WO_T_OOM; + } + for (uint32_t i = 0; i < out->len; i++) + if (out->data[i] >= 'A' && out->data[i] <= 'Z') out->data[i] += 32; + R[A] = (uint64_t)(uintptr_t)out; + return 0; + } + case WO_B_CHAR_OF: { + char c = (char)(uint8_t)R[B]; + wo_str *out = wo_str_new(rt, &c, 1); + if (!out) { + *msg = "out of memory"; + return WO_T_OOM; + } + R[A] = (uint64_t)(uintptr_t)out; + return 0; + } + case WO_B_PARSE_INT: { /* optional-shaped: unparseable is 0, which is + * exactly how a `?Int` spells nil */ + wo_str *s = native_check(R[B], WO_CLS_STR, msg); + if (!s) return WO_T_BOUNDS; + uint32_t i = 0; + int neg = 0; + while (i < s->len && ws_byte(s->data[i])) i++; + if (i < s->len && (s->data[i] == '-' || s->data[i] == '+')) neg = s->data[i++] == '-'; + int64_t acc = 0; + int digits = 0; + while (i < s->len && s->data[i] >= '0' && s->data[i] <= '9') { + acc = acc * 10 + (s->data[i++] - '0'); + digits++; + } + R[A] = digits ? (uint64_t)(neg ? -acc : acc) : 0; + return 0; + } + case WO_B_SPLIT: + case WO_B_SPLIT_WS: { + wo_str *s = native_check(R[B], WO_CLS_STR, msg); + if (!s) return WO_T_BOUNDS; + wo_str *sep = NULL; + if (C == WO_B_SPLIT) { + sep = native_check(R[B + 1], WO_CLS_STR, msg); + if (!sep) return WO_T_BOUNDS; + if (sep->len == 0) { + *msg = "`split` needs a non-empty separator"; + return WO_T_BOUNDS; + } + } + wo_multi *out = wo_multi_new(rt, WO_K_TEXT); + if (!out) { + *msg = "out of memory"; + return WO_T_OOM; + } + uint32_t i = 0; + while (i <= s->len) { + uint32_t start = i, end; + if (C == WO_B_SPLIT) { + end = s->len; + for (uint32_t j = i; j + sep->len <= s->len; j++) + if (!memcmp(s->data + j, sep->data, sep->len)) { + end = j; + break; + } + i = end + sep->len; + } else { + while (start < s->len && ws_byte(s->data[start])) start++; + if (start >= s->len) break; + end = start; + while (end < s->len && !ws_byte(s->data[end])) end++; + i = end; + } + wo_str *part = wo_str_new(rt, s->data + start, end - start); + if (!part || wo_multi_push(out, (uint64_t)(uintptr_t)part) != 0) { + if (part) wo_str_free(rt, part); + wo_drop_obj(rt, &out->h); + *msg = "out of memory"; + return WO_T_OOM; + } + if (C == WO_B_SPLIT && end == s->len) break; + } + R[A] = (uint64_t)(uintptr_t)out; + return 0; + } + case WO_B_JOIN: { + wo_multi *m = native_check(R[B], WO_CLS_MULTI, msg); + if (!m) return WO_T_BOUNDS; + wo_str *sep = native_check(R[B + 1], WO_CLS_STR, msg); + if (!sep) return WO_T_BOUNDS; + if (m->elem_kind != WO_K_TEXT) { + *msg = "`join` needs a `multi Text`"; + return WO_T_BOUNDS; + } + uint32_t total = m->len ? (m->len - 1) * sep->len : 0; + for (uint32_t i = 0; i < m->len; i++) { + const wo_str *e = (const wo_str *)(uintptr_t)m->items[i]; + if (e) total += e->len; + } + wo_str *out = wo_str_alloc(rt, total); + if (!out) { + *msg = "out of memory"; + return WO_T_OOM; + } + uint32_t at = 0; + for (uint32_t i = 0; i < m->len; i++) { + if (i && sep->len) { + memcpy(out->data + at, sep->data, sep->len); + at += sep->len; + } + const wo_str *e = (const wo_str *)(uintptr_t)m->items[i]; + if (e && e->len) { + memcpy(out->data + at, e->data, e->len); + at += e->len; + } + } + R[A] = (uint64_t)(uintptr_t)out; + return 0; + } + /* ---- systems stdlib: containers ---------------------------------- */ + case WO_B_SLICE: { /* [from, to) — Text elements are COPIED so the slice + * and its source never both own one value */ + wo_multi *m = native_check(R[B], WO_CLS_MULTI, msg); + if (!m) return WO_T_BOUNDS; + if (m->elem_kind != WO_K_TEXT && m->elem_kind != WO_K_SCALAR) { + *msg = "`slice` needs a `multi Text` or a `multi` of scalars"; + return WO_T_BOUNDS; + } + int64_t from = (int64_t)R[B + 1], to = (int64_t)R[B + 2]; + if (from < 0) from = 0; + if (to > (int64_t)m->len) to = m->len; + wo_multi *out = wo_multi_new(rt, m->elem_kind); + if (!out) { + *msg = "out of memory"; + return WO_T_OOM; + } + for (int64_t i = from; i < to; i++) { + uint64_t v = m->items[i]; + if (m->elem_kind == WO_K_TEXT && v) { + const wo_str *e = (const wo_str *)(uintptr_t)v; + wo_str *cp = wo_str_new(rt, e->data, e->len); + if (!cp) { + wo_drop_obj(rt, &out->h); + *msg = "out of memory"; + return WO_T_OOM; + } + v = (uint64_t)(uintptr_t)cp; + } + if (wo_multi_push(out, v) != 0) { + wo_drop_obj(rt, &out->h); + *msg = "out of memory"; + return WO_T_OOM; + } + } + R[A] = (uint64_t)(uintptr_t)out; + return 0; + } + case WO_B_POP: + case WO_B_SHIFT: { /* the element LEAVES the container: its ownership goes + * to the caller's register, so nothing is dropped here */ + wo_multi *m = native_check(R[B], WO_CLS_MULTI, msg); + if (!m) return WO_T_BOUNDS; + if (m->len == 0) { + *msg = C == WO_B_POP ? "`pop` on an empty multi" : "`shift` on an empty multi"; + return WO_T_BOUNDS; + } + if (C == WO_B_POP) R[A] = m->items[--m->len]; + else { + R[A] = m->items[0]; + memmove(m->items, m->items + 1, (size_t)(m->len - 1) * sizeof(uint64_t)); + m->len--; + } + return 0; + } + case WO_B_SORT: { /* insertion sort: in place, stable, and the workload's + * lists are short (a directory's file names) */ + wo_multi *m = native_check(R[B], WO_CLS_MULTI, msg); + if (!m) return WO_T_BOUNDS; + for (uint32_t i = 1; i < m->len; i++) { + uint64_t v = m->items[i]; + uint32_t j = i; + while (j > 0 && elem_cmp(m->elem_kind, m->items[j - 1], v) > 0) { + m->items[j] = m->items[j - 1]; + j--; + } + m->items[j] = v; + } + R[A] = 0; + return 0; + } + case WO_B_REVERSE: { + wo_multi *m = native_check(R[B], WO_CLS_MULTI, msg); + if (!m) return WO_T_BOUNDS; + for (uint32_t i = 0, j = m->len; i + 1 < j; i++, j--) { + uint64_t t = m->items[i]; + m->items[i] = m->items[j - 1]; + m->items[j - 1] = t; + } + R[A] = 0; + return 0; + } + case WO_B_MAP_REMOVE: { /* the map owned the key and the value, so both die + * here (their kinds are the map's own drop plan) */ + wo_map *m = native_check(R[B], WO_CLS_MAP, msg); + if (!m) return WO_T_BOUNDS; + uint64_t k = R[B + 1]; + for (uint32_t i = 0; i < m->len; i++) { + int hit = m->key_kind == WO_K_TEXT + ? (m->keys[i] && k && + wo_str_eq((const wo_str *)(uintptr_t)m->keys[i], + (const wo_str *)(uintptr_t)k)) + : m->keys[i] == k; + if (!hit) continue; + wo_drop_kind(rt, m->key_kind, m->keys[i]); + wo_drop_kind(rt, m->val_kind, m->vals[i]); + memmove(m->keys + i, m->keys + i + 1, (size_t)(m->len - i - 1) * sizeof(uint64_t)); + memmove(m->vals + i, m->vals + i + 1, (size_t)(m->len - i - 1) * sizeof(uint64_t)); + m->len--; + R[A] = 1; + return 0; + } + R[A] = 0; + return 0; + } + case WO_B_MAP_KEY_AT: + case WO_B_MAP_VAL_AT: { /* slot-indexed enumeration — what `for k, v in m` + * lowers onto (the parallel arrays are insertion + * ordered, cont.h) */ + wo_map *m = native_check(R[B], WO_CLS_MAP, msg); + if (!m) return WO_T_BOUNDS; + uint64_t i = R[B + 1]; + if (i >= m->len) { + *msg = "map slot out of range"; + return WO_T_BOUNDS; + } + R[A] = C == WO_B_MAP_KEY_AT ? m->keys[i] : m->vals[i]; + return 0; + } default: /* unreachable: loader validated the id */ *msg = "unknown builtin"; return WO_T_EXPLICIT; diff --git a/runtime/src/loader.c b/runtime/src/loader.c index 2b3033c..c705b5f 100644 --- a/runtime/src/loader.c +++ b/runtime/src/loader.c @@ -42,6 +42,15 @@ static const uint8_t b_arity[WO_B_MAX + 1] = { [WO_B_MAP_NEW] = 0, [WO_B_MAP_SET] = 3, [WO_B_MAP_GET] = 2, [WO_B_MAP_HAS] = 2, [WO_B_INT_TO_TEXT] = 1, [WO_B_VARIANT_TAG] = 1, [WO_B_ERR_FILL] = 1, + /* systems stdlib */ + [WO_B_LEN] = 1, [WO_B_BYTE_AT] = 2, [WO_B_PRINT_ERR] = 1, + [WO_B_STARTS_WITH] = 2, [WO_B_ENDS_WITH] = 2, [WO_B_INDEX_OF] = 2, + [WO_B_LAST_INDEX_OF] = 2, [WO_B_SUBSTR] = 3, [WO_B_TRIM] = 1, + [WO_B_TO_LOWER] = 1, [WO_B_CHAR_OF] = 1, [WO_B_PARSE_INT] = 1, + [WO_B_SPLIT] = 2, [WO_B_SPLIT_WS] = 1, [WO_B_JOIN] = 2, + [WO_B_SLICE] = 3, [WO_B_POP] = 1, [WO_B_SHIFT] = 1, + [WO_B_SORT] = 1, [WO_B_REVERSE] = 1, [WO_B_MAP_REMOVE] = 2, + [WO_B_MAP_KEY_AT] = 2, [WO_B_MAP_VAL_AT] = 2, }; static int vtab_cmp(const void *a, const void *b) { diff --git a/runtime/src/obj.c b/runtime/src/obj.c index 9a10dd5..686f9f9 100644 --- a/runtime/src/obj.c +++ b/runtime/src/obj.c @@ -76,7 +76,7 @@ wo_hdr *wo_obj_new(wo_rt *rt, uint32_t class_id) { return o; } -static wo_str *str_alloc(wo_rt *rt, uint32_t len) { +wo_str *wo_str_alloc(wo_rt *rt, uint32_t len) { wo_str *s = wo_arena_alloc(&rt->arena, sizeof(wo_str) + len); if (!s) return NULL; memset(&s->h, 0, sizeof(s->h)); @@ -86,14 +86,14 @@ static wo_str *str_alloc(wo_rt *rt, uint32_t len) { } wo_str *wo_str_new(wo_rt *rt, const char *bytes, uint32_t len) { - wo_str *s = str_alloc(rt, len); + wo_str *s = wo_str_alloc(rt, len); if (!s) return NULL; memcpy(s->data, bytes, len); return s; } wo_str *wo_str_concat(wo_rt *rt, const wo_str *a, const wo_str *b) { - wo_str *s = str_alloc(rt, a->len + b->len); + wo_str *s = wo_str_alloc(rt, a->len + b->len); if (!s) return NULL; memcpy(s->data, a->data, a->len); memcpy(s->data + a->len, b->data, b->len); diff --git a/runtime/src/obj.h b/runtime/src/obj.h index 45807e6..89fe6c6 100644 --- a/runtime/src/obj.h +++ b/runtime/src/obj.h @@ -56,6 +56,10 @@ typedef struct wo_str { } wo_str; wo_str *wo_str_new(wo_rt *rt, const char *bytes, uint32_t len); /* NULL=OOM */ +/* A Text of [len] UNINITIALIZED bytes for a caller that writes them itself + * (the systems-stdlib `join`, which knows the total length up front and + * would otherwise need one allocation per element). NULL = OOM. */ +wo_str *wo_str_alloc(wo_rt *rt, uint32_t len); wo_str *wo_str_concat(wo_rt *rt, const wo_str *a, const wo_str *b); int wo_str_eq(const wo_str *a, const wo_str *b); /* content equality */ void wo_str_free(wo_rt *rt, wo_str *s); /* no-op on WO_F_CONST */ diff --git a/runtime/src/wob.h b/runtime/src/wob.h index db52caf..e804148 100644 --- a/runtime/src/wob.h +++ b/runtime/src/wob.h @@ -176,8 +176,37 @@ enum { * a 4-field class object, WO_T_OOM if either Text cannot be * allocated. */ WO_B_ERR_FILL = 15, + /* ---- systems stdlib: text and container surface (the driving + * workload's own vocabulary — docs/plan/oop-vm/08-builtin-surface.md + * lists the source spelling of each). Every one that returns a fresh + * Text or a fresh `multi` allocates it here, so a `?T` result spells + * absence as 0 like every other nullable. ---- */ + WO_B_LEN = 16, /* (text|multi|map) -> i64 length */ + WO_B_BYTE_AT = 17, /* (text, i) -> i64 byte; out of range traps BOUNDS */ + WO_B_PRINT_ERR = 18, /* (text) -> stderr, newline-terminated */ + WO_B_STARTS_WITH = 19, /* (text, prefix) -> 1/0 */ + WO_B_ENDS_WITH = 20, /* (text, suffix) -> 1/0 */ + WO_B_INDEX_OF = 21, /* (text, needle) -> first byte offset, -1 = absent */ + WO_B_LAST_INDEX_OF = 22, /* (text, needle) -> last byte offset, -1 = absent */ + WO_B_SUBSTR = 23, /* (text, start, len) -> fresh Text, clamped */ + WO_B_TRIM = 24, /* (text) -> fresh Text without leading/trailing space */ + WO_B_TO_LOWER = 25, /* (text) -> fresh Text, ASCII-lowercased */ + WO_B_CHAR_OF = 26, /* (i64) -> fresh one-byte Text */ + WO_B_PARSE_INT = 27, /* (text) -> i64; unparseable is 0, `?Int`'s own nil */ + WO_B_SPLIT = 28, /* (text, sep) -> fresh multi Text */ + WO_B_SPLIT_WS = 29, /* (text) -> fresh multi Text, whitespace-separated */ + WO_B_JOIN = 30, /* (multi Text, sep) -> fresh Text */ + WO_B_SLICE = 31, /* (multi, from, to) -> fresh multi; Text elements are + * COPIED, so the two containers never share a value */ + WO_B_POP = 32, /* (multi) -> last element, removed; empty traps BOUNDS */ + WO_B_SHIFT = 33, /* (multi) -> first element, removed; empty traps BOUNDS */ + WO_B_SORT = 34, /* (multi) -> 0; in place, Text by content else by value */ + WO_B_REVERSE = 35, /* (multi) -> 0; in place */ + WO_B_MAP_REMOVE = 36, /* (map, key) -> 1/0; drops the removed key and value */ + WO_B_MAP_KEY_AT = 37, /* (map, i) -> key at slot i (insertion order) */ + WO_B_MAP_VAL_AT = 38, /* (map, i) -> value at slot i */ }; -#define WO_B_MAX 15u +#define WO_B_MAX 38u /* ---- instruction encode/decode: op:8 A:8 then B:8 C:8 or Bx:16 ---- */ static inline uint32_t wo_ins_abc(uint8_t op, uint8_t a, uint8_t b, uint8_t c) {