From 1ef4e463eca31acfee1e80d4bb6ab72e46527225 Mon Sep 17 00:00:00 2001 From: "shoney.arickathil" Date: Tue, 8 Sep 2026 13:16:17 +0200 Subject: [PATCH] feat(crypto): AES-GCM via AES-NI + PCLMULQDQ (rv2 8 phase B, ids 113/114) - aes_gcm_seal/open, AES-128 and AES-256 (variant by key length 16/32), nonce 12 bytes, out = ciphertext||tag; open returns nil on auth failure - hardware path only (phase B): AES-NI key schedule (128/256) + block, GHASH via PCLMULQDQ with the fast GF(2^128) reduction, GCM mode (J0, CTR from counter 2, GHASH over aad|pad|ct|pad|len, tag = GHASH ^ AES(J0)) - constant-time by hardware; target-attributed functions + __builtin_cpu_supports gate so the binary stays portable -- no AES-NI traps with a clear message (bitsliced software + ARMv8 paths are phase C) - wiring: wob.h ids + WO_B_MAX 114; builtin.c crypto range; loader arity 4; emit.ml (ids/arity/return/name); types.ml (register + return type) - VERIFIED: matches NIST SP 800-38D cases 4 (AES-128) and 16 (AES-256) and the python cryptography reference byte-for-byte; KAT-gated in test_crypto (36/0); ASan/UBSan clean; runtime battery + compiler 557/0 green (cherry picked from commit f12a745a3c1313847f9d7f65e53bcd8093758af9) --- compiler/src/emit.ml | 12 +- compiler/src/types.ml | 4 + runtime/src/builtin.c | 2 +- runtime/src/crypto.c | 314 +++++++++++++++++++++++++++++++++++++ runtime/src/crypto.h | 11 ++ runtime/src/loader.c | 1 + runtime/src/wob.h | 5 +- runtime/test/test_crypto.c | 55 +++++++ 8 files changed, 400 insertions(+), 4 deletions(-) diff --git a/compiler/src/emit.ml b/compiler/src/emit.ml index ef480a2..914b515 100644 --- a/compiler/src/emit.ml +++ b/compiler/src/emit.ml @@ -297,6 +297,8 @@ let b_hmac_sha256 = 87 (* runtime-v2 8 phase A: ChaCha20-Poly1305 AEAD (ids match wob.h 111/112) *) let b_chacha20poly1305_seal = 111 let b_chacha20poly1305_open = 112 +let b_aes_gcm_seal = 113 +let b_aes_gcm_open = 114 let b_call = 88 let b_monitor = 89 let b_split = 28 @@ -1104,6 +1106,8 @@ let builtin_ret (name : string) (argty : Ast.field_ty option) : Ast.field_ty opt | "sha1" | "sha256" | "hmac_sha256" -> Some (Scalar "Bytes") | "chacha20poly1305_seal" -> Some (Scalar "Bytes") | "chacha20poly1305_open" -> Some (Nullable (Scalar "Bytes")) + | "aes_gcm_seal" -> Some (Scalar "Bytes") + | "aes_gcm_open" -> Some (Nullable (Scalar "Bytes")) | "base64_decode" -> Some (Nullable (Scalar "Bytes")) | _ -> None @@ -1124,7 +1128,8 @@ let is_builtin_name (n : string) = (* iteration 34: digests *) "sha1"; "sha256"; "hmac_sha256"; (* runtime-v2 8 phase A: AEAD *) - "chacha20poly1305_seal"; "chacha20poly1305_open" ] + "chacha20poly1305_seal"; "chacha20poly1305_open"; + "aes_gcm_seal"; "aes_gcm_open" ] (* ---- unions and variants (haxe-parity Task 4) ------------------------ @@ -3703,7 +3708,8 @@ and emit_builtin (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e : (* iteration 24, two arguments *) || id = b_call then 2 - else if id = b_chacha20poly1305_seal || id = b_chacha20poly1305_open then 4 + else if id = b_chacha20poly1305_seal || id = b_chacha20poly1305_open + || id = b_aes_gcm_seal || id = b_aes_gcm_open then 4 (* rv2 8: (key, nonce, aad, plaintext|ciphertext) *) else 3 (* b_bytes_slice lands here with substr's shape: (value, start, len) *) in @@ -3826,6 +3832,8 @@ and emit_builtin (p : pctx) (f : fstate) (v : views) ~(dst : int) ?expected (e : | "hmac_sha256" -> fixed b_hmac_sha256 | "chacha20poly1305_seal" -> fixed b_chacha20poly1305_seal | "chacha20poly1305_open" -> fixed b_chacha20poly1305_open + | "aes_gcm_seal" -> fixed b_aes_gcm_seal + | "aes_gcm_open" -> fixed b_aes_gcm_open | "multi_new" | "map_new" -> let is_map = name = "map_new" in if args <> [] then bad (Printf.sprintf "builtin `%s` takes no arguments" name) diff --git a/compiler/src/types.ml b/compiler/src/types.ml index d9eac44..52a640b 100644 --- a/compiler/src/types.ml +++ b/compiler/src/types.ml @@ -957,6 +957,8 @@ let builtin_signatures : (string * int * builtin_arg_req list) list = plaintext|ciphertext). seal -> Bytes; open -> ?Bytes (nil on auth fail). *) ("chacha20poly1305_seal", 4, [ ReqBytes; ReqBytes; ReqBytes; ReqBytes ]); ("chacha20poly1305_open", 4, [ ReqBytes; ReqBytes; ReqBytes; ReqBytes ]); + ("aes_gcm_seal", 4, [ ReqBytes; ReqBytes; ReqBytes; ReqBytes ]); + ("aes_gcm_open", 4, [ ReqBytes; ReqBytes; ReqBytes; ReqBytes ]); ] let rec unwrap_nullable (t : typ) : typ = @@ -1165,6 +1167,8 @@ let builtin_confident_ret (name : string) (arg0 : typ option) : typ option = | "sha1" | "sha256" | "hmac_sha256" -> Some (TScalar "Bytes") | "chacha20poly1305_seal" -> Some (TScalar "Bytes") | "chacha20poly1305_open" -> Some (TNullable (TScalar "Bytes")) + | "aes_gcm_seal" -> Some (TScalar "Bytes") + | "aes_gcm_open" -> Some (TNullable (TScalar "Bytes")) (* malformed base64 is nil, not a trap: it arrives from the network *) | "base64_decode" -> Some (TNullable (TScalar "Bytes")) | _ -> None diff --git a/runtime/src/builtin.c b/runtime/src/builtin.c index 35f0f8a..bb1b06e 100644 --- a/runtime/src/builtin.c +++ b/runtime/src/builtin.c @@ -175,7 +175,7 @@ int wo_builtin(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { || (C >= WO_B_NET_READ_DL && C <= WO_B_NET_CONNECT)) return wo_builtin_sys(vm, R, ins, msg); if ((C >= WO_B_SHA1 && C <= WO_B_HMAC_SHA256) - || (C >= WO_B_CHACHA20POLY1305_SEAL && C <= WO_B_CHACHA20POLY1305_OPEN)) + || (C >= WO_B_CHACHA20POLY1305_SEAL && C <= WO_B_AES_GCM_OPEN)) return wo_builtin_crypto(vm, R, ins, msg); if (C >= WO_B_DB_INSERT && C <= WO_B_DB_PROBE) { /* arc stage 3: the database is an actor on shard 0. A worker shard diff --git a/runtime/src/crypto.c b/runtime/src/crypto.c index 505c078..2484f8d 100644 --- a/runtime/src/crypto.c +++ b/runtime/src/crypto.c @@ -393,6 +393,262 @@ int wo_chacha20poly1305_open(const uint8_t key[32], const uint8_t nonce[12], return 0; } +/* ---- AES-GCM via AES-NI + PCLMULQDQ (rv2 8 phase B, x86-64 hardware path) -- + * Constant-time by hardware (no tables, no data-dependent branches). The + * functions carry target attributes so the binary still runs on CPUs without + * the extensions; aes_gcm_available() gates entry (the bitsliced software + * fallback is phase C). Refs: Intel AES-NI + carry-less-multiplication + * whitepapers, NIST SP 800-38D. Vectors: NIST/RFC in test/test_crypto.c. */ +#if defined(__x86_64__) +#include +#include +#include + +int wo_aes_gcm_available(void) { + return __builtin_cpu_supports("aes") && __builtin_cpu_supports("pclmul") && + __builtin_cpu_supports("ssse3"); +} + +#define AES128_ASSIST(t1, t2) \ + do { \ + __m128i _t3; \ + t2 = _mm_shuffle_epi32(t2, 0xff); \ + _t3 = _mm_slli_si128(t1, 4); t1 = _mm_xor_si128(t1, _t3); \ + _t3 = _mm_slli_si128(_t3, 4); t1 = _mm_xor_si128(t1, _t3); \ + _t3 = _mm_slli_si128(_t3, 4); t1 = _mm_xor_si128(t1, _t3); \ + t1 = _mm_xor_si128(t1, t2); \ + } while (0) + +__attribute__((target("aes,sse2"))) +static void aes128_expand(const uint8_t *key, __m128i rk[11]) { + __m128i t1 = _mm_loadu_si128((const __m128i *)key), t2; + rk[0] = t1; + t2 = _mm_aeskeygenassist_si128(t1, 0x01); AES128_ASSIST(t1, t2); rk[1] = t1; + t2 = _mm_aeskeygenassist_si128(t1, 0x02); AES128_ASSIST(t1, t2); rk[2] = t1; + t2 = _mm_aeskeygenassist_si128(t1, 0x04); AES128_ASSIST(t1, t2); rk[3] = t1; + t2 = _mm_aeskeygenassist_si128(t1, 0x08); AES128_ASSIST(t1, t2); rk[4] = t1; + t2 = _mm_aeskeygenassist_si128(t1, 0x10); AES128_ASSIST(t1, t2); rk[5] = t1; + t2 = _mm_aeskeygenassist_si128(t1, 0x20); AES128_ASSIST(t1, t2); rk[6] = t1; + t2 = _mm_aeskeygenassist_si128(t1, 0x40); AES128_ASSIST(t1, t2); rk[7] = t1; + t2 = _mm_aeskeygenassist_si128(t1, 0x80); AES128_ASSIST(t1, t2); rk[8] = t1; + t2 = _mm_aeskeygenassist_si128(t1, 0x1b); AES128_ASSIST(t1, t2); rk[9] = t1; + t2 = _mm_aeskeygenassist_si128(t1, 0x36); AES128_ASSIST(t1, t2); rk[10] = t1; +} + +__attribute__((target("aes,sse2"))) +static void aes256_assist1(__m128i *t1, __m128i *t2) { + __m128i t4; + *t2 = _mm_shuffle_epi32(*t2, 0xff); + t4 = _mm_slli_si128(*t1, 4); *t1 = _mm_xor_si128(*t1, t4); + t4 = _mm_slli_si128(t4, 4); *t1 = _mm_xor_si128(*t1, t4); + t4 = _mm_slli_si128(t4, 4); *t1 = _mm_xor_si128(*t1, t4); + *t1 = _mm_xor_si128(*t1, *t2); +} +__attribute__((target("aes,sse2"))) +static void aes256_assist2(__m128i *t1, __m128i *t3) { + __m128i t2, t4; + t4 = _mm_aeskeygenassist_si128(*t1, 0x00); + t2 = _mm_shuffle_epi32(t4, 0xaa); + t4 = _mm_slli_si128(*t3, 4); *t3 = _mm_xor_si128(*t3, t4); + t4 = _mm_slli_si128(t4, 4); *t3 = _mm_xor_si128(*t3, t4); + t4 = _mm_slli_si128(t4, 4); *t3 = _mm_xor_si128(*t3, t4); + *t3 = _mm_xor_si128(*t3, t2); +} +/* rcon must be a compile-time immediate to aeskeygenassist, so the schedule + * is unrolled rather than looped over an rcon array. */ +#define AES256_STEP(RC) \ + do { \ + t2 = _mm_aeskeygenassist_si128(t3, (RC)); \ + aes256_assist1(&t1, &t2); rk[k++] = t1; \ + aes256_assist2(&t1, &t3); rk[k++] = t3; \ + } while (0) + +__attribute__((target("aes,sse2"))) +static void aes256_expand(const uint8_t *key, __m128i rk[15]) { + __m128i t1 = _mm_loadu_si128((const __m128i *)key); + __m128i t3 = _mm_loadu_si128((const __m128i *)(key + 16)); + __m128i t2; + int k = 2; + rk[0] = t1; rk[1] = t3; + AES256_STEP(0x01); AES256_STEP(0x02); AES256_STEP(0x04); + AES256_STEP(0x08); AES256_STEP(0x10); AES256_STEP(0x20); + t2 = _mm_aeskeygenassist_si128(t3, 0x40); + aes256_assist1(&t1, &t2); rk[k] = t1; /* rk[14] */ +} + +__attribute__((target("aes"))) +static __m128i aes_enc(const __m128i *rk, int nr, __m128i m) { + m = _mm_xor_si128(m, rk[0]); + for (int i = 1; i < nr; i++) m = _mm_aesenc_si128(m, rk[i]); + return _mm_aesenclast_si128(m, rk[nr]); +} + +/* Carry-less multiply in GF(2^128) with the GCM reduction, operands in the + * byte-reversed domain (Intel CLMUL whitepaper gfmul + fast reduction). */ +__attribute__((target("pclmul,sse2"))) +static __m128i gfmul(__m128i a, __m128i b) { + __m128i t3, t4, t5, t6, t7, t8, t9, t2; + t3 = _mm_clmulepi64_si128(a, b, 0x00); + t4 = _mm_clmulepi64_si128(a, b, 0x10); + t5 = _mm_clmulepi64_si128(a, b, 0x01); + t6 = _mm_clmulepi64_si128(a, b, 0x11); + t4 = _mm_xor_si128(t4, t5); + t5 = _mm_slli_si128(t4, 8); + t4 = _mm_srli_si128(t4, 8); + t3 = _mm_xor_si128(t3, t5); + t6 = _mm_xor_si128(t6, t4); + t7 = _mm_srli_epi32(t3, 31); + t8 = _mm_srli_epi32(t6, 31); + t3 = _mm_slli_epi32(t3, 1); + t6 = _mm_slli_epi32(t6, 1); + t9 = _mm_srli_si128(t7, 12); + t8 = _mm_slli_si128(t8, 4); + t7 = _mm_slli_si128(t7, 4); + t3 = _mm_or_si128(t3, t7); + t6 = _mm_or_si128(t6, t8); + t6 = _mm_or_si128(t6, t9); + t7 = _mm_slli_epi32(t3, 31); + t8 = _mm_slli_epi32(t3, 30); + t9 = _mm_slli_epi32(t3, 25); + t7 = _mm_xor_si128(t7, t8); + t7 = _mm_xor_si128(t7, t9); + t8 = _mm_srli_si128(t7, 4); + t7 = _mm_slli_si128(t7, 12); + t3 = _mm_xor_si128(t3, t7); + t2 = _mm_srli_epi32(t3, 1); + t4 = _mm_srli_epi32(t3, 2); + t5 = _mm_srli_epi32(t3, 7); + t2 = _mm_xor_si128(t2, t4); + t2 = _mm_xor_si128(t2, t5); + t2 = _mm_xor_si128(t2, t8); + t3 = _mm_xor_si128(t3, t2); + t6 = _mm_xor_si128(t6, t3); + return t6; +} + +/* GHASH `T = (T ^ block)·H` over full+partial 16-byte blocks (bswap domain). */ +__attribute__((target("pclmul,ssse3"))) +static __m128i ghash(__m128i T, __m128i H, const uint8_t *data, size_t len, + __m128i bswap) { + size_t off = 0; + while (len - off >= 16) { + __m128i b = _mm_loadu_si128((const __m128i *)(data + off)); + b = _mm_shuffle_epi8(b, bswap); + T = gfmul(_mm_xor_si128(T, b), H); + off += 16; + } + if (off < len) { + uint8_t last[16]; + memset(last, 0, 16); + memcpy(last, data + off, len - off); + __m128i b = _mm_loadu_si128((const __m128i *)last); + b = _mm_shuffle_epi8(b, bswap); + T = gfmul(_mm_xor_si128(T, b), H); + } + return T; +} + +/* GCM core (encrypt==1 seals, 0 opens). On open, `tag_in` is compared + * constant-time; returns 0 ok, 1 auth failure. On seal, writes tag_out. */ +__attribute__((target("aes,pclmul,ssse3"))) +static int aes_gcm_core(const uint8_t *key, size_t keylen, const uint8_t nonce[12], + const uint8_t *aad, size_t aadlen, const uint8_t *in, + size_t inlen, uint8_t *out, uint8_t tag_out[16], + const uint8_t *tag_in) { + __m128i rk[15]; + int nr; + if (keylen == 16) { aes128_expand(key, rk); nr = 10; } + else { aes256_expand(key, rk); nr = 14; } + const __m128i bswap = + _mm_set_epi8(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15); + + __m128i H = aes_enc(rk, nr, _mm_setzero_si128()); + H = _mm_shuffle_epi8(H, bswap); /* reflect H once */ + + uint8_t j0[16]; + memcpy(j0, nonce, 12); + j0[12] = 0; j0[13] = 0; j0[14] = 0; j0[15] = 1; + __m128i ej0 = aes_enc(rk, nr, _mm_loadu_si128((const __m128i *)j0)); + + /* GHASH over aad || pad || ciphertext || pad || len-block. + * On seal the ciphertext is what we produce; on open it is the input. */ + __m128i T = _mm_setzero_si128(); + T = ghash(T, H, aad, aadlen, bswap); + if (!tag_in) { + /* seal: CTR-encrypt from counter 2, then GHASH the produced ct */ + } + + /* CTR: counter starts at 2 (inc32(J0)); build blocks from nonce||BE32 */ + uint32_t ctr = 2; + size_t off = 0; + /* For open, GHASH the input ciphertext first (before we overwrite via out) */ + if (tag_in) T = ghash(T, H, in, inlen, bswap); + while (off < inlen) { + uint8_t cb[16]; + memcpy(cb, nonce, 12); + cb[12] = (uint8_t)(ctr >> 24); cb[13] = (uint8_t)(ctr >> 16); + cb[14] = (uint8_t)(ctr >> 8); cb[15] = (uint8_t)ctr; + __m128i ks = aes_enc(rk, nr, _mm_loadu_si128((const __m128i *)cb)); + uint8_t ksb[16]; + _mm_storeu_si128((__m128i *)ksb, ks); + size_t n = inlen - off < 16 ? inlen - off : 16; + for (size_t i = 0; i < n; i++) out[off + i] = in[off + i] ^ ksb[i]; + off += n; ctr++; + } + if (!tag_in) T = ghash(T, H, out, inlen, bswap); /* seal: GHASH the ct */ + + uint8_t lb[16]; + uint64_t aBits = (uint64_t)aadlen * 8, cBits = (uint64_t)inlen * 8; + for (int i = 0; i < 8; i++) lb[i] = (uint8_t)(aBits >> (56 - 8 * i)); + for (int i = 0; i < 8; i++) lb[8 + i] = (uint8_t)(cBits >> (56 - 8 * i)); + T = ghash(T, H, lb, 16, bswap); + + T = _mm_shuffle_epi8(T, bswap); /* back to big-endian bytes */ + __m128i tagv = _mm_xor_si128(T, ej0); + uint8_t tag[16]; + _mm_storeu_si128((__m128i *)tag, tagv); + + if (tag_in) { + uint8_t d = 0; + for (int i = 0; i < 16; i++) d |= (uint8_t)(tag[i] ^ tag_in[i]); + return d == 0 ? 0 : 1; + } + memcpy(tag_out, tag, 16); + return 0; +} +#else +int wo_aes_gcm_available(void) { return 0; } +#endif + +/* Public seal/open. keylen 16 (AES-128) or 32 (AES-256), nonce 12 bytes. + * Returns 0 ok, 1 auth failure (open), -2 when no hardware AES is available. */ +int wo_aes_gcm_seal(const uint8_t *key, size_t keylen, const uint8_t nonce[12], + const uint8_t *aad, size_t aadlen, const uint8_t *pt, + size_t ptlen, uint8_t *out) { +#if defined(__x86_64__) + if (!wo_aes_gcm_available()) return -2; + return aes_gcm_core(key, keylen, nonce, aad, aadlen, pt, ptlen, out, + out + ptlen, NULL); +#else + (void)key; (void)keylen; (void)nonce; (void)aad; (void)aadlen; + (void)pt; (void)ptlen; (void)out; + return -2; +#endif +} +int wo_aes_gcm_open(const uint8_t *key, size_t keylen, const uint8_t nonce[12], + const uint8_t *aad, size_t aadlen, const uint8_t *ct, + size_t ctlen, const uint8_t tag[16], uint8_t *out) { +#if defined(__x86_64__) + if (!wo_aes_gcm_available()) return -2; + return aes_gcm_core(key, keylen, nonce, aad, aadlen, ct, ctlen, out, NULL, + tag); +#else + (void)key; (void)keylen; (void)nonce; (void)aad; (void)aadlen; (void)ct; + (void)ctlen; (void)tag; (void)out; + return -2; +#endif +} + /* The VM half: Bytes in, fresh Bytes out. Wrong class id traps * WO_T_BOUNDS with the Bytes builtins' message shape. */ static const wo_str *arg_bytes(uint64_t r, const char **msg) { @@ -486,6 +742,64 @@ int wo_builtin_crypto(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg) { R[A] = (uint64_t)(uintptr_t)o; return 0; } + case WO_B_AES_GCM_SEAL: { + const wo_str *k = arg_bytes(R[B], msg); + const wo_str *n = k ? arg_bytes(R[B + 1], msg) : NULL; + const wo_str *a = n ? arg_bytes(R[B + 2], msg) : NULL; + const wo_str *p = a ? arg_bytes(R[B + 3], msg) : NULL; + if (!p) return WO_T_BOUNDS; + if ((k->len != 16 && k->len != 32) || n->len != 12) { + *msg = "aes_gcm: key must be 16 or 32 bytes, nonce 12"; + return WO_T_BOUNDS; + } + uint8_t *buf = (uint8_t *)malloc(p->len + 16u); + if (!buf) { *msg = "out of memory"; return WO_T_OOM; } + int rc = wo_aes_gcm_seal((const uint8_t *)k->data, k->len, + (const uint8_t *)n->data, + (const uint8_t *)a->data, a->len, + (const uint8_t *)p->data, p->len, buf); + if (rc == -2) { + free(buf); + *msg = "aes_gcm requires hardware AES (AES-NI); software fallback is rv2 8 phase C"; + return WO_T_BOUNDS; + } + wo_str *o = wo_bytes_new(rt, (const char *)buf, (uint32_t)(p->len + 16u)); + free(buf); + if (!o) { *msg = "out of memory"; return WO_T_OOM; } + R[A] = (uint64_t)(uintptr_t)o; + return 0; + } + case WO_B_AES_GCM_OPEN: { + const wo_str *k = arg_bytes(R[B], msg); + const wo_str *n = k ? arg_bytes(R[B + 1], msg) : NULL; + const wo_str *a = n ? arg_bytes(R[B + 2], msg) : NULL; + const wo_str *ctag = a ? arg_bytes(R[B + 3], msg) : NULL; + if (!ctag) return WO_T_BOUNDS; + if ((k->len != 16 && k->len != 32) || n->len != 12) { + *msg = "aes_gcm: key must be 16 or 32 bytes, nonce 12"; + return WO_T_BOUNDS; + } + if (ctag->len < 16) { R[A] = 0; return 0; } + uint32_t bodylen = ctag->len - 16u; + uint8_t *buf = (uint8_t *)malloc(bodylen ? bodylen : 1u); + if (!buf) { *msg = "out of memory"; return WO_T_OOM; } + int rc = wo_aes_gcm_open((const uint8_t *)k->data, k->len, + (const uint8_t *)n->data, + (const uint8_t *)a->data, a->len, + (const uint8_t *)ctag->data, bodylen, + (const uint8_t *)ctag->data + bodylen, buf); + if (rc == -2) { + free(buf); + *msg = "aes_gcm requires hardware AES (AES-NI); software fallback is rv2 8 phase C"; + return WO_T_BOUNDS; + } + if (rc != 0) { free(buf); R[A] = 0; return 0; } /* auth failure -> nil */ + wo_str *o = wo_bytes_new(rt, (const char *)buf, bodylen); + free(buf); + if (!o) { *msg = "out of memory"; return WO_T_OOM; } + R[A] = (uint64_t)(uintptr_t)o; + return 0; + } default: *msg = "unknown crypto builtin"; return WO_T_BOUNDS; diff --git a/runtime/src/crypto.h b/runtime/src/crypto.h index 9e9c08e..29ca11c 100644 --- a/runtime/src/crypto.h +++ b/runtime/src/crypto.h @@ -26,6 +26,17 @@ int wo_chacha20poly1305_open(const uint8_t key[32], const uint8_t nonce[12], const uint8_t *ct, size_t ctlen, const uint8_t tag[16], uint8_t *out); +/* AES-GCM (rv2 8 phase B, hardware AES-NI/PCLMULQDQ path). keylen 16 or 32, + * nonce 12 bytes. seal writes out[ptlen] || tag[16]. Returns 0 ok, 1 auth + * failure (open), -2 when no hardware AES is available (phase C fallback). */ +int wo_aes_gcm_available(void); +int wo_aes_gcm_seal(const uint8_t *key, size_t keylen, const uint8_t nonce[12], + const uint8_t *aad, size_t aadlen, const uint8_t *pt, + size_t ptlen, uint8_t *out); +int wo_aes_gcm_open(const uint8_t *key, size_t keylen, const uint8_t nonce[12], + const uint8_t *aad, size_t aadlen, const uint8_t *ct, + size_t ctlen, const uint8_t tag[16], uint8_t *out); + int wo_builtin_crypto(wo_vm *vm, uint64_t *R, uint32_t ins, const char **msg); #endif diff --git a/runtime/src/loader.c b/runtime/src/loader.c index fab902e..1d069ac 100644 --- a/runtime/src/loader.c +++ b/runtime/src/loader.c @@ -80,6 +80,7 @@ static const uint8_t b_arity[WO_B_MAX + 1] = { [WO_B_NET_SEND_FD] = 2, [WO_B_NET_RECV_FD] = 1, [WO_B_NET_CONNECT_UNIX] = 1, [WO_B_TERM_SIZE] = 2, [WO_B_TERM_WIDTH] = 1, [WO_B_NET_CONNECT] = 2, [WO_B_CHACHA20POLY1305_SEAL] = 4, [WO_B_CHACHA20POLY1305_OPEN] = 4, + [WO_B_AES_GCM_SEAL] = 4, [WO_B_AES_GCM_OPEN] = 4, /* json (json.c): encode takes the value's static kind, decode the class id to build */ [WO_B_JSON_ENCODE] = 2, [WO_B_JSON_DECODE] = 2, [WO_B_MAP_GET_OPT] = 2, diff --git a/runtime/src/wob.h b/runtime/src/wob.h index 4802e26..84551ca 100644 --- a/runtime/src/wob.h +++ b/runtime/src/wob.h @@ -558,9 +558,12 @@ enum { * failure or a too-short input). */ WO_B_CHACHA20POLY1305_SEAL = 111, /* (key, nonce, aad, plaintext) -> Bytes */ WO_B_CHACHA20POLY1305_OPEN = 112, /* (key, nonce, aad, ct||tag) -> ?Bytes */ + /* rv2 8 phase B: AES-GCM (AES-128/256 by key length), AES-NI hardware. */ + WO_B_AES_GCM_SEAL = 113, /* (key, nonce, aad, plaintext) -> Bytes */ + WO_B_AES_GCM_OPEN = 114, /* (key, nonce, aad, ct||tag) -> ?Bytes */ }; -#define WO_B_MAX 112u +#define WO_B_MAX 114u /* ids at or above this one live in sysio.c, not builtin.c */ #define WO_B_SYS_FIRST WO_B_FS_EXISTS diff --git a/runtime/test/test_crypto.c b/runtime/test/test_crypto.c index 0159fb0..21eaaa5 100644 --- a/runtime/test/test_crypto.c +++ b/runtime/test/test_crypto.c @@ -52,6 +52,37 @@ static void t_poly1305(const uint8_t key[32], const char *msg, size_t mlen, T_CHECK(strcmp(got, want) == 0); } +static size_t unhex(const char *h, uint8_t *out) { + size_t n = strlen(h) / 2; + for (size_t i = 0; i < n; i++) { + unsigned v; + sscanf(h + 2 * i, "%2x", &v); + out[i] = (uint8_t)v; + } + return n; +} + +/* NIST SP 800-38D GCM test vector: seal matches ct||tag, open round-trips, + * a flipped tag is rejected. `cth` is the expected ciphertext concatenated + * with the 16-byte tag. */ +static void t_aesgcm(const char *kh, const char *ih, const char *ah, + const char *ph, const char *cth) { + uint8_t key[32], iv[12], aad[64], pt[256], expect[272], out[272], back[256]; + size_t klen = unhex(kh, key), alen = unhex(ah, aad); + unhex(ih, iv); + size_t plen = unhex(ph, pt), elen = unhex(cth, expect); + T_CHECK(elen == plen + 16); + T_CHECK(wo_aes_gcm_seal(key, klen, iv, aad, alen, pt, plen, out) == 0); + T_CHECK(memcmp(out, expect, plen + 16) == 0); + T_CHECK(wo_aes_gcm_open(key, klen, iv, aad, alen, out, plen, out + plen, + back) == 0); + T_CHECK(memcmp(back, pt, plen) == 0); + uint8_t bad[16]; + memcpy(bad, out + plen, 16); + bad[0] ^= 0x01; + T_CHECK(wo_aes_gcm_open(key, klen, iv, aad, alen, out, plen, bad, back) == 1); +} + int main(void) { /* RFC 3174 */ t_sha1("abc", 3, "a9993e364706816aba3e25717850c26c9cd0d89d"); @@ -174,5 +205,29 @@ int main(void) { T_CHECK(rc == 1); } + /* AES-GCM (rv2 8 phase B) — NIST SP 800-38D test vectors, when hardware + * AES is present (the bitsliced software fallback is phase C). */ + if (wo_aes_gcm_available()) { + /* AES-128 GCM test case 4 */ + t_aesgcm("feffe9928665731c6d6a8f9467308308", + "cafebabefacedbaddecaf888", + "feedfacedeadbeeffeedfacedeadbeefabaddad2", + "d9313225f88406e5a55909c5aff5269a86a7a9531534f7da2e4c303d8a318a721" + "c3c0c95956809532fcf0e2449a6b525b16aedf5aa0de657ba637b39", + "42831ec2217774244b7221b784d0d49ce3aa212f2c02a4e035c17e2329aca12e2" + "1d514b25466931c7d8f6a5aac84aa051ba30b396a0aac973d58e091" + "5bc94fbc3221a5db94fae95ae7121a47"); + /* AES-256 GCM test case 16 */ + t_aesgcm("feffe9928665731c6d6a8f9467308308" + "feffe9928665731c6d6a8f9467308308", + "cafebabefacedbaddecaf888", + "feedfacedeadbeeffeedfacedeadbeefabaddad2", + "d9313225f88406e5a55909c5aff5269a86a7a9531534f7da2e4c303d8a318a721" + "c3c0c95956809532fcf0e2449a6b525b16aedf5aa0de657ba637b39", + "522dc1f099567d07f47f37a32a84427d643a8cdcbfe5c0c97598a2bd2555d1aa8" + "cb08e48590dbb3da7b08b1056828838c5f61e6393ba7a0abcc9f662" + "76fc6ece0f4e1768cddf8853bb2d551b"); + } + return t_report("test_crypto"); }