code wiki / (root) / nx_ml_kem_768_wasm.nx

nx_ml_kem_768_wasm.nx source

↩ module page · 1258 lines · 52831 B

1// nx_ml_kem_768_wasm.nx -- Provides cryptographic operations for secure key exchange and messaging in the Nishi sovereign ecosystem. 2const KYBER_MAGIC_32768: i64 = 32768 3const KYBER_MAGIC_65536: i64 = 65536 4const KYBER_MAGIC_20159: i64 = 20159 5const KYBER_MAGIC_33554432: i64 = 33554432 6const KYBER_MAGIC_1044: i64 = 1044 7const KYBER_MAGIC_1517: i64 = 1517 8const KYBER_MAGIC_1493: i64 = 1493 9const KYBER_MAGIC_1422: i64 = 1422 10const KYBER_MAGIC_1577: i64 = 1577 11const KYBER_MAGIC_1202: i64 = 1202 12const KYBER_MAGIC_1474: i64 = 1474 13const KYBER_MAGIC_1468: i64 = 1468 14const KYBER_MAGIC_1325: i64 = 1325 15const KYBER_MAGIC_1458: i64 = 1458 16const KYBER_MAGIC_1602: i64 = 1602 17const KYBER_MAGIC_1542: i64 = 1542 18const KYBER_MAGIC_1571: i64 = 1571 19const KYBER_MAGIC_1223: i64 = 1223 20const KYBER_MAGIC_1293: i64 = 1293 21const KYBER_MAGIC_1491: i64 = 1491 22const KYBER_MAGIC_1544: i64 = 1544 23const KYBER_MAGIC_1618: i64 = 1618 24const KYBER_MAGIC_1162: i64 = 1162 25const KYBER_MAGIC_1469: i64 = 1469 26const KYBER_MAGIC_1421: i64 = 1421 27const KYBER_MAGIC_1508: i64 = 1508 28const KYBER_MAGIC_1065: i64 = 1065 29const KYBER_MAGIC_1275: i64 = 1275 30const KYBER_MAGIC_1103: i64 = 1103 31const KYBER_MAGIC_1251: i64 = 1251 32const KYBER_MAGIC_1550: i64 = 1550 33const KYBER_MAGIC_1574: i64 = 1574 34const KYBER_MAGIC_1653: i64 = 1653 35const KYBER_MAGIC_1159: i64 = 1159 36const KYBER_MAGIC_1483: i64 = 1483 37const KYBER_MAGIC_1119: i64 = 1119 38const KYBER_MAGIC_1590: i64 = 1590 39const KYBER_MAGIC_1097: i64 = 1097 40const KYBER_MAGIC_1322: i64 = 1322 41const KYBER_MAGIC_1285: i64 = 1285 42const KYBER_MAGIC_1465: i64 = 1465 43const KYBER_MAGIC_1215: i64 = 1215 44const KYBER_MAGIC_1218: i64 = 1218 45const KYBER_MAGIC_1335: i64 = 1335 46const KYBER_MAGIC_1187: i64 = 1187 47const KYBER_MAGIC_1659: i64 = 1659 48const KYBER_MAGIC_1185: i64 = 1185 49const KYBER_MAGIC_1530: i64 = 1530 50const KYBER_MAGIC_1278: i64 = 1278 51const KYBER_MAGIC_1510: i64 = 1510 52const KYBER_MAGIC_1460: i64 = 1460 53const KYBER_MAGIC_1522: i64 = 1522 54const KYBER_MAGIC_1628: i64 = 1628 55const KYBER_MAGIC_1441: i64 = 1441 56const KYBER_MAGIC_1535: i64 = 1535 57const KYBER_MAGIC_1536: i64 = 1536 58const KYBER_MAGIC_1567: i64 = 1567 59const KYBER_MAGIC_1568: i64 = 1568 60const KYBER_MAGIC_1599: i64 = 1599 61const KYBER_MAGIC_1600: i64 = 1600 62const KYBER_MAGIC_1663: i64 = 1663 63const KYBER_MAGIC_1664: i64 = 1664 64const KYBER_MAGIC_2175: i64 = 2175 65const KYBER_MAGIC_2176: i64 = 2176 66const KYBER_MAGIC_3711: i64 = 3711 67const KYBER_MAGIC_3712: i64 = 3712 68const KYBER_MAGIC_5247: i64 = 5247 69const KYBER_MAGIC_5248: i64 = 5248 70const KYBER_MAGIC_6783: i64 = 6783 71const KYBER_MAGIC_6784: i64 = 6784 72const KYBER_MAGIC_7295: i64 = 7295 73const KYBER_MAGIC_7296: i64 = 7296 74const KYBER_MAGIC_8831: i64 = 8831 75const KYBER_MAGIC_8832: i64 = 8832 76const KYBER_MAGIC_10367: i64 = 10367 77const KYBER_MAGIC_10368: i64 = 10368 78const KYBER_MAGIC_10879: i64 = 10879 79const KYBER_MAGIC_10880: i64 = 10880 80const KYBER_MAGIC_11391: i64 = 11391 81const KYBER_MAGIC_11392: i64 = 11392 82const KYBER_MAGIC_11903: i64 = 11903 83const KYBER_MAGIC_11904: i64 = 11904 84const KYBER_MAGIC_13439: i64 = 13439 85const KYBER_MAGIC_13440: i64 = 13440 86const KYBER_MAGIC_14975: i64 = 14975 87const KYBER_MAGIC_14976: i64 = 14976 88const KYBER_MAGIC_16511: i64 = 16511 89const KYBER_MAGIC_16512: i64 = 16512 90const KYBER_MAGIC_18047: i64 = 18047 91const KYBER_MAGIC_18048: i64 = 18048 92const KYBER_MAGIC_18559: i64 = 18559 93const KYBER_MAGIC_18560: i64 = 18560 94const KYBER_MAGIC_19071: i64 = 19071 95const KYBER_MAGIC_19072: i64 = 19072 96const KYBER_MAGIC_1152: i64 = 1152 97const KYBER_MAGIC_1184: i64 = 1184 98const KYBER_MAGIC_20224: i64 = 20224 99const KYBER_MAGIC_20096: i64 = 20096 100const KYBER_MAGIC_20160: i64 = 20160 101const KYBER_MAGIC_19103: i64 = 19103 102const KYBER_MAGIC_19104: i64 = 19104 103const KYBER_MAGIC_20191: i64 = 20191 104const KYBER_MAGIC_20192: i64 = 20192 105const KYBER_MAGIC_20255: i64 = 20255 106const KYBER_MAGIC_20256: i64 = 20256 107const KYBER_MAGIC_20319: i64 = 20319 108const KYBER_MAGIC_20320: i64 = 20320 109const KYBER_MAGIC_20351: i64 = 20351 110const KYBER_MAGIC_22000: i64 = 22000 111const KYBER_MAGIC_23119: i64 = 23119 112const KYBER_MAGIC_28000: i64 = 28000 113const KYBER_MAGIC_30000: i64 = 30000 114const KYBER_MAGIC_2400: i64 = 2400 115const KYBER_MAGIC_32400: i64 = 32400 116const KYBER_MAGIC_33000: i64 = 33000 117const KYBER_MAGIC_1088: i64 = 1088 118const KYBER_MAGIC_34088: i64 = 34088 119const KYBER_MAGIC_34200: i64 = 34200 120const KYBER_MAGIC_34300: i64 = 34300 121// nx_ml_kem_768_wasm.nx -- Sovereign single-WASM FIPS 203 ML-KEM-768. 122// 123// END STATE per user directive 2026-05-16: "i want all sovering nishi 124// lang from the bits up". 125// 126// Single self-contained NishiLang file compiling to ONE substrate-WASM 127// that exports the complete ML-KEM-768 surface. No JS orchestrator. 128// No cross-module imports. Every cryptographic operation runs inside 129// this one .wasm in the browser, called once per keygen/encaps/decaps. 130// 131// Composes (inlined, since WAT-target requires self-contained modules): 132// * Keccak-f[1600] permutation (shared by SHA-3-256/512 + SHAKE128/256) 133// * Sponge wrappers for SHA-3-256, SHA-3-512, SHAKE128, SHAKE256 134// * Kyber NTT / INVNTT / basemul + zetas[0..127] 135// * Polynomial ops: add, sub, CBD-eta2, compress10/4, decompress10/4, 136// tobytes12, frombytes12, tomont, msg_to_poly, msg_from_poly 137// * SampleNTT rejection sampler 138// * K-PKE keygen, encrypt, decrypt 139// * ML-KEM-768 keygen, encaps, decaps with FO transform 140// * Round-trip self-test (KAT in the wasm itself) 141// 142// Public API: 143// nx_mlkem_keygen(seed_64, scratch, ek_out, dk_out) -> i64 144// nx_mlkem_encaps(ek, m_32, scratch, ct_out, ss_out) -> i64 145// nx_mlkem_decaps(dk, ct, scratch, ss_out) -> i64 146// nx_mlkem_round_trip_test(seed_64, msg_32, scratch) -> i64 147// returns 0 if encaps/decaps produce same shared secret, non-zero 148// bitmap of failure axes otherwise. 149// 150// Scratch buffer requirement: 65536 bytes (browser allocates one big 151// buffer and the wasm uses it for all intermediates). 152// 153// Verification path: 154// 1. nx_mlkem_round_trip_test returns 0 (built-in KAT) 155// 2. nx_mlkem_encaps(ek, m_fixed)+nx_mlkem_decaps(dk, ct) byte-match 156// 3. (TODO) FIPS 203 Appendix A KAT vectors when available 157// 158// Pillar 4 alignment: this module supersedes the kyber.js JS stitch 159// shipped L123. JS now does only I/O (load wasm, supply randomness 160// from CSPRNG via window.crypto.getRandomValues, allocate scratch), 161// no algorithm. The sovereign-substrate doctrine is reached. 162// 163// license_tier: INDEPENDENT_REDERIVE 164// genealogy_id: international-research-sources/nist/fips_203 165// lineage_id: nishi_ml_kem_768_wasm_q1 166// safe_shift_audit: shared Keccak _rotl64 uses gold-standard mask 167// nxc_alias_safe: all internal poly_op-equivalents write to dedicated slots 168 169// ============================================================================ 170// SECTION 0: Constants 171// ============================================================================ 172 173const KYBER_Q: i64 = 3329 174const KYBER_QINV: i64 = 62209 175const KYBER_N: i64 = 256 176const KYBER_K: i64 = 3 177const KYBER_ETA: i64 = 2 178const KYBER_F: i64 = 1353 // R^2 mod q (for tomont) 179const KYBER_HALF: i64 = 1665 // (q+1)/2 (for msg codec) 180 181const MLKEM_PK_BYTES: i64 = 1184 182const MLKEM_SK_BYTES: i64 = 2400 183const MLKEM_CT_BYTES: i64 = 1088 184const MLKEM_SS_BYTES: i64 = 32 185 186// Polynomial in-memory size (256 i16 LE) 187const POLY_BUF: i64 = 512 188const POLY_BYTES_12: i64 = 384 // 12-bit canonical encode 189const POLY_COMPR_U: i64 = 320 // d_u = 10 190const POLY_COMPR_V: i64 = 128 // d_v = 4 191 192// ============================================================================ 193// SECTION 1: Shared bit / byte helpers 194// ============================================================================ 195 196func _u16_load_le(p: *u8, i: i64) -> i64 { 197 let lo: i64 = p[i * 2] 198 let hi: i64 = p[i * 2 + 1] 199 let raw: i64 = lo | (hi << 8) 200 if raw >= KYBER_MAGIC_32768 { return raw - KYBER_MAGIC_65536 } 201 return raw 202} 203 204func _u16_store_le(p: *u8, i: i64, v: i64) -> i64 { 205 var vv: i64 = v 206 if vv < 0 { vv = vv + KYBER_MAGIC_65536 } 207 p[i * 2] = vv & 0xff 208 p[i * 2 + 1] = (vv >> 8) & 0xff 209 return 0 210} 211 212func _canon_q(v: i64) -> i64 { 213 var x: i64 = v % KYBER_Q 214 if x < 0 { x = x + KYBER_Q } 215 return x 216} 217 218// 64-bit logical right shift via gold-standard mask (per F-meta-5 / nx_bitops). 219func _rotl64(x: i64, n: i64) -> i64 { 220 let nn: i64 = n & 63 221 if nn == 0 { return x } 222 let shr_amt: i64 = 64 - nn 223 let mask: i64 = (1 << nn) - 1 224 return ((x << nn) | ((x >> shr_amt) & mask)) & 0xffffffffffffffff 225} 226 227// ============================================================================ 228// SECTION 2: Keccak-f[1600] permutation (shared by SHA-3 + SHAKE) 229// ============================================================================ 230 231func _lane_load(p: *u8) -> i64 { 232 let b0: i64 = p[0] 233 let b1: i64 = p[1] 234 let b2: i64 = p[2] 235 let b3: i64 = p[3] 236 let b4: i64 = p[4] 237 let b5: i64 = p[5] 238 let b6: i64 = p[6] 239 let b7: i64 = p[7] 240 return b0 | (b1 << 8) | (b2 << 16) | (b3 << 24) 241 | (b4 << 32) | (b5 << 40) | (b6 << 48) | (b7 << 56) 242} 243 244func _lane_store(p: *u8, v: i64) -> i64 { 245 p[0] = v & 0xff 246 p[1] = (v >> 8) & 0xff 247 p[2] = (v >> 16) & 0xff 248 p[3] = (v >> 24) & 0xff 249 p[4] = (v >> 32) & 0xff 250 p[5] = (v >> 40) & 0xff 251 p[6] = (v >> 48) & 0xff 252 p[7] = (v >> 56) & 0xff 253 return 0 254} 255 256func _keccak_rc(i: i64) -> i64 { 257 if i == 0 { return 0x0000000000000001 } if i == 1 { return 0x0000000000008082 } 258 if i == 2 { return 0x800000000000808a } if i == 3 { return 0x8000000080008000 } 259 if i == 4 { return 0x000000000000808b } if i == 5 { return 0x0000000080000001 } 260 if i == 6 { return 0x8000000080008081 } if i == 7 { return 0x8000000000008009 } 261 if i == 8 { return 0x000000000000008a } if i == 9 { return 0x0000000000000088 } 262 if i == 10 { return 0x0000000080008009 } if i == 11 { return 0x000000008000000a } 263 if i == 12 { return 0x000000008000808b } if i == 13 { return 0x800000000000008b } 264 if i == 14 { return 0x8000000000008089 } if i == 15 { return 0x8000000000008003 } 265 if i == 16 { return 0x8000000000008002 } if i == 17 { return 0x8000000000000080 } 266 if i == 18 { return 0x000000000000800a } if i == 19 { return 0x800000008000000a } 267 if i == 20 { return 0x8000000080008081 } if i == 21 { return 0x8000000000008080 } 268 if i == 22 { return 0x0000000080000001 } 269 return 0x8000000080008008 270} 271 272func _rho_off(lane_idx: i64) -> i64 { 273 if lane_idx == 0 { return 0 } if lane_idx == 1 { return 1 } 274 if lane_idx == 2 { return 62 } if lane_idx == 3 { return 28 } 275 if lane_idx == 4 { return 27 } if lane_idx == 5 { return 36 } 276 if lane_idx == 6 { return 44 } if lane_idx == 7 { return 6 } 277 if lane_idx == 8 { return 55 } if lane_idx == 9 { return 20 } 278 if lane_idx == 10 { return 3 } if lane_idx == 11 { return 10 } 279 if lane_idx == 12 { return 43 } if lane_idx == 13 { return 25 } 280 if lane_idx == 14 { return 39 } if lane_idx == 15 { return 41 } 281 if lane_idx == 16 { return 45 } if lane_idx == 17 { return 15 } 282 if lane_idx == 18 { return 21 } if lane_idx == 19 { return 8 } 283 if lane_idx == 20 { return 18 } if lane_idx == 21 { return 2 } 284 if lane_idx == 22 { return 61 } if lane_idx == 23 { return 56 } 285 return 14 286} 287 288// state_ptr layout: 200 bytes state + 40 bytes C + 200 bytes B = 440 bytes 289func _keccak_f1600(state_ptr: *u8) -> i64 { 290 let A: *u8 = state_ptr 291 let C: *u8 = (state_ptr as i64 + 200) as *u8 292 let B: *u8 = (state_ptr as i64 + 240) as *u8 293 294 var round: i64 = 0 295 while round < 24 { 296 // theta 297 var x: i64 = 0 298 while x < 5 { 299 let c0: i64 = _lane_load((A as i64 + 8 * (x + 0)) as *u8) 300 let c1: i64 = _lane_load((A as i64 + 8 * (x + 5)) as *u8) 301 let c2: i64 = _lane_load((A as i64 + 8 * (x + 10)) as *u8) 302 let c3: i64 = _lane_load((A as i64 + 8 * (x + 15)) as *u8) 303 let c4: i64 = _lane_load((A as i64 + 8 * (x + 20)) as *u8) 304 _lane_store((C as i64 + 8 * x) as *u8, c0 ^ c1 ^ c2 ^ c3 ^ c4) 305 x = x + 1 306 } 307 var x2: i64 = 0 308 while x2 < 5 { 309 let xm: i64 = (x2 + 4) % 5 310 let xp: i64 = (x2 + 1) % 5 311 let cl: i64 = _lane_load((C as i64 + 8 * xm) as *u8) 312 let cr: i64 = _lane_load((C as i64 + 8 * xp) as *u8) 313 let d: i64 = cl ^ _rotl64(cr, 1) 314 var y: i64 = 0 315 while y < 5 { 316 let off: i64 = 8 * (x2 + 5 * y) 317 let v: i64 = _lane_load((A as i64 + off) as *u8) ^ d 318 _lane_store((A as i64 + off) as *u8, v) 319 y = y + 1 320 } 321 x2 = x2 + 1 322 } 323 // rho + pi 324 var y3: i64 = 0 325 while y3 < 5 { 326 var x3: i64 = 0 327 while x3 < 5 { 328 let src_idx: i64 = x3 + 5 * y3 329 let rot: i64 = _rho_off(src_idx) 330 let lane: i64 = _lane_load((A as i64 + 8 * src_idx) as *u8) 331 let rotated: i64 = _rotl64(lane, rot) 332 let new_x: i64 = y3 333 let new_y: i64 = (2 * x3 + 3 * y3) % 5 334 let dst_idx: i64 = new_x + 5 * new_y 335 _lane_store((B as i64 + 8 * dst_idx) as *u8, rotated) 336 x3 = x3 + 1 337 } 338 y3 = y3 + 1 339 } 340 // chi 341 var y4: i64 = 0 342 while y4 < 5 { 343 var x4: i64 = 0 344 while x4 < 5 { 345 let xp1: i64 = (x4 + 1) % 5 346 let xp2: i64 = (x4 + 2) % 5 347 let b0: i64 = _lane_load((B as i64 + 8 * (x4 + 5 * y4)) as *u8) 348 let b1: i64 = _lane_load((B as i64 + 8 * (xp1 + 5 * y4)) as *u8) 349 let b2: i64 = _lane_load((B as i64 + 8 * (xp2 + 5 * y4)) as *u8) 350 let nb1: i64 = (~b1) & 0xffffffffffffffff 351 let v: i64 = b0 ^ (nb1 & b2) 352 _lane_store((A as i64 + 8 * (x4 + 5 * y4)) as *u8, v) 353 x4 = x4 + 1 354 } 355 y4 = y4 + 1 356 } 357 // iota 358 let a00: i64 = _lane_load(A) ^ _keccak_rc(round) 359 _lane_store(A, a00) 360 361 round = round + 1 362 } 363 return 0 364} 365 366// ============================================================================ 367// SECTION 3: Sponge wrappers (SHA-3-256, SHA-3-512, SHAKE128, SHAKE256) 368// ============================================================================ 369 370// Generic sponge absorb+squeeze for known-output-length variants. 371// state_ptr needs 440 bytes (200 state + 240 scratch for Keccak). 372// dom_byte: 0x06 for SHA-3, 0x1f for SHAKE. 373func _sponge_one_shot(msg: *u8, msg_len: i64, 374 rate: i64, dom_byte: i64, 375 state: *u8, 376 out: *u8, out_len: i64) -> i64 { 377 var i: i64 = 0 378 while i < 200 { state[i] = 0; i = i + 1 } 379 380 var pos: i64 = 0 381 while pos + rate <= msg_len { 382 var b: i64 = 0 383 while b < rate { 384 state[b] = (state[b] ^ msg[pos + b]) & 0xff 385 b = b + 1 386 } 387 _keccak_f1600(state) 388 pos = pos + rate 389 } 390 let tail: i64 = msg_len - pos 391 var t: i64 = 0 392 while t < tail { 393 state[t] = (state[t] ^ msg[pos + t]) & 0xff 394 t = t + 1 395 } 396 state[tail] = (state[tail] ^ dom_byte) & 0xff 397 state[rate - 1] = (state[rate - 1] ^ 0x80) & 0xff 398 _keccak_f1600(state) 399 400 var written: i64 = 0 401 while written < out_len { 402 let remaining: i64 = out_len - written 403 var take: i64 = rate 404 if remaining < rate { take = remaining } 405 var k: i64 = 0 406 while k < take { 407 out[written + k] = state[k] 408 k = k + 1 409 } 410 written = written + take 411 if written < out_len { _keccak_f1600(state) } 412 } 413 return 0 414} 415 416func _sha3_256(msg: *u8, msg_len: i64, state: *u8, out: *u8) -> i64 { 417 return _sponge_one_shot(msg, msg_len, 136, 0x06, state, out, 32) 418} 419func _sha3_512(msg: *u8, msg_len: i64, state: *u8, out: *u8) -> i64 { 420 return _sponge_one_shot(msg, msg_len, 72, 0x06, state, out, 64) 421} 422func _shake128(msg: *u8, msg_len: i64, state: *u8, out: *u8, out_len: i64) -> i64 { 423 return _sponge_one_shot(msg, msg_len, 168, 0x1f, state, out, out_len) 424} 425func _shake256(msg: *u8, msg_len: i64, state: *u8, out: *u8, out_len: i64) -> i64 { 426 return _sponge_one_shot(msg, msg_len, 136, 0x1f, state, out, out_len) 427} 428 429// ============================================================================ 430// SECTION 4: Modular arithmetic (Montgomery + Barrett) 431// ============================================================================ 432 433func _mont(a: i64) -> i64 { 434 var u: i64 = (a * KYBER_QINV) & 0xffff 435 if u >= KYBER_MAGIC_32768 { u = u - KYBER_MAGIC_65536 } 436 return (a - u * KYBER_Q) >> 16 437} 438 439func _barrett(a: i64) -> i64 { 440 let v: i64 = KYBER_MAGIC_20159 441 let t: i64 = (v * a + KYBER_MAGIC_33554432) >> 26 442 return a - t * KYBER_Q 443} 444 445func _fqmul(a: i64, b: i64) -> i64 { return _mont(a * b) } 446 447// ============================================================================ 448// SECTION 5: Kyber zetas (all 128) and NTT / INVNTT 449// ============================================================================ 450 451func _zeta(i: i64) -> i64 { 452 if i == 0 { return -KYBER_MAGIC_1044 } if i == 1 { return -758 } if i == 2 { return -359 } if i == 3 { return -KYBER_MAGIC_1517 } 453 if i == 4 { return KYBER_MAGIC_1493 } if i == 5 { return KYBER_MAGIC_1422 } if i == 6 { return 287 } if i == 7 { return 202 } 454 if i == 8 { return -171 } if i == 9 { return 622 } if i == 10 { return KYBER_MAGIC_1577 } if i == 11 { return 182 } 455 if i == 12 { return 962 } if i == 13 { return -KYBER_MAGIC_1202 } if i == 14 { return -KYBER_MAGIC_1474 } if i == 15 { return KYBER_MAGIC_1468 } 456 if i == 16 { return 573 } if i == 17 { return -KYBER_MAGIC_1325 } if i == 18 { return 264 } if i == 19 { return 383 } 457 if i == 20 { return -829 } if i == 21 { return KYBER_MAGIC_1458 } if i == 22 { return -KYBER_MAGIC_1602 } if i == 23 { return -130 } 458 if i == 24 { return -681 } if i == 25 { return 1017 } if i == 26 { return 732 } if i == 27 { return 608 } 459 if i == 28 { return -KYBER_MAGIC_1542 } if i == 29 { return 411 } if i == 30 { return -205 } if i == 31 { return -KYBER_MAGIC_1571 } 460 if i == 32 { return KYBER_MAGIC_1223 } if i == 33 { return 652 } if i == 34 { return -552 } if i == 35 { return 1015 } 461 if i == 36 { return -KYBER_MAGIC_1293 } if i == 37 { return KYBER_MAGIC_1491 } if i == 38 { return -282 } if i == 39 { return -KYBER_MAGIC_1544 } 462 if i == 40 { return 516 } if i == 41 { return -8 } if i == 42 { return -320 } if i == 43 { return -666 } 463 if i == 44 { return -KYBER_MAGIC_1618 } if i == 45 { return -KYBER_MAGIC_1162 } if i == 46 { return 126 } if i == 47 { return KYBER_MAGIC_1469 } 464 if i == 48 { return -853 } if i == 49 { return -90 } if i == 50 { return -271 } if i == 51 { return 830 } 465 if i == 52 { return 107 } if i == 53 { return -KYBER_MAGIC_1421 } if i == 54 { return -247 } if i == 55 { return -951 } 466 if i == 56 { return -398 } if i == 57 { return 961 } if i == 58 { return -KYBER_MAGIC_1508 } if i == 59 { return -725 } 467 if i == 60 { return 448 } if i == 61 { return -KYBER_MAGIC_1065 } if i == 62 { return 677 } if i == 63 { return -KYBER_MAGIC_1275 } 468 if i == 64 { return -KYBER_MAGIC_1103 } if i == 65 { return 430 } if i == 66 { return 555 } if i == 67 { return 843 } 469 if i == 68 { return -KYBER_MAGIC_1251 } if i == 69 { return 871 } if i == 70 { return KYBER_MAGIC_1550 } if i == 71 { return 105 } 470 if i == 72 { return 422 } if i == 73 { return 587 } if i == 74 { return 177 } if i == 75 { return -235 } 471 if i == 76 { return -291 } if i == 77 { return -460 } if i == 78 { return KYBER_MAGIC_1574 } if i == 79 { return KYBER_MAGIC_1653 } 472 if i == 80 { return -246 } if i == 81 { return 778 } if i == 82 { return KYBER_MAGIC_1159 } if i == 83 { return -147 } 473 if i == 84 { return -777 } if i == 85 { return KYBER_MAGIC_1483 } if i == 86 { return -602 } if i == 87 { return KYBER_MAGIC_1119 } 474 if i == 88 { return -KYBER_MAGIC_1590 } if i == 89 { return 644 } if i == 90 { return -872 } if i == 91 { return 349 } 475 if i == 92 { return 418 } if i == 93 { return 329 } if i == 94 { return -156 } if i == 95 { return -75 } 476 if i == 96 { return 817 } if i == 97 { return KYBER_MAGIC_1097 } if i == 98 { return 603 } if i == 99 { return 610 } 477 if i == 100 { return KYBER_MAGIC_1322 } if i == 101 { return -KYBER_MAGIC_1285 } if i == 102 { return -KYBER_MAGIC_1465 } if i == 103 { return 384 } 478 if i == 104 { return -KYBER_MAGIC_1215 } if i == 105 { return -136 } if i == 106 { return KYBER_MAGIC_1218 } if i == 107 { return -KYBER_MAGIC_1335 } 479 if i == 108 { return -874 } if i == 109 { return 220 } if i == 110 { return -KYBER_MAGIC_1187 } if i == 111 { return -KYBER_MAGIC_1659 } 480 if i == 112 { return -KYBER_MAGIC_1185 } if i == 113 { return -KYBER_MAGIC_1530 } if i == 114 { return -KYBER_MAGIC_1278 } if i == 115 { return 794 } 481 if i == 116 { return -KYBER_MAGIC_1510 } if i == 117 { return -854 } if i == 118 { return -870 } if i == 119 { return 478 } 482 if i == 120 { return -108 } if i == 121 { return -308 } if i == 122 { return 996 } if i == 123 { return 991 } 483 if i == 124 { return 958 } if i == 125 { return -KYBER_MAGIC_1460 } if i == 126 { return KYBER_MAGIC_1522 } 484 return KYBER_MAGIC_1628 485} 486 487func _ntt(poly: *u8) -> i64 { 488 var k: i64 = 1 489 var len: i64 = 128 490 while len >= 2 { 491 var start: i64 = 0 492 while start < KYBER_N { 493 let zeta: i64 = _zeta(k) 494 k = k + 1 495 var j: i64 = start 496 while j < start + len { 497 let aj: i64 = _u16_load_le(poly, j) 498 let ajl: i64 = _u16_load_le(poly, j + len) 499 let t: i64 = _fqmul(zeta, ajl) 500 _u16_store_le(poly, j + len, aj - t) 501 _u16_store_le(poly, j, aj + t) 502 j = j + 1 503 } 504 start = j + len 505 } 506 len = len >> 1 507 } 508 return 0 509} 510 511func _invntt(poly: *u8) -> i64 { 512 let f: i64 = KYBER_MAGIC_1441 513 var k: i64 = 127 514 var len: i64 = 2 515 while len <= 128 { 516 var start: i64 = 0 517 while start < KYBER_N { 518 let zeta: i64 = _zeta(k) 519 k = k - 1 520 var j: i64 = start 521 while j < start + len { 522 let aj: i64 = _u16_load_le(poly, j) 523 let ajl: i64 = _u16_load_le(poly, j + len) 524 _u16_store_le(poly, j, _barrett(aj + ajl)) 525 let diff: i64 = ajl - aj 526 _u16_store_le(poly, j + len, _fqmul(zeta, diff)) 527 j = j + 1 528 } 529 start = j + len 530 } 531 len = len << 1 532 } 533 var i: i64 = 0 534 while i < KYBER_N { 535 let v: i64 = _u16_load_le(poly, i) 536 _u16_store_le(poly, i, _fqmul(v, f)) 537 i = i + 1 538 } 539 return 0 540} 541 542// ============================================================================ 543// SECTION 6: Basemul + tomont 544// ============================================================================ 545 546func _basemul_pair_acc(acc: *u8, off: i64, a: *u8, b: *u8, zeta: i64) -> i64 { 547 let a0: i64 = _u16_load_le(a, off) 548 let a1: i64 = _u16_load_le(a, off + 1) 549 let b0: i64 = _u16_load_le(b, off) 550 let b1: i64 = _u16_load_le(b, off + 1) 551 let r0: i64 = _fqmul(_fqmul(a1, b1), zeta) + _fqmul(a0, b0) 552 let r1: i64 = _fqmul(a0, b1) + _fqmul(a1, b0) 553 _u16_store_le(acc, off, _u16_load_le(acc, off) + r0) 554 _u16_store_le(acc, off + 1, _u16_load_le(acc, off + 1) + r1) 555 return 0 556} 557 558func _poly_basemul_acc(acc: *u8, a: *u8, b: *u8) -> i64 { 559 var i: i64 = 0 560 while i < 64 { 561 let z: i64 = _zeta(64 + i) 562 _basemul_pair_acc(acc, i * 4, a, b, z) 563 _basemul_pair_acc(acc, i * 4 + 2, a, b, -z) 564 i = i + 1 565 } 566 return 0 567} 568 569func _poly_zero(p: *u8) -> i64 { 570 var i: i64 = 0 571 while i < KYBER_N { _u16_store_le(p, i, 0); i = i + 1 } 572 return 0 573} 574 575func _poly_tomont(poly: *u8) -> i64 { 576 var i: i64 = 0 577 while i < KYBER_N { 578 let c: i64 = _u16_load_le(poly, i) 579 _u16_store_le(poly, i, _mont(c * KYBER_F)) 580 i = i + 1 581 } 582 return 0 583} 584 585func _poly_add(out: *u8, a: *u8, b: *u8) -> i64 { 586 var i: i64 = 0 587 while i < KYBER_N { 588 _u16_store_le(out, i, _u16_load_le(a, i) + _u16_load_le(b, i)) 589 i = i + 1 590 } 591 return 0 592} 593 594func _poly_sub(out: *u8, a: *u8, b: *u8) -> i64 { 595 var i: i64 = 0 596 while i < KYBER_N { 597 _u16_store_le(out, i, _u16_load_le(a, i) - _u16_load_le(b, i)) 598 i = i + 1 599 } 600 return 0 601} 602 603// ============================================================================ 604// SECTION 7: Polynomial encoding / compress 605// ============================================================================ 606 607func _compress10_one(v: i64) -> i64 { 608 let x: i64 = _canon_q(v) 609 return (((x << 11) / KYBER_Q + 1) >> 1) & 0x3ff 610} 611func _decompress10_one(x: i64) -> i64 { return (KYBER_Q * x + 512) >> 10 } 612 613func _poly_compress10(bytes_out: *u8, poly: *u8) -> i64 { 614 var i: i64 = 0 615 while i < KYBER_N { 616 let c0: i64 = _compress10_one(_u16_load_le(poly, i + 0)) 617 let c1: i64 = _compress10_one(_u16_load_le(poly, i + 1)) 618 let c2: i64 = _compress10_one(_u16_load_le(poly, i + 2)) 619 let c3: i64 = _compress10_one(_u16_load_le(poly, i + 3)) 620 let off: i64 = (i >> 2) * 5 621 bytes_out[off + 0] = c0 & 0xff 622 bytes_out[off + 1] = ((c0 >> 8) | (c1 << 2)) & 0xff 623 bytes_out[off + 2] = ((c1 >> 6) | (c2 << 4)) & 0xff 624 bytes_out[off + 3] = ((c2 >> 4) | (c3 << 6)) & 0xff 625 bytes_out[off + 4] = (c3 >> 2) & 0xff 626 i = i + 4 627 } 628 return 0 629} 630 631func _poly_decompress10(poly: *u8, bytes_in: *u8) -> i64 { 632 var i: i64 = 0 633 while i < KYBER_N { 634 let off: i64 = (i >> 2) * 5 635 let b0: i64 = bytes_in[off + 0] 636 let b1: i64 = bytes_in[off + 1] 637 let b2: i64 = bytes_in[off + 2] 638 let b3: i64 = bytes_in[off + 3] 639 let b4: i64 = bytes_in[off + 4] 640 let c0: i64 = b0 | ((b1 & 0x03) << 8) 641 let c1: i64 = (b1 >> 2) | ((b2 & 0x0f) << 6) 642 let c2: i64 = (b2 >> 4) | ((b3 & 0x3f) << 4) 643 let c3: i64 = (b3 >> 6) | (b4 << 2) 644 _u16_store_le(poly, i + 0, _decompress10_one(c0 & 0x3ff)) 645 _u16_store_le(poly, i + 1, _decompress10_one(c1 & 0x3ff)) 646 _u16_store_le(poly, i + 2, _decompress10_one(c2 & 0x3ff)) 647 _u16_store_le(poly, i + 3, _decompress10_one(c3 & 0x3ff)) 648 i = i + 4 649 } 650 return 0 651} 652 653func _compress4_one(v: i64) -> i64 { 654 let x: i64 = _canon_q(v) 655 return (((x << 5) / KYBER_Q + 1) >> 1) & 0xf 656} 657func _decompress4_one(x: i64) -> i64 { return (KYBER_Q * x + 8) >> 4 } 658 659func _poly_compress4(bytes_out: *u8, poly: *u8) -> i64 { 660 var i: i64 = 0 661 while i < KYBER_N { 662 let c0: i64 = _compress4_one(_u16_load_le(poly, i + 0)) 663 let c1: i64 = _compress4_one(_u16_load_le(poly, i + 1)) 664 bytes_out[i >> 1] = (c0 | (c1 << 4)) & 0xff 665 i = i + 2 666 } 667 return 0 668} 669 670func _poly_decompress4(poly: *u8, bytes_in: *u8) -> i64 { 671 var i: i64 = 0 672 while i < KYBER_N { 673 let b: i64 = bytes_in[i >> 1] 674 _u16_store_le(poly, i + 0, _decompress4_one(b & 0xf)) 675 _u16_store_le(poly, i + 1, _decompress4_one((b >> 4) & 0xf)) 676 i = i + 2 677 } 678 return 0 679} 680 681func _poly_tobytes12(bytes_out: *u8, poly: *u8) -> i64 { 682 var i: i64 = 0 683 while i < KYBER_N { 684 let c0: i64 = _canon_q(_u16_load_le(poly, i + 0)) & 0xfff 685 let c1: i64 = _canon_q(_u16_load_le(poly, i + 1)) & 0xfff 686 let off: i64 = (i >> 1) * 3 687 bytes_out[off + 0] = c0 & 0xff 688 bytes_out[off + 1] = ((c0 >> 8) | (c1 << 4)) & 0xff 689 bytes_out[off + 2] = (c1 >> 4) & 0xff 690 i = i + 2 691 } 692 return 0 693} 694 695func _poly_frombytes12(poly: *u8, bytes_in: *u8) -> i64 { 696 var i: i64 = 0 697 while i < KYBER_N { 698 let off: i64 = (i >> 1) * 3 699 let b0: i64 = bytes_in[off + 0] 700 let b1: i64 = bytes_in[off + 1] 701 let b2: i64 = bytes_in[off + 2] 702 let c0: i64 = b0 | ((b1 & 0x0f) << 8) 703 let c1: i64 = (b1 >> 4) | (b2 << 4) 704 _u16_store_le(poly, i + 0, c0 & 0xfff) 705 _u16_store_le(poly, i + 1, c1 & 0xfff) 706 i = i + 2 707 } 708 return 0 709} 710 711// ============================================================================ 712// SECTION 8: CBD-eta2 + SampleNTT + msg codec 713// ============================================================================ 714 715func _poly_cbd_eta2(out: *u8, buf: *u8) -> i64 { 716 var i: i64 = 0 717 while i < KYBER_N { 718 let byte_idx: i64 = i >> 1 719 let upper: i64 = i & 1 720 let nibble: i64 = (buf[byte_idx] >> (upper * 4)) & 0xf 721 let b0: i64 = nibble & 1 722 let b1: i64 = (nibble >> 1) & 1 723 let b2: i64 = (nibble >> 2) & 1 724 let b3: i64 = (nibble >> 3) & 1 725 _u16_store_le(out, i, (b0 + b1) - (b2 + b3)) 726 i = i + 1 727 } 728 return 0 729} 730 731// Sample uniform NTT-form polynomial from SHAKE128 byte stream. 732// Returns 0 on success, -1 if buf exhausted. 733func _sample_ntt(poly: *u8, buf: *u8, buf_len: i64) -> i64 { 734 var i: i64 = 0 735 var pos: i64 = 0 736 while i < KYBER_N { 737 if pos + 3 > buf_len { return -1 } 738 let b0: i64 = buf[pos] 739 let b1: i64 = buf[pos + 1] 740 let b2: i64 = buf[pos + 2] 741 pos = pos + 3 742 let d1: i64 = b0 | ((b1 & 0x0f) << 8) 743 let d2: i64 = (b1 >> 4) | (b2 << 4) 744 if d1 < KYBER_Q { 745 _u16_store_le(poly, i, d1) 746 i = i + 1 747 } 748 if i < KYBER_N { 749 if d2 < KYBER_Q { 750 _u16_store_le(poly, i, d2) 751 i = i + 1 752 } 753 } 754 } 755 return 0 756} 757 758func _msg_to_poly(poly: *u8, msg: *u8) -> i64 { 759 var i: i64 = 0 760 while i < 32 { 761 let byte: i64 = msg[i] 762 var b: i64 = 0 763 while b < 8 { 764 let bit: i64 = (byte >> b) & 1 765 _u16_store_le(poly, i * 8 + b, bit * KYBER_HALF) 766 b = b + 1 767 } 768 i = i + 1 769 } 770 return 0 771} 772 773func _msg_from_poly(msg: *u8, poly: *u8) -> i64 { 774 var i: i64 = 0 775 while i < 32 { 776 var byte: i64 = 0 777 var b: i64 = 0 778 while b < 8 { 779 let c: i64 = _canon_q(_u16_load_le(poly, i * 8 + b)) 780 let bit: i64 = (((c << 1) + KYBER_HALF) / KYBER_Q) & 1 781 byte = byte | (bit << b) 782 b = b + 1 783 } 784 msg[i] = byte & 0xff 785 i = i + 1 786 } 787 return 0 788} 789 790// ============================================================================ 791// SECTION 9: Higher-level Kyber compose: matrix expansion, noise sampling 792// ============================================================================ 793 794// Build matrix A entry (rho || x_byte || y_byte) -> 256-coef polynomial. 795// SHAKE128 squeeze 504 bytes, run rejection sampler. Retries with more 796// bytes if needed (up to 840 -- 5 rate blocks). 797// scratch >= 1300 bytes (240 keccak + 840 shake squeeze + slack). 798func _gen_matrix_entry(poly: *u8, rho: *u8, x: i64, y: i64, 799 keccak_state: *u8, shake_buf: *u8) -> i64 { 800 let seed_buf: *u8 = (shake_buf as i64 + 840) as *u8 801 var i: i64 = 0 802 while i < 32 { seed_buf[i] = rho[i]; i = i + 1 } 803 seed_buf[32] = x & 0xff 804 seed_buf[33] = y & 0xff 805 806 var attempt: i64 = 0 807 while attempt < 3 { 808 let try_len: i64 = 504 + attempt * 168 809 _shake128(seed_buf, 34, keccak_state, shake_buf, try_len) 810 let rc: i64 = _sample_ntt(poly, shake_buf, try_len) 811 if rc == 0 { return 0 } 812 attempt = attempt + 1 813 } 814 return -1 815} 816 817// CBD-eta2 noise polynomial from (sigma || nonce_byte) via SHAKE256. 818// scratch >= 240 keccak + 128 shake squeeze. 819func _gen_noise_eta2(poly: *u8, sigma: *u8, nonce: i64, 820 keccak_state: *u8, shake_buf: *u8) -> i64 { 821 let seed_buf: *u8 = (shake_buf as i64 + 128) as *u8 822 var i: i64 = 0 823 while i < 32 { seed_buf[i] = sigma[i]; i = i + 1 } 824 seed_buf[32] = nonce & 0xff 825 _shake256(seed_buf, 33, keccak_state, shake_buf, 128) 826 _poly_cbd_eta2(poly, shake_buf) 827 return 0 828} 829 830// Constant-time conditional-copy: if mask==0xff out=src1, if mask==0x00 out=src0. 831func _cmov(out: *u8, src1: *u8, len: i64, mask: i64) -> i64 { 832 let m: i64 = mask & 0xff 833 var i: i64 = 0 834 while i < len { 835 out[i] = (out[i] ^ ((out[i] ^ src1[i]) & m)) & 0xff 836 i = i + 1 837 } 838 return 0 839} 840 841// ============================================================================ 842// SECTION 10: K-PKE keygen / encrypt / decrypt 843// ============================================================================ 844 845// Scratch buffer layout for K-PKE keygen + encaps + decaps. Caller passes 846// one big buffer; we partition. 847// 848// IMPORTANT bug-class learned 2026-05-16: Keccak's _f1600 permutation uses 849// state_ptr[0..439] internally (200 A + 40 C + 200 B work). Any other 850// buffer that overlaps state_ptr+240..439 will be CORRUPTED whenever the 851// permutation runs. So all SHAKE squeeze-into-buffer destinations MUST 852// live PAST offset 440 from the Keccak state base. shake_buf below 853// starts at scratch+512 for safe clearance. Original layout put it at 854// scratch+240 -- silently corrupted every multi-block squeeze. 855// 856// Offset Size Purpose 857// 0 440 Keccak state + work area (do NOT overlap) 858// 440 72 padding 859// 512 1024 SHAKE squeeze buffer (matrix + noise expansion) 860// 1280 32 rho 861// 1312 32 sigma (G output second half) 862// 1344 64 G output (rho || sigma) 863// 1408 512 poly scratch A (per-iteration matrix entry) 864// 1920 1536 s vector polys (3 * 512) 865// 3456 1536 e vector polys 866// 4992 1536 t vector polys (output) 867// 6528 512 acc poly 868// 7040 1536 rVec polys (encrypt) 869// 8576 1536 e1 polys (encrypt) 870// 10112 512 e2 poly (encrypt) 871// 10624 512 mu poly (encrypt) 872// 11136 512 v poly (encrypt) 873// 11648 1536 u polys (encrypt) 874// 13184 1536 tHat polys (encrypt) 875// 14720 1536 sHat polys (decrypt) 876// 16256 1536 u' polys (decrypt) 877// 17792 512 vp poly (decrypt) 878// 18304 512 w poly (decrypt) 879// 18816 32 m' (decrypt output) 880// 18848 1088 ct' (decap re-encrypt verification) 881// 19936 32 z (rejection randomness) 882// 19968 32 K shared secret 883// 20000 1024 slack 884// 885// Total: ~21000 bytes. Caller supplies 32 KB minimum, 64 KB recommended. 886 887func _kpke_keygen(d_seed_32: *u8, scratch: *u8, ek_out: *u8, dk_pke_out: *u8) -> i64 { 888 let keccak_state: *u8 = scratch // scratch_range: keccak 0..439 889 let shake_buf: *u8 = (scratch as i64 + 512) as *u8 // scratch_range: shake_buf 512..KYBER_MAGIC_1535 890 let rho: *u8 = (scratch as i64 + KYBER_MAGIC_1536) as *u8 // scratch_range: rho KYBER_MAGIC_1536..KYBER_MAGIC_1567 891 let sigma: *u8 = (scratch as i64 + KYBER_MAGIC_1568) as *u8 // scratch_range: sigma KYBER_MAGIC_1568..KYBER_MAGIC_1599 892 let g_out: *u8 = (scratch as i64 + KYBER_MAGIC_1600) as *u8 // scratch_range: g_out_kg KYBER_MAGIC_1600..KYBER_MAGIC_1663 893 let A_entry: *u8 = (scratch as i64 + KYBER_MAGIC_1664) as *u8 // scratch_range: A_entry KYBER_MAGIC_1664..KYBER_MAGIC_2175 894 let s_vec: *u8 = (scratch as i64 + KYBER_MAGIC_2176) as *u8 // scratch_range: s_vec KYBER_MAGIC_2176..KYBER_MAGIC_3711 895 let e_vec: *u8 = (scratch as i64 + KYBER_MAGIC_3712) as *u8 // scratch_range: e_vec KYBER_MAGIC_3712..KYBER_MAGIC_5247 896 let t_vec: *u8 = (scratch as i64 + KYBER_MAGIC_5248) as *u8 // scratch_range: t_vec KYBER_MAGIC_5248..KYBER_MAGIC_6783 897 let acc: *u8 = (scratch as i64 + KYBER_MAGIC_6784) as *u8 // scratch_range: acc_kg KYBER_MAGIC_6784..KYBER_MAGIC_7295 898 899 // 1. (rho, sigma) = G(d || k=3). Reuse shake_buf as the (d||k) input scratch. 900 var i: i64 = 0 901 while i < 32 { shake_buf[i] = d_seed_32[i]; i = i + 1 } 902 shake_buf[32] = KYBER_K & 0xff 903 _sha3_512(shake_buf, 33, keccak_state, g_out) 904 var k: i64 = 0 905 while k < 32 { rho[k] = g_out[k]; sigma[k] = g_out[k + 32]; k = k + 1 } 906 907 // 2. Sample s_vec, e_vec via CBD eta_2 (eta_1 = eta_2 = 2 for ML-KEM-768) 908 var nonce: i64 = 0 909 var v: i64 = 0 910 while v < KYBER_K { 911 _gen_noise_eta2((s_vec as i64 + v * POLY_BUF) as *u8, sigma, nonce, 912 keccak_state, shake_buf) 913 nonce = nonce + 1 914 v = v + 1 915 } 916 v = 0 917 while v < KYBER_K { 918 _gen_noise_eta2((e_vec as i64 + v * POLY_BUF) as *u8, sigma, nonce, 919 keccak_state, shake_buf) 920 nonce = nonce + 1 921 v = v + 1 922 } 923 924 // 3. NTT(s), NTT(e) 925 v = 0 926 while v < KYBER_K { _ntt((s_vec as i64 + v * POLY_BUF) as *u8); v = v + 1 } 927 v = 0 928 while v < KYBER_K { _ntt((e_vec as i64 + v * POLY_BUF) as *u8); v = v + 1 } 929 930 // 4. t_vec[i] = sum_j A[i][j] * s[j] then tomont, then + e[i] 931 var rr: i64 = 0 932 while rr < KYBER_K { 933 _poly_zero(acc) 934 var jj: i64 = 0 935 while jj < KYBER_K { 936 _gen_matrix_entry(A_entry, rho, jj, rr, keccak_state, shake_buf) 937 _poly_basemul_acc(acc, A_entry, (s_vec as i64 + jj * POLY_BUF) as *u8) 938 jj = jj + 1 939 } 940 _poly_tomont(acc) 941 _poly_add((t_vec as i64 + rr * POLY_BUF) as *u8, 942 acc, 943 (e_vec as i64 + rr * POLY_BUF) as *u8) 944 rr = rr + 1 945 } 946 947 // 5. ek = ByteEncode_12(t̂) || rho 948 var ii: i64 = 0 949 while ii < KYBER_K { 950 _poly_tobytes12((ek_out as i64 + ii * POLY_BYTES_12) as *u8, 951 (t_vec as i64 + ii * POLY_BUF) as *u8) 952 ii = ii + 1 953 } 954 var rho_i: i64 = 0 955 while rho_i < 32 { ek_out[KYBER_K * POLY_BYTES_12 + rho_i] = rho[rho_i]; rho_i = rho_i + 1 } 956 957 // 6. dk_pke = ByteEncode_12(ŝ) 958 ii = 0 959 while ii < KYBER_K { 960 _poly_tobytes12((dk_pke_out as i64 + ii * POLY_BYTES_12) as *u8, 961 (s_vec as i64 + ii * POLY_BUF) as *u8) 962 ii = ii + 1 963 } 964 return 0 965} 966 967func _kpke_encrypt(ek: *u8, msg_32: *u8, coins_32: *u8, 968 scratch: *u8, ct_out: *u8) -> i64 { 969 let keccak_state: *u8 = scratch // 0..439 970 let shake_buf: *u8 = (scratch as i64 + 512) as *u8 // scratch_range: shake_buf 512..KYBER_MAGIC_1535 971 let A_entry: *u8 = (scratch as i64 + KYBER_MAGIC_1664) as *u8 // scratch_range: A_entry KYBER_MAGIC_1664..KYBER_MAGIC_2175 972 let acc: *u8 = (scratch as i64 + KYBER_MAGIC_6784) as *u8 // scratch_range: acc_en KYBER_MAGIC_6784..KYBER_MAGIC_7295 973 let rVec: *u8 = (scratch as i64 + KYBER_MAGIC_7296) as *u8 // scratch_range: rVec KYBER_MAGIC_7296..KYBER_MAGIC_8831 974 let e1: *u8 = (scratch as i64 + KYBER_MAGIC_8832) as *u8 // scratch_range: e1 KYBER_MAGIC_8832..KYBER_MAGIC_10367 975 let e2: *u8 = (scratch as i64 + KYBER_MAGIC_10368) as *u8 // scratch_range: e2 KYBER_MAGIC_10368..KYBER_MAGIC_10879 976 let mu: *u8 = (scratch as i64 + KYBER_MAGIC_10880) as *u8 // scratch_range: mu KYBER_MAGIC_10880..KYBER_MAGIC_11391 977 let v_poly: *u8 = (scratch as i64 + KYBER_MAGIC_11392) as *u8 // scratch_range: v_poly KYBER_MAGIC_11392..KYBER_MAGIC_11903 978 let u_vec: *u8 = (scratch as i64 + KYBER_MAGIC_11904) as *u8 // scratch_range: u_vec_en KYBER_MAGIC_11904..KYBER_MAGIC_13439 979 let tHat: *u8 = (scratch as i64 + KYBER_MAGIC_13440) as *u8 // scratch_range: tHat KYBER_MAGIC_13440..KYBER_MAGIC_14975 980 981 // 1. Parse ek -> tHat + rho 982 var ii: i64 = 0 983 while ii < KYBER_K { 984 _poly_frombytes12((tHat as i64 + ii * POLY_BUF) as *u8, 985 (ek as i64 + ii * POLY_BYTES_12) as *u8) 986 ii = ii + 1 987 } 988 let rho: *u8 = (ek as i64 + KYBER_K * POLY_BYTES_12) as *u8 989 990 // 2. Sample r, e1 (eta2), e2 (eta2) 991 var nonce: i64 = 0 992 var i: i64 = 0 993 while i < KYBER_K { 994 _gen_noise_eta2((rVec as i64 + i * POLY_BUF) as *u8, coins_32, nonce, 995 keccak_state, shake_buf) 996 nonce = nonce + 1 997 i = i + 1 998 } 999 i = 0 1000 while i < KYBER_K { 1001 _gen_noise_eta2((e1 as i64 + i * POLY_BUF) as *u8, coins_32, nonce, 1002 keccak_state, shake_buf) 1003 nonce = nonce + 1 1004 i = i + 1 1005 } 1006 _gen_noise_eta2(e2, coins_32, nonce, keccak_state, shake_buf) 1007 1008 // 3. r̂ = NTT(r) 1009 i = 0 1010 while i < KYBER_K { _ntt((rVec as i64 + i * POLY_BUF) as *u8); i = i + 1 } 1011 1012 // 4. u = INVNTT(A^T ∘ r̂) + e1 1013 var rr: i64 = 0 1014 while rr < KYBER_K { 1015 _poly_zero(acc) 1016 var jj: i64 = 0 1017 while jj < KYBER_K { 1018 // Transpose: A^T[i][j] uses rho || i || j (swapped order vs keygen) 1019 _gen_matrix_entry(A_entry, rho, rr, jj, keccak_state, shake_buf) 1020 _poly_basemul_acc(acc, A_entry, (rVec as i64 + jj * POLY_BUF) as *u8) 1021 jj = jj + 1 1022 } 1023 _invntt(acc) 1024 _poly_add((u_vec as i64 + rr * POLY_BUF) as *u8, 1025 acc, 1026 (e1 as i64 + rr * POLY_BUF) as *u8) 1027 rr = rr + 1 1028 } 1029 1030 // 5. v = INVNTT(tHat^T ∘ r̂) + e2 + Decompress_1(msg) 1031 _poly_zero(acc) 1032 var jj: i64 = 0 1033 while jj < KYBER_K { 1034 _poly_basemul_acc(acc, 1035 (tHat as i64 + jj * POLY_BUF) as *u8, 1036 (rVec as i64 + jj * POLY_BUF) as *u8) 1037 jj = jj + 1 1038 } 1039 _invntt(acc) 1040 _poly_add(v_poly, acc, e2) 1041 _msg_to_poly(mu, msg_32) 1042 _poly_add(v_poly, v_poly, mu) 1043 1044 // 6. c1 = ByteEncode_du(Compress_du(u)) (3 * 320 = 960 B) 1045 // c2 = ByteEncode_dv(Compress_dv(v)) (128 B) 1046 var k: i64 = 0 1047 while k < KYBER_K { 1048 _poly_compress10((ct_out as i64 + k * POLY_COMPR_U) as *u8, 1049 (u_vec as i64 + k * POLY_BUF) as *u8) 1050 k = k + 1 1051 } 1052 _poly_compress4((ct_out as i64 + KYBER_K * POLY_COMPR_U) as *u8, v_poly) 1053 return 0 1054} 1055 1056func _kpke_decrypt(dk_pke: *u8, ct: *u8, scratch: *u8, msg_out: *u8) -> i64 { 1057 let acc: *u8 = (scratch as i64 + KYBER_MAGIC_6784) as *u8 // scratch_range: acc_dc KYBER_MAGIC_6784..KYBER_MAGIC_7295 1058 let sHat: *u8 = (scratch as i64 + KYBER_MAGIC_14976) as *u8 // scratch_range: sHat KYBER_MAGIC_14976..KYBER_MAGIC_16511 1059 let u_vec: *u8 = (scratch as i64 + KYBER_MAGIC_16512) as *u8 // scratch_range: u_vec_dc KYBER_MAGIC_16512..KYBER_MAGIC_18047 1060 let vp: *u8 = (scratch as i64 + KYBER_MAGIC_18048) as *u8 // scratch_range: vp KYBER_MAGIC_18048..KYBER_MAGIC_18559 1061 let w: *u8 = (scratch as i64 + KYBER_MAGIC_18560) as *u8 // scratch_range: w KYBER_MAGIC_18560..KYBER_MAGIC_19071 1062 1063 // 1. Decompress c1 -> u; c2 -> vp 1064 var i: i64 = 0 1065 while i < KYBER_K { 1066 _poly_decompress10((u_vec as i64 + i * POLY_BUF) as *u8, 1067 (ct as i64 + i * POLY_COMPR_U) as *u8) 1068 i = i + 1 1069 } 1070 _poly_decompress4(vp, (ct as i64 + KYBER_K * POLY_COMPR_U) as *u8) 1071 1072 // 2. sHat from dk_pke 1073 i = 0 1074 while i < KYBER_K { 1075 _poly_frombytes12((sHat as i64 + i * POLY_BUF) as *u8, 1076 (dk_pke as i64 + i * POLY_BYTES_12) as *u8) 1077 i = i + 1 1078 } 1079 1080 // 3. NTT(u) 1081 i = 0 1082 while i < KYBER_K { _ntt((u_vec as i64 + i * POLY_BUF) as *u8); i = i + 1 } 1083 1084 // 4. acc = sHat^T ∘ NTT(u); INVNTT -> canonical poly 1085 _poly_zero(acc) 1086 var jj: i64 = 0 1087 while jj < KYBER_K { 1088 _poly_basemul_acc(acc, 1089 (sHat as i64 + jj * POLY_BUF) as *u8, 1090 (u_vec as i64 + jj * POLY_BUF) as *u8) 1091 jj = jj + 1 1092 } 1093 _invntt(acc) 1094 1095 // 5. w = vp - acc; msg = Compress_1(w) 1096 _poly_sub(w, vp, acc) 1097 _msg_from_poly(msg_out, w) 1098 return 0 1099} 1100 1101// ============================================================================ 1102// SECTION 11: ML-KEM-768 keygen / encaps / decaps (FO transform) 1103// ============================================================================ 1104 1105// Public ML-KEM-768 keygen. Takes 64-byte seed (d || z) and produces 1106// (ek_1184, dk_2400). Scratch >= 32 KB. 1107func nx_mlkem_keygen(seed_64: *u8, scratch: *u8, ek_out: *u8, dk_out: *u8) -> i64 { 1108 let keccak_state: *u8 = scratch 1109 let h_ek: *u8 = (scratch as i64 + KYBER_MAGIC_19072) as *u8 // 32 B inside scratch (past decrypt area) 1110 1111 // dk layout: dk_pke (1152) || ek (1184) || H(ek) (32) || z (32) = 2400 B 1112 let dk_pke: *u8 = dk_out 1113 let dk_ek: *u8 = (dk_out as i64 + KYBER_MAGIC_1152) as *u8 1114 let dk_h: *u8 = (dk_out as i64 + KYBER_MAGIC_1152 + KYBER_MAGIC_1184) as *u8 1115 let dk_z: *u8 = (dk_out as i64 + KYBER_MAGIC_1152 + KYBER_MAGIC_1184 + 32) as *u8 1116 1117 let d: *u8 = seed_64 1118 let z: *u8 = (seed_64 as i64 + 32) as *u8 1119 1120 _kpke_keygen(d, scratch, ek_out, dk_pke) 1121 // Copy ek into dk 1122 var i: i64 = 0 1123 while i < MLKEM_PK_BYTES { dk_ek[i] = ek_out[i]; i = i + 1 } 1124 // H(ek) -> dk_h 1125 _sha3_256(ek_out, MLKEM_PK_BYTES, keccak_state, dk_h) 1126 // z -> dk_z 1127 var j: i64 = 0 1128 while j < 32 { dk_z[j] = z[j]; j = j + 1 } 1129 return 0 1130} 1131 1132// Public ML-KEM-768 encaps. Takes 32-byte msg + ek, produces ct + ss. 1133// Scratch >= 32 KB. 1134func nx_mlkem_encaps(ek: *u8, msg_32: *u8, scratch: *u8, 1135 ct_out: *u8, ss_out: *u8) -> i64 { 1136 let keccak_state: *u8 = scratch 1137 let h_ek: *u8 = (scratch as i64 + KYBER_MAGIC_20224) as *u8 // 32 B 1138 let g_buf: *u8 = (scratch as i64 + KYBER_MAGIC_20096) as *u8 // 64 B (m || H(ek)) 1139 let g_out: *u8 = (scratch as i64 + KYBER_MAGIC_20160) as *u8 // 64 B 1140 1141 // (K, r) = G(m || H(ek)) 1142 _sha3_256(ek, MLKEM_PK_BYTES, keccak_state, h_ek) 1143 var i: i64 = 0 1144 while i < 32 { g_buf[i] = msg_32[i]; g_buf[32 + i] = h_ek[i]; i = i + 1 } 1145 _sha3_512(g_buf, 64, keccak_state, g_out) 1146 var k: i64 = 0 1147 while k < 32 { ss_out[k] = g_out[k]; k = k + 1 } 1148 let r: *u8 = (g_out as i64 + 32) as *u8 1149 1150 // ct = K-PKE.encrypt(ek, m, r) 1151 _kpke_encrypt(ek, msg_32, r, scratch, ct_out) 1152 return 0 1153} 1154 1155// Public ML-KEM-768 decaps. Returns 32-byte ss. Scratch >= 32 KB. 1156func nx_mlkem_decaps(dk: *u8, ct: *u8, scratch: *u8, ss_out: *u8) -> i64 { 1157 let keccak_state: *u8 = scratch 1158 let m_prime: *u8 = (scratch as i64 + KYBER_MAGIC_19072) as *u8 // scratch_range: m_prime KYBER_MAGIC_19072..KYBER_MAGIC_19103 1159 let ct_prime:*u8 = (scratch as i64 + KYBER_MAGIC_19104) as *u8 // scratch_range: ct_prime KYBER_MAGIC_19104..KYBER_MAGIC_20191 1160 // g_buf / g_out MUST live past ct_prime+1088 -- the decap encrypt step 1161 // writes ct_prime AFTER g_out is needed for K', so g_out must be 1162 // outside the ct_prime write range or its bytes get clobbered. 1163 let g_buf: *u8 = (scratch as i64 + KYBER_MAGIC_20192) as *u8 // scratch_range: g_buf_dc KYBER_MAGIC_20192..KYBER_MAGIC_20255 1164 let g_out: *u8 = (scratch as i64 + KYBER_MAGIC_20256) as *u8 // scratch_range: g_out_dc KYBER_MAGIC_20256..KYBER_MAGIC_20319 1165 let k_reject:*u8 = (scratch as i64 + KYBER_MAGIC_20320) as *u8 // scratch_range: k_reject KYBER_MAGIC_20320..KYBER_MAGIC_20351 1166 let z_ct: *u8 = (scratch as i64 + KYBER_MAGIC_22000) as *u8 // scratch_range: z_ct KYBER_MAGIC_22000..KYBER_MAGIC_23119 1167 1168 let dk_pke: *u8 = dk 1169 let dk_ek: *u8 = (dk as i64 + KYBER_MAGIC_1152) as *u8 1170 let dk_h: *u8 = (dk as i64 + KYBER_MAGIC_1152 + KYBER_MAGIC_1184) as *u8 1171 let dk_z: *u8 = (dk as i64 + KYBER_MAGIC_1152 + KYBER_MAGIC_1184 + 32) as *u8 1172 1173 // m' = K-PKE.decrypt(dk_pke, ct) 1174 _kpke_decrypt(dk_pke, ct, scratch, m_prime) 1175 1176 // (K', r') = G(m' || h) 1177 var i: i64 = 0 1178 while i < 32 { g_buf[i] = m_prime[i]; g_buf[32 + i] = dk_h[i]; i = i + 1 } 1179 _sha3_512(g_buf, 64, keccak_state, g_out) 1180 1181 // ct' = K-PKE.encrypt(ek, m', r') 1182 let r_prime: *u8 = (g_out as i64 + 32) as *u8 1183 _kpke_encrypt(dk_ek, m_prime, r_prime, scratch, ct_prime) 1184 1185 // diff = constant-time compare ct vs ct' 1186 var diff: i64 = 0 1187 var c: i64 = 0 1188 while c < MLKEM_CT_BYTES { 1189 diff = diff | (ct[c] ^ ct_prime[c]) 1190 c = c + 1 1191 } 1192 1193 // K_reject = J(z || ct) = SHAKE256(z || ct, 32) 1194 var z_i: i64 = 0 1195 while z_i < 32 { z_ct[z_i] = dk_z[z_i]; z_i = z_i + 1 } 1196 var ct_i: i64 = 0 1197 while ct_i < MLKEM_CT_BYTES { z_ct[32 + ct_i] = ct[ct_i]; ct_i = ct_i + 1 } 1198 _shake256(z_ct, 32 + MLKEM_CT_BYTES, keccak_state, k_reject, 32) 1199 1200 // Constant-time select: mask = 0 if diff==0 (success) else 0xff 1201 var mask_acc: i64 = 0 1202 var bit_iter: i64 = 0 1203 while bit_iter < 8 { 1204 mask_acc = mask_acc | (diff >> bit_iter) 1205 bit_iter = bit_iter + 1 1206 } 1207 let mask: i64 = (mask_acc & 1) * 0xff // 0x00 or 0xff 1208 1209 // ss_out = K' if mask==0 else k_reject 1210 var k_i: i64 = 0 1211 while k_i < 32 { 1212 let kp: i64 = g_out[k_i] 1213 let kr: i64 = k_reject[k_i] 1214 ss_out[k_i] = ((kp & (~mask & 0xff)) | (kr & mask)) & 0xff 1215 k_i = k_i + 1 1216 } 1217 return 0 1218} 1219 1220// ============================================================================ 1221// SECTION 12: Round-trip self-test (KAT inside the wasm) 1222// ============================================================================ 1223 1224// Given a 64-byte seed and a 32-byte msg, run keygen+encaps+decaps and 1225// verify the shared secret is recovered byte-exact. 1226// Returns 0 on success, non-zero bitmap of failure axes: 1227// bit 0: keygen returned non-zero 1228// bit 1: encaps returned non-zero 1229// bit 2: decaps returned non-zero 1230// bit 3: shared secret mismatch 1231// scratch >= 64 KB (we use the bottom 32 KB for internals + reserve 1232// extra room for the keys/ct/ss output buffers at the top). 1233func nx_mlkem_round_trip_test(seed_64: *u8, msg_32: *u8, scratch: *u8) -> i64 { 1234 // Carve out output buffers at high addresses inside scratch. 1235 // The kpke / decap internal layout uses up to ~23120; we start outputs at 28000+. 1236 let ek: *u8 = (scratch as i64 + KYBER_MAGIC_28000) as *u8 1237 let dk: *u8 = (scratch as i64 + KYBER_MAGIC_30000) as *u8 // KYBER_MAGIC_2400 B -> ends at KYBER_MAGIC_32400 1238 let ct: *u8 = (scratch as i64 + KYBER_MAGIC_33000) as *u8 // KYBER_MAGIC_1088 B -> ends at KYBER_MAGIC_34088 1239 let ss1: *u8 = (scratch as i64 + KYBER_MAGIC_34200) as *u8 // 32 B 1240 let ss2: *u8 = (scratch as i64 + KYBER_MAGIC_34300) as *u8 // 32 B 1241 1242 var fail_mask: i64 = 0 1243 let rc1: i64 = nx_mlkem_keygen(seed_64, scratch, ek, dk) 1244 if rc1 != 0 { fail_mask = fail_mask | 1 } 1245 let rc2: i64 = nx_mlkem_encaps(ek, msg_32, scratch, ct, ss1) 1246 if rc2 != 0 { fail_mask = fail_mask | 2 } 1247 let rc3: i64 = nx_mlkem_decaps(dk, ct, scratch, ss2) 1248 if rc3 != 0 { fail_mask = fail_mask | 4 } 1249 1250 var diff: i64 = 0 1251 var i: i64 = 0 1252 while i < 32 { 1253 if ss1[i] != ss2[i] { diff = 1 } 1254 i = i + 1 1255 } 1256 if diff != 0 { fail_mask = fail_mask | 8 } 1257 return fail_mask 1258}