nx_ml_kem_768_wasm.nx source
↩ module page · 1258 lines · 52831 B
1// nx_ml_kem_768_wasm.nx -- Provides cryptographic operations for secure key exchange and messaging in the Nishi sovereign ecosystem.
2const KYBER_MAGIC_32768: i64 = 32768
3const KYBER_MAGIC_65536: i64 = 65536
4const KYBER_MAGIC_20159: i64 = 20159
5const KYBER_MAGIC_33554432: i64 = 33554432
6const KYBER_MAGIC_1044: i64 = 1044
7const KYBER_MAGIC_1517: i64 = 1517
8const KYBER_MAGIC_1493: i64 = 1493
9const KYBER_MAGIC_1422: i64 = 1422
10const KYBER_MAGIC_1577: i64 = 1577
11const KYBER_MAGIC_1202: i64 = 1202
12const KYBER_MAGIC_1474: i64 = 1474
13const KYBER_MAGIC_1468: i64 = 1468
14const KYBER_MAGIC_1325: i64 = 1325
15const KYBER_MAGIC_1458: i64 = 1458
16const KYBER_MAGIC_1602: i64 = 1602
17const KYBER_MAGIC_1542: i64 = 1542
18const KYBER_MAGIC_1571: i64 = 1571
19const KYBER_MAGIC_1223: i64 = 1223
20const KYBER_MAGIC_1293: i64 = 1293
21const KYBER_MAGIC_1491: i64 = 1491
22const KYBER_MAGIC_1544: i64 = 1544
23const KYBER_MAGIC_1618: i64 = 1618
24const KYBER_MAGIC_1162: i64 = 1162
25const KYBER_MAGIC_1469: i64 = 1469
26const KYBER_MAGIC_1421: i64 = 1421
27const KYBER_MAGIC_1508: i64 = 1508
28const KYBER_MAGIC_1065: i64 = 1065
29const KYBER_MAGIC_1275: i64 = 1275
30const KYBER_MAGIC_1103: i64 = 1103
31const KYBER_MAGIC_1251: i64 = 1251
32const KYBER_MAGIC_1550: i64 = 1550
33const KYBER_MAGIC_1574: i64 = 1574
34const KYBER_MAGIC_1653: i64 = 1653
35const KYBER_MAGIC_1159: i64 = 1159
36const KYBER_MAGIC_1483: i64 = 1483
37const KYBER_MAGIC_1119: i64 = 1119
38const KYBER_MAGIC_1590: i64 = 1590
39const KYBER_MAGIC_1097: i64 = 1097
40const KYBER_MAGIC_1322: i64 = 1322
41const KYBER_MAGIC_1285: i64 = 1285
42const KYBER_MAGIC_1465: i64 = 1465
43const KYBER_MAGIC_1215: i64 = 1215
44const KYBER_MAGIC_1218: i64 = 1218
45const KYBER_MAGIC_1335: i64 = 1335
46const KYBER_MAGIC_1187: i64 = 1187
47const KYBER_MAGIC_1659: i64 = 1659
48const KYBER_MAGIC_1185: i64 = 1185
49const KYBER_MAGIC_1530: i64 = 1530
50const KYBER_MAGIC_1278: i64 = 1278
51const KYBER_MAGIC_1510: i64 = 1510
52const KYBER_MAGIC_1460: i64 = 1460
53const KYBER_MAGIC_1522: i64 = 1522
54const KYBER_MAGIC_1628: i64 = 1628
55const KYBER_MAGIC_1441: i64 = 1441
56const KYBER_MAGIC_1535: i64 = 1535
57const KYBER_MAGIC_1536: i64 = 1536
58const KYBER_MAGIC_1567: i64 = 1567
59const KYBER_MAGIC_1568: i64 = 1568
60const KYBER_MAGIC_1599: i64 = 1599
61const KYBER_MAGIC_1600: i64 = 1600
62const KYBER_MAGIC_1663: i64 = 1663
63const KYBER_MAGIC_1664: i64 = 1664
64const KYBER_MAGIC_2175: i64 = 2175
65const KYBER_MAGIC_2176: i64 = 2176
66const KYBER_MAGIC_3711: i64 = 3711
67const KYBER_MAGIC_3712: i64 = 3712
68const KYBER_MAGIC_5247: i64 = 5247
69const KYBER_MAGIC_5248: i64 = 5248
70const KYBER_MAGIC_6783: i64 = 6783
71const KYBER_MAGIC_6784: i64 = 6784
72const KYBER_MAGIC_7295: i64 = 7295
73const KYBER_MAGIC_7296: i64 = 7296
74const KYBER_MAGIC_8831: i64 = 8831
75const KYBER_MAGIC_8832: i64 = 8832
76const KYBER_MAGIC_10367: i64 = 10367
77const KYBER_MAGIC_10368: i64 = 10368
78const KYBER_MAGIC_10879: i64 = 10879
79const KYBER_MAGIC_10880: i64 = 10880
80const KYBER_MAGIC_11391: i64 = 11391
81const KYBER_MAGIC_11392: i64 = 11392
82const KYBER_MAGIC_11903: i64 = 11903
83const KYBER_MAGIC_11904: i64 = 11904
84const KYBER_MAGIC_13439: i64 = 13439
85const KYBER_MAGIC_13440: i64 = 13440
86const KYBER_MAGIC_14975: i64 = 14975
87const KYBER_MAGIC_14976: i64 = 14976
88const KYBER_MAGIC_16511: i64 = 16511
89const KYBER_MAGIC_16512: i64 = 16512
90const KYBER_MAGIC_18047: i64 = 18047
91const KYBER_MAGIC_18048: i64 = 18048
92const KYBER_MAGIC_18559: i64 = 18559
93const KYBER_MAGIC_18560: i64 = 18560
94const KYBER_MAGIC_19071: i64 = 19071
95const KYBER_MAGIC_19072: i64 = 19072
96const KYBER_MAGIC_1152: i64 = 1152
97const KYBER_MAGIC_1184: i64 = 1184
98const KYBER_MAGIC_20224: i64 = 20224
99const KYBER_MAGIC_20096: i64 = 20096
100const KYBER_MAGIC_20160: i64 = 20160
101const KYBER_MAGIC_19103: i64 = 19103
102const KYBER_MAGIC_19104: i64 = 19104
103const KYBER_MAGIC_20191: i64 = 20191
104const KYBER_MAGIC_20192: i64 = 20192
105const KYBER_MAGIC_20255: i64 = 20255
106const KYBER_MAGIC_20256: i64 = 20256
107const KYBER_MAGIC_20319: i64 = 20319
108const KYBER_MAGIC_20320: i64 = 20320
109const KYBER_MAGIC_20351: i64 = 20351
110const KYBER_MAGIC_22000: i64 = 22000
111const KYBER_MAGIC_23119: i64 = 23119
112const KYBER_MAGIC_28000: i64 = 28000
113const KYBER_MAGIC_30000: i64 = 30000
114const KYBER_MAGIC_2400: i64 = 2400
115const KYBER_MAGIC_32400: i64 = 32400
116const KYBER_MAGIC_33000: i64 = 33000
117const KYBER_MAGIC_1088: i64 = 1088
118const KYBER_MAGIC_34088: i64 = 34088
119const KYBER_MAGIC_34200: i64 = 34200
120const KYBER_MAGIC_34300: i64 = 34300
121// nx_ml_kem_768_wasm.nx -- Sovereign single-WASM FIPS 203 ML-KEM-768.
122//
123// END STATE per user directive 2026-05-16: "i want all sovering nishi
124// lang from the bits up".
125//
126// Single self-contained NishiLang file compiling to ONE substrate-WASM
127// that exports the complete ML-KEM-768 surface. No JS orchestrator.
128// No cross-module imports. Every cryptographic operation runs inside
129// this one .wasm in the browser, called once per keygen/encaps/decaps.
130//
131// Composes (inlined, since WAT-target requires self-contained modules):
132// * Keccak-f[1600] permutation (shared by SHA-3-256/512 + SHAKE128/256)
133// * Sponge wrappers for SHA-3-256, SHA-3-512, SHAKE128, SHAKE256
134// * Kyber NTT / INVNTT / basemul + zetas[0..127]
135// * Polynomial ops: add, sub, CBD-eta2, compress10/4, decompress10/4,
136// tobytes12, frombytes12, tomont, msg_to_poly, msg_from_poly
137// * SampleNTT rejection sampler
138// * K-PKE keygen, encrypt, decrypt
139// * ML-KEM-768 keygen, encaps, decaps with FO transform
140// * Round-trip self-test (KAT in the wasm itself)
141//
142// Public API:
143// nx_mlkem_keygen(seed_64, scratch, ek_out, dk_out) -> i64
144// nx_mlkem_encaps(ek, m_32, scratch, ct_out, ss_out) -> i64
145// nx_mlkem_decaps(dk, ct, scratch, ss_out) -> i64
146// nx_mlkem_round_trip_test(seed_64, msg_32, scratch) -> i64
147// returns 0 if encaps/decaps produce same shared secret, non-zero
148// bitmap of failure axes otherwise.
149//
150// Scratch buffer requirement: 65536 bytes (browser allocates one big
151// buffer and the wasm uses it for all intermediates).
152//
153// Verification path:
154// 1. nx_mlkem_round_trip_test returns 0 (built-in KAT)
155// 2. nx_mlkem_encaps(ek, m_fixed)+nx_mlkem_decaps(dk, ct) byte-match
156// 3. (TODO) FIPS 203 Appendix A KAT vectors when available
157//
158// Pillar 4 alignment: this module supersedes the kyber.js JS stitch
159// shipped L123. JS now does only I/O (load wasm, supply randomness
160// from CSPRNG via window.crypto.getRandomValues, allocate scratch),
161// no algorithm. The sovereign-substrate doctrine is reached.
162//
163// license_tier: INDEPENDENT_REDERIVE
164// genealogy_id: international-research-sources/nist/fips_203
165// lineage_id: nishi_ml_kem_768_wasm_q1
166// safe_shift_audit: shared Keccak _rotl64 uses gold-standard mask
167// nxc_alias_safe: all internal poly_op-equivalents write to dedicated slots
168
169// ============================================================================
170// SECTION 0: Constants
171// ============================================================================
172
173const KYBER_Q: i64 = 3329
174const KYBER_QINV: i64 = 62209
175const KYBER_N: i64 = 256
176const KYBER_K: i64 = 3
177const KYBER_ETA: i64 = 2
178const KYBER_F: i64 = 1353 // R^2 mod q (for tomont)
179const KYBER_HALF: i64 = 1665 // (q+1)/2 (for msg codec)
180
181const MLKEM_PK_BYTES: i64 = 1184
182const MLKEM_SK_BYTES: i64 = 2400
183const MLKEM_CT_BYTES: i64 = 1088
184const MLKEM_SS_BYTES: i64 = 32
185
186// Polynomial in-memory size (256 i16 LE)
187const POLY_BUF: i64 = 512
188const POLY_BYTES_12: i64 = 384 // 12-bit canonical encode
189const POLY_COMPR_U: i64 = 320 // d_u = 10
190const POLY_COMPR_V: i64 = 128 // d_v = 4
191
192// ============================================================================
193// SECTION 1: Shared bit / byte helpers
194// ============================================================================
195
196func _u16_load_le(p: *u8, i: i64) -> i64 {
197 let lo: i64 = p[i * 2]
198 let hi: i64 = p[i * 2 + 1]
199 let raw: i64 = lo | (hi << 8)
200 if raw >= KYBER_MAGIC_32768 { return raw - KYBER_MAGIC_65536 }
201 return raw
202}
203
204func _u16_store_le(p: *u8, i: i64, v: i64) -> i64 {
205 var vv: i64 = v
206 if vv < 0 { vv = vv + KYBER_MAGIC_65536 }
207 p[i * 2] = vv & 0xff
208 p[i * 2 + 1] = (vv >> 8) & 0xff
209 return 0
210}
211
212func _canon_q(v: i64) -> i64 {
213 var x: i64 = v % KYBER_Q
214 if x < 0 { x = x + KYBER_Q }
215 return x
216}
217
218// 64-bit logical right shift via gold-standard mask (per F-meta-5 / nx_bitops).
219func _rotl64(x: i64, n: i64) -> i64 {
220 let nn: i64 = n & 63
221 if nn == 0 { return x }
222 let shr_amt: i64 = 64 - nn
223 let mask: i64 = (1 << nn) - 1
224 return ((x << nn) | ((x >> shr_amt) & mask)) & 0xffffffffffffffff
225}
226
227// ============================================================================
228// SECTION 2: Keccak-f[1600] permutation (shared by SHA-3 + SHAKE)
229// ============================================================================
230
231func _lane_load(p: *u8) -> i64 {
232 let b0: i64 = p[0]
233 let b1: i64 = p[1]
234 let b2: i64 = p[2]
235 let b3: i64 = p[3]
236 let b4: i64 = p[4]
237 let b5: i64 = p[5]
238 let b6: i64 = p[6]
239 let b7: i64 = p[7]
240 return b0 | (b1 << 8) | (b2 << 16) | (b3 << 24)
241 | (b4 << 32) | (b5 << 40) | (b6 << 48) | (b7 << 56)
242}
243
244func _lane_store(p: *u8, v: i64) -> i64 {
245 p[0] = v & 0xff
246 p[1] = (v >> 8) & 0xff
247 p[2] = (v >> 16) & 0xff
248 p[3] = (v >> 24) & 0xff
249 p[4] = (v >> 32) & 0xff
250 p[5] = (v >> 40) & 0xff
251 p[6] = (v >> 48) & 0xff
252 p[7] = (v >> 56) & 0xff
253 return 0
254}
255
256func _keccak_rc(i: i64) -> i64 {
257 if i == 0 { return 0x0000000000000001 } if i == 1 { return 0x0000000000008082 }
258 if i == 2 { return 0x800000000000808a } if i == 3 { return 0x8000000080008000 }
259 if i == 4 { return 0x000000000000808b } if i == 5 { return 0x0000000080000001 }
260 if i == 6 { return 0x8000000080008081 } if i == 7 { return 0x8000000000008009 }
261 if i == 8 { return 0x000000000000008a } if i == 9 { return 0x0000000000000088 }
262 if i == 10 { return 0x0000000080008009 } if i == 11 { return 0x000000008000000a }
263 if i == 12 { return 0x000000008000808b } if i == 13 { return 0x800000000000008b }
264 if i == 14 { return 0x8000000000008089 } if i == 15 { return 0x8000000000008003 }
265 if i == 16 { return 0x8000000000008002 } if i == 17 { return 0x8000000000000080 }
266 if i == 18 { return 0x000000000000800a } if i == 19 { return 0x800000008000000a }
267 if i == 20 { return 0x8000000080008081 } if i == 21 { return 0x8000000000008080 }
268 if i == 22 { return 0x0000000080000001 }
269 return 0x8000000080008008
270}
271
272func _rho_off(lane_idx: i64) -> i64 {
273 if lane_idx == 0 { return 0 } if lane_idx == 1 { return 1 }
274 if lane_idx == 2 { return 62 } if lane_idx == 3 { return 28 }
275 if lane_idx == 4 { return 27 } if lane_idx == 5 { return 36 }
276 if lane_idx == 6 { return 44 } if lane_idx == 7 { return 6 }
277 if lane_idx == 8 { return 55 } if lane_idx == 9 { return 20 }
278 if lane_idx == 10 { return 3 } if lane_idx == 11 { return 10 }
279 if lane_idx == 12 { return 43 } if lane_idx == 13 { return 25 }
280 if lane_idx == 14 { return 39 } if lane_idx == 15 { return 41 }
281 if lane_idx == 16 { return 45 } if lane_idx == 17 { return 15 }
282 if lane_idx == 18 { return 21 } if lane_idx == 19 { return 8 }
283 if lane_idx == 20 { return 18 } if lane_idx == 21 { return 2 }
284 if lane_idx == 22 { return 61 } if lane_idx == 23 { return 56 }
285 return 14
286}
287
288// state_ptr layout: 200 bytes state + 40 bytes C + 200 bytes B = 440 bytes
289func _keccak_f1600(state_ptr: *u8) -> i64 {
290 let A: *u8 = state_ptr
291 let C: *u8 = (state_ptr as i64 + 200) as *u8
292 let B: *u8 = (state_ptr as i64 + 240) as *u8
293
294 var round: i64 = 0
295 while round < 24 {
296 // theta
297 var x: i64 = 0
298 while x < 5 {
299 let c0: i64 = _lane_load((A as i64 + 8 * (x + 0)) as *u8)
300 let c1: i64 = _lane_load((A as i64 + 8 * (x + 5)) as *u8)
301 let c2: i64 = _lane_load((A as i64 + 8 * (x + 10)) as *u8)
302 let c3: i64 = _lane_load((A as i64 + 8 * (x + 15)) as *u8)
303 let c4: i64 = _lane_load((A as i64 + 8 * (x + 20)) as *u8)
304 _lane_store((C as i64 + 8 * x) as *u8, c0 ^ c1 ^ c2 ^ c3 ^ c4)
305 x = x + 1
306 }
307 var x2: i64 = 0
308 while x2 < 5 {
309 let xm: i64 = (x2 + 4) % 5
310 let xp: i64 = (x2 + 1) % 5
311 let cl: i64 = _lane_load((C as i64 + 8 * xm) as *u8)
312 let cr: i64 = _lane_load((C as i64 + 8 * xp) as *u8)
313 let d: i64 = cl ^ _rotl64(cr, 1)
314 var y: i64 = 0
315 while y < 5 {
316 let off: i64 = 8 * (x2 + 5 * y)
317 let v: i64 = _lane_load((A as i64 + off) as *u8) ^ d
318 _lane_store((A as i64 + off) as *u8, v)
319 y = y + 1
320 }
321 x2 = x2 + 1
322 }
323 // rho + pi
324 var y3: i64 = 0
325 while y3 < 5 {
326 var x3: i64 = 0
327 while x3 < 5 {
328 let src_idx: i64 = x3 + 5 * y3
329 let rot: i64 = _rho_off(src_idx)
330 let lane: i64 = _lane_load((A as i64 + 8 * src_idx) as *u8)
331 let rotated: i64 = _rotl64(lane, rot)
332 let new_x: i64 = y3
333 let new_y: i64 = (2 * x3 + 3 * y3) % 5
334 let dst_idx: i64 = new_x + 5 * new_y
335 _lane_store((B as i64 + 8 * dst_idx) as *u8, rotated)
336 x3 = x3 + 1
337 }
338 y3 = y3 + 1
339 }
340 // chi
341 var y4: i64 = 0
342 while y4 < 5 {
343 var x4: i64 = 0
344 while x4 < 5 {
345 let xp1: i64 = (x4 + 1) % 5
346 let xp2: i64 = (x4 + 2) % 5
347 let b0: i64 = _lane_load((B as i64 + 8 * (x4 + 5 * y4)) as *u8)
348 let b1: i64 = _lane_load((B as i64 + 8 * (xp1 + 5 * y4)) as *u8)
349 let b2: i64 = _lane_load((B as i64 + 8 * (xp2 + 5 * y4)) as *u8)
350 let nb1: i64 = (~b1) & 0xffffffffffffffff
351 let v: i64 = b0 ^ (nb1 & b2)
352 _lane_store((A as i64 + 8 * (x4 + 5 * y4)) as *u8, v)
353 x4 = x4 + 1
354 }
355 y4 = y4 + 1
356 }
357 // iota
358 let a00: i64 = _lane_load(A) ^ _keccak_rc(round)
359 _lane_store(A, a00)
360
361 round = round + 1
362 }
363 return 0
364}
365
366// ============================================================================
367// SECTION 3: Sponge wrappers (SHA-3-256, SHA-3-512, SHAKE128, SHAKE256)
368// ============================================================================
369
370// Generic sponge absorb+squeeze for known-output-length variants.
371// state_ptr needs 440 bytes (200 state + 240 scratch for Keccak).
372// dom_byte: 0x06 for SHA-3, 0x1f for SHAKE.
373func _sponge_one_shot(msg: *u8, msg_len: i64,
374 rate: i64, dom_byte: i64,
375 state: *u8,
376 out: *u8, out_len: i64) -> i64 {
377 var i: i64 = 0
378 while i < 200 { state[i] = 0; i = i + 1 }
379
380 var pos: i64 = 0
381 while pos + rate <= msg_len {
382 var b: i64 = 0
383 while b < rate {
384 state[b] = (state[b] ^ msg[pos + b]) & 0xff
385 b = b + 1
386 }
387 _keccak_f1600(state)
388 pos = pos + rate
389 }
390 let tail: i64 = msg_len - pos
391 var t: i64 = 0
392 while t < tail {
393 state[t] = (state[t] ^ msg[pos + t]) & 0xff
394 t = t + 1
395 }
396 state[tail] = (state[tail] ^ dom_byte) & 0xff
397 state[rate - 1] = (state[rate - 1] ^ 0x80) & 0xff
398 _keccak_f1600(state)
399
400 var written: i64 = 0
401 while written < out_len {
402 let remaining: i64 = out_len - written
403 var take: i64 = rate
404 if remaining < rate { take = remaining }
405 var k: i64 = 0
406 while k < take {
407 out[written + k] = state[k]
408 k = k + 1
409 }
410 written = written + take
411 if written < out_len { _keccak_f1600(state) }
412 }
413 return 0
414}
415
416func _sha3_256(msg: *u8, msg_len: i64, state: *u8, out: *u8) -> i64 {
417 return _sponge_one_shot(msg, msg_len, 136, 0x06, state, out, 32)
418}
419func _sha3_512(msg: *u8, msg_len: i64, state: *u8, out: *u8) -> i64 {
420 return _sponge_one_shot(msg, msg_len, 72, 0x06, state, out, 64)
421}
422func _shake128(msg: *u8, msg_len: i64, state: *u8, out: *u8, out_len: i64) -> i64 {
423 return _sponge_one_shot(msg, msg_len, 168, 0x1f, state, out, out_len)
424}
425func _shake256(msg: *u8, msg_len: i64, state: *u8, out: *u8, out_len: i64) -> i64 {
426 return _sponge_one_shot(msg, msg_len, 136, 0x1f, state, out, out_len)
427}
428
429// ============================================================================
430// SECTION 4: Modular arithmetic (Montgomery + Barrett)
431// ============================================================================
432
433func _mont(a: i64) -> i64 {
434 var u: i64 = (a * KYBER_QINV) & 0xffff
435 if u >= KYBER_MAGIC_32768 { u = u - KYBER_MAGIC_65536 }
436 return (a - u * KYBER_Q) >> 16
437}
438
439func _barrett(a: i64) -> i64 {
440 let v: i64 = KYBER_MAGIC_20159
441 let t: i64 = (v * a + KYBER_MAGIC_33554432) >> 26
442 return a - t * KYBER_Q
443}
444
445func _fqmul(a: i64, b: i64) -> i64 { return _mont(a * b) }
446
447// ============================================================================
448// SECTION 5: Kyber zetas (all 128) and NTT / INVNTT
449// ============================================================================
450
451func _zeta(i: i64) -> i64 {
452 if i == 0 { return -KYBER_MAGIC_1044 } if i == 1 { return -758 } if i == 2 { return -359 } if i == 3 { return -KYBER_MAGIC_1517 }
453 if i == 4 { return KYBER_MAGIC_1493 } if i == 5 { return KYBER_MAGIC_1422 } if i == 6 { return 287 } if i == 7 { return 202 }
454 if i == 8 { return -171 } if i == 9 { return 622 } if i == 10 { return KYBER_MAGIC_1577 } if i == 11 { return 182 }
455 if i == 12 { return 962 } if i == 13 { return -KYBER_MAGIC_1202 } if i == 14 { return -KYBER_MAGIC_1474 } if i == 15 { return KYBER_MAGIC_1468 }
456 if i == 16 { return 573 } if i == 17 { return -KYBER_MAGIC_1325 } if i == 18 { return 264 } if i == 19 { return 383 }
457 if i == 20 { return -829 } if i == 21 { return KYBER_MAGIC_1458 } if i == 22 { return -KYBER_MAGIC_1602 } if i == 23 { return -130 }
458 if i == 24 { return -681 } if i == 25 { return 1017 } if i == 26 { return 732 } if i == 27 { return 608 }
459 if i == 28 { return -KYBER_MAGIC_1542 } if i == 29 { return 411 } if i == 30 { return -205 } if i == 31 { return -KYBER_MAGIC_1571 }
460 if i == 32 { return KYBER_MAGIC_1223 } if i == 33 { return 652 } if i == 34 { return -552 } if i == 35 { return 1015 }
461 if i == 36 { return -KYBER_MAGIC_1293 } if i == 37 { return KYBER_MAGIC_1491 } if i == 38 { return -282 } if i == 39 { return -KYBER_MAGIC_1544 }
462 if i == 40 { return 516 } if i == 41 { return -8 } if i == 42 { return -320 } if i == 43 { return -666 }
463 if i == 44 { return -KYBER_MAGIC_1618 } if i == 45 { return -KYBER_MAGIC_1162 } if i == 46 { return 126 } if i == 47 { return KYBER_MAGIC_1469 }
464 if i == 48 { return -853 } if i == 49 { return -90 } if i == 50 { return -271 } if i == 51 { return 830 }
465 if i == 52 { return 107 } if i == 53 { return -KYBER_MAGIC_1421 } if i == 54 { return -247 } if i == 55 { return -951 }
466 if i == 56 { return -398 } if i == 57 { return 961 } if i == 58 { return -KYBER_MAGIC_1508 } if i == 59 { return -725 }
467 if i == 60 { return 448 } if i == 61 { return -KYBER_MAGIC_1065 } if i == 62 { return 677 } if i == 63 { return -KYBER_MAGIC_1275 }
468 if i == 64 { return -KYBER_MAGIC_1103 } if i == 65 { return 430 } if i == 66 { return 555 } if i == 67 { return 843 }
469 if i == 68 { return -KYBER_MAGIC_1251 } if i == 69 { return 871 } if i == 70 { return KYBER_MAGIC_1550 } if i == 71 { return 105 }
470 if i == 72 { return 422 } if i == 73 { return 587 } if i == 74 { return 177 } if i == 75 { return -235 }
471 if i == 76 { return -291 } if i == 77 { return -460 } if i == 78 { return KYBER_MAGIC_1574 } if i == 79 { return KYBER_MAGIC_1653 }
472 if i == 80 { return -246 } if i == 81 { return 778 } if i == 82 { return KYBER_MAGIC_1159 } if i == 83 { return -147 }
473 if i == 84 { return -777 } if i == 85 { return KYBER_MAGIC_1483 } if i == 86 { return -602 } if i == 87 { return KYBER_MAGIC_1119 }
474 if i == 88 { return -KYBER_MAGIC_1590 } if i == 89 { return 644 } if i == 90 { return -872 } if i == 91 { return 349 }
475 if i == 92 { return 418 } if i == 93 { return 329 } if i == 94 { return -156 } if i == 95 { return -75 }
476 if i == 96 { return 817 } if i == 97 { return KYBER_MAGIC_1097 } if i == 98 { return 603 } if i == 99 { return 610 }
477 if i == 100 { return KYBER_MAGIC_1322 } if i == 101 { return -KYBER_MAGIC_1285 } if i == 102 { return -KYBER_MAGIC_1465 } if i == 103 { return 384 }
478 if i == 104 { return -KYBER_MAGIC_1215 } if i == 105 { return -136 } if i == 106 { return KYBER_MAGIC_1218 } if i == 107 { return -KYBER_MAGIC_1335 }
479 if i == 108 { return -874 } if i == 109 { return 220 } if i == 110 { return -KYBER_MAGIC_1187 } if i == 111 { return -KYBER_MAGIC_1659 }
480 if i == 112 { return -KYBER_MAGIC_1185 } if i == 113 { return -KYBER_MAGIC_1530 } if i == 114 { return -KYBER_MAGIC_1278 } if i == 115 { return 794 }
481 if i == 116 { return -KYBER_MAGIC_1510 } if i == 117 { return -854 } if i == 118 { return -870 } if i == 119 { return 478 }
482 if i == 120 { return -108 } if i == 121 { return -308 } if i == 122 { return 996 } if i == 123 { return 991 }
483 if i == 124 { return 958 } if i == 125 { return -KYBER_MAGIC_1460 } if i == 126 { return KYBER_MAGIC_1522 }
484 return KYBER_MAGIC_1628
485}
486
487func _ntt(poly: *u8) -> i64 {
488 var k: i64 = 1
489 var len: i64 = 128
490 while len >= 2 {
491 var start: i64 = 0
492 while start < KYBER_N {
493 let zeta: i64 = _zeta(k)
494 k = k + 1
495 var j: i64 = start
496 while j < start + len {
497 let aj: i64 = _u16_load_le(poly, j)
498 let ajl: i64 = _u16_load_le(poly, j + len)
499 let t: i64 = _fqmul(zeta, ajl)
500 _u16_store_le(poly, j + len, aj - t)
501 _u16_store_le(poly, j, aj + t)
502 j = j + 1
503 }
504 start = j + len
505 }
506 len = len >> 1
507 }
508 return 0
509}
510
511func _invntt(poly: *u8) -> i64 {
512 let f: i64 = KYBER_MAGIC_1441
513 var k: i64 = 127
514 var len: i64 = 2
515 while len <= 128 {
516 var start: i64 = 0
517 while start < KYBER_N {
518 let zeta: i64 = _zeta(k)
519 k = k - 1
520 var j: i64 = start
521 while j < start + len {
522 let aj: i64 = _u16_load_le(poly, j)
523 let ajl: i64 = _u16_load_le(poly, j + len)
524 _u16_store_le(poly, j, _barrett(aj + ajl))
525 let diff: i64 = ajl - aj
526 _u16_store_le(poly, j + len, _fqmul(zeta, diff))
527 j = j + 1
528 }
529 start = j + len
530 }
531 len = len << 1
532 }
533 var i: i64 = 0
534 while i < KYBER_N {
535 let v: i64 = _u16_load_le(poly, i)
536 _u16_store_le(poly, i, _fqmul(v, f))
537 i = i + 1
538 }
539 return 0
540}
541
542// ============================================================================
543// SECTION 6: Basemul + tomont
544// ============================================================================
545
546func _basemul_pair_acc(acc: *u8, off: i64, a: *u8, b: *u8, zeta: i64) -> i64 {
547 let a0: i64 = _u16_load_le(a, off)
548 let a1: i64 = _u16_load_le(a, off + 1)
549 let b0: i64 = _u16_load_le(b, off)
550 let b1: i64 = _u16_load_le(b, off + 1)
551 let r0: i64 = _fqmul(_fqmul(a1, b1), zeta) + _fqmul(a0, b0)
552 let r1: i64 = _fqmul(a0, b1) + _fqmul(a1, b0)
553 _u16_store_le(acc, off, _u16_load_le(acc, off) + r0)
554 _u16_store_le(acc, off + 1, _u16_load_le(acc, off + 1) + r1)
555 return 0
556}
557
558func _poly_basemul_acc(acc: *u8, a: *u8, b: *u8) -> i64 {
559 var i: i64 = 0
560 while i < 64 {
561 let z: i64 = _zeta(64 + i)
562 _basemul_pair_acc(acc, i * 4, a, b, z)
563 _basemul_pair_acc(acc, i * 4 + 2, a, b, -z)
564 i = i + 1
565 }
566 return 0
567}
568
569func _poly_zero(p: *u8) -> i64 {
570 var i: i64 = 0
571 while i < KYBER_N { _u16_store_le(p, i, 0); i = i + 1 }
572 return 0
573}
574
575func _poly_tomont(poly: *u8) -> i64 {
576 var i: i64 = 0
577 while i < KYBER_N {
578 let c: i64 = _u16_load_le(poly, i)
579 _u16_store_le(poly, i, _mont(c * KYBER_F))
580 i = i + 1
581 }
582 return 0
583}
584
585func _poly_add(out: *u8, a: *u8, b: *u8) -> i64 {
586 var i: i64 = 0
587 while i < KYBER_N {
588 _u16_store_le(out, i, _u16_load_le(a, i) + _u16_load_le(b, i))
589 i = i + 1
590 }
591 return 0
592}
593
594func _poly_sub(out: *u8, a: *u8, b: *u8) -> i64 {
595 var i: i64 = 0
596 while i < KYBER_N {
597 _u16_store_le(out, i, _u16_load_le(a, i) - _u16_load_le(b, i))
598 i = i + 1
599 }
600 return 0
601}
602
603// ============================================================================
604// SECTION 7: Polynomial encoding / compress
605// ============================================================================
606
607func _compress10_one(v: i64) -> i64 {
608 let x: i64 = _canon_q(v)
609 return (((x << 11) / KYBER_Q + 1) >> 1) & 0x3ff
610}
611func _decompress10_one(x: i64) -> i64 { return (KYBER_Q * x + 512) >> 10 }
612
613func _poly_compress10(bytes_out: *u8, poly: *u8) -> i64 {
614 var i: i64 = 0
615 while i < KYBER_N {
616 let c0: i64 = _compress10_one(_u16_load_le(poly, i + 0))
617 let c1: i64 = _compress10_one(_u16_load_le(poly, i + 1))
618 let c2: i64 = _compress10_one(_u16_load_le(poly, i + 2))
619 let c3: i64 = _compress10_one(_u16_load_le(poly, i + 3))
620 let off: i64 = (i >> 2) * 5
621 bytes_out[off + 0] = c0 & 0xff
622 bytes_out[off + 1] = ((c0 >> 8) | (c1 << 2)) & 0xff
623 bytes_out[off + 2] = ((c1 >> 6) | (c2 << 4)) & 0xff
624 bytes_out[off + 3] = ((c2 >> 4) | (c3 << 6)) & 0xff
625 bytes_out[off + 4] = (c3 >> 2) & 0xff
626 i = i + 4
627 }
628 return 0
629}
630
631func _poly_decompress10(poly: *u8, bytes_in: *u8) -> i64 {
632 var i: i64 = 0
633 while i < KYBER_N {
634 let off: i64 = (i >> 2) * 5
635 let b0: i64 = bytes_in[off + 0]
636 let b1: i64 = bytes_in[off + 1]
637 let b2: i64 = bytes_in[off + 2]
638 let b3: i64 = bytes_in[off + 3]
639 let b4: i64 = bytes_in[off + 4]
640 let c0: i64 = b0 | ((b1 & 0x03) << 8)
641 let c1: i64 = (b1 >> 2) | ((b2 & 0x0f) << 6)
642 let c2: i64 = (b2 >> 4) | ((b3 & 0x3f) << 4)
643 let c3: i64 = (b3 >> 6) | (b4 << 2)
644 _u16_store_le(poly, i + 0, _decompress10_one(c0 & 0x3ff))
645 _u16_store_le(poly, i + 1, _decompress10_one(c1 & 0x3ff))
646 _u16_store_le(poly, i + 2, _decompress10_one(c2 & 0x3ff))
647 _u16_store_le(poly, i + 3, _decompress10_one(c3 & 0x3ff))
648 i = i + 4
649 }
650 return 0
651}
652
653func _compress4_one(v: i64) -> i64 {
654 let x: i64 = _canon_q(v)
655 return (((x << 5) / KYBER_Q + 1) >> 1) & 0xf
656}
657func _decompress4_one(x: i64) -> i64 { return (KYBER_Q * x + 8) >> 4 }
658
659func _poly_compress4(bytes_out: *u8, poly: *u8) -> i64 {
660 var i: i64 = 0
661 while i < KYBER_N {
662 let c0: i64 = _compress4_one(_u16_load_le(poly, i + 0))
663 let c1: i64 = _compress4_one(_u16_load_le(poly, i + 1))
664 bytes_out[i >> 1] = (c0 | (c1 << 4)) & 0xff
665 i = i + 2
666 }
667 return 0
668}
669
670func _poly_decompress4(poly: *u8, bytes_in: *u8) -> i64 {
671 var i: i64 = 0
672 while i < KYBER_N {
673 let b: i64 = bytes_in[i >> 1]
674 _u16_store_le(poly, i + 0, _decompress4_one(b & 0xf))
675 _u16_store_le(poly, i + 1, _decompress4_one((b >> 4) & 0xf))
676 i = i + 2
677 }
678 return 0
679}
680
681func _poly_tobytes12(bytes_out: *u8, poly: *u8) -> i64 {
682 var i: i64 = 0
683 while i < KYBER_N {
684 let c0: i64 = _canon_q(_u16_load_le(poly, i + 0)) & 0xfff
685 let c1: i64 = _canon_q(_u16_load_le(poly, i + 1)) & 0xfff
686 let off: i64 = (i >> 1) * 3
687 bytes_out[off + 0] = c0 & 0xff
688 bytes_out[off + 1] = ((c0 >> 8) | (c1 << 4)) & 0xff
689 bytes_out[off + 2] = (c1 >> 4) & 0xff
690 i = i + 2
691 }
692 return 0
693}
694
695func _poly_frombytes12(poly: *u8, bytes_in: *u8) -> i64 {
696 var i: i64 = 0
697 while i < KYBER_N {
698 let off: i64 = (i >> 1) * 3
699 let b0: i64 = bytes_in[off + 0]
700 let b1: i64 = bytes_in[off + 1]
701 let b2: i64 = bytes_in[off + 2]
702 let c0: i64 = b0 | ((b1 & 0x0f) << 8)
703 let c1: i64 = (b1 >> 4) | (b2 << 4)
704 _u16_store_le(poly, i + 0, c0 & 0xfff)
705 _u16_store_le(poly, i + 1, c1 & 0xfff)
706 i = i + 2
707 }
708 return 0
709}
710
711// ============================================================================
712// SECTION 8: CBD-eta2 + SampleNTT + msg codec
713// ============================================================================
714
715func _poly_cbd_eta2(out: *u8, buf: *u8) -> i64 {
716 var i: i64 = 0
717 while i < KYBER_N {
718 let byte_idx: i64 = i >> 1
719 let upper: i64 = i & 1
720 let nibble: i64 = (buf[byte_idx] >> (upper * 4)) & 0xf
721 let b0: i64 = nibble & 1
722 let b1: i64 = (nibble >> 1) & 1
723 let b2: i64 = (nibble >> 2) & 1
724 let b3: i64 = (nibble >> 3) & 1
725 _u16_store_le(out, i, (b0 + b1) - (b2 + b3))
726 i = i + 1
727 }
728 return 0
729}
730
731// Sample uniform NTT-form polynomial from SHAKE128 byte stream.
732// Returns 0 on success, -1 if buf exhausted.
733func _sample_ntt(poly: *u8, buf: *u8, buf_len: i64) -> i64 {
734 var i: i64 = 0
735 var pos: i64 = 0
736 while i < KYBER_N {
737 if pos + 3 > buf_len { return -1 }
738 let b0: i64 = buf[pos]
739 let b1: i64 = buf[pos + 1]
740 let b2: i64 = buf[pos + 2]
741 pos = pos + 3
742 let d1: i64 = b0 | ((b1 & 0x0f) << 8)
743 let d2: i64 = (b1 >> 4) | (b2 << 4)
744 if d1 < KYBER_Q {
745 _u16_store_le(poly, i, d1)
746 i = i + 1
747 }
748 if i < KYBER_N {
749 if d2 < KYBER_Q {
750 _u16_store_le(poly, i, d2)
751 i = i + 1
752 }
753 }
754 }
755 return 0
756}
757
758func _msg_to_poly(poly: *u8, msg: *u8) -> i64 {
759 var i: i64 = 0
760 while i < 32 {
761 let byte: i64 = msg[i]
762 var b: i64 = 0
763 while b < 8 {
764 let bit: i64 = (byte >> b) & 1
765 _u16_store_le(poly, i * 8 + b, bit * KYBER_HALF)
766 b = b + 1
767 }
768 i = i + 1
769 }
770 return 0
771}
772
773func _msg_from_poly(msg: *u8, poly: *u8) -> i64 {
774 var i: i64 = 0
775 while i < 32 {
776 var byte: i64 = 0
777 var b: i64 = 0
778 while b < 8 {
779 let c: i64 = _canon_q(_u16_load_le(poly, i * 8 + b))
780 let bit: i64 = (((c << 1) + KYBER_HALF) / KYBER_Q) & 1
781 byte = byte | (bit << b)
782 b = b + 1
783 }
784 msg[i] = byte & 0xff
785 i = i + 1
786 }
787 return 0
788}
789
790// ============================================================================
791// SECTION 9: Higher-level Kyber compose: matrix expansion, noise sampling
792// ============================================================================
793
794// Build matrix A entry (rho || x_byte || y_byte) -> 256-coef polynomial.
795// SHAKE128 squeeze 504 bytes, run rejection sampler. Retries with more
796// bytes if needed (up to 840 -- 5 rate blocks).
797// scratch >= 1300 bytes (240 keccak + 840 shake squeeze + slack).
798func _gen_matrix_entry(poly: *u8, rho: *u8, x: i64, y: i64,
799 keccak_state: *u8, shake_buf: *u8) -> i64 {
800 let seed_buf: *u8 = (shake_buf as i64 + 840) as *u8
801 var i: i64 = 0
802 while i < 32 { seed_buf[i] = rho[i]; i = i + 1 }
803 seed_buf[32] = x & 0xff
804 seed_buf[33] = y & 0xff
805
806 var attempt: i64 = 0
807 while attempt < 3 {
808 let try_len: i64 = 504 + attempt * 168
809 _shake128(seed_buf, 34, keccak_state, shake_buf, try_len)
810 let rc: i64 = _sample_ntt(poly, shake_buf, try_len)
811 if rc == 0 { return 0 }
812 attempt = attempt + 1
813 }
814 return -1
815}
816
817// CBD-eta2 noise polynomial from (sigma || nonce_byte) via SHAKE256.
818// scratch >= 240 keccak + 128 shake squeeze.
819func _gen_noise_eta2(poly: *u8, sigma: *u8, nonce: i64,
820 keccak_state: *u8, shake_buf: *u8) -> i64 {
821 let seed_buf: *u8 = (shake_buf as i64 + 128) as *u8
822 var i: i64 = 0
823 while i < 32 { seed_buf[i] = sigma[i]; i = i + 1 }
824 seed_buf[32] = nonce & 0xff
825 _shake256(seed_buf, 33, keccak_state, shake_buf, 128)
826 _poly_cbd_eta2(poly, shake_buf)
827 return 0
828}
829
830// Constant-time conditional-copy: if mask==0xff out=src1, if mask==0x00 out=src0.
831func _cmov(out: *u8, src1: *u8, len: i64, mask: i64) -> i64 {
832 let m: i64 = mask & 0xff
833 var i: i64 = 0
834 while i < len {
835 out[i] = (out[i] ^ ((out[i] ^ src1[i]) & m)) & 0xff
836 i = i + 1
837 }
838 return 0
839}
840
841// ============================================================================
842// SECTION 10: K-PKE keygen / encrypt / decrypt
843// ============================================================================
844
845// Scratch buffer layout for K-PKE keygen + encaps + decaps. Caller passes
846// one big buffer; we partition.
847//
848// IMPORTANT bug-class learned 2026-05-16: Keccak's _f1600 permutation uses
849// state_ptr[0..439] internally (200 A + 40 C + 200 B work). Any other
850// buffer that overlaps state_ptr+240..439 will be CORRUPTED whenever the
851// permutation runs. So all SHAKE squeeze-into-buffer destinations MUST
852// live PAST offset 440 from the Keccak state base. shake_buf below
853// starts at scratch+512 for safe clearance. Original layout put it at
854// scratch+240 -- silently corrupted every multi-block squeeze.
855//
856// Offset Size Purpose
857// 0 440 Keccak state + work area (do NOT overlap)
858// 440 72 padding
859// 512 1024 SHAKE squeeze buffer (matrix + noise expansion)
860// 1280 32 rho
861// 1312 32 sigma (G output second half)
862// 1344 64 G output (rho || sigma)
863// 1408 512 poly scratch A (per-iteration matrix entry)
864// 1920 1536 s vector polys (3 * 512)
865// 3456 1536 e vector polys
866// 4992 1536 t vector polys (output)
867// 6528 512 acc poly
868// 7040 1536 rVec polys (encrypt)
869// 8576 1536 e1 polys (encrypt)
870// 10112 512 e2 poly (encrypt)
871// 10624 512 mu poly (encrypt)
872// 11136 512 v poly (encrypt)
873// 11648 1536 u polys (encrypt)
874// 13184 1536 tHat polys (encrypt)
875// 14720 1536 sHat polys (decrypt)
876// 16256 1536 u' polys (decrypt)
877// 17792 512 vp poly (decrypt)
878// 18304 512 w poly (decrypt)
879// 18816 32 m' (decrypt output)
880// 18848 1088 ct' (decap re-encrypt verification)
881// 19936 32 z (rejection randomness)
882// 19968 32 K shared secret
883// 20000 1024 slack
884//
885// Total: ~21000 bytes. Caller supplies 32 KB minimum, 64 KB recommended.
886
887func _kpke_keygen(d_seed_32: *u8, scratch: *u8, ek_out: *u8, dk_pke_out: *u8) -> i64 {
888 let keccak_state: *u8 = scratch // scratch_range: keccak 0..439
889 let shake_buf: *u8 = (scratch as i64 + 512) as *u8 // scratch_range: shake_buf 512..KYBER_MAGIC_1535
890 let rho: *u8 = (scratch as i64 + KYBER_MAGIC_1536) as *u8 // scratch_range: rho KYBER_MAGIC_1536..KYBER_MAGIC_1567
891 let sigma: *u8 = (scratch as i64 + KYBER_MAGIC_1568) as *u8 // scratch_range: sigma KYBER_MAGIC_1568..KYBER_MAGIC_1599
892 let g_out: *u8 = (scratch as i64 + KYBER_MAGIC_1600) as *u8 // scratch_range: g_out_kg KYBER_MAGIC_1600..KYBER_MAGIC_1663
893 let A_entry: *u8 = (scratch as i64 + KYBER_MAGIC_1664) as *u8 // scratch_range: A_entry KYBER_MAGIC_1664..KYBER_MAGIC_2175
894 let s_vec: *u8 = (scratch as i64 + KYBER_MAGIC_2176) as *u8 // scratch_range: s_vec KYBER_MAGIC_2176..KYBER_MAGIC_3711
895 let e_vec: *u8 = (scratch as i64 + KYBER_MAGIC_3712) as *u8 // scratch_range: e_vec KYBER_MAGIC_3712..KYBER_MAGIC_5247
896 let t_vec: *u8 = (scratch as i64 + KYBER_MAGIC_5248) as *u8 // scratch_range: t_vec KYBER_MAGIC_5248..KYBER_MAGIC_6783
897 let acc: *u8 = (scratch as i64 + KYBER_MAGIC_6784) as *u8 // scratch_range: acc_kg KYBER_MAGIC_6784..KYBER_MAGIC_7295
898
899 // 1. (rho, sigma) = G(d || k=3). Reuse shake_buf as the (d||k) input scratch.
900 var i: i64 = 0
901 while i < 32 { shake_buf[i] = d_seed_32[i]; i = i + 1 }
902 shake_buf[32] = KYBER_K & 0xff
903 _sha3_512(shake_buf, 33, keccak_state, g_out)
904 var k: i64 = 0
905 while k < 32 { rho[k] = g_out[k]; sigma[k] = g_out[k + 32]; k = k + 1 }
906
907 // 2. Sample s_vec, e_vec via CBD eta_2 (eta_1 = eta_2 = 2 for ML-KEM-768)
908 var nonce: i64 = 0
909 var v: i64 = 0
910 while v < KYBER_K {
911 _gen_noise_eta2((s_vec as i64 + v * POLY_BUF) as *u8, sigma, nonce,
912 keccak_state, shake_buf)
913 nonce = nonce + 1
914 v = v + 1
915 }
916 v = 0
917 while v < KYBER_K {
918 _gen_noise_eta2((e_vec as i64 + v * POLY_BUF) as *u8, sigma, nonce,
919 keccak_state, shake_buf)
920 nonce = nonce + 1
921 v = v + 1
922 }
923
924 // 3. NTT(s), NTT(e)
925 v = 0
926 while v < KYBER_K { _ntt((s_vec as i64 + v * POLY_BUF) as *u8); v = v + 1 }
927 v = 0
928 while v < KYBER_K { _ntt((e_vec as i64 + v * POLY_BUF) as *u8); v = v + 1 }
929
930 // 4. t_vec[i] = sum_j A[i][j] * s[j] then tomont, then + e[i]
931 var rr: i64 = 0
932 while rr < KYBER_K {
933 _poly_zero(acc)
934 var jj: i64 = 0
935 while jj < KYBER_K {
936 _gen_matrix_entry(A_entry, rho, jj, rr, keccak_state, shake_buf)
937 _poly_basemul_acc(acc, A_entry, (s_vec as i64 + jj * POLY_BUF) as *u8)
938 jj = jj + 1
939 }
940 _poly_tomont(acc)
941 _poly_add((t_vec as i64 + rr * POLY_BUF) as *u8,
942 acc,
943 (e_vec as i64 + rr * POLY_BUF) as *u8)
944 rr = rr + 1
945 }
946
947 // 5. ek = ByteEncode_12(t̂) || rho
948 var ii: i64 = 0
949 while ii < KYBER_K {
950 _poly_tobytes12((ek_out as i64 + ii * POLY_BYTES_12) as *u8,
951 (t_vec as i64 + ii * POLY_BUF) as *u8)
952 ii = ii + 1
953 }
954 var rho_i: i64 = 0
955 while rho_i < 32 { ek_out[KYBER_K * POLY_BYTES_12 + rho_i] = rho[rho_i]; rho_i = rho_i + 1 }
956
957 // 6. dk_pke = ByteEncode_12(ŝ)
958 ii = 0
959 while ii < KYBER_K {
960 _poly_tobytes12((dk_pke_out as i64 + ii * POLY_BYTES_12) as *u8,
961 (s_vec as i64 + ii * POLY_BUF) as *u8)
962 ii = ii + 1
963 }
964 return 0
965}
966
967func _kpke_encrypt(ek: *u8, msg_32: *u8, coins_32: *u8,
968 scratch: *u8, ct_out: *u8) -> i64 {
969 let keccak_state: *u8 = scratch // 0..439
970 let shake_buf: *u8 = (scratch as i64 + 512) as *u8 // scratch_range: shake_buf 512..KYBER_MAGIC_1535
971 let A_entry: *u8 = (scratch as i64 + KYBER_MAGIC_1664) as *u8 // scratch_range: A_entry KYBER_MAGIC_1664..KYBER_MAGIC_2175
972 let acc: *u8 = (scratch as i64 + KYBER_MAGIC_6784) as *u8 // scratch_range: acc_en KYBER_MAGIC_6784..KYBER_MAGIC_7295
973 let rVec: *u8 = (scratch as i64 + KYBER_MAGIC_7296) as *u8 // scratch_range: rVec KYBER_MAGIC_7296..KYBER_MAGIC_8831
974 let e1: *u8 = (scratch as i64 + KYBER_MAGIC_8832) as *u8 // scratch_range: e1 KYBER_MAGIC_8832..KYBER_MAGIC_10367
975 let e2: *u8 = (scratch as i64 + KYBER_MAGIC_10368) as *u8 // scratch_range: e2 KYBER_MAGIC_10368..KYBER_MAGIC_10879
976 let mu: *u8 = (scratch as i64 + KYBER_MAGIC_10880) as *u8 // scratch_range: mu KYBER_MAGIC_10880..KYBER_MAGIC_11391
977 let v_poly: *u8 = (scratch as i64 + KYBER_MAGIC_11392) as *u8 // scratch_range: v_poly KYBER_MAGIC_11392..KYBER_MAGIC_11903
978 let u_vec: *u8 = (scratch as i64 + KYBER_MAGIC_11904) as *u8 // scratch_range: u_vec_en KYBER_MAGIC_11904..KYBER_MAGIC_13439
979 let tHat: *u8 = (scratch as i64 + KYBER_MAGIC_13440) as *u8 // scratch_range: tHat KYBER_MAGIC_13440..KYBER_MAGIC_14975
980
981 // 1. Parse ek -> tHat + rho
982 var ii: i64 = 0
983 while ii < KYBER_K {
984 _poly_frombytes12((tHat as i64 + ii * POLY_BUF) as *u8,
985 (ek as i64 + ii * POLY_BYTES_12) as *u8)
986 ii = ii + 1
987 }
988 let rho: *u8 = (ek as i64 + KYBER_K * POLY_BYTES_12) as *u8
989
990 // 2. Sample r, e1 (eta2), e2 (eta2)
991 var nonce: i64 = 0
992 var i: i64 = 0
993 while i < KYBER_K {
994 _gen_noise_eta2((rVec as i64 + i * POLY_BUF) as *u8, coins_32, nonce,
995 keccak_state, shake_buf)
996 nonce = nonce + 1
997 i = i + 1
998 }
999 i = 0
1000 while i < KYBER_K {
1001 _gen_noise_eta2((e1 as i64 + i * POLY_BUF) as *u8, coins_32, nonce,
1002 keccak_state, shake_buf)
1003 nonce = nonce + 1
1004 i = i + 1
1005 }
1006 _gen_noise_eta2(e2, coins_32, nonce, keccak_state, shake_buf)
1007
1008 // 3. r̂ = NTT(r)
1009 i = 0
1010 while i < KYBER_K { _ntt((rVec as i64 + i * POLY_BUF) as *u8); i = i + 1 }
1011
1012 // 4. u = INVNTT(A^T ∘ r̂) + e1
1013 var rr: i64 = 0
1014 while rr < KYBER_K {
1015 _poly_zero(acc)
1016 var jj: i64 = 0
1017 while jj < KYBER_K {
1018 // Transpose: A^T[i][j] uses rho || i || j (swapped order vs keygen)
1019 _gen_matrix_entry(A_entry, rho, rr, jj, keccak_state, shake_buf)
1020 _poly_basemul_acc(acc, A_entry, (rVec as i64 + jj * POLY_BUF) as *u8)
1021 jj = jj + 1
1022 }
1023 _invntt(acc)
1024 _poly_add((u_vec as i64 + rr * POLY_BUF) as *u8,
1025 acc,
1026 (e1 as i64 + rr * POLY_BUF) as *u8)
1027 rr = rr + 1
1028 }
1029
1030 // 5. v = INVNTT(tHat^T ∘ r̂) + e2 + Decompress_1(msg)
1031 _poly_zero(acc)
1032 var jj: i64 = 0
1033 while jj < KYBER_K {
1034 _poly_basemul_acc(acc,
1035 (tHat as i64 + jj * POLY_BUF) as *u8,
1036 (rVec as i64 + jj * POLY_BUF) as *u8)
1037 jj = jj + 1
1038 }
1039 _invntt(acc)
1040 _poly_add(v_poly, acc, e2)
1041 _msg_to_poly(mu, msg_32)
1042 _poly_add(v_poly, v_poly, mu)
1043
1044 // 6. c1 = ByteEncode_du(Compress_du(u)) (3 * 320 = 960 B)
1045 // c2 = ByteEncode_dv(Compress_dv(v)) (128 B)
1046 var k: i64 = 0
1047 while k < KYBER_K {
1048 _poly_compress10((ct_out as i64 + k * POLY_COMPR_U) as *u8,
1049 (u_vec as i64 + k * POLY_BUF) as *u8)
1050 k = k + 1
1051 }
1052 _poly_compress4((ct_out as i64 + KYBER_K * POLY_COMPR_U) as *u8, v_poly)
1053 return 0
1054}
1055
1056func _kpke_decrypt(dk_pke: *u8, ct: *u8, scratch: *u8, msg_out: *u8) -> i64 {
1057 let acc: *u8 = (scratch as i64 + KYBER_MAGIC_6784) as *u8 // scratch_range: acc_dc KYBER_MAGIC_6784..KYBER_MAGIC_7295
1058 let sHat: *u8 = (scratch as i64 + KYBER_MAGIC_14976) as *u8 // scratch_range: sHat KYBER_MAGIC_14976..KYBER_MAGIC_16511
1059 let u_vec: *u8 = (scratch as i64 + KYBER_MAGIC_16512) as *u8 // scratch_range: u_vec_dc KYBER_MAGIC_16512..KYBER_MAGIC_18047
1060 let vp: *u8 = (scratch as i64 + KYBER_MAGIC_18048) as *u8 // scratch_range: vp KYBER_MAGIC_18048..KYBER_MAGIC_18559
1061 let w: *u8 = (scratch as i64 + KYBER_MAGIC_18560) as *u8 // scratch_range: w KYBER_MAGIC_18560..KYBER_MAGIC_19071
1062
1063 // 1. Decompress c1 -> u; c2 -> vp
1064 var i: i64 = 0
1065 while i < KYBER_K {
1066 _poly_decompress10((u_vec as i64 + i * POLY_BUF) as *u8,
1067 (ct as i64 + i * POLY_COMPR_U) as *u8)
1068 i = i + 1
1069 }
1070 _poly_decompress4(vp, (ct as i64 + KYBER_K * POLY_COMPR_U) as *u8)
1071
1072 // 2. sHat from dk_pke
1073 i = 0
1074 while i < KYBER_K {
1075 _poly_frombytes12((sHat as i64 + i * POLY_BUF) as *u8,
1076 (dk_pke as i64 + i * POLY_BYTES_12) as *u8)
1077 i = i + 1
1078 }
1079
1080 // 3. NTT(u)
1081 i = 0
1082 while i < KYBER_K { _ntt((u_vec as i64 + i * POLY_BUF) as *u8); i = i + 1 }
1083
1084 // 4. acc = sHat^T ∘ NTT(u); INVNTT -> canonical poly
1085 _poly_zero(acc)
1086 var jj: i64 = 0
1087 while jj < KYBER_K {
1088 _poly_basemul_acc(acc,
1089 (sHat as i64 + jj * POLY_BUF) as *u8,
1090 (u_vec as i64 + jj * POLY_BUF) as *u8)
1091 jj = jj + 1
1092 }
1093 _invntt(acc)
1094
1095 // 5. w = vp - acc; msg = Compress_1(w)
1096 _poly_sub(w, vp, acc)
1097 _msg_from_poly(msg_out, w)
1098 return 0
1099}
1100
1101// ============================================================================
1102// SECTION 11: ML-KEM-768 keygen / encaps / decaps (FO transform)
1103// ============================================================================
1104
1105// Public ML-KEM-768 keygen. Takes 64-byte seed (d || z) and produces
1106// (ek_1184, dk_2400). Scratch >= 32 KB.
1107func nx_mlkem_keygen(seed_64: *u8, scratch: *u8, ek_out: *u8, dk_out: *u8) -> i64 {
1108 let keccak_state: *u8 = scratch
1109 let h_ek: *u8 = (scratch as i64 + KYBER_MAGIC_19072) as *u8 // 32 B inside scratch (past decrypt area)
1110
1111 // dk layout: dk_pke (1152) || ek (1184) || H(ek) (32) || z (32) = 2400 B
1112 let dk_pke: *u8 = dk_out
1113 let dk_ek: *u8 = (dk_out as i64 + KYBER_MAGIC_1152) as *u8
1114 let dk_h: *u8 = (dk_out as i64 + KYBER_MAGIC_1152 + KYBER_MAGIC_1184) as *u8
1115 let dk_z: *u8 = (dk_out as i64 + KYBER_MAGIC_1152 + KYBER_MAGIC_1184 + 32) as *u8
1116
1117 let d: *u8 = seed_64
1118 let z: *u8 = (seed_64 as i64 + 32) as *u8
1119
1120 _kpke_keygen(d, scratch, ek_out, dk_pke)
1121 // Copy ek into dk
1122 var i: i64 = 0
1123 while i < MLKEM_PK_BYTES { dk_ek[i] = ek_out[i]; i = i + 1 }
1124 // H(ek) -> dk_h
1125 _sha3_256(ek_out, MLKEM_PK_BYTES, keccak_state, dk_h)
1126 // z -> dk_z
1127 var j: i64 = 0
1128 while j < 32 { dk_z[j] = z[j]; j = j + 1 }
1129 return 0
1130}
1131
1132// Public ML-KEM-768 encaps. Takes 32-byte msg + ek, produces ct + ss.
1133// Scratch >= 32 KB.
1134func nx_mlkem_encaps(ek: *u8, msg_32: *u8, scratch: *u8,
1135 ct_out: *u8, ss_out: *u8) -> i64 {
1136 let keccak_state: *u8 = scratch
1137 let h_ek: *u8 = (scratch as i64 + KYBER_MAGIC_20224) as *u8 // 32 B
1138 let g_buf: *u8 = (scratch as i64 + KYBER_MAGIC_20096) as *u8 // 64 B (m || H(ek))
1139 let g_out: *u8 = (scratch as i64 + KYBER_MAGIC_20160) as *u8 // 64 B
1140
1141 // (K, r) = G(m || H(ek))
1142 _sha3_256(ek, MLKEM_PK_BYTES, keccak_state, h_ek)
1143 var i: i64 = 0
1144 while i < 32 { g_buf[i] = msg_32[i]; g_buf[32 + i] = h_ek[i]; i = i + 1 }
1145 _sha3_512(g_buf, 64, keccak_state, g_out)
1146 var k: i64 = 0
1147 while k < 32 { ss_out[k] = g_out[k]; k = k + 1 }
1148 let r: *u8 = (g_out as i64 + 32) as *u8
1149
1150 // ct = K-PKE.encrypt(ek, m, r)
1151 _kpke_encrypt(ek, msg_32, r, scratch, ct_out)
1152 return 0
1153}
1154
1155// Public ML-KEM-768 decaps. Returns 32-byte ss. Scratch >= 32 KB.
1156func nx_mlkem_decaps(dk: *u8, ct: *u8, scratch: *u8, ss_out: *u8) -> i64 {
1157 let keccak_state: *u8 = scratch
1158 let m_prime: *u8 = (scratch as i64 + KYBER_MAGIC_19072) as *u8 // scratch_range: m_prime KYBER_MAGIC_19072..KYBER_MAGIC_19103
1159 let ct_prime:*u8 = (scratch as i64 + KYBER_MAGIC_19104) as *u8 // scratch_range: ct_prime KYBER_MAGIC_19104..KYBER_MAGIC_20191
1160 // g_buf / g_out MUST live past ct_prime+1088 -- the decap encrypt step
1161 // writes ct_prime AFTER g_out is needed for K', so g_out must be
1162 // outside the ct_prime write range or its bytes get clobbered.
1163 let g_buf: *u8 = (scratch as i64 + KYBER_MAGIC_20192) as *u8 // scratch_range: g_buf_dc KYBER_MAGIC_20192..KYBER_MAGIC_20255
1164 let g_out: *u8 = (scratch as i64 + KYBER_MAGIC_20256) as *u8 // scratch_range: g_out_dc KYBER_MAGIC_20256..KYBER_MAGIC_20319
1165 let k_reject:*u8 = (scratch as i64 + KYBER_MAGIC_20320) as *u8 // scratch_range: k_reject KYBER_MAGIC_20320..KYBER_MAGIC_20351
1166 let z_ct: *u8 = (scratch as i64 + KYBER_MAGIC_22000) as *u8 // scratch_range: z_ct KYBER_MAGIC_22000..KYBER_MAGIC_23119
1167
1168 let dk_pke: *u8 = dk
1169 let dk_ek: *u8 = (dk as i64 + KYBER_MAGIC_1152) as *u8
1170 let dk_h: *u8 = (dk as i64 + KYBER_MAGIC_1152 + KYBER_MAGIC_1184) as *u8
1171 let dk_z: *u8 = (dk as i64 + KYBER_MAGIC_1152 + KYBER_MAGIC_1184 + 32) as *u8
1172
1173 // m' = K-PKE.decrypt(dk_pke, ct)
1174 _kpke_decrypt(dk_pke, ct, scratch, m_prime)
1175
1176 // (K', r') = G(m' || h)
1177 var i: i64 = 0
1178 while i < 32 { g_buf[i] = m_prime[i]; g_buf[32 + i] = dk_h[i]; i = i + 1 }
1179 _sha3_512(g_buf, 64, keccak_state, g_out)
1180
1181 // ct' = K-PKE.encrypt(ek, m', r')
1182 let r_prime: *u8 = (g_out as i64 + 32) as *u8
1183 _kpke_encrypt(dk_ek, m_prime, r_prime, scratch, ct_prime)
1184
1185 // diff = constant-time compare ct vs ct'
1186 var diff: i64 = 0
1187 var c: i64 = 0
1188 while c < MLKEM_CT_BYTES {
1189 diff = diff | (ct[c] ^ ct_prime[c])
1190 c = c + 1
1191 }
1192
1193 // K_reject = J(z || ct) = SHAKE256(z || ct, 32)
1194 var z_i: i64 = 0
1195 while z_i < 32 { z_ct[z_i] = dk_z[z_i]; z_i = z_i + 1 }
1196 var ct_i: i64 = 0
1197 while ct_i < MLKEM_CT_BYTES { z_ct[32 + ct_i] = ct[ct_i]; ct_i = ct_i + 1 }
1198 _shake256(z_ct, 32 + MLKEM_CT_BYTES, keccak_state, k_reject, 32)
1199
1200 // Constant-time select: mask = 0 if diff==0 (success) else 0xff
1201 var mask_acc: i64 = 0
1202 var bit_iter: i64 = 0
1203 while bit_iter < 8 {
1204 mask_acc = mask_acc | (diff >> bit_iter)
1205 bit_iter = bit_iter + 1
1206 }
1207 let mask: i64 = (mask_acc & 1) * 0xff // 0x00 or 0xff
1208
1209 // ss_out = K' if mask==0 else k_reject
1210 var k_i: i64 = 0
1211 while k_i < 32 {
1212 let kp: i64 = g_out[k_i]
1213 let kr: i64 = k_reject[k_i]
1214 ss_out[k_i] = ((kp & (~mask & 0xff)) | (kr & mask)) & 0xff
1215 k_i = k_i + 1
1216 }
1217 return 0
1218}
1219
1220// ============================================================================
1221// SECTION 12: Round-trip self-test (KAT inside the wasm)
1222// ============================================================================
1223
1224// Given a 64-byte seed and a 32-byte msg, run keygen+encaps+decaps and
1225// verify the shared secret is recovered byte-exact.
1226// Returns 0 on success, non-zero bitmap of failure axes:
1227// bit 0: keygen returned non-zero
1228// bit 1: encaps returned non-zero
1229// bit 2: decaps returned non-zero
1230// bit 3: shared secret mismatch
1231// scratch >= 64 KB (we use the bottom 32 KB for internals + reserve
1232// extra room for the keys/ct/ss output buffers at the top).
1233func nx_mlkem_round_trip_test(seed_64: *u8, msg_32: *u8, scratch: *u8) -> i64 {
1234 // Carve out output buffers at high addresses inside scratch.
1235 // The kpke / decap internal layout uses up to ~23120; we start outputs at 28000+.
1236 let ek: *u8 = (scratch as i64 + KYBER_MAGIC_28000) as *u8
1237 let dk: *u8 = (scratch as i64 + KYBER_MAGIC_30000) as *u8 // KYBER_MAGIC_2400 B -> ends at KYBER_MAGIC_32400
1238 let ct: *u8 = (scratch as i64 + KYBER_MAGIC_33000) as *u8 // KYBER_MAGIC_1088 B -> ends at KYBER_MAGIC_34088
1239 let ss1: *u8 = (scratch as i64 + KYBER_MAGIC_34200) as *u8 // 32 B
1240 let ss2: *u8 = (scratch as i64 + KYBER_MAGIC_34300) as *u8 // 32 B
1241
1242 var fail_mask: i64 = 0
1243 let rc1: i64 = nx_mlkem_keygen(seed_64, scratch, ek, dk)
1244 if rc1 != 0 { fail_mask = fail_mask | 1 }
1245 let rc2: i64 = nx_mlkem_encaps(ek, msg_32, scratch, ct, ss1)
1246 if rc2 != 0 { fail_mask = fail_mask | 2 }
1247 let rc3: i64 = nx_mlkem_decaps(dk, ct, scratch, ss2)
1248 if rc3 != 0 { fail_mask = fail_mask | 4 }
1249
1250 var diff: i64 = 0
1251 var i: i64 = 0
1252 while i < 32 {
1253 if ss1[i] != ss2[i] { diff = 1 }
1254 i = i + 1
1255 }
1256 if diff != 0 { fail_mask = fail_mask | 8 }
1257 return fail_mask
1258}