code wiki / _hdl_build / nx_intfp_mla_gradcheck_gate.nx

nx_intfp_mla_gradcheck_gate.nx

buildroot/runtime/_hdl_build/nx_intfp_mla_gradcheck_gate.nx

9318 B98 linesdepth 2pulls 2 transitivereach 0 importersview sourcekind gate/prooftopic intfp
docsdependenciesstructsconstsfunctions

about

nx_intfp_mla_gradcheck_gate.nx -- 2026 FRONTIER: MLA (Multi-head Latent Attention, DeepSeek's marquee KV- compression) in Q20 INTEGER, gradchecked, NO float. Instead of storing full K,V, x is DOWN-projected to a small latent c (dim LC<DM), and K,V are UP-projected from c -- so the KV cache is just c (LC-dim), a big compression. Full causal attention on the reconstructed K,V, full backward through the bottleneck. Gradcheck the DOWN-proj Wdkv (the compression novelty -- its gradient flows through BOTH the K and V up-projections and all of attention) plus Wuk, Wuv, Wq. license_tier: ORIGINAL

dependencies 1 imports · 0 importers

nx_syscalls.nx nx_intfp_mla_gradcheck_gate.nx

imports: nx_syscalls.nx

imported by: nobody (leaf or entry point)

call flow from main pre-order; caps 40 nodes / depth 6 declared; ↻ = already shown

main w sys_write sys_mmap mla_fwd fp_exp mla_bwd sys_mmap ↻ wn sys_write ↻ sys_mmap ↻ gcheck iabs mla_fwd ↻ w ↻ wn ↻

structs

none

consts

13const S: i64 = 16777216 // Q24 -- Q/K gradients flow thru softmax and are tiny; need the extra bits (as attention gate)
14const T: i64 = 3
15const DM: i64 = 4
16const LC: i64 = 2 // latent (compressed) dim < DM -- the KV-cache size
17const SCALE: i64 = 8388608 // 1/sqrt(4)=0.5

functions

9func w(s: *u8) -> i64 { var n: i64=0; while s[n]!=(0 as u8){n=n+1} sys_write(1,s,n); return 0 }
called by 2: gcheckmain calls 1: sys_write
10func wn(v: i64) -> i64 { if v==0 { sys_write(1,"0" as *u8,1); return 0 } var m: i64=v; if m<0{sys_write(1,"-" as *u8,1);m=0-m} let t: *u8=sys_mmap(24); var k: i64=0; while m>0{t[k]=(48+(m%10)) as u8;m=m/10;k=k+1} let o: *u8=sys_mmap(24); var q: i64=k-1; var i: i64=0; while q>=0{o[i]=t[q];i=i+1;q=q-1} sys_write(1,o,i); return 0 }
called by 2: gcheckmain calls 2: sys_writesys_mmap
11func iabs(v: i64) -> i64 { if v<0 { return 0-v } return v }
called by 1: gcheck
20func fp_exp(xq: i64) -> i64 { let y: i64=(xq*24204406)/S; var yi: i64=0; if y>=0 { yi=y/S } else { yi=0-(((0-y)+S-1)/S) } let yf: i64=y-yi*S; var p: i64=161380; p=931144+(p*yf)/S; p=4030770+(p*yf)/S; p=11632166+(p*yf)/S; p=S+(p*yf)/S; if yi>=0 { if yi>=31 { return 2000000000 } return p*(1<<yi) } let k: i64=0-yi; if k>=31 { return 0 } return p/(1<<k) }
called by 1: mla_fwd
23func mla_fwd(P: *i64) -> i64
called by 2: gcheckmain calls 1: fp_exp
35func mla_bwd(P: *i64) -> i64
called by 1: main calls 1: sys_mmap
58func gcheck(name: *u8, P: *i64, Wt: *i64, dW: *i64, ncell: i64) -> i64
called by 1: main calls 4: iabsmla_fwdwwn
75func main() -> i64