code wiki / _hdl_build / _gpu_bm_compute_gate.nx
_gpu_bm_compute_gate.nx
buildroot/runtime/_hdl_build/_gpu_bm_compute_gate.nx
about
_gpu_bm_compute_gate.nx -- BARE-METAL SOVEREIGN-GPU rung BM-GPU-C1: compute-class dispatch + GEMM through the
sovereign GPFIFO pipeline (model side, toward BM-GPU-7).
Extends the FIFO model with engine-class binding: NVC56F_SET_OBJECT @0x0 (REAL FIFO method) binds a class id to
a subchannel; methods on a COMPUTE-bound subchannel route to a modeled compute engine. A sovereign driver fills
matrices A,B in modeled VRAM, emits a compute-dispatch pushbuffer (SET_OBJECT compute + matrix addrs/dims + LAUNCH),
and the modeled engine runs C = A*B, writing C to its GPU VA. The driver reads C and checks it against an
INDEPENDENT reference GEMM -> proves the compute COMMAND STREAM was decoded + dispatched correctly.
NISHI-ECOSYSTEM-ONLY: our model+driver+command-stream. HONEST SCOPE: SET_OBJECT/class-bind is spec-faithful; the
compute-class method OFFSETS here are REPRESENTATIVE (the production NVC6C0/Blackwell QMD + a real SASS GEMM kernel
is BM-GPU-7 on real silicon). The modeled engine STANDS IN for SASS execution. This gates the dispatch+result LOGIC.
GREEN iff: after the dispatch, C == independent reference GEMM (4x4 int); AND tampers (no SET_OBJECT class-bind /
no LAUNCH) leave C unwritten (dispatch is gated on class-bind + launch).
Marker -> knowledge/status/gpu_baremetal.log (BMCOMPUTEGATE). license_tier: ORIGINAL
dependencies 1 imports · 0 importers
imports: nx_syscalls.nx
imported by: nobody (leaf or entry point)
call flow from main pre-order; caps 40 nodes / depth 6 declared; ↻ = already shown
structs
| none |
consts
| 19 | const COMPUTE_CLASS: i64 = 0xCDC0 // Blackwell compute class (representative id) |
| 20 | const DIM: i64 = 4 // 4x4 matrices |
functions
| 22 | func p(s: *u8) -> i64 { var nn: i64=0; while s[nn]!=(0 as u8){nn=nn+1} sys_write(1,s,nn); return 0 } |
| 23 | func fp(fd: i64, s: *u8) -> i64 { var nn: i64=0; while s[nn]!=(0 as u8){nn=nn+1} sys_write(fd,s,nn); return 0 } |
| 24 | func n(v: i64) -> i64 { let bb: *u8=sys_mmap(28); var m: i64=v; if m<0{m=0-m;sys_write(1,"-" as *u8,1)}; let t: *u8=sys_mmap(28); var k: i64=0; if m==0{t[0]=48;k=1}; while m>0{t[k]=(48+(m%10)) as u8;m=m/10;k=k+1}; var i: i64=0; while i<k{bb[i]=t[k-1-i];i=i+1}; sys_write(1,bb,k); return 0 } |
| 25 | func x(v: i64) -> i64 { p("0x" as *u8); let bb:*u8=sys_mmap(20); var k:i64=0; var m:i64=v; if m==0{bb[0]=48;k=1}; while m>0{ let d:i64=m&15; if d<10{bb[k]=(48+d) as u8}else{bb[k]=(87+d) as u8}; m=(m>>4); k=k+1 } var i:i64=0; let o:*u8=sys_mmap(20); while i<k{o[i]=bb[k-1-i];i=i+1} sys_write(1,o,k); return 0 } |
| 26 | func fx(fd: i64, v: i64) -> i64 { let bb:*u8=sys_mmap(20); var k:i64=0; var m:i64=v; if m==0{bb[0]=48;k=1}; while m>0{ let d:i64=m&15; if d<10{bb[k]=(48+d) as u8}else{bb[k]=(87+d) as u8}; m=(m>>4); k=k+1 } let o:*u8=sys_mmap(20); var i:i64=0; while i<k{o[i]=bb[k-1-i];i=i+1} fp(fd,"0x" as *u8); sys_write(fd,o,k); return 0 } |
| 27 | func rd32(b: *u8, o: i64) -> i64 { return (b[o] as i64)|((b[o+1] as i64)<<8)|((b[o+2] as i64)<<16)|((b[o+3] as i64)<<24) } |
| 28 | func wr32(b: *u8, o: i64, v: i64) -> i64 { b[o]=(v&0xff) as u8; b[o+1]=((v>>8)&0xff) as u8; b[o+2]=((v>>16)&0xff) as u8; b[o+3]=((v>>24)&0xff) as u8; return 0 } |
| 33 | func gpfifo_run_compute(gmem: *u8, pb_off: i64, pb_dwords: i64, cs: *i64) -> i64 |
| 78 | func emit_hdr(gmem: *u8, off: i64, mode: i64, count: i64, subch: i64, method: i64) -> i64 |
| 82 | func main() -> i64 |