nx_tool_run.nx source
↩ module page · 688 lines · 35901 B
1// nx_tool_run.nx -- R0 of the EXECUTABLE-API rung: the sovereign exec+capture primitive the ecosystem
2// is missing. Today nishifamily.com/api/tools + /mcp only LIST tools (a read-only registry) and MCP
3// tools/call returns a SAFE STUB ("invoked X (capability-authorized)") -- there is NO way to actually
4// RUN an organ and hand back its real stdout. This organ is that missing capability, built hardware-up
5// from raw syscalls (fork/pipe/dup3/execve/wait4), reusing the EXACT proven wrappers from nx_hostctl's
6// hc_dfork_exec so it inherits the same never-brick discipline. NO /bin/sh, NO shell string, NO PATH
7// search -- callers pass an ABSOLUTE ELF path (the allowlist layer that maps tool-name -> path is R1,
8// nx_tool_registry). Synchronous (wait4), unlike hc_dfork_exec's detach -- because an API tools/call
9// needs the child's OUTPUT and EXIT CODE, not a fire-and-forget daemon.
10// license_tier: ORIGINAL
11import "nx_syscalls.nx"
12import "nx_buf_dyn.nx"
13
14// EINTR is -4. sys_read returns -errno, so a NEGATIVE result is an ERROR and r==0 alone is EOF.
15
16const TR_ERR_DRAIN: i64 = 0 - 8
17const TR_ERR_FD_SETUP: i64 = 0 - 12
18const TR_ERR_PARENT_LIFETIME: i64 = 0 - 11
19const TR_DRAIN_BYTES: i64 = 4096
20
21const TR_EINTR: i64 = 0 - 4
22// bounded so a genuinely unreadable fd cannot spin forever (same shape as sys_sleep_ms's guard)
23const TR_EINTR_MAX: i64 = 4096
24
25// ---- EXEC FAILURE MUST SPEAK (2026-08-28, 503 lane) ----------------------------------------------
26// WHAT THIS DELETES. Both capture primitives below used to end the child with a bare sys_exit(127)
27// when execve returned, writing NOTHING to the pipe -- and the header above records that as acceptable
28// (a bad path can only produce a 127 exit + empty capture). But the tools daemon's async lane writes
29// that empty capture to _jobs/job_<id>.out, and its own receipt tells every caller that an EMPTY
30// ARTIFACT MEANS THE JOB IS STILL RUNNING. So a tool that never started and a tool still working are
31// THE SAME OBSERVATION, forever: the caller either waits without bound or concludes the work landed.
32// A PROCESS THAT DIES BEFORE IT CAN SPEAK IS INDISTINGUISHABLE FROM ONE THAT IS STILL THINKING, AND
33// THE SILENCE IS READ AS THE MORE FLATTERING OF THE TWO.
34// MEASURED 2026-08-28 with a control pair: a 130000-byte single argument writes its file and returns
35// OK, while a 132000-byte one produced an empty artifact and no diagnostic anywhere. The CAUSE is not
36// ours -- Linux caps ONE argv element at MAX_ARG_STRLEN = 32 pages = 131072 bytes, a PER-ARGUMENT
37// limit no larger total-argv budget relaxes -- but the SILENCE was ours, and that is what this fixes.
38// The child is already past dup3 here, so fd 1 IS the capture pipe: the parent drains this text and the
39// caller reads a named cause instead of nothing. ONE message, TWO call sites, so they cannot drift.
40const TR_EXIT_EXECFAIL: i64 = 127
41const TR_MODE_0644: i64 = 420
42
43func tr_exec_failed(path: *u8) -> i64 {
44 let m1: *u8 = "NX-EXEC-FAILED rc=127 path=" as *u8
45 var n1: i64 = 0
46 while m1[n1] != (0 as u8) { n1 = n1 + 1 }
47 sys_write(1, m1, n1)
48 var pn: i64 = 0
49 while path[pn] != (0 as u8) { pn = pn + 1 }
50 sys_write(1, path, pn)
51 let m2: *u8 = " -- execve RETURNED instead of replacing this process, so the tool NEVER RAN. This capture is empty BY CAUSE, not because work is still in flight: do NOT read it as RUNNING. Two causes produce it here. (1) The binary is missing, not executable, or not an ELF -- check nx_catalog <name>, and nx_offc_install <name> promoted if a runner forks the _offc mirror. (2) A SINGLE argument exceeds the kernel MAX_ARG_STRLEN of 32 pages = 131072 bytes; that cap is PER-ARGUMENT and no total-size budget relaxes it -- split the payload, or push a large source as anchored edits rather than one whole-file argument. MEASURED 2026-08-28: a 130000-byte argument succeeds and a 132000-byte argument lands here.\n" as *u8
52 var n2: i64 = 0
53 while m2[n2] != (0 as u8) { n2 = n2 + 1 }
54 sys_write(1, m2, n2)
55 sys_exit(TR_EXIT_EXECFAIL)
56 return TR_EXIT_EXECFAIL
57}
58
59// tr_run_capture: fork -> child wires its stdout(+stderr) to a pipe and execve's `path` with `argv`
60// (a NUL-terminated *i64 array of *u8-as-i64, argv[0] conventionally = path) -> parent closes the write
61// end, drains the pipe into out[0..cap), wait4's the child, and returns wait_exit_code (0..255), or a
62// negative sentinel on a harness failure. On child execve failure the child exits 127 (captured as such).
63// *outlen (if non-null) receives the number of bytes captured.
64//
65// never-brick: path is an absolute ELF chosen by the caller's allowlist; a bad path can only produce a
66// 127 exit + empty capture, never a shell injection and never a write to persistent hardware state.
67// RUN A TOOL WHOSE STDOUT IS ITS PRODUCT, NOT ITS COMMENTARY.
68//
69// WHY THIS EXISTS BESIDE tr_run_capture RATHER THAN INSTEAD OF IT. Every capture helper in this lib
70// merges the child's stdout and stderr onto ONE pipe, which is exactly right when the output is a
71// report to be scanned: a diagnostic and a result belong in the same buffer and the caller reads
72// both. It is exactly WRONG when stdout carries an ARTIFACT. The sovereign compiler writes assembly
73// to stdout and progress to stderr, so a caller that captures it merged and writes the buffer to a
74// .s file produces a file with diagnostics glued to the front -- which the assembler then rejects
75// with an error about the SOURCE, sending the reader at the compiler instead of at the plumbing.
76// Measured 2026-09-04: that mistake cost a gate five failing teeth and read exactly like a broken
77// compiler while the compiler was correct.
78//
79// THE ESTATE ALREADY HAD THIS AND COULD NOT REACH IT. `sbr_run` inside nx_sov_build_run does fork +
80// redirect + execve correctly, but it lives inside a PROGRAM, so every other consumer must either
81// duplicate it or work around it -- the duplicate-ruler defect in its most common form. This is the
82// same mechanism lifted into the lib both consumers already import, so there is one of it.
83//
84// Contract: stdout -> out_path (created/truncated), stderr -> err_path when non-zero, else the
85// child's stderr is left on the parent's. Returns the child's exit status, or 128+signal if it died
86// to one -- because a SEGFAULTED tool whose status decodes as 0 is a silent fake success, which is
87// the defect the shell convention exists to prevent. Negative returns are the same named plumbing
88// failures tr_run_capture uses, so a caller can tell "the tool failed" from "I could not run it".
89func tr_run_redirect(path: *u8, argv: *i64, out_path: *u8, err_path: *u8) -> i64 {
90 let ofd: i64 = sys_openat_wr(out_path, TR_MODE_0644)
91 if ofd < 0 { return 0 - 5 } // TR_ERR_OUTOPEN
92 var efd: i64 = 0 - 1
93 if (err_path as i64) != 0 {
94 efd = sys_openat_wr(err_path, TR_MODE_0644)
95 if efd < 0 { sys_close(ofd); return 0 - 6 } // TR_ERR_ERROPEN
96 }
97 let pid: i64 = sys_fork()
98 if pid < 0 {
99 sys_close(ofd)
100 if efd >= 0 { sys_close(efd) }
101 return 0 - 3
102 }
103 if pid == 0 {
104 sys_dup3(ofd, 1, 0)
105 if efd >= 0 { sys_dup3(efd, 2, 0) }
106 sys_close(ofd)
107 if efd >= 0 { sys_close(efd) }
108 var fdc: i64 = 3
109 while fdc < 256 { sys_close(fdc); fdc = fdc + 1 }
110 let envp: *i64 = sys_mmap(16) as *i64
111 envp[0] = "PATH=/usr/bin:/bin" as *u8 as i64
112 envp[1] = 0
113 sys_execve_clean(path, argv, envp)
114 // execve failed. There is no pipe to name the cause on here, so exit with the shell's
115 // not-executable status and let the caller's own existence check say which file it was.
116 sys_exit(127)
117 return 0
118 }
119 sys_close(ofd)
120 if efd >= 0 { sys_close(efd) }
121 let stp: *i64 = sys_mmap(16) as *i64
122 let w: i64 = sys_wait4(pid, stp, 0)
123 if w < 0 { return 0 - 4 }
124 return wait_status_rc(stp[0])
125}
126
127func tr_run_capture(path: *u8, argv: *i64, out: *u8, cap: i64, outlen: *i64) -> i64 {
128 if (outlen as i64) != 0 { outlen[0] = 0 }
129 let fds: *i64 = sys_mmap(16) as *i64
130 if sys_pipe2(fds, 0) != 0 { return 0 - 2 } // TR_ERR_PIPE
131 // pipe2 writes int[2] (TWO 32-bit fds) into the first 8 bytes -> read end = low 32 bits of fds[0],
132 // write end = high 32 bits. Reading them as two i64 slots leaves the write fd un-tracked (never closed),
133 // so the reader never sees EOF and blocks forever. Unpack the 32-bit fds explicitly.
134 let packed: i64 = fds[0]
135 let rfd: i64 = packed & 0xFFFFFFFF
136 let wfd: i64 = (packed >> 32) & 0xFFFFFFFF
137
138 let pid: i64 = sys_fork()
139 if pid < 0 { sys_close(rfd); sys_close(wfd); return 0 - 3 } // TR_ERR_FORK
140 if pid == 0 {
141 // ---- CHILD ---- wire stdout(1) + stderr(2) to the pipe write end, close both raw ends, exec.
142 sys_dup3(wfd, 1, 0)
143 sys_dup3(wfd, 2, 0)
144 sys_close(rfd)
145 sys_close(wfd)
146 var fdc: i64 = 3
147 while fdc < 256 { sys_close(fdc); fdc = fdc + 1 }
148 let envp: *i64 = sys_mmap(16) as *i64
149 envp[0] = "PATH=/usr/bin:/bin" as *u8 as i64
150 envp[1] = 0
151 sys_execve_clean(path, argv, envp)
152 tr_exec_failed(path) // execve failed -> NAMED cause on the pipe, then 127
153 return 0
154 }
155
156 // ---- PARENT ---- close the write end (so read() sees EOF when the child exits), drain the pipe.
157 sys_close(wfd)
158 var total: i64 = 0
159 var run: i64 = 1
160 var eintr: i64 = 0
161 while run == 1 {
162 if total >= cap { run = 0 } else {
163 let r: i64 = sys_read(rfd, ((out as i64) + total) as *u8, cap - total)
164 // r < 0 is an ERROR, NOT EOF. `r <= 0` conflated them: EINTR -- overwhelmingly SIGCHLD from a
165 // reaped child, which THIS function creates -- ended the drain, returning a SHORT capture that
166 // outlen then reported as COMPLETE. That is how one organ's output arrives truncated at a
167 // DIFFERENT point every run and reads as a flaky gate (MEASURED 2026-08-08: 11 runs of one
168 // nx_coa_gate binary -> 4 different prefixes, ZERO FAIL lines). RESUME on EINTR exactly as
169 // sys_sleep_ms does for clock_nanosleep, whose comment already names this cause.
170 if r > 0 { total = total + r } else {
171 if r == 0 { run = 0 } else {
172 if r == TR_EINTR { if eintr > TR_EINTR_MAX { run = 0 } else { eintr = eintr + 1 } } else { run = 0 }
173 }
174 }
175 }
176 }
177 sys_close(rfd)
178 if (outlen as i64) != 0 { outlen[0] = total }
179
180 let stp: *i64 = sys_mmap(16) as *i64
181 let w: i64 = sys_wait4(pid, stp, 0)
182 if w < 0 { return 0 - 4 } // TR_ERR_WAIT
183 return wait_status_rc(stp[0])
184}
185
186// ---- BOUNDED EXEC (seq1412) ----------------------------------------------------------------
187// tr_run_capture has NO timeout: the parent blocks in the read() drain until EOF. A child that never
188// exits -- or that forks something holding stdout open -- hangs its caller forever. That primitive has
189// 51+ call sites including nx_seat (every session boot), nx_gate_rollup, and the tools-daemon exec path
190// (tea_run), so ONE hanging organ can wedge tools/call for every MCP client.
191//
192// ADDITIVE ON PURPOSE: tr_run_capture's signature and behaviour are untouched, so none of those 51
193// callers change. New/critical callers opt in here.
194//
195// Bounded capture shares a deadline-polled drain and an owned process-group
196// watchdog. Unbounded and promotable capture retain their existing contracts.
197const TR_ERR_TIMEOUT: i64 = 0 - 5
198const TR_SIGKILL: i64 = 9
199
200// timeout_ms <= 0 -> delegates to the unbounded tr_run_capture (explicit opt-out, never a silent one).
201// Returns the child's exit code, TR_ERR_TIMEOUT if the deadline fired, or the -2/-3/-4 harness sentinels.
202func tr_run_capture_to(path: *u8, argv: *i64, out: *u8, cap: i64, outlen: *i64, timeout_ms: i64) -> i64 {
203 return tr_run_capture_tr(path, argv, out, cap, outlen, timeout_ms, 0 as *i64)
204}
205
206// ---- BOUNDED, PROMOTABLE DRAIN (2026-08-22) -----------------------------------------------------
207// tr_run_capture_to's header above records that bounding the drain "would need O_NONBLOCK on the read
208// end, and there is no sys_fcntl in nx_syscalls -- so the non-blocking design is not buildable." That
209// premise is FALSE, and it cost the estate every synchronous-lane 503: poll(2) needs no O_NONBLOCK, and
210// sys_poll ships in nx_syscalls with 50 call sites (corpus_complete=1). A retrieval failure was written
211// into the source as an impossibility, and the watchdog+SIGKILL below it DESTROYED THE ANSWER of every
212// call that outran the window while the work itself ran on and landed.
213// ★A LAW RECORDED IN A HEADER IS STILL A HYPOTHESIS -- THIS ONE WAS REFUTED BY ONE GREP.
214// The promotable implementation remains separate. Bounded and cwd capture now
215// share tr_run_capture_core; callers keep their existing signatures.
216// * deadline_ms is a WHOLE-CALL budget, not a per-read idle timer -- the distinction the edge's
217// SO_RCVTIMEO gets wrong. remaining = deadline_ms - (now - t0), recomputed every pass.
218// * poll > 0 -> read (POLLIN is set, so it cannot block); the r>0 / r==0 / EINTR discipline is
219// copied from tr_run_capture_to verbatim, truncation-at-a-different-point bug included.
220// * poll == 0 -> THE DEADLINE. DO NOT KILL. The caller receives TR_PROMOTE plus the live worker pid and
221// the live read end, and now owns a worker that will finish and a pipe that will carry its
222// tail. No watchdog fork exists on this path: one FEWER process per sync call.
223// * deadline_ms <= 0 -> delegates to the unbounded tr_run_capture -- the explicit opt-out contract
224// tr_run_capture_to already has, never a silent one.
225// pollfd is the kernel ABI struct { i32 fd; i16 events; i16 revents } = 8 bytes.
226// ⚠INCUMBENT NAMED, NOT HIDDEN: nx_ts_drain_lib.nx carries tsd_pollfd_set / tsd_pollfd_ready for this
227// same struct. It is deliberately NOT imported here: nx_tool_run sits in the closure of 51 consumers
228// and that lib transitively imports nx_resmon_lib + nx_itoa_lib, so a symbol collision in any ONE of
229// those closures would break a build nobody in this lane can see. The consolidation is nx_oo_extract
230// lifting BOTH copies into a tiny nx_pollfd_lib -- that is the named remedy; this is the named debt.
231const TR_PROMOTE: i64 = 0 - 7 // distinct from TR_ERR_TIMEOUT (-5) and TR_ERR_CHDIR (-6)
232const TR_POLLFD_BYTES: i64 = 8 // sizeof(struct pollfd)
233const TR_POLLIN: i64 = 1 // POLLIN
234const TR_PF_EV_OFF: i64 = 4 // offsetof(struct pollfd, events)
235const TR_PF_RE_OFF: i64 = 6 // offsetof(struct pollfd, revents)
236const TR_PF_FD_BYTES: i64 = 4 // sizeof(i32 fd)
237const TR_BITS_PER_BYTE: i64 = 8
238const TR_BYTE_MASK: i64 = 0xff
239func tr_pollfd_set(p: *u8, fd: i64) -> i64 {
240 var k: i64 = 0
241 while k < TR_PF_FD_BYTES { p[k] = ((fd >> (k * TR_BITS_PER_BYTE)) & TR_BYTE_MASK) as u8; k = k + 1 }
242 p[TR_PF_EV_OFF] = TR_POLLIN as u8
243 p[TR_PF_EV_OFF + 1] = 0 as u8
244 p[TR_PF_RE_OFF] = 0 as u8
245 p[TR_PF_RE_OFF + 1] = 0 as u8
246 return 0
247}
248// Returns the child's exit code if it finished inside the deadline, TR_PROMOTE (with *out_pid / *out_rfd
249// filled) if the deadline fired first, or the -2/-3/-4 harness sentinels. On TR_PROMOTE the caller OWNS
250// the worker and the read end: it must drain rfd to EOF (the tail of the answer) or close it.
251func tr_run_capture_deadline(path: *u8, argv: *i64, out: *u8, cap: i64, outlen: *i64, deadline_ms: i64, out_pid: *i64, out_rfd: *i64) -> i64 {
252 if (out_pid as i64) != 0 { out_pid[0] = 0 }
253 if (out_rfd as i64) != 0 { out_rfd[0] = 0 - 1 }
254 if deadline_ms <= 0 { return tr_run_capture(path, argv, out, cap, outlen) }
255 if (outlen as i64) != 0 { outlen[0] = 0 }
256 let fds: *i64 = sys_mmap(16) as *i64
257 if sys_pipe2(fds, 0) != 0 { return 0 - 2 }
258 let packed: i64 = fds[0]
259 let rfd: i64 = packed & 0xFFFFFFFF
260 let wfd: i64 = (packed >> 32) & 0xFFFFFFFF
261 let t0: i64 = sys_now_ms()
262 let pid: i64 = sys_fork()
263 if pid < 0 { sys_close(rfd); sys_close(wfd); return 0 - 3 }
264 if pid == 0 {
265 sys_default_signal(13)
266 sys_dup3(wfd, 1, 0)
267 sys_dup3(wfd, 2, 0)
268 sys_close(rfd)
269 sys_close(wfd)
270 var fdc: i64 = 3
271 while fdc < 256 { sys_close(fdc); fdc = fdc + 1 }
272 let envp: *i64 = sys_mmap(16) as *i64
273 envp[0] = "PATH=/usr/bin:/bin" as *u8 as i64
274 envp[1] = 0
275 sys_execve_clean(path, argv, envp)
276 tr_exec_failed(path)
277 return 0
278 }
279 sys_close(wfd)
280 let pfd: *u8 = sys_mmap(TR_POLLFD_BYTES)
281 var total: i64 = 0
282 var run: i64 = 1
283 var eintr: i64 = 0
284 var promoted: i64 = 0
285 while run == 1 {
286 if total >= cap { run = 0 } else {
287 var remaining: i64 = deadline_ms - (sys_now_ms() - t0)
288 if remaining < 0 { remaining = 0 }
289 tr_pollfd_set(pfd, rfd)
290 let pr: i64 = sys_poll(pfd, 1, remaining)
291 if pr > 0 {
292 let r: i64 = sys_read(rfd, ((out as i64) + total) as *u8, cap - total)
293 if r > 0 { total = total + r } else {
294 if r == 0 { run = 0 } else {
295 if r == TR_EINTR { if eintr > TR_EINTR_MAX { run = 0 } else { eintr = eintr + 1 } } else { run = 0 }
296 }
297 }
298 } else {
299 if pr == 0 { promoted = 1; run = 0 } else {
300 if pr == TR_EINTR { if eintr > TR_EINTR_MAX { run = 0 } else { eintr = eintr + 1 } } else { run = 0 }
301 }
302 }
303 }
304 }
305 if (outlen as i64) != 0 { outlen[0] = total }
306 if promoted == 1 {
307 if (out_pid as i64) != 0 { out_pid[0] = pid }
308 if (out_rfd as i64) != 0 { out_rfd[0] = rfd }
309 return TR_PROMOTE
310 }
311 sys_close(rfd)
312 let stp: *i64 = sys_mmap(16) as *i64
313 let w: i64 = sys_wait4(pid, stp, 0)
314 if w < 0 { return 0 - 4 }
315 return wait_status_rc(stp[0])
316}
317
318// ---- SANDBOXED RUN (2026-08-06) ---------------------------------------------------------------
319// ***CONTAINMENT BEATS ENUMERATION.*** On 2026-08-06 a coverage sweep ran nx_cap_grant_e2e_gate -- a
320// destructive end-to-end test that exercises the REAL capability system in place. It rewrote
321// nishihost/tool_allowlist.conf with a 52-byte fixture (750 GREEN rows lost) and left
322// tools_cap_secret.key ABSENT, so nx_tools_api fell back to its forgeable placeholder and EVERY
323// capability in the estate was denied. A denylist of such gates was written afterwards, and a denylist
324// is an ENUMERATION -- it only ever protects against the instances somebody already thought of, and its
325// detector is a source-literal proxy, so it is a FLOOR not a total.
326// This is the containment: run the child with its OWN cwd, so a gate reaching for ../<production-file>
327// lands inside a scratch tree instead of the live one. It cannot stop an ABSOLUTE path -- nothing short
328// of a namespace can -- so it composes with the denylist rather than replacing it. Defence in depth,
329// stated honestly, because a containment that oversells itself is how the next one gets skipped.
330// ***FAIL-CLOSED: if the chdir does not take, the child EXITS rather than running in the wrong tree.***
331// Running the subject in the directory you were trying to protect is the exact failure this prevents,
332// so "could not chdir, so proceeded" must never be reachable.
333// ⚠<path> is resolved AFTER the chdir -- pass it absolute, or relative to <cwd> (e.g. ../_build/x.elf).
334const TR_ERR_CHDIR: i64 = 0 - 6
335func tr_run_capture_cwd(path: *u8, argv: *i64, out: *u8, cap: i64, outlen: *i64, timeout_ms: i64, cwd: *u8) -> i64 {
336 if (cwd as i64) == 0 { if (outlen as i64) != 0 { outlen[0]=0 }; return TR_ERR_CHDIR }
337 return tr_run_capture_core(path,argv,out,cap,outlen,timeout_ms,0 as *i64,cwd)
338}
339
340// Bounded capture retains the caller-sized prefix while draining excess bytes.
341// FIT/CUT is based on observed output, not whether the buffer filled exactly.
342const TR_FIT: i64 = 0
343const TR_CUT: i64 = 1
344func tr_drain_tr(rfd: i64, out: *u8, cap: i64, trunc: *i64) -> i64 {
345 if (trunc as i64) != 0 { trunc[0] = TR_FIT }
346 if cap < 0 { return TR_ERR_DRAIN }
347 let scratch: *u8 = sys_mmap(TR_DRAIN_BYTES)
348 var total: i64 = 0
349 var result: i64 = 0
350 var run: i64 = 1
351 var eintr: i64 = 0
352 while run == 1 {
353 var dest: *u8 = scratch
354 var room: i64 = TR_DRAIN_BYTES
355 if total < cap { dest = ((out as i64)+total) as *u8; room = cap-total }
356 let r: i64 = sys_read(rfd,dest,room)
357 if r > 0 {
358 eintr = 0
359 if total < cap { total = total+r } else {
360 if (trunc as i64) != 0 { trunc[0] = TR_CUT }
361 }
362 } else {
363 if r == 0 { run = 0 } else {
364 if r == TR_EINTR {
365 eintr = eintr+1
366 if eintr > TR_EINTR_MAX { result = TR_ERR_DRAIN; run = 0 }
367 } else { result = TR_ERR_DRAIN; run = 0 }
368 }
369 }
370 }
371 sys_munmap(scratch,TR_DRAIN_BYTES)
372 if result < 0 { return result }
373 return total
374}
375
376// tr_run_capture_tr: bounded capture that REPORTS whether it was cut off. trunc may be null.
377func tr_run_capture_tr(path: *u8, argv: *i64, out: *u8, cap: i64, outlen: *i64, timeout_ms: i64, trunc: *i64) -> i64 {
378 return tr_run_capture_core(path,argv,out,cap,outlen,timeout_ms,trunc,0 as *u8)
379}
380
381// Capture owns one private workspace and one shared control page per invocation.
382// The control page distinguishes a fired deadline from an unrelated SIGKILL.
383const TR_CONTROL_BYTES: i64 = 4096
384const TR_ERR_GROUP: i64 = 0 - 9
385const TR_ERR_ALLOC: i64 = 0 - 10
386const TR_ECHILD: i64 = 0 - 10 // Linux errno, distinct from the public result namespace
387func tr_clock_ms(ts: *i64) -> i64 {
388 if sys_clock_gettime_mono(ts) < 0 { return TR_ERR_DRAIN }
389 return ts[0]*1000 + ts[1]/SYS_MAGIC_1000000
390}
391// Observation keeps the PID reserved until every watchdog capable of signalling it is gone.
392func tr_observe_exit(pid: i64, info: *u8) -> i64 {
393 var rc: i64=sys_waitid(NX_WAIT_P_PID,pid,info,NX_WAIT_EXITED | NX_WAIT_NOWAIT)
394 while rc == TR_EINTR { rc=sys_waitid(NX_WAIT_P_PID,pid,info,NX_WAIT_EXITED | NX_WAIT_NOWAIT) }
395 return rc
396}
397func tr_reap(pid: i64, status: *i64) -> i64 {
398 var rc: i64=sys_wait4(pid,status,0)
399 while rc == TR_EINTR { rc=sys_wait4(pid,status,0) }
400 return rc
401}
402// A pipe can stay open after its direct producer exits. Poll against the whole-call
403// deadline; neither a full capture nor continuous output resets that deadline.
404// Fixed-buffer callers retain the original contract; owned capture shares the same drain and deadline.
405func tr_drain_until(rfd:i64,out:*u8,cap:i64,trunc:*i64,outlen:*i64,deadline:i64)->i64{
406 return tr_drain_until_owned(rfd,out,cap,trunc,outlen,deadline,0 as *NxBufOwned,0)
407}
408func tr_drain_until_owned(rfd:i64,out:*u8,cap:i64,trunc:*i64,outlen:*i64,deadline:i64,owned:*NxBufOwned,max_bytes:i64)->i64 {
409 let scratch: *u8=sys_mmap(TR_DRAIN_BYTES)
410 let state: *u8=sys_mmap(TR_CONTROL_BYTES)
411 let ts: *i64=((state as i64)+TR_POLLFD_BYTES) as *i64
412 var total: i64=0
413 var result: i64=0
414 var run: i64=1
415 var eintr: i64=0
416 while run == 1 {
417 var remaining: i64=0-1
418 if deadline > 0 {
419 let now: i64=tr_clock_ms(ts)
420 if now < 0 { result=TR_ERR_DRAIN; run=0 } else {
421 remaining=deadline-now
422 if remaining <= 0 { result=TR_ERR_TIMEOUT; run=0 }
423 }
424 }
425 if run == 1 {
426 tr_pollfd_set(state,rfd)
427 let ready: i64=sys_poll(state,1,remaining)
428 if ready > 0 {
429 var dest: *u8=scratch
430 var room: i64=TR_DRAIN_BYTES
431 if (owned as i64)==0 { if total < cap { dest=((out as i64)+total) as *u8; room=cap-total } }
432 let count: i64=sys_read(rfd,dest,room)
433 if count > 0 {
434 eintr=0
435 if (owned as i64)!=0 {
436 let appended:i64=nx_bo_append(owned,scratch,count,max_bytes)
437 if appended!=0 {result=appended;run=0}else{total=owned.len}
438 }else{
439 if total < cap { total=total+count } else {
440 if (trunc as i64) != 0 { trunc[0]=TR_CUT }
441 }
442 }
443 } else {
444 if count == 0 { run=0 } else {
445 if count == TR_EINTR { eintr=eintr+1 } else { result=TR_ERR_DRAIN; run=0 }
446 }
447 }
448 } else {
449 if ready == 0 { result=TR_ERR_TIMEOUT; run=0 } else {
450 if ready == TR_EINTR { eintr=eintr+1 } else { result=TR_ERR_DRAIN; run=0 }
451 }
452 }
453 if eintr > TR_EINTR_MAX { result=TR_ERR_DRAIN; run=0 }
454 }
455 }
456 if (outlen as i64) != 0 { outlen[0]=total }
457 sys_munmap(state,TR_CONTROL_BYTES)
458 sys_munmap(scratch,TR_DRAIN_BYTES)
459 return result
460}
461func tr_capture_stop(pid: i64, control: *i64) -> i64 {
462 // Only signal a group established by this invocation, never the caller's group.
463 if control[1] == 1 { nx_kill(0-pid,TR_SIGKILL) }
464 nx_kill(pid,TR_SIGKILL)
465 return 0
466}
467// The watchdog acknowledges lifetime binding before the parent relies on it.
468// Its startup pipe is created after the worker fork, so the worker cannot retain it.
469func tr_watchdog_child(owner: i64, pid: i64, control: *i64, deadline: i64, wfd: i64, scratch: *u8) -> i64 {
470 if sys_bind_parent_lifetime(owner,TR_SIGKILL) < 0 { control[2]=TR_ERR_PARENT_LIFETIME; return 126 }
471 if wfd != 1 { if sys_dup3(wfd,1,0) < 0 { control[2]=TR_ERR_FD_SETUP; return 126 } }
472 sys_close(0); sys_close(2)
473 if sys_close_inherited(3) < 0 { control[2]=TR_ERR_FD_SETUP; return 126 }
474 scratch[128]=82 as u8
475 if sys_write(1,scratch+128,1) != 1 { control[2]=TR_ERR_FD_SETUP; return 126 }
476 sys_close(1)
477 let ts: *i64=(scratch+160) as *i64
478 var running: i64=1
479 while running == 1 {
480 let now: i64=tr_clock_ms(ts)
481 if now < 0 { control[2]=TR_ERR_DRAIN; running=0 } else {
482 if now >= deadline { control[0]=1; running=0 } else {
483 if sys_sleep_ms(deadline-now) < 0 { control[2]=TR_ERR_DRAIN; running=0 }
484 }
485 }
486 }
487 tr_capture_stop(pid,control)
488 return 0
489}
490func tr_watchdog_start(owner: i64, pid: i64, control: *i64, deadline: i64) -> i64 {
491 let scratch: *u8=sys_mmap(TR_CONTROL_BYTES)
492 let fds: *i64=scratch as *i64
493 if sys_pipe2(fds,0) < 0 { sys_munmap(scratch,TR_CONTROL_BYTES); return 0-2 }
494 let rfd: i64=fds[0] & 0xFFFFFFFF
495 let wfd: i64=(fds[0] >> 32) & 0xFFFFFFFF
496 let wd: i64=sys_fork()
497 if wd == 0 {
498 let rc: i64=tr_watchdog_child(owner,pid,control,deadline,wfd,scratch)
499 sys_exit(rc); return rc
500 }
501 sys_close(wfd)
502 if wd < 0 { sys_close(rfd); sys_munmap(scratch,TR_CONTROL_BYTES); return 0-3 }
503 let length: *i64=(scratch+16) as *i64
504 let cut: *i64=(scratch+32) as *i64
505 let status: *i64=(scratch+64) as *i64
506 var result: i64=tr_drain_until(rfd,scratch+128,1,cut,length,deadline)
507 sys_close(rfd)
508 if result == 0 {
509 if length[0] != 1 || cut[0] != 0 || scratch[128] != 82 as u8 { result=TR_ERR_PARENT_LIFETIME }
510 }
511 if control[2] < 0 { result=control[2] }
512 if result < 0 { nx_kill(wd,TR_SIGKILL); tr_reap(wd,status) }
513 sys_munmap(scratch,TR_CONTROL_BYTES)
514 if result < 0 { return result }
515 return wd
516}
517func tr_run_capture_core(path:*u8,argv:*i64,out:*u8,cap:i64,outlen:*i64,timeout_ms:i64,trunc:*i64,cwd:*u8)->i64{
518 return tr_run_capture_core_owned(path,argv,out,cap,outlen,timeout_ms,trunc,cwd,0 as *NxBufOwned,0)
519}
520// The caller owns partial evidence even on error and releases it with nx_bo_release.
521// A new capture requires an empty owner; never re-execute a child to discover its output length.
522func tr_run_capture_owned(path:*u8,argv:*i64,owned:*NxBufOwned,timeout_ms:i64,max_bytes:i64,cwd:*u8)->i64{
523 if (owned as i64)<=0||max_bytes<0{return TR_ERR_DRAIN}
524 if (owned.buf as i64)!=0||owned.len!=0||owned.cap!=0{return TR_ERR_DRAIN}
525 var length:i64=0;var cut:i64=TR_FIT
526 return tr_run_capture_core_owned(path,argv,0 as *u8,0,&length,timeout_ms,&cut,cwd,owned,max_bytes)
527}
528
529// Typed evidence supplements the legacy flattened rc; no stdout parsing.
530struct NxRunEvidence { result:i64, capture_error:i64, setup_error:i64, exec_error:i64, wait_error:i64, raw_wait:i64, reaped:i64, exited:i64, exit_code:i64, signal:i64, timed_out:i64, drain_error:i64 }
531func tr_evidence_init(e:*NxRunEvidence)->i64{
532 e.result=TR_ERR_DRAIN;e.capture_error=TR_ERR_DRAIN;e.setup_error=0;e.exec_error=0;e.wait_error=0;e.raw_wait=-1;e.reaped=0;e.exited=0;e.exit_code=-1;e.signal=0;e.timed_out=0;e.drain_error=0;return 0
533}
534func tr_evidence_wait(e:*NxRunEvidence,control:*i64,pid:i64,waited:i64,status:i64)->i64{
535 if (e as i64)==0{return 0}
536 e.setup_error=control[2];e.exec_error=control[3];e.timed_out=control[0]
537 if waited<0{e.wait_error=waited}
538 if waited==pid{
539 e.reaped=1;e.raw_wait=status
540 let sig:i64=status&127
541 if sig==0{e.exited=1;e.exit_code=wait_exit_code(status)}
542 if sig>0{if sig<127{e.signal=sig}}
543 }
544 return 0
545}
546func tr_run_capture_owned_result(path:*u8,argv:*i64,owned:*NxBufOwned,timeout_ms:i64,max_bytes:i64,cwd:*u8,e:*NxRunEvidence)->i64{
547 tr_evidence_init(e)
548 if (owned as i64)<=0||max_bytes<0{return e.result}
549 if (owned.buf as i64)!=0||owned.len!=0||owned.cap!=0{return e.result}
550 var length:i64=0;var cut:i64=TR_FIT
551 let rc:i64=tr_run_capture_core_evidence(path,argv,0 as *u8,0,&length,timeout_ms,&cut,cwd,owned,max_bytes,e)
552 e.result=rc
553 if rc==TR_ERR_TIMEOUT{e.timed_out=1}
554 if rc<0{e.capture_error=rc}
555 if rc>=0{e.capture_error=0}
556 return rc
557}
558func tr_run_capture_core_owned(path:*u8,argv:*i64,out:*u8,cap:i64,outlen:*i64,timeout_ms:i64,trunc:*i64,cwd:*u8,owned:*NxBufOwned,max_bytes:i64)->i64 {
559 return tr_run_capture_core_evidence(path,argv,out,cap,outlen,timeout_ms,trunc,cwd,owned,max_bytes,0 as *NxRunEvidence)
560}
561func tr_capture_redirect(wfd:i64)->i64{
562 if wfd!=1{if sys_dup3(wfd,1,0)<0{return TR_ERR_FD_SETUP}}
563 if wfd!=2{if sys_dup3(wfd,2,0)<0{return TR_ERR_FD_SETUP}}
564 return 0
565}
566func tr_run_capture_core_evidence(path:*u8,argv:*i64,out:*u8,cap:i64,outlen:*i64,timeout_ms:i64,trunc:*i64,cwd:*u8,owned:*NxBufOwned,max_bytes:i64,e:*NxRunEvidence)->i64 {
567 if (outlen as i64) != 0 { outlen[0]=0 }
568 if (trunc as i64) != 0 { trunc[0]=TR_FIT }
569 if cap < 0 { return TR_ERR_DRAIN }
570 let control: *i64=sys_mmap_shared(TR_CONTROL_BYTES) as *i64
571 if (control as i64) <= 0 { return TR_ERR_ALLOC }
572 let fds: *i64=((control as i64)+32) as *i64
573 let stp: *i64=((control as i64)+48) as *i64
574 let wstp: *i64=((control as i64)+64) as *i64
575 let ts: *i64=((control as i64)+80) as *i64
576 var deadline: i64=0
577 if timeout_ms > 0 {
578 let now: i64=tr_clock_ms(ts)
579 if now < 0 { sys_munmap(control as *u8,TR_CONTROL_BYTES); return TR_ERR_DRAIN }
580 deadline=now+timeout_ms
581 if deadline < now { sys_munmap(control as *u8,TR_CONTROL_BYTES); return TR_ERR_DRAIN }
582 }
583 let owner: i64=__syscall(172,0,0,0,0,0,0)
584 if owner <= 0 { sys_munmap(control as *u8,TR_CONTROL_BYTES); return TR_ERR_PARENT_LIFETIME }
585 if sys_pipe2(fds,0) != 0 { sys_munmap(control as *u8,TR_CONTROL_BYTES); return 0-2 }
586 let packed: i64=fds[0]
587 let rfd: i64=packed & 0xFFFFFFFF
588 let wfd: i64=(packed >> 32) & 0xFFFFFFFF
589 let pid: i64=sys_fork()
590 if pid < 0 {
591 sys_close(rfd); sys_close(wfd)
592 sys_munmap(control as *u8,TR_CONTROL_BYTES)
593 return 0-3
594 }
595 if pid == 0 {
596 sys_default_signal(13)
597 if sys_bind_parent_lifetime(owner,TR_SIGKILL) < 0 { control[2]=TR_ERR_PARENT_LIFETIME; sys_exit(126); return 0 }
598 if sys_setpgid(0,0) != 0 { control[2]=TR_ERR_GROUP; sys_exit(126); return 0 }
599 control[1]=1
600 // Close the read end first: it may occupy stdout/stderr when inherited fds are closed.
601 sys_close(rfd)
602 let redirected:i64=tr_capture_redirect(wfd)
603 if redirected<0{control[2]=redirected;sys_exit(126);return 0}
604 if wfd > 2 { sys_close(wfd) }
605 let envp: *i64=((control as i64)+96) as *i64
606 envp[0]="PATH=/usr/bin:/bin" as *u8 as i64; envp[1]=0
607 // Setup failure travels independently of the program's legitimate exit status.
608 if (cwd as i64) != 0 {
609 if sys_chdir(cwd) != 0 { control[2]=TR_ERR_CHDIR; sys_exit(126); return 0 }
610 }
611 // control[3] is the reserved word before the fd block at byte32.
612 let closed:i64=sys_close_inherited(3)
613 if closed<0{control[2]=closed;sys_exit(126);return 0}
614 let executed:i64=sys_execve(path,argv,envp)
615 control[3]=executed
616 tr_exec_failed(path)
617 return 0
618 }
619 sys_close(wfd)
620 var wd: i64=0
621 if timeout_ms > 0 {
622 wd=tr_watchdog_start(owner,pid,control,deadline)
623 if wd < 0 {
624 sys_close(rfd); tr_capture_stop(pid,control)
625 let early_wait:i64=tr_reap(pid,stp)
626 tr_evidence_wait(e,control,pid,early_wait,stp[0])
627 sys_munmap(control as *u8,TR_CONTROL_BYTES)
628 return wd
629 }
630 }
631 let drained: i64=tr_drain_until_owned(rfd,out,cap,trunc,outlen,deadline,owned,max_bytes)
632 sys_close(rfd)
633 if drained < 0 { tr_capture_stop(pid,control) }
634 let info: *u8=((control as i64)+128) as *u8
635 let observed: i64=tr_observe_exit(pid,info)
636 var watchwait: i64=0
637 if wd > 0 { nx_kill(wd,TR_SIGKILL); watchwait=tr_reap(wd,wstp) }
638 // ECHILD means ownership is already absent: never signal a potentially reused PID.
639 if observed < 0 { if observed != TR_ECHILD { tr_capture_stop(pid,control) } }
640 let waited: i64=tr_reap(pid,stp)
641 var result: i64=wait_status_rc(stp[0])
642 if observed < 0 || waited < 0 || watchwait < 0 { result=0-4 }
643 if control[2] < 0 { result=control[2] }
644 if control[0] == 1 { result=TR_ERR_TIMEOUT }
645 if drained < 0 { result=drained }
646 tr_evidence_wait(e,control,pid,waited,stp[0])
647 if (e as i64)!=0{
648 if drained<0{e.drain_error=drained}
649 if observed<0{e.wait_error=observed}
650 if watchwait<0{e.wait_error=watchwait}
651 }
652 sys_munmap(control as *u8,TR_CONTROL_BYTES)
653 return result
654}
655
656// tr_run1: convenience for the common "run ELF with a single string arg" case. Builds argv = [path, arg, 0].
657// arg may be null -> argv = [path, 0].
658func tr_run1(path: *u8, arg: *u8, out: *u8, cap: i64, outlen: *i64) -> i64 {
659 let argv: *i64 = sys_mmap(32) as *i64
660 argv[0] = path as i64
661 if (arg as i64) == 0 { argv[1] = 0 } else { argv[1] = arg as i64; argv[2] = 0 }
662 return tr_run_capture(path, argv, out, cap, outlen)
663}
664
665// tr_run1_to: bounded twin of tr_run1. Same argv shaping, with a deadline.
666func tr_run1_to(path: *u8, arg: *u8, out: *u8, cap: i64, outlen: *i64, timeout_ms: i64) -> i64 {
667 let argv: *i64 = sys_mmap(32) as *i64
668 argv[0] = path as i64
669 if (arg as i64) == 0 { argv[1] = 0 } else { argv[1] = arg as i64; argv[2] = 0 }
670 return tr_run_capture_to(path, argv, out, cap, outlen, timeout_ms)
671}
672
673// tr_contains: 1 if the NUL-terminated needle occurs in buf[0..n), else 0. For gates asserting on captured stdout.
674func tr_contains(buf: *u8, n: i64, needle: *u8) -> i64 {
675 var nl: i64 = 0
676 while needle[nl] != (0 as u8) { nl = nl + 1 }
677 if nl == 0 { return 1 }
678 if n < nl { return 0 }
679 var i: i64 = 0
680 while i <= n - nl {
681 var m: i64 = 1
682 var c: i64 = 0
683 while c < nl { if buf[i + c] != needle[c] { m = 0; c = nl } else { c = c + 1 } }
684 if m == 1 { return 1 }
685 i = i + 1
686 }
687 return 0
688}