From 0ab4409aec3b26bf70cb77f946d796ac651ec85f Mon Sep 17 00:00:00 2001 From: wakamex Date: Thu, 1 Oct 2026 01:27:43 -0400 Subject: [PATCH] A spin whose loop calls spins is one shared call on the devices when one function runs it twice Bend's GPU lanes run unrelated tasks, so a warp's threads only execute together at the same instruction. When one function runs two copies of a spin holding a loop in turn, a lane that finishes the first copy early moves on to the second while its neighbours are still in the first, and the warp runs the copies one after another. raytrace's shader, inlined four times in its pixel loop around the bounce loop, ran 2.1 of 32 threads per warp under ncu; one shared copy lifts that to 3.7 and cuts issued instructions 1.7x. So a spin is SHARE when it holds a loop that calls spins and some function runs two or more copies of it. Each segment records the spins it calls as emit_fuse emits them, and copies count through inlined spins and stop at shared or FAR ones, deciding callers first. SHARE is FAR on the devices and INLINE on the host, where clang decides (as host calls, raytrace's shared loops cost 22% on 16 threads). Each condition rests on a measured case: - copies in different functions run in different tasks, so sharing them buys no reconvergence: mandelbrot's escape wrapper, called from two functions, ran 32 of 32 threads either way and lost 1.6% to the call; - a loop that calls no spin is a few statements per step: sharing hashmap's bucket walk issued 10.7% more instructions at the same 7.5 of 32 active threads and lost 2.2%; - loop-free code rejoins after each branch: gameoflife runs 32 of 32 and a call there costs 5.6x the instructions. SPIN_FAR, FAR, term_drop and the meaning of a segment's spin flag are unchanged. comp.ts 63980 -> 64305 ttok, +29 -4 lines. Measured with bend-bench paired (fast-gpu-v2 sizes, RTX 3090, CUDA 13.1, 4GB heap, shared GPU) against a3f1782a: after a check and a warmup per side, 10 blocks of A B B A / B A A B, each scored (A1 + A2) / (B1 + B2), median and 95% bootstrap interval of the median: raytrace 1.844x (1.827x to 1.875x); editdist 1.145x (1.125x to 1.151x). The other 20 workloads compile to the same program as a3f1782a (the same GPU program and host code; the SHARE definitions they don't use only change the source copy the binary embeds), so they were not timed. bend-bench compiler-check against a3f1782a: every test keeps its status (691 pass, 11 wrong, 841 build-fail), and only raytrace, editdist, three demos and two tests get SHARE spins. Of the demos it touches, app_slash_boss_3d's 120-frame headless probe gave the same output and frames of 60.9 -> 60.2 ms on a loaded host (measured on 1adb0a61); pure_hvm5_mini runs in 12 ms, too short to time; app_ray_tracer_3d needs a display. Metal is unmeasured: no Apple hardware here. --- bend2/comp.ts | 33 +++++++++++++++++++++++++++++---- 1 file changed, 29 insertions(+), 4 deletions(-) diff --git a/bend2/comp.ts b/bend2/comp.ts index 74cd52bd2..83620b493 100644 --- a/bend2/comp.ts +++ b/bend2/comp.ts @@ -32,6 +32,7 @@ type Seg = { ks: Kind[]; frame: { pop: number; at: number[] } | null; refs: Set; + calls: string[]; spin?: boolean; fork?: boolean; }; @@ -99,9 +100,12 @@ type Fun = { n: number; h: HTerm | null; live: Dom[]; lays: Lay[]; ret: Lay }; // so that no file declares them. FOLD_FUEL caps the nodes that unfolds // add to a segment, so a literal-bounded loop does not unroll into its // caller. A spin of SPIN_FAR lines is a call (at 128, raytrace lost -// 31% on PAR-CPU). WIDE is the widest flat layout or segment; a node past -// it pads to its size class and keeps 240 plus log2 of it in CID_T. An -// argument nested past TPL_DEEP brackets goes to a local (clang allows 256). +// 31% on PAR-CPU). On the devices, a spin with a loop that calls spins is +// shared when one function runs it twice: a lane runs the copies in turn, +// and lanes in different copies cannot run together. WIDE is the widest +// flat layout or segment; a node past it pads to its size class and keeps +// 240 plus log2 of it in CID_T. An argument nested past TPL_DEEP brackets +// goes to a local (clang allows 256). const CLO_APPLY = "Clo~apply"; @@ -1489,7 +1493,7 @@ function spare_flush(fl: File): void { function seg_new(name: string, ret: Lay, params: string[], ks: Kind[] = params.map(() => "w64"), frame: Seg["frame"] = null): Seg { return { fid: seg_fid(name), def: name, ret, lines: [], params, ks, frame, - refs: new Set() }; + refs: new Set(), calls: [] }; } function seg_fid(k: Name): string { @@ -2137,6 +2141,7 @@ function emit_fuse(fl: File, ck: Spine, dst: Val | null, tail = false): void { } const out = emit_dst(fl, ret); const name = emit_native(fl, k, ers); + fl.seg.calls.push(name); const o = name_local(fl, "o"); file_push(fl, `Term ${o}[${out.ws.length}];`); const ks = lays.flatMap((l) => l.ks); @@ -2799,6 +2804,24 @@ export function compile_book(book: Bend.Book): string { const dev = reach([...[...fl.bangs].map(seg_fid), ...wide ? fl.clos : []]); fl.segs = fl.segs.filter((s) => live.has(s.fid)); fl.spins = fl.spins.filter((s) => live.has(s.fid)); + const deep = new Set(); + fl.spins.forEach((s) => fl.spins.some((t) => t !== s && s.refs.has(t.fid) + && (s.spin || deep.has(t.fid))) && deep.add(s.fid)); + const roots = new Map>(); + for (const s of [...fl.spins].reverse()) { + const n = new Map(); + for (const c of [...fl.segs, ...fl.spins]) { + const k = c.calls.filter((f) => f === s.fid).length; + for (const [r, m] of roots.get(c.fid) ?? [[c.fid, 1] as const]) { + n.set(r, (n.get(r) ?? 0) + k * m); + } + } + if (deep.has(s.fid) && Math.max(...n.values()) > 1) { + s.lines[0] = s.lines[0].replace(/^INLINE/, "SHARE"); + } + roots.set(s.fid, s.lines[0].startsWith("INLINE") ? n + : new Map([[s.fid, 1]])); + } const desc = show === null ? [] : ["#if !DEVICE", `static const u32 SHOW_DESC[] = { ${show.map((c) => typeof c === "string" ? cid_mac(c) : c).join(", ")} };`, @@ -3349,6 +3372,7 @@ using namespace metal; #define FAR static __attribute__((noinline)) #if DEVICE +#define SHARE FAR #define LOCK(l) #define UNLOCK(l) #define WL_CASE(F) case F: @@ -3356,6 +3380,7 @@ using namespace metal; #define WL_JMP(F) { fid = (F); break; } #define WL_DYN WL_JMP #else +#define SHARE INLINE #define LOCK(l) while (__atomic_exchange_n(&(l), 1, __ATOMIC_ACQUIRE)) {} #define UNLOCK(l) __atomic_store_n(&(l), 0, __ATOMIC_RELEASE) #define WL_FN static PRESERVE(preserve_none) __attribute__((noinline)) Term