"""igc_gpu_frame.py -- M2: full frames rendered on the GPU. Two paths, both validated against the CPU golden renders: BARS -- the pure instruction-stream path: walk the captured DMA stream (conformance/sends_cap7.pkl), dispatch the M1 tile shader per tile with real origins, read texu, apply the SMPTE ramp. FX -- the primitive path: the fxtest per-draw packets (conformance/ fx_program.pkl) as primitive records (2 strip edges + z + u/v planes) rasterized full-frame in one dispatch. Mirrors render_fx.py (the op-0x2c strip semantics are winding-agnostic pending exact pinning -- documented in IGC-ENCODING-DERIVATION.md). """ import os, sys, struct, pickle import numpy as np sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) import igc_exec, igc_gpu HERE = os.path.dirname(os.path.abspath(__file__)) FIX = os.path.join(HERE, 'conformance') W, H = 832, 512 FX_SHADER = """ #version 430 layout(local_size_x = 64, local_size_y = 1) in; layout(std430, binding = 0) readonly buffer Prims { vec4 prims[]; }; // per primitive: 7 vec4 records: // [0..3] edges 0..3: A B C used (used=0 -> slot inactive) // [4] z.A z.B z.C nedges // [5] u.A u.B u.C hasuv // [6] v.A v.B v.C 0 layout(std430, binding = 1) buffer Outs { uint outbuf[]; }; // z(f32 as uint), r, g, b uniform int n_prims; uniform int width; void main() { int x = int(gl_GlobalInvocationID.x); int y = int(gl_GlobalInvocationID.y); if (x >= width) return; float fx = float(x), fy = float(y); float zbest = -1e30; vec3 col = vec3(0.0); for (int p = 0; p < n_prims; p++) { bool allpos = true, allneg = true; for (int e = 0; e < 4; e++) { vec4 re = prims[p*7+e]; if (re.w < 0.5) continue; float ve = re.x * fx + re.y * fy + re.z; if (ve >= 0.0) allneg = false; else allpos = false; } bool inside = allpos || allneg; if (!inside) continue; vec4 r2 = prims[p*7+4]; vec4 r3 = prims[p*7+5]; vec4 r4 = prims[p*7+6]; float z = r2.x * fx + r2.y * fy + r2.z; if (z <= zbest) continue; zbest = z; if (r3.w > 0.5) { float u = r3.x * fx + r3.y * fy + r3.z; float v = r4.x * fx + r4.y * fy + r4.z; col = vec3(u, v, 1.0); // raw plane values; normalized on host } else { col = vec3(-1.0, -1.0, 2.0); // flat marker } } int pix = y * width + x; outbuf[pix*4+0] = floatBitsToUint(zbest); outbuf[pix*4+1] = floatBitsToUint(col.x); outbuf[pix*4+2] = floatBitsToUint(col.y); outbuf[pix*4+3] = floatBitsToUint(col.z); } """ def f32(w): return struct.unpack('> 28) & 0xf) == 2}) texu = np.zeros((H, W), np.int32) ntiles = 0 for tid in tiles: ox = (tid & 0x1f) * 64 oy = ((tid >> 5) & 0x1f) * 128 if ox >= W or oy >= H: continue data = g.run(instrs, ox=ox, oy=oy, pre_seed_texz_x=True) for py in range(igc_gpu.TILE_H): gy = oy + py if gy >= H: break row = data[py * igc_gpu.TILE_W:(py + 1) * igc_gpu.TILE_W] for px in range(igc_gpu.TILE_W): gx = ox + px if gx >= W: break texu[gy, gx] = igc_gpu.GpuTile.rdbits(row[px], 57, 20) ntiles += 1 # SMPTE ramp PAL = [(180, 180, 180), (180, 180, 16), (16, 180, 180), (16, 180, 16), (180, 16, 180), (180, 16, 16), (16, 16, 180)] X0 = 48 rgb = np.zeros((H, W, 3), np.uint8) for gx in range(W): u = int(np.median(texu[:, gx])) rgb[:, gx] = (0, 0, 0) if u < X0 else PAL[min(6, (u - X0) * 7 // (W - X0))] return rgb, ntiles def render_fx_gpu(ctx): d = pickle.load(open(os.path.join(FIX, 'fx_program.pkl'), 'rb')) prog = d['prog'] def rd(a): return prog.get(a, 0) setups = [a for a in sorted(prog) if rd(a) == 0x100 and ((rd(a + 4) >> 8) & 0xff) == 0x2c] recs = [] for a in setups: edges = [] p = a + 4 while ((rd(p) >> 8) & 0xff) == 0x2c and len(edges) < 6: edges.append((f32(rd(p + 4)), f32(rd(p + 8)), f32(rd(p + 12)))) p += 16 zp = up = vp = None q = p for _ in range(40): w = rd(q); op = (w >> 8) & 0xff; ad = w & 0xff if op == 0x21 and zp is None: zp = (f32(rd(q + 4)), f32(rd(q + 8)), f32(rd(q + 12))) if op == 0x43 and ad == 58 and up is None: up = (f32(rd(q + 4)), f32(rd(q + 8)), f32(rd(q + 12))) if op == 0x43 and ad == 78 and vp is None: vp = (f32(rd(q + 4)), f32(rd(q + 8)), f32(rd(q + 12))) if zp and up and vp: break q += 4 if zp is None or len(edges) < 2: continue hasuv = 1.0 if (up and vp) else 0.0 up = up or (0, 0, 0); vp = vp or (0, 0, 0) eslots = [(e[0], e[1], e[2], 1.0) for e in edges[:4]] while len(eslots) < 4: eslots.append((0.0, 0.0, 0.0, 0.0)) recs += eslots + [(*zp, float(len(edges))), (*up, hasuv), (*vp, 0.0)] prims = np.array(recs, dtype=np.float32).reshape(-1, 4) n_prims = len(prims) // 7 prog_gl = ctx.compute_shader(FX_SHADER) b0 = ctx.buffer(prims.tobytes()) out = ctx.buffer(reserve=W * H * 16) b0.bind_to_storage_buffer(0) out.bind_to_storage_buffer(1) prog_gl['n_prims'] = n_prims prog_gl['width'] = W prog_gl.run(group_x=(W + 63) // 64, group_y=H) raw = np.frombuffer(out.read(), dtype=np.uint32).reshape(H, W, 4) z = raw[..., 0].view(np.float32) u = raw[..., 1].view(np.float32) v = raw[..., 2].view(np.float32) m = z > -1e29 rgb = np.zeros((H, W, 3), np.uint8) if m.any(): un = np.zeros_like(u); vn = np.zeros_like(v) un[m] = (u[m] - u[m].min()) / max(1e-9, np.ptp(u[m])) vn[m] = (v[m] - v[m].min()) / max(1e-9, np.ptp(v[m])) rgb[..., 0][m] = (60 + 180 * un[m]).astype(np.uint8) rgb[..., 1][m] = (40 + 160 * vn[m]).astype(np.uint8) rgb[..., 2][m] = (220 - 140 * un[m]).astype(np.uint8) b0.release(); out.release() return rgb, n_prims, int(m.sum()) if __name__ == '__main__': from PIL import Image g = igc_gpu.GpuTile() print("GPU:", g.ctx.info['GL_RENDERER']) bars, ntiles = render_bars_gpu(g) Image.fromarray(bars, 'RGB').save(os.path.join(HERE, 'gpu_bars.png')) # validation: 7 bars + black border, boundaries at the expected columns idx = [tuple(bars[10, x]) for x in (20, 100, 220, 340, 450, 560, 680, 800)] assert idx[0] == (0, 0, 0), idx[0] assert len(set(idx[1:])) == 7, idx print("M2 bars: PASS -- %d tiles, 7 distinct bars + border (gpu_bars.png)" % ntiles) fx, n_prims, covered = render_fx_gpu(g.ctx) Image.fromarray(fx, 'RGB').save(os.path.join(HERE, 'gpu_fx.png')) assert n_prims >= 9 and covered > 100000, (n_prims, covered) print("M2 fx: PASS -- %d prims, %d px covered (gpu_fx.png)" % (n_prims, covered)) print("M2 CONFORMANT: both fixture frames render on GPU")