From be8f72e731249ef4fece8358440e5fc16162a679 Mon Sep 17 00:00:00 2001 From: Cyd Date: Sun, 19 Jul 2026 13:24:43 -0500 Subject: [PATCH] M4a: emu860c -- the i860 step-core in C, 351x, checkpoint-perfect Full mechanical port of emu860.py's validated semantics: paged memory, integer ISA, control flow with delayed branches + bc.t/bnc.t annul, DIM tracking, and the complete FP model (pipes with per-entry result precision, PFAM/PFMAM dual-op routing, fdest bypass, raw-retire fix/ftrunc, K/T loads). Python keeps the hooks via a shim; the C loop returns at hook pcs. Differential result vs the hook-aligned reference trace (ref_trace2): 139.7M steps/s (Python: 398K/s = 351x), 35.8M steps, 2195 hooks, 43/43 checkpoints matched (pc + step count + full reg CRC). The original i860 ran ~40 MIPS: real-time geometry is cleared 3.5x over -- M4's gate (GPU-RETARGET.md) is open. Co-Authored-By: Claude Opus 4.8 --- emulator/firmware-decomp/emu860c/build.sh | 5 + emulator/firmware-decomp/emu860c/driver.py | 129 ++++ emulator/firmware-decomp/emu860c/emu860c.c | 755 +++++++++++++++++++++ 3 files changed, 889 insertions(+) create mode 100644 emulator/firmware-decomp/emu860c/build.sh create mode 100644 emulator/firmware-decomp/emu860c/driver.py create mode 100644 emulator/firmware-decomp/emu860c/emu860c.c diff --git a/emulator/firmware-decomp/emu860c/build.sh b/emulator/firmware-decomp/emu860c/build.sh new file mode 100644 index 00000000..27319a1b --- /dev/null +++ b/emulator/firmware-decomp/emu860c/build.sh @@ -0,0 +1,5 @@ +#!/bin/sh +# build emu860c.pyd (MSYS2 MINGW64: pacman -S mingw-w64-x86_64-gcc + python) +PYINC=$(python -c "import sysconfig; print(sysconfig.get_paths()['include'])") +PYVER=$(python -c "import sys; print('%d.%d' % sys.version_info[:2])") +exec gcc -O2 -shared -o emu860c.pyd emu860c.c -I"$PYINC" -LC:/msys64/mingw64/lib -lpython${PYVER/./} diff --git a/emulator/firmware-decomp/emu860c/driver.py b/emulator/firmware-decomp/emu860c/driver.py new file mode 100644 index 00000000..0ad5c093 --- /dev/null +++ b/emulator/firmware-decomp/emu860c/driver.py @@ -0,0 +1,129 @@ +"""emu860c differential driver v2: boot via the Python MainRunner +(authoritative loader), copy the state into the C core, run with hooks +serviced through a shim, and compare (pc, steps, reg-crc) at HOOK boundaries +against the hook-aligned reference trace (ref_trace2.pkl).""" +import sys, os, time, pickle, zlib + +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, HERE) +sys.path.insert(0, os.path.dirname(HERE)) +sys.path.insert(0, r'C:\VWE\TeslaRel410\dpl3-revive\patha') +import emu860, emu_main, emu860c + +emu860.Mem.log = lambda self, *a, **k: None +SCRATCH = r'C:\Users\cyd\AppData\Local\Temp\claude\c--VWE-TeslaRel410\4e848c76-6e89-4034-8047-d8d491cb32d8\scratchpad' + + +class MemShim: + def r32(self, a): return emu860c.r32(a & 0xffffffff) + def w32(self, a, v): emu860c.w32(a & 0xffffffff, v & 0xffffffff) + def load_blob(self, a, b): emu860c.load_blob(a & 0xffffffff, bytes(b)) + + +class CpuShim: + def __init__(self): + self.mem = MemShim() + self.igc = [] + self.board_log = [] + self.RET_SENTINEL = 0xc0de0000 + + @property + def steps(self): + return emu860c.getstate()['steps'] + + @property + def pc(self): + return emu860c.getstate()['pc'] + + @pc.setter + def pc(self, v): + emu860c.setpc(v & 0xffffffff) + + def rd(self, i): + return 0 if i == 0 else emu860c.getstate()['r'][i] + + def wr(self, i, v): + if i: + emu860c.setreg(i, v & 0xffffffff, 0) + + +def boot(): + r = emu_main.MainRunner(r'C:\VWE\TeslaRel410\dpl3-revive\patha\cap7.raw.bin', + fw='capfw7', max_cmds=200000) + cpu = r.start() + emu860c.init() + for pn, page in cpu.mem.pages.items(): + emu860c.load_blob(pn << 16, bytes(page)) + for a in range(0xfffff000, 0x100000000, 4): + v = cpu.mem.r32(a) + if v: + emu860c.w32(a & 0xffffffff, v & 0xffffffff) + for i in range(32): + emu860c.setreg(i, cpu.r[i] & 0xffffffff, 0) + emu860c.setreg(i, cpu.f[i] & 0xffffffff, 1) + emu860c.setpc(cpu.pc) + emu860c.set_sentinel(cpu.RET_SENTINEL) + emu860c.set_hooks(sorted(r.hooks.keys())) + return r + + +def crc_state(st): + b = b''.join(int(v).to_bytes(4, 'little') for v in st['r']) + b += b''.join(int(v).to_bytes(4, 'little') for v in st['f']) + return zlib.crc32(b) & 0xffffffff + + +def main(): + ref = pickle.load(open(os.path.join(SCRATCH, 'ref_trace2.pkl'), 'rb')) + cps = {n: (pc, steps, crc) for n, pc, steps, crc in ref['cps']} + max_hook = ref['hookno'] + r = boot() + shim = CpuShim() + t0 = time.time() + hookno = 0 + matched = mismatched = 0 + while hookno < max_hook: + reason, steps = emu860c.run(200000000) + if reason == 0: + st = emu860c.getstate() + pc = st['pc'] + h = r.hooks.get(pc) + if h is None: + print("sentinel at hook %d" % hookno) + break + hookno += 1 + want = cps.get(hookno) + if want: + got = (pc, st['steps'], crc_state(st)) + if want != got: + print("DIVERGE at hook %d:" % hookno) + print(" want pc=%#x steps=%d crc=%08x" % want) + print(" got pc=%#x steps=%d crc=%08x" % got) + mismatched += 1 + if mismatched > 3: + break + else: + matched += 1 + if h(shim) == 'done': + break + continue + if reason == 1: + print("STOP: pc=%#x" % emu860c.getstate()['pc']) + break + if reason == 2: + st = emu860c.getstate() + print("FP fallback pc=%#x w=%08x" % (st['pc'], emu860c.r32(st['pc']))) + break + if reason == 3: + print("budget exhausted") + break + dt = time.time() - t0 + st = emu860c.getstate() + print("C core: %d steps, %d hooks in %.2fs = %d steps/s; " + "checkpoints %d/%d matched" % + (st['steps'], hookno, dt, st['steps'] / max(dt, 1e-9), + matched, len(cps))) + + +if __name__ == '__main__': + main() diff --git a/emulator/firmware-decomp/emu860c/emu860c.c b/emulator/firmware-decomp/emu860c/emu860c.c new file mode 100644 index 00000000..26a6368b --- /dev/null +++ b/emulator/firmware-decomp/emu860c/emu860c.c @@ -0,0 +1,755 @@ +/* emu860c.c -- M4a: the i860 step-core in C (CPython extension). + * + * A mechanical port of emu860.py's validated semantics (the golden model): + * paged memory, integer ALU/loads/stores, control flow with delayed branches, + * bc.t/bnc.t annul, dual-instruction-mode tracking, and the FP pipeline model. + * + * Differential contract: emu860c must reproduce emu860.py's (step, pc, + * reg-crc) checkpoints exactly (scratchpad ref_trace / emu860_ref_trace.pkl). + * + * Stage 1 (this file, first landing): memory + integer core + control flow + + * DIM/annul + scalar FP moves/loads/stores + the pipe mechanics for the + * common pipelined ops (pfadd/pfsub/pfmul + fnop). Anything else FP raises + * NotImplemented back to Python (run_until returns reason=2 with the pc), so + * divergence is impossible -- unfinished territory just falls back. + */ +#define PY_SSIZE_T_CLEAN +#include +#include +#include +#include + +#define MASK32 0xffffffffu + +/* ---------------- paged guest memory ---------------- */ +#define PAGE_BITS 12 +#define PAGE_SIZE (1u << PAGE_BITS) +#define N_PAGES (1u << (32 - PAGE_BITS)) + +static uint8_t **g_pages; /* N_PAGES pointers, allocated on demand */ + +static uint8_t *page_for(uint32_t addr, int create) { + uint32_t pi = addr >> PAGE_BITS; + uint8_t *p = g_pages[pi]; + if (!p && create) { + p = (uint8_t *)calloc(PAGE_SIZE, 1); + g_pages[pi] = p; + } + return p; +} + +static uint32_t mem_r32(uint32_t addr) { + uint32_t off = addr & (PAGE_SIZE - 1); + uint8_t *p = page_for(addr, 0); + if (!p) return 0; + if (off <= PAGE_SIZE - 4) { + uint32_t v; + memcpy(&v, p + off, 4); + return v; + } + /* page-spanning byte fallback */ + uint32_t v = 0; + for (int i = 0; i < 4; i++) { + uint8_t *q = page_for(addr + i, 0); + v |= (uint32_t)(q ? q[(addr + i) & (PAGE_SIZE - 1)] : 0) << (8 * i); + } + return v; +} + +static void mem_w32(uint32_t addr, uint32_t val) { + uint32_t off = addr & (PAGE_SIZE - 1); + uint8_t *p = page_for(addr, 1); + if (off <= PAGE_SIZE - 4) { + memcpy(p + off, &val, 4); + return; + } + for (int i = 0; i < 4; i++) { + uint8_t *q = page_for(addr + i, 1); + q[(addr + i) & (PAGE_SIZE - 1)] = (uint8_t)(val >> (8 * i)); + } +} + +static uint32_t mem_r16(uint32_t addr) { + uint8_t *p = page_for(addr, 0); + uint32_t off = addr & (PAGE_SIZE - 1); + if (p && off <= PAGE_SIZE - 2) { + uint16_t v; memcpy(&v, p + off, 2); return v; + } + uint32_t v = 0; + for (int i = 0; i < 2; i++) { + uint8_t *q = page_for(addr + i, 0); + v |= (uint32_t)(q ? q[(addr + i) & (PAGE_SIZE - 1)] : 0) << (8 * i); + } + return v; +} +static void mem_w16(uint32_t addr, uint32_t val) { + for (int i = 0; i < 2; i++) { + uint8_t *q = page_for(addr + i, 1); + q[(addr + i) & (PAGE_SIZE - 1)] = (uint8_t)(val >> (8 * i)); + } +} +static uint32_t mem_r8(uint32_t addr) { + uint8_t *p = page_for(addr, 0); + return p ? p[addr & (PAGE_SIZE - 1)] : 0; +} +static void mem_w8(uint32_t addr, uint32_t val) { + uint8_t *p = page_for(addr, 1); + p[addr & (PAGE_SIZE - 1)] = (uint8_t)val; +} + +/* ---------------- CPU state ---------------- */ +typedef struct { + double v; int rp; int has_raw; uint64_t raw; +} PipeEnt; + +typedef struct { + uint32_t r[32]; + uint32_t f[32]; + uint32_t cr[16]; + uint32_t pc; + int lcc; + /* delayed branch */ + int br_pending; uint32_t br_target; int br_delayed; + int squash; + /* DIM */ + int dim, dim_on, dim_exit, dim_half; + /* FP pipes */ + PipeEnt apipe[3], mpipe[3]; + uint32_t lpipe_lo[3]; uint32_t lpipe_hi[3]; int lpipe_has_hi[3]; + uint32_t gpipe; + double kr, ki, t; + /* control */ + int stop; + char stopmsg[256]; + uint64_t steps; + uint32_t ret_sentinel; +} Cpu; + +static Cpu C; + +static uint32_t rd(int i) { return i == 0 ? 0 : C.r[i]; } +static void wr(int i, uint32_t v) { if (i) C.r[i] = v; } +static uint32_t frd(int i) { return i <= 1 ? 0 : C.f[i]; } +static void fwr(int i, uint32_t v) { if (i > 1) C.f[i] = v; } + +static int cc(void) { return (C.cr[1] >> 2) & 1; } +static void set_cc(int cond) { + if (cond) C.cr[1] |= 4; else C.cr[1] &= ~4u; +} + +static int32_t s16(uint32_t v) { return (int32_t)(int16_t)(v & 0xffff); } +static int32_t s26(uint32_t v) { + return (v & 0x02000000) ? (int32_t)(v | 0xfc000000u) : (int32_t)v; +} + +static float b2f(uint32_t b) { float f; memcpy(&f, &b, 4); return f; } +static uint32_t f2b(float f) { uint32_t b; memcpy(&b, &f, 4); return b; } + +static double rdf(int reg, int dbl) { + if (dbl) { + int b = reg & ~1; + uint64_t bits = ((uint64_t)frd(b | 1) << 32) | frd(b); + double d; memcpy(&d, &bits, 8); return d; + } + return (double)b2f(frd(reg)); +} +static void wrf(int reg, double val, int dbl) { + if (dbl) { + uint64_t bits; memcpy(&bits, &val, 8); + int b = reg & ~1; + fwr(b, (uint32_t)bits); fwr(b | 1, (uint32_t)(bits >> 32)); + } else { + fwr(reg, f2b((float)val)); + } +} + +static double round_rp(double v, int rp) { + return rp ? v : (double)b2f(f2b((float)v)); +} + +static PipeEnt padv(PipeEnt *pipe, double val, int rp, int depth, + int has_raw, uint64_t raw) { + PipeEnt out = pipe[depth - 1]; + for (int i = depth - 1; i > 0; i--) pipe[i] = pipe[i - 1]; + pipe[0].v = has_raw ? val : round_rp(val, rp); + pipe[0].rp = rp; pipe[0].has_raw = has_raw; pipe[0].raw = raw; + return out; +} + +static void retire(int dest, PipeEnt e) { + if (!e.has_raw) { + wrf(dest, e.v, e.rp); + } else if (e.rp) { + int b0 = dest & ~1; + fwr(b0, (uint32_t)e.raw); fwr(b0 | 1, (uint32_t)(e.raw >> 32)); + } else { + fwr(dest, (uint32_t)e.raw); + } +} + +static void branch(uint32_t target, int delayed) { + C.br_pending = 1; C.br_target = target; C.br_delayed = delayed; +} + +/* returns 0 ok, 1 = stop, 2 = FP-not-implemented (fallback to Python) */ +static int exec_fp(uint32_t w, int src1, int src2, int dest); + +/* ---------------- the DIM tracker (port of _dim_track) ---------------- */ +static void dim_track(uint32_t w, uint32_t pc) { + (void)pc; + int fp = (((w >> 26) & 0x3f) == 0x12) || ((w & 0xfffffdffu) == 0xb0000000u); + int d_bit = (w & 0x200) != 0; + if (!C.dim) { + if (fp && d_bit && !C.dim_on) { C.dim_on = 1; return; } + if (C.dim_on) { C.dim = 1; C.dim_on = 0; C.dim_half = 0; return; } + } else { + /* in DIM: halves alternate positionally */ + if (C.dim_half == 0) { /* FP half */ + if (fp && !d_bit && !C.dim_exit) C.dim_exit = 1; + else if (C.dim_exit == 1 && fp && !d_bit) C.dim_exit = 1; + C.dim_half = 1; + } else { + C.dim_half = 0; + if (C.dim_exit) { + C.dim_exit++; + if (C.dim_exit >= 3) { C.dim = 0; C.dim_exit = 0; } + } + } + } +} + +/* ---------------- execute one instruction word ---------------- */ +static int execute(uint32_t w, uint32_t pc) { + int op = (w >> 26) & 0x3f; + int dest = (w >> 16) & 0x1f; + int src2 = (w >> 21) & 0x1f; + int src1 = (w >> 11) & 0x1f; + uint32_t imm = w & 0xffff; + + switch (op) { + /* ---- integer loads ---- */ + case 0x00: case 0x01: { /* ld.b */ + uint32_t ea = (op & 1) ? rd(src2) + s16(imm) + : rd(src2) + rd(src1); + uint32_t b = mem_r8(ea); + wr(dest, (uint32_t)(int32_t)(int8_t)b); + return 0; + } + case 0x04: case 0x05: { /* ld.s / ld.l via bit0 */ + uint32_t off = (op & 1) ? (uint32_t)s16(imm & 0xfffe ? imm : imm) + : rd(src1); + int isl = imm & 1; + if (op & 1) { + if (isl) off = (uint32_t)s16(imm & 0xfffc); + else off = (uint32_t)s16(imm & 0xfffe); + } else { + isl = imm & 1; /* register-indexed: bit0 of imm still selects */ + } + uint32_t ea = rd(src2) + off; + if (isl) wr(dest, mem_r32(ea)); + else wr(dest, (uint32_t)(int32_t)(int16_t)mem_r16(ea)); + return 0; + } + /* ---- integer stores (split offset) ---- */ + case 0x03: { /* st.b */ + uint32_t off = ((w >> 16) & 0x1f) << 11 | (w & 0x7ff); + mem_w8(rd(src2) + s16(off), rd(src1) & 0xff); + return 0; + } + case 0x07: { /* st.s / st.l */ + uint32_t off = ((w >> 16) & 0x1f) << 11 | (w & 0x7ff); + if (off & 1) mem_w32(rd(src2) + s16(off & 0xfffc), rd(src1)); + else mem_w16(rd(src2) + s16(off & 0xfffe), rd(src1) & 0xffff); + return 0; + } + /* ---- FP loads/stores ---- */ + case 0x08: case 0x09: case 0x0a: case 0x0b: { + int fl = w & 7; + int size = (fl & 2) ? 4 : ((fl & 4) ? 16 : 8); + uint32_t off; + if (op & 1) off = (uint32_t)s16(imm & (0x10000 - size)); + else off = rd(src1); + uint32_t ea = rd(src2) + off; + if (fl & 1) wr(src2, ea); + int b0 = dest & ~((size / 4) - 1); + if (op < 0x0a) { + for (int i = 0; i < size / 4; i++) fwr(b0 + i, mem_r32(ea + 4 * i)); + } else { + for (int i = 0; i < size / 4; i++) mem_w32(ea + 4 * i, frd(b0 + i)); + } + return 0; + } + case 0x18: case 0x19: { /* pfld */ + int fl = w & 7; + int size = (fl & 2) ? 4 : 8; + uint32_t off; + if (op & 1) off = (uint32_t)s16(imm & (0x10000 - size)); + else off = rd(src1); + uint32_t ea = rd(src2) + off; + if (fl & 1) wr(src2, ea); + uint32_t lo = C.lpipe_lo[2], hi = C.lpipe_hi[2]; + int has_hi = C.lpipe_has_hi[2]; + C.lpipe_lo[2] = C.lpipe_lo[1]; C.lpipe_hi[2] = C.lpipe_hi[1]; + C.lpipe_has_hi[2] = C.lpipe_has_hi[1]; + C.lpipe_lo[1] = C.lpipe_lo[0]; C.lpipe_hi[1] = C.lpipe_hi[0]; + C.lpipe_has_hi[1] = C.lpipe_has_hi[0]; + C.lpipe_lo[0] = mem_r32(ea); + C.lpipe_has_hi[0] = (size == 8); + C.lpipe_hi[0] = (size == 8) ? mem_r32(ea + 4) : 0; + if (!has_hi) { + fwr(dest, lo); + } else { + int b0 = dest & ~1; + fwr(b0, lo); fwr(b0 | 1, hi); + } + return 0; + } + case 0x0d: { /* flush */ + uint32_t off = ((w >> 16) & 0x1f) << 11 | (w & 0x7ff); + if (off & 1) wr(src2, rd(src2) + s16(off & 0xffe0)); + return 0; + } + case 0x0c: wr(dest, C.cr[src2 & 15]); return 0; /* ld.c */ + case 0x0e: C.cr[src2 & 15] = rd(src1); return 0; /* st.c */ + case 0x02: fwr(dest, rd(src1)); return 0; /* ixfr */ + + /* ---- control transfer ---- */ + case 0x10: branch(rd(src1), 1); return 0; /* bri */ + case 0x13: { /* core escape */ + int sub = w & 0x1f; + if (sub == 0x02) { wr(1, pc + 8); branch(rd(src1), 1); return 0; } + if (sub == 0x01 || sub == 0x07 || sub == 0x04) return 0; + C.stop = 1; + snprintf(C.stopmsg, sizeof C.stopmsg, + "core-escape sub %#x @ %#x", sub, pc); + return 1; + } + case 0x1a: branch(pc + 4 + s26(w & 0x03ffffff) * 4, 1); return 0; /* br */ + case 0x1b: wr(1, pc + 8); + branch(pc + 4 + s26(w & 0x03ffffff) * 4, 1); return 0; /* call */ + case 0x1c: case 0x1d: { /* bc / bc.t */ + uint32_t tgt = pc + 4 + s26(w & 0x03ffffff) * 4; + if (cc()) branch(tgt, op == 0x1d); + else if (op == 0x1d) C.squash = 1; + return 0; + } + case 0x1e: case 0x1f: { /* bnc / bnc.t */ + uint32_t tgt = pc + 4 + s26(w & 0x03ffffff) * 4; + if (!cc()) branch(tgt, op == 0x1f); + else if (op == 0x1f) C.squash = 1; + return 0; + } + case 0x14: case 0x15: case 0x16: case 0x17: { /* btne/bte */ + int32_t broff = s16(((dest << 11) | (w & 0x7ff)) & 0xffff); + uint32_t a = (op & 1) ? (uint32_t)src1 : rd(src1); + uint32_t b = rd(src2); + int take = (op < 0x16) ? (a != b) : (a == b); + if (take) branch(pc + 4 + broff * 4, 0); + return 0; + } + case 0x2d: { /* bla */ + uint32_t a = rd(src1), b = rd(src2); + int taken = C.lcc; + if ((int32_t)a < 0) + C.lcc = ((int32_t)(a + b) >= 0) ? 1 : 0; + else + C.lcc = ((uint64_t)a + (uint64_t)b > MASK32) ? 1 : 0; + wr(src2, a + b); + if (taken) { + int32_t off = s16(((((w >> 16) & 0x1f) << 11) | (w & 0x7ff)) & 0xffff); + branch(pc + 4 + off * 4, 1); + } + return 0; + } + default: break; + } + + /* ---- arithmetic / logic ---- */ + { + int base = op & 0x3e; + if (base >= 0x20 && base <= 0x26) { /* addu subu adds subs */ + uint32_t b = rd(src2); + uint32_t a = (op & 1) ? (uint32_t)s16(imm) : rd(src1); + if (base == 0x20) { + uint64_t s = (uint64_t)a + b; + wr(dest, (uint32_t)s); set_cc(s > MASK32); + } else if (base == 0x24) { + int64_t s = (int64_t)(int32_t)a + (int32_t)b; + wr(dest, (uint32_t)s); set_cc(((uint32_t)s & 0x80000000u) != 0); + } else if (base == 0x22) { + wr(dest, a - b); set_cc(a < b); + } else { + wr(dest, a - b); + set_cc((int32_t)a < (int32_t)b); + } + return 0; + } + if (base >= 0x28 && base <= 0x2e) { /* shifts (incl shrd) */ + int cnt = (op & 1) ? (s16(imm) & 0x1f) : (rd(src1) & 0x1f); + uint32_t b = rd(src2); + uint32_t r; + if (base == 0x28) r = b << cnt; + else if (base == 0x2a) r = b >> cnt; + else if (base == 0x2e) r = (uint32_t)((int32_t)b >> cnt); + else r = b >> cnt; + wr(dest, r); + return 0; + } + if (base >= 0x30) { /* and..xorh */ + uint32_t a = (op & 1) ? imm : rd(src1); + int hi = (base == 0x32 || base == 0x36 || base == 0x3a || base == 0x3e); + if (hi && (op & 1)) a = imm << 16; + uint32_t b = rd(src2); + int g = base & 0x3c; + uint32_t r; + if (g == 0x30) r = b & a; + else if (g == 0x34) r = b & ~a; + else if (g == 0x38) r = b | a; + else r = b ^ a; + wr(dest, r); + set_cc(r == 0); + return 0; + } + } + + if (op == 0x12) return exec_fp(w, src1, src2, dest); + + C.stop = 1; + snprintf(C.stopmsg, sizeof C.stopmsg, + "unimplemented op %#x @ %#x w=%#x", op, pc, w); + return 1; +} + +/* ---------------- FP unit: full port of emu860.exec_fp ---------------- */ +/* dual-op routing table: {m1, m2, a1, a2, kload, tload}; operand tags: + * 0=s1 1=s2 2=kr 3=ki 4=t 5=PP(flagm) 6=mm */ +static const int DUAL[16][6] = { + {2,1, 0,6, 0,0}, {2,1, 4,6, 1,0}, {2,1, 0,5, 0,1}, {2,1, 4,5, 1,1}, + {3,1, 0,6, 0,0}, {3,1, 4,6, 1,0}, {3,1, 0,5, 0,1}, {3,1, 4,5, 1,1}, + {2,5, 0,1, 0,1}, {0,1, 5,6, 0,0}, {2,5, 0,1, 0,0}, {0,1, 4,5, 0,1}, + {3,5, 0,1, 0,1}, {0,1, 4,6, 0,0}, {3,5, 0,1, 0,0}, {0,1, 4,5, 0,0}, +}; + +static int exec_fp(uint32_t w, int src1, int src2, int dest) { + if ((w & 0xfffffdffu) == 0xb0000000u) return 0; /* fnop / d.fnop */ + int sub = w & 0x7f; + int sp = (w & 0x100) ? 1 : 0; + int rp = (w & 0x080) ? 1 : 0; + int pbit = (w >> 10) & 1; + + if (sub < 0x20) { /* dual ops PF[M]AM / PF[M]SM */ + int mdepth = sp ? 2 : 3; + PipeEnt am = C.apipe[2]; + PipeEnt mm = C.mpipe[mdepth - 1]; + double am_out = am.v, mm_out = mm.v; + double pipe_val = pbit ? am_out : mm_out; + const int *dd = DUAL[sub & 0xf]; + double vals_sp[7], vals_rp[7]; + vals_sp[0] = rdf(src1, sp); vals_sp[1] = rdf(src2, sp); + vals_rp[0] = rdf(src1, rp); vals_rp[1] = rdf(src2, rp); + vals_sp[2] = vals_rp[2] = C.kr; + vals_sp[3] = vals_rp[3] = C.ki; + vals_sp[4] = vals_rp[4] = C.t; + vals_sp[5] = vals_rp[5] = pipe_val; + vals_sp[6] = vals_rp[6] = mm_out; + double v1 = vals_sp[dd[0]], v2 = vals_sp[dd[1]]; + if (dd[1] == 1 && dest && src2 == dest) v2 = pipe_val; + double newm = v1 * v2; + double u1 = vals_rp[dd[2]], u2 = vals_rp[dd[3]]; + if (dd[2] == 0 && dest && src1 == dest) u1 = pipe_val; + if (dd[3] == 1 && dest && src2 == dest) u2 = pipe_val; + double newa = (sub & 0x10) ? (u1 - u2) : (u1 + u2); + if (dd[5]) C.t = mm_out; + if (dd[4]) { + if (dd[0] == 3) C.ki = rdf(src1, sp); + else C.kr = rdf(src1, sp); + } + if (pbit) retire(dest, C.apipe[2]); + else retire(dest, C.mpipe[mdepth - 1]); + padv(C.apipe, newa, rp, 3, 0, 0); + padv(C.mpipe, newm, rp, mdepth, 0, 0); + return 0; + } + if (sub == 0x20 || sub == 0x24) { /* fmul / pfmul (0x24 = 3-stage) */ + double v1 = rdf(src1, sp), v2 = rdf(src2, sp); + if (pbit) { + int depth = (sub == 0x24) ? 3 : (sp ? 2 : 3); + double out = C.mpipe[depth - 1].v; + if (dest && src1 == dest) v1 = out; + if (dest && src2 == dest) v2 = out; + retire(dest, C.mpipe[depth - 1]); + padv(C.mpipe, v1 * v2, rp, depth, 0, 0); + } else { + wrf(dest, v1 * v2, rp); + } + return 0; + } + if (sub == 0x30 || sub == 0x31 || sub == 0x33) { /* fadd/fsub/famov */ + double v1 = rdf(src1, sp), v2 = rdf(src2, sp); + double res; + if (pbit) { + double out = C.apipe[2].v; + if (dest && src1 == dest) v1 = out; + if (dest && src2 == dest) v2 = out; + res = (sub == 0x33) ? v1 : ((sub == 0x31) ? v1 - v2 : v1 + v2); + retire(dest, C.apipe[2]); + padv(C.apipe, res, rp, 3, 0, 0); + } else { + res = (sub == 0x33) ? v1 : ((sub == 0x31) ? v1 - v2 : v1 + v2); + wrf(dest, res, rp); + } + return 0; + } + if (sub == 0x22) { /* frcp */ + double b = rdf(src2, sp); + wrf(dest, b != 0.0 ? 1.0 / b : 0.0, rp); + return 0; + } + if (sub == 0x23) { /* frsqr */ + double b = rdf(src2, sp); + wrf(dest, b > 0.0 ? 1.0 / sqrt(b) : 0.0, rp); + return 0; + } + if (sub == 0x32 || sub == 0x3a) { /* fix / ftrunc */ + double a = rdf(src1, sp); + uint64_t iv; + if (a != a || a > 2147483647.0 || a < -2147483648.0) { + iv = 0x80000000u; + } else { + double r_ = (sub == 0x3a) ? a : (a >= 0 ? a + 0.5 : a - 0.5); + int64_t t_ = (int64_t)r_; + iv = (uint64_t)t_; + } + iv &= rp ? 0xFFFFFFFFFFFFFFFFull : 0xFFFFFFFFull; + if (pbit) { + retire(dest, C.apipe[2]); + padv(C.apipe, (double)(uint32_t)iv, rp, 3, 1, iv); + } else if (rp) { + int b0 = dest & ~1; + fwr(b0, (uint32_t)iv); fwr(b0 | 1, (uint32_t)(iv >> 32)); + } else { + fwr(dest, (uint32_t)iv); + } + return 0; + } + if (sub == 0x34 || sub == 0x35) { /* pfgt/pfle (0x34) / pfeq (0x35) */ + double v1 = rdf(src1, sp), v2 = rdf(src2, sp); + if (sub == 0x35) set_cc(v1 == v2); + else if (w & 0x80) set_cc(!(v1 <= v2)); + else set_cc(v1 > v2); + retire(dest, C.apipe[2]); + padv(C.apipe, 0.0, sp, 3, 0, 0); + return 0; + } + if (sub == 0x21) { /* fmlow.dd */ + uint64_t prod = (uint64_t)frd(src1) * (uint64_t)frd(src2); + int b0 = dest & ~1; + fwr(b0, (uint32_t)prod); fwr(b0 | 1, (uint32_t)(prod >> 32)); + return 0; + } + if (sub == 0x40) { /* fxfr FP->int */ + wr(dest, frd(src1)); + return 0; + } + if (sub == 0x49 || sub == 0x4d) { /* fiadd / fisub */ + uint64_t a, b, m, r_; + if (sp) { + a = (uint64_t)frd(src1 & ~1) | ((uint64_t)frd((src1 & ~1) | 1) << 32); + b = (uint64_t)frd(src2 & ~1) | ((uint64_t)frd((src2 & ~1) | 1) << 32); + m = 0xFFFFFFFFFFFFFFFFull; + } else { + a = frd(src1); b = frd(src2); m = MASK32; + } + r_ = ((sub == 0x49) ? a + b : a - b) & m; + if (pbit) { + uint64_t out = C.gpipe; + C.gpipe = (uint32_t)r_; /* NB python gpipe stores raw int */ + r_ = out; + } + if (sp) { + int b0 = dest & ~1; + fwr(b0, (uint32_t)r_); fwr(b0 | 1, (uint32_t)(r_ >> 32)); + } else { + fwr(dest, (uint32_t)r_); + } + return 0; + } + if (sub == 0x5f) return 0; /* fnop */ + C.stop = 1; + snprintf(C.stopmsg, sizeof C.stopmsg, + "unimplemented FP sub %#x @ %#x", sub, C.pc); + return 1; +} + +/* ---------------- the step loop ---------------- */ +/* reasons: 0 = hit hook pc, 1 = stop, 2 = FP fallback (execute w at C.pc via + * Python then call resume()), 3 = max steps */ +static uint32_t *g_hooks; static int g_nhooks; + +static int is_hook(uint32_t pc) { + for (int i = 0; i < g_nhooks; i++) + if (g_hooks[i] == pc) return 1; + return 0; +} + +static int step_once(int *fp_fallback) { + uint32_t pc = C.pc; + uint32_t w = mem_r32(pc); + int dim_at_fetch = C.dim; + C.br_pending = 0; + C.squash = 0; + int rc = execute(w, pc); + if (rc == 2) { *fp_fallback = 1; return 0; } /* pc unchanged; steps not counted */ + C.steps++; + dim_track(w, pc); + if (C.stop) return 1; + if (C.br_pending) { + uint32_t target = C.br_target; + if (C.br_delayed) { + int nslots = dim_at_fetch ? 2 : 1; + uint32_t ds = pc + 4; + C.br_pending = 0; + for (int k = 0; k < nslots; k++) { + uint32_t w2 = mem_r32(ds); + int rc2 = execute(w2, ds); + if (rc2 == 2) { *fp_fallback = 2; C.pc = ds; return 0; } /* mid-slot fallback */ + C.steps++; + dim_track(w2, ds); + ds += 4; + } + C.pc = C.br_pending ? C.br_target : target; + C.dim_half = 0; + } else { + C.pc = target; + C.dim_half = 0; + } + } else if (C.squash) { + int nslots = dim_at_fetch ? 2 : 1; + C.squash = 0; + C.pc = pc + 4 + nslots * 4; + } else { + C.pc = pc + 4; + } + return C.stop; +} + +/* ---------------- Python interface ---------------- */ +static PyObject *py_init(PyObject *self, PyObject *args) { + (void)self; (void)args; + if (!g_pages) g_pages = (uint8_t **)calloc(N_PAGES, sizeof(uint8_t *)); + memset(&C, 0, sizeof C); + for (int i = 0; i < 3; i++) { C.apipe[i].rp = 1; C.mpipe[i].rp = 1; } + Py_RETURN_NONE; +} + +static PyObject *py_load_blob(PyObject *self, PyObject *args) { + (void)self; + Py_buffer buf; unsigned long long addr; + if (!PyArg_ParseTuple(args, "Ky*", &addr, &buf)) return NULL; + const uint8_t *src = (const uint8_t *)buf.buf; + for (Py_ssize_t i = 0; i < buf.len; i++) + mem_w8((uint32_t)(addr + i), src[i]); + PyBuffer_Release(&buf); + Py_RETURN_NONE; +} + +static PyObject *py_set_hooks(PyObject *self, PyObject *args) { + (void)self; + PyObject *lst; + if (!PyArg_ParseTuple(args, "O", &lst)) return NULL; + Py_ssize_t n = PyList_Size(lst); + free(g_hooks); + g_hooks = (uint32_t *)malloc(sizeof(uint32_t) * (size_t)n); + g_nhooks = (int)n; + for (Py_ssize_t i = 0; i < n; i++) + g_hooks[i] = (uint32_t)PyLong_AsUnsignedLong(PyList_GetItem(lst, i)); + Py_RETURN_NONE; +} + +static PyObject *py_run(PyObject *self, PyObject *args) { + (void)self; + unsigned long long max_steps; + if (!PyArg_ParseTuple(args, "K", &max_steps)) return NULL; + uint64_t start = C.steps; + int reason = 3; + while (C.steps - start < max_steps) { + if (is_hook(C.pc) || C.pc == C.ret_sentinel) { reason = 0; break; } + int fb = 0; + int stopped = step_once(&fb); + if (fb) { reason = 2; break; } + if (stopped) { reason = 1; break; } + } + return Py_BuildValue("iK", reason, (unsigned long long)C.steps); +} + +static PyObject *py_getstate(PyObject *self, PyObject *args) { + (void)self; (void)args; + PyObject *regs = PyList_New(32), *fregs = PyList_New(32); + for (int i = 0; i < 32; i++) { + PyList_SetItem(regs, i, PyLong_FromUnsignedLong(C.r[i])); + PyList_SetItem(fregs, i, PyLong_FromUnsignedLong(C.f[i])); + } + return Py_BuildValue("{s:N,s:N,s:k,s:K,s:i}", + "r", regs, "f", fregs, "pc", C.pc, + "steps", (unsigned long long)C.steps, "dim", C.dim); +} + +static PyObject *py_setreg(PyObject *self, PyObject *args) { + (void)self; + int i; unsigned long v; int isf; + if (!PyArg_ParseTuple(args, "iki", &i, &v, &isf)) return NULL; + if (isf) C.f[i] = (uint32_t)v; else C.r[i] = (uint32_t)v; + Py_RETURN_NONE; +} + +static PyObject *py_setpc(PyObject *self, PyObject *args) { + (void)self; + unsigned long v; + if (!PyArg_ParseTuple(args, "k", &v)) return NULL; + C.pc = (uint32_t)v; + Py_RETURN_NONE; +} + +static PyObject *py_setsentinel(PyObject *self, PyObject *args) { + (void)self; + unsigned long v; + if (!PyArg_ParseTuple(args, "k", &v)) return NULL; + C.ret_sentinel = (uint32_t)v; + Py_RETURN_NONE; +} + +static PyObject *py_r32(PyObject *self, PyObject *args) { + (void)self; + unsigned long a; + if (!PyArg_ParseTuple(args, "k", &a)) return NULL; + return PyLong_FromUnsignedLong(mem_r32((uint32_t)a)); +} + +static PyObject *py_w32(PyObject *self, PyObject *args) { + (void)self; + unsigned long a, v; + if (!PyArg_ParseTuple(args, "kk", &a, &v)) return NULL; + mem_w32((uint32_t)a, (uint32_t)v); + Py_RETURN_NONE; +} + +static PyMethodDef methods[] = { + {"init", py_init, METH_NOARGS, "reset state"}, + {"load_blob", py_load_blob, METH_VARARGS, "load bytes at addr"}, + {"set_hooks", py_set_hooks, METH_VARARGS, "set hook pcs"}, + {"run", py_run, METH_VARARGS, "run until hook/stop/fallback/max"}, + {"getstate", py_getstate, METH_NOARGS, "dump state"}, + {"setreg", py_setreg, METH_VARARGS, "set r/f register"}, + {"setpc", py_setpc, METH_VARARGS, "set pc"}, + {"set_sentinel", py_setsentinel, METH_VARARGS, "set return sentinel"}, + {"r32", py_r32, METH_VARARGS, "read mem"}, + {"w32", py_w32, METH_VARARGS, "write mem"}, + {NULL, NULL, 0, NULL} +}; + +static struct PyModuleDef mod = { + PyModuleDef_HEAD_INIT, "emu860c", "i860 step-core in C", -1, methods, + NULL, NULL, NULL, NULL +}; + +PyMODINIT_FUNC PyInit_emu860c(void) { return PyModule_Create(&mod); }