#!/usr/bin/env python3 """gapcensus.py -- the #60 export gap census. Maps the Ghidra export's coverage of BTL4OPT.EXE's code section and inventories every DARK region (bytes belonging to no indexed function), so "absence in the export" claims can be checked against a known inventory instead of tripping over gaps one dig at a time (reconstruction-gotchas #20). Layers: 1. INDEX coverage -- reference/decomp/functions_index.tsv (addr+size) vs the PE .text extent -> raw dark regions. 2. EXPORT coverage -- /* @ADDR */ headers in reference/decomp/all/part_*.c; indexed-but-unexported = Ghidra knew the function but the pseudocode is absent (decompile failure/filter). 3. DISCOVERY -- function starts inside dark regions, from E8 call targets + data-section code pointers (vtables/handler tables/performance pointers), with caller counts. 4. ANNOTATION -- nearest TU attribution (file= tags), and "VISITED" flags for addresses already cited anywhere in the port/KB (game/, context/, docs/, CLASSMAP). Output: reference/decomp/GAP_CENSUS.md (+ gap_census.tsv, machine-readable). Deterministic: re-runs diff cleanly. """ import glob import os import re import struct import sys ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) EXE = os.path.join(ROOT, 'content', 'BTL4OPT.EXE') IDX = os.path.join(ROOT, 'reference', 'decomp', 'functions_index.tsv') PARTS = os.path.join(ROOT, 'reference', 'decomp', 'all', 'part_*.c') OUT_MD = os.path.join(ROOT, 'reference', 'decomp', 'GAP_CENSUS.md') OUT_TSV = os.path.join(ROOT, 'reference', 'decomp', 'gap_census.tsv') MIN_REGION = 16 # ignore alignment slivers below this many bytes TOP_REGIONS = 40 # detail table size def pe_sections(data): e_lfanew = struct.unpack_from(' 0] # ---- layer 1: the index ------------------------------------------------ index = {} # addr -> (size, name) for line in open(IDX, encoding='utf-8', errors='replace'): f = line.rstrip('\n').split('\t') if len(f) < 4: continue try: a = int(f[0], 16) sz = int(f[1]) except ValueError: continue index[a] = (sz, f[3]) idx_sorted = sorted(index) covered = [] # merged [lo,hi) intervals for a in idx_sorted: sz = index[a][0] lo, hi = a, a + max(sz, 1) if covered and lo <= covered[-1][1]: covered[-1][1] = max(covered[-1][1], hi) else: covered.append([lo, hi]) dark = [] # [lo,hi) uncovered in .text cur = tlo for lo, hi in covered: if lo > cur: dark.append((cur, min(lo, thi))) cur = max(cur, hi) if cur >= thi: break if cur < thi: dark.append((cur, thi)) dark = [(lo, hi) for lo, hi in dark if hi - lo >= MIN_REGION] # ---- layer 2: the export ---------------------------------------------- exported = {} # addr -> file tag hdr = re.compile(r'/\* @(00[0-9a-f]{6}) file=(\S+) name=') for p in sorted(glob.glob(PARTS)): for m in hdr.finditer(open(p, encoding='utf-8', errors='replace').read()): exported[int(m.group(1), 16)] = m.group(2) unexported = [a for a in idx_sorted if a not in exported] # TU attribution anchors (file= != '?') tagged = sorted((a, f) for a, f in exported.items() if f != '?') def nearest_tu(addr): before = after = None for a, f in tagged: if a <= addr: before = f else: after = f break if before and after and before == after: return before return '%s|%s' % (before or '?', after or '?') # ---- layer 3: discovery ------------------------------------------------ callers = {} # target -> caller count b = data for off in range(t_raw, t_raw + (thi - tlo) - 5): if b[off] == 0xE8: rel = struct.unpack_from('= 2 or va in dptr) # ---- layer 4: repo citations ------------------------------------------- cite = set() pat = re.compile(r'(?:@|FUN_|0x)0{0,2}(4[0-9a-f]{5})\b', re.I) for base in ('game', 'context', 'docs'): for dirpath, _dirs, files in os.walk(os.path.join(ROOT, base)): for fn in files: if not fn.lower().endswith(('.cpp', '.hpp', '.h', '.md')): continue try: txt = open(os.path.join(dirpath, fn), encoding='utf-8', errors='replace').read() except OSError: continue for m in pat.finditer(txt): cite.add(int(m.group(1), 16)) # ---- assemble region records ------------------------------------------- regions = [] for lo, hi in dark: seg = b[va_off(lo):va_off(hi)] padb = sum(1 for c in seg if c in (0xCC, 0x90, 0x00)) starts = sorted(set( [v for v in callers if lo <= v < hi and strong_start(v)] + [v for v in dptr if lo <= v < hi])) visited = sorted(v for v in set(starts) | {a for a in cite if lo <= a < hi} if lo <= v < hi and v in cite) regions.append({ 'lo': lo, 'hi': hi, 'size': hi - lo, 'code': (hi - lo) - padb, 'starts': starts, 'call_in': sum(callers.get(v, 0) for v in starts), 'visited': visited, 'tu': nearest_tu(lo), }) regions.sort(key=lambda r: (-r['code'], r['lo'])) text_bytes = thi - tlo idx_bytes = sum(hi - lo for lo, hi in covered if lo < thi) - max(0, covered[-1][1] - thi if covered else 0) dark_bytes = sum(r['size'] for r in regions) unexp_bytes = sum(index[a][0] for a in unexported) # ---- TSV ---------------------------------------------------------------- with open(OUT_TSV, 'w', encoding='utf-8', newline='\n') as t: t.write('lo\thi\tsize\tcode\tstarts\tcall_in\tvisited\ttu\n') for r in regions: t.write('%06x\t%06x\t%d\t%d\t%s\t%d\t%s\t%s\n' % ( r['lo'], r['hi'], r['size'], r['code'], ','.join('%06x' % v for v in r['starts']), r['call_in'], ','.join('%06x' % v for v in r['visited']), r['tu'])) # ---- report ------------------------------------------------------------- L = [] A = L.append A('# GAP CENSUS -- the export dark-region inventory (#60)') A('') A('Generated by `tools/gapcensus.py` (deterministic; re-run after any re-export).') A('Machine-readable twin: `gap_census.tsv`.') A('') A('## Headline') A('| Metric | Value |') A('|---|---|') A('| Code section (.text) | `0x%06x`-`0x%06x` (%d KB) |' % (tlo, thi, text_bytes // 1024)) A('| Indexed functions (functions_index.tsv) | %d, covering %d KB (%.1f%%) |' % (len(index), idx_bytes // 1024, 100.0 * idx_bytes / text_bytes)) code_bytes = sum(r['code'] for r in regions) A('| **RAW DARK (no indexed function at all)** | **%d regions >= %dB, %d KB (%.1f%%) -- %d KB REAL CODE after padding** |' % (len(regions), MIN_REGION, dark_bytes // 1024, 100.0 * dark_bytes / text_bytes, code_bytes // 1024)) A('| Exported pseudocode functions (part_*.c) | %d |' % len(exported)) A('| **Indexed but NOT exported** (Ghidra knew it; no pseudocode) | **%d functions, %d KB** |' % (len(unexported), unexp_bytes // 1024)) A('| Function starts discovered inside dark regions | %d (call-graph + data-pointer evidence) |' % sum(len(r['starts']) for r in regions)) A('') A('Reading the tables: `visited` = the address is already cited somewhere in') A('game/ context/ docs/ (a prior dig reached it); everything else is UNCHARTED.') A('`tu` = nearest file-tagged export neighbors (`before|after` when they disagree).') A('') # dark code by TU family -- engine-source TUs matter less (we compile the # real MUNGA/WinTesla source); bt/ + bt_l4/ dark is the reconstruction target. fam = {} for r in regions: tus = set(t for t in r['tu'].replace('|', ' ').split() if t != '?') cats = set() for t in tus: if t.startswith('bt/'): cats.add('bt/ (GAME -- reconstruction target)') elif t.startswith('bt_l4/'): cats.add('bt_l4/ (GAME video/glue -- reconstruction target)') elif t.startswith('munga_l4/'): cats.add('munga_l4/ (engine L4 -- source in repo)') elif t.startswith('munga/'): cats.add('munga/ (engine core -- source in repo)') else: cats.add('other/unknown') key = sorted(cats)[0] if len(cats) == 1 else 'BOUNDARY/mixed' f = fam.setdefault(key, [0, 0]) f[0] += 1 f[1] += r['code'] A('## Dark code by suspected TU family') A('') A('| Family | Regions | Real code bytes |') A('|---|---|---|') for k in sorted(fam, key=lambda k: -fam[k][1]): A('| %s | %d | %d |' % (k, fam[k][0], fam[k][1])) A('') A('## Top %d dark regions by REAL CODE bytes (padding excluded)' % TOP_REGIONS) A('') A('| # | Range | Code bytes | Fn starts | Call-ins | Visited | Suspected TU |') A('|---|---|---|---|---|---|---|') for i, r in enumerate(regions[:TOP_REGIONS], 1): A('| %d | `0x%06x`-`0x%06x` | %d | %d | %d | %s | %s |' % ( i, r['lo'], r['hi'], r['code'], len(r['starts']), r['call_in'], ('%d: ' % len(r['visited'])) + ' '.join('`%x`' % v for v in r['visited'][:4]) + ('...' if len(r['visited']) > 4 else '') if r['visited'] else '-', r['tu'])) A('') A('## Indexed-but-unexported functions (first 60 by address)') A('') A('These have index rows (address + size) but no pseudocode in `all/part_*.c` --') A('decompile them individually (raw disasm or a targeted re-export) when a dig arrives.') A('') A('| Addr | Size | Index name | Cited in repo |') A('|---|---|---|---|') for a in unexported[:60]: sz, nm = index[a] A('| `0x%06x` | %d | %s | %s |' % (a, sz, nm, 'YES' if a in cite else '-')) if len(unexported) > 60: A('') A('(+%d more -- see gap_census.tsv)' % (len(unexported) - 60)) A('') open(OUT_MD, 'w', encoding='utf-8', newline='\n').write('\n'.join(L)) print('text %dKB indexed %.1f%% dark %d regions/%dKB unexported %d fns/%dKB' % (text_bytes // 1024, 100.0 * idx_bytes / text_bytes, len(regions), dark_bytes // 1024, len(unexported), unexp_bytes // 1024)) print('wrote', OUT_MD) print('wrote', OUT_TSV) if __name__ == '__main__': main()