Tooling built to recover editable source for six 'Mech chassis that exist
in the parallel FS_Build_V4H build but not in this repo. Reverse-engineers
every compiled record type in the .mw4 package format back to the .data /
.instance / .subsystems / .damage / .contents / .torso / .engine /
.armature sources the content pipeline consumes.
Nothing here is wired into the game build. It is a standalone analysis
harness run from Linux.
Package format
--------------
"#VBD" container. Directory records are [len][name][FILETIME][origSize]
[storedSize][offset], payload base at dword 0x0C. A record is stored raw
when storedSize == origSize, otherwise LZW (9->12-bit LSB-first codes,
256=clear, 257=EOF, dict from 258), per Database.cpp:451.
GameModel records are flat /Zp4 structs following the C++ inheritance
chain Entity(0) -> Mover(28) -> MWObject(80) -> Vehicle(664) -> Mech(756),
1636 bytes total. CreateMessage records follow Replicator -> Entity ->
Mover -> MWMover -> MWObject -> Vehicle -> Mech from start=16 (the
undeclared Connection__Message header), ending at 341 and padded to 344.
tools/decompile/
----------------
datamap.py header-driven layout engine; CHAIN + ANCHORS
{Vehicle:664, Mech:756} assert the struct offsets
mw4msg.py CreateMessage reader/walker
data.py .data constants.py define/table symbol resolution
damage.py .damage contents.py .contents
smallmodel.py .torso + .engine instance.py .instance
armature.py / armature_parts.py .armature + armaturedata/armaturevideo
assembly.py joint hierarchy renderer
make_generic_doll.py builds generic MFD/Radar damage dolls
verify_*.py per-type round-trip verifiers
Verified round-trip across all 64 shared chassis:
.armature 2938/2976 pages .subsystems 7579/7585 keys
.data map 6071/6071 values .data trip 8291/8306 keys
.damage 6605/6605 keys .contents 7480/7480 keys
.torso+.engine 1280/1280 keys .instance 896/896 keys, 64/64 pages
armature_parts 1202/1202 .data, 1149/1202 .video
Layout-discovery lessons (documented in DECOMPILING.md)
-------------------------------------------------------
- Never let a field map be discovered by the values that verify it. A
value-matching pass reported 4288/4288 while mis-assigning 34 keys. The
map was rebuilt from header declaration order, anchored on uniquely
resolved fields.
- Read the factory, not the data. 12 .data fields and 5 Torso fields are
declared plain Stuff::Scalar but multiplied by Radians_Per_Degree in
Mech_Tool.cpp:889 / Torso_Tool.cpp.
- Strip typedefs before walking a header. A stray `typedef int AttributeID;`
masked a missing ClassID - two 4-byte errors cancelling out, caught only
by the ANCHORS assertion.
- A verifier that silently narrows its own input reports success. Braced
blocks must be hidden before splitting pages, replacing both CR and LF,
because a `Shadow={...}` block contains a line reading `[shadow]` and
splitlines() also splits on bare CR.
- NSWIZZLE is undefined, so the #else branch is live and orders members
differently. bool is 1 byte; char x[MaxStringLength] is 256.
- V4H carries stale Mech IDs (their Atlas is 5, ours 6), so 64 of 65 shared
chassis are off by one; --retarget-ids emits $(M_<Chassis>)/$(IDS_<Chassis>).
reports/ holds generated diffs. The two ~5 MB manifest-*.tsv intermediates
are gitignored; regenerate everything with run-comparison.sh.
Co-authored-by: Claude Opus 5 (Anthropic) <noreply@anthropic.com>
Co-authored-by: GitHub Copilot <copilot@github.com>
101 lines
3.7 KiB
Python
101 lines
3.7 KiB
Python
#!/usr/bin/env python3
|
|
"""Round-trip verifier for the .contents decompiler.
|
|
|
|
Regenerates each `.contents` and compares page names and both key values against
|
|
the authored source. Page order is not compared -- NotationFile is
|
|
order-independent, and the packer groups pages by parent joint rather than
|
|
preserving the authored sequence.
|
|
|
|
python3 verify_contents.py [--show CHASSIS]
|
|
"""
|
|
import argparse, collections, glob, os, re, sys
|
|
|
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
import contents, subsystems
|
|
|
|
REC = "/home/rich/Repositories/FS_Ours_extracted/core/mechs"
|
|
SRC = "/home/rich/Repositories/firestorm/Gameleap/mw4/Content/Mechs"
|
|
|
|
|
|
def source_pages(path):
|
|
txt = open(path, "rb").read().decode("latin-1")
|
|
txt = re.sub(r'//[^\n]*', '', txt)
|
|
out = {}
|
|
for m in re.finditer(r'^\[([^\]]+)\]\r?\n(.*?)(?=^\[|\Z)', txt, re.M | re.S):
|
|
name, body = m.group(1), m.group(2)
|
|
if name.lower() == "includes":
|
|
continue
|
|
kv = dict(re.findall(r'^([A-Za-z_]\w*)=([^\r\n]*)', body, re.M))
|
|
out[name.lower()] = {k.lower(): v.strip().lower().replace("/", "\\")
|
|
for k, v in kv.items()}
|
|
return out
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--show")
|
|
args = ap.parse_args()
|
|
|
|
manifest = subsystems.load_manifest(contents.MANIFEST)
|
|
t = collections.Counter()
|
|
bad = collections.Counter()
|
|
examples = []
|
|
missing_pages = collections.Counter()
|
|
extra_pages = collections.Counter()
|
|
|
|
for d in sorted(glob.glob(REC + "/*")):
|
|
ch = os.path.basename(d)
|
|
src = [p for p in glob.glob(SRC + "/*/*.contents")
|
|
if os.path.basename(p).lower() == ch.lower() + ".contents"]
|
|
if not src:
|
|
continue
|
|
t["chassis"] += 1
|
|
want = source_pages(src[0])
|
|
chassis, pages = contents.decompile(d, manifest)
|
|
got = {n.lower(): {k.lower(): (v or "").lower().replace("/", "\\")
|
|
for k, v in kv} for n, kv in pages}
|
|
|
|
if args.show and args.show.lower() == ch.lower():
|
|
sys.stdout.write(contents.emit(chassis, pages))
|
|
return
|
|
|
|
for name in want:
|
|
if name not in got:
|
|
missing_pages[name] += 1
|
|
for name in got:
|
|
if name not in want:
|
|
extra_pages[name] += 1
|
|
for name in set(want) & set(got):
|
|
t["pages"] += 1
|
|
for key in ("model", "executionstate"):
|
|
t["keys"] += 1
|
|
if want[name].get(key) == got[name].get(key):
|
|
t["ok"] += 1
|
|
else:
|
|
bad[key] += 1
|
|
if len(examples) < 10:
|
|
examples.append((ch, name, key,
|
|
want[name].get(key), got[name].get(key)))
|
|
|
|
print(f"chassis : {t['chassis']}")
|
|
print(f"pages compared : {t['pages']}")
|
|
print(f"keys compared : {t['keys']} exact: {t['ok']} wrong: {t['keys'] - t['ok']}")
|
|
if missing_pages:
|
|
print(f"\npages in source but not decoded: {sum(missing_pages.values())}")
|
|
for n, c in missing_pages.most_common(10):
|
|
print(f" {n:28s} x{c}")
|
|
if extra_pages:
|
|
print(f"\npages decoded but not in source: {sum(extra_pages.values())}")
|
|
for n, c in extra_pages.most_common(10):
|
|
print(f" {n:28s} x{c}")
|
|
if bad:
|
|
print("\nvalue mismatches:")
|
|
for k, c in bad.most_common():
|
|
print(f" {k:20s} x{c}")
|
|
for e in examples:
|
|
print(f" {e[0]:14s} {e[1]:24s} {e[2]:16s} want={e[3]!r} got={e[4]!r}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|