#!/usr/bin/env python3 """Solve the runtime `Unit` (craft/vessel definition) struct layout against the disc's `unit\\UN_*.tbl` records, then read the fields the disc leaves defaulted. Same method as weapon_runtime.py, with two differences forced by the data: * the runtime class is *discovered*, not assumed — see unit_discover.py; the definition objects carry vtable 0x820af844 and hold a pointer to their name record at +0x04 (string at name+0x10), exactly like `Weapon`; * a unit `.tbl` is one object made of several sub-records (Generic, Maneuver, Shield, Explosion, Mass, Effect, SE, Turret_*), all flattened into ONE runtime object, so the disc side is merged per unit ID. Only the *definition* objects are used. The other class found in RAM (0x820af030) is the spawned-entity instance — same ID string, live state, not definition data — and is deliberately ignored. Usage: unit_runtime.py (snapshots are unioned) """ import math import os import re import struct import sys from collections import defaultdict sys.path.insert(0, "/home/fabi/RE - Project Sylpheed/sylpheed-reborn/tools/re-capture") import gmem # noqa: E402 DEF_VTABLE = 0x820AF844 STRIDE = 0x380 SUBRECS = ("Generic", "Maneuver", "Shield", "Explosion", "Mass", "Effect", "SE") # `Maneuver` fields that EVERY unit table leaves at its default, so the solver # can never bind them: there is no disc value anywhere to score against. Their # offsets come from the declaration-order rule (see schema_order.py), which 29 # solved anchors confirm, and each of these sits *between* two of those anchors # -- `YawDragFactor +0x104` .. `ArterBurner_Vc +0x114` and # `ReverseThrust_Vc +0x118` .. `ReverseThrust_Acc +0x120`. Reported separately # and marked `interpolated`, never mixed into the solved set. INTERPOLATED = [ (0x108, "PitchDragFactor", "f32"), (0x10C, "RollDragFactor", "f32"), (0x110, "DragFactorThreshold", "f32"), (0x11C, "ArterBurner_Acc", "f32"), (0x128, "DecPitchFactor", "f32"), ] # ---------------------------------------------------------------- disc side def read_tokens(path): """{unit_id: {field: value}} merged over the tbl's sub-records. A field name that two sub-records of the same unit define with different values is ambiguous *within* the object, so it is dropped rather than silently resolved. """ per_hash = defaultdict(lambda: {"id": None, "f": defaultdict(set)}) cur = None for line in open(path): if line.startswith("REC "): _, h, _, ty, rid = line.split() cur = per_hash[h] if ty == "Generic" and rid.startswith("UN_"): cur["id"] = rid elif line.startswith("F ") and cur is not None: _, k, v = line.rstrip("\n").split(" ", 2) cur["f"][k].add(v) out, ambiguous = {}, {} for rec in per_hash.values(): if not rec["id"]: continue fields, bad = {}, [] for k, vs in rec["f"].items(): if len(vs) == 1: fields[k] = next(iter(vs)) else: bad.append(k) out[rec["id"]] = fields if bad: ambiguous[rec["id"]] = sorted(bad) return out, ambiguous # ------------------------------------------------------------- runtime side def scan_vtable(f, size, vt): pat = struct.pack(">I", vt) hits = [] for a, b in gmem.extents(f.fileno(), size): f.seek(a) blob = f.read(b - a) for m in re.finditer(re.escape(pat), blob): off = a + m.start() if off % 4 == 0: hits.append(off) return sorted(set(hits)) def runtime_objects(paths): """{id: raw} unioned over snapshots; identical IDs must agree byte-for-byte.""" objs, conflicts = {}, [] for path in paths: size = os.path.getsize(path) with open(path, "rb") as f: for off in scan_vtable(f, size, DEF_VTABLE): f.seek(off) raw = f.read(STRIDE) (p,) = struct.unpack_from(">I", raw, 4) try: f.seek(gmem.va_to_off(p + 0x10)) except ValueError: continue nm = f.read(96).split(b"\0")[0] try: nm = nm.decode("ascii") except UnicodeDecodeError: continue if not nm.startswith("UN_"): continue if nm in objs and objs[nm] != raw: conflicts.append((nm, os.path.basename(path))) objs[nm] = raw return objs, conflicts # ------------------------------------------------------------------- solver ENC = { "f32": lambda raw, o: struct.unpack_from(">f", raw, o)[0], "u32": lambda raw, o: float(struct.unpack_from(">I", raw, o)[0]), "i32": lambda raw, o: float(struct.unpack_from(">i", raw, o)[0]), "rad": lambda raw, o: math.degrees(struct.unpack_from(">f", raw, o)[0]), } def near(a, b): if a == b: return True return abs(a - b) <= 2e-4 * max(abs(a), abs(b), 1e-6) def solve(disc, run): numeric = {} for rid, fields in disc.items(): if rid not in run: continue for k, v in fields.items(): try: numeric.setdefault(k, {})[rid] = float(v.rstrip("fF")) except ValueError: pass solved = {} for field, samples in numeric.items(): if len(samples) < 2: continue cands = [] # +0x00 is the vtable and +0x04 the name-record pointer; no field lives # there. They are excluded explicitly because a huge-exponent pointer # word read as f32 is a denormal, which the tolerance test happily # "matches" against any disc value of 0.0. for off in range(8, STRIDE - 3, 4): for enc, fn in ENC.items(): agree, bad = 0, [] for rid, want in samples.items(): got = fn(run[rid], off) if math.isfinite(got) and near(got, want): agree += 1 else: bad.append((rid, want, got)) if agree >= 2: cands.append((len(bad), -agree, off, enc, agree, bad)) if not cands: continue cands.sort() nbad, _, off, enc, agree, bad = cands[0] if len(bad) > 0.2 * len(samples): continue good = {rid for rid in samples} - {b[0] for b in bad} distinct = len({round(samples[r], 6) for r in good}) solved[field] = (off, enc, agree, bad, distinct) # two fields cannot share one byte offset -- and the offset alone is the # identity, not (offset, encoding): the same word read as f32 and as i32 is # still one field. best_at = {} for field, (off, enc, agree, bad, distinct) in solved.items(): score = (distinct, agree, -len(bad)) if off not in best_at or score > best_at[off][1]: best_at[off] = (field, score) winners = {f for f, _ in best_at.values()} return {f: v for f, v in solved.items() if f in winners}, numeric def main(): tokens = sys.argv[1] snaps = [a for a in sys.argv[2:] if not a.startswith("--")] disc, ambiguous = read_tokens(tokens) run, conflicts = runtime_objects(snaps) matched = sorted(set(disc) & set(run)) print(f"# disc units: {len(disc)} runtime objects: {len(run)} matched: {len(matched)}") if conflicts: print(f"# !! objects that differ between snapshots: {conflicts}") unknown = sorted(set(run) - set(disc)) if unknown: print(f"# runtime IDs with no disc record: {unknown}") solved, numeric = solve(disc, run) print(f"# numeric disc fields on matched units: {len(numeric)} solved: {len(solved)}\n") print(f"{'offset':>8} {'enc':>4} {'field':<32} {'agree':>5} {'dist':>4} {'bad':>3}") for field, (off, enc, agree, bad, distinct) in sorted(solved.items(), key=lambda kv: kv[1][0]): print(f"{off:#8x} {enc:>4} {field:<32} {agree:5d} {distinct:4d} {len(bad):3d}") if "--csv" in sys.argv: import csv w = csv.writer(open("unit-runtime-fields.csv", "w", newline="")) w.writerow(["unit", "field", "offset", "enc", "value", "source", "conf"]) for rid in matched: for field, (off, enc, agree, bad, distinct) in sorted( solved.items(), key=lambda kv: kv[1][0] ): conf = "confirmed" if (not bad and agree >= 10 and distinct >= 3) else "tentative" src = "disc" if field in disc[rid] else "defaulted-on-disc" w.writerow([rid, field, f"{off:#05x}", enc, f"{ENC[enc](run[rid], off):g}", src, conf]) for off, field, enc in INTERPOLATED: w.writerow([rid, field, f"{off:#05x}", enc, f"{ENC[enc](run[rid], off):g}", "never-valued-on-disc", "interpolated"]) for arg in sys.argv: if arg.startswith("--unit="): rid = arg.split("=", 1)[1] print(f"\n# every solved field of {rid}") for field, (off, enc, agree, bad, distinct) in sorted( solved.items(), key=lambda kv: kv[1][0] ): conf = "OK " if (not bad and agree >= 10 and distinct >= 3) else "thin" src = "disc" if field in disc.get(rid, {}) else "DEFAULTED" print(f" +{off:#05x} {enc} {conf} {field:<32} " f"{ENC[enc](run[rid], off):>14g} {src}") if "--values" in sys.argv: print("\n# defaulted-on-disc values read from the runtime objects") for field, (off, enc, agree, bad, distinct) in sorted(solved.items(), key=lambda kv: kv[1][0]): if bad: continue miss = [(r, ENC[enc](run[r], off)) for r in matched if field not in disc[r]] if miss: print(f"\n{field} (+{off:#x}, {enc}):") for r, v in miss: print(f" {r:<48} {v:g}") if __name__ == "__main__": main()