Files
Syplheed-Reborn/tools/re-capture/unit_runtime.py
Claude (auto-RE) 1ee2d5e406 re: widen unit coverage to all six tutorials, and prove the fields are run-invariant
Captures the five remaining tutorials (grab_tutorial.sh, one cold boot each)
and re-solves over seven snapshots from seven separate emulator runs. Coverage
18 -> 21 units, confirmed fields 22 -> 27 (Size_Y, MassScore, MinimumVelocity,
AV_PitchPlus_Min, AV_PitchMinus_Min join the zero-contradiction set).

The multi-run union exposed something the single-run check could not: the same
unit's object is NOT byte-identical between runs. unit_runtime.py --crosscheck
pins down why -- exactly 15 words differ, 13 of them holding guest heap
pointers, and none of them is a solved or interpolated field offset. So every
value reported is run-invariant, which is a stronger statement than the
within-run identity check that came before it.

The two non-pointer stragglers are a real caveat, now documented rather than
smoothed over: +0x2c8 reads 8000.0 for UN_f001_TCAF_DeltaSaber_T_Ttrl in two
tutorials and 10000.0 in a third, with a 0/1 flag at +0x2d0. A couple of words
in the object are set per stage, so it is mostly but not entirely the parsed
table. Unidentified, marked NEEDS-HUMAN.

Coverage honesty: all six tutorials together add only 3 units the missions do
not already have -- they reuse one training box, one drone and the player
craft. Further coverage needs story progress, not more tutorials.

Two navigation facts encoded in grab_tutorial.sh, both learned by breaking
them: the main menu is not input-ready for ~10 s after the title tap and early
d-pad presses are dropped (which sends the A to NEW GAME); and NEW GAME is not
a shortcut to Stage 01 -- it gates on DIFFICULTY then plays the prologue. A
NEW GAME excursion to the READY ROOM leaves game01/savedata byte-identical,
verified by diff against a backup.
2026-07-29 18:22:47 +00:00

305 lines
12 KiB
Python

#!/usr/bin/env python3
"""Solve the runtime `Unit` (craft/vessel definition) struct layout against the
disc's `unit\\UN_*.tbl` records, then read the fields the disc leaves defaulted.
Same method as weapon_runtime.py, with two differences forced by the data:
* the runtime class is *discovered*, not assumed — see unit_discover.py; the
definition objects carry vtable 0x820af844 and hold a pointer to their name
record at +0x04 (string at name+0x10), exactly like `Weapon`;
* a unit `.tbl` is one object made of several sub-records (Generic, Maneuver,
Shield, Explosion, Mass, Effect, SE, Turret_*), all flattened into ONE
runtime object, so the disc side is merged per unit ID.
Only the *definition* objects are used. The other class found in RAM
(0x820af030) is the spawned-entity instance — same ID string, live state, not
definition data — and is deliberately ignored.
Usage: unit_runtime.py <tokens.txt> <snapshot...> (snapshots are unioned)
"""
import math
import os
import re
import struct
import sys
from collections import defaultdict
sys.path.insert(0, "/home/fabi/RE - Project Sylpheed/sylpheed-reborn/tools/re-capture")
import gmem # noqa: E402
DEF_VTABLE = 0x820AF844
STRIDE = 0x380
SUBRECS = ("Generic", "Maneuver", "Shield", "Explosion", "Mass", "Effect", "SE")
# `Maneuver` fields that EVERY unit table leaves at its default, so the solver
# can never bind them: there is no disc value anywhere to score against. Their
# offsets come from the declaration-order rule (see schema_order.py), which 29
# solved anchors confirm, and each of these sits *between* two of those anchors
# -- `YawDragFactor +0x104` .. `ArterBurner_Vc +0x114` and
# `ReverseThrust_Vc +0x118` .. `ReverseThrust_Acc +0x120`. Reported separately
# and marked `interpolated`, never mixed into the solved set.
INTERPOLATED = [
(0x108, "PitchDragFactor", "f32"),
(0x10C, "RollDragFactor", "f32"),
(0x110, "DragFactorThreshold", "f32"),
(0x11C, "ArterBurner_Acc", "f32"),
(0x128, "DecPitchFactor", "f32"),
]
# ---------------------------------------------------------------- disc side
def read_tokens(path):
"""{unit_id: {field: value}} merged over the tbl's sub-records.
A field name that two sub-records of the same unit define with different
values is ambiguous *within* the object, so it is dropped rather than
silently resolved.
"""
per_hash = defaultdict(lambda: {"id": None, "f": defaultdict(set)})
cur = None
for line in open(path):
if line.startswith("REC "):
_, h, _, ty, rid = line.split()
cur = per_hash[h]
if ty == "Generic" and rid.startswith("UN_"):
cur["id"] = rid
elif line.startswith("F ") and cur is not None:
_, k, v = line.rstrip("\n").split(" ", 2)
cur["f"][k].add(v)
out, ambiguous = {}, {}
for rec in per_hash.values():
if not rec["id"]:
continue
fields, bad = {}, []
for k, vs in rec["f"].items():
if len(vs) == 1:
fields[k] = next(iter(vs))
else:
bad.append(k)
out[rec["id"]] = fields
if bad:
ambiguous[rec["id"]] = sorted(bad)
return out, ambiguous
# ------------------------------------------------------------- runtime side
def scan_vtable(f, size, vt):
pat = struct.pack(">I", vt)
hits = []
for a, b in gmem.extents(f.fileno(), size):
f.seek(a)
blob = f.read(b - a)
for m in re.finditer(re.escape(pat), blob):
off = a + m.start()
if off % 4 == 0:
hits.append(off)
return sorted(set(hits))
def runtime_objects(paths):
"""{id: raw} unioned over snapshots; identical IDs must agree byte-for-byte."""
objs, conflicts = {}, []
for path in paths:
size = os.path.getsize(path)
with open(path, "rb") as f:
for off in scan_vtable(f, size, DEF_VTABLE):
f.seek(off)
raw = f.read(STRIDE)
(p,) = struct.unpack_from(">I", raw, 4)
try:
f.seek(gmem.va_to_off(p + 0x10))
except ValueError:
continue
nm = f.read(96).split(b"\0")[0]
try:
nm = nm.decode("ascii")
except UnicodeDecodeError:
continue
if not nm.startswith("UN_"):
continue
if nm in objs and objs[nm] != raw:
conflicts.append((nm, os.path.basename(path)))
objs[nm] = raw
return objs, conflicts
# ------------------------------------------------------------------- solver
ENC = {
"f32": lambda raw, o: struct.unpack_from(">f", raw, o)[0],
"u32": lambda raw, o: float(struct.unpack_from(">I", raw, o)[0]),
"i32": lambda raw, o: float(struct.unpack_from(">i", raw, o)[0]),
"rad": lambda raw, o: math.degrees(struct.unpack_from(">f", raw, o)[0]),
}
def near(a, b):
if a == b:
return True
return abs(a - b) <= 2e-4 * max(abs(a), abs(b), 1e-6)
def solve(disc, run):
numeric = {}
for rid, fields in disc.items():
if rid not in run:
continue
for k, v in fields.items():
try:
numeric.setdefault(k, {})[rid] = float(v.rstrip("fF"))
except ValueError:
pass
solved = {}
for field, samples in numeric.items():
if len(samples) < 2:
continue
cands = []
# +0x00 is the vtable and +0x04 the name-record pointer; no field lives
# there. They are excluded explicitly because a huge-exponent pointer
# word read as f32 is a denormal, which the tolerance test happily
# "matches" against any disc value of 0.0.
for off in range(8, STRIDE - 3, 4):
for enc, fn in ENC.items():
agree, bad = 0, []
for rid, want in samples.items():
got = fn(run[rid], off)
if math.isfinite(got) and near(got, want):
agree += 1
else:
bad.append((rid, want, got))
if agree >= 2:
cands.append((len(bad), -agree, off, enc, agree, bad))
if not cands:
continue
cands.sort()
nbad, _, off, enc, agree, bad = cands[0]
if len(bad) > 0.2 * len(samples):
continue
good = {rid for rid in samples} - {b[0] for b in bad}
distinct = len({round(samples[r], 6) for r in good})
solved[field] = (off, enc, agree, bad, distinct)
# two fields cannot share one byte offset -- and the offset alone is the
# identity, not (offset, encoding): the same word read as f32 and as i32 is
# still one field.
best_at = {}
for field, (off, enc, agree, bad, distinct) in solved.items():
score = (distinct, agree, -len(bad))
if off not in best_at or score > best_at[off][1]:
best_at[off] = (field, score)
winners = {f for f, _ in best_at.values()}
return {f: v for f, v in solved.items() if f in winners}, numeric
def crosscheck(paths, solved):
"""Which words of a unit's object vary between snapshots, and are any of
them fields we claim to have solved?
Within one run the objects are byte-identical, but across runs the same
unit differs -- pointer words hold heap addresses, which move. This
separates "the emulator relocated it" from "the value is not definition
data", and it is the check that keeps the solved set honest.
"""
per = {}
for p in paths:
for k, v in runtime_objects([p])[0].items():
per.setdefault(k, {})[os.path.basename(p)] = v
diff = set()
shared = 0
for k, d in per.items():
if len(d) < 2:
continue
shared += 1
vs = list(d.values())
for v in vs[1:]:
diff |= {i // 4 * 4 for i in range(min(len(v), len(vs[0]))) if v[i] != vs[0][i]}
def looks_ptr(off):
vals = [struct.unpack_from(">I", v, off)[0]
for d in per.values() for v in d.values() if len(v) > off + 3]
return sum(1 for x in vals if 0x80000000 <= x < 0xC0000000), len(vals)
print(f"\n# cross-run check: {shared} unit(s) appear in more than one snapshot")
print(f"# words that differ between runs: {len(diff)}")
for off in sorted(diff):
p, n = looks_ptr(off)
kind = "guest pointer" if p == n else ("mixed" if p else "non-pointer")
print(f"# +{off:#05x} {kind} ({p}/{n} look like pointers)")
offs = {o for o, _, _, _, _ in solved.values()} | {o for o, _, _ in INTERPOLATED}
clash = sorted(offs & diff)
print("# solved/interpolated offsets that vary between runs: "
+ (", ".join(f"{o:#05x}" for o in clash) if clash
else "NONE — every reported field is run-invariant"))
def main():
tokens = sys.argv[1]
snaps = [a for a in sys.argv[2:] if not a.startswith("--")]
disc, ambiguous = read_tokens(tokens)
run, conflicts = runtime_objects(snaps)
matched = sorted(set(disc) & set(run))
print(f"# disc units: {len(disc)} runtime objects: {len(run)} matched: {len(matched)}")
if conflicts:
print(f"# !! objects that differ between snapshots: {conflicts}")
unknown = sorted(set(run) - set(disc))
if unknown:
print(f"# runtime IDs with no disc record: {unknown}")
solved, numeric = solve(disc, run)
print(f"# numeric disc fields on matched units: {len(numeric)} solved: {len(solved)}\n")
print(f"{'offset':>8} {'enc':>4} {'field':<32} {'agree':>5} {'dist':>4} {'bad':>3}")
for field, (off, enc, agree, bad, distinct) in sorted(solved.items(), key=lambda kv: kv[1][0]):
print(f"{off:#8x} {enc:>4} {field:<32} {agree:5d} {distinct:4d} {len(bad):3d}")
if "--crosscheck" in sys.argv:
crosscheck(snaps, solved)
if "--csv" in sys.argv:
import csv
w = csv.writer(open("unit-runtime-fields.csv", "w", newline=""))
w.writerow(["unit", "field", "offset", "enc", "value", "source", "conf"])
for rid in matched:
for field, (off, enc, agree, bad, distinct) in sorted(
solved.items(), key=lambda kv: kv[1][0]
):
conf = "confirmed" if (not bad and agree >= 10 and distinct >= 3) else "tentative"
src = "disc" if field in disc[rid] else "defaulted-on-disc"
w.writerow([rid, field, f"{off:#05x}", enc,
f"{ENC[enc](run[rid], off):g}", src, conf])
for off, field, enc in INTERPOLATED:
w.writerow([rid, field, f"{off:#05x}", enc,
f"{ENC[enc](run[rid], off):g}",
"never-valued-on-disc", "interpolated"])
for arg in sys.argv:
if arg.startswith("--unit="):
rid = arg.split("=", 1)[1]
print(f"\n# every solved field of {rid}")
for field, (off, enc, agree, bad, distinct) in sorted(
solved.items(), key=lambda kv: kv[1][0]
):
conf = "OK " if (not bad and agree >= 10 and distinct >= 3) else "thin"
src = "disc" if field in disc.get(rid, {}) else "DEFAULTED"
print(f" +{off:#05x} {enc} {conf} {field:<32} "
f"{ENC[enc](run[rid], off):>14g} {src}")
if "--values" in sys.argv:
print("\n# defaulted-on-disc values read from the runtime objects")
for field, (off, enc, agree, bad, distinct) in sorted(solved.items(), key=lambda kv: kv[1][0]):
if bad:
continue
miss = [(r, ENC[enc](run[r], off)) for r in matched if field not in disc[r]]
if miss:
print(f"\n{field} (+{off:#x}, {enc}):")
for r, v in miss:
print(f" {r:<48} {v:g}")
if __name__ == "__main__":
main()