The graded output is the improvement. Instead of a single bit it reports how many witnesses agree, and the sequence tells a coherent story: 11, 9, 7, 4, 1, then 0 of 31, with the drop to zero at t=183 s coinciding exactly with the last loss and 106 s of nothing after it. That is a real freeze, identified. The threshold was wrong though. Flagging a stall at "fewer than half" marked the entire run stalled, including samples in which craft were destroyed, so 11 of 31 advancing is a healthy guest rather than a stalled one. The cause is the cluster choice: the modal rate was 93/s, far above the ~16.5/s frame rate timer_probe measured, and those are subsystem counters that tick in bursts and sit idle in most 15 s windows even while the game runs. Picking the modal cluster was convenient rather than principled. Fixed to prefer the cluster whose rate falls in the frame-rate band of 8-40/s, falling back to modal only if none exists, and to flag a stall only when zero witnesses advance, which is the signal the data actually supports. Not yet run. The uncomfortable part: this run used the cheap probe and still froze, at about 183 s. The previous iteration's "0 stalled samples" came from the unreliable single-word witness and cannot stand as validation. What the evidence supports now is that the no-probe control ran 300 s clean, the heavy probe froze at 27 to 255 s, and the cheap probe froze at 183 s -- one run on each arm. Cheap sampling plausibly helps but does not remove the freeze, and it is equally possible the freeze is stochastic and the control was lucky. Recorded as unresolved rather than resolved in the probe's favour. Practical consequence: the usable window is roughly three minutes per run, sometimes less, whether or not the probe is cheap. Experiments needing longer have to survive a freeze or be redesigned around one.
165 lines
7.9 KiB
Python
Executable File
165 lines
7.9 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Cheap per-record watch: the heavy rescan was stalling the guest.
|
|
|
|
A no-probe control ran 300 s clean while probed runs stalled at 27-255 s
|
|
(guest-stalls.md), so the 32 MB rescan every 12 s has to go. After one full
|
|
enumeration the craft bases and their roster links are known, and a sample only
|
|
needs the hull word at each base -- about 1.2 KB instead of 32 MB. A full rescan
|
|
runs only every RESCAN seconds to catch anything genuinely new.
|
|
"""
|
|
import os, sys, time, struct, collections, importlib.util
|
|
SD = __file__.rsplit('/', 1)[0]
|
|
sys.path.insert(0, SD)
|
|
import gmem, gworld, entities2
|
|
_w3 = importlib.util.spec_from_file_location('w3', SD + '/wave3_probe.py')
|
|
wave3 = importlib.util.module_from_spec(_w3); _w3.loader.exec_module(wave3)
|
|
|
|
ROSTER_VT = struct.pack('>I', 0x820AF030)
|
|
DELTA, WIN, LINK, HULL = 0x130, 0x400, 0x08, 0x154
|
|
BASELINE, RESCAN = 116, 90
|
|
|
|
def scan_vt(fd, size, vt):
|
|
out = []
|
|
for a, b in gmem.extents(fd, size):
|
|
pos = a
|
|
while pos < b:
|
|
m = min(1 << 24, b - pos)
|
|
blob = os.pread(fd, m, pos)
|
|
i = blob.find(vt)
|
|
while i != -1:
|
|
if (pos + i) % 4 == 0: out.append(pos + i)
|
|
i = blob.find(vt, i + 1)
|
|
pos += m
|
|
return sorted(out)
|
|
|
|
def enumerate_craft(fd, defs, want):
|
|
"""Heavy: full heap scan. Returns [(base, unit, roster_off)]."""
|
|
lo, hi = gmem.va_to_off(entities2.ENT_VA_LO), gmem.va_to_off(entities2.ENT_VA_HI)
|
|
out, pos = [], lo
|
|
while pos < hi:
|
|
m = min(1 << 24, hi - pos)
|
|
blob = os.pread(fd, m, pos)
|
|
for k in range(0, len(blob) - 3, 4):
|
|
nm = defs.get(blob[k:k+4])
|
|
if not nm: continue
|
|
base = pos + k - DELTA
|
|
head = os.pread(fd, WIN, base)
|
|
own = None
|
|
for j in range(0, len(head) - 3, 4):
|
|
(p,) = struct.unpack_from('>I', head, j)
|
|
if p in want: own = want[p]; break
|
|
out.append((base, nm, own))
|
|
pos += m
|
|
return out
|
|
|
|
def alive(fd, base):
|
|
try: return struct.unpack('>f', os.pread(fd, 4, base + HULL))[0] > 0
|
|
except Exception: return False
|
|
|
|
def main():
|
|
secs = int(sys.argv[1]) if len(sys.argv) > 1 else 300
|
|
every = int(sys.argv[2]) if len(sys.argv) > 2 else 15
|
|
w = gworld.World(); fd = w.fd
|
|
defs = entities2.definitions(w)
|
|
if not defs: print('NOT IN A MISSION'); return 2
|
|
roster = scan_vt(fd, w.size, ROSTER_VT)
|
|
print('roster records: %d (baseline %d)' % (len(roster), BASELINE))
|
|
if len(roster) != BASELINE:
|
|
print('DISCARD: not the reproduced baseline'); return 3
|
|
f = os.fdopen(os.dup(fd), 'rb')
|
|
label, want = {}, {}
|
|
for o in roster:
|
|
va = gmem.primary_va(o)
|
|
if va is None: continue
|
|
label[o] = wave3.resolve_id(f, w.size, o)[0] or '?'
|
|
want[va + LINK] = o
|
|
|
|
# A 1 MB window was too narrow and found nothing, yet the summary still
|
|
# printed "stalled samples=0" -- a witness that does not exist cannot report
|
|
# zero stalls, and that is false reassurance, not a clean run. Search wider,
|
|
# and if there is still no witness say the run is UNVALIDATED.
|
|
# Taking the FIRST word that advances was wrong: it flagged 13 samples as
|
|
# stalled in a run that was recording losses in those same samples, which a
|
|
# frozen guest cannot do. Some counters simply advance intermittently. Use
|
|
# timer_probe's approach instead -- collect every candidate, keep the modal
|
|
# rate cluster, and call a stall only when a MAJORITY of that cluster fails
|
|
# to advance.
|
|
lo = gmem.va_to_off(0xBC000000)
|
|
a = os.pread(fd, 1 << 22, lo); time.sleep(3.0); b = os.pread(fd, 1 << 22, lo)
|
|
cands = []
|
|
for k in range(0, min(len(a), len(b)) - 3, 4):
|
|
va, vb = struct.unpack_from('>I', a, k)[0], struct.unpack_from('>I', b, k)[0]
|
|
if va < vb and 5 < (vb - va) / 3.0 < 200:
|
|
cands.append((lo + k, round((vb - va) / 3.0)))
|
|
# The MODAL cluster is not the frame counter. One run picked a modal rate of
|
|
# 93/s, and only 11 of 31 of those advanced during active combat -- they are
|
|
# subsystem counters that tick in bursts. timer_probe measured the frame-rate
|
|
# cluster at ~17/s, matching Canary's 14-19 fps on this box, so prefer a
|
|
# cluster in that band and fall back to modal only if none exists.
|
|
rates = collections.Counter(r for _, r in cands)
|
|
band = [r for r in rates if 8 <= r <= 40]
|
|
pick = max(band, key=lambda r: rates[r]) if band else (
|
|
rates.most_common(1)[0][0] if rates else None)
|
|
ticks = [o for o, r in cands if r == pick][:32]
|
|
print('tick witnesses: %d candidates, using %d at %s/s (frame-rate band)'
|
|
% (len(cands), len(ticks), pick)
|
|
if ticks else 'tick witnesses: NONE -- RUN UNVALIDATED')
|
|
last_ticks = [struct.unpack('>I', os.pread(fd, 4, o))[0] for o in ticks]
|
|
|
|
craft = enumerate_craft(fd, defs, want)
|
|
print('craft %d, linked %d' % (len(craft), sum(1 for c in craft if c[2])))
|
|
prev = collections.Counter(c[2] for c in craft if c[2] and alive(fd, c[0]))
|
|
print('t= 0s deployed=%d strengths %s'
|
|
% (len(prev), sorted(collections.Counter(prev.values()).items())), flush=True)
|
|
t0 = time.time(); last_rescan = t0; arr = los = 0; stalls = 0
|
|
pending, confirmed = {}, 0
|
|
while time.time() - t0 < secs:
|
|
time.sleep(every)
|
|
el = round(time.time() - t0)
|
|
if time.time() - last_rescan > RESCAN:
|
|
craft = enumerate_craft(fd, defs, want); last_rescan = time.time()
|
|
cur = collections.Counter(c[2] for c in craft if c[2] and alive(fd, c[0]))
|
|
st = ''
|
|
if ticks:
|
|
now = [struct.unpack('>I', os.pread(fd, 4, o))[0] for o in ticks]
|
|
moved = sum(1 for x, y in zip(last_ticks, now) if y > x)
|
|
# "Fewer than half" was too strict: 11 of 31 advanced while craft
|
|
# were being destroyed. The unambiguous signal in that run was
|
|
# 0 of 31, which coincided exactly with losses stopping.
|
|
if moved == 0:
|
|
st = ' *** GUEST STALLED (%d/%d witnesses moved) ***' % (moved, len(ticks))
|
|
stalls += 1
|
|
last_ticks = now
|
|
# Report EVERY increase, not just 0 -> n. The first run logged a record
|
|
# reading 13 then 14 with no event printed, which is how flicker in the
|
|
# hull-based liveness read hides: only decreases were being surfaced, so
|
|
# a spurious 0 -> 2 looked like an arrival while 13 -> 14 looked like
|
|
# nothing. An arrival must also PERSIST to count.
|
|
a_ = [(o, prev.get(o, 0), cur[o]) for o in cur if cur[o] > prev.get(o, 0)]
|
|
l_ = [(o, prev[o], cur.get(o, 0)) for o in prev if cur.get(o, 0) < prev[o]]
|
|
for o, x, y in a_:
|
|
if x == 0: pending[o] = pending.get(o, 0) + 1
|
|
for o in list(pending):
|
|
if cur.get(o, 0) == 0: pending.pop(o, None) # vanished: flicker
|
|
elif pending[o] == 2:
|
|
confirmed += 1
|
|
print(' *** CONFIRMED ARRIVAL %-26s now %d (persisted 2 samples)'
|
|
% (label.get(o, '?'), cur[o]), flush=True)
|
|
pending[o] = 3
|
|
arr += sum(1 for x in a_ if x[1] == 0); los += len(l_)
|
|
print('t=%4ds deployed=%d up=%d down=%d (cum up %d / down %d, confirmed %d)%s'
|
|
% (el, len(cur), len(a_), len(l_), arr, los, confirmed, st), flush=True)
|
|
for o, x, y in a_:
|
|
print(' up %-30s %d -> %d%s' % (label.get(o, '?'), x, y,
|
|
' <- candidate arrival' if x == 0 else ''), flush=True)
|
|
for o, x, y in l_: print(' loss %-30s %d -> %d' % (label.get(o, '?'), x, y), flush=True)
|
|
prev = cur
|
|
if not ticks:
|
|
print('\nRUN UNVALIDATED: no tick witness, so flat samples prove nothing.')
|
|
print('\nTOTAL candidate-up=%d down=%d CONFIRMED arrivals=%d stalled samples=%s'
|
|
% (arr, los, confirmed, stalls if ticks else 'UNKNOWN'))
|
|
return 0
|
|
|
|
if __name__ == '__main__':
|
|
sys.exit(main())
|