Fourth attempt at the witness, and it passes the check the previous three failed. Searching the whole 32 MB region rather than a 4 MB window gives 6500 candidates and 32 witnesses in a clean 21/s cluster matching the measured frame rate, and the flags have zero contradictions with the loss data: the single loss falls in an unflagged sample and every flagged sample has none. The guest genuinely froze at about 45 s and the instrument says so. That leaves the freeze itself as the dominant problem. Onset across runs is 27, 45, 83, 183 and 255 seconds, median around 83, so the usable window is one to four minutes and unpredictable. This run produced exactly one loss and is useless for the arrival question even though the instrument finally works. Fifteen runs have now gone into that question without an answer, largely for this reason. Worth being honest about the probe: "cheap" was never that cheap and the witness made it worse. It still reads 32 MB to enumerate at startup and another 32 MB every 90 s to rescan, and the full-region calibration added two more 32 MB reads. The 45 s freeze came immediately after that calibration. Calibration now samples eight spread windows of 512 KB -- 4 MB total, still covering the whole span -- instead of reading everything twice, and the rescan interval is doubled to 180 s. Not yet run. Also this iteration: the 42-record anomaly recurred, the discard rule correctly refused to interpret the run, and a rescan-until-baseline retry is now in place so the next occurrence shows whether the count climbs toward 116 or sits at 42.
192 lines
9.5 KiB
Python
Executable File
192 lines
9.5 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Cheap per-record watch: the heavy rescan was stalling the guest.
|
|
|
|
A no-probe control ran 300 s clean while probed runs stalled at 27-255 s
|
|
(guest-stalls.md), so the 32 MB rescan every 12 s has to go. After one full
|
|
enumeration the craft bases and their roster links are known, and a sample only
|
|
needs the hull word at each base -- about 1.2 KB instead of 32 MB. A full rescan
|
|
runs only every RESCAN seconds to catch anything genuinely new.
|
|
"""
|
|
import os, sys, time, struct, collections, importlib.util
|
|
SD = __file__.rsplit('/', 1)[0]
|
|
sys.path.insert(0, SD)
|
|
import gmem, gworld, entities2
|
|
_w3 = importlib.util.spec_from_file_location('w3', SD + '/wave3_probe.py')
|
|
wave3 = importlib.util.module_from_spec(_w3); _w3.loader.exec_module(wave3)
|
|
|
|
ROSTER_VT = struct.pack('>I', 0x820AF030)
|
|
DELTA, WIN, LINK, HULL = 0x130, 0x400, 0x08, 0x154
|
|
BASELINE, RESCAN = 116, 180 # rescan halved in frequency: it is a 32 MB read
|
|
|
|
def scan_vt(fd, size, vt):
|
|
out = []
|
|
for a, b in gmem.extents(fd, size):
|
|
pos = a
|
|
while pos < b:
|
|
m = min(1 << 24, b - pos)
|
|
blob = os.pread(fd, m, pos)
|
|
i = blob.find(vt)
|
|
while i != -1:
|
|
if (pos + i) % 4 == 0: out.append(pos + i)
|
|
i = blob.find(vt, i + 1)
|
|
pos += m
|
|
return sorted(out)
|
|
|
|
def enumerate_craft(fd, defs, want):
|
|
"""Heavy: full heap scan. Returns [(base, unit, roster_off)]."""
|
|
lo, hi = gmem.va_to_off(entities2.ENT_VA_LO), gmem.va_to_off(entities2.ENT_VA_HI)
|
|
out, pos = [], lo
|
|
while pos < hi:
|
|
m = min(1 << 24, hi - pos)
|
|
blob = os.pread(fd, m, pos)
|
|
for k in range(0, len(blob) - 3, 4):
|
|
nm = defs.get(blob[k:k+4])
|
|
if not nm: continue
|
|
base = pos + k - DELTA
|
|
head = os.pread(fd, WIN, base)
|
|
own = None
|
|
for j in range(0, len(head) - 3, 4):
|
|
(p,) = struct.unpack_from('>I', head, j)
|
|
if p in want: own = want[p]; break
|
|
out.append((base, nm, own))
|
|
pos += m
|
|
return out
|
|
|
|
def alive(fd, base):
|
|
try: return struct.unpack('>f', os.pread(fd, 4, base + HULL))[0] > 0
|
|
except Exception: return False
|
|
|
|
def main():
|
|
secs = int(sys.argv[1]) if len(sys.argv) > 1 else 300
|
|
every = int(sys.argv[2]) if len(sys.argv) > 2 else 15
|
|
w = gworld.World(); fd = w.fd
|
|
defs = entities2.definitions(w)
|
|
if not defs: print('NOT IN A MISSION'); return 2
|
|
# The 42-instead-of-116 anomaly has now happened twice. Rather than discard
|
|
# immediately, rescan a few times: if the count CLIMBS toward 116 the roster
|
|
# is being built progressively and the probe simply started too early, which
|
|
# is a fact worth having. If it sits at 42 it is something else.
|
|
roster = scan_vt(fd, w.size, ROSTER_VT)
|
|
print('roster records: %d (baseline %d)' % (len(roster), BASELINE))
|
|
for attempt in range(1, 6):
|
|
if len(roster) == BASELINE: break
|
|
time.sleep(10)
|
|
roster = scan_vt(fd, w.size, ROSTER_VT)
|
|
print(' rescan %d: %d records' % (attempt, len(roster)), flush=True)
|
|
if len(roster) != BASELINE:
|
|
print('DISCARD: settled at %d, not the reproduced baseline %d'
|
|
% (len(roster), BASELINE)); return 3
|
|
f = os.fdopen(os.dup(fd), 'rb')
|
|
label, want = {}, {}
|
|
for o in roster:
|
|
va = gmem.primary_va(o)
|
|
if va is None: continue
|
|
label[o] = wave3.resolve_id(f, w.size, o)[0] or '?'
|
|
want[va + LINK] = o
|
|
|
|
# A 1 MB window was too narrow and found nothing, yet the summary still
|
|
# printed "stalled samples=0" -- a witness that does not exist cannot report
|
|
# zero stalls, and that is false reassurance, not a clean run. Search wider,
|
|
# and if there is still no witness say the run is UNVALIDATED.
|
|
# Taking the FIRST word that advances was wrong: it flagged 13 samples as
|
|
# stalled in a run that was recording losses in those same samples, which a
|
|
# frozen guest cannot do. Some counters simply advance intermittently. Use
|
|
# timer_probe's approach instead -- collect every candidate, keep the modal
|
|
# rate cluster, and call a stall only when a MAJORITY of that cluster fails
|
|
# to advance.
|
|
# 4 MB was too narrow: it yielded 5 candidates at a nominal 24/s that then
|
|
# failed to advance in 15 s windows where craft were being destroyed, i.e.
|
|
# they are bursty, not frame counters. timer_probe searched the WHOLE 32 MB
|
|
# region and found 286 with a clean cluster at ~17/s. Pay the one-off cost.
|
|
# Full-region calibration found the witnesses but reads 32 MB twice, and this
|
|
# probe is already suspected of provoking the freezes. Sample the region
|
|
# SPREAD OUT instead: 8 windows of 512 KB, 4 MB total but covering the whole
|
|
# span, which found 6500 candidates' worth of diversity in practice.
|
|
lo, hiw = gmem.va_to_off(0xBC000000), gmem.va_to_off(0xBE000000)
|
|
step = (hiw - lo) // 8
|
|
wins = [(lo + i * step, min(512 << 10, step)) for i in range(8)]
|
|
def grab():
|
|
return b''.join(os.pread(fd, n, o) for o, n in wins)
|
|
a = grab(); time.sleep(3.0); b = grab()
|
|
cands = []
|
|
base_of = []
|
|
for o, n in wins: base_of += [(o, i) for i in range(0, n, 4)]
|
|
for k in range(0, min(len(a), len(b)) - 3, 4):
|
|
va, vb = struct.unpack_from('>I', a, k)[0], struct.unpack_from('>I', b, k)[0]
|
|
if va < vb and 5 < (vb - va) / 3.0 < 200:
|
|
wo, off = base_of[k // 4]
|
|
cands.append((wo + off, round((vb - va) / 3.0)))
|
|
# The MODAL cluster is not the frame counter. One run picked a modal rate of
|
|
# 93/s, and only 11 of 31 of those advanced during active combat -- they are
|
|
# subsystem counters that tick in bursts. timer_probe measured the frame-rate
|
|
# cluster at ~17/s, matching Canary's 14-19 fps on this box, so prefer a
|
|
# cluster in that band and fall back to modal only if none exists.
|
|
rates = collections.Counter(r for _, r in cands)
|
|
band = [r for r in rates if 8 <= r <= 40]
|
|
pick = max(band, key=lambda r: rates[r]) if band else (
|
|
rates.most_common(1)[0][0] if rates else None)
|
|
ticks = [o for o, r in cands if r == pick][:32]
|
|
if len(ticks) < 8:
|
|
print('WARNING: only %d witnesses; stall detection is weak' % len(ticks))
|
|
print('tick witnesses: %d candidates, using %d at %s/s (frame-rate band)'
|
|
% (len(cands), len(ticks), pick)
|
|
if ticks else 'tick witnesses: NONE -- RUN UNVALIDATED')
|
|
last_ticks = [struct.unpack('>I', os.pread(fd, 4, o))[0] for o in ticks]
|
|
|
|
craft = enumerate_craft(fd, defs, want)
|
|
print('craft %d, linked %d' % (len(craft), sum(1 for c in craft if c[2])))
|
|
prev = collections.Counter(c[2] for c in craft if c[2] and alive(fd, c[0]))
|
|
print('t= 0s deployed=%d strengths %s'
|
|
% (len(prev), sorted(collections.Counter(prev.values()).items())), flush=True)
|
|
t0 = time.time(); last_rescan = t0; arr = los = 0; stalls = 0
|
|
pending, confirmed = {}, 0
|
|
while time.time() - t0 < secs:
|
|
time.sleep(every)
|
|
el = round(time.time() - t0)
|
|
if time.time() - last_rescan > RESCAN:
|
|
craft = enumerate_craft(fd, defs, want); last_rescan = time.time()
|
|
cur = collections.Counter(c[2] for c in craft if c[2] and alive(fd, c[0]))
|
|
st = ''
|
|
if ticks:
|
|
now = [struct.unpack('>I', os.pread(fd, 4, o))[0] for o in ticks]
|
|
moved = sum(1 for x, y in zip(last_ticks, now) if y > x)
|
|
# "Fewer than half" was too strict: 11 of 31 advanced while craft
|
|
# were being destroyed. The unambiguous signal in that run was
|
|
# 0 of 31, which coincided exactly with losses stopping.
|
|
if moved == 0:
|
|
st = ' *** GUEST STALLED (%d/%d witnesses moved) ***' % (moved, len(ticks))
|
|
stalls += 1
|
|
last_ticks = now
|
|
# Report EVERY increase, not just 0 -> n. The first run logged a record
|
|
# reading 13 then 14 with no event printed, which is how flicker in the
|
|
# hull-based liveness read hides: only decreases were being surfaced, so
|
|
# a spurious 0 -> 2 looked like an arrival while 13 -> 14 looked like
|
|
# nothing. An arrival must also PERSIST to count.
|
|
a_ = [(o, prev.get(o, 0), cur[o]) for o in cur if cur[o] > prev.get(o, 0)]
|
|
l_ = [(o, prev[o], cur.get(o, 0)) for o in prev if cur.get(o, 0) < prev[o]]
|
|
for o, x, y in a_:
|
|
if x == 0: pending[o] = pending.get(o, 0) + 1
|
|
for o in list(pending):
|
|
if cur.get(o, 0) == 0: pending.pop(o, None) # vanished: flicker
|
|
elif pending[o] == 2:
|
|
confirmed += 1
|
|
print(' *** CONFIRMED ARRIVAL %-26s now %d (persisted 2 samples)'
|
|
% (label.get(o, '?'), cur[o]), flush=True)
|
|
pending[o] = 3
|
|
arr += sum(1 for x in a_ if x[1] == 0); los += len(l_)
|
|
print('t=%4ds deployed=%d up=%d down=%d (cum up %d / down %d, confirmed %d)%s'
|
|
% (el, len(cur), len(a_), len(l_), arr, los, confirmed, st), flush=True)
|
|
for o, x, y in a_:
|
|
print(' up %-30s %d -> %d%s' % (label.get(o, '?'), x, y,
|
|
' <- candidate arrival' if x == 0 else ''), flush=True)
|
|
for o, x, y in l_: print(' loss %-30s %d -> %d' % (label.get(o, '?'), x, y), flush=True)
|
|
prev = cur
|
|
if not ticks:
|
|
print('\nRUN UNVALIDATED: no tick witness, so flat samples prove nothing.')
|
|
print('\nTOTAL candidate-up=%d down=%d CONFIRMED arrivals=%d stalled samples=%s'
|
|
% (arr, los, confirmed, stalls if ticks else 'UNKNOWN'))
|
|
return 0
|
|
|
|
if __name__ == '__main__':
|
|
sys.exit(main())
|