This repository has been archived on 2026-09-16. You can view files and clone it. You cannot open issues or pull requests or push a commit.
Files
Syplheed-Reborn/tools/re-capture/wave7_probe.py
Sylpheed RE agent 015fb7d21e re: persistence rule works; the stall witness gives false positives
Run 12, cheap probe with a bound pilot: 16 losses over 290 s and zero confirmed
arrivals, making twelve runs without one. The persistence rule earned its place
immediately -- an increase of 13 to 15 was surfaced and correctly not counted,
since it does not start from zero. Under the previous rule it would have been
invisible, and a similar flicker straddling zero was nearly written up last
iteration as the first arrival.

The stall witness, on the other hand, is unreliable. Thirteen samples were
flagged GUEST STALLED while recording losses in those same samples, and a frozen
guest cannot destroy craft, so they are false positives and the run was healthy.

The cause is the selection rule: it took the first word in a 4 MB window whose
rate fell in a plausible band, and plenty of counters advance intermittently
without saying anything about whether frames are being rendered. timer_probe had
already solved this properly -- 286 candidates, a rate histogram with a dominant
cluster near 17/s -- and that lesson was not carried over when the witness was
bolted onto the probe.

Now fixed to a majority vote: collect every candidate, keep the modal-rate
cluster, sample up to 32 of them, and report a stall only when fewer than half
advance. It also prints RUN UNVALIDATED when no witness is found, because an
earlier run printed "stalled samples=0" alongside "tick witness: NONE", and a
witness that does not exist cannot report zero stalls. Not yet run.

Consequence worth flagging: the "0 stalled samples" that validated the cheap
probe last iteration came from this same unreliable witness and should be
re-confirmed under the majority rule. The pilot-log speed analysis that
established the stalls in the first place is unaffected.

Also fixed: the entity bind now retries three times and aborts if it never
takes, instead of silently flying an unattended craft -- one run was wasted that
way this iteration, producing no kills and no information.
2026-08-24 16:54:52 +00:00

154 lines
7.2 KiB
Python
Executable File

#!/usr/bin/env python3
"""Cheap per-record watch: the heavy rescan was stalling the guest.
A no-probe control ran 300 s clean while probed runs stalled at 27-255 s
(guest-stalls.md), so the 32 MB rescan every 12 s has to go. After one full
enumeration the craft bases and their roster links are known, and a sample only
needs the hull word at each base -- about 1.2 KB instead of 32 MB. A full rescan
runs only every RESCAN seconds to catch anything genuinely new.
"""
import os, sys, time, struct, collections, importlib.util
SD = __file__.rsplit('/', 1)[0]
sys.path.insert(0, SD)
import gmem, gworld, entities2
_w3 = importlib.util.spec_from_file_location('w3', SD + '/wave3_probe.py')
wave3 = importlib.util.module_from_spec(_w3); _w3.loader.exec_module(wave3)
ROSTER_VT = struct.pack('>I', 0x820AF030)
DELTA, WIN, LINK, HULL = 0x130, 0x400, 0x08, 0x154
BASELINE, RESCAN = 116, 90
def scan_vt(fd, size, vt):
out = []
for a, b in gmem.extents(fd, size):
pos = a
while pos < b:
m = min(1 << 24, b - pos)
blob = os.pread(fd, m, pos)
i = blob.find(vt)
while i != -1:
if (pos + i) % 4 == 0: out.append(pos + i)
i = blob.find(vt, i + 1)
pos += m
return sorted(out)
def enumerate_craft(fd, defs, want):
"""Heavy: full heap scan. Returns [(base, unit, roster_off)]."""
lo, hi = gmem.va_to_off(entities2.ENT_VA_LO), gmem.va_to_off(entities2.ENT_VA_HI)
out, pos = [], lo
while pos < hi:
m = min(1 << 24, hi - pos)
blob = os.pread(fd, m, pos)
for k in range(0, len(blob) - 3, 4):
nm = defs.get(blob[k:k+4])
if not nm: continue
base = pos + k - DELTA
head = os.pread(fd, WIN, base)
own = None
for j in range(0, len(head) - 3, 4):
(p,) = struct.unpack_from('>I', head, j)
if p in want: own = want[p]; break
out.append((base, nm, own))
pos += m
return out
def alive(fd, base):
try: return struct.unpack('>f', os.pread(fd, 4, base + HULL))[0] > 0
except Exception: return False
def main():
secs = int(sys.argv[1]) if len(sys.argv) > 1 else 300
every = int(sys.argv[2]) if len(sys.argv) > 2 else 15
w = gworld.World(); fd = w.fd
defs = entities2.definitions(w)
if not defs: print('NOT IN A MISSION'); return 2
roster = scan_vt(fd, w.size, ROSTER_VT)
print('roster records: %d (baseline %d)' % (len(roster), BASELINE))
if len(roster) != BASELINE:
print('DISCARD: not the reproduced baseline'); return 3
f = os.fdopen(os.dup(fd), 'rb')
label, want = {}, {}
for o in roster:
va = gmem.primary_va(o)
if va is None: continue
label[o] = wave3.resolve_id(f, w.size, o)[0] or '?'
want[va + LINK] = o
# A 1 MB window was too narrow and found nothing, yet the summary still
# printed "stalled samples=0" -- a witness that does not exist cannot report
# zero stalls, and that is false reassurance, not a clean run. Search wider,
# and if there is still no witness say the run is UNVALIDATED.
# Taking the FIRST word that advances was wrong: it flagged 13 samples as
# stalled in a run that was recording losses in those same samples, which a
# frozen guest cannot do. Some counters simply advance intermittently. Use
# timer_probe's approach instead -- collect every candidate, keep the modal
# rate cluster, and call a stall only when a MAJORITY of that cluster fails
# to advance.
lo = gmem.va_to_off(0xBC000000)
a = os.pread(fd, 1 << 22, lo); time.sleep(3.0); b = os.pread(fd, 1 << 22, lo)
cands = []
for k in range(0, min(len(a), len(b)) - 3, 4):
va, vb = struct.unpack_from('>I', a, k)[0], struct.unpack_from('>I', b, k)[0]
if va < vb and 5 < (vb - va) / 3.0 < 200:
cands.append((lo + k, round((vb - va) / 3.0)))
modal = collections.Counter(r for _, r in cands).most_common(1)
ticks = [o for o, r in cands if modal and r == modal[0][0]][:32]
print('tick witnesses: %d candidates, modal rate %s/s, using %d'
% (len(cands), modal[0][0] if modal else '-', len(ticks))
if ticks else 'tick witnesses: NONE -- RUN UNVALIDATED')
last_ticks = [struct.unpack('>I', os.pread(fd, 4, o))[0] for o in ticks]
craft = enumerate_craft(fd, defs, want)
print('craft %d, linked %d' % (len(craft), sum(1 for c in craft if c[2])))
prev = collections.Counter(c[2] for c in craft if c[2] and alive(fd, c[0]))
print('t= 0s deployed=%d strengths %s'
% (len(prev), sorted(collections.Counter(prev.values()).items())), flush=True)
t0 = time.time(); last_rescan = t0; arr = los = 0; stalls = 0
pending, confirmed = {}, 0
while time.time() - t0 < secs:
time.sleep(every)
el = round(time.time() - t0)
if time.time() - last_rescan > RESCAN:
craft = enumerate_craft(fd, defs, want); last_rescan = time.time()
cur = collections.Counter(c[2] for c in craft if c[2] and alive(fd, c[0]))
st = ''
if ticks:
now = [struct.unpack('>I', os.pread(fd, 4, o))[0] for o in ticks]
moved = sum(1 for x, y in zip(last_ticks, now) if y > x)
if moved * 2 < len(ticks):
st = ' *** GUEST STALLED (%d/%d witnesses moved) ***' % (moved, len(ticks))
stalls += 1
last_ticks = now
# Report EVERY increase, not just 0 -> n. The first run logged a record
# reading 13 then 14 with no event printed, which is how flicker in the
# hull-based liveness read hides: only decreases were being surfaced, so
# a spurious 0 -> 2 looked like an arrival while 13 -> 14 looked like
# nothing. An arrival must also PERSIST to count.
a_ = [(o, prev.get(o, 0), cur[o]) for o in cur if cur[o] > prev.get(o, 0)]
l_ = [(o, prev[o], cur.get(o, 0)) for o in prev if cur.get(o, 0) < prev[o]]
for o, x, y in a_:
if x == 0: pending[o] = pending.get(o, 0) + 1
for o in list(pending):
if cur.get(o, 0) == 0: pending.pop(o, None) # vanished: flicker
elif pending[o] == 2:
confirmed += 1
print(' *** CONFIRMED ARRIVAL %-26s now %d (persisted 2 samples)'
% (label.get(o, '?'), cur[o]), flush=True)
pending[o] = 3
arr += sum(1 for x in a_ if x[1] == 0); los += len(l_)
print('t=%4ds deployed=%d up=%d down=%d (cum up %d / down %d, confirmed %d)%s'
% (el, len(cur), len(a_), len(l_), arr, los, confirmed, st), flush=True)
for o, x, y in a_:
print(' up %-30s %d -> %d%s' % (label.get(o, '?'), x, y,
' <- candidate arrival' if x == 0 else ''), flush=True)
for o, x, y in l_: print(' loss %-30s %d -> %d' % (label.get(o, '?'), x, y), flush=True)
prev = cur
if not ticks:
print('\nRUN UNVALIDATED: no tick witness, so flat samples prove nothing.')
print('\nTOTAL candidate-up=%d down=%d CONFIRMED arrivals=%d stalled samples=%s'
% (arr, los, confirmed, stalls if ticks else 'UNKNOWN'))
return 0
if __name__ == '__main__':
sys.exit(main())