The linear walk's 10% unknown was a floor imposed by the method: a block entered only
by a branch has a well-defined state, just not one a straight-line pass can see.
tools/re-capture/isl_cfg.py replaces it with a worklist fixpoint that joins each
block's state over its ACTUAL predecessors -- a value survives only if every
predecessor agrees.
Over all 28 stages:
instructions reached by the CFG 85.0%
condition sites, unknown LHS 756 (10.00%) -> 402 (5.32%)
of those, never reached at all 389
joined away (predecessors disagree) 13
both resolve but DISAGREE 161 <- linear walk was wrong here
Those 161 are on top of the 889 the previous jmp fix caught.
Two zero-results on the way, both my own bug, both caught because the number looked
wrong rather than because a test failed:
* The first CFG run reached only 36% of instructions and made things WORSE (35%
unknown). Cause: the phase bases reach almost nothing. Most routines are
COROUTINES the engine starts from its trigger queue, with no static predecessor,
so every start_coroutine target has to be seeded as an entry.
* That seeding then found ZERO entries in a file with 216 start_coroutine calls,
because the target is staged in TWO steps -- special[0] = imm, then
local[0] = special[0] -- and I matched only the direct-immediate form.
Reachability went 36% -> 64% -> 85% as each was fixed.
The 389 still unreached are an honest limit rather than a gap: nothing in the bytecode
starts them; they are entered from the trigger queue at phase+272, by data rather than
code, so no purely static analysis reaches them.
isl_report.py conditions now uses isl_cfg; calls and phase-ends regenerate
byte-identical. Stage 02 unknowns drop from 71 to 25.
178 lines
7.6 KiB
Python
178 lines
7.6 KiB
Python
"""ISL condition recovery by CFG DATAFLOW, replacing `isl.conditions`'s linear walk.
|
|
|
|
`isl.conditions` walks the flat stream and resets its tracker at every
|
|
control-flow boundary, so a block entered only by a branch reports an unknown.
|
|
This module instead builds the CFG and runs a worklist fixpoint, joining each
|
|
block's state over its ACTUAL predecessors (a value survives the join only if
|
|
every predecessor agrees).
|
|
|
|
Measured over all 28 stages, against the linear walk:
|
|
|
|
instructions reached by the CFG 85.0%
|
|
condition sites with an unknown LHS 402 (5.32%) linear walk: 756 (10.00%)
|
|
of those, still never reached 389
|
|
sites where both resolve but DISAGREE 161 <- the linear walk was wrong
|
|
|
|
Two things the entry-point search had to get right, both of which read as zero
|
|
results first:
|
|
|
|
* The phase bases reach only ~36% of the code. Most routines are COROUTINES
|
|
the engine starts from its trigger queue, so they have no static predecessor
|
|
and must be seeded from every `start_coroutine` target.
|
|
* That target is staged in TWO steps -- `special[0] = imm` then
|
|
`local[0] = special[0]`. Matching only the direct-immediate form found ZERO
|
|
entries in a file with 216 of them.
|
|
|
|
The 389 that remain unreached are the honest limit: nothing in the bytecode
|
|
starts them, so they are entered by data (the trigger queue at `phase+272`),
|
|
not by code.
|
|
"""
|
|
import struct, collections
|
|
import isl
|
|
|
|
REL = isl.REL
|
|
|
|
def _val(op, kind, operand, lo, sp, loc):
|
|
if kind == 1:
|
|
if op == 1:
|
|
return '%.6g' % struct.unpack('>d', struct.pack('>II', operand, lo))[0]
|
|
return str(operand)
|
|
if kind == 2: return sp.get(operand)
|
|
if kind == 3: return loc.get(operand)
|
|
return 'global[%d]' % operand
|
|
|
|
def _pack(sp, loc, stack):
|
|
return (tuple(sorted(sp.items())), tuple(sorted(loc.items())), tuple(stack))
|
|
|
|
def _join(a, bst):
|
|
if a is None: return bst
|
|
if bst is None: return a
|
|
if a == bst: return a
|
|
def m(x, y):
|
|
dx, dy = dict(x), dict(y)
|
|
return tuple(sorted((k, v) for k, v in dx.items() if dy.get(k) == v))
|
|
st = a[2] if a[2] == bst[2] else ()
|
|
return (m(a[0], bst[0]), m(a[1], bst[1]), st)
|
|
|
|
def coroutine_entries(b):
|
|
"""Every `start_coroutine` target, found by a linear pre-pass."""
|
|
offs = isl.linear_offsets(b)
|
|
bases = isl.phase_bases(b)
|
|
out, loc, sp = [], {}, {}
|
|
for off in offs:
|
|
w = struct.unpack_from('>I', b, off)[0]
|
|
op, ln = w & 0xFF, (w >> 8) & 0xFF
|
|
sk, dk = (w >> 24) & 0xFF, (w >> 16) & 0xFF
|
|
words = [struct.unpack_from('>I', b, off + i)[0]
|
|
for i in range(4, max(ln, 4), 4) if off + i + 4 <= len(b)]
|
|
if op == 0 and len(words) >= 2:
|
|
# staging is TWO-step: `special[0] = imm` then `local[0] = special[0]`.
|
|
# Matching only the direct-immediate form found ZERO entries in a file
|
|
# with 216 start_coroutine sites.
|
|
v = words[1] if sk == 1 else sp.get(words[1]) if sk == 2 else None
|
|
if v is None: (sp if dk == 2 else loc).pop(words[0], None)
|
|
else: (sp if dk == 2 else loc)[words[0]] = v
|
|
elif op == 19 and words:
|
|
if words[0] == 1 and 0 in loc:
|
|
ph = sum(1 for x in bases if x <= off)
|
|
out.append(bases[ph - 1] + loc[0])
|
|
loc = {}
|
|
return out
|
|
|
|
|
|
def conditions(b, s1=None, s2=None):
|
|
offs = isl.linear_offsets(b)
|
|
nxt = {offs[i]: offs[i + 1] for i in range(len(offs) - 1)}
|
|
bases = isl.phase_bases(b)
|
|
EMPTY = ((), (), ())
|
|
IN = {}
|
|
work = collections.deque()
|
|
# Entry points. The phase bases alone reach only ~36% of the code: most
|
|
# routines are COROUTINES the engine starts from its trigger queue, so they
|
|
# have no static predecessor. Seed every `start_coroutine` target as well --
|
|
# a linear pre-pass finds them because the target is staged into local[0]
|
|
# immediately before the call, which no control-flow boundary intervenes in.
|
|
entries = list(bases) + coroutine_entries(b)
|
|
for e in entries:
|
|
if e in nxt or e == offs[-1]:
|
|
IN[e] = EMPTY; work.append(e)
|
|
conds = {}
|
|
pend_at = {}
|
|
seen_entry = set(bases)
|
|
rounds = 0
|
|
while work:
|
|
rounds += 1
|
|
if rounds > 400000: break
|
|
off = work.popleft()
|
|
state = IN[off]
|
|
sp, loc, stack = dict(state[0]), dict(state[1]), list(state[2])
|
|
w = struct.unpack_from('>I', b, off)[0]
|
|
op, ln = w & 0xFF, (w >> 8) & 0xFF
|
|
sk, dk = (w >> 24) & 0xFF, (w >> 16) & 0xFF
|
|
words = [struct.unpack_from('>I', b, off + i)[0]
|
|
for i in range(4, max(ln, 4), 4) if off + i + 4 <= len(b)]
|
|
succ, fall = [], nxt.get(off)
|
|
if op in (0, 1) and len(words) >= 2:
|
|
v = _val(op, sk, words[1], words[2] if len(words) > 2 else 0, sp, loc)
|
|
d = sp if dk == 2 else loc
|
|
if v is None: d.pop(words[0], None)
|
|
else: d[words[0]] = v
|
|
elif op == 19 and words:
|
|
bid = words[0]
|
|
nm = isl.BUILTIN.get(bid, 'builtin%d' % bid)
|
|
args = []
|
|
tags = {sl - 4 for sl, ids in isl.UNIT_SLOTS.items() if bid in ids and sl in loc}
|
|
for sl in sorted(loc):
|
|
v = loc[sl]
|
|
if sl in tags and v == '1': continue
|
|
if v.isdigit():
|
|
i = int(v)
|
|
if s2 and bid in isl.UNIT_SLOTS.get(sl, ()) and i in s2: v = s2[i][1]
|
|
elif s1 and bid in isl.SYM1_SLOTS.get(sl, ()) and i in s1: v = s1[i][1]
|
|
args.append(v)
|
|
sp[0] = '%s(%s)' % (nm, ', '.join(args))
|
|
loc = {}
|
|
if bid == 1: # start_coroutine: a FRESH thread
|
|
t = state[1] and dict(state[1]).get(0)
|
|
if t and t.isdigit():
|
|
e = bases[max(0, sum(1 for x in bases if x <= off) - 1)] + int(t)
|
|
if e in nxt or e in IN:
|
|
if e not in seen_entry:
|
|
seen_entry.add(e); IN[e] = EMPTY; work.append(e)
|
|
if bid == 11: # end_coroutine: thread destroyed
|
|
fall = None
|
|
elif op in (21, 22):
|
|
stack.append(sp.get(1))
|
|
elif op in (23, 24):
|
|
if stack: sp[1] = stack.pop()
|
|
else: sp.pop(1, None)
|
|
elif op == 12 and words:
|
|
ph = sum(1 for x in bases if x <= off)
|
|
succ.append(bases[ph - 1] + words[0]); fall = None
|
|
elif op in (10, 11):
|
|
pend_at[off] = (_val(op, dk, words[0], 0, sp, loc),
|
|
_val(op, sk, words[1], words[2] if len(words) > 2 else 0, sp, loc))
|
|
elif op in REL and words:
|
|
ph = sum(1 for x in bases if x <= off)
|
|
succ.append(bases[ph - 1] + words[0])
|
|
out = _pack(sp, loc, stack)
|
|
for s in ([fall] if fall else []) + succ:
|
|
if s is None or s not in nxt and s not in IN and s != offs[-1]: continue
|
|
j = _join(IN.get(s), out)
|
|
if IN.get(s) != j:
|
|
IN[s] = j; work.append(s)
|
|
# read conditions off the fixpoint
|
|
out = []
|
|
prev = None
|
|
for off in offs:
|
|
w = struct.unpack_from('>I', b, off)[0]; op = w & 0xFF
|
|
if op in (10, 11): prev = off
|
|
elif op in REL and prev is not None:
|
|
lhs, rhs = pend_at.get(prev, (None, None))
|
|
ph = sum(1 for x in bases if x <= off)
|
|
words = [struct.unpack_from('>I', b, off + 4)[0]]
|
|
out.append({'off': prev, 'phase': ph, 'lhs': lhs, 'rel': REL[op],
|
|
'rhs': rhs, 'target': bases[ph - 1] + words[0]})
|
|
prev = None
|
|
return out
|