feat(phase-31 T8): LEN+N lane — 587 near-misses routed; 345 wrong-drafts reclassified; detectors live

- match_one --emit-streams (additive; stdout-identity NC'd): word streams for the
  len lane
- family_align.addr_true_rel: reloc-vs-constant range discriminator — FULL
  conservative set kept for pair semantics (NC-1 157/157 regression), address-
  true subset for indel eligibility only (a constant li-cluster must not read as
  reloc-in-indel); synthetic probes green both directions
- tools/len_tells.py: aligned classification + §172b tell tagging (EXTPAIR/
  SELECT/NOP) on target-side indels; detectors imported from atlas_features
  (R33); cookbook text embedded in cards
- tools/lenmiss_route.py: pool-parallel (A8) — 587 audit LEN rows re-verified
  live + routed in 24s: redraft 345 (frac>0.35, APPEND-ONLY backlog
  reclassification — near-miss metrics stop lying) / permuter-length 49 (grinder
  fuel) / cards 192 incl 14 tell-tagged (the audit's own detectors had emitted
  ZERO) / mechanical 0 — an HONEST NULL: stored drafts rarely get constants
  wrong; LEN drift is shape, family_align's value here is classifier/detector
- R32 accounting 587/587
This commit is contained in:
Drew T
2026-08-14 19:54:57 -06:00
parent 682d0fa1fd
commit 0840eda5bb
6 changed files with 647 additions and 3 deletions
+45 -2
View File
@@ -74,11 +74,54 @@ def _li_const(words, idxs):
return (dest, v & 0xFFFFFFFF) if dest is not None else None
_ADDR_RANGES = ((0x80010000, 0x80200000), (0x1F800000, 0x1F810000))
def addr_true_rel(words):
"""`reloc_indices` minus the lui-anchor pairs whose combined hi+lo value is NOT a plausible
address (T8: `reloc_indices` conservatively flags EVERY lui+consumer as an anchor — right for
h_norm masking, wrong for li-cluster recognition, where a 32-bit CONSTANT materialization must
stay eligible). jal indices are always kept. The pairing walk mirrors norm_stream's tracker."""
rel = FR.reloc_indices(words)
keep = set()
pending = {} # reg -> (lui_idx, hi)
for k, w in enumerate(words):
op = w >> 26
if op in (2, 3): # j/jal
keep.add(k)
continue
if op == 0x0F: # lui
pending[(w >> 16) & 31] = (k, (w & 0xFFFF) << 16)
continue
rs, rt = (w >> 21) & 31, (w >> 16) & 31
if k in rel and rs in pending:
lk, hi = pending[rs]
imm = w & 0xFFFF
lo = imm - 0x10000 if imm >= 0x8000 and op in (0x09, 0x23, 0x21, 0x25, 0x20, 0x24,
0x2B, 0x29, 0x28, 0x22, 0x26, 0x2A, 0x2E) else imm
v = (hi + lo) & 0xFFFFFFFF
if any(a <= v < b for a, b in _ADDR_RANGES):
keep.add(lk)
keep.add(k)
# else: a constant materialization — both indices stay OUT of the address-true set
# register kill tracking (approximate, matches reloc_indices' conservatism)
if op == 0:
pending.pop((w >> 11) & 31, None)
else:
pending.pop(rt, None)
return keep
def classify_aligned(ex_words, sib_words):
"""-> (verdict, detail). detail: {'pairs', 'clusters': [(exv,sibv)], 'imm_pairs', 'reasons'}.
Equal-length identical-tok inputs reproduce classify_member's PURE/IMM/STRUCT verdicts (NC-1)."""
Equal-length identical-tok inputs reproduce classify_member's PURE/IMM/STRUCT verdicts (NC-1).
TWO reloc sets per side: the FULL conservative set (classify_member-equivalent pair semantics,
NC-1) and the ADDRESS-TRUE subset (indel-region eligibility only — a constant li-cluster must
not read as 'reloc-in-indel')."""
rel_ex = FR.reloc_indices(ex_words)
rel_sib = FR.reloc_indices(sib_words)
rel_ex_addr = addr_true_rel(ex_words)
rel_sib_addr = addr_true_rel(sib_words)
ops = align_blocks(ex_words, sib_words)
pairs = [] # aligned (i, j) index pairs
@@ -126,7 +169,7 @@ def classify_aligned(ex_words, sib_words):
if all(w == 0 for w in region_words_ex + region_words_sib):
verdict_flags.add("NOP")
continue
if any(k in rel_ex for k in ex_idx) or any(k in rel_sib for k in sib_idx):
if any(k in rel_ex_addr for k in ex_idx) or any(k in rel_sib_addr for k in sib_idx):
reasons.append("reloc-in-indel")
verdict_flags.add("STRUCT")
continue
+84
View File
@@ -0,0 +1,84 @@
#!/usr/bin/env python3
"""P31 T8 — classify a LEN±N near-miss (draft vs its OWN target) and build the routing card.
This is where `family_align` earns its keep (the T7 probe refuted the cousin use — but a draft vs
its own target is the SAME function, so registers agree outside the drift and the aligned
classifier's premise holds). Per draft:
- align draft↔target word streams (family_align.classify_aligned; draft = ex side, so the
mechanical repair swaps DRAFT literals toward the TARGET's)
- LEN-LI with resolvable clusters -> MECHANICAL repair candidate (swap the C literal, recompile)
- LEN-NOP / |Δ|<=2 clean drift -> permuter `length` profile
- LEN-STRUCT -> tag the target-side indels with the §172b tells
EXTPAIR target-extra sll/sra promotion pair (a multi-def s16/s8 the draft collapsed)
SELECT target-extra slt+branch select (a swapped-arm textual repeat)
-> agent tell-cards (the §172b cookbook text rides on the card)
- frac > 0.35 -> redraft (SIZE-MISMATCH — the draft is not the function)
Detectors are IMPORTED from atlas_features (one implementation, R33).
"""
import os, sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import family_align as FA
from atlas_features import extpair_count, dupselect_count
COOKBOOK = {
"EXTPAIR": "§172b-1: the target's extra sll/sra-16|24 pair = a REGISTER-held short promotion "
"(a multi-def s16/s8 variable, e.g. branch-merged); keep the short in a named local "
"used both ways so the promotion pair exists.",
"SELECT": "§172b-2: the target's extra slt+branch+move select = the source textually REPEATED "
"a compare with SWAPPED arms ((b>a)?b:a vs (a<b)?a:b); repeat the expression with "
"swapped operands at the extra site.",
"NOP": "schedule artifact — route to the permuter length profile, not an edit.",
}
def analyze(mine_words, tgt_words):
"""-> card dict {verdict, delta, frac, clusters, tells, blocks, route}."""
delta = len(tgt_words) - len(mine_words)
frac = abs(delta) / max(1, len(tgt_words))
verdict, detail = FA.classify_aligned(mine_words, tgt_words)
tells = []
blocks = []
if verdict in ("LEN-STRUCT", "STRUCT-ALIGNED"):
ops = FA.align_blocks(mine_words, tgt_words)
for tag, i1, i2, j1, j2 in ops:
if tag == "equal":
continue
tw = tgt_words[max(0, j1 - 2):min(len(tgt_words), j2 + 2)]
kinds = []
if extpair_count(tw):
kinds.append("EXTPAIR")
if dupselect_count(tgt_words[max(0, j1 - 8):min(len(tgt_words), j2 + 8)]):
kinds.append("SELECT")
if all(w == 0 for w in tgt_words[j1:j2]) and i1 == i2:
kinds.append("NOP")
blocks.append({"mine": [i1, i2], "tgt": [j1, j2], "kinds": kinds,
"tgt_words": [f"{w:08x}" for w in tgt_words[j1:j2][:12]]})
tells += kinds
clusters = [(hex(a), hex(b)) for a, b, _ in detail.get("clusters", [])]
if frac > 0.35:
route = "redraft"
elif verdict == "LEN-LI" and clusters and all(a != b for a, b in clusters):
route = "mechanical"
elif verdict == "LEN-NOP" or (verdict.startswith("LEN") and abs(delta) <= 2 and not tells):
route = "permuter-length"
elif tells:
route = "tell-card"
elif verdict in ("PURE", "IMM"):
route = "imm-or-reloc" # equal length reached here = not a LEN case at all
else:
route = "card"
return {"verdict": verdict, "delta": delta, "frac": round(frac, 3),
"clusters": clusters, "tells": sorted(set(tells)),
"tell_refs": [COOKBOOK[t] for t in sorted(set(tells)) if t in COOKBOOK],
"blocks": blocks[:8], "route": route,
"detail": detail if route == "mechanical" else None}
def repair_mechanical(draft_text, mine_words, tgt_words, detail):
"""Swap the draft's cluster/imm literals toward the target's. -> (new_text, unresolved)."""
return FA.imm_map_aligned(draft_text, mine_words, tgt_words, detail)
+163
View File
@@ -0,0 +1,163 @@
#!/usr/bin/env python3
"""P31 T8 — route the LEN±N near-miss pile through the §172b lenses (plan Leg B / A2).
Consumes the c294 gcc-read audit (`.run/c294/audit_results.json`, the classified near-miss
ledger), re-verifies each LEN row against the CURRENT tree (still-stub + draft exists — stored
verdicts decay, R35), re-derives fresh streams via `match_one --emit-streams` (isolated compile,
process pool per A8), classifies with `len_tells.analyze`, and routes:
mechanical LEN-LI cluster swap -> repaired draft -> re-match_one; only a fresh MATCH
enters the gate slate (.run/lenmiss/mech_slate.json — gate_lane-shaped)
permuter-length |Δ|<=2 clean drift -> .run/lenmiss/permuter.json (grinder fuel)
tell-card EXTPAIR/SELECT-tagged -> .run/lenmiss/cards.json (campaign agent fuel,
§172b text embedded)
redraft frac>0.35 (the draft is not the function) -> APPEND-ONLY backlog records
(status=failed, klass=SIZE-MISMATCH) so near-miss metrics stop counting them
card / other .run/lenmiss/cards.json with the raw verdict
R32: every audit LEN row is accounted (routed | gone | no-draft); totals printed and asserted.
"""
import argparse, collections, json, os, re, subprocess, sys
from concurrent.futures import ProcessPoolExecutor
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import corpus
import len_tells
PY = ".venv/bin/python"
OUT = ".run/lenmiss"
_sc = {}
def is_stub(b, fn):
if b not in _sc:
try:
_sc[b] = {s.symbol: s for s in corpus.stubs(b).values()}
except Exception:
_sc[b] = {}
return _sc[b].get(fn)
def emit_streams(job):
"""Worker: run match_one --emit-streams for one (binary, fn, draft). -> (key, streams|err)."""
b, fn, draft, asm_dir, o0 = job
sp = f"{OUT}/streams/{b}__{fn}.json"
cmd = [PY, "tools/match_one.py", fn, "--c", draft, "--asm-subdir", asm_dir,
"--emit-streams", sp, "--json"]
if o0:
cmd.append("--o0")
r = subprocess.run(cmd, capture_output=True, text=True)
if not os.path.exists(sp):
return (b, fn), {"err": (r.stdout + r.stderr)[-160:]}
return (b, fn), json.load(open(sp))
def main():
ap = argparse.ArgumentParser(description=__doc__.split("\n")[0])
ap.add_argument("--jobs", type=int, default=12)
ap.add_argument("--limit", type=int, default=0)
a = ap.parse_args()
os.makedirs(f"{OUT}/streams", exist_ok=True)
os.makedirs(f"{OUT}/mech", exist_ok=True)
audit = json.load(open(".run/c294/audit_results.json"))
lenrows = [r for r in audit if (r.get("cls") or "").startswith("LEN")]
best = {}
for ln in open(".run/backlog.jsonl"):
r = json.loads(ln)
best[(r.get("binary"), r.get("name"))] = r.get("best_draft")
acct = collections.Counter()
jobs = []
meta = {}
for r in lenrows:
b, fn = r["binary"], r["fn"]
st = is_stub(b, fn)
if st is None:
acct["gone"] += 1
continue
d = best.get((b, fn))
if not d or not os.path.exists(d):
acct["no-draft"] += 1
continue
asm_dir = os.path.dirname(st.asm_path)
o0 = corpus.is_o0(st.path)
jobs.append((b, fn, d, asm_dir, o0))
meta[(b, fn)] = {"draft": d, "audit_cls": r.get("cls"), "asm_dir": asm_dir, "o0": o0}
if a.limit:
jobs = jobs[:a.limit]
print(f"lenmiss: {len(lenrows)} audit LEN rows -> {len(jobs)} live jobs "
f"(gone {acct['gone']}, no-draft {acct['no-draft']})")
with ProcessPoolExecutor(max_workers=a.jobs) as ex:
results = dict(ex.map(emit_streams, jobs))
routes = collections.Counter()
cards, mech, perm, redraft = [], [], [], []
for (b, fn), st in sorted(results.items()):
if "err" in st:
routes["stream-err"] += 1
continue
card = len_tells.analyze(st["mine"], st["tgt"])
card.update({"fn": fn, "binary": b, **meta[(b, fn)]})
routes[card["route"]] += 1
if card["route"] == "mechanical":
txt = open(card["draft"]).read()
fixed, unresolved = len_tells.repair_mechanical(
txt, st["mine"], st["tgt"], card.pop("detail"))
if unresolved:
card["route"] = "card"
card["mech_unresolved"] = [str(u) for u in unresolved[:4]]
routes["mechanical"] -= 1
routes["mech-unresolved"] += 1
cards.append(card)
continue
p = f"{OUT}/mech/{fn}.c"
open(p, "w").write(fixed)
r2 = subprocess.run([PY, "tools/match_one.py", fn, "--c", p,
"--asm-subdir", card["asm_dir"], "--json"]
+ (["--o0"] if card["o0"] else []),
capture_output=True, text=True)
ok = '"status": "match"' in r2.stdout
if ok:
mech.append({"fn": fn, "binary": b, "draft": p})
else:
card["route"] = "card"
card["mech_recheck"] = "no-match"
routes["mechanical"] -= 1
routes["mech-nomatch"] += 1
cards.append(card)
elif card["route"] == "permuter-length":
perm.append({"fn": fn, "binary": b, "draft": card["draft"], "delta": card["delta"]})
elif card["route"] == "redraft":
redraft.append({"fn": fn, "binary": b})
else:
card.pop("detail", None)
cards.append(card)
json.dump(mech, open(f"{OUT}/mech_slate.json", "w"), indent=1)
json.dump(perm, open(f"{OUT}/permuter.json", "w"), indent=1)
json.dump(cards, open(f"{OUT}/cards.json", "w"), indent=1)
json.dump(dict(routes), open(f"{OUT}/route_summary.json", "w"), indent=1)
# redraft reclassification: APPEND-ONLY backlog records (never rewrite history)
if redraft:
import time
with open(".run/backlog.jsonl", "a") as f:
for r in redraft:
f.write(json.dumps({"ts": None, "addr": None, "name": r["fn"],
"reach": None, "klass": "SIZE-MISMATCH", "nins": None,
"status": "failed", "closeness": None,
"where_stuck": "lenmiss_route: frac>0.35 — draft is not the fn",
"best_draft": None, "binary": r["binary"],
"source": "lenmiss-route", "residual": None,
"passes_tried": None}) + "\n")
total = sum(routes.values()) + acct["gone"] + acct["no-draft"]
print(f"routes: {dict(routes)}")
print(f"mech slate {len(mech)} | permuter {len(perm)} | cards {len(cards)} | redraft {len(redraft)}")
assert total >= len(lenrows) - 2, f"R32 accounting: {total} routed+skipped vs {len(lenrows)} rows"
print(f"lenmiss_route: accounted {total}/{len(lenrows)} -> {OUT}/")
if __name__ == "__main__":
main()
+7
View File
@@ -33,6 +33,9 @@ ap.add_argument('--work', default=None,
ap.add_argument('--o0', action='store_true',
help='compile at -O0 (for the _o0 split subsegments: ov_SC01_077_o0.c, whale _o0b — '
'their target bytes are -O0; an -O2 compile can never match them, Makefile:445)')
ap.add_argument('--emit-streams', default=None,
help='P31 T8 (additive): dump {"fn","mine":[words],"tgt":[words]} to this path — '
'the len_tells/family_align input. No effect on stdout.')
ap.add_argument('--json', action='store_true',
help='emit one JSON result line {status,closeness,nins,residual} (Task-12 structured '
'residual telemetry for the permuter-autopsy). Still exits 0 on MATCH, 1 otherwise.')
@@ -96,6 +99,10 @@ tgt = masked_diff.insns_from_s('%s/%s.s' % (a.asm_subdir, a.fn))
if not mine:
print('FAIL: my object has no function', a.fn, '(compile produced nothing?)'); sys.exit(1)
if a.emit_streams: # P31 T8: word streams for len_tells
json.dump({"fn": a.fn, "mine": [i["word"] for i in mine], "tgt": [i["word"] for i in tgt]},
open(a.emit_streams, "w"))
# structured residual (shared with gate_stage's near-record + the Task-12 autopsy telemetry)
diffs = masked_diff.structured_diff(mine, tgt)