phase11: tools/sf3_family — search for a matched row's siblings (cookbook 125)

Worker D's finding 125 said three members of one family were found by three different means
and 'the finder varies, the price does not', concluding that families should be SEARCHED FOR
explicitly rather than waited for. This implements that: for every unmatched worklist row,
find the already-matched row with the highest similarity, where similarity is an
opcode-histogram cosine (registers erased, nops dropped, per finding 120) multiplied by the
size ratio so a shared multiset at a different scale does not count.

A BUG WORTH RECORDING: the first version read config/match_worklist.tsv with the regions
column layout, so column 0 (the RANK) was read as the address. It returned ZERO candidates at
every threshold, which is what exposed it -- a silently wrong address yields no matches rather
than an error. Both layouts are now parsed by named functions with the offset documented.
This commit is contained in:
Christopher Williams
2026-09-24 10:30:58 -04:00
parent bc4c046625
commit 6589668e85
+141
View File
@@ -0,0 +1,141 @@
#!/usr/bin/env python3
"""Find unmatched worklist rows that are FAMILY MEMBERS of already-matched rows.
Phase 11 finding 125 (worker D): three members of one family were found by three different
means -- the size ranker, adjacency, and the redundancy rank -- and all three closed the same
way. *"The finder varies, the price does not."* Once a lever is in hand, a sibling row is a
one-attempt row wherever it sits.
Worker D's operational conclusion is what this tool implements: **families should be SEARCHED
FOR explicitly rather than waited for.** Three rankers found three siblings of one family by
accident; a direct search finds all of them at once.
SIMILARITY is measured on the instruction stream with registers erased, because (finding 120)
the allocator makes copies non-identical, so opcode repetition survives where word repetition
does not. Two signals are combined:
* opcode-histogram cosine -- "does this body do the same KINDS of things, in the same
proportions?"
* size ratio -- "is it the same scale?"
Both must be high. A body that merely shares an instruction multiset with a much larger one is
not a sibling, it is a coincidence.
Usage:
sf3_family [--top N] [--min-score F] [--min-size B] [--max-size B]
"""
from __future__ import annotations
import math
import struct
import sys
from collections import Counter
from pathlib import Path
REPO = Path(__file__).resolve().parent.parent
EXE = REPO / "extracted/SCUS_946.40;1"
REGIONS = REPO / "config/regions.tsv"
WORKLIST = REPO / "config/match_worklist.tsv"
LOAD_ADDRESS = 0x80010000
HEADER_SIZE = 0x800
def body_words(payload: bytes, start: int, end: int) -> list[int]:
offset = HEADER_SIZE + (start - LOAD_ADDRESS)
count = (end - start) // 4
return list(struct.unpack_from(f"<{count}I", payload, offset))
def histogram(words: list[int]) -> Counter[int]:
"""Opcode histogram with registers erased and nops dropped."""
return Counter(word >> 26 for word in words if word)
def cosine(left: Counter[int], right: Counter[int]) -> float:
if not left or not right:
return 0.0
common = set(left) & set(right)
dot = sum(left[k] * right[k] for k in common)
norm = math.sqrt(sum(v * v for v in left.values())) * math.sqrt(sum(v * v for v in right.values()))
return dot / norm if norm else 0.0
def read_regions(path: Path) -> list[tuple[int, int, str]]:
"""config/regions.tsv: start, end, source[, overrides]."""
rows = []
for line in path.read_text().splitlines():
if line.startswith("#") or not line.strip():
continue
fields = line.split("\t")
rows.append((int(fields[0], 16), int(fields[1], 16), fields[2]))
return rows
def read_worklist(path: Path) -> list[tuple[int, int]]:
"""config/match_worklist.tsv: rank, address, end, size, ...
NOTE the column offset -- column 0 is the RANK, not the address. Reading this file with
the regions layout silently produces garbage (rank numbers as addresses) and the search
returns nothing at all, which is how this bug was found.
"""
rows = []
for line in path.read_text().splitlines():
if line.startswith("#") or not line.strip():
continue
fields = line.split("\t")
rows.append((int(fields[1], 16), int(fields[2], 16)))
return rows
def main(argv: list[str]) -> int:
top, min_score, min_size, max_size = 40, 0.97, 0, 1 << 30
index = 0
while index < len(argv):
if argv[index] == "--top":
index += 1
top = int(argv[index])
elif argv[index] == "--min-score":
index += 1
min_score = float(argv[index])
elif argv[index] == "--min-size":
index += 1
min_size = int(argv[index])
elif argv[index] == "--max-size":
index += 1
max_size = int(argv[index])
index += 1
payload = EXE.read_bytes()
matched = []
for start, end, source in read_regions(REGIONS):
words = body_words(payload, start, end)
matched.append((start, end, source, histogram(words), end - start))
results = []
for start, end in read_worklist(WORKLIST):
size = end - start
if size < min_size or size > max_size:
continue
mine = histogram(body_words(payload, start, end))
best = None
for m_start, m_end, m_source, m_hist, m_size in matched:
ratio = min(size, m_size) / max(size, m_size)
score = cosine(mine, m_hist) * ratio
if best is None or score > best[0]:
best = (score, m_start, m_end, m_source, m_size, ratio)
if best and best[0] >= min_score:
results.append((best[0], start, size, best))
results.sort(key=lambda row: (-row[0], row[2]))
print(f"{'score':>6} {'ratio':>6} {'size':>6} candidate -> matched sibling")
for score, start, size, best in results[:top]:
_, m_start, m_end, m_source, m_size, ratio = best
print(f"{score:6.3f} {ratio:6.2f} {size:6d} {hex(start)} -> {hex(m_start)} "
f"({m_size} B, {m_source})")
print(f"\ncandidates={len(results)}")
return 0
if __name__ == "__main__":
raise SystemExit(main(sys.argv[1:]))