phase11: tools/sf3_family — search for a matched row's siblings (cookbook 125)
Worker D's finding 125 said three members of one family were found by three different means and 'the finder varies, the price does not', concluding that families should be SEARCHED FOR explicitly rather than waited for. This implements that: for every unmatched worklist row, find the already-matched row with the highest similarity, where similarity is an opcode-histogram cosine (registers erased, nops dropped, per finding 120) multiplied by the size ratio so a shared multiset at a different scale does not count. A BUG WORTH RECORDING: the first version read config/match_worklist.tsv with the regions column layout, so column 0 (the RANK) was read as the address. It returned ZERO candidates at every threshold, which is what exposed it -- a silently wrong address yields no matches rather than an error. Both layouts are now parsed by named functions with the offset documented.
This commit is contained in:
Executable
+141
@@ -0,0 +1,141 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Find unmatched worklist rows that are FAMILY MEMBERS of already-matched rows.
|
||||
|
||||
Phase 11 finding 125 (worker D): three members of one family were found by three different
|
||||
means -- the size ranker, adjacency, and the redundancy rank -- and all three closed the same
|
||||
way. *"The finder varies, the price does not."* Once a lever is in hand, a sibling row is a
|
||||
one-attempt row wherever it sits.
|
||||
|
||||
Worker D's operational conclusion is what this tool implements: **families should be SEARCHED
|
||||
FOR explicitly rather than waited for.** Three rankers found three siblings of one family by
|
||||
accident; a direct search finds all of them at once.
|
||||
|
||||
SIMILARITY is measured on the instruction stream with registers erased, because (finding 120)
|
||||
the allocator makes copies non-identical, so opcode repetition survives where word repetition
|
||||
does not. Two signals are combined:
|
||||
|
||||
* opcode-histogram cosine -- "does this body do the same KINDS of things, in the same
|
||||
proportions?"
|
||||
* size ratio -- "is it the same scale?"
|
||||
|
||||
Both must be high. A body that merely shares an instruction multiset with a much larger one is
|
||||
not a sibling, it is a coincidence.
|
||||
|
||||
Usage:
|
||||
sf3_family [--top N] [--min-score F] [--min-size B] [--max-size B]
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
import struct
|
||||
import sys
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
|
||||
REPO = Path(__file__).resolve().parent.parent
|
||||
EXE = REPO / "extracted/SCUS_946.40;1"
|
||||
REGIONS = REPO / "config/regions.tsv"
|
||||
WORKLIST = REPO / "config/match_worklist.tsv"
|
||||
LOAD_ADDRESS = 0x80010000
|
||||
HEADER_SIZE = 0x800
|
||||
|
||||
|
||||
def body_words(payload: bytes, start: int, end: int) -> list[int]:
|
||||
offset = HEADER_SIZE + (start - LOAD_ADDRESS)
|
||||
count = (end - start) // 4
|
||||
return list(struct.unpack_from(f"<{count}I", payload, offset))
|
||||
|
||||
|
||||
def histogram(words: list[int]) -> Counter[int]:
|
||||
"""Opcode histogram with registers erased and nops dropped."""
|
||||
return Counter(word >> 26 for word in words if word)
|
||||
|
||||
|
||||
def cosine(left: Counter[int], right: Counter[int]) -> float:
|
||||
if not left or not right:
|
||||
return 0.0
|
||||
common = set(left) & set(right)
|
||||
dot = sum(left[k] * right[k] for k in common)
|
||||
norm = math.sqrt(sum(v * v for v in left.values())) * math.sqrt(sum(v * v for v in right.values()))
|
||||
return dot / norm if norm else 0.0
|
||||
|
||||
|
||||
def read_regions(path: Path) -> list[tuple[int, int, str]]:
|
||||
"""config/regions.tsv: start, end, source[, overrides]."""
|
||||
rows = []
|
||||
for line in path.read_text().splitlines():
|
||||
if line.startswith("#") or not line.strip():
|
||||
continue
|
||||
fields = line.split("\t")
|
||||
rows.append((int(fields[0], 16), int(fields[1], 16), fields[2]))
|
||||
return rows
|
||||
|
||||
|
||||
def read_worklist(path: Path) -> list[tuple[int, int]]:
|
||||
"""config/match_worklist.tsv: rank, address, end, size, ...
|
||||
|
||||
NOTE the column offset -- column 0 is the RANK, not the address. Reading this file with
|
||||
the regions layout silently produces garbage (rank numbers as addresses) and the search
|
||||
returns nothing at all, which is how this bug was found.
|
||||
"""
|
||||
rows = []
|
||||
for line in path.read_text().splitlines():
|
||||
if line.startswith("#") or not line.strip():
|
||||
continue
|
||||
fields = line.split("\t")
|
||||
rows.append((int(fields[1], 16), int(fields[2], 16)))
|
||||
return rows
|
||||
|
||||
|
||||
def main(argv: list[str]) -> int:
|
||||
top, min_score, min_size, max_size = 40, 0.97, 0, 1 << 30
|
||||
index = 0
|
||||
while index < len(argv):
|
||||
if argv[index] == "--top":
|
||||
index += 1
|
||||
top = int(argv[index])
|
||||
elif argv[index] == "--min-score":
|
||||
index += 1
|
||||
min_score = float(argv[index])
|
||||
elif argv[index] == "--min-size":
|
||||
index += 1
|
||||
min_size = int(argv[index])
|
||||
elif argv[index] == "--max-size":
|
||||
index += 1
|
||||
max_size = int(argv[index])
|
||||
index += 1
|
||||
|
||||
payload = EXE.read_bytes()
|
||||
|
||||
matched = []
|
||||
for start, end, source in read_regions(REGIONS):
|
||||
words = body_words(payload, start, end)
|
||||
matched.append((start, end, source, histogram(words), end - start))
|
||||
|
||||
results = []
|
||||
for start, end in read_worklist(WORKLIST):
|
||||
size = end - start
|
||||
if size < min_size or size > max_size:
|
||||
continue
|
||||
mine = histogram(body_words(payload, start, end))
|
||||
best = None
|
||||
for m_start, m_end, m_source, m_hist, m_size in matched:
|
||||
ratio = min(size, m_size) / max(size, m_size)
|
||||
score = cosine(mine, m_hist) * ratio
|
||||
if best is None or score > best[0]:
|
||||
best = (score, m_start, m_end, m_source, m_size, ratio)
|
||||
if best and best[0] >= min_score:
|
||||
results.append((best[0], start, size, best))
|
||||
|
||||
results.sort(key=lambda row: (-row[0], row[2]))
|
||||
print(f"{'score':>6} {'ratio':>6} {'size':>6} candidate -> matched sibling")
|
||||
for score, start, size, best in results[:top]:
|
||||
_, m_start, m_end, m_source, m_size, ratio = best
|
||||
print(f"{score:6.3f} {ratio:6.2f} {size:6d} {hex(start)} -> {hex(m_start)} "
|
||||
f"({m_size} B, {m_source})")
|
||||
print(f"\ncandidates={len(results)}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main(sys.argv[1:]))
|
||||
Reference in New Issue
Block a user