Files
BFM-decomp/tools/progress.py
T
Drew T 379185a799 feat(phase-7): matching reports + 24 matches (38 real) — session A checkpoint
Mid-phase checkpoint (Drew-directed; phase NOT closed — R8 exception).

- reports (Task 3): tools/{progress,difficulty,dup_report}.py + `make report`
  (deterministic) + `make sig-refresh`; authoritative baseline 38 real / 2050
  matchable; empties audit 42/42 genuine jr;nop; docs/{progress,difficulty,
  duplicates}.md digests
- matches (Tasks 4-5): +24 byte-identical (22 accessor leaves via the harvest
  pattern + ResourceGetCdLoc + LoaderResetReadState); make check BYTE-IDENTICAL
  throughout (143dbb89...)
- config/symbols.us.txt: declare func_80047CAC (R15) — fixes a latent
  non-reproducibility (spimdisasm 1.41.0 auto-detect of an 8B inter-fn blob was
  unstable across clean extracts; the committed Phase-6 baseline did not
  deterministically rebuild)
- Task 1 rodata-island foundation: investigated (migration co-locates jump
  tables + resolves refs, but byte-identity blocked by splat global-migration +
  contiguous linker placement vs BFM's monolithic segment) -> DEFERRED to Task 2'
  with a candidate fix; full findings in phase-ends/CURRENT_PHASE.md
- phase-ends/CURRENT_PHASE.md: reordered plan, loader-cluster triage, per-session
  green-check log (1 of >=3 needed for the Gen1-exit milestone)
2026-06-14 17:36:12 -06:00

129 lines
5.1 KiB
Python

#!/usr/bin/env python3
"""BFM matching-progress report (authoritative; supersedes the Phase-6 PhaseEnd estimate).
Classifies every matchable function in src/800.c and prints a deterministic summary
(also written to docs/progress.md). Ghidra-free. The REAL count is the number the
Gen1-exit "≥25 matched functions" bar counts (splat-auto empties do NOT count).
Usage:
tools/progress.py # print summary + write docs/progress.md
tools/progress.py --audit # also verify every empty no-op's asm is exactly {jr,nop}
tools/progress.py --check # also hash build/us/SLUS_007.26 vs config/check.us.sha
"""
import re, sys, hashlib, pathlib
ROOT = pathlib.Path(__file__).resolve().parent.parent
SRC = ROOT / "src" / "800.c"
ASM = ROOT / "asm" / "nonmatchings" / "800"
OUT = ROOT / "docs" / "progress.md"
BUILD = ROOT / "build" / "us" / "SLUS_007.26"
CHECK = ROOT / "config" / "check.us.sha"
INSTR = re.compile(r'^\s*/\*\s*[0-9A-Fa-f]+\s+[0-9A-Fa-f]+\s+[0-9A-Fa-f]+\s*\*/\s+[a-z]')
def strip_comments(s):
s = re.sub(r'/\*.*?\*/', '', s, flags=re.S)
return re.sub(r'//[^\n]*', '', s)
def is_data_blob(name):
"""A .s with a code label (glabel/jlabel) is a function; data-only (dlabel, no code) is a blob."""
p = ASM / f"{name}.s"
if not p.exists():
return False
txt = p.read_text()
return ('glabel' not in txt and 'jlabel' not in txt and 'dlabel' in txt)
def asm_is_trivial(name):
"""True iff the function's asm is exactly {jr, nop} (the empty-no-op shape splat emits void{} for)."""
p = ASM / f"{name}.s"
if not p.exists():
return None
mnem = []
for ln in p.read_text().splitlines():
m = re.match(r'^\s*/\*\s*[0-9A-Fa-f]+\s+[0-9A-Fa-f]+\s+[0-9A-Fa-f]+\s*\*/\s+([a-z0-9.]+)', ln)
if m:
mnem.append(m.group(1))
return set(mnem) <= {'jr', 'nop'} and len(mnem) <= 2
SIG = re.compile(r'^\s*[A-Za-z_][\w \t\*]*\b([A-Za-z_]\w*)\s*\(')
def classify():
lines = SRC.read_text().split('\n')
n = len(lines); i = 0
real, empty, nonmatching, stubs, blobs = [], [], [], [], []
while i < n:
s = lines[i].strip()
if s.startswith('#ifdef NON_MATCHING'):
blk = []
while i < n and not lines[i].strip().startswith('#endif'):
blk.append(lines[i]); i += 1
i += 1
m = re.search(r'INCLUDE_ASM\("[^"]+",\s*(\w+)\)', '\n'.join(blk))
if m: nonmatching.append(m.group(1))
continue
m = re.match(r'INCLUDE_ASM\("[^"]+",\s*(\w+)\)', s)
if m:
(blobs if is_data_blob(m.group(1)) else stubs).append(m.group(1)); i += 1; continue
if s.startswith('INCLUDE_RODATA'):
i += 1; continue
fm = SIG.match(lines[i])
if fm and '(' in lines[i]:
start = i; depth = 0; opened = False
while i < n:
c = strip_comments(lines[i]); depth += c.count('{') - c.count('}')
if '{' in c: opened = True
i += 1
if opened and depth <= 0: break
body = '\n'.join(lines[start:i])
a, b = body.index('{'), body.rindex('}')
(real if strip_comments(body[a+1:b]).strip() else empty).append(fm.group(1))
continue
i += 1
return real, empty, nonmatching, stubs, blobs
def main():
audit = '--audit' in sys.argv
check = '--check' in sys.argv
real, empty, nonmatching, stubs, blobs = classify()
matchable = len(real) + len(empty) + len(nonmatching) + len(stubs)
incl = len(real) + len(empty)
out = []
out.append("# BFM matching progress (generated by tools/progress.py — authoritative)")
out.append("")
out.append(f"REAL substantive matches : {len(real):5d} <- the Gen1-exit >=25 bar counts THIS")
out.append(f"NON_MATCHING (near-miss) : {len(nonmatching):5d}")
out.append(f"splat-auto empty no-ops : {len(empty):5d}")
out.append(f"INCLUDE_ASM stubs : {len(stubs):5d}")
out.append(f"data blobs (excluded) : {len(blobs):5d}")
out.append("-" * 40)
out.append(f"matchable functions : {matchable:5d}")
out.append(f"REAL / matchable : {len(real)} / {matchable} = {100*len(real)/matchable:.2f}%")
out.append(f"incl. empties : {incl} / {matchable} = {100*incl/matchable:.2f}%")
out.append("")
out.append("REAL matches: " + " ".join(sorted(real)))
out.append("NON_MATCHING: " + " ".join(sorted(nonmatching)))
if BUILD.exists():
h = hashlib.sha1(BUILD.read_bytes()).hexdigest()
want = CHECK.read_text().split()[0] if CHECK.exists() else ""
out.append("")
out.append(f"build SHA1: {h} ({'byte-identical' if h == want else 'MISMATCH'})")
if audit:
bad = [n for n in empty if asm_is_trivial(n) is False]
out.append("")
out.append(f"empties audit: {len(empty)-len(bad)}/{len(empty)} genuine jr;nop"
+ (f" !! SUSPICIOUS: {bad}" if bad else " (all clean)"))
text = "\n".join(out) + "\n"
print(text, end="")
OUT.write_text(text)
if check:
import subprocess
sys.exit(subprocess.run(["make", "-C", str(ROOT), "check"]).returncode)
if __name__ == "__main__":
main()