From 22b20e8f31907d0ec6a31cae80e377f1f612bc00 Mon Sep 17 00:00:00 2001 From: Drew T <50529377+Druthulu@users.noreply.github.com> Date: Mon, 15 Jun 2026 20:46:48 -0600 Subject: [PATCH] =?UTF-8?q?refactor(phase-9):=20T6=20=E2=80=94=20--binary?= =?UTF-8?q?=20selector=20for=20dup=5Freport=20+=20difficulty;=20report=20t?= =?UTF-8?q?hreads=20BINARY?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - dup_report.py + difficulty.py: argparse --binary (default main) + BINARIES table (main = current sig/src/asm/md/csv paths); overlay src/asm subtree deferred to P10 - Makefile report target threads --binary $(BINARY) to progress/difficulty/dup_report - no-op proof: docs/progress.md + docs/duplicates.md UNCHANGED; difficulty.py old-code vs new-code output IDENTICAL (pure path-table refactor) - docs/difficulty.md regenerated (the Phase-8-deferred regen: 9 harvested funcs left the unmatched queue — 12/12 row shift); build path untouched (still 143dbb89) --- Makefile | 6 +++--- docs/difficulty.md | 24 ++++++++++++------------ phase-ends/CURRENT_PHASE.md | 4 ++-- tools/difficulty.py | 26 ++++++++++++++++++++------ tools/dup_report.py | 22 +++++++++++++++++----- 5 files changed, 54 insertions(+), 28 deletions(-) diff --git a/Makefile b/Makefile index 1cb4cd43ec..42a1bcc432 100644 --- a/Makefile +++ b/Makefile @@ -98,9 +98,9 @@ GHIDRA := $(or $(GHIDRA_INSTALL_DIR),$(HOME)/ghidra_12.1_PUBLIC) GHIDRA_PROJ := $(HOME)/bfm-decomp/ghidra report: - $(VENV_PY) tools/progress.py --audit - $(VENV_PY) tools/difficulty.py - $(VENV_PY) tools/dup_report.py + $(VENV_PY) tools/progress.py --binary $(BINARY) --audit + $(VENV_PY) tools/difficulty.py --binary $(BINARY) + $(VENV_PY) tools/dup_report.py --binary $(BINARY) sig-refresh: @if ss -tln 2>/dev/null | grep -qE ':8080([^0-9]|$$)'; then diff --git a/docs/difficulty.md b/docs/difficulty.md index 6556666198..e39a801b6d 100644 --- a/docs/difficulty.md +++ b/docs/difficulty.md @@ -1,8 +1,8 @@ # Unmatched difficulty inventory (generated by tools/difficulty.py — harvest queue) -unmatched functions : 2001 -trivial (<=5 ins) : 297 -non-jtbl leaves : 1123 (best harvest targets) +unmatched functions : 1995 +trivial (<=5 ins) : 291 +non-jtbl leaves : 1117 (best harvest targets) jump-table funcs : 80 (deferred — need the rodata-island workflow, Task 2') ## Easiest 120 unmatched (score asc) — the work queue @@ -57,6 +57,7 @@ jump-table funcs : 80 (deferred — need the rodata-island workflow, Task | 3 | SetData32 | 3 | 0 | 0 | - | Y | | 3 | SetIR0 | 3 | 0 | 0 | - | Y | | 3 | UT_KEYV_OBJ_174 | 3 | 0 | 0 | - | Y | +| 3 | func_80049610 | 3 | 0 | 0 | - | Y | | 3 | func_8005E188 | 3 | 0 | 0 | - | Y | | 3 | func_80062388 | 3 | 0 | 0 | - | Y | | 4 | CD_set_test_parmnum | 4 | 0 | 0 | - | Y | @@ -94,15 +95,6 @@ jump-table funcs : 80 (deferred — need the rodata-island workflow, Task | 4 | UT_KEYV_OBJ_3FC | 4 | 0 | 0 | - | Y | | 4 | VSYNC_OBJ_1D4 | 4 | 0 | 0 | - | Y | | 4 | VS_VH_OBJ_180 | 4 | 0 | 0 | - | Y | -| 4 | func_8002AEF8 | 4 | 0 | 0 | - | Y | -| 4 | func_8002AF08 | 4 | 0 | 0 | - | Y | -| 4 | func_8002AF60 | 4 | 0 | 0 | - | Y | -| 4 | func_8002D4B8 | 4 | 0 | 0 | - | Y | -| 4 | func_8002D7FC | 4 | 0 | 0 | - | Y | -| 4 | func_8002D834 | 4 | 0 | 0 | - | Y | -| 4 | func_8002F648 | 4 | 0 | 0 | - | Y | -| 4 | func_80037358 | 4 | 0 | 0 | - | Y | -| 4 | func_80037CC8 | 4 | 0 | 0 | - | Y | | 4 | func_8003D424 | 4 | 0 | 0 | - | Y | | 4 | func_8003FA54 | 4 | 0 | 0 | - | Y | | 4 | func_800426D4 | 4 | 0 | 0 | - | Y | @@ -112,6 +104,7 @@ jump-table funcs : 80 (deferred — need the rodata-island workflow, Task | 4 | func_80043440 | 4 | 0 | 0 | - | Y | | 4 | func_800491EC | 4 | 0 | 0 | - | Y | | 4 | func_8004923C | 4 | 0 | 0 | - | Y | +| 4 | func_80049600 | 4 | 0 | 0 | - | Y | | 4 | func_8005C4CC | 4 | 0 | 0 | - | Y | | 4 | func_8005CF08 | 4 | 0 | 0 | - | Y | | 4 | func_8005CF18 | 4 | 0 | 0 | - | Y | @@ -128,3 +121,10 @@ jump-table funcs : 80 (deferred — need the rodata-island workflow, Task | 5 | INTR_OBJ_6D0 | 5 | 0 | 0 | - | Y | | 5 | ISO9660_OBJ_29C | 3 | 0 | 1 | - | - | | 5 | LIBMCRD_OBJ_5C0 | 5 | 0 | 0 | - | Y | +| 5 | LIBMCRD_OBJ_840 | 3 | 0 | 1 | - | - | +| 5 | LIBMCRD_OBJ_A50 | 3 | 0 | 1 | - | - | +| 5 | LIBMCRD_OBJ_C90 | 3 | 0 | 1 | - | - | +| 5 | LIBMCRD_OBJ_F4C | 3 | 0 | 1 | - | - | +| 5 | OBJT_OBJ_108 | 2 | 1 | 0 | - | Y | +| 5 | OBJT_OBJ_110 | 2 | 1 | 0 | - | Y | +| 5 | OBJT_OBJ_124 | 2 | 1 | 0 | - | Y | diff --git a/phase-ends/CURRENT_PHASE.md b/phase-ends/CURRENT_PHASE.md index dbe4b58495..81a463fce5 100644 --- a/phase-ends/CURRENT_PHASE.md +++ b/phase-ends/CURRENT_PHASE.md @@ -28,8 +28,8 @@ - [x] **T3** — `psyq_identify.py` [A2]: argparse `--vram-base`/`--exe` (default to kept globals, transitional); window stays optional positional. **Caller argv updates deferred to T4/T8** (one-touch-per-file; psyq_identify's defaults keep them green). DONE: narrow window 0x50000..0x60000 ⇒ 0/25 located (window threaded); default + explicit ⇒ 18/25; clean build ⇒ `143dbb89`. - [x] **T4** — `psyq_link_region.py` + `psyq_link_lib.py` [A3]: threaded vram_base/exe through build_region/placement/classify (+ link_lib); pass `--vram-base`/`--exe` to psyq_identify argv; argparse mains. **VRAM_BASE import KEPT** as transitional default source (removed T8, not dropped now — lower churn, consistent). DONE: link_lib libcd 18/18; wrong base ⇒ 6/18 (threaded); link_region byte-identical True; clean build ⇒ `143dbb89`. - [x] **T5** — `psyq_integrate.py` + `progress.py` + Makefile (ONE COMMIT) [A4+B1]: integrate argparse (flags before positionals, `nargs="*"` window) threads vram_base/exe/symbols; 9 Makefile call sites pass `--vram-base/--exe/--symbols $(main_*)`; progress regex `(?:\s+--\S+\s+\S+)*` consumes leading flags + `--binary` selector. DONE: positive `143dbb89`; report 52/**959**/7/50.24% (regex OK); **negative** main_VRAM_BASE=0x8000F804 ⇒ `cae22f7e`≠target; **without-SDK fallback** ⇒ `143dbb89`. -- [ ] **T6** — `dup_report.py` + `difficulty.py` [B2+B3]: `--binary` selector + BINARIES table; defer overlay subtree to P10. Gate: `make report` + `git diff --exit-code` docs digests. ← **CURRENT** -- [ ] **T7** — `ld_interleave.py` [C3]: `--front`/`--tail`/`--ld`; gate call under `ifeq ($(BINARY),main)`. Gate: build no-op (extract critical path). +- [x] **T6** — `dup_report.py` + `difficulty.py` [B2+B3]: argparse `--binary` (default main) + BINARIES table; Makefile `report` threads `--binary $(BINARY)`. DONE: progress.md + duplicates.md UNCHANGED (pure no-op); difficulty.md old-code==new-code output (proven) ⇒ its 12/12 diff is the **Phase-8 deferred regen** (9 harvested fns left the queue) — regenerated + committed here (resolves the Phase-8 deferral). Build untouched (report-only). +- [ ] **T7** — `ld_interleave.py` [C3]: `--front`/`--tail`/`--ld`; gate call under `ifeq ($(BINARY),main)`. Gate: build no-op (extract critical path). ← **CURRENT** - [ ] **T8** — curation helpers de-default [C1/C2/E1/D]: make_snd_used / make_apicard_used / gen_lib_subsegs (`--vram-base`) / split_src_region (`--symbols`). Gate: helpers regenerate identical curated dirs; build no-op. - [ ] **T9** — `diff_settings.py`: `BFM_BINARY` env-var selector + BINARIES table. Gate: asm-differ scores 0 on a known match. - [ ] **T10** — Docs (SETUP/cookbook/psyq-worklist/README; R16/R21) + `make expected` refresh + final both-ways milestone proof + negative control demonstrated/reverted. diff --git a/tools/difficulty.py b/tools/difficulty.py index 7aa064b9c3..5ee1c92839 100644 --- a/tools/difficulty.py +++ b/tools/difficulty.py @@ -6,15 +6,18 @@ flow, jump-table presence, and call count, then ranks easiest-first. Jump-table score high (they need the deferred rodata-island workflow, Task 2'). Writes the actionable easy queue to docs/difficulty.md and the full CSV to .run/difficulty.csv. -Usage: tools/difficulty.py [TOP] (TOP = how many easy rows in the md digest, default 120) +Usage: tools/difficulty.py [TOP] [--binary ] (TOP default 120; binary default main) """ import re, sys, pathlib ROOT = pathlib.Path(__file__).resolve().parent.parent -SRCS = sorted((ROOT / "src").glob("*.c")) # every c-segment (src/boot.c, src/800.c, ...) -ASM_ROOT = ROOT / "asm" / "nonmatchings" # per-segment subdirs (boot/, 800/, ...) -MD = ROOT / "docs" / "difficulty.md" -CSV = ROOT / ".run" / "difficulty.csv" + +# Per-binary config (Phase 9). main = the retail EXE (current paths = no-op default). +# The overlay src/asm subtree LAYOUT is a Phase-10 decision (main = the originals). +BINARIES = { + "main": dict(src="src", asm="asm/nonmatchings", md="docs/difficulty.md", csv=".run/difficulty.csv"), +} +SRCS = ASM_ROOT = MD = CSV = None # set by main() from --binary def find_s(name): """Locate .s in any asm/nonmatchings// subdir.""" @@ -65,7 +68,18 @@ def analyze(name): jtbl=jtbl, leaf=(ncalls == 0), score=score) def main(): - top = int(sys.argv[1]) if len(sys.argv) > 1 else 120 + import argparse + global SRCS, ASM_ROOT, MD, CSV + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("top", nargs="?", type=int, default=120) + ap.add_argument("--binary", default="main", choices=list(BINARIES)) + a = ap.parse_args() + cfg = BINARIES[a.binary] + SRCS = sorted((ROOT / cfg["src"]).glob("*.c")) + ASM_ROOT = ROOT / cfg["asm"] + MD = ROOT / cfg["md"] + CSV = ROOT / cfg["csv"] + top = a.top rows = [r for r in (analyze(n) for n in unmatched_stubs()) if r] rows.sort(key=lambda r: (r['score'], r['name'])) diff --git a/tools/dup_report.py b/tools/dup_report.py index daba9a9797..ddef478528 100644 --- a/tools/dup_report.py +++ b/tools/dup_report.py @@ -5,24 +5,36 @@ structurally identical (h_norm) to each other, so one match can be shared to the (match-once-share-many). Ghidra-free; consumes .run/sig.SLUS_007.26.jsonl (regenerate with `make sig-refresh` when the Ghidra DB changes). Writes docs/duplicates.md. -Usage: tools/dup_report.py [MIN_INS] (default 8 — ignore trivial stubs) +Usage: tools/dup_report.py [MIN_INS] [--binary ] (MIN_INS default 8; binary default main) """ import json, sys, hashlib, pathlib from collections import defaultdict ROOT = pathlib.Path(__file__).resolve().parent.parent -SIG = ROOT / ".run" / "sig.SLUS_007.26.jsonl" -MD = ROOT / "docs" / "duplicates.md" + +# Per-binary config (Phase 9). main = the retail EXE (current paths = no-op default); +# a second binary (Phase 10) adds its sig/md entry. +BINARIES = { + "main": dict(sig=".run/sig.SLUS_007.26.jsonl", md="docs/duplicates.md"), +} def main(): - minins = int(sys.argv[1]) if len(sys.argv) > 1 else 8 + import argparse + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("min_ins", nargs="?", type=int, default=8) + ap.add_argument("--binary", default="main", choices=list(BINARIES)) + a = ap.parse_args() + minins = a.min_ins + cfg = BINARIES[a.binary] + SIG = ROOT / cfg["sig"] + MD = ROOT / cfg["md"] raw = SIG.read_bytes() sig_sha = hashlib.sha1(raw).hexdigest() rows = [json.loads(l) for l in raw.decode().splitlines() if l.strip()] funcs = [r for r in rows if r['nins'] >= minins and r.get('src') != 'IMPORTED'] out = ["# EXE-wide duplicate function groups (generated by tools/dup_report.py)", - f"# sig: .run/sig.SLUS_007.26.jsonl sha1={sig_sha} ({len(rows)} funcs, {len(funcs)} with nins>={minins})", + f"# sig: {cfg['sig']} sha1={sig_sha} ({len(rows)} funcs, {len(funcs)} with nins>={minins})", "# Match the representative once -> share the body to the duplicates (h_exact = guaranteed", "# byte-match; h_norm = same source, re-point any single address constant).", ""]