phase7: census duplicate bodies and expose a zero band of false positives

Matching conventions require a duplicate check before registering, because a
shared body is matched once and registered once per address. Phase 6 did that
check by hand and found one 12-byte pair. tools/sf3_dupes now hashes every
derived extent body and groups exact duplicates.

Results: 2284 extents, 65 multi-address groups, 2104 singletons. Only 10 groups
contain code (24 addresses, all exact-graded); 55 are all-zero bodies. The
hand-found pair 0x800262E0/0x800262EC is reproduced as g0002, which is the check
that the census measures what it claims. The largest real groups are 712 bytes
(0x8001084C/0x800189E8) and 436 bytes.

The zero groups are a real finding: 252 extents have all-zero bodies, 245 inside
the zero band 0x80147000..0x80170000. The cause is the inventory's jal grade,
which decodes every word as an instruction -- in a data region a word with
opcode 3 is graded as a call whose target lands in the zero band. The census
flags those groups rather than hiding them, and the worklist must exclude
degenerate bodies.

The census is tracked rather than ignored as the plan said, because it holds
addresses, sizes and grades only (the same class as the tracked inventory and
extents tables) and the worklist must be reproducible from tracked inputs. The
content hash is computed and never written.
This commit is contained in:
Christopher Williams
2026-09-23 22:15:51 -04:00
parent 6988ca96b0
commit 2000cc4101
6 changed files with 761 additions and 2 deletions
+261
View File
@@ -0,0 +1,261 @@
#!/usr/bin/env python3
"""Duplicate-body census for the USA executable.
Matching conventions require checking for duplicates *before* registering a
match, because a body shared by several addresses is matched **once** and
registered once per address (`config/regions.tsv`, N rows to one source). Phase 6
did that check by hand and found one shared body; there was no way to know how
many others exist.
This tool hashes the body of every derived extent in
`config/function_extents.tsv` and groups exact duplicates.
## What is grouped
Every extent row -- any grade -- is hashed over exactly the bytes
`[address, end)` that `tools/sf3_extents` derived. Each member keeps its grade so
a consumer can prefer groups whose members are all `exact`. `--exact-only`
restricts the census to `exact` extents, and `--min-size` drops bodies shorter
than N bytes (very short bodies collide for boring reasons, e.g. `jr ra; nop`).
## What is written
A tracked, sorted TSV of **multi-address groups only**:
group<TAB>size<TAB>count<TAB>addresses<TAB>grades<TAB>flags
* `group` is a stable label (`g0001`, ...) assigned in `(size, first address)`
order, so the file is deterministic;
* `addresses` is a comma-separated list in ascending order;
* `grades` is the comma-separated grade of each address, in the same order;
* `flags` is `zero` when the shared body is nothing but zero bytes, otherwise
`-`. An all-zero "body" is not a function: it is a zero-filled region the
walk ran through, which happens when a data word decodes as a `jal` whose
target lands in the zero band. Those groups are recorded, not hidden, so the
finding stays visible and the worklist can exclude them.
The content hash is computed and **never written**: grouping is the finding, and
addresses, sizes and grades are the only ROM-derived quantities this project
tracks (the same rule `config/function_inventory.tsv` and
`config/function_extents.tsv` follow). No instruction bytes are read into the
output, and the executable is read only from the caller-supplied path.
The census is a measurement, not a match claim: two identical bodies are
evidence of shared code, and nothing more. A shared *tail* is not a duplicate
function, so a group member still has to be verified as a real function start by
the ordinary workflow.
Exit codes: 0 success, 2 usage or environment error.
"""
from __future__ import annotations
import argparse
import hashlib
from pathlib import Path
import struct
import sys
from typing import Sequence
EXE_MAGIC = b"PS-X EXE"
HEADER_SIZE = 0x800
PAYLOAD_LMA = 0x800
EXTENT_GRADES = frozenset({"exact", "fallthrough", "indirect", "escape"})
HEADER_LINES = (
"# Syphon Filter 3 (USA) duplicate-body census.",
"# Columns: group<TAB>size<TAB>count<TAB>addresses<TAB>grades<TAB>flags.",
"# Multi-address groups only; addresses, sizes and grades only; no bytes.",
"# A group is evidence of a shared body, not a claim that every member is",
"# a real function start: verify a member before registering it.",
"# flags=zero means the body is all zero bytes -- a zero-filled region the",
"# walk ran through, not a function. Exclude those from matching.",
"# Regenerate: ./tools/sf3_dupes census --exe '<exe>' \\",
"# --extents config/function_extents.tsv --out config/duplicate_bodies.tsv --force",
)
class ToolError(Exception):
"""A usage or environment problem; maps to exit code 2."""
def parse_hex(text: str, label: str) -> int:
try:
return int(text, 16)
except ValueError as exc:
raise ToolError(f"{label}: not a hex address: {text!r}") from exc
def require_file(path: Path, label: str) -> Path:
if not path.is_file():
raise ToolError(f"{label} is not a regular file: {path}")
return path
def resolve_output(path: Path, force: bool) -> Path:
if path.exists() or path.is_symlink():
if not force:
raise ToolError(f"output already exists (use --force to overwrite): {path}")
if not path.is_file() or path.is_symlink():
raise ToolError(f"output is not a regular file: {path}")
return path
def parse_psx_exe(header: bytes) -> tuple[int, int, int]:
if len(header) < HEADER_SIZE:
raise ToolError("executable is smaller than a PS-X EXE header")
if header[:8] != EXE_MAGIC:
raise ToolError("executable does not carry the PS-X EXE magic")
entry, _gp, text_address, text_size = struct.unpack_from("<IIII", header, 0x10)
if text_size == 0:
raise ToolError("PS-X EXE header declares an empty payload")
return entry, text_address, text_size
def load_extents(path: Path) -> list[tuple[int, int, str]]:
"""Read a generated extents table; keep rows that carry an extent."""
rows: list[tuple[int, int, str]] = []
for number, raw in enumerate(path.read_text(encoding="utf-8").splitlines(), 1):
line = raw.split("#", 1)[0].strip()
if not line:
continue
fields = line.split("\t")
if len(fields) != 7:
raise ToolError(f"extents line {number}: expected seven fields")
address = parse_hex(fields[0], f"extents line {number}")
grade = fields[5]
if fields[1] == "-":
continue
end = parse_hex(fields[1], f"extents line {number}")
if end <= address:
raise ToolError(f"extents line {number}: end is not after address")
if grade not in EXTENT_GRADES:
raise ToolError(f"extents line {number}: {grade!r} carries no extent")
rows.append((address, end, grade))
if not rows:
raise ToolError("extents table contains no rows with an extent")
return rows
class Group:
"""One set of addresses whose bodies are byte-identical."""
__slots__ = ("size", "members", "all_zero")
def __init__(self, size: int, all_zero: bool) -> None:
self.size = size
self.members: list[tuple[int, str]] = []
self.all_zero = all_zero
def census(payload: bytes, text_address: int, rows: Sequence[tuple[int, int, str]],
min_size: int, exact_only: bool) -> tuple[list[Group], int]:
"""Group rows by (size, body bytes). Returns the groups and the singleton count."""
buckets: dict[tuple[int, bytes], Group] = {}
for address, end, grade in rows:
if exact_only and grade != "exact":
continue
size = end - address
if size < min_size:
continue
start = address - text_address
body = payload[start:start + size]
if len(body) != size:
raise ToolError(f"extent 0x{address:08X}..0x{end:08X} is outside the payload")
key = (size, hashlib.sha1(body).digest())
group = buckets.get(key)
if group is None:
group = buckets[key] = Group(size, not any(body))
group.members.append((address, grade))
groups = [group for group in buckets.values() if len(group.members) > 1]
singletons = sum(1 for group in buckets.values() if len(group.members) == 1)
for group in groups:
group.members.sort()
groups.sort(key=lambda group: (group.size, group.members[0][0]))
return groups, singletons
def format_census(groups: Sequence[Group]) -> str:
lines = list(HEADER_LINES)
for index, group in enumerate(groups, 1):
addresses = ",".join(f"0x{address:08X}" for address, _grade in group.members)
grades = ",".join(grade for _address, grade in group.members)
flags = "zero" if group.all_zero else "-"
lines.append("\t".join((
f"g{index:04d}", str(group.size), str(len(group.members)), addresses, grades, flags,
)))
return "\n".join(lines) + "\n"
def command_census(args: argparse.Namespace) -> int:
exe_path = require_file(args.exe, "executable")
extents_path = require_file(args.extents, "function extents")
out = resolve_output(args.out, args.force)
if args.min_size < 0:
raise ToolError("--min-size cannot be negative")
data = exe_path.read_bytes()
_entry, text_address, text_size = parse_psx_exe(data)
payload = data[PAYLOAD_LMA:PAYLOAD_LMA + text_size]
rows = load_extents(extents_path)
groups, singletons = census(payload, text_address, rows, args.min_size, args.exact_only)
out.parent.mkdir(parents=True, exist_ok=True)
out.write_text(format_census(groups), encoding="ascii")
addresses = sum(len(group.members) for group in groups)
duplicated = sum(group.size * (len(group.members) - 1) for group in groups)
zero_groups = sum(1 for group in groups if group.all_zero)
print(f"extents={len(rows)}")
print(f"groups={len(groups)}")
print(f"groups_zero_body={zero_groups}")
print(f"groups_with_code={len(groups) - zero_groups}")
print(f"addresses_in_groups={addresses}")
print(f"singletons={singletons}")
print(f"duplicated_bytes={duplicated}")
if groups:
print(f"largest_group={max(len(group.members) for group in groups)}")
print(f"output={out}")
return 0
def build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
)
subparsers = parser.add_subparsers(dest="command", required=True)
census_parser = subparsers.add_parser("census", help="group byte-identical bodies")
census_parser.add_argument("--exe", required=True, type=Path)
census_parser.add_argument("--extents", required=True, type=Path,
help="derived function extents (tools/sf3_extents output)")
census_parser.add_argument("--out", required=True, type=Path)
census_parser.add_argument("--force", action="store_true",
help="overwrite an existing output file")
census_parser.add_argument("--min-size", type=int, default=1, dest="min_size",
help="ignore bodies shorter than N bytes (default 1)")
census_parser.add_argument("--exact-only", action="store_true",
help="census only extents graded exact")
census_parser.set_defaults(handler=command_census)
return parser
def main(argv: Sequence[str] | None = None) -> int:
parser = build_parser()
args = parser.parse_args(argv)
try:
return args.handler(args)
except ToolError as exc:
print(f"error: {exc}", file=sys.stderr)
return 2
except OSError as exc:
print(f"error: {exc}", file=sys.stderr)
return 2
if __name__ == "__main__":
raise SystemExit(main())
+262
View File
@@ -0,0 +1,262 @@
"""Synthetic-only tests for the duplicate-body census tool.
Fixtures are self-authored bytes in temporary directories. They never read the
local disc or the extracted game executable.
"""
from __future__ import annotations
import contextlib
import importlib.machinery
import importlib.util
import io
from pathlib import Path
import struct
import sys
import tempfile
import unittest
TOOL_PATH = Path(__file__).resolve().parents[1] / "sf3_dupes"
def _load_tool() -> object:
loader = importlib.machinery.SourceFileLoader("sf3_dupes_under_test", str(TOOL_PATH))
spec = importlib.util.spec_from_loader(loader.name, loader)
if spec is None:
raise RuntimeError("could not create an import specification")
module = importlib.util.module_from_spec(spec)
sys.modules[spec.name] = module
loader.exec_module(module)
return module
sf3_dupes = _load_tool()
PAYLOAD = 0x80010000
# Payload-relative layout of the fixture.
SHARED_A = 0x00 # 8 bytes of code, duplicated at SHARED_B
SHARED_B = 0x10
UNIQUE = 0x20 # 12 bytes, no twin
ZERO_A = 0x30 # 16 zero bytes, duplicated at ZERO_B
ZERO_B = 0x40
def _payload() -> bytes:
words = [0] * (0x50 // 4)
words[SHARED_A // 4] = 0x03E00008 # jr ra
words[SHARED_B // 4] = 0x03E00008 # jr ra
words[UNIQUE // 4] = 0x24020001
words[UNIQUE // 4 + 1] = 0x24030002
words[UNIQUE // 4 + 2] = 0x03E00008
return b"".join(struct.pack("<I", word) for word in words)
def _synthetic_exe(payload: bytes, entry: int) -> bytes:
header = bytearray(0x800)
header[:8] = b"PS-X EXE"
struct.pack_into("<IIII", header, 0x10, entry, 0, PAYLOAD, len(payload))
return bytes(header) + payload
def _extents_text() -> str:
lines = [
"# synthetic extents",
"# Columns: address<TAB>end<TAB>size<TAB>next<TAB>gap<TAB>grade<TAB>evidence.",
]
def row(offset: int, size: int, grade: str, evidence: str) -> str:
start = PAYLOAD + offset
end = start + size
return f"0x{start:08X}\t0x{end:08X}\t{size}\t-\t-\t{grade}\t{evidence}"
lines.append(row(SHARED_A, 8, "exact", "term=jr_ra"))
lines.append(row(SHARED_B, 8, "exact", "term=jr_ra"))
lines.append(row(UNIQUE, 12, "exact", "term=jr_ra"))
lines.append(row(ZERO_A, 16, "fallthrough", "hit=0x80010040"))
lines.append(row(ZERO_B, 16, "fallthrough", "hit=0x80010050"))
lines.append(f"0x{PAYLOAD + 0x60:08X}\t-\t-\t-\t-\tcontained\tinside=0x80010000")
return "\n".join(lines) + "\n"
def _rows() -> list[tuple[int, int, str]]:
return [
(PAYLOAD + SHARED_A, PAYLOAD + SHARED_A + 8, "exact"),
(PAYLOAD + SHARED_B, PAYLOAD + SHARED_B + 8, "exact"),
(PAYLOAD + UNIQUE, PAYLOAD + UNIQUE + 12, "exact"),
(PAYLOAD + ZERO_A, PAYLOAD + ZERO_A + 16, "fallthrough"),
(PAYLOAD + ZERO_B, PAYLOAD + ZERO_B + 16, "fallthrough"),
]
class ParseTests(unittest.TestCase):
def test_rejects_missing_magic(self) -> None:
with self.assertRaises(sf3_dupes.ToolError):
sf3_dupes.parse_psx_exe(bytes(0x800))
def test_rejects_empty_payload(self) -> None:
header = bytearray(0x800)
header[:8] = b"PS-X EXE"
with self.assertRaises(sf3_dupes.ToolError):
sf3_dupes.parse_psx_exe(bytes(header))
class ExtentLoadingTests(unittest.TestCase):
def test_skips_rows_without_an_extent(self) -> None:
with tempfile.TemporaryDirectory() as tmp:
path = Path(tmp) / "extents.tsv"
path.write_text(_extents_text(), encoding="ascii")
rows = sf3_dupes.load_extents(path)
self.assertEqual(len(rows), 5)
def test_rejects_a_row_whose_grade_carries_no_extent(self) -> None:
with tempfile.TemporaryDirectory() as tmp:
path = Path(tmp) / "extents.tsv"
path.write_text(
f"0x{PAYLOAD:08X}\t0x{PAYLOAD + 8:08X}\t8\t-\t-\tcontained\tinside=0x1\n",
encoding="ascii")
with self.assertRaises(sf3_dupes.ToolError):
sf3_dupes.load_extents(path)
def test_rejects_an_empty_extents_table(self) -> None:
with tempfile.TemporaryDirectory() as tmp:
path = Path(tmp) / "extents.tsv"
path.write_text("# nothing\n", encoding="ascii")
with self.assertRaises(sf3_dupes.ToolError):
sf3_dupes.load_extents(path)
class CensusTests(unittest.TestCase):
def setUp(self) -> None:
self.payload = _payload()
def _groups(self, **kwargs: object) -> list[object]:
options = {"min_size": 1, "exact_only": False}
options.update(kwargs)
groups, _singletons = sf3_dupes.census(
self.payload, PAYLOAD, _rows(), options["min_size"], options["exact_only"])
return groups
def test_identical_bodies_are_grouped(self) -> None:
groups = self._groups()
addresses = [[address for address, _grade in group.members] for group in groups]
self.assertIn([PAYLOAD + SHARED_A, PAYLOAD + SHARED_B], addresses)
self.assertIn([PAYLOAD + ZERO_A, PAYLOAD + ZERO_B], addresses)
def test_unique_bodies_are_not_grouped(self) -> None:
groups = self._groups()
for group in groups:
self.assertNotIn(PAYLOAD + UNIQUE, [address for address, _g in group.members])
def test_a_zero_body_is_flagged(self) -> None:
groups = {group.members[0][0]: group for group in self._groups()}
self.assertTrue(groups[PAYLOAD + ZERO_A].all_zero)
self.assertFalse(groups[PAYLOAD + SHARED_A].all_zero)
def test_min_size_drops_short_bodies(self) -> None:
groups = self._groups(min_size=10)
sizes = {group.size for group in groups}
self.assertEqual(sizes, {16})
def test_exact_only_drops_other_grades(self) -> None:
groups = self._groups(exact_only=True)
self.assertEqual([group.size for group in groups], [8])
for group in groups:
for _address, grade in group.members:
self.assertEqual(grade, "exact")
def test_singletons_are_counted_not_listed(self) -> None:
groups, singletons = sf3_dupes.census(self.payload, PAYLOAD, _rows(), 1, False)
self.assertEqual(singletons, 1)
self.assertEqual(len(groups), 2)
def test_groups_are_ordered_by_size_then_address(self) -> None:
groups = self._groups()
keys = [(group.size, group.members[0][0]) for group in groups]
self.assertEqual(keys, sorted(keys))
def test_an_extent_outside_the_payload_is_rejected(self) -> None:
rows = [(PAYLOAD + 0x1000, PAYLOAD + 0x1008, "exact")]
with self.assertRaises(sf3_dupes.ToolError):
sf3_dupes.census(self.payload, PAYLOAD, rows, 1, False)
class FormatTests(unittest.TestCase):
def test_header_and_column_count(self) -> None:
groups, _ = sf3_dupes.census(_payload(), PAYLOAD, _rows(), 1, False)
text = sf3_dupes.format_census(groups)
body = [line for line in text.splitlines() if not line.startswith("#")]
self.assertTrue(body)
for line in body:
self.assertEqual(len(line.split("\t")), 6)
def test_group_labels_are_stable_and_sorted(self) -> None:
groups, _ = sf3_dupes.census(_payload(), PAYLOAD, _rows(), 1, False)
labels = [line.split("\t")[0] for line in sf3_dupes.format_census(groups).splitlines()
if not line.startswith("#")]
self.assertEqual(labels, ["g0001", "g0002"])
def test_no_body_bytes_are_written(self) -> None:
groups, _ = sf3_dupes.census(_payload(), PAYLOAD, _rows(), 1, False)
text = sf3_dupes.format_census(groups)
self.assertNotIn("03e00008", text.lower())
self.assertNotIn("24020001", text.lower())
class MainTests(unittest.TestCase):
def _write_inputs(self, root: Path) -> tuple[Path, Path]:
exe = root / "synthetic.exe"
exe.write_bytes(_synthetic_exe(_payload(), PAYLOAD))
extents = root / "extents.tsv"
extents.write_text(_extents_text(), encoding="ascii")
return exe, extents
def test_census_writes_the_table(self) -> None:
with tempfile.TemporaryDirectory() as tmp:
root = Path(tmp)
exe, extents = self._write_inputs(root)
out = root / "dupes.tsv"
stdout = io.StringIO()
with contextlib.redirect_stdout(stdout):
rc = sf3_dupes.main(["census", "--exe", str(exe), "--extents", str(extents),
"--out", str(out)])
self.assertEqual(rc, 0)
text = out.read_text(encoding="ascii")
self.assertIn(f"0x{PAYLOAD + SHARED_A:08X},0x{PAYLOAD + SHARED_B:08X}", text)
self.assertIn("groups=2", stdout.getvalue())
self.assertIn("groups_zero_body=1", stdout.getvalue())
def test_census_refuses_an_existing_output(self) -> None:
with tempfile.TemporaryDirectory() as tmp:
root = Path(tmp)
exe, extents = self._write_inputs(root)
out = root / "dupes.tsv"
out.write_text("", encoding="ascii")
rc = sf3_dupes.main(["census", "--exe", str(exe), "--extents", str(extents),
"--out", str(out)])
self.assertEqual(rc, 2)
def test_census_force_overwrites(self) -> None:
with tempfile.TemporaryDirectory() as tmp:
root = Path(tmp)
exe, extents = self._write_inputs(root)
out = root / "dupes.tsv"
out.write_text("stale\n", encoding="ascii")
with contextlib.redirect_stdout(io.StringIO()):
rc = sf3_dupes.main(["census", "--exe", str(exe), "--extents", str(extents),
"--out", str(out), "--force"])
self.assertEqual(rc, 0)
self.assertNotIn("stale", out.read_text(encoding="ascii"))
def test_rejects_a_negative_min_size(self) -> None:
with tempfile.TemporaryDirectory() as tmp:
root = Path(tmp)
exe, extents = self._write_inputs(root)
rc = sf3_dupes.main(["census", "--exe", str(exe), "--extents", str(extents),
"--out", str(root / "dupes.tsv"), "--min-size", "-1"])
self.assertEqual(rc, 2)
if __name__ == "__main__":
unittest.main()