diff --git a/config/regions.tsv b/config/regions.tsv index 0ae1427..2e3c149 100644 --- a/config/regions.tsv +++ b/config/regions.tsv @@ -606,6 +606,7 @@ 0x800FF47C 0x800FF4BC src/func_800FF47C.c cc1bin=gcc-2.8.1-psx 0x800FF6A4 0x800FF6DC src/func_800FF6A4.c 0x800FFBEC 0x800FFC3C src/func_800FFBEC.c maspsx=epilogue +0x800FFF60 0x800FFFEC src/func_800FFF60.c maspsx=epilogue 0x80100038 0x801000A0 src/func_80100038.c maspsx=epilogue 0x80100318 0x80100334 src/func_80100318.c 0x8010036C 0x801003A4 src/func_8010036C.c maspsx=epilogue @@ -619,6 +620,8 @@ 0x80101C5C 0x80101C7C src/func_80101C5C.c 0x80101CAC 0x80101CDC src/func_80101CAC.c 0x801027CC 0x801027F8 src/func_801027CC.c +0x80102A00 0x80102A80 src/func_80102A00.c maspsx=regread +0x80102A80 0x80102B04 src/func_80102A80.c maspsx=epilogue 0x80102B10 0x80102B2C src/func_80102B10.c maspsx=off 0x80102B30 0x80102B5C src/func_80102B30.c maspsx=off 0x80102FA4 0x80102FD4 src/func_80102FA4.c diff --git a/docs/MATCHING_COOKBOOK.md b/docs/MATCHING_COOKBOOK.md index 0bbf08c..3db6342 100644 --- a/docs/MATCHING_COOKBOOK.md +++ b/docs/MATCHING_COOKBOOK.md @@ -1095,6 +1095,56 @@ approximation** — `s = |dx|+|dz|`, `d = ||dx|-|dz||`, `h = s>>1`, `q = s>>2`, build is 120; the residual is cc1 hoisting all four loads and merging the two abs tests, i.e. the scheduler class. **The index's "unresolved" is now a named class rather than a gap.** +### 61c. THE REGISTRY DECIDES THE SPELLING (symbol vs literal for the same address) + +Phase 12, worker A, two rows closed **in opposite directions on the same afternoon**. The deciding +fact is the **registry entry**, not the code's shape: + +* **`0x800FB758`** — the address **IS** `gp`-marked in `config/symbols.tsv`, and the original writes it + **ABSOLUTELY** (`lui at,0x8012 / sw v0,0x2140(at)`), so the source must use the **LITERAL address**. + The symbol spelling lets the harness rewrite the access to `%gp_rel` and the row fails. + (Same shape as this entry's neighbour `D_80122140`.) +* **`0x800FFF60`** — the three addresses `0x801221A8` / `0x801221AA` / `0x801221AC` are **NOT** in the + registry, so the source must use **SYMBOLS**. As symbols the assembler macros expand **per site** — + a symbol *load* through the destination register (`lui v1,0x8012 / lw v1,0x21ac(v1)`, finding 2) and a + symbol *store* through `$at` (`lui at,0x8012 / sw zero,0x21ac(at)`, finding 3) — which **is** the + original's three separate `lui`s, and no CSE is possible across them. As **literals used twice** the + address is a VALUE, so cc1 commons it into a register: the `li s0,0x80120000 / ori s0,s0,0x21ac` the + index had recorded, with a `cc1=-G4` attempt that was chasing the wrong thing entirely. + +**The rule:** check the registry FIRST. `gp`-marked → literal. Not in the registry → symbol. This is +the source-side companion to finding 46's per-site access rule, and between them they cover both +directions: **the registry's marker is a statement about the SYMBOL, the access form is a property of +the SITE, and the spelling has to satisfy both.** + +### 61d. maspsx's `nop_on_reg_read` had a SECOND gap: the JUMP's delay-slot filler + +Phase 12, worker B's `0x80102A00` (128 B). Finding 27's gap was fixed by `line_jumps_via_reg`: a load +feeding a **register jump** (`jr`/`jalr`) got no delay nop. **This is the complementary case, in the +opposite direction** — the jump ignores the register entirely, but the instruction in its **delay slot** +reads it: + +``` +lw $2, D_8011FD2C <- loads $2 +jal func_80103FCC +sb $0, 0($2) <- the SLOT FILLER reads $2 +``` + +cc1 emits this with no `#nop` and no marker, correctly by the documented load-to-use rule. **ASPSX was +conservative across a jump**: the slot filler is part of the jump. maspsx even says so in its own +trace — `#nop # DEBUG: 'jal func_80103FCC' does not load from $2` — i.e. it declined the row for +exactly the reason the original accepted it. + +The predicate is now `_jump_slot_filler_reads_reg` in maspsx, **on the existing opt-in +`--nop-on-reg-read` / `maspsx=regread` token**, so the default path is byte-identical and the green +gate cannot move. **Restricted to `j`/`jal`: `jr`/`jalr` slot fillers are the same shape but +UNMEASURED**, and an opt-in predicate whose whole purpose is that the default stays identical has no +place for an unmeasured widening. Results on the row: default 124 (LENGTH-MISMATCH), `maspsx=regread` +**128 / 0 differing / MATCH** with **the source unchanged**. + +The tracked patch `tools/patches/maspsx-phase10-r1r2.patch` was **regenerated in the same commit** — +the exact step Phase 11's `ea51ac9` forgot, which left a fresh clone unable to rebuild the gate. + ### 62. Fail fast on an invalid region row A worker placed the source md5 in a claim row's 4th column. `sf3_merge` passed it through as a region diff --git a/src/func_80013114.c b/src/func_80013114.c new file mode 100644 index 0000000..f936269 --- /dev/null +++ b/src/func_80013114.c @@ -0,0 +1,8 @@ +void func_80013114(int *p0, int *p1) +{ + unsigned int a; + + a = *p1; + *p1 = (a & 0xff000000U) | (*p0 & 0xffffffU); + *p0 = (*p0 & 0xff000000U) | ((unsigned int)p1 & 0xffffffU); +} \ No newline at end of file diff --git a/src/func_80085B44.c b/src/func_80085B44.c new file mode 100644 index 0000000..a7173b7 --- /dev/null +++ b/src/func_80085B44.c @@ -0,0 +1,64 @@ +/* + * func_80085B44 — 60 bytes at 0x80085B44..0x80085B80 + * + * Copies three words from `src` into the consecutive globals at 0x80138450 / + * 0x80138454 / 0x80138458 when `src` is non-null, then unconditionally stores + * `val` into 0x8013845C. Matched on the FOURTH spelling; the first three were the + * right function in the WRONG SCHEDULE, and the lever is new: + * + * **THE LOAD/STORE INTERLEAVING REQUIRES *BOTH* SIDES VOLATILE.** + * + * The original alternates load and store, reusing ONE register for all three + * loaded values: + * lw v0,0(a0) / lui at,0x8014 / sw v0,-31664(at) + * lw v0,4(a0) / lui at,0x8014 / sw v0,-31660(at) + * lw v0,8(a0) / lui at,0x8014 / sw v0,-31656(at) + * + * Plain C, whatever the statement order, gives cc1's SCHEDULED form: the three + * loads batched into three DIFFERENT registers (v0, v1, a0) and then the three + * stores -- 25 differing bytes at the correct length. Measured spellings: + * (a) `int *src` + plain globals ................. 25 differing bytes (batched) + * (b) `volatile int *src` + plain globals ........ 25 differing bytes (batched) + * (c) a 12-byte STRUCT assignment ................ 34 differing bytes + * (d) plain `int *src` + `volatile` GLOBALS ...... 25 differing bytes (batched) + * (e) `volatile int *src` + `volatile` GLOBALS ... MATCH + * + * (b) and (d) are the informative failures: a volatile LOAD may still be hoisted + * above a plain store, and a volatile STORE does not stop a plain load from being + * hoisted above IT. cc1 orders volatile accesses only against OTHER volatile + * accesses, so the alternating form appears only when BOTH the reads and the writes + * are volatile. The register reuse (one v0, not three) follows from the order: + * each value dies in the store that immediately consumes it. + * + * NOTE the second `lui at,0x8014` repeats before every store: the symbol-store + * macro re-materialises the page every time, so a shared base register is NOT what + * the original did -- do not "optimise" the four globals into an array with one + * base, which emits one `lui` and a base register. + * + * LIMITS: `volatile` here is a MEASUREMENT, not a recovery. The bytes prove the + * original's accesses were ordered against one another; they cannot prove the + * original source said `volatile`, and any construct that produces the same order + * would be equally consistent with the bytes. The four words at 0x80138450 are + * named for their addresses and read as a state block only because they are + * consecutive and written together; nothing here establishes what they mean, and + * `val`'s fourth slot may or may not belong to the same object. The first three + * values are read through ONE base with constant offsets 0/4/8 (there is no + * `addiu` in the body), which is why the parameter is an `int *` and not an + * incrementing walk. The `nop` in the `beqz` delay slot is cc1's; the branch is the + * `src != 0` guard, and `!= 0` rather than a truth test is the shape that emits a + * comparison against `$0` here. + */ +extern volatile int D_80138450; +extern volatile int D_80138454; +extern volatile int D_80138458; +extern int D_8013845C; + +void func_80085B44(volatile int *src, int val) +{ + if (src != 0) { + D_80138450 = src[0]; + D_80138454 = src[1]; + D_80138458 = src[2]; + } + D_8013845C = val; +} diff --git a/src/func_800FBF5C.c b/src/func_800FBF5C.c deleted file mode 100644 index e9be921..0000000 --- a/src/func_800FBF5C.c +++ /dev/null @@ -1,20 +0,0 @@ -typedef struct { int f0; int f4; char pad[24]; } Rec_801455F0; -extern Rec_801455F0 D_801455F0[8]; -extern int D_80122790; -extern int D_80122794; - -void func_800FBF5C(void) -{ - Rec_801455F0 *p; - Rec_801455F0 *base; - int i; - - base = D_801455F0; - p = base + 7; - for (i = 7; i >= 0; i--) { - p->f4 = 0; - p--; - } - D_80122790 = (int)base; - D_80122794 = (int)base; -} \ No newline at end of file diff --git a/src/func_800FFF60.c b/src/func_800FFF60.c new file mode 100644 index 0000000..0ee02b1 --- /dev/null +++ b/src/func_800FFF60.c @@ -0,0 +1,112 @@ +/* + * func_800FFF60 — 140 bytes at 0x800FFF60..0x800FFFEC + * + * A guarded one-shot teardown: if the "open" flag is set, run two helper calls, + * move the flag's bit from one status word to another, clear the flag, and then — + * only if a third word is still set — read two shorts, call a handler and clear + * that word. + * + * The observed instructions are: + * lw v0,0x840(gp) \ + * addiu sp,sp,-24 | (the flag is tested BEFORE the frame is set up) + * beq v0,zero,epilogue | + * _sw ra,16(sp) / (delay slot) + * lw a1,0x850(gp) \ + * jal 0x80108990 / 0x80108990(8, D_80122188) + * _addiu a0,zero,8 (delay slot) + * lw a1,0x840(gp) \ + * jal 0x801087D0 / 0x801087D0(1, D_80122178) + * _addiu a0,zero,1 (delay slot) + * lw a0,0x840(gp) x = D_80122178 + * lw v0,0x84c(gp) \ + * sw zero,0x840(gp) | D_80122178 = 0 + * nor v1,zero,a0 | + * and v0,v0,v1 | + * sw v0,0x84c(gp) / D_80122184 &= ~x + * lw v0,0x848(gp) \ + * lui v1,0x8012 | + * lw v1,0x21ac(v1) | a THIRD separate lui — see below + * or v0,v0,a0 | + * sw v0,0x848(gp) / D_80122180 |= x + * beq v1,zero,epilogue \ + * _nop | + * lui a0,0x8012 | + * lh a0,0x21a8(a0) | + * lui a1,0x8012 | 0x80107874(D_801221A8, D_801221AA) + * lh a1,0x21aa(a1) | + * jal 0x80107874 | + * _nop | + * lui at,0x8012 | + * sw zero,0x21ac(at) / 0x801221AC = 0 + * epilogue: + * lw ra,16(sp) + * nop + * jr ra + * addiu sp,sp,24 <- the frame release is IN THE JUMP DELAY SLOT + * + * REGION TOKEN: `maspsx=epilogue` (shape A of cookbook 144 — `lw ra` immediately + * precedes the release), decided from the CANDIDATE's tail. + * + * THE LEVER IS SYMBOL-VERSUS-LITERAL, AND IT IS THE OPPOSITE DIRECTION FROM MY + * LAST ROW. The index recorded this row as *"maspsx on + default -G0: + * candidate_bytes=148 (cc1 CSEs the address of 0x801221AC into `s0`: lui s0/ori + * s0/sw s0/lw s0)"*, with a `cc1=-G4` attempt. The fix is not a flag and not the + * `-G` route: **0x801221A8, 0x801221AA and 0x801221AC must be SYMBOLS, not literal + * addresses.** A literal that cc1 sees used twice is an address VALUE and gets + * commoned into a register (the `li s0,0x80120000` / `ori s0,s0,0x21ac` the index + * describes). As distant symbols they are expanded **per site by the assembler + * macros** — a symbol LOAD expands through the DESTINATION register + * (`lui v1,0x8012` / `lw v1,0x21ac(v1)`, cookbook 2) and a symbol STORE through + * `$at` (`lui at,0x8012` / `sw zero,0x21ac(at)`, cookbook 3) — which is exactly + * the original's THREE separate `lui`s, and no CSE is possible across them. This + * is the mirror image of `0x800FB758`, where the registry marks the symbol `gp` and + * the source must use a LITERAL to keep the absolute encoding; here the registry has + * no entry at all and the source must use a SYMBOL. **The registry (and the + * harness's gp rewrite) decides which spelling a given address needs — the two + * rows together are the rule.** + * + * The four `gp` globals ARE read gp-relatively and are registered: D_80122178 + * (0x840), D_80122180 (0x848), D_80122184 (0x84c) and D_80122188 (0x850). + * + * The bit move is two statements, not one: `D_80122184 &= ~x;` then + * `D_80122180 |= x;`, with the old flag captured in a local BEFORE the flag word is + * zeroed (the load of `x` precedes the `sw zero` in the emitted order). + * + * LIMITS: the globals' names and widths, the three helpers' signatures (only the + * argument registers actually set are evidenced) and the meaning of the flag + * bits are hypotheses read off the instruction shape. Only the compiled bytes are + * evidence. + */ + +extern int D_80122178; +extern int D_80122180; +extern int D_80122184; +extern int D_80122188; +extern short D_801221A8; +extern short D_801221AA; +extern int D_801221AC; + +void func_80108990(int a0, int a1); +void func_801087D0(int a0, int a1); +void func_80107874(short a0, short a1); + +void func_800FFF60(void) +{ + int x; + + if (D_80122178 == 0) + return; + + func_80108990(8, D_80122188); + func_801087D0(1, D_80122178); + + x = D_80122178; + D_80122178 = 0; + D_80122184 &= ~x; + D_80122180 |= x; + + if (D_801221AC != 0) { + func_80107874(D_801221A8, D_801221AA); + D_801221AC = 0; + } +} diff --git a/src/func_80102A00.c b/src/func_80102A00.c new file mode 100644 index 0000000..806fd2b --- /dev/null +++ b/src/func_80102A00.c @@ -0,0 +1,52 @@ +/* + * func_80102A00 -- 128 bytes at 0x80102A00..0x80102A80. + * + * PHASE 12: **THIS REGION REQUIRES THE REGION TOKEN `maspsx=regread`.** Without it the + * same source is 124 bytes; with it, 128 and byte-exact. Do not "clean up" the token. + * + * Written by worker B, whose candidate this is; it was RELEASED in the ledger as an + * unmatched row and became a body only after the harness predicate below was patched. + * The whole finding is worker B's, taken from the harness's own --work intermediates. + * + * cc1 emits, with NO `#nop` and no marker: + * + * lw $2, D_8011FD2C <- loads $2 + * jal func_80103FCC + * sb $0, 0($2) <- the DELAY-SLOT FILLER reads $2 + * + * A one-instruction load-to-use gap is fine by the documented rule, so cc1 owes no nop + * and does not ask for one. But ASPSX was CONSERVATIVE ACROSS A JUMP: the slot filler is + * part of the jump, so a load one instruction before the jump whose result the slot reads + * still received a delay nop. maspsx declined the row for exactly the reason the original + * accepted it -- see its own trace line + * `#nop # DEBUG: 'jal func_80103FCC' does not load from $2`. + * + * The fix is in maspsx's opt-in `nop_on_reg_read` predicate + * (`tools/maspsx/maspsx/__init__.py`, `_jump_slot_filler_reads_reg`), carried + * reproducibly by `tools/patches/maspsx-phase10-r1r2.patch`. + */ +extern int D_8011FEB0; +extern char *D_8011FD20; +extern char *D_8011FD2C; + +void func_80103FEC(void); +void func_80103FCC(void); +void func_80109314(int); +void func_80109300(int); +void func_800F8F9C(int); +void func_800F8B6C(int); + +void func_80102A00(void) +{ + func_80103FEC(); + if (D_8011FEB0 == 1) { + func_80109314(0); + func_80109300(0); + } else { + func_800F8F9C(0); + func_800F8B6C(0); + } + *D_8011FD20 = 0; + *D_8011FD2C = 0; + func_80103FCC(); +} diff --git a/tools/patches/maspsx-phase10-r1r2.patch b/tools/patches/maspsx-phase10-r1r2.patch index 97d686e..400564b 100644 --- a/tools/patches/maspsx-phase10-r1r2.patch +++ b/tools/patches/maspsx-phase10-r1r2.patch @@ -1,119 +1,5 @@ ---- a/maspsx/__init__.py -+++ b/maspsx/__init__.py -@@ -78,6 +78,38 @@ - return line.strip() - - -+def line_jumps_via_reg(line: str, r_source: str) -> bool: -+ """True if `line` is a register jump whose target is `r_source`. -+ -+ Cookbook finding 27's gap: the delay-nop predicate (`line_loads_from_reg`) -+ recognises loads and branches but not `jr`/`jalr`, and `jr`/`jalr` are not in -+ `jump_mnemonics` either, so a load feeding a register jump got no delay nop. -+ -+ This is OPT-IN (`--nop-on-reg-read`): the default must stay byte-identical for -+ the corpus already matched against it. Local patch to a pinned vendored tool; -+ see docs/SETUP.md for provenance. -+ """ -+ line = strip_comments(line) -+ -+ # escape dollar -+ r_source = r_source.replace("$", r"\$") -+ -+ if match := re.match(r"^([A-z][A-z0-9]*)\s+(.*)$", line): -+ op, rest = match.group(1, 2) -+ else: -+ return False -+ -+ if op in ("jr", "jalr"): -+ # jr $31 -+ if re.match(rf"^{r_source}$", rest): -+ return True -+ # jalr $2,$31 (destination first, target last) -+ if re.match(rf"^.*,\s*{r_source}\s*$", rest): -+ return True -+ -+ return False -+ -+ - def line_loads_from_reg(line: str, r_source: str, loads_to_reg=False) -> bool: - """ - NOTE: Returns True even if line might use $at expansion -@@ -448,6 +480,9 @@ - gp_allow_la=False, - use_comm_section=False, - use_comm_for_lcomm=False, -+ no_jump_slot_nop=False, -+ nop_on_reg_read=False, -+ honour_nop_marker=False, - ): - self.lines = [x.strip() for x in lines] - -@@ -460,6 +495,18 @@ - self.nop_mflo_mfhi = nop_mflo_mfhi - self.nop_lw_lw = nop_lw_lw - -+ # Local opt-in modes (Phase 10, developer-authorised). Both default off so -+ # the matched corpus reproduces byte-identically. -+ # no_jump_slot_nop: suppress the unconditional reorder nop after a -+ # branch/jump, so GNU `as` can fill the slot itself (worker C's R1). -+ # nop_on_reg_read: additionally treat a following `jr`/`jalr` that uses -+ # the loaded register as needing the delay nop (worker C's R2). -+ self.no_jump_slot_nop = no_jump_slot_nop -+ self.nop_on_reg_read = nop_on_reg_read -+ # Opt-in (Phase 11 correction): honour cc1's explicit `#nop` marker even when -+ # the following instruction's own macro expansion would fill the delay slot. -+ self.honour_nop_marker = honour_nop_marker -+ - self.sltu_at = sltu_at - self.addiu_at = addiu_at - self.div_uses_tge = div_uses_tge -@@ -696,9 +743,35 @@ - ) -> List[str]: - res: List[str] = [] - -- if line_loads_from_reg(next_instruction, r_dest, loads_to_reg=self.nop_lw_lw): -+ reuse = line_loads_from_reg(next_instruction, r_dest, loads_to_reg=self.nop_lw_lw) -+ if not reuse and self.nop_on_reg_read: -+ # Cookbook finding 27's gap: the predicate above recognises loads and -+ # branches but not `jr`/`jalr`, so a load feeding a register jump got -+ # no delay nop. Opt-in, so the default path is unchanged. -+ reuse = line_jumps_via_reg(next_instruction, r_dest) -+ -+ if reuse: - nop_required = False - -+ # Phase 10/11: cc1 emits an explicit `#nop` marker when it thinks a -+ # load-delay nop is needed. For a BARE-SYMBOL store consumer -+ # (`sw $2,D_801221C4`) `uses_at` is True (the store expands via $at) while -+ # `nop_at_expansion` is False for ASPSX >= 2.30, so neither test below -+ # fires and the marker is ignored -- which is the correct default, because -+ # the store's own `lui $at` expansion fills the delay slot (worker C's -+ # 0x800AFDBC: the original is `lhu` / `lui at` / `sh` with NO nop, and -+ # honouring the marker costs 2 instructions there). -+ # -+ # Phase 10 made this unconditional and it was TOO BROAD: 0x80107C5C DOES -+ # want the nop (108 vs 112). So it is opt-in per region now -+ # (`maspsx=nopmarker`). Default off; no registered region depends on -+ # either behaviour (the gate is green at 510 regions both ways). -+ if self.honour_nop_marker and self.get_next_instruction( -+ skip=0, ignore_nop=False, ignore_set=True, ignore_label=True -+ ) == "#nop": -+ reason = "cc1 emitted an explicit '#nop' marker" -+ nop_required = True -+ - if not uses_at(next_instruction): - reason = f"'{next_instruction}' does not use $at" - nop_required = True -@@ -1102,7 +1175,7 @@ - - elif op in branch_mnemonics or op in jump_mnemonics: - res.append(line) -- if self.is_reorder: -+ if self.is_reorder and not self.no_jump_slot_nop: - res.append("nop # DEBUG: branch/jump") - - elif op == "move": +diff --git a/maspsx.py b/maspsx.py +index ed7f23a..8c3e899 100644 --- a/maspsx.py +++ b/maspsx.py @@ -1,4 +1,5 @@ @@ -122,7 +8,7 @@ import shutil import subprocess import sys -@@ -49,6 +50,72 @@ +@@ -49,6 +50,72 @@ def config_for_aspsx_version(aspsx_version_arg: str | None) -> AspsxVersionConfi return config @@ -195,7 +81,7 @@ def main() -> None: parser = argparse.ArgumentParser() parser.add_argument("--aspsx-version", type=str) -@@ -62,6 +129,12 @@ +@@ -62,6 +129,12 @@ def main() -> None: parser.add_argument("--passthrough", action="store_true") parser.add_argument("--use-comm-section", action="store_true") parser.add_argument("--use-comm-for-lcomm", action="store_true") @@ -208,7 +94,7 @@ # decomp.me debugging parser.add_argument("--print-output", action="store_true") parser.add_argument("--print-input", action="store_true") -@@ -152,6 +225,9 @@ +@@ -152,6 +225,9 @@ def main() -> None: gp_allow_la=version_config.gp_allow_la, use_comm_section=args.use_comm_section, use_comm_for_lcomm=args.use_comm_for_lcomm, @@ -218,7 +104,7 @@ ) try: -@@ -195,6 +271,8 @@ +@@ -195,6 +271,8 @@ def main() -> None: if process.returncode != 0: sys.exit(process.returncode) else: @@ -227,3 +113,177 @@ sys.stdout.write(out_text) +diff --git a/maspsx/__init__.py b/maspsx/__init__.py +index a45d7cc..5cc4bca 100644 +--- a/maspsx/__init__.py ++++ b/maspsx/__init__.py +@@ -78,6 +78,38 @@ def strip_comments(line: str) -> str: + return line.strip() + + ++def line_jumps_via_reg(line: str, r_source: str) -> bool: ++ """True if `line` is a register jump whose target is `r_source`. ++ ++ Cookbook finding 27's gap: the delay-nop predicate (`line_loads_from_reg`) ++ recognises loads and branches but not `jr`/`jalr`, and `jr`/`jalr` are not in ++ `jump_mnemonics` either, so a load feeding a register jump got no delay nop. ++ ++ This is OPT-IN (`--nop-on-reg-read`): the default must stay byte-identical for ++ the corpus already matched against it. Local patch to a pinned vendored tool; ++ see docs/SETUP.md for provenance. ++ """ ++ line = strip_comments(line) ++ ++ # escape dollar ++ r_source = r_source.replace("$", r"\$") ++ ++ if match := re.match(r"^([A-z][A-z0-9]*)\s+(.*)$", line): ++ op, rest = match.group(1, 2) ++ else: ++ return False ++ ++ if op in ("jr", "jalr"): ++ # jr $31 ++ if re.match(rf"^{r_source}$", rest): ++ return True ++ # jalr $2,$31 (destination first, target last) ++ if re.match(rf"^.*,\s*{r_source}\s*$", rest): ++ return True ++ ++ return False ++ ++ + def line_loads_from_reg(line: str, r_source: str, loads_to_reg=False) -> bool: + """ + NOTE: Returns True even if line might use $at expansion +@@ -448,6 +480,9 @@ class MaspsxProcessor: + gp_allow_la=False, + use_comm_section=False, + use_comm_for_lcomm=False, ++ no_jump_slot_nop=False, ++ nop_on_reg_read=False, ++ honour_nop_marker=False, + ): + self.lines = [x.strip() for x in lines] + +@@ -460,6 +495,18 @@ class MaspsxProcessor: + self.nop_mflo_mfhi = nop_mflo_mfhi + self.nop_lw_lw = nop_lw_lw + ++ # Local opt-in modes (Phase 10, developer-authorised). Both default off so ++ # the matched corpus reproduces byte-identically. ++ # no_jump_slot_nop: suppress the unconditional reorder nop after a ++ # branch/jump, so GNU `as` can fill the slot itself (worker C's R1). ++ # nop_on_reg_read: additionally treat a following `jr`/`jalr` that uses ++ # the loaded register as needing the delay nop (worker C's R2). ++ self.no_jump_slot_nop = no_jump_slot_nop ++ self.nop_on_reg_read = nop_on_reg_read ++ # Opt-in (Phase 11 correction): honour cc1's explicit `#nop` marker even when ++ # the following instruction's own macro expansion would fill the delay slot. ++ self.honour_nop_marker = honour_nop_marker ++ + self.sltu_at = sltu_at + self.addiu_at = addiu_at + self.div_uses_tge = div_uses_tge +@@ -691,14 +738,91 @@ class MaspsxProcessor: + + return False + ++ def _jump_slot_filler_reads_reg(self, jump_instruction: str, r_dest: str) -> bool: ++ """True if `jump_instruction` is a `j`/`jal` whose DELAY SLOT reads `r_dest`. ++ ++ PHASE 12, worker B's `0x80102A00` (128 B). The `line_jumps_via_reg` gap below is ++ "the JUMP's own target register is the loaded one". **This is the complementary ++ case, and it is the opposite direction:** the jump ignores the register entirely, ++ but the instruction ASPSX placed in its DELAY SLOT reads it. ++ ++ The sequence cc1 actually emits for this row is ++ ++ lw $2, D_8011FD2C <- loads $2 ++ jal func_80103FCC ++ sb $0, 0($2) <- the SLOT FILLER reads $2 ++ ++ with no `#nop` and no marker, because by the documented load-to-use rule a ++ one-instruction gap is fine. It IS fine for cc1's own schedule. But ASPSX was ++ **conservative across a jump**: the slot filler is part of the jump, so a load ++ one instruction before the jump whose result the slot reads still got a delay ++ nop. maspsx says so in its own trace -- ++ `#nop # DEBUG: 'jal func_80103FCC' does not load from $2` -- i.e. the existing ++ predicate declines the row for exactly the reason the original accepted it. ++ ++ Restricted to `j`/`jal`. **`jr`/`jalr` slot fillers are the same shape but are ++ UNMEASURED**, and this is an opt-in predicate whose whole point is that the ++ default path stays byte-identical, so an unmeasured widening has no place in ++ it. Widen it only with a row that measures the difference. ++ ++ Opt-in via `--nop-on-reg-read` (region token `maspsx=regread`), like ++ `line_jumps_via_reg`; no registered region names that token today, which is why ++ this cannot move the green gate. ++ """ ++ match = re.match(r"^([A-z][A-z0-9]*)\b", strip_comments(jump_instruction).strip()) ++ if not match or match.group(1) not in ("j", "jal"): ++ return False ++ ++ # `skip=1` from the current line is the jump; `skip=2` would be the filler. The ++ # caller has already resolved the jump as `skip=0`, so ask for one past it. ++ # `ignore_nop=False` on purpose: a real `nop` in the slot must be seen as the ++ # filler, not skipped over, so that a slot already filled reports False. ++ slot = self.get_next_instruction( ++ skip=1, ignore_nop=False, ignore_set=True, ignore_label=True ++ ) ++ if not slot: ++ return False ++ return line_loads_from_reg(slot, r_dest, loads_to_reg=self.nop_lw_lw) ++ + def _handle_nop_before_next_instruction( + self, next_instruction: str, r_dest: str + ) -> List[str]: + res: List[str] = [] + +- if line_loads_from_reg(next_instruction, r_dest, loads_to_reg=self.nop_lw_lw): ++ reuse = line_loads_from_reg(next_instruction, r_dest, loads_to_reg=self.nop_lw_lw) ++ if not reuse and self.nop_on_reg_read: ++ # Cookbook finding 27's gap: the predicate above recognises loads and ++ # branches but not `jr`/`jalr`, so a load feeding a register jump got ++ # no delay nop. Opt-in, so the default path is unchanged. ++ reuse = line_jumps_via_reg(next_instruction, r_dest) ++ if not reuse: ++ # Phase 12, worker B: the SIBLING gap, in the opposite direction -- ++ # the jump does not use the register, but its delay-slot filler does. ++ # Same opt-in token, so the default path is still unchanged. ++ reuse = self._jump_slot_filler_reads_reg(next_instruction, r_dest) ++ ++ if reuse: + nop_required = False + ++ # Phase 10/11: cc1 emits an explicit `#nop` marker when it thinks a ++ # load-delay nop is needed. For a BARE-SYMBOL store consumer ++ # (`sw $2,D_801221C4`) `uses_at` is True (the store expands via $at) while ++ # `nop_at_expansion` is False for ASPSX >= 2.30, so neither test below ++ # fires and the marker is ignored -- which is the correct default, because ++ # the store's own `lui $at` expansion fills the delay slot (worker C's ++ # 0x800AFDBC: the original is `lhu` / `lui at` / `sh` with NO nop, and ++ # honouring the marker costs 2 instructions there). ++ # ++ # Phase 10 made this unconditional and it was TOO BROAD: 0x80107C5C DOES ++ # want the nop (108 vs 112). So it is opt-in per region now ++ # (`maspsx=nopmarker`). Default off; no registered region depends on ++ # either behaviour (the gate is green at 510 regions both ways). ++ if self.honour_nop_marker and self.get_next_instruction( ++ skip=0, ignore_nop=False, ignore_set=True, ignore_label=True ++ ) == "#nop": ++ reason = "cc1 emitted an explicit '#nop' marker" ++ nop_required = True ++ + if not uses_at(next_instruction): + reason = f"'{next_instruction}' does not use $at" + nop_required = True +@@ -1102,7 +1226,7 @@ class MaspsxProcessor: + + elif op in branch_mnemonics or op in jump_mnemonics: + res.append(line) +- if self.is_reorder: ++ if self.is_reorder and not self.no_jump_slot_nop: + res.append("nop # DEBUG: branch/jump") + + elif op == "move":