phase10: cycle-1 merges 1-7 — 432 distinct bodies / 441 regions
32 new bodies from 400, all verified on the candidate whole-binary gate before promotion. SHA-1 e173426c157384ebf1b6caf8c6fea18a85a14af9 stable. Registry requests granted (each byte-verified with a failing control): cc1=-G8 on 0x800A6BEC; gp=-D_80121B88 on 0x80015D50 symbols D_80122700, D_80122704, D_80121AD4 (gp) Harness: per-region maspsx modes wired through sf3_match (maspsx=noreordernop, maspsx=regread) plus --no-jump-slot-nop/--nop-on-reg-read for range. Both are opt-in and default-off; make check green at 441 with them off, suite 229 -> 232 tests. Carried as a TRACKED patch (tools/patches/maspsx-phase10-r1r2.patch) because tools/maspsx/ is git-ignored, so an in-place edit would not survive a fresh clone; patch verified to reproduce the working tree byte-identically. R1/R2 are recorded as a MEASURED NEGATIVE: neither closes a region (cookbook finding 40 has the mechanism and the remaining developer-owned route). Docs: cookbook finding 40 (rare-epilogue mechanism + why the obvious maspsx fix fails); SETUP.md maspsx patch provenance and apply step. Negatives: 0x8010AA28 imported; index sorted by address (140 rows, 0 registered). Full clean audit green: make clean && make all exit 0, cmp exit 0, both SHA-1 match, registry 441/0 overlaps/0 bad extents/0 missing sources, 0 firewall.
This commit is contained in:
+1138
-1154
File diff suppressed because it is too large
Load Diff
@@ -8,26 +8,48 @@
|
||||
0x80010810 60 near-match register-tiebreak
|
||||
0x8001084C 712 blocked trapping-arithmetic
|
||||
0x80011084 - near-match -
|
||||
0x8001278C - near-match -
|
||||
0x80012834 - near-match -
|
||||
0x80012A10 - near-match -
|
||||
0x80012A48 80 near-match primitive-init scheduler-bound family
|
||||
0x80012A98 68 near-match cc1-scheduler-bound (constant stores before stack-arg loads; 2 coordinator spellings incl. named locals, 72B both; same family as 0x80012A10/0x80012AE0)
|
||||
0x80012AE0 - near-match -
|
||||
0x80012CFC - near-match -
|
||||
0x80012D54 56 near-match -
|
||||
0x80013114 64 near-match -
|
||||
0x800161E0 - near-match -
|
||||
0x80016224 - near-match -
|
||||
0x80016268 - near-match -
|
||||
0x80016E24 - near-match -
|
||||
0x800179E0 - near-match -
|
||||
0x80017A38 - near-match -
|
||||
0x80018284 80 near-match struct-copy jr-slot shape (8 loads/8 stores + v0-zero in slot)
|
||||
0x80018CB0 120 blocked trapping-arithmetic
|
||||
0x80019700 - near-match -
|
||||
0x8001D98C 436 blocked,deferred trapping-arithmetic
|
||||
0x8001DC20 - near-match -
|
||||
0x8001EAFC 12 deferred shared-block-not-a-function
|
||||
0x80022E44 52 near-match adjacent-zero-store merge
|
||||
0x800245D8 - near-match -
|
||||
0x80024630 56 near-match stack-routed min (temp elided by optimizer)
|
||||
0x8002515C - near-match -
|
||||
0x800254B0 - near-match -
|
||||
0x80025758 - near-match -
|
||||
0x800259DC - near-match -
|
||||
0x80025C08 - near-match -
|
||||
0x80025EAC - near-match -
|
||||
0x8002622C - near-match -
|
||||
0x800266A8 68 near-match byte-replication alloc (first sll register; fresh-context re-spelling 3B; named-locals regresses)
|
||||
0x80027BE8 - near-match -
|
||||
0x8002D014 - near-match -
|
||||
0x8002D060 72 near-match -
|
||||
0x80037984 - near-match -
|
||||
0x800379E0 - near-match -
|
||||
0x80039308 - near-match -
|
||||
0x8003A5F4 80 near-match alloc+layout (j-to-done second guard; worker-C residual confirmed by coordinator)
|
||||
0x8003EB18 - near-match -
|
||||
0x80042CE0 80 near-match paired-array geometry (stride 1428, single walked at-register)
|
||||
0x80042D88 76 near-match strength-reduce (sra5+sll2 vs cc1 folded shift)
|
||||
0x80042DD4 - near-match -
|
||||
0x80042E10 - near-match -
|
||||
0x80042E68 - near-match -
|
||||
@@ -35,37 +57,61 @@
|
||||
0x80045F00 - near-match -
|
||||
0x800460AC 40 near-match constant-materialisation-order
|
||||
0x80047468 72 near-match alloc (move-zero-in-delay + *76 strength-reduce; 2 coordinator spellings 76B; the *76 and pointer-slot table-lever patterns confirmed)
|
||||
0x8004820C - near-match -
|
||||
0x800496CC - near-match -
|
||||
0x800518BC 88 near-match return-merge/sltiu (3 spellings: goto-shared 80B, combined cond 80B, two-exit if 80B; original keeps move-zero at separate target + j-to-shared-return; sltiu needs unsigned val which alone fixes the compare byte but not the 8-byte layout)
|
||||
0x800582AC 64 near-match -
|
||||
0x8005DEF8 124 near-match register-tiebreak
|
||||
0x80065494 80 near-match popcount scheduling (base in beqz delay slot)
|
||||
0x800690E4 68 near-match loop-rotation+head-match-early-return (3 spellings: do/while arg-test-top 84B, while+conditional 76B, while-test 64B; the j/li early path and bnez-back-to-bne rotation are the residual)
|
||||
0x8006AD5C - near-match -
|
||||
0x8006AE04 - near-match -
|
||||
0x8006B470 - near-match -
|
||||
0x8006F95C - near-match -
|
||||
0x8006FAA0 76 near-match cc1-scheduler (load/subu/store interleave + sra ordering; 1 coordinator attempt 72B)
|
||||
0x8006FAEC 88 near-match -
|
||||
0x8007374C - near-match -
|
||||
0x800751E8 - near-match -
|
||||
0x8007A114 - near-match -
|
||||
0x8007E85C - near-match -
|
||||
0x800807E0 - near-match -
|
||||
0x80085B44 60 near-match -
|
||||
0x8008A198 56 near-match -
|
||||
0x8008F478 - near-match -
|
||||
0x80090990 - near-match -
|
||||
0x80092104 - near-match -
|
||||
0x80094370 84 near-match triple-deref copy loop (dest = *(*(a0+0xc)+0x160)+0x160; 1 coordinator attempt 100B)
|
||||
0x80099E34 40 near-match alloc-tiebreak (symbol-form address in a1)
|
||||
0x8009B218 - near-match -
|
||||
0x8009C69C 60 near-match register-tiebreak
|
||||
0x8009C750 436 deferred trapping-arithmetic
|
||||
0x8009F064 - near-match -
|
||||
0x800A6658 88 near-match alloc+symbol-recompute (3 coordinator spellings; original recomputes lui/addu per iteration into two separate symbol arrays; cc1 pre-materializes base pointers — worker-A finding-8 CSE class)
|
||||
0x800A82D0 64 near-match -
|
||||
0x800A86B4 - near-match -
|
||||
0x800A8AEC - near-match -
|
||||
0x800AAC44 - near-match -
|
||||
0x800AC9D8 56 near-match constant-materialisation-order
|
||||
0x800B0B88 - near-match -
|
||||
0x800B34A4 88 near-match alloc-tiebreak (bit-index/base register pair a0/a1 vs cc1 a0/v1; 2 spellings both 80B; original keeps base in a1 via direct +0x10 load)
|
||||
0x800C1E54 - near-match -
|
||||
0x800C3514 88 near-match alloc+layout (priority selector; original: all stack loads hoisted, beqz+nop+li groups, sltu first test; cc1 interleaves with bnez — 2 coordinator spellings 84B)
|
||||
0x800F3BB4 164 near-match global-ra-save guard (ra through D_8012A57C is CRT/library asm, not usable C; GTE $0..$7 block verified as the rotation/translation macros — gte_ldTRX/TRY/TRZ added to gtemac.h from this row)
|
||||
0x800F4B54 - near-match -
|
||||
0x800F4C08 - near-match -
|
||||
0x800F50B0 - near-match -
|
||||
0x800F6ED0 44 near-match cc1-scheduling
|
||||
0x800F6F20 - near-match -
|
||||
0x800F7610 - near-match -
|
||||
0x800F7930 96 near-match record-builder alloc (flag/0x100 branch shape + packed pair; 1 coordinator attempt 80B)
|
||||
0x800F7E00 - near-match -
|
||||
0x800F88F0 - near-match -
|
||||
0x800F8928 - near-match -
|
||||
0x800F9134 - near-match -
|
||||
0x800FBD80 - near-match -
|
||||
0x800FBDC0 - near-match -
|
||||
0x800FBF5C 56 near-match -
|
||||
0x800FC280 - near-match -
|
||||
0x800FCC30 - near-match -
|
||||
0x800FD220 88 near-match exit-duplication
|
||||
0x800FDE54 - near-match -
|
||||
@@ -75,10 +121,12 @@
|
||||
0x800FFB74 72 near-match base-materialization (original: lui v0,0x8014 + addiu 0x5ac0 + addiu 0x678 in v0; cc1: lui v1 + one addiu or ori; gp stores need D_80122168/6C gp rows; 4 coordinator spellings)
|
||||
0x800FFBBC 48 near-match reorg-thread-fill
|
||||
0x80100334 - near-match -
|
||||
0x8010036C 56 near-match rare-epilogue-order (F21)
|
||||
0x80100998 80 near-match -
|
||||
0x80101C5C 32 blocked -
|
||||
0x80101C7C - near-match -
|
||||
0x80101E50 76 near-match custom-compare loop shape (back-up-mismatch path; cc1 rewrites to 52B)
|
||||
0x80102F58 - near-match -
|
||||
0x80102FE4 - near-match -
|
||||
0x80103AA4 - near-match -
|
||||
0x8010400C 44 near-match constant-base-in-register
|
||||
@@ -90,56 +138,9 @@
|
||||
0x80108034 24 blocked gp-thunk
|
||||
0x8010804C 16 blocked gp-thunk
|
||||
0x80108578 56 near-match two-epilogue
|
||||
0x801086B0 - near-match -
|
||||
0x801092C0 52 near-match -
|
||||
0x801097A0 - near-match -
|
||||
0x8010AA28 112 near-match maspsx mutual exclusion (finding 17 in its purest form: maspsx=off -> 2 differing bytes at 0x8010AA6C; maspsx on -> 124 bytes, three unfilled delay slots; maspsx 2.56 has no flag suppressing only the jump-slot nop)
|
||||
0x8010B420 - near-match -
|
||||
0x801BB450 - near-match -
|
||||
0x80042D88 76 near-match strength-reduce (sra5+sll2 vs cc1 folded shift)
|
||||
0x80018284 80 near-match struct-copy jr-slot shape (8 loads/8 stores + v0-zero in slot)
|
||||
0x80042CE0 80 near-match paired-array geometry (stride 1428, single walked at-register)
|
||||
0x80065494 80 near-match popcount scheduling (base in beqz delay slot)
|
||||
0x80012A48 80 near-match primitive-init scheduler-bound family
|
||||
0x8003A5F4 80 near-match alloc+layout (j-to-done second guard; worker-C residual confirmed by coordinator)
|
||||
0x800F9134 - near-match -
|
||||
0x80094370 84 near-match triple-deref copy loop (dest = *(*(a0+0xc)+0x160)+0x160; 1 coordinator attempt 100B)
|
||||
0x80099E34 40 near-match alloc-tiebreak (symbol-form address in a1)
|
||||
0x80016E24 - near-match -
|
||||
0x800F88F0 - near-match -
|
||||
0x80024630 56 near-match stack-routed min (temp elided by optimizer)
|
||||
0x80022E44 52 near-match adjacent-zero-store merge
|
||||
0x8010036C 56 near-match rare-epilogue-order (F21)
|
||||
0x800FBDC0 - near-match -
|
||||
0x800F4C08 - near-match -
|
||||
0x800254B0 - near-match -
|
||||
0x800751E8 - near-match -
|
||||
0x80017A38 - near-match -
|
||||
0x8004820C - near-match -
|
||||
0x80102F58 - near-match -
|
||||
0x800F7610 - near-match -
|
||||
0x800F8928 - near-match -
|
||||
0x80025758 - near-match -
|
||||
0x8006AE04 - near-match -
|
||||
0x8006AD5C - near-match -
|
||||
0x8006B470 - near-match -
|
||||
0x800245D8 - near-match -
|
||||
0x8007374C - near-match -
|
||||
0x800A86B4 - near-match -
|
||||
0x80027BE8 - near-match -
|
||||
0x80037984 - near-match -
|
||||
0x8007E85C - near-match -
|
||||
0x801086B0 - near-match -
|
||||
0x800379E0 - near-match -
|
||||
0x8001278C - near-match -
|
||||
0x80025EAC - near-match -
|
||||
0x800B0B88 - near-match -
|
||||
0x80012CFC - near-match -
|
||||
0x8006F95C - near-match -
|
||||
0x800FC280 - near-match -
|
||||
0x800259DC - near-match -
|
||||
0x800179E0 - near-match -
|
||||
0x800C1E54 - near-match -
|
||||
0x800AAC44 - near-match -
|
||||
0x8003EB18 - near-match -
|
||||
0x80012834 - near-match -
|
||||
0x8007A114 - near-match -
|
||||
0x800A8AEC - near-match -
|
||||
|
||||
|
@@ -18,8 +18,11 @@
|
||||
0x800129C8 0x80012A10 src/func_800129C8.c
|
||||
0x80012D8C 0x80012DBC src/func_80012D8C.c
|
||||
0x80012DBC 0x80012DE8 src/func_80012DBC.c
|
||||
0x80012F24 0x80012F80 src/func_80012F24.c
|
||||
0x80013C88 0x80013C90 src/func_80013C88.c
|
||||
0x80013D04 0x80013D44 src/func_80013D04.c
|
||||
0x8001587C 0x8001590C src/func_8001587C.c
|
||||
0x80015D50 0x80015DBC src/func_80015D50.c gp=-D_80121B88
|
||||
0x800160E8 0x80016110 src/func_800160E8.c
|
||||
0x80016110 0x80016120 src/func_80016110.c
|
||||
0x80016120 0x80016158 src/func_80016120.c
|
||||
@@ -52,6 +55,7 @@
|
||||
0x8001AA9C 0x8001AAA8 src/func_8001AA9C.c
|
||||
0x8001AE3C 0x8001AE50 src/func_8001AE3C.c
|
||||
0x8001CE40 0x8001CE70 src/func_8001CE40.c
|
||||
0x80021C64 0x80021CDC src/func_80021C64.c
|
||||
0x80021F24 0x80021F50 src/func_80021F24.c
|
||||
0x80021F50 0x80021F64 src/func_80021F50.c
|
||||
0x80021F64 0x80021F88 src/func_80021F64.c
|
||||
@@ -61,6 +65,7 @@
|
||||
0x80022A60 0x80022A98 src/func_80022A60.c
|
||||
0x80022FB8 0x80022FCC src/func_80022FB8.c
|
||||
0x80022FCC 0x80022FFC src/func_80022FCC.c
|
||||
0x80023080 0x800230E4 src/func_80023080.c
|
||||
0x800230E4 0x8002311C src/func_800230E4.c
|
||||
0x80024C14 0x80024C34 src/func_80024C14.c
|
||||
0x80025070 0x800250AC src/func_80025070.c
|
||||
@@ -76,6 +81,7 @@
|
||||
0x80026C2C 0x80026C7C src/func_80026C2C.c
|
||||
0x80026F14 0x80026F3C src/func_80026F14.c
|
||||
0x800276B0 0x800276D4 src/func_800276B0.c
|
||||
0x80027E1C 0x80027ECC src/func_80027E1C.c
|
||||
0x80028150 0x800281A4 src/func_80028150.c
|
||||
0x800281A4 0x800281E4 src/func_800281A4.c
|
||||
0x800282EC 0x80028344 src/func_800282EC.c
|
||||
@@ -86,6 +92,7 @@
|
||||
0x8002AC84 0x8002ACBC src/func_8002AC84.c
|
||||
0x8002C6EC 0x8002C728 src/func_8002C6EC.c
|
||||
0x8002C728 0x8002C764 src/func_8002C728.c
|
||||
0x8002C764 0x8002C7BC src/func_8002C764.c
|
||||
0x8002C7BC 0x8002C7EC src/func_8002C7BC.c
|
||||
0x8002C888 0x8002C894 src/func_8002C888.c
|
||||
0x8002C894 0x8002C8A4 src/func_8002C894.c
|
||||
@@ -100,7 +107,9 @@
|
||||
0x8002E3FC 0x8002E44C src/func_8002E3FC.c
|
||||
0x8002E4C8 0x8002E4F0 src/func_8002E4C8.c
|
||||
0x8002E7C4 0x8002E7E4 src/func_8002E7C4.c
|
||||
0x8002E870 0x8002E8EC src/func_8002E870.c
|
||||
0x8002E968 0x8002E9AC src/func_8002E968.c
|
||||
0x8002F0D0 0x8002F118 src/func_8002F0D0.c
|
||||
0x8002F160 0x8002F1A4 src/func_8002F160.c
|
||||
0x8002F1A4 0x8002F1D8 src/func_8002F1A4.c
|
||||
0x8002F2F8 0x8002F300 src/func_8002F2F8.c
|
||||
@@ -129,9 +138,11 @@
|
||||
0x8003B2F0 0x8003B320 src/func_8003B2F0.c
|
||||
0x8003B320 0x8003B34C src/func_8003B320.c
|
||||
0x8003CB8C 0x8003CBE4 src/func_8003CB8C.c
|
||||
0x800419D0 0x80041A24 src/func_800419D0.c
|
||||
0x80041A24 0x80041A58 src/func_80041A24.c
|
||||
0x80042088 0x80042090 src/func_80042088.c
|
||||
0x80042964 0x800429B0 src/func_80042964.c
|
||||
0x800429B0 0x800429F0 src/func_800429B0.c
|
||||
0x80042D64 0x80042D88 src/func_80042D64.c
|
||||
0x80043D8C 0x80043DC4 src/func_80043D8C.c
|
||||
0x80044F58 0x80044FA4 src/func_80044F58.c gp=-D_80121BFC
|
||||
@@ -139,6 +150,7 @@
|
||||
0x80045388 0x800453C0 src/func_80045388.c
|
||||
0x800453C0 0x800453F8 src/func_800453C0.c
|
||||
0x800474B0 0x80047508 src/func_800474B0.c
|
||||
0x80047984 0x80047A14 src/func_80047984.c
|
||||
0x80048E20 0x80048E70 src/func_80048E20.c
|
||||
0x80049298 0x800492E4 src/func_80049298.c gp=-D_80121BFC
|
||||
0x8004C060 0x8004C090 src/func_8004C060.c
|
||||
@@ -151,6 +163,7 @@
|
||||
0x8005182C 0x80051864 src/func_8005182C.c
|
||||
0x80052C98 0x80052CAC src/func_80052C98.c
|
||||
0x80057524 0x80057564 src/func_80057524.c
|
||||
0x80057564 0x800575C4 src/func_80057564.c
|
||||
0x80057748 0x80057798 src/func_80057748.c
|
||||
0x80057798 0x800577E8 src/func_80057798.c
|
||||
0x800577E8 0x8005784C src/func_800577E8.c
|
||||
@@ -171,6 +184,7 @@
|
||||
0x80058230 0x80058288 src/func_80058230.c
|
||||
0x80058288 0x800582AC src/func_80058288.c
|
||||
0x8005E3D0 0x8005E3F4 src/func_8005E3D0.c
|
||||
0x8005E79C 0x8005E820 src/func_8005E79C.c
|
||||
0x8005ED6C 0x8005EDBC src/func_8005ED6C.c
|
||||
0x800658FC 0x80065930 src/func_800658FC.c
|
||||
0x80065930 0x80065980 src/func_80065930.c
|
||||
@@ -202,6 +216,7 @@
|
||||
0x80073250 0x80073284 src/func_80073250.c
|
||||
0x80073364 0x800733C4 src/func_80073364.c
|
||||
0x800734A4 0x800734DC src/func_800734A4.c
|
||||
0x80074C00 0x80074C74 src/func_80074C00.c
|
||||
0x8007A404 0x8007A428 src/func_8007A404.c
|
||||
0x8007C4A8 0x8007C4EC src/func_8007C4A8.c
|
||||
0x8007C4EC 0x8007C524 src/func_8007C4EC.c
|
||||
@@ -213,8 +228,10 @@
|
||||
0x80082750 0x800827A8 src/func_80082750.c
|
||||
0x800827A8 0x800827C4 src/func_800827A8.c
|
||||
0x80082914 0x80082944 src/func_80082914.c
|
||||
0x800833CC 0x80083440 src/func_800833CC.c
|
||||
0x80083440 0x80083470 src/func_80083440.c
|
||||
0x80083470 0x800834B8 src/func_80083470.c
|
||||
0x800834B8 0x80083504 src/func_800834B8.c
|
||||
0x80083504 0x8008352C src/func_80083504.c
|
||||
0x8008352C 0x8008355C src/func_8008352C.c
|
||||
0x80085B80 0x80085B90 src/func_80085B80.c
|
||||
@@ -230,6 +247,8 @@
|
||||
0x8008B8E4 0x8008B8F4 src/func_8008B8E4.c
|
||||
0x8008B8F4 0x8008B910 src/func_8008B8F4.c
|
||||
0x8008B910 0x8008B960 src/func_8008B910.c
|
||||
0x8008D9AC 0x8008DA0C src/func_8008D9AC.c
|
||||
0x8008F2E0 0x8008F338 src/func_8008F2E0.c
|
||||
0x8008F4A0 0x8008F4AC src/func_8008F4A0.c
|
||||
0x8008F4AC 0x8008F4F4 src/func_8008F4AC.c
|
||||
0x8008F4F4 0x8008F508 src/func_8008F4F4.c
|
||||
@@ -239,10 +258,12 @@
|
||||
0x80090028 0x80090048 src/func_80090028.c
|
||||
0x80090048 0x80090058 src/func_80090048.c
|
||||
0x800900A0 0x800900CC src/func_800900A0.c
|
||||
0x80090894 0x800908E4 src/func_80090894.c
|
||||
0x80090A44 0x80090A70 src/func_80090A44.c
|
||||
0x80090B64 0x80090B7C src/func_80090B64.c
|
||||
0x80090B7C 0x80090BB4 src/func_80090B7C.c
|
||||
0x80090C8C 0x80090CAC src/func_80090C8C.c
|
||||
0x80090CAC 0x80090D00 src/func_80090CAC.c
|
||||
0x8009107C 0x800910B0 src/func_8009107C.c
|
||||
0x800912D4 0x800912FC src/func_800912D4.c
|
||||
0x800912FC 0x8009132C src/func_800912FC.c
|
||||
@@ -252,10 +273,13 @@
|
||||
0x800920BC 0x800920DC src/func_800920BC.c
|
||||
0x800920DC 0x80092104 src/func_800920DC.c
|
||||
0x80092130 0x8009214C src/func_80092130.c
|
||||
0x8009214C 0x800921C0 src/func_8009214C.c
|
||||
0x80093A08 0x80093A64 src/func_80093A08.c
|
||||
0x80093A64 0x80093A84 src/func_80093A64.c
|
||||
0x800943C4 0x800943E0 src/func_800943C4.c
|
||||
0x80096324 0x80096364 src/func_80096324.c
|
||||
0x80099024 0x80099078 src/func_80099024.c
|
||||
0x80099078 0x8009916C src/func_80099078.c
|
||||
0x80099A94 0x80099AE4 src/func_80099A94.c
|
||||
0x80099DC4 0x80099E14 src/func_80099DC4.c
|
||||
0x80099E14 0x80099E34 src/func_80099E14.c
|
||||
@@ -273,6 +297,9 @@
|
||||
0x800A6268 0x800A6294 src/func_800A6268.c
|
||||
0x800A648C 0x800A64C8 src/func_800A648C.c
|
||||
0x800A6840 0x800A6880 src/func_800A6840.c
|
||||
0x800A6934 0x800A6998 src/func_800A6934.c
|
||||
0x800A6A18 0x800A6A70 src/func_800A6A18.c
|
||||
0x800A6BEC 0x800A6C34 src/func_800A6BEC.c cc1=-G8
|
||||
0x800A745C 0x800A74BC src/func_800A745C.c
|
||||
0x800A74BC 0x800A74D0 src/func_800A74BC.c
|
||||
0x800A8B48 0x800A8B8C src/func_800A8B48.c
|
||||
@@ -284,14 +311,17 @@
|
||||
0x800AA56C 0x800AA59C src/func_800AA56C.c
|
||||
0x800AC818 0x800AC85C src/func_800AC818.c
|
||||
0x800AC85C 0x800AC884 src/func_800AC85C.c
|
||||
0x800ACAC8 0x800ACB34 src/func_800ACAC8.c
|
||||
0x800ACC00 0x800ACC20 src/func_800ACC00.c
|
||||
0x800AE0F4 0x800AE10C src/func_800AE0F4.c
|
||||
0x800AE4DC 0x800AE548 src/func_800AE4DC.c
|
||||
0x800AE548 0x800AE574 src/func_800AE548.c
|
||||
0x800AF1FC 0x800AF20C src/func_800AF1FC.c
|
||||
0x800AF20C 0x800AF260 src/func_800AF20C.c
|
||||
0x800AFB1C 0x800AFB6C src/func_800AFB1C.c
|
||||
0x800AFFB8 0x800AFFE4 src/func_800AFFB8.c
|
||||
0x800B0E30 0x800B0E64 src/func_800B0E30.c
|
||||
0x800B1D5C 0x800B1DD0 src/func_800B1D5C.c
|
||||
0x800B24EC 0x800B2534 src/func_800B24EC.c
|
||||
0x800B2534 0x800B255C src/func_800B2534.c
|
||||
0x800B3474 0x800B34A4 src/func_800B3474.c
|
||||
@@ -302,6 +332,7 @@
|
||||
0x800B6838 0x800B6894 src/func_800B6838.c
|
||||
0x800B6BDC 0x800B6C14 src/func_800B6BDC.c
|
||||
0x800B6C14 0x800B6C60 src/func_800B6C14.c
|
||||
0x800B6C60 0x800B6CB4 src/func_800B6C60.c
|
||||
0x800B7230 0x800B7264 src/func_800B7230.c
|
||||
0x800B74D0 0x800B7524 src/func_800B74D0.c
|
||||
0x800BBDEC 0x800BBDF8 src/func_800BBDEC.c
|
||||
@@ -388,6 +419,7 @@
|
||||
0x80104C38 0x80104C60 src/func_80104C38.c maspsx=off
|
||||
0x80104C6C 0x80104CA0 src/func_80104C6C.c maspsx=off
|
||||
0x8010513C 0x80105148 src/func_8010513C.c
|
||||
0x80105148 0x801051CC src/func_80105148.c
|
||||
0x8010543C 0x80105474 src/func_8010543C.c
|
||||
0x80105B34 0x80105B54 src/func_80105B34.c
|
||||
0x80105B54 0x80105B68 src/func_80105B54.c
|
||||
|
||||
|
@@ -46,6 +46,7 @@ D_80121A80 0x80121A80 gp
|
||||
D_80121A88 0x80121A88 gp
|
||||
D_80121AAC 0x80121AAC gp
|
||||
D_80121AD0 0x80121AD0 gp
|
||||
D_80121AD4 0x80121AD4 gp
|
||||
D_80121B14 0x80121B14 gp
|
||||
D_80121B18 0x80121B18 gp
|
||||
D_80121B1C 0x80121B1C gp
|
||||
@@ -356,6 +357,8 @@ D_801226EC 0x801226EC gp
|
||||
D_801226F0 0x801226F0 gp
|
||||
D_801226F4 0x801226F4 gp
|
||||
D_801226F8 0x801226F8 gp
|
||||
D_80122700 0x80122700 gp
|
||||
D_80122704 0x80122704 gp
|
||||
D_80122708 0x80122708 gp
|
||||
D_80122714 0x80122714 gp
|
||||
D_80122716 0x80122716 gp
|
||||
|
||||
|
@@ -639,3 +639,47 @@ Phase 10 takes the rare-epilogue classes on.
|
||||
Sony/SN toolchain story is the hypothesis for Phase 10's Goal B.
|
||||
- The exact `-G`, the CRT entry, and library-versus-game-code remain
|
||||
unresolved, with the class exclusions now precisely bounded.
|
||||
|
||||
## Phase 10 — findings 40+ (coordinated tail squeeze, 2026-09-24)
|
||||
|
||||
### 40. The rare epilogue's mechanism, and why the obvious maspsx fix does not work
|
||||
|
||||
Finding 35's class is now mechanistically closed. GNU `as` in reorder mode fills a jump delay
|
||||
slot with the immediately-preceding instruction, but **refuses when doing so would place `jr ra`
|
||||
in the load-delay slot of `lw ra`**. So the class splits on whether any instruction intervenes
|
||||
between `lw ra` and the frame release:
|
||||
|
||||
| source before the `j $31` | `as` output | filled? |
|
||||
|---|---|---|
|
||||
| `lw ra,20(sp)` / `lw s0,16(sp)` / `addu sp,sp,24` | `lw ra` / `lw s0` / `jr ra` / `addiu sp,sp,24` | yes |
|
||||
| `lw ra,16(sp)` / `addu sp,sp,24` | `lw ra` / `addiu sp,sp,24` / `jr ra` / `nop` | no |
|
||||
| `addu sp,sp,24` alone | `jr ra` / `addiu sp,sp,24` | yes |
|
||||
| `lw ra,16(sp)` / `nop` / `j $31` / `addiu` | `lw ra` / `nop` / `jr ra` / `nop` / `addiu` | no (does not remove an existing nop) |
|
||||
|
||||
**Decision rule (costs one compile):** compile the row and read the CANDIDATE's `.s`, not the
|
||||
original. If the epilogue sits inside cc1's own `.set noreorder` block (which happens when a
|
||||
macro-using insn such as `move` is present), cc1 filled the slot itself and the row is ordinary.
|
||||
If cc1 left the slot empty, check the hazard: an intervening instruction means `as` *can* fill it;
|
||||
`lw ra` immediately before the release means it cannot, and the row is a harness row.
|
||||
|
||||
**The obvious fix is closed, measured.** maspsx normally forces the whole function into
|
||||
`.set noreorder` (it emits it after every `.ent`) and then supplies every delay slot itself, so
|
||||
`as` never gets a chance to fill. Suppressing only maspsx's jump-slot `nop` therefore restores the
|
||||
correct *length* but leaves the `jr ra` with an **empty** delay slot, because `as` will not insert
|
||||
one under `.set noreorder`. Also suppressing the function-level `.set noreorder` lets `as` fill
|
||||
the epilogue *correctly* (`lw ra` / `lw s2` / `lw s1` / `lw s0` / `jr ra` / `addiu sp,sp,32`, exactly
|
||||
the original) — but then `as` over-fills *other* slots and the region comes out +4 bytes. Switching
|
||||
to `.set reorder` at the jump alone does not re-enable the fill; reorder mode must be in effect from
|
||||
the function start.
|
||||
|
||||
*Basis:* coordinator measurements with the repo's binutils, plus worker A's per-guard classifier,
|
||||
worker B's standalone `as` test and maspsx source citation (`maspsx/__init__.py`, the branch/jump
|
||||
`nop` adjacent to the `move` expansion), and worker C's `0x800FFF60`/`0x800F6948` observations —
|
||||
the three reports disagreed and the disagreement is what produced the rule above.
|
||||
*Limit:* the only remaining route is a tracked post-pass that performs exactly one transform (move
|
||||
the frame release into the jump slot, inserting the load-delay `nop` where required), i.e. modelling
|
||||
ASPSX's fill for that site on maspsx's output. That is a developer-owned harness decision and was not
|
||||
taken in Phase 10. The two opt-in maspsx modes shipped in Phase 10 (`maspsx=noreordernop`,
|
||||
`maspsx=regread`) are implemented and default-off but **neither has been shown to close a region**:
|
||||
on `0x800FFF60` R1 makes the length *worse* (140 → 128) because it also removes `nop`s the original
|
||||
needs, and R2's predicate did not fire.
|
||||
|
||||
@@ -70,6 +70,29 @@ P3-T3 also added ignored local GNU Binutils and Maspsx candidates after explicit
|
||||
| GNU Binutils cross tools | 2.46.0, target `mipsel-none-elf` | AUR recipe `mipsel-none-elf-binutils 2.46.0-1`, fetched with `paru -G` into ignored `tools/mipsel-none-elf-binutils/`. Its GNU FTP source archive passed recipe SHA-256 `d75a94f4d73e7a4086f7513e67e439e8fcdcbb726ffe63f4661744e6256b2cf2` and PGP verification using imported recipe-declared Nick Clifton fingerprint `3A24BC1E8FB409FA9F14371813FCEF89DD9E3C4F`. Built locally with `makepkg --noconfirm --nocheck` and extracted to ignored `prefix/`; never installed system-wide. |
|
||||
| Maspsx | tagged `aspsx` commit `86ccd7d8c89682c0562d1425bbb15a09f42eb522` | Cloned from `https://github.com/mkst/maspsx.git` into ignored `tools/maspsx/`. It is a GNU-as compatibility transformer for GCC output, not the original ASPSX or a verified USA matching tool. It was inspected but not required for the P3-T3 payload-data build. |
|
||||
|
||||
### Maspsx local patch (Phase 10)
|
||||
|
||||
`tools/maspsx/` is **ignored**, so a local edit there is invisible to a fresh clone and would silently
|
||||
break reproducibility. Phase 10's two opt-in maspsx modes are therefore carried as a **tracked patch**:
|
||||
|
||||
```bash
|
||||
# from tools/
|
||||
patch -p1 < patches/maspsx-phase10-r1r2.patch
|
||||
```
|
||||
|
||||
The patch applies to the pinned `86ccd7d8` checkout above and touches exactly two files
|
||||
(`maspsx/__init__.py`, `maspsx.py`). It was verified by reconstructing the pristine files, applying the
|
||||
patch, and diffing the result against the working tree (byte-identical). It adds:
|
||||
|
||||
- `--no-jump-slot-nop` (region token `maspsx=noreordernop`) — suppress the unconditional reorder `nop`
|
||||
after a branch/jump so GNU `as` can fill the slot.
|
||||
- `--nop-on-reg-read` (region token `maspsx=regread`) — extend the load-delay predicate (cookbook
|
||||
finding 27) so a following `jr`/`jalr` that *uses* the loaded register also gets the delay `nop`.
|
||||
|
||||
Both default **off**, and `make check` is green at 439 regions with them off, so the matched corpus is
|
||||
byte-identical. **Status: implemented, but neither mode has been shown to close a region.** See
|
||||
`docs/MATCHING_COOKBOOK.md` finding 40 for the measured negative result before relying on them.
|
||||
|
||||
The local GNU assembler accepted self-authored COP2 and Splat GTE-macro synthetic sources, while LLVM's MIPS assembler did not. This is assembler-capability evidence only; it does not identify the original assembler or compiler.
|
||||
|
||||
P3-T5 added tracked synthetic-only `tools/sf3_fingerprint_probe`. It writes a self-authored C fixture into a caller-selected new output directory, compiles twice with explicitly supplied paths, assembles both outputs, and requires byte-identical objects. Its validated invocation is:
|
||||
|
||||
@@ -107,6 +107,34 @@ candidate-gate-before-promote step is wired correctly for this phase.
|
||||
| Cycle | New bodies | Total bodies | New regions | Total regions | Result |
|
||||
|---|---|---|---|---|---|
|
||||
| (baseline) | — | 400 | — | 409 | green, head `25bf4a3` |
|
||||
| 1 (in progress) | +32 | **432** | +32 | **441** | gate MATCH, `make check` green, full clean audit green |
|
||||
|
||||
Worker yield at this point: A 7 claims, B 11 claims (1 negative, 1 deferred, 1 R1-pending),
|
||||
C 14 claims. All 32 new bodies verified by the coordinator on the candidate whole-binary gate
|
||||
before promotion; the registry has never been corrupted.
|
||||
|
||||
## R1/R2 harness outcome (P10-T2) — measured negative result
|
||||
|
||||
The developer authorised both maspsx modes. Both are implemented as **opt-in** modes
|
||||
(`maspsx=noreordernop`, `maspsx=regread`), carried as a **tracked patch**
|
||||
(`tools/patches/maspsx-phase10-r1r2.patch`, verified to reproduce the working tree byte-identically
|
||||
from the pristine pinned checkout), and documented in `docs/SETUP.md`.
|
||||
|
||||
**Neither mode has been shown to close a region, and this is recorded as a negative result.**
|
||||
|
||||
- `make check` is green at 441 regions with both modes off, and the suite grew 229 → 232 tests, so the
|
||||
default path is provably unchanged.
|
||||
- **R1** (suppress the jump-slot `nop`) restores the correct LENGTH on worker B's acceptance row
|
||||
`0x80102A80` (136 → 132) but leaves the `jr ra` with an **empty** delay slot, because maspsx forces
|
||||
the function into `.set noreorder` and `as` will not insert one there. Also suppressing the
|
||||
function-level `.set noreorder` makes `as` fill the epilogue *exactly* right — and then over-fill
|
||||
other slots, for +4 bytes.
|
||||
- On worker C's `0x800FFF60`, R1 makes the length **worse** (140 → 128) because it also removes `nop`s
|
||||
the original needs — worker A's "second victim" warning, confirmed.
|
||||
- **R2**'s extended predicate did not fire on the tested row.
|
||||
|
||||
The mechanism and the remaining (developer-owned) route — a tracked post-pass modelling ASPSX's fill
|
||||
for that one site — are recorded as cookbook finding 40.
|
||||
|
||||
## Open protocol notes
|
||||
|
||||
|
||||
@@ -192,3 +192,51 @@ register numbering.
|
||||
| Cycle | New bodies | Total bodies | New regions | Total regions | Result |
|
||||
|---|---|---|---|---|---|
|
||||
| (baseline) | — | 400 | — | 409 | green, head `25bf4a3` |
|
||||
|
||||
---
|
||||
|
||||
## Cycle 1 — merges 1-7 (2026-09-24)
|
||||
|
||||
**432 distinct bodies / 441 regions**, from 400 / 409. All merges followed the hardened flow:
|
||||
`sf3_merge apply` → candidate → **whole-binary gate** → promote → `make check`. The candidate gate was
|
||||
run before every promotion and the tracked registry was never touched by a failing batch.
|
||||
|
||||
| Merge | Worker | Added | Result |
|
||||
|---|---|---|---|
|
||||
| 1 | A(2) + B(5) | 7 regions, 2 symbols | gate MATCH |
|
||||
| 2 | C(8) | 8 regions | gate MATCH |
|
||||
| 3 | B(6-8) | 3 regions (5 skipped as registered) | gate MATCH |
|
||||
| 4 | C(9-12) + A(3) | 5 regions | gate MATCH |
|
||||
| 5 | B(9-11) | 3 regions, 1 symbol | gate MATCH |
|
||||
| 6 | A(4-7) | 4 regions | gate MATCH |
|
||||
| 7 | C(13-14) | 2 regions | gate MATCH |
|
||||
|
||||
**Registry-row requests granted** (each byte-verified by the coordinator with a failing control):
|
||||
`cc1=-G8` on `0x800A6BEC`; `gp=-D_80121B88` on `0x80015D50`; symbol rows `D_80122700`, `D_80122704`,
|
||||
`D_80121AD4` (all `gp`).
|
||||
|
||||
**Negatives reconciled:** `0x8010AA28` imported (maspsx mutual exclusion, finding 17's purest form).
|
||||
The tracked index was also **sorted by address** and its 6 stale registered rows were dropped at
|
||||
P10-T1, so it now carries a checkable ordering invariant: 140 rows, address-ordered, 0 duplicates,
|
||||
0 registered.
|
||||
|
||||
**Worklist/partition discipline.** The tracked worklist regenerates every merge
|
||||
(`excluded_already_registered` always equals the registry size). Partitions were **filtered against the
|
||||
new worklist** rather than re-interleaved mid-cycle, so an address cannot migrate between workers while
|
||||
they are working down it; the union/disjointness proof was re-run after each filter. A full re-interleave
|
||||
is reserved for the cycle boundary.
|
||||
|
||||
**Full clean audit (due at the 3rd merge, run at merge 7):**
|
||||
`make clean && make all` exit 0; `cmp` exit 0; both SHA-1
|
||||
`e173426c157384ebf1b6caf8c6fea18a85a14af9`; registry 441 rows / 0 overlaps / 0 unsorted / 0 bad extents /
|
||||
0 missing sources / 432 distinct sources; 0 tracked paths under any prohibited root; 508 tracked files.
|
||||
|
||||
**Deviations and decisions.** The developer approved the three-worker roster (plan said two); the
|
||||
per-region `-G` override route (global `-G0` unchanged, so the 400-body corpus is untouched); and both
|
||||
R1/R2 maspsx modes. R1/R2 are recorded as a **measured negative result** — implemented, default-off,
|
||||
`make check` green at 441, but neither closes a region (cookbook finding 40, `CURRENT_PHASE.md`).
|
||||
|
||||
**Coordinator-side dispatch aid (ignored, `.run/p10/`).** The family-set map, built up front instead of
|
||||
reactively: 378 worklist rows call an already-registered function (the P1 pool). Two bugs were found and
|
||||
fixed before dispatch (little-endian payload decode; call-site double-count). Per-partition lever files
|
||||
rank each worker's rows P1 (known callee) → P2 (intra-worklist family) → P3 (small tier-1 leaf) → P4.
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
/* func_80012F24 — 0x80012F24..0x80012F80 (92 bytes).
|
||||
*
|
||||
* Original words (objdump of the validated payload, little-endian):
|
||||
* addiu sp,sp,-40
|
||||
* addiu a0,sp,24 a0 = &e (sp+24)
|
||||
* sw ra,32(sp)
|
||||
* jal 0x80021F88
|
||||
* _addiu a1,sp,26 (delay slot) a1 = &f (sp+26)
|
||||
* addiu a0,sp,16 a0 = &s (sp+16)
|
||||
* move a1,zero
|
||||
* move a2,zero
|
||||
* lhu v1,24(sp) v1 = e (UNSIGNED halfword)
|
||||
* lhu v0,26(sp) v0 = f (UNSIGNED halfword)
|
||||
* move a3,zero
|
||||
* sh zero,18(sp) s.b = 0 (b BEFORE a)
|
||||
* sh zero,16(sp) s.a = 0
|
||||
* sll v0,v0,0x1 f * 2
|
||||
* sh v1,20(sp) s.c = e
|
||||
* jal 0x800F421C
|
||||
* _sh v0,22(sp) (delay slot) s.d = f * 2
|
||||
* jal 0x800F4098
|
||||
* _move a0,zero (delay slot)
|
||||
* lw ra,32(sp)
|
||||
* addiu sp,sp,40
|
||||
* jr ra
|
||||
* _nop
|
||||
*
|
||||
* Builds a four-halfword record from two values filled in by an out-parameter
|
||||
* callee and hands it to a four-argument callee, then runs a final call with a
|
||||
* zero argument. The frame is 40 bytes = 16 (o32 outgoing args) + 12 of locals
|
||||
* (the 8-byte record at sp+16 plus the two halfwords at sp+24/26) + the saved ra
|
||||
* at sp+32. `func_80021F88` writes through its two `short *` out parameters, and
|
||||
* the record is passed by address as argument 0 of `func_800F421C`.
|
||||
*
|
||||
* TWO LEVERS this body turns on, both found by dumping the word streams:
|
||||
*
|
||||
* 1. `lhu`, not `lh`: the two out-parameters are read as UNSIGNED halfwords, so
|
||||
* `e` and `f` are `unsigned short`. Declared `short` the same source emits
|
||||
* `lh v0,26(sp)` — 1 differing byte at 0x80012F48 — and (knock-on) also
|
||||
* swaps the two `sh zero` stores. The callee's own registered prototype is
|
||||
* `short *`, but only the caller-side declaration is visible here, so the
|
||||
* out-parameters are declared `unsigned short *` to match the `lhu`.
|
||||
*
|
||||
* 2. CHAINED ASSIGNMENT REVERSES STORE ORDER: the two zeroing stores are
|
||||
* `sh zero,18(sp)` THEN `sh zero,16(sp)` — field b before field a. Written
|
||||
* as two statements in field order (`s.a = 0; s.b = 0;`) the stores come out
|
||||
* 16 then 18 and differ by 2 bytes at 0x80012F50. Written `s.a = s.b = 0;`
|
||||
* the inner assignment is evaluated first, so b is stored before a and the
|
||||
* body is byte-identical. Same family as the commutative-operand lever: the
|
||||
* bytes encode an ORDER that only the source spelling controls.
|
||||
*
|
||||
* LIMITS (unresolved, recorded rather than guessed): whether the two out-values
|
||||
* are separate locals copied into the record (as written here: 8-byte record at
|
||||
* sp+16, e at sp+24, f at sp+26) or the record's own fifth and sixth halfword
|
||||
* fields (a 12-byte record) is NOT observable — both layouts put e/f at sp+24/26
|
||||
* and both compile byte-identically, so the frame is 40 either way. The
|
||||
* separate-local reading is shipped as the more natural one. The purpose of the
|
||||
* record, the meaning of `f * 2`, the two callees and the zero argument are not
|
||||
* observable. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
struct S { short a, b, c, d; };
|
||||
|
||||
extern int func_80021F88(unsigned short *first, unsigned short *second);
|
||||
extern void func_800F421C(struct S *s, int a1, int a2, int a3);
|
||||
extern void func_800F4098(int a0);
|
||||
|
||||
void func_80012F24(void)
|
||||
{
|
||||
struct S s;
|
||||
unsigned short e, f;
|
||||
|
||||
func_80021F88(&e, &f);
|
||||
s.a = s.b = 0;
|
||||
s.c = e;
|
||||
s.d = f * 2;
|
||||
func_800F421C(&s, 0, 0, 0);
|
||||
func_800F4098(0);
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
/*
|
||||
* func_8001587C — 144 bytes at 0x8001587C..0x8001590C
|
||||
*
|
||||
* Byte-identical reconstruction of a framed range-checked converter: an index
|
||||
* at or above 2048 is rejected with 1, otherwise two stack temporaries are
|
||||
* filled, a halved index is looked up, and on success the halved temporary is
|
||||
* converted and stored through the caller's pointer.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* addiu sp,sp,-40
|
||||
* sw s0,24(sp)
|
||||
* move s0,a0 s0 = index
|
||||
* sw s1,28(sp)
|
||||
* move s1,a1 s1 = out
|
||||
* slti v0,s0,2048
|
||||
* bnez v0,0x800158A4
|
||||
* sw ra,32(sp) (delay slot)
|
||||
* j 0x800158F4
|
||||
* li v0,1 (delay slot) return 1
|
||||
* addiu a0,sp,16
|
||||
* jal 0x80021F88
|
||||
* addiu a1,sp,18 (delay slot)
|
||||
* srl a0,s0,0x1f
|
||||
* addu a0,s0,a0
|
||||
* sra a0,a0,0x1 index / 2
|
||||
* jal 0x80023080
|
||||
* addiu a1,sp,20 (delay slot)
|
||||
* bnez v0,0x800158F4 if (r != 0) return r
|
||||
* nop
|
||||
* lhu v0,16(sp) the first temporary
|
||||
* lw a1,20(sp) the looked-up word
|
||||
* sll v0,v0,0x10
|
||||
* sra a0,v0,0x10 (short)tmp
|
||||
* srl v0,v0,0x1f
|
||||
* addu a0,a0,v0
|
||||
* jal 0x80010698
|
||||
* sra a0,a0,0x1 (delay slot) (short)tmp / 2
|
||||
* sw v0,0(s1) *out = converted
|
||||
* move v0,zero return 0
|
||||
* lw ra,32(sp)
|
||||
* lw s1,28(sp)
|
||||
* lw s0,24(sp)
|
||||
* addiu sp,sp,40
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* The frame is 40 bytes: the 16-byte o32 outgoing argument area, three locals
|
||||
* at 16/18/20 (two 16-bit temporaries and a word), `s0` at 24(sp), `s1` at
|
||||
* 28(sp) and `ra` at 32(sp). `lhu` followed by `sll`/`sra` is why the first
|
||||
* temporary is declared `unsigned short` and then explicitly cast to `short`
|
||||
* before the halving: a plain `short` would have loaded with `lh` and needed
|
||||
* no extension. Both halvings are the signed `/ 2` idiom (`srl`+`addu`+`sra`),
|
||||
* not a shift.
|
||||
*
|
||||
* LIMITS: the function names, the callees' arities and parameter types, the
|
||||
* limit 2048 and the locals' meanings are hypotheses reconstructed from the
|
||||
* disassembly. Only the compiled bytes are evidence. The `if (r != 0) return
|
||||
* r;` shape follows from the `bnez` reaching the shared epilogue without
|
||||
* materialising a constant.
|
||||
*/
|
||||
|
||||
void func_80021F88(short *, short *);
|
||||
int func_80023080(int, int *);
|
||||
int func_80010698(int, int);
|
||||
|
||||
int func_8001587C(int index, int *out) {
|
||||
unsigned short tmp;
|
||||
unsigned short unused;
|
||||
int word;
|
||||
int r;
|
||||
|
||||
if (index >= 2048)
|
||||
return 1;
|
||||
|
||||
func_80021F88((short *)&tmp, (short *)&unused);
|
||||
r = func_80023080(index / 2, &word);
|
||||
if (r != 0)
|
||||
return r;
|
||||
|
||||
*out = func_80010698((short)tmp / 2, word);
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
/*
|
||||
* func_80015D50 — 108 bytes at 0x80015D50..0x80015DBC
|
||||
*
|
||||
* Accumulator flush: adds a pending delta into three counters, clears the delta,
|
||||
* then re-arms only when a mode global equals 4.
|
||||
*
|
||||
* Original words:
|
||||
* 8F860074 lw a2,116(gp) ; a2 = D_801219AC (hoisted above the frame)
|
||||
* 27BDFFE8 addiu sp,sp,-24
|
||||
* 10C00014 beq a2,zero,0x80015DAC ; nothing pending -> return
|
||||
* AFBF0010 sw ra,16(sp) ; (delay)
|
||||
* 8F85006C lw a1,108(gp) ; a1 = D_801219A4
|
||||
* 8F820070 lw v0,112(gp) ; v0 = D_801219A8
|
||||
* 8F8308B4 lw v1,2228(gp) ; v1 = D_801221EC (a DIFFERENT address)
|
||||
* AF800074 sw zero,116(gp) ; D_801219AC = 0
|
||||
* 00A62821 addu a1,a1,a2 ; a1 += delta
|
||||
* 00461021 addu v0,v0,a2 ; v0 += delta
|
||||
* 00651821 addu v1,v1,a1 ; v1 += the NEW D_801219A4
|
||||
* AF85006C sw a1,108(gp) ; D_801219A4 = a1
|
||||
* AF820070 sw v0,112(gp) ; D_801219A8 = v0
|
||||
* AF8308B8 sw v1,2232(gp) ; D_801221F0 = v1
|
||||
* 0C03D026 jal 0x800F4098
|
||||
* 00002021 addu a0,zero,zero ; (delay) func_800F4098(0)
|
||||
* 3C038012 lui v1,0x8012
|
||||
* 8C631B88 lw v1,7048(v1) ; v1 = D_80121B88 (ABSOLUTE, not gp-relative)
|
||||
* 24020004 addiu v0,zero,4
|
||||
* 14620003 bne v1,v0,0x80015DAC ; mode != 4 -> return
|
||||
* 00000000 nop
|
||||
* 0C00B1BB jal 0x8002C6EC
|
||||
* 24040006 addiu a0,zero,6 ; (delay) func_8002C6EC(6)
|
||||
* 8FBF0010 lw ra,16(sp)
|
||||
* 27BD0018 addiu sp,sp,24
|
||||
* 03E00008 jr ra
|
||||
* 00000000 nop
|
||||
*
|
||||
* The read at `2228(gp)` and the write at `2232(gp)` are 4 bytes apart and are
|
||||
* DIFFERENT globals: D_801221F0 is assigned `D_801221EC + <the new D_801219A4>`,
|
||||
* not an increment of one address. The three loads are hoisted above the
|
||||
* `sw zero,116(gp)` and above the additions by the scheduler.
|
||||
*
|
||||
* PER-SITE gp OVERRIDE: D_80121B88 carries a registry `gp` marker but the original
|
||||
* reads it ABSOLUTELY (`lui v1,0x8012` / `lw v1,7048(v1)`), so this region needs
|
||||
* `gp=-D_80121B88` — the same per-site form difference as cookbook finding 16.
|
||||
* The other five globals are genuinely gp-relative.
|
||||
*
|
||||
* LIMITS: the function name, both callees, all six globals and the meaning of the
|
||||
* constants 0, 4 and 6 are hypotheses read from the instruction shape; only the
|
||||
* bytes are evidence. The delta is a local in a2 and the accumulator in a1; the
|
||||
* third counter's addend is a distinct global load.
|
||||
*/
|
||||
|
||||
extern int D_801219AC;
|
||||
extern int D_801219A4;
|
||||
extern int D_801219A8;
|
||||
extern int D_801221EC;
|
||||
extern int D_801221F0;
|
||||
extern int D_80121B88;
|
||||
extern void func_800F4098(int a0);
|
||||
extern void func_8002C6EC(int a0);
|
||||
|
||||
void func_80015D50(void)
|
||||
{
|
||||
int delta = D_801219AC;
|
||||
|
||||
if (delta == 0)
|
||||
return;
|
||||
|
||||
D_801219AC = 0;
|
||||
D_801219A4 += delta;
|
||||
D_801219A8 += delta;
|
||||
D_801221F0 = D_801221EC + D_801219A4;
|
||||
func_800F4098(0);
|
||||
|
||||
if (D_80121B88 == 4)
|
||||
func_8002C6EC(6);
|
||||
}
|
||||
@@ -0,0 +1,86 @@
|
||||
/* func_80021C64 — 0x80021C64..0x80021CDC (120 bytes).
|
||||
*
|
||||
* Original words (objdump of the validated payload, little-endian):
|
||||
* addiu sp,sp,-32
|
||||
* sw s0,24(sp)
|
||||
* move s0,a0 s0 = the parameter (kept across all four calls)
|
||||
* sw ra,28(sp)
|
||||
* addiu a1,gp,412 a1 = &D_80121AD4 (gp + 412 = 0x80121AD4)
|
||||
* lwl v0,3(a1) \ 8-byte UNALIGNED block copy
|
||||
* lwr v0,0(a1) | (lwl/lwr + swl/swr, not lw/sw)
|
||||
* lwl v1,7(a1) |
|
||||
* lwr v1,4(a1) |
|
||||
* swl v0,19(sp) |
|
||||
* swr v0,16(sp) |
|
||||
* swl v1,23(sp) |
|
||||
* swr v1,20(sp) /
|
||||
* jal 0x800F3E8C
|
||||
* _move a0,zero (delay slot)
|
||||
* addiu a0,sp,16 a0 = &s (sp+16)
|
||||
* move a1,zero
|
||||
* move a2,zero
|
||||
* jal 0x800F421C
|
||||
* _move a3,zero (delay slot)
|
||||
* jal 0x800F4098
|
||||
* _move a0,zero (delay slot)
|
||||
* sll s0,s0,0x10 \ narrow the parameter to 16 bits
|
||||
* jal 0x800F8FE4 |
|
||||
* _sra a0,s0,0x10 (delay slot) a0 = (short)parameter
|
||||
* lw ra,28(sp)
|
||||
* lw s0,24(sp)
|
||||
* addiu sp,sp,32
|
||||
* jr ra
|
||||
* _nop
|
||||
*
|
||||
* Copies an 8-byte object out of the small-data block, hands it to a
|
||||
* four-argument callee, brackets it with two zero-argument calls, and finally
|
||||
* forwards the parameter narrowed to 16 bits. The frame is 32 bytes = 16 (o32
|
||||
* outgoing args) + the 8-byte local at sp+16 + the saved s0/ra pair.
|
||||
*
|
||||
* WHY THE COPY IS `lwl/lwr` AND NOT `lw`: the eight-byte move uses the unaligned
|
||||
* halfword/word forms on BOTH sides — including the destination, which is a
|
||||
* frame slot at sp+16 that is in fact 4-byte aligned. cc1 only does that when the
|
||||
* move's alignment is 1, so both the local and the source global are
|
||||
* byte-aligned objects: `struct S { char c[8]; }` on both sides reproduces it.
|
||||
* The global is addressed by `addiu a1,gp,412`, i.e. its ADDRESS is taken
|
||||
* gp-relative, which is the address-taking form of the registry `gp` marker.
|
||||
*
|
||||
* The final `sll 16` / `sra 16` pair is the caller-side narrowing of an argument
|
||||
* to a 16-bit parameter, so the last callee takes a `short` (or the source casts
|
||||
* to `short`). `func_800F8FE4` is registered taking `int`, and passing a narrowed
|
||||
* value to it is what the bytes show.
|
||||
*
|
||||
* SYMBOL DEPENDENCY: `D_80121AD4` is NOT yet in config/symbols.tsv. It must be
|
||||
* added with the `gp` marker (see .run/p10/w-b/symbols-request.tsv) or the
|
||||
* address is materialised absolutely (`lui`+`addiu`) instead of `addiu a1,gp,412`.
|
||||
* This claim was verified against a local overlay of the tracked registry plus
|
||||
* that row.
|
||||
*
|
||||
* LIMITS (unresolved, recorded rather than guessed): the parameter's declared
|
||||
* type is NOT observable — `int a0` with `func_800F8FE4((short)a0)` and
|
||||
* `short a0` with `func_800F8FE4(a0)` both compile byte-identically, because the
|
||||
* sign-extension happens lazily at the call site rather than on entry; the
|
||||
* `short` form is shipped as the cleaner reading. The object's real type (char
|
||||
* array vs a packed structure) is not observable beyond its 8-byte size and
|
||||
* alignment 1, and the three callees' names and semantics are hypotheses. Only
|
||||
* the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
struct S { char c[8]; };
|
||||
|
||||
extern struct S D_80121AD4;
|
||||
extern void func_800F3E8C(int a0);
|
||||
extern void func_800F421C(struct S *s, int a1, int a2, int a3);
|
||||
extern void func_800F4098(int a0);
|
||||
extern int func_800F8FE4(int value);
|
||||
|
||||
void func_80021C64(short a0)
|
||||
{
|
||||
struct S s;
|
||||
|
||||
s = D_80121AD4;
|
||||
func_800F3E8C(0);
|
||||
func_800F421C(&s, 0, 0, 0);
|
||||
func_800F4098(0);
|
||||
func_800F8FE4(a0);
|
||||
}
|
||||
@@ -0,0 +1,78 @@
|
||||
/*
|
||||
* func_80023080 — 100 bytes at 0x80023080..0x800230E4
|
||||
*
|
||||
* Two-step lookup whose second result decides between storing a value and
|
||||
* returning 0, or storing zero and returning 7.
|
||||
*
|
||||
* Original words:
|
||||
* 27BDFFE0 addiu sp,sp,-32
|
||||
* AFB00010 sw s0,16(sp)
|
||||
* 00808021 addu s0,a0,zero ; s0 = a0
|
||||
* AFB10014 sw s1,20(sp)
|
||||
* AFBF0018 sw ra,24(sp)
|
||||
* 0C03E4D1 jal 0x800F9344
|
||||
* 00A08821 addu s1,a1,zero ; (delay) s1 = a1
|
||||
* 02002021 addu a0,s0,zero ; a0 = the ORIGINAL a0
|
||||
* 0C03E4A9 jal 0x800F92A4
|
||||
* 00408021 addu s0,v0,zero ; (delay) s0 = first result
|
||||
* 10400006 beq v0,zero,0x800230C4 ; second result zero -> the 7 path
|
||||
* 02002021 addu a0,s0,zero ; (delay) a0 = first result
|
||||
* 0C0041A6 jal 0x80010698
|
||||
* 00402821 addu a1,v0,zero ; (delay) a1 = second result
|
||||
* AE220000 sw v0,0(s1) ; *a1 = third result
|
||||
* 08008C33 j 0x800230CC
|
||||
* 00001021 addu v0,zero,zero ; (delay) return 0
|
||||
* AE200000 sw zero,0(s1) ; *a1 = 0
|
||||
* 24020007 addiu v0,zero,7 ; return 7
|
||||
* 8FBF0018 lw ra,24(sp)
|
||||
* 8FB10014 lw s1,20(sp)
|
||||
* 8FB00010 lw s0,16(sp)
|
||||
* 27BD0020 addiu sp,sp,32
|
||||
* 03E00008 jr ra
|
||||
* 00000000 nop
|
||||
*
|
||||
* `s0` carries two different values with disjoint live ranges: the incoming `a0`
|
||||
* across the first call, then that call's result across the second — which is why
|
||||
* the second call is made with `a0,s0` and its delay slot overwrites `s0` with
|
||||
* `v0`. `s1` holds the pointer argument across all three calls.
|
||||
*
|
||||
* The two arms are `*a1 = <third result>; r = 0;` and `*a1 = 0; r = 7;` followed by a
|
||||
* single `return r;`, with the `v0 == 0` test jumping to the second arm, so the
|
||||
* non-zero arm is the fall-through. The shared epilogue is the normal order
|
||||
* (`lw ra` / release / `jr ra` / `nop`).
|
||||
*
|
||||
* THE SHARED-RESULT LEVER (measured, both spellings compiled against this range).
|
||||
* Writing the two arms as `return 0;` / `return 7;` makes cc1 REVERSE the layout:
|
||||
* it emits `bne v0,zero,<call arm>` and puts the `r = 7` arm in the fall-through
|
||||
* position, 23 differing bytes with every instruction present. Writing the arms as
|
||||
* assignments to a shared local with one `return r;` after the `if`/`else`
|
||||
* produces the original exactly. So when both arms of an `if`/`else` end in a
|
||||
* `return`, cc1's jump optimisation is free to reverse the arms; a single shared
|
||||
* result local pins the original order. `v1.c` (both arms return) and `v4.c`
|
||||
* (shared local) were both compiled; only v4 matches.
|
||||
*
|
||||
* LIMITS: the function name, the three callees, the pointer argument and the
|
||||
* meaning of the constants 0 and 7 are hypotheses read from the instruction
|
||||
* shape; only the bytes are evidence. func_800F92A4 and func_80010698 are not
|
||||
* registered and are referenced by their address-named spellings.
|
||||
*/
|
||||
|
||||
extern int func_800F9344(int a0);
|
||||
extern int func_800F92A4(int a0);
|
||||
extern int func_80010698(int a0, int a1);
|
||||
|
||||
int func_80023080(int a0, int *a1)
|
||||
{
|
||||
int first = func_800F9344(a0);
|
||||
int second = func_800F92A4(a0);
|
||||
int r;
|
||||
|
||||
if (second != 0) {
|
||||
*a1 = func_80010698(first, second);
|
||||
r = 0;
|
||||
} else {
|
||||
*a1 = 0;
|
||||
r = 7;
|
||||
}
|
||||
return r;
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
/*
|
||||
* func_80027E1C — 176 bytes at 0x80027E1C..0x80027ECC
|
||||
*
|
||||
* Byte-identical reconstruction of a framed three-component transform: it
|
||||
* subtracts one vector from another, passes each component through a shared
|
||||
* two-argument helper with the corresponding component of the second vector,
|
||||
* sums the three results, and stores the sum through an out pointer.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* addiu sp,sp,-48
|
||||
* move a3,a0 p
|
||||
* sw s1,36(sp)
|
||||
* move s1,a1 q
|
||||
* sw ra,44(sp)
|
||||
* sw s2,40(sp)
|
||||
* sw s0,32(sp)
|
||||
* lw a0,0(a3)
|
||||
* lw v0,0(s1)
|
||||
* nop (load delay)
|
||||
* subu a0,a0,v0 d[0] = p[0] - q[0]
|
||||
* sw a0,16(sp)
|
||||
* lw v0,4(a3)
|
||||
* lw v1,4(s1)
|
||||
* nop
|
||||
* subu v0,v0,v1 d[1]
|
||||
* sw v0,20(sp)
|
||||
* lw v0,8(a3)
|
||||
* lw v1,8(s1)
|
||||
* nop
|
||||
* subu v0,v0,v1 d[2]
|
||||
* sw v0,24(sp)
|
||||
* lw a1,16(s1) q[4]
|
||||
* jal 0x80010654
|
||||
* move s2,a2 (delay slot) out
|
||||
* lw a0,20(sp) d[1]
|
||||
* lw a1,20(s1) q[5]
|
||||
* jal 0x80010654
|
||||
* move s0,v0 (delay slot) r0
|
||||
* lw a0,24(sp) d[2]
|
||||
* lw a1,24(s1) q[6]
|
||||
* jal 0x80010654
|
||||
* move s1,v0 (delay slot) r1 (q is dead here)
|
||||
* addu s0,s0,s1 r0 + r1
|
||||
* addu s0,s0,v0 + r2
|
||||
* move v0,zero
|
||||
* sw s0,0(s2) *out = sum
|
||||
* lw ra,44(sp)
|
||||
* lw s2,40(sp)
|
||||
* lw s1,36(sp)
|
||||
* lw s0,32(sp)
|
||||
* addiu sp,sp,48
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* The frame is 48 bytes: the 16-byte o32 outgoing argument area, three
|
||||
* spilled component words at 16/20/24, `s0` at 32(sp), `s1` at 36(sp), `s2` at
|
||||
* 40(sp) and `ra` at 44(sp). **The three components must be an ARRAY, not
|
||||
* three scalars.** Written as scalars, cc1 keeps them in registers and the
|
||||
* frame collapses to 40 bytes with four saved registers (160 bytes total);
|
||||
* written as `int d[3]`, cc1 cannot keep an array in registers, so it spills
|
||||
* and reloads exactly as the original does. `move s1,v0` in the third call's
|
||||
* delay slot reuses the dead `q` register for the second result.
|
||||
*
|
||||
* LIMITS: the function name, the helper's arity and semantics, the vector
|
||||
* strides and the out pointer's meaning are hypotheses reconstructed from the
|
||||
* disassembly. Only the compiled bytes are evidence. The helper's two
|
||||
* arguments are `(component, q[i + 4])`; whether the second is a scale, a
|
||||
* basis element or a bias is not observable from this body.
|
||||
*/
|
||||
|
||||
int func_80010654(int, int);
|
||||
|
||||
int func_80027E1C(int *p, int *q, int *out) {
|
||||
int d[3];
|
||||
int r0;
|
||||
int r1;
|
||||
|
||||
d[0] = p[0] - q[0];
|
||||
d[1] = p[1] - q[1];
|
||||
d[2] = p[2] - q[2];
|
||||
|
||||
r0 = func_80010654(d[0], q[4]);
|
||||
r1 = func_80010654(d[1], q[5]);
|
||||
*out = r0 + r1 + func_80010654(d[2], q[6]);
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,61 @@
|
||||
/* func_8002C764 — 0x8002C764..0x8002C7BC (88 bytes).
|
||||
*
|
||||
* Original words (objdump of the validated payload, little-endian):
|
||||
* addiu sp,sp,-24
|
||||
* sw s0,16(sp)
|
||||
* sw ra,20(sp)
|
||||
* jal 0x800A745C
|
||||
* _move s0,a0 (delay slot; parameter kept across the call)
|
||||
* jal 0x8002C6EC
|
||||
* _li a0,4 (delay slot; literal argument 4)
|
||||
* lbu v0,576(gp) v0 = D_80121B78 (gp-relative, 0x80121938+576)
|
||||
* _nop load-delay slot
|
||||
* beqz v0,0x8002C7A0 zero -> else branch
|
||||
* _nop
|
||||
* jal 0x80159FD0
|
||||
* _move a0,s0 (delay slot)
|
||||
* j 0x8002C7A8 skip the else block
|
||||
* _nop
|
||||
* 0x8002C7A0:
|
||||
* jal 0x80156D78
|
||||
* _move a0,s0 (delay slot)
|
||||
* 0x8002C7A8:
|
||||
* lw ra,20(sp)
|
||||
* lw s0,16(sp)
|
||||
* addiu sp,sp,24
|
||||
* jr ra
|
||||
* _nop
|
||||
*
|
||||
* Two setup calls, then a two-way dispatch on a gp-relative byte global, both
|
||||
* arms forwarding the incoming parameter. The parameter is copied to s0 in the
|
||||
* first `jal` delay slot because it is live across both setup calls and is the
|
||||
* argument to whichever arm runs. `func_8002C6EC` is called with the literal 4
|
||||
* materialised in its `jal` delay slot.
|
||||
*
|
||||
* The `lbu` is gp-relative, so `D_80121B78` is already carried in the tracked
|
||||
* symbol registry with its `gp` marker (0x80121938 + 576 = 0x80121B78) — the
|
||||
* harness rewrites the access to `%gp_rel`. The byte is tested with `beqz` on
|
||||
* the zero-extended `lbu` result, so the global is an unsigned char.
|
||||
*
|
||||
* LIMITS: the purpose of the flag, the meaning of the literal 4, and the names
|
||||
* of the two dispatch targets are not observable. `func_800A745C` is registered
|
||||
* as taking no arguments, so it is called with none here; the call site leaves
|
||||
* a0 holding the incoming parameter either way, so a source that passed it would
|
||||
* compile identically. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
extern unsigned char D_80121B78;
|
||||
extern void func_800A745C(void);
|
||||
extern void func_8002C6EC(int a0);
|
||||
extern void func_80159FD0(int a0);
|
||||
extern void func_80156D78(int a0);
|
||||
|
||||
void func_8002C764(int a0)
|
||||
{
|
||||
func_800A745C();
|
||||
func_8002C6EC(4);
|
||||
if (D_80121B78)
|
||||
func_80159FD0(a0);
|
||||
else
|
||||
func_80156D78(a0);
|
||||
}
|
||||
@@ -0,0 +1,95 @@
|
||||
/* func_8002E870 — 0x8002E870..0x8002E8EC (124 bytes).
|
||||
*
|
||||
* Original words (objdump of the validated payload, little-endian):
|
||||
* lui v1,0x8012 \
|
||||
* lw v1,9044(v1) / v1 = g_80122354 (absolute, hoisted above the frame)
|
||||
* lw v0,2444(gp) v0 = D_801222C4 (gp-relative, 0x801222C4)
|
||||
* addiu sp,sp,-32
|
||||
* sw s1,20(sp)
|
||||
* move s1,a0 s1 = the parameter
|
||||
* sw ra,24(sp)
|
||||
* sltu v0,v0,v1 v0 = (D_801222C4 < g_80122354) UNSIGNED
|
||||
* beqz v0,0x8002E8D4 not below -> epilogue
|
||||
* _sw s0,16(sp) (delay slot)
|
||||
* jal 0x800FB5B4
|
||||
* _nop
|
||||
* sll s0,v0,0x1 size = returned * 2 ...
|
||||
* lui a0,0x8012 \
|
||||
* lw a0,9044(a0) / a0 = g_80122354 (RELOADED after the call)
|
||||
* addiu s0,s0,24 ... + 24
|
||||
* addu a0,a0,s0 a0 = g_80122354 + size
|
||||
* addiu a0,a0,16 a0 += 16
|
||||
* sw a0,2444(gp) D_801222C4 = a0
|
||||
* jal 0x800ACC00 call with that same value
|
||||
* _nop
|
||||
* move a0,zero
|
||||
* move a1,s1
|
||||
* jal 0x80063870
|
||||
* _move a2,s0 (delay slot) a2 = size
|
||||
* lw ra,24(sp) (epilogue)
|
||||
* lw s1,20(sp)
|
||||
* lw s0,16(sp)
|
||||
* addiu sp,sp,32
|
||||
* jr ra
|
||||
* _nop
|
||||
*
|
||||
* A guarded allocator step: while the cursor `D_801222C4` is still below the
|
||||
* limit `g_80122354`, ask for a count, compute `count * 2 + 24`, advance the
|
||||
* cursor past that many bytes plus 16, and hand both the new cursor value and
|
||||
* the size to two callees. The frame is 32 bytes = 16 (o32 outgoing args) + the
|
||||
* s0/s1 pair at sp+16/20 + ra at sp+24. `s1` holds the parameter across the
|
||||
* first call; `s0` holds the computed size across the last one.
|
||||
*
|
||||
* THREE THINGS THE BYTES PIN DOWN:
|
||||
*
|
||||
* 1. THE COMPARISON IS WRITTEN MIRRORED — `g_80122354 > D_801222C4`, not
|
||||
* `D_801222C4 < g_80122354`. Both give the same `sltu v0,v0,v1` (cc1
|
||||
* canonicalises `>` by swapping the operands), but the SOURCE order decides
|
||||
* which global is loaded first: the mirrored spelling loads `g_80122354` into
|
||||
* v1 first and then `D_801222C4` into v0, exactly as the original does. The
|
||||
* natural spelling swaps those two instructions and leaves 18 differing bytes
|
||||
* from 0x8002E870. This is the same operand-order family as claims 1 and 2,
|
||||
* now on a comparison's operand evaluation order.
|
||||
*
|
||||
* 2. THE COMPUTED VALUE IS ALSO AN ARGUMENT. The sum is built in **a0** — the
|
||||
* first argument register — not in v0: `lui a0` / `lw a0` / `addu a0,a0,s0` /
|
||||
* `addiu a0,a0,16` / `sw a0,2444(gp)` / `jal 0x800ACC00`. That only happens
|
||||
* when the same value is passed to the next call, so the source names it and
|
||||
* passes it: `func_800ACC00(v)`. This is CONFIRMED by the registry —
|
||||
* `src/func_800ACC00.c` is `void func_800ACC00(unsigned int value)`, i.e. the
|
||||
* callee does take that argument. Building the sum into a temporary and
|
||||
* storing only (no call argument) leaves the value in v0 and 6 differing
|
||||
* bytes at 0x8002E8A4.
|
||||
*
|
||||
* 3. THE COMPARISON IS UNSIGNED: `sltu`, not `slt`. The cursor is
|
||||
* `unsigned int` while the limit is `int`, so the usual arithmetic conversions
|
||||
* make the comparison unsigned.
|
||||
*
|
||||
* The limit global is loaded TWICE (once for the guard, once for the update) and
|
||||
* is deliberately NOT kept in a register across the middle call — the source
|
||||
* re-reads it. `g_80122354` is read absolutely (`lui`+`lw`) because its registry
|
||||
* row has no `gp` marker; `D_801222C4` is gp-relative (+2444 = 0x98C) and is
|
||||
* already carried with its `gp` marker.
|
||||
*
|
||||
* LIMITS: the purpose of the +16 (an object header or alignment pad), the `* 2`
|
||||
* scaling, the meaning of the limit and cursor, and the three callees' semantics
|
||||
* are not observable. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
extern int g_80122354;
|
||||
extern unsigned int D_801222C4;
|
||||
extern int func_800FB5B4(void);
|
||||
extern void func_800ACC00(unsigned int value);
|
||||
extern void func_80063870(int a0, int a1, int a2);
|
||||
|
||||
void func_8002E870(int a0)
|
||||
{
|
||||
if (g_80122354 > D_801222C4) {
|
||||
int size = func_800FB5B4() * 2 + 24;
|
||||
unsigned int v = g_80122354 + size + 16;
|
||||
|
||||
D_801222C4 = v;
|
||||
func_800ACC00(v);
|
||||
func_80063870(0, a0, size);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
/*
|
||||
* func_8002F0D0 — 72 bytes at 0x8002F0D0..0x8002F118
|
||||
*
|
||||
* Byte-identical reconstruction of a framed routine that calls a lookup with
|
||||
* three arguments, conditionally calls a second routine with the lookup's
|
||||
* result, and finally clears a gp-relative byte.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* addiu sp,sp,-24
|
||||
* move a0,zero
|
||||
* lui a1,0x8014
|
||||
* lw a1,-15564(a1) a1 = *(int *)0x8013C334
|
||||
* sw ra,16(sp)
|
||||
* jal 0x80063870
|
||||
* move a2,zero (delay slot)
|
||||
* move a0,v0 r = lookup(...)
|
||||
* li v0,-1
|
||||
* beq a0,v0,0x8002F104
|
||||
* nop
|
||||
* jal 0x800A9F7C
|
||||
* li a1,1 (delay slot)
|
||||
* sb zero,2524(gp) D_80122314 = 0
|
||||
* lw ra,16(sp)
|
||||
* addiu sp,sp,24
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* The frame is 24 bytes: the 16-byte o32 outgoing argument area plus `ra` at
|
||||
* 16(sp). The global read is a **literal** constant address, which folds into
|
||||
* the load's displacement (cookbook finding 18) — no symbol registry row is
|
||||
* needed and none would change the encoding. The final store is gp-relative
|
||||
* (gp = 0x80121938, so 2524(gp) is 0x80122314) and needs the registry's `gp`
|
||||
* marker on that symbol's name (finding 30).
|
||||
*
|
||||
* LIMITS: the function names, the global's type, and the argument meanings are
|
||||
* hypotheses reconstructed from the disassembly. Only the compiled bytes are
|
||||
* evidence. The -1 sentinel is written as a comparison against -1 because that
|
||||
* is the only value the original materialises into `v0`.
|
||||
*/
|
||||
|
||||
extern char D_80122314;
|
||||
|
||||
int func_80063870(int, int, int);
|
||||
void func_800A9F7C(int, int);
|
||||
|
||||
void func_8002F0D0(void) {
|
||||
int r = func_80063870(0, *(int *)0x8013C334, 0);
|
||||
|
||||
if (r != -1)
|
||||
func_800A9F7C(r, 1);
|
||||
|
||||
D_80122314 = 0;
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
/*
|
||||
* func_800419D0 — 84 bytes at 0x800419D0..0x80041A24
|
||||
*
|
||||
* Byte-identical reconstruction of a framed accessor: it accepts a pointer
|
||||
* that is either of two registered globals, in which case it walks two fields
|
||||
* of the pointed-to structure, and otherwise delegates to a helper.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lui v0,0x8013
|
||||
* lw v0,-10340(v0) v0 = *(int *)0x8012D79C
|
||||
* addiu sp,sp,-24
|
||||
* beq a0,v0,0x800419F8 if (p == g1) goto body
|
||||
* sw ra,16(sp) (delay slot)
|
||||
* lui v0,0x8013
|
||||
* lw v0,-8912(v0) v0 = *(int *)0x8012DD30
|
||||
* nop (load-delay, finding 27)
|
||||
* bne a0,v0,0x80041A0C if (p != g2) goto call
|
||||
* nop
|
||||
* lw v0,32(a0) body: v0 = *(int *)(p + 32)
|
||||
* nop (load-delay)
|
||||
* lw v0,24(v0) v0 = *(int *)(v0 + 24)
|
||||
* j 0x80041A14 goto return
|
||||
* nop
|
||||
* jal 0x8007EB8C call: v0 = helper(p)
|
||||
* nop
|
||||
* lw ra,16(sp) return:
|
||||
* addiu sp,sp,24
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* The two `if` arms share one return, so the original lays the body block out
|
||||
* first, jumps over the call block, and lets the call block fall through into
|
||||
* the epilogue — the shape `if (a || b) { ... } return helper(p);` produces.
|
||||
* Both global reads are **literal** constant addresses folding into the load
|
||||
* displacement (cookbook finding 18); the first one is scheduled above the
|
||||
* prologue by cc1's reorganisation pass, which is why the `addiu sp,sp,-24`
|
||||
* appears between the two loads.
|
||||
*
|
||||
* LIMITS: the function name, the helper's name, the global names and the
|
||||
* structure field offsets are hypotheses reconstructed from the disassembly.
|
||||
* Only the compiled bytes are evidence. Whether the two globals are pointers
|
||||
* or ints is not observable here — only that each is loaded whole and compared
|
||||
* against the incoming pointer.
|
||||
*/
|
||||
|
||||
int func_8007EB8C(char *);
|
||||
|
||||
int func_800419D0(char *p) {
|
||||
if (p == (char *)*(int *)0x8012D79C || p == (char *)*(int *)0x8012DD30)
|
||||
return *(int *)(*(int *)(p + 32) + 24);
|
||||
|
||||
return func_8007EB8C(p);
|
||||
}
|
||||
@@ -0,0 +1,51 @@
|
||||
/*
|
||||
* func_800429B0 — 64 bytes at 0x800429B0..0x800429F0
|
||||
*
|
||||
* Framed routine that calls the same two setters as its sibling func_80042964
|
||||
* (0x80042964..0x800429B0), but passes the raw signed halfwords with no scaling.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lui v0,0x8012 3c028012 \
|
||||
* lw v0,0x2430(v0) 8c422430 / v0 = *(int *)0x80122430 (D_80122430)
|
||||
* addiu sp,sp,-0x18 27bdffe8 frame, 24 bytes
|
||||
* sw ra,0x10(sp) afbf0010 save ra
|
||||
* lh a0,0x4e(v0) 8444004e a0 = *(short *)(v0 + 0x4e)
|
||||
* lh a1,0x50(v0) 84450050 a1 = *(short *)(v0 + 0x50)
|
||||
* jal 0x80017c50 0c005f14 call func_80017C50
|
||||
* nop 00000000 (delay slot)
|
||||
* lui a0,0x8012 3c048012 \
|
||||
* lh a0,0x242c(a0) 8484242c / a0 = *(short *)0x8012242C (D_8012242C)
|
||||
* jal 0x80017c60 0c005f18 call func_80017C60
|
||||
* nop 00000000 (delay slot)
|
||||
* lw ra,0x10(sp) 8fbf0010 restore ra
|
||||
* addiu sp,sp,0x18 27bd0018 frame release
|
||||
* jr ra 03e00008
|
||||
* nop 00000000 (delay slot)
|
||||
*
|
||||
* The sibling func_80042964 shares this prologue and the same two global loads
|
||||
* byte-for-byte; the only difference is that the sibling scales its three
|
||||
* operands (`sll` by 1, plus the `sll 17`/`sra 16` narrowing of the `short`
|
||||
* argument) and this body does not. The pointer load is hoisted above the frame
|
||||
* setup here too, and both `jal`s carry an unfilled `nop` delay slot.
|
||||
*
|
||||
* Both callees take 16-bit parameters and all three values come from signed
|
||||
* halfword loads, so no narrowing instruction is required at either call site.
|
||||
*
|
||||
* LIMITS: the function name, both callees, the two globals and the meaning of
|
||||
* the stores are hypotheses; only the bytes are evidence. `int v0` is a local
|
||||
* used only to keep the pointer global loaded once — the disassembly fixes the
|
||||
* load count, not the declaration.
|
||||
*/
|
||||
|
||||
extern int D_80122430;
|
||||
extern short D_8012242C;
|
||||
extern void func_80017C50(int a0, int a1);
|
||||
extern void func_80017C60(short a0);
|
||||
|
||||
void func_800429B0(void)
|
||||
{
|
||||
int v0 = D_80122430;
|
||||
|
||||
func_80017C50(*(short *)(v0 + 0x4e), *(short *)(v0 + 0x50));
|
||||
func_80017C60(D_8012242C);
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
/*
|
||||
* func_80047984 — 144 bytes at 0x80047984..0x80047A14
|
||||
*
|
||||
* Byte-identical reconstruction of a framed two-arm dispatch on a mode
|
||||
* argument: each arm walks a 76-byte record table with the object's index,
|
||||
* reads the pointer at record offset 36, dereferences it, and calls one
|
||||
* routine with a different pair of constants.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* addiu sp,sp,-24
|
||||
* li v0,1
|
||||
* bne a1,v0,0x800479C8 if (mode != 1) goto arm 2
|
||||
* sw ra,16(sp) (delay slot)
|
||||
* lh v1,2(a0) idx = *(short *)(p + 2)
|
||||
* li a1,220
|
||||
* sll v0,v1,0x2
|
||||
* addu v0,v0,v1
|
||||
* sll v0,v0,0x2
|
||||
* subu v0,v0,v1 idx * 19
|
||||
* lui v1,0x8012
|
||||
* lw v1,7164(v1) *(int *)0x80121BFC
|
||||
* sll v0,v0,0x2 idx * 76
|
||||
* addu v0,v0,v1 (char *)(idx * 76) + base
|
||||
* lw v0,36(v0) the record's pointer
|
||||
* j 0x800479F8
|
||||
* li a2,13 (delay slot)
|
||||
* ... the whole sequence again with 200 / 14 ...
|
||||
* lw a0,0(v0) *record
|
||||
* jal 0x80017AE8
|
||||
* li a3,3 (delay slot)
|
||||
* lw ra,16(sp)
|
||||
* addiu sp,sp,24
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* The frame is 24 bytes: the 16-byte o32 outgoing argument area plus `ra` at
|
||||
* 16(sp). Three source-shape facts are load-bearing. (1) The index scaling
|
||||
* `((idx*4 + idx)*4 - idx)*4` is cc1's strength reduction of `idx * 76`.
|
||||
* (2) `addu v0,v0,v1` is the stride-first spelling of cookbook finding 22 —
|
||||
* `(char *)(idx * 76) + base` — not the base-first one. (3) The address
|
||||
* computation must be written **inline in both arms**: binding it to a local
|
||||
* before the `if` lets cc1's CSE hoist it (108 bytes) and hoisting the two
|
||||
* loads into locals shrinks the body differently (124 bytes). Written inline,
|
||||
* the arms stay duplicated and cc1's **cross-jumping** merges only their
|
||||
* identical tails — the `lw a0,0(v0)` / `jal` / `li a3,3` sequence — which is
|
||||
* exactly the shared 0x800479F8 block. 0x80121BFC is read as a **literal**
|
||||
* constant address because its registry symbol carries a `gp` marker and the
|
||||
* original uses the absolute form (cookbook finding 30).
|
||||
*
|
||||
* LIMITS: the function name, the callee's arities and parameter types, the
|
||||
* record stride 76, the field offset 36, the mode/constant pairs and the index
|
||||
* field's width are hypotheses reconstructed from the disassembly. Only the
|
||||
* compiled bytes are evidence. Whether the record's offset-36 field is a
|
||||
* pointer or the first word of an embedded struct is not observable; only the
|
||||
* double `lw` is.
|
||||
*/
|
||||
|
||||
void func_80017AE8(int, int, int, int);
|
||||
|
||||
void func_80047984(char *a0, int a1) {
|
||||
if (a1 == 1)
|
||||
func_80017AE8(
|
||||
**(int **)((char *)(*(short *)(a0 + 2) * 76) + *(int *)0x80121BFC + 36),
|
||||
220, 13, 3);
|
||||
else
|
||||
func_80017AE8(
|
||||
**(int **)((char *)(*(short *)(a0 + 2) * 76) + *(int *)0x80121BFC + 36),
|
||||
200, 14, 3);
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
/* func_80057564 — 0x80057564..0x800575C4 (96 bytes).
|
||||
*
|
||||
* Original words (objdump of the validated payload, little-endian):
|
||||
* lui v0,0x8013
|
||||
* lw v0,-10340(v0) v0 = *(int *)0x8012D79C (hoisted above the frame)
|
||||
* addiu sp,sp,-24
|
||||
* sw s0,16(sp)
|
||||
* move s0,a0 s0 = the argument
|
||||
* beq s0,v0,0x80057594 argument == first sentinel -> body
|
||||
* _sw ra,20(sp) (delay slot)
|
||||
* lui v0,0x8013
|
||||
* lw v0,-8912(v0) v0 = *(int *)0x8012DD30
|
||||
* _nop load-delay slot
|
||||
* bne s0,v0,0x800575B0 argument != second sentinel -> epilogue
|
||||
* _nop
|
||||
* 0x80057594:
|
||||
* jal 0x80057524
|
||||
* _move a0,s0 (delay slot)
|
||||
* lw v0,32(s0) v0 = *(int *)(s0 + 0x20)
|
||||
* _nop load-delay slot
|
||||
* lw a0,244(v0) a0 = *(int *)(v0 + 0xf4)
|
||||
* jal 0x80050CA8
|
||||
* _nop
|
||||
* 0x800575B0:
|
||||
* lw ra,20(sp)
|
||||
* lw s0,16(sp)
|
||||
* addiu sp,sp,24
|
||||
* jr ra
|
||||
* _nop
|
||||
*
|
||||
* A sentinel-guarded two-call body: when the argument equals either of two
|
||||
* global sentinel pointers, forward it to one callee and then call a second
|
||||
* callee with a word reached through two indirections from it. The argument is
|
||||
* kept in s0 across the first call because it is both the second call's base and
|
||||
* the guard's operand. The `||` short-circuit is visible as the layout: the
|
||||
* first comparison's taken edge jumps *forward* to the shared body, the second
|
||||
* comparison's not-taken edge falls into it, and the not-taken edge of the
|
||||
* second jumps over the body to the epilogue.
|
||||
*
|
||||
* The two sentinels are the SAME globals as in the registered
|
||||
* `func_800419D0.c` — `*(int *)0x8012D79C` (0x80130000 - 10340) and
|
||||
* `*(int *)0x8012DD30` (0x80130000 - 8912) — which also compares its argument
|
||||
* against both and then does the same `*(int *)(*(int *)(p + 0x20) + off)`
|
||||
* double indirection (offset 24 there, 0xf4 here). That cross-reference is what
|
||||
* fixes these as literal constant addresses rather than registry symbols.
|
||||
*
|
||||
* LIMITS (unresolved, recorded rather than guessed): the parameter's declared
|
||||
* type is NOT observable — `int p` compared against `*(int *)0x8012D79C` and
|
||||
* `char *p` compared against `(char *)*(int *)0x8012D79C` compile
|
||||
* byte-identically. `int` with hex offsets is shipped to match the house style of
|
||||
* the directly-called `func_80057524.c`, which uses the same double-indirection
|
||||
* spelling. The sentinels' meanings, the field at +0xf4, and the callees' names
|
||||
* are hypotheses read from the instruction shapes. Only the compiled bytes are
|
||||
* evidence.
|
||||
*/
|
||||
|
||||
extern void func_80057524(int a0);
|
||||
extern void func_80050CA8(int a0);
|
||||
|
||||
void func_80057564(int p)
|
||||
{
|
||||
if (p == *(int *)0x8012D79C || p == *(int *)0x8012DD30) {
|
||||
func_80057524(p);
|
||||
func_80050CA8(*(int *)(*(int *)(p + 0x20) + 0xf4));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,17 @@
|
||||
void func_80057F08(int, char *);
|
||||
void func_80058424(int);
|
||||
void func_80057DFC(char *);
|
||||
void func_8005E79C(char *a0) {
|
||||
char *table = a0 + 2604;
|
||||
int i = 0;
|
||||
int offset = 252;
|
||||
|
||||
for (; i < 7; i++) {
|
||||
char *p = table + offset;
|
||||
|
||||
offset += 12;
|
||||
func_80057F08(*(int *)(a0 + 28), p);
|
||||
func_80058424(*(int *)(p + 8));
|
||||
func_80057DFC(p);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
/*
|
||||
* func_80074C00 — 116 bytes at 0x80074C00..0x80074C74
|
||||
*
|
||||
* Byte-identical reconstruction of a framed setter that records four incoming
|
||||
* 16-bit values into a small-data block, calls a four-argument routine, stores
|
||||
* the result, and finally packs two of its stack arguments into a 16-bit word.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* addiu sp,sp,-32
|
||||
* move v0,a0
|
||||
* sw s0,16(sp)
|
||||
* lw s0,48(sp) s0 = arg5
|
||||
* move v1,a1
|
||||
* sw s1,20(sp)
|
||||
* lw s1,52(sp) s1 = arg6
|
||||
* li a0,1
|
||||
* sh a2,3014(gp) D_801224FE = a2
|
||||
* move a2,v0 a2 = original a0
|
||||
* sh v0,3010(gp) D_801224FA = a0
|
||||
* sh v1,3012(gp) D_801224FC = a1
|
||||
* sh a3,3016(gp) D_80122500 = a3
|
||||
* lw a1,56(sp) a1 = arg7
|
||||
* sw ra,24(sp)
|
||||
* jal 0x800F75D0
|
||||
* move a3,v1 (delay slot) a3 = original a1
|
||||
* sll s1,s1,0x6 arg6 << 6
|
||||
* sra s0,s0,0x4 arg5 >> 4
|
||||
* andi s0,s0,0x3f
|
||||
* or s1,s1,s0
|
||||
* sh v0,3006(gp) D_801224F6 = result
|
||||
* sh s1,3008(gp) D_801224F8 = packed
|
||||
* lw ra,24(sp)
|
||||
* lw s1,20(sp)
|
||||
* lw s0,16(sp)
|
||||
* addiu sp,sp,32
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* The frame is 32 bytes: the 16-byte o32 outgoing argument area, `s0` at
|
||||
* 16(sp), `s1` at 20(sp), `ra` at 24(sp), and 4 bytes of alignment pad. The
|
||||
* incoming stack arguments sit at 48/52/56(sp) — the caller's outgoing area
|
||||
* offset by this frame — so the routine takes seven arguments: four in
|
||||
* registers and three on the stack. Both stack words are loaded as full words
|
||||
* (`lw`), so they are `int`, while the four register arguments are stored with
|
||||
* `sh` (their globals are 16-bit). All six stores are `gp`-relative
|
||||
* (gp = 0x80121938) and need the registry's `gp` markers.
|
||||
*
|
||||
* LIMITS: the function name, the callee's parameter meanings, the globals'
|
||||
* types and the packing expression's purpose are hypotheses reconstructed from
|
||||
* the disassembly. Only the compiled bytes are evidence. Whether the four
|
||||
* register arguments are declared `int` or `short` is not observable: the
|
||||
* `sh` stores follow from the globals' width either way.
|
||||
*/
|
||||
|
||||
extern short D_801224F6;
|
||||
extern short D_801224F8;
|
||||
extern short D_801224FA;
|
||||
extern short D_801224FC;
|
||||
extern short D_801224FE;
|
||||
extern short D_80122500;
|
||||
|
||||
int func_800F75D0(int, int, int, int);
|
||||
|
||||
void func_80074C00(int a0, int a1, int a2, int a3,
|
||||
int arg5, int arg6, int arg7) {
|
||||
int r;
|
||||
|
||||
D_801224FA = a0;
|
||||
D_801224FC = a1;
|
||||
D_801224FE = a2;
|
||||
D_80122500 = a3;
|
||||
|
||||
r = func_800F75D0(1, arg7, a0, a1);
|
||||
|
||||
D_801224F6 = r;
|
||||
D_801224F8 = (arg6 << 6) | ((arg5 >> 4) & 0x3f);
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
/*
|
||||
* func_800833CC — 116 bytes at 0x800833CC..0x80083440
|
||||
*
|
||||
* Guarded call returning a flag. Original words:
|
||||
* 27BDFFE8 addiu sp,sp,-24
|
||||
* AFBF0010 sw ra,16(sp)
|
||||
* 8C820008 lw v0,8(a0)
|
||||
* 00000000 nop
|
||||
* 8C420000 lw v0,0(v0)
|
||||
* 00000000 nop
|
||||
* 10400012 beq v0,zero,0x80083430 ; first == 0 -> return 0
|
||||
* 00002821 addu a1,zero,zero ; (delay) result = 0
|
||||
* 3C028013 lui v0,0x8013
|
||||
* 8442E2C4 lh v0,-7484(v0) ; v0 = D_8012E2C4 (SIGNED halfword)
|
||||
* 00000000 nop
|
||||
* 1440000A bne v0,zero,0x80083424 ; D != 0 -> call
|
||||
* 3C030040 lui v1,0x40 ; (delay) v1 = 0x400000
|
||||
* 8C82001C lw v0,28(a0)
|
||||
* 00000000 nop
|
||||
* 8C420008 lw v0,8(v0)
|
||||
* 00000000 nop
|
||||
* 8C420024 lw v0,36(v0)
|
||||
* 00000000 nop
|
||||
* 00431024 and v0,v0,v1 ; v0 &= 0x400000
|
||||
* 14400004 bne v0,zero,0x80083430 ; mask set -> return 0
|
||||
* 00000000 nop
|
||||
* 0C02973B jal 0x800A5CEC ; (L2)
|
||||
* 00000000 nop
|
||||
* 24050001 addiu a1,zero,1 ; result = 1
|
||||
* 8FBF0010 lw ra,16(sp) ; (END)
|
||||
* 00A01021 addu v0,a1,zero ; return result
|
||||
* 03E00008 jr ra
|
||||
* 27BD0018 addiu sp,sp,24 ; (delay slot)
|
||||
*
|
||||
* The two guards are one short-circuit `||` in source order: the `D != 0` test
|
||||
* comes first and jumps straight to the call, and the three-load mask chain is
|
||||
* evaluated only on the `D == 0` path, so the source is
|
||||
* `if (D_8012E2C4 != 0 || (mask chain) == 0)`. The result is a local in `a1`
|
||||
* (`a0` holds the pointer, so `a1` is the free argument register), initialised in
|
||||
* the first `beq` delay slot and set to 1 after the call; the epilogue returns it
|
||||
* with `addu v0,a1,zero`, which is `return result;` and not `return 0;`.
|
||||
*
|
||||
* THE EPILOGUE IS THE `rare-real` SHAPE of cookbook findings 11/35:
|
||||
* `lw ra,16(sp)` / `addu v0,a1,zero` / `jr ra` / `addiu sp,sp,24` — the frame
|
||||
* release is in the jump delay slot, and the instruction between the `ra` load
|
||||
* and the jump is a REAL instruction that does not read `ra`. That is what makes
|
||||
* this shape reachable: GNU `as` in reorder mode can move the frame release into
|
||||
* the jump slot without putting `jr ra` in the `lw ra` load-delay slot, so the
|
||||
* default toolchain reproduces the original with no override. The blocked variant
|
||||
* of the same class is the one where the original has a `nop` there (0x800FE970
|
||||
* in this partition), which needs insert-nop-then-fill and is a harness limit.
|
||||
*
|
||||
* LIMITS: the function name, the callee, the global, the three struct offsets
|
||||
* (8, 0x1c, 0x24), the inner offset 8 and the mask 0x400000 are hypotheses read
|
||||
* from the instruction shape; only the bytes are evidence. D_8012E2C4 is loaded
|
||||
* ABSOLUTELY (`lui`/`lh`, not gp-relative), so it needs no registry gp marker and
|
||||
* the implicit address-named resolution is correct here. The `and` encodes
|
||||
* `and v0,v0,v1` (rs = the loaded value, rt = the mask), so the source operand
|
||||
* order is `value & 0x400000`.
|
||||
*/
|
||||
|
||||
extern short D_8012E2C4;
|
||||
extern void func_800A5CEC(char *arg);
|
||||
|
||||
int func_800833CC(char *a0)
|
||||
{
|
||||
int result = 0;
|
||||
|
||||
if (*(int *)(*(int *)(a0 + 8)) != 0) {
|
||||
if (D_8012E2C4 != 0
|
||||
|| (*(int *)(*(int *)(*(int *)(a0 + 0x1c) + 8) + 0x24) & 0x400000) == 0) {
|
||||
func_800A5CEC(a0);
|
||||
result = 1;
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
/* func_800834B8 — 0x800834B8..0x80083504 (76 bytes).
|
||||
*
|
||||
* Original words (objdump of the validated payload, little-endian):
|
||||
* addiu sp,sp,-24
|
||||
* sw s0,16(sp)
|
||||
* sw ra,20(sp)
|
||||
* jal 0x80083440
|
||||
* _move s0,a0 (delay slot; index kept across the call)
|
||||
* lui v1,0xffff
|
||||
* lui v0,0x8012
|
||||
* lw v0,8968(v0) v0 = D_80122308
|
||||
* sll s0,s0,0x4 index *= 16
|
||||
* addu s0,s0,v0 element = index16 + base (INDEX16 FIRST)
|
||||
* lw v0,0x0(s0)
|
||||
* ori v1,v1,0x7fff v1 = 0xFFFF7FFF = ~0x8000
|
||||
* and v0,v0,v1
|
||||
* sw v0,0x0(s0)
|
||||
* lw ra,20(sp)
|
||||
* lw s0,16(sp)
|
||||
* addiu sp,sp,24
|
||||
* jr ra
|
||||
* _nop (delay slot)
|
||||
*
|
||||
* A sibling of the registered `func_80083440` (0x80083440..0x80083470) and the
|
||||
* second half of the same table operation: it calls that function for its side
|
||||
* effect only (the `jal` result is dead — v0 is immediately reloaded with the
|
||||
* table base), then clears bit 15 of the same 16-byte-strided element. Both
|
||||
* functions index `D_80122308`, the same global *value* (the `lui`+`lw` pair is
|
||||
* a symbol load, so 0x80122308 holds a pointer, not the table), with the same
|
||||
* `index << 4` stride.
|
||||
*
|
||||
* The mask is written as `~0x8000`, which is what makes the two-piece constant
|
||||
* fall out as `lui v1,0xffff` + `ori v1,v1,0x7fff` (0xFFFF7FFF does not fit a
|
||||
* sign-extended `addiu`); cc1 splits it into a high-part load and an `ori`, and
|
||||
* the scheduler separates them around the table load. The store is a plain
|
||||
* `sw`, not a delay-slot store (unlike the sibling, which is a single basic
|
||||
* block and puts its store in the `jr ra` slot).
|
||||
*
|
||||
* OPERAND-ORDER LEVER (the byte-difference that this body turns on): the
|
||||
* address addition is written `(index << 4) + D_80122308`, i.e. index16 first.
|
||||
* cc1 does not canonicalize the operand order of a commutative `+`, and
|
||||
* local-alloc hands the `addu` output the register of its *first* dying source
|
||||
* operand. Written base-first (the sibling's spelling) the element pointer
|
||||
* takes v1, the loaded value takes v0 and the mask is pushed to a0 — 12
|
||||
* differing bytes of pure register numbering with an otherwise identical
|
||||
* instruction sequence. Written index16-first the element pointer stays in the
|
||||
* s0 already holding the index, the base coalesces into v0 with the loaded
|
||||
* value, and the mask lands in v1 — byte-identical. Six spellings of the same
|
||||
* expression (`*element & ~0x8000`, a compound `&=`, a pointer-typed base, a
|
||||
* duplicated address, `0xFFFF7FFF`, a named mask local) all produce the
|
||||
* base-first allocation and DIFF, so the operand order is the whole lever.
|
||||
*
|
||||
* LIMITS: the stride 16, the displaced base and the cleared bit are read from
|
||||
* the bytes; what the field means, and why the sibling sets bits 11/13 while
|
||||
* this clears bit 15 of the same word, is not observable. The call's return
|
||||
* value is provably discarded (v0 is overwritten before any use), so the callee
|
||||
* is declared `int` but its result is ignored. `D_80122308` is typed `int` for
|
||||
* the same reason as in func_80083440.c: the only proven use is as an integer
|
||||
* base added to a scaled index.
|
||||
*/
|
||||
|
||||
extern int D_80122308;
|
||||
extern int func_80083440(int index);
|
||||
|
||||
int func_800834B8(int index)
|
||||
{
|
||||
int *element;
|
||||
int value;
|
||||
|
||||
func_80083440(index);
|
||||
element = (int *)((index << 4) + D_80122308);
|
||||
value = *element;
|
||||
value &= ~0x8000;
|
||||
*element = value;
|
||||
return value;
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
/*
|
||||
* func_8008D9AC — 96 bytes at 0x8008D9AC..0x8008DA0C
|
||||
*
|
||||
* Guarded five-argument call: the first call's result decides whether the second
|
||||
* runs, and the second call's result becomes the return value.
|
||||
*
|
||||
* Original words:
|
||||
* 27BDFFD8 addiu sp,sp,-40
|
||||
* AFB00018 sw s0,24(sp)
|
||||
* 00808021 addu s0,a0,zero ; s0 = a0
|
||||
* AFB1001C sw s1,28(sp)
|
||||
* 00A08821 addu s1,a1,zero ; s1 = a1
|
||||
* AFB20020 sw s2,32(sp)
|
||||
* 00C09021 addu s2,a2,zero ; s2 = a2
|
||||
* AFBF0024 sw ra,36(sp)
|
||||
* 0C01608C jal 0x80058230
|
||||
* 00E02021 addu a0,a3,zero ; (delay) func_80058230(a3)
|
||||
* 10400006 beq v0,zero,0x8008D9F0 ; zero -> return it untouched
|
||||
* 02002021 addu a0,s0,zero ; (delay) a0 = a0
|
||||
* AFA00010 sw zero,16(sp) ; 5th outgoing argument = 0
|
||||
* 02202821 addu a1,s1,zero ; a1 = a1
|
||||
* 02403021 addu a2,s2,zero ; a2 = a2
|
||||
* 0C005C80 jal 0x80017200
|
||||
* 00403821 addu a3,v0,zero ; (delay) a3 = first result
|
||||
* 8FBF0024 lw ra,36(sp)
|
||||
* 8FB20020 lw s2,32(sp)
|
||||
* 8FB1001C lw s1,28(sp)
|
||||
* 8FB00018 lw s0,24(sp)
|
||||
* 27BD0028 addiu sp,sp,40
|
||||
* 03E00008 jr ra
|
||||
* 00000000 nop
|
||||
*
|
||||
* The three callee-saved registers hold the three leading arguments across the
|
||||
* first call; the fourth argument is consumed by that call in its delay slot, so
|
||||
* it needs no save. `sw zero,16(sp)` is the OUTGOING ARGUMENT AREA (the callee's
|
||||
* fifth argument at sp+0x10), not a dead local store — the call therefore takes
|
||||
* five arguments, and the frame arithmetic agrees: 0x10 for the register
|
||||
* arguments, 0x14 for the fifth, then s0/s1/s2/ra at 0x18..0x28.
|
||||
*
|
||||
* The result of the first call is tested and, when non-zero, is both the fourth
|
||||
* argument of the second call and the value the second call's result replaces, so
|
||||
* the source is the `if (v0 != 0) v0 = call(...); return v0;` shape.
|
||||
*
|
||||
* LIMITS: the function name, both callees, the five-argument shape and the
|
||||
* argument/return roles are hypotheses read from the instruction shape; only the
|
||||
* bytes are evidence. func_80017200 is not registered and is referenced by its
|
||||
* address-named spelling, which resolves implicitly.
|
||||
*/
|
||||
|
||||
extern int func_80058230(int a3);
|
||||
extern int func_80017200(int a0, int a1, int a2, int a3, int a4);
|
||||
|
||||
int func_8008D9AC(int a0, int a1, int a2, int a3)
|
||||
{
|
||||
int v0 = func_80058230(a3);
|
||||
|
||||
if (v0 != 0)
|
||||
v0 = func_80017200(a0, a1, a2, v0, 0);
|
||||
return v0;
|
||||
}
|
||||
@@ -0,0 +1,68 @@
|
||||
/*
|
||||
* func_8008F2E0 — 88 bytes at 0x8008F2E0..0x8008F338
|
||||
*
|
||||
* Byte-identical reconstruction of a framed teardown: five void calls in a
|
||||
* fixed order (the first with a zero argument, the last with four) followed by
|
||||
* a block of small-data state resets.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* addiu sp,sp,-24
|
||||
* sw ra,16(sp)
|
||||
* jal 0x800FCFDC
|
||||
* move a0,zero (delay slot)
|
||||
* jal 0x80100740
|
||||
* nop
|
||||
* jal 0x800FDE54
|
||||
* nop
|
||||
* jal 0x800FB5E4
|
||||
* nop
|
||||
* jal 0x800FC2CC
|
||||
* li a0,4 (delay slot)
|
||||
* li v0,-1
|
||||
* sw v0,3352(gp) D_80122650 = -1
|
||||
* sw v0,3344(gp) D_80122648 = -1
|
||||
* sw zero,3340(gp) D_80122644 = 0
|
||||
* sw zero,3376(gp) D_80122668 = 0
|
||||
* sb zero,3364(gp) D_8012265C = 0
|
||||
* lw ra,16(sp)
|
||||
* addiu sp,sp,24
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* The frame is 24 bytes: the 16-byte o32 outgoing argument area plus `ra` at
|
||||
* 16(sp). The `-1` is materialised once into `v0` and stored twice, so the two
|
||||
* globals are written from a single constant; the three zero stores use the
|
||||
* zero register directly. All five stores are `gp`-relative (gp = 0x80121938),
|
||||
* so every target needs the registry's `gp` marker on its name.
|
||||
*
|
||||
* LIMITS: the function names, the callees' arities and parameter types, the
|
||||
* global types and the state block's meaning are hypotheses reconstructed from
|
||||
* the disassembly. Only the compiled bytes are evidence. The order of the five
|
||||
* stores is the original's; it is not derivable from the addresses.
|
||||
*/
|
||||
|
||||
extern int D_80122650;
|
||||
extern int D_80122648;
|
||||
extern int D_80122644;
|
||||
extern int D_80122668;
|
||||
extern char D_8012265C;
|
||||
|
||||
void func_800FCFDC(int);
|
||||
void func_80100740(void);
|
||||
void func_800FDE54(void);
|
||||
void func_800FB5E4(void);
|
||||
void func_800FC2CC(int);
|
||||
|
||||
void func_8008F2E0(void) {
|
||||
func_800FCFDC(0);
|
||||
func_80100740();
|
||||
func_800FDE54();
|
||||
func_800FB5E4();
|
||||
func_800FC2CC(4);
|
||||
|
||||
D_80122650 = -1;
|
||||
D_80122648 = -1;
|
||||
D_80122644 = 0;
|
||||
D_80122668 = 0;
|
||||
D_8012265C = 0;
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
/* func_80090894 — 0x80090894..0x800908E4 (80 bytes).
|
||||
*
|
||||
* Original words (objdump of the validated payload, little-endian):
|
||||
* addiu sp,sp,-24
|
||||
* sw s0,16(sp)
|
||||
* sw ra,20(sp)
|
||||
* jal 0x80090028
|
||||
* _move s0,a0 (delay slot; parameter kept across the call)
|
||||
* andi v0,v0,0xff narrow the call result to 8 bits
|
||||
* beqz v0,0x800908d0 result == 0 -> epilogue
|
||||
* _nop
|
||||
* lui v0,0x8014
|
||||
* lh v0,-30554(v0) v0 = *(short *)0x801388A6 (D_801388A6)
|
||||
* _nop load-delay slot
|
||||
* bne v0,s0,0x800908d0 global != parameter -> epilogue
|
||||
* _nop
|
||||
* jal 0x800907a4
|
||||
* _nop
|
||||
* 0x800908d0: (both early exits land here)
|
||||
* lw ra,20(sp)
|
||||
* lw s0,16(sp)
|
||||
* addiu sp,sp,24
|
||||
* jr ra
|
||||
* _nop
|
||||
*
|
||||
* A guarded forwarding call: the body runs `func_800907a4` only when the
|
||||
* 8-bit-narrowed result of `func_80090028()` is nonzero AND the signed halfword
|
||||
* global `D_801388A6` equals the incoming parameter. The parameter must survive
|
||||
* the first call, which is why it is copied to s0 in the `jal` delay slot; the
|
||||
* `lh` is sign-extended and compared against the full 32-bit parameter, so the
|
||||
* global is a `short` promoted to `int` and the parameter is an `int`.
|
||||
*
|
||||
* The `& 0xFF` is real code, not a redundant mask: `func_80090028` is registered
|
||||
* returning `unsigned char`, but cc1 still emits the `andi`, so the source masks
|
||||
* explicitly (a bare `if (func_80090028())` would test the full register).
|
||||
*
|
||||
* OPERAND-ORDER LEVER, second instance (see func_800834B8.c for the first): the
|
||||
* comparison must be written `D_801388A6 == a0`. Written `a0 != D_801388A6` in
|
||||
* the inverted-branch spelling the body is otherwise byte-identical and leaves
|
||||
* exactly 2 differing bytes at 0x800908C2 — `bne v0,s0` becomes `bne s0,v0`.
|
||||
* cc1 does not canonicalize the operand order of a comparison, and the MIPS
|
||||
* branch encoding is operand-ordered, so the two spellings are different bytes.
|
||||
*
|
||||
* LIMITS: the guard's purpose, the meaning of the halfword global, and whether
|
||||
* the callee `func_80090028` is really called with no arguments are not
|
||||
* observable. The call site leaves a0 holding the incoming parameter, so a
|
||||
* source that passed it as an argument would compile identically — the no-arg
|
||||
* call is chosen only to match the callee's own registered prototype. The name
|
||||
* `D_801388A6` encodes its address, per the registry convention.
|
||||
*/
|
||||
|
||||
extern short D_801388A6;
|
||||
extern unsigned char func_80090028(void);
|
||||
extern void func_800907a4(void);
|
||||
|
||||
void func_80090894(int a0)
|
||||
{
|
||||
if ((func_80090028() & 0xFF) != 0 && D_801388A6 == a0)
|
||||
func_800907a4();
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
/*
|
||||
* func_80090CAC — 84 bytes at 0x80090CAC..0x80090D00
|
||||
*
|
||||
* Byte-identical reconstruction of a framed routine that clamps its argument
|
||||
* against a value obtained from a helper minus a five-unit margin.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* addiu sp,sp,-24
|
||||
* sw s0,16(sp)
|
||||
* sw ra,20(sp)
|
||||
* jal 0x800FE86C
|
||||
* move s0,a0 (delay slot) s = argument
|
||||
* move a0,v0 t = helper()
|
||||
* slt v0,s0,a0 (s < t)
|
||||
* beqz v0,0x80090CD8
|
||||
* slt v0,a0,s0 (delay slot) (t < s)
|
||||
* addiu a0,a0,-5 t - 5
|
||||
* slt v0,a0,s0 (t - 5 < s)
|
||||
* beqz v0,0x80090CE4
|
||||
* nop
|
||||
* move a0,s0 s
|
||||
* andi a0,a0,0xff
|
||||
* lw ra,20(sp)
|
||||
* lw s0,16(sp)
|
||||
* addiu sp,sp,24
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* The frame is 24 bytes: the 16-byte o32 outgoing argument area plus `s0` at
|
||||
* 16(sp) and `ra` at 20(sp). The body is two straight-line guards on the same
|
||||
* value, which is why cc1 merges the second guard's test into the first
|
||||
* branch's delay slot and re-tests only on the path that changed the value.
|
||||
*
|
||||
* LIMITS: the helper's and the callee's names, the argument types and the
|
||||
* `- 5` margin's meaning are hypotheses reconstructed from the disassembly.
|
||||
* Only the compiled bytes are evidence. The final mask is written as an
|
||||
* explicit `& 0xff` because the original emits `andi`; whether the callee
|
||||
* declares an `unsigned char` parameter or the source masks by hand is not
|
||||
* observable from this body alone.
|
||||
*/
|
||||
|
||||
int func_800FE86C(void);
|
||||
void func_800FE844(int);
|
||||
|
||||
void func_80090CAC(int s) {
|
||||
int t = func_800FE86C();
|
||||
|
||||
if (s < t)
|
||||
t = t - 5;
|
||||
if (t < s)
|
||||
t = s;
|
||||
|
||||
func_800FE844(t & 0xff);
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
/*
|
||||
* func_8009214C — 116 bytes at 0x8009214C..0x800921C0
|
||||
*
|
||||
* Byte-identical reconstruction of a framed clamp: a status bit on a nested
|
||||
* structure disables it, otherwise a helper's result is offset by a third
|
||||
* argument, rejected when non-positive, and finally clamped to the second
|
||||
* argument.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* addiu sp,sp,-32
|
||||
* sw s0,16(sp)
|
||||
* move s0,a1 s0 = limit
|
||||
* sw ra,24(sp)
|
||||
* sw s1,20(sp)
|
||||
* lw v0,8(a0) v0 = *(int *)(p + 8)
|
||||
* nop (load delay)
|
||||
* lw v0,20(v0) v0 = *(int *)(v0 + 20)
|
||||
* lui v1,0x100
|
||||
* and v0,v0,v1
|
||||
* bnez v0,0x800921A4 if (status & 0x01000000) return limit
|
||||
* move s1,a2 (delay slot) s1 = offset
|
||||
* jal 0x80092130
|
||||
* nop
|
||||
* move v1,v0 r = helper()
|
||||
* blez v1,0x800921A4 if (r <= 0) return limit
|
||||
* addu v1,v1,s1 (delay slot) r += offset
|
||||
* blez v1,0x800921A4 if (r <= 0) return limit
|
||||
* slt v0,v1,s0 (delay slot) (r < limit)
|
||||
* beqz v0,0x800921A8
|
||||
* move v0,s0 (delay slot) limit
|
||||
* move s0,v1 limit = r
|
||||
* move v0,s0 return limit
|
||||
* lw ra,24(sp)
|
||||
* lw s1,20(sp)
|
||||
* lw s0,16(sp)
|
||||
* addiu sp,sp,32
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* The frame is 32 bytes: the 16-byte o32 outgoing argument area, `s0` at
|
||||
* 16(sp), `s1` at 20(sp), `ra` at 24(sp), and 4 bytes of alignment pad. Both
|
||||
* `s0` and `s1` hold live values across the call, which is why they are saved.
|
||||
* The status mask is `0x01000000`, not `0x100`: `lui v1,0x100` loads the high
|
||||
* half (cookbook finding 30), and the register form (rather than `andi`)
|
||||
* follows from the mask being wider than 16 bits. The function has exactly one
|
||||
* `return limit;`, which is why all three refusal paths branch to one shared
|
||||
* `move v0,s0` block instead of each carrying its own copy in a delay slot.
|
||||
*
|
||||
* LIMITS: the function name, the helper's arity, the structure offsets, the
|
||||
* status mask and the arguments' meanings are hypotheses reconstructed from
|
||||
* the disassembly. Only the compiled bytes are evidence. Whether the guard
|
||||
* reads a bitfield or a masked word is not observable; only `lui 0x100` +
|
||||
* `and` is.
|
||||
*/
|
||||
|
||||
int func_80092130(void);
|
||||
|
||||
int func_8009214C(int *p, int limit, int offset) {
|
||||
int status = *(int *)(*(int *)((char *)p + 8) + 20);
|
||||
int r;
|
||||
|
||||
if ((status & 0x01000000) == 0) {
|
||||
r = func_80092130();
|
||||
if (r > 0) {
|
||||
r += offset;
|
||||
if (r > 0) {
|
||||
if (r < limit)
|
||||
limit = r;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return limit;
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
/* func_80093A08 — 0x80093A08..0x80093A64 (92 bytes).
|
||||
*
|
||||
* Original words (objdump of the validated payload, little-endian):
|
||||
* addiu sp,sp,-24
|
||||
* sll a0,a0,0x10 sign-extend the halfword parameter ...
|
||||
* lui v0,0x8012
|
||||
* lw v0,7168(v0) ... v0 = *(int *)0x80121C00 (the table base)
|
||||
* sra a0,a0,0xe ... and scale it by 4 ((short)p * 4)
|
||||
* sw ra,20(sp)
|
||||
* sw s0,16(sp)
|
||||
* addu a0,a0,v0 a0 = base + (short)p * 4
|
||||
* lw s0,0(a0) s0 = table[(short)p]
|
||||
* jal 0x800697A4
|
||||
* _move a0,s0 (delay slot)
|
||||
* lw v1,12(s0) v1 = *(int *)(s0 + 12)
|
||||
* move a0,s0
|
||||
* lw v0,260(v1) v0 = *(int *)(v1 + 260)
|
||||
* move a1,zero
|
||||
* ori v0,v0,0x200
|
||||
* jal 0x80092A48
|
||||
* _sw v0,260(v1) (delay slot) read-modify-write
|
||||
* lw ra,20(sp)
|
||||
* lw s0,16(sp)
|
||||
* addiu sp,sp,24
|
||||
* jr ra
|
||||
* _nop
|
||||
*
|
||||
* Indexes a pointer table with a **halfword** index, forwards the selected
|
||||
* element to one callee, sets bit 9 of a word reached through two indirections,
|
||||
* then forwards the element again with a zero second argument. The `sll 16` /
|
||||
* `sra 14` pair is this compiler's combined narrow-and-scale: sign-extend the
|
||||
* low 16 bits of the argument and multiply by 4, so the parameter is a `short`
|
||||
* and the table holds 4-byte elements. The element pointer is kept in s0 because
|
||||
* it is live across both calls. The `lui`+`lw` pair is a symbol load, so
|
||||
* 0x80121C00 holds a *pointer* (the table base), not the table.
|
||||
*
|
||||
* LIMITS (unresolved, recorded rather than guessed): the global's declared type
|
||||
* is NOT observable. Three spellings all compile byte-identically — the literal
|
||||
* `*(int *)(*(int *)0x80121C00 + index * 4)`, a pointer-typed
|
||||
* `extern int *D_80121C00` indexed directly, and the symbolic
|
||||
* `extern int D_80121C00` used as an integer base (shipped here, matching the
|
||||
* `int`-holding-a-pointer style documented in func_80083440.c and
|
||||
* func_800419D0.c). The table's element type, the two structure offsets (12 and
|
||||
* 260), the meaning of bit 9 and the callees' names are all hypotheses read from
|
||||
* the instruction shapes. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
extern int D_80121C00;
|
||||
extern void func_800697A4(int x);
|
||||
extern void func_80092A48(int a0, int a1);
|
||||
|
||||
void func_80093A08(short index)
|
||||
{
|
||||
int p = *(int *)(D_80121C00 + index * 4);
|
||||
|
||||
func_800697A4(p);
|
||||
*(int *)(*(int *)(p + 12) + 260) |= 0x200;
|
||||
func_80092A48(p, 0);
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
/*
|
||||
* func_80099078 — 244 bytes at 0x80099078..0x8009916C
|
||||
*
|
||||
* Byte-identical reconstruction of a framed signed three-component scale: a
|
||||
* helper fills a local, and depending on how a second argument compares with
|
||||
* it the routine either scales each component of a vector by the negated
|
||||
* argument and then re-scales by the local, or simply negates each component.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* addiu sp,sp,-40
|
||||
* sw s1,28(sp)
|
||||
* move s1,a0 p
|
||||
* sw s0,24(sp)
|
||||
* move s0,a1 arg
|
||||
* sw s2,32(sp)
|
||||
* move s2,a2 out
|
||||
* sw ra,36(sp)
|
||||
* jal 0x800231BC
|
||||
* addiu a1,sp,16 (delay slot) &tmp (a0 still holds p)
|
||||
* lw v0,16(sp)
|
||||
* nop (load delay)
|
||||
* slt v0,s0,v0 arg < tmp
|
||||
* beqz v0,0x8009911C else: negate-only arm
|
||||
* negu s0,s0 (delay slot) -arg, kept for all three calls
|
||||
* lw a0,0(s1)
|
||||
* jal 0x80010654
|
||||
* move a1,s0 (delay slot)
|
||||
* sw v0,0(s2) out[0] = f(p[0], -arg)
|
||||
* ... the same for p[1] and p[2] ...
|
||||
* lw a0,0(s2)
|
||||
* lw a1,16(sp) tmp
|
||||
* jal 0x80010698
|
||||
* sw v0,8(s2) (delay slot) out[2] = f(p[2], -arg)
|
||||
* lw a0,4(s2)
|
||||
* sw v0,0(s2) out[0] = g(out[0], tmp)
|
||||
* ... the same for out[1] and out[2] ...
|
||||
* j 0x8009914C
|
||||
* sw v0,8(s2) (delay slot) out[2] = g(out[2], tmp)
|
||||
* lw v0,0(s1) negate-only arm
|
||||
* nop
|
||||
* negu v0,v0
|
||||
* sw v0,0(s2)
|
||||
* ... the same for p[1] and p[2] ...
|
||||
* li v0,1
|
||||
* lw ra,36(sp)
|
||||
* lw s2,32(sp)
|
||||
* lw s1,28(sp)
|
||||
* lw s0,24(sp)
|
||||
* addiu sp,sp,40
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* The frame is 40 bytes: the 16-byte o32 outgoing argument area, the local
|
||||
* `tmp` at 16(sp), `s0` at 24(sp), `s1` at 28(sp), `s2` at 32(sp) and `ra` at
|
||||
* 36(sp). The first call receives `p` in `a0` unchanged (the `move s1,a0` does
|
||||
* not disturb it), so it takes the object AND the out-parameter. The three
|
||||
* `out[i]` values are written and then read back through the pointer rather
|
||||
* than kept in registers, because each store is followed by a call.
|
||||
*
|
||||
* LIMITS: the function names, the helpers' arities and semantics, the vector
|
||||
* stride and the meaning of the comparison and the negation are hypotheses
|
||||
* reconstructed from the disassembly. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
int func_800231BC(int *, int *);
|
||||
int func_80010654(int, int);
|
||||
int func_80010698(int, int);
|
||||
|
||||
int func_80099078(int *p, int arg, int *out) {
|
||||
int tmp;
|
||||
|
||||
func_800231BC(p, &tmp);
|
||||
|
||||
if (arg < tmp) {
|
||||
out[0] = func_80010654(p[0], -arg);
|
||||
out[1] = func_80010654(p[1], -arg);
|
||||
out[2] = func_80010654(p[2], -arg);
|
||||
|
||||
out[0] = func_80010698(out[0], tmp);
|
||||
out[1] = func_80010698(out[1], tmp);
|
||||
out[2] = func_80010698(out[2], tmp);
|
||||
} else {
|
||||
out[0] = -p[0];
|
||||
out[1] = -p[1];
|
||||
out[2] = -p[2];
|
||||
}
|
||||
|
||||
return 1;
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
/*
|
||||
* func_800A6934 — 100 bytes at 0x800A6934..0x800A6998
|
||||
*
|
||||
* Byte-identical reconstruction of a framed predicate: it refuses when a
|
||||
* small-data flag is set, otherwise selects one of two small-data words by a
|
||||
* second state value, refuses again when that word is non-zero, and finally
|
||||
* runs one call and reports success.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lbu v0,1544(gp) D_80121F40
|
||||
* addiu sp,sp,-24
|
||||
* bnez v0,0x800A6984 if (flag) return 0
|
||||
* sw ra,16(sp) (delay slot)
|
||||
* lui v0,0x8012
|
||||
* lw v0,7048(v0) *(int *)0x80121B88
|
||||
* nop (load delay)
|
||||
* bnez v0,0x800A6964
|
||||
* nop
|
||||
* lw v0,3516(gp) D_801226F4
|
||||
* j 0x800A6968
|
||||
* nop
|
||||
* lw v0,3536(gp) D_80122708
|
||||
* nop
|
||||
* bnez v0,0x800A6988 if (v) return 0
|
||||
* move v0,zero (delay slot)
|
||||
* jal 0x800A745C
|
||||
* nop
|
||||
* j 0x800A6988
|
||||
* li v0,1 (delay slot)
|
||||
* move v0,zero
|
||||
* lw ra,16(sp)
|
||||
* addiu sp,sp,24
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* The frame is 24 bytes: the 16-byte o32 outgoing argument area plus `ra` at
|
||||
* 16(sp). The `nop` at 0x800A6968 is both the load-delay filler for the
|
||||
* `lw v0,3536(gp)` above it and the target of the `j` at 0x800A695C — the
|
||||
* two paths share one merge point, which is why one `nop` serves both. Both
|
||||
* selected words are `gp`-relative (gp = 0x80121938), and 0x80121B88 is read
|
||||
* as a **literal** constant address so it keeps the absolute encoding rather
|
||||
* than the `gp`-relative form its registry symbol would force.
|
||||
*
|
||||
* LIMITS: the function name, the callee's arity, the flag's and words'
|
||||
* meanings and the literal 13-family constants are hypotheses reconstructed
|
||||
* from the disassembly. Only the compiled bytes are evidence. The two-arm
|
||||
* selection is written with the **inverted** `== 0` condition because the
|
||||
* original's `bnez` jumps over the first arm to the second — the mirrored
|
||||
* block layout of cookbook finding 28, not the natural spelling.
|
||||
*/
|
||||
|
||||
extern char D_80121F40;
|
||||
extern int D_801226F4;
|
||||
extern int D_80122708;
|
||||
|
||||
void func_800A745C(void);
|
||||
|
||||
int func_800A6934(void) {
|
||||
int v;
|
||||
|
||||
if (D_80121F40 != 0)
|
||||
return 0;
|
||||
|
||||
if (*(int *)0x80121B88 == 0)
|
||||
v = D_801226F4;
|
||||
else
|
||||
v = D_80122708;
|
||||
|
||||
if (v != 0)
|
||||
return 0;
|
||||
|
||||
func_800A745C();
|
||||
return 1;
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
/* func_800A6A18 — 0x800A6A18..0x800A6A70 (88 bytes).
|
||||
*
|
||||
* Original words (objdump of the validated payload, little-endian):
|
||||
* addiu sp,sp,-48
|
||||
* move v0,a0 v0 = parameter 0
|
||||
* sw s0,40(sp)
|
||||
* move s0,a1 s0 = parameter 1
|
||||
* addiu a0,sp,16 a0 = &local (sp+16)
|
||||
* sw ra,44(sp)
|
||||
* jal 0x800267CC
|
||||
* _move a1,v0 (delay slot; parameter 0 as the second argument)
|
||||
* beqz v0,0x800A6A58 result == 0 -> return 5
|
||||
* _nop
|
||||
* sw s0,3532(gp) D_80122704 = parameter 1
|
||||
* jal 0x800FB13C
|
||||
* _addiu a0,sp,16 (delay slot; &local again)
|
||||
* sw v0,3528(gp) D_80122700 = result
|
||||
* j 0x800A6A5C
|
||||
* _move v0,zero (delay slot; return 0)
|
||||
* 0x800A6A58:
|
||||
* li v0,5 return 5
|
||||
* 0x800A6A5C:
|
||||
* lw ra,44(sp)
|
||||
* lw s0,40(sp)
|
||||
* addiu sp,sp,48
|
||||
* jr ra
|
||||
* _nop
|
||||
*
|
||||
* A framed guard that fills a 24-byte local through an unregistered callee and,
|
||||
* on success, publishes the second parameter and a value derived from the local
|
||||
* into the small-data block. Both globals are gp-relative full-word stores
|
||||
* (gp = 0x80121938; 0x80121938 + 3528 = 0x80122700, + 3532 = 0x80122704), so
|
||||
* both are `int` — the same small-data run as the registered `func_800A6934`
|
||||
* (D_801226F4 at +3516, D_80122708 at +3536), which types them `int` too. Both
|
||||
* `sw`s store full words: parameter 1 is an `int` and the callee returns `int`.
|
||||
*
|
||||
* The local is passed by address to both callees. `func_800FB13C` reads it as
|
||||
* `unsigned char *p` and consumes p[0..2] as packed BCD; `func_800267CC` writes
|
||||
* at least a word at offset 4 of it (`sw v0,4(s3)` inside that body), so the
|
||||
* object is at least 8 bytes and is byte-addressable at the front.
|
||||
*
|
||||
* FRAME SIZE / LIMITS (unresolved, recorded rather than guessed): the frame is
|
||||
* 48 bytes = the 16-byte o32 outgoing argument area + the local at sp+16 + the
|
||||
* 8-byte saved area at sp+40/44. That constrains the local to **17..24 bytes**,
|
||||
* and it does not narrow further: declared sizes 17, 20 and 24 all compile to
|
||||
* byte-identical output (16 and 12 differ only in the frame size, 6 bytes at
|
||||
* 0x800A6A18), because cc1 rounds the local area up when placing the saved
|
||||
* registers. 24 is chosen as the largest consistent size and the round number;
|
||||
* the true declared size is not observable from these bytes.
|
||||
*
|
||||
* SYMBOL DEPENDENCY: D_80122700 and D_80122704 are NOT yet in
|
||||
* config/symbols.tsv. They must be added with the `gp` marker (see
|
||||
* .run/p10/w-b/symbols-request.tsv) or the two stores encode absolutely instead
|
||||
* of gp-relative. This claim was verified against a local overlay of the
|
||||
* tracked registry plus those two rows.
|
||||
*
|
||||
* LIMITS (semantics): the purpose of the guard, the local's layout, the meaning
|
||||
* of the two published words, and the 5/0 return codes are not observable.
|
||||
* Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
extern int D_80122700;
|
||||
extern int D_80122704;
|
||||
extern int func_800267CC(void *a0, int a1);
|
||||
extern int func_800FB13C(unsigned char *p);
|
||||
|
||||
int func_800A6A18(int a0, int a1)
|
||||
{
|
||||
unsigned char buf[24];
|
||||
|
||||
if (func_800267CC(buf, a0) == 0)
|
||||
return 5;
|
||||
D_80122704 = a1;
|
||||
D_80122700 = func_800FB13C(buf);
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
/*
|
||||
* func_800A6BEC — 72 bytes at 0x800A6BEC..0x800A6C34
|
||||
*
|
||||
* Framed routine that clears one byte of an 8-byte stack buffer and makes three
|
||||
* calls, the first two with a shared gp-relative pointer and the buffer address.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* addiu sp,sp,-0x20 27bdffe0 frame, 32 bytes
|
||||
* li a0,0xe 2404000e a0 = 14
|
||||
* addiu a1,sp,0x10 27a50010 a1 = &buf
|
||||
* addiu a2,gp,0xdc0 27860dc0 a2 = D_801226F8 (gp+0xdc0)
|
||||
* sw ra,0x18(sp) afbf0018 save ra
|
||||
* jal 0x800f8df0 0c03e37c call func_800F8DF0
|
||||
* sb zero,0x10(sp) a3a00010 buf[0] = 0 (delay slot)
|
||||
* li a0,8 24040008 a0 = 8
|
||||
* addiu a2,gp,0xdc0 27860dc0 a2 = D_801226F8
|
||||
* jal 0x800f8df0 0c03e37c call func_800F8DF0
|
||||
* addiu a1,sp,0x10 27a50010 a1 = &buf (delay slot)
|
||||
* addiu a1,gp,0xdc0 27850dc0 a1 = D_801226F8
|
||||
* jal 0x800f8b18 0c03e2c6 call func_800F8B18
|
||||
* addu a0,zero,zero 00002021 a0 = 0 (delay slot)
|
||||
* lw ra,0x18(sp) 8fbf0018 restore ra
|
||||
* addiu sp,sp,0x20 27bd0020 frame release
|
||||
* jr ra 03e00008
|
||||
* nop 00000000 (delay slot)
|
||||
*
|
||||
* Frame arithmetic: `sw ra,0x18(sp)` bounds the save area at 0x18, so the local
|
||||
* area is sp+0x10..sp+0x18 — 8 bytes — and the frame rounds to 0x20.
|
||||
*
|
||||
* Two encoding notes that were checked against the payload bytes rather than the
|
||||
* Ghidra mnemonics: the constant loads are `addiu rd,zero,imm` (so `li`), and the
|
||||
* third call's zero argument is `addu a0,zero,zero` (Ghidra's `clear`), not
|
||||
* `addiu a0,zero,0`.
|
||||
*
|
||||
* THE `-G` REQUIREMENT (the reason this region carries a `cc1=` override). The
|
||||
* three `addiu rt,gp,0xdc0` are the address of D_801226F8 materialised once per
|
||||
* call site, straight into the argument register. Under the harness default
|
||||
* `-G0` the symbol is not small data, so cc1 emits `la` as a two-instruction
|
||||
* large-data address, CSE hoists it into a callee-saved register, and the body
|
||||
* comes out 84 bytes (`sw s0`/`lw s0` added, ra moved 0x18 -> 0x1c). With a
|
||||
* nonzero `-G` that admits the symbol, cc1 treats the address as small data and
|
||||
* emits one `la` per use with no hoist. Measured matrix (all against this range):
|
||||
*
|
||||
* G=0 size=4 -> 84 bytes LENGTH-MISMATCH
|
||||
* G=4 size=1 -> 72 bytes MATCH G=8 size=1 -> 72 bytes MATCH
|
||||
* G=4 size=4 -> 72 bytes MATCH G=8 size=4 -> 72 bytes MATCH
|
||||
* G=4 size=8 -> 84 bytes LENGTH-MISMATCH
|
||||
* G=8 size=8 -> 72 bytes MATCH G=8 size=16 -> 84 bytes LENGTH-MISMATCH
|
||||
*
|
||||
* The rule is exactly `declared size <= G`. A `__attribute__((section(".sdata")))`
|
||||
* on the declaration does not substitute for it (tested at -G0: still 84 bytes).
|
||||
* This is the first byte-visible evidence that the original build ran cc1 with a
|
||||
* nonzero `-G`: with `-G0` no symbol is small data and the gp-relative form
|
||||
* cannot be produced by cc1 at all. The symbol's true size is unknown; the next
|
||||
* symbol above it is at 0x80122708, so it is at most 16 bytes.
|
||||
*
|
||||
* LIMITS: the callees, the global and the meaning of the two small integer
|
||||
* arguments are hypotheses; only the bytes are evidence. `buf`'s size is fixed
|
||||
* by the frame arithmetic; only `buf[0]` is ever written, so its type is a
|
||||
* hypothesis and the buffer is declared `char` because the single observed store
|
||||
* is a byte store. The declared size 8 is the largest the `-G8` override admits
|
||||
* and is not an independent measurement.
|
||||
*/
|
||||
|
||||
extern char D_801226F8[8];
|
||||
extern void func_800F8DF0(int a0, char *a1, char *a2);
|
||||
extern void func_800F8B18(int a0, char *a1);
|
||||
|
||||
void func_800A6BEC(void)
|
||||
{
|
||||
char buf[8];
|
||||
|
||||
buf[0] = 0;
|
||||
func_800F8DF0(0xe, buf, D_801226F8);
|
||||
func_800F8DF0(8, buf, D_801226F8);
|
||||
func_800F8B18(0, D_801226F8);
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
/*
|
||||
* func_800ACAC8 — 108 bytes at 0x800ACAC8..0x800ACB34
|
||||
*
|
||||
* Teardown: conditionally closes a handle, then for a non-null argument resolves a
|
||||
* value and passes it on with a flag, and finally poisons the handle global.
|
||||
*
|
||||
* Original words:
|
||||
* 27BDFFE8 addiu sp,sp,-24
|
||||
* AFB00010 sw s0,16(sp)
|
||||
* 00808021 addu s0,a0,zero ; s0 = a0 (survives two calls)
|
||||
* 8F8406D8 lw a0,1752(gp) ; a0 = D_80122010
|
||||
* AFBF0014 sw ra,20(sp)
|
||||
* 0C02A791 jal 0x800A9E44
|
||||
* 00000000 nop ; (delay) func_800A9E44(handle)
|
||||
* 10400004 beq v0,zero,0x800ACAF8 ; zero -> skip the close
|
||||
* 00000000 nop
|
||||
* 8F8406D8 lw a0,1752(gp) ; a0 = D_80122010 (RELOADED)
|
||||
* 0C02A756 jal 0x800A9D58
|
||||
* 00000000 nop ; (delay) func_800A9D58(handle)
|
||||
* 12000007 beq s0,zero,0x800ACB18 ; null argument -> skip
|
||||
* 00002021 addu a0,zero,zero ; (delay) a0 = 0
|
||||
* 02002821 addu a1,s0,zero ; a1 = a0 (the argument)
|
||||
* 0C018E1C jal 0x80063870
|
||||
* 00003021 addu a2,zero,zero ; (delay) a2 = 0
|
||||
* 00402021 addu a0,v0,zero ; a0 = the resolved value
|
||||
* 0C02A7DF jal 0x800A9F7C
|
||||
* 24050001 addiu a1,zero,1 ; (delay) func_800A9F7C(value, 1)
|
||||
* 2402FFFF addiu v0,zero,-1 ; D_80122010 = -1
|
||||
* AF8206D8 sw v0,1752(gp)
|
||||
* 8FBF0014 lw ra,20(sp)
|
||||
* 8FB00010 lw s0,16(sp)
|
||||
* 27BD0018 addiu sp,sp,24
|
||||
* 03E00008 jr ra
|
||||
* 00000000 nop
|
||||
*
|
||||
* The handle global is loaded TWICE (once per call) rather than held in a
|
||||
* register, so it is read at each use. The three-argument call takes the zero
|
||||
* constant in `a0` and the incoming argument in `a1`, and its result is the first
|
||||
* argument of the two-argument call whose second argument is the constant 1.
|
||||
* `D_80122010` is set to -1 through `v0` at the single exit, which is shared by
|
||||
* both early paths (the frame is entered before either test).
|
||||
*
|
||||
* LIMITS: the function name, the four callees, the handle global and the meaning
|
||||
* of the constants 0 and 1 are hypotheses read from the instruction shape; only
|
||||
* the bytes are evidence. func_800A9E44, func_80063870 and func_800A9F7C are not
|
||||
* registered and are referenced by their address-named spellings. The frame's
|
||||
* `s0` slot holds the argument because it must survive two calls.
|
||||
*/
|
||||
|
||||
extern int D_80122010;
|
||||
extern int func_800A9E44(int a0);
|
||||
extern void func_800A9D58(int a0);
|
||||
extern int func_80063870(int a0, int a1, int a2);
|
||||
extern void func_800A9F7C(int a0, int a1);
|
||||
|
||||
void func_800ACAC8(int a0)
|
||||
{
|
||||
if (func_800A9E44(D_80122010) != 0)
|
||||
func_800A9D58(D_80122010);
|
||||
|
||||
if (a0 != 0)
|
||||
func_800A9F7C(func_80063870(0, a0, 0), 1);
|
||||
|
||||
D_80122010 = -1;
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
/* func_800AE4DC — 0x800AE4DC..0x800AE548 (108 bytes).
|
||||
*
|
||||
* Original words (objdump of the validated payload, little-endian):
|
||||
* addiu sp,sp,-32
|
||||
* sw s0,24(sp)
|
||||
* move s0,a0 s0 = the argument
|
||||
* sw ra,28(sp)
|
||||
* lbu v0,0(s0) v0 = p[0] (UNSIGNED byte)
|
||||
* _nop
|
||||
* ori v0,v0,0x20 v0 |= 0x20
|
||||
* jal 0x800AE458
|
||||
* _sb v0,0(s0) (delay slot) p[0] = v0
|
||||
* addiu a1,sp,16 a1 = &a (sp+16)
|
||||
* lbu a0,1(s0) a0 = p[1]
|
||||
* jal 0x800AE0F4
|
||||
* _addiu a2,sp,18 (delay slot) a2 = &b (sp+18)
|
||||
* lui v0,0x8012
|
||||
* lbu v0,9076(v0) v0 = *(unsigned char *)0x80122374
|
||||
* _nop load-delay slot
|
||||
* beqz v0,0x800AE534 flag == 0 -> epilogue
|
||||
* _li a1,600 (delay slot)
|
||||
* lui a0,0x800b
|
||||
* addiu a0,a0,-8288 a0 = &D_800ADFA0 (%hi/%lo form)
|
||||
* jal 0x8002D0A8
|
||||
* _move a2,s0 (delay slot) a2 = p
|
||||
* 0x800AE534:
|
||||
* lw ra,28(sp)
|
||||
* lw s0,24(sp)
|
||||
* addiu sp,sp,32
|
||||
* jr ra
|
||||
* _nop
|
||||
*
|
||||
* Sets bit 5 of the first byte, runs a no-argument call, splits the second byte
|
||||
* into two halfword out-parameters, and — only when a byte flag is set — reports
|
||||
* through a three-argument callee. The argument is kept in s0 across the first
|
||||
* two calls because it is still the third argument of the last one. The frame is
|
||||
* 32 bytes = 16 (o32 outgoing args) + 8 (the two `short` out-parameters at
|
||||
* sp+16/18) + the saved s0/ra pair at sp+24/28. The `lbu`/`sb` pair fixes an
|
||||
* unsigned byte, and `func_800AE0F4`'s registered prototype
|
||||
* (`unsigned int value, short *p1, short *p2`) fixes the argument order
|
||||
* (value in a0, then p1, then p2).
|
||||
*
|
||||
* ENCODING LEVER (the byte-difference this body turns on): the third argument's
|
||||
* first pointer must be a SYMBOL reference, not a folded literal constant. The
|
||||
* original materialises it as `lui a0,0x800b` + `addiu a0,a0,-8288` — the
|
||||
* `%hi`/`%lo` form, whose high half is incremented because the low half has bit
|
||||
* 15 set. Writing the literal `(char *)0x800ADFA0` makes cc1 fold the constant
|
||||
* and emit `lui a0,0x800a` + `ori a0,a0,0xdfa0` instead, which also knocks the
|
||||
* scheduler off: the two instructions land in the `beqz` delay slot and push
|
||||
* `li a1,600` after the branch, for 11 differing bytes. Declaring
|
||||
* `extern char D_800ADFA0;` and passing `&D_800ADFA0` restores the symbol form
|
||||
* and matches byte-for-byte. Diagnostic: `lui hi, ori lo` = folded literal;
|
||||
* `lui hi+1, addiu -lo` = symbol.
|
||||
*
|
||||
* SYMBOL NOTE: `D_800ADFA0` is not in config/symbols.tsv; the harness resolved
|
||||
* it implicitly from the name-encoded address, exactly as it resolves
|
||||
* `func_XXXXXXXX` callees. No registry change was needed for this claim. If the
|
||||
* merged whole-binary build needs an explicit row, add `D_800ADFA0 0x800ADFA0`.
|
||||
*
|
||||
* LIMITS: the object at 0x800ADFA0 is only proven to be an address; whether it
|
||||
* is a string, a table or a structure is not observable, and the `char` type is
|
||||
* the minimal declaration that yields the right pointer. The meaning of the
|
||||
* literal 600, the byte flag at 0x80122374, bit 5, and the callees' names are
|
||||
* hypotheses read from the instruction shapes. `func_800AE458` is called with no
|
||||
* arguments; the call site leaves a0 holding the parameter either way, so a
|
||||
* source that passed it would compile identically. Only the compiled bytes are
|
||||
* evidence.
|
||||
*/
|
||||
|
||||
extern char D_800ADFA0;
|
||||
extern void func_800AE458(void);
|
||||
extern void func_800AE0F4(unsigned int value, short *p1, short *p2);
|
||||
extern void func_8002D0A8(char *a0, int a1, int a2);
|
||||
|
||||
void func_800AE4DC(unsigned char *p)
|
||||
{
|
||||
short a, b;
|
||||
|
||||
p[0] |= 0x20;
|
||||
func_800AE458();
|
||||
func_800AE0F4(p[1], &a, &b);
|
||||
if (*(unsigned char *)0x80122374)
|
||||
func_8002D0A8(&D_800ADFA0, 600, (char *)p);
|
||||
}
|
||||
@@ -0,0 +1,90 @@
|
||||
/* func_800B1D5C — 0x800B1D5C..0x800B1DD0 (116 bytes).
|
||||
*
|
||||
* Original words (objdump of the validated payload, little-endian):
|
||||
* addiu sp,sp,-72
|
||||
* sw ra,64(sp)
|
||||
* lw v0,12(a1) v0 = *(int *)(a1 + 12)
|
||||
* _nop load-delay slot
|
||||
* lw v1,0(v0) \ 16-byte block copy
|
||||
* lw a0,4(v0) | (all four loads, then all four stores —
|
||||
* lw a1,8(v0) | cc1's block_move, i.e. a STRUCT ASSIGNMENT)
|
||||
* lw a2,12(v0) |
|
||||
* sw v1,16(sp) |
|
||||
* sw a0,20(sp) |
|
||||
* sw a1,24(sp) |
|
||||
* sw a2,28(sp) /
|
||||
* addiu a1,sp,32 a1 = &t (sp+32)
|
||||
* addiu a2,sp,56 a2 = &v (sp+56)
|
||||
* lw v0,20(sp) v0 = s.f1 (word)
|
||||
* lw v1,16(sp) v1 = s.f0 (word)
|
||||
* lhu a0,24(sp) a0 = s.f2 (UNSIGNED halfword)
|
||||
* negu v0,v0 -s.f1
|
||||
* sh a0,52(sp) u.h2 = s.f2
|
||||
* addiu a0,sp,48 a0 = &u (sp+48)
|
||||
* sw v0,20(sp) s.f1 = -s.f1 (the negation is stored BACK)
|
||||
* sh v1,48(sp) u.h0 = s.f0
|
||||
* jal 0x80101C2C
|
||||
* _sh v0,50(sp) (delay slot) u.h1 = -s.f1
|
||||
* lw v0,40(sp) v0 = t.g2 (sp+40 = t+8)
|
||||
* lw ra,64(sp)
|
||||
* addiu sp,sp,72
|
||||
* jr ra
|
||||
* _nop
|
||||
*
|
||||
* Copies a 16-byte structure out of a doubly-indirected pointer, negates its
|
||||
* second word **in place**, projects three of its fields into a halfword record,
|
||||
* passes that record plus two output records to a callee, and returns a word
|
||||
* from one of the outputs. The frame is 72 bytes = 16 (o32 outgoing args) + 48 of
|
||||
* locals + the saved ra at sp+64. The locals are the copied structure at sp+16
|
||||
* (16 bytes), t at sp+32 (16), u at sp+48 (8) and v at sp+56 (8) — the
|
||||
* declaration order s, t, u, v.
|
||||
*
|
||||
* FOUR THINGS THE BYTES PIN DOWN:
|
||||
* 1. `s = *(struct S *)p` is a real STRUCT ASSIGNMENT, not four int assignments:
|
||||
* cc1's block_move emits all four loads before all four stores (using
|
||||
* v1/a0/a1/a2), which the four-separate-assignments spelling does not produce.
|
||||
* 2. Offset 8 of the copied structure is read with `lhu`, so that field is an
|
||||
* `unsigned short` while offsets 0 and 4 are full words. Declared as four ints
|
||||
* the same source emits `lw` there.
|
||||
* 3. The negation is stored BACK to the structure (`sw v0,20(sp)`), so the source
|
||||
* modifies the copy in place — and `u.h1 = s.f1` then reuses the negated value
|
||||
* (one `lw` feeds both the `negu` and the two stores), so the source reads the
|
||||
* field back rather than negating twice.
|
||||
* 4. STATEMENT ORDER IS THE LEVER: the three record stores and the negation must
|
||||
* be written `s.f1 = -s.f1; u.h0 = s.f0; u.h1 = s.f1; u.h2 = s.f2;`. All ten
|
||||
* permutations of these four statements were compiled; only that one is
|
||||
* byte-identical (the others leave 4-31 differing bytes), because cc1's
|
||||
* scheduler keeps this store order. The "natural" order is also the correct
|
||||
* one here — worth checking first on similar record-building bodies.
|
||||
*
|
||||
* LIMITS: the field names, `t`'s layout (only t+8 is proven, by the returned
|
||||
* `lw`), `u`'s fourth halfword (padding, never read), and `v`'s size (8 bytes,
|
||||
* inferred from the frame rather than read) are hypotheses. The callee is
|
||||
* declared with its REGISTERED prototype `(const int *, int *, int *)` and the
|
||||
* struct pointers are cast at the call site; a struct-typed prototype compiles
|
||||
* identically, so the pointer types are not observable. What the structures mean
|
||||
* is unknown and not guessed. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
struct S { int f0; int f1; unsigned short f2; unsigned short f3; int f4; };
|
||||
struct T { int g0, g1, g2, g3; };
|
||||
struct U { short h0; short h1; unsigned short h2; short h3; };
|
||||
struct V { int i0, i1; };
|
||||
|
||||
extern void func_80101C2C(const int *a0, int *a1, int *a2);
|
||||
|
||||
int func_800B1D5C(int a0, int a1)
|
||||
{
|
||||
struct S s;
|
||||
struct T t;
|
||||
struct U u;
|
||||
struct V v;
|
||||
|
||||
s = *(struct S *)(*(int *)(a1 + 12));
|
||||
s.f1 = -s.f1;
|
||||
u.h0 = s.f0;
|
||||
u.h1 = s.f1;
|
||||
u.h2 = s.f2;
|
||||
func_80101C2C((const int *)&u, (int *)&t, (int *)&v);
|
||||
return t.g2;
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
/*
|
||||
* func_800B6C60 — 84 bytes at 0x800B6C60..0x800B6CB4
|
||||
*
|
||||
* Byte-identical reconstruction of a framed routine that queries a lookup,
|
||||
* and on a non-null answer announces it to a small-data buffer, flags the
|
||||
* event, and dispatches a three-argument call with the original argument.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* addiu sp,sp,-24
|
||||
* sw s0,16(sp)
|
||||
* sw ra,20(sp)
|
||||
* jal 0x800B6BDC
|
||||
* move s0,a0 (delay slot) s = argument
|
||||
* beqz v0,0x800B6CA0 if (r == 0) return
|
||||
* nop
|
||||
* addiu a0,gp,1788 a0 = &D_80122034 (small-data address)
|
||||
* jal 0x80026560
|
||||
* move a1,v0 (delay slot) a1 = r
|
||||
* li v0,1
|
||||
* sb v0,1794(gp) D_8012203A = 1
|
||||
* move a0,s0 a0 = s
|
||||
* li a1,2
|
||||
* jal 0x800B6894
|
||||
* move a2,zero (delay slot)
|
||||
* lw ra,20(sp)
|
||||
* lw s0,16(sp)
|
||||
* addiu sp,sp,24
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* The frame is 24 bytes: the 16-byte o32 outgoing argument area plus `s0` at
|
||||
* 16(sp) and `ra` at 20(sp). `addiu a0,gp,1788` is the small-data *address*
|
||||
* form (`la` of a `gp`-relative symbol, 1788 + 0x80121938 = 0x80122034), not a
|
||||
* load: taking the address of a `gp` symbol keeps the `gp` base and an
|
||||
* immediate, so it needs the registry's `gp` marker on the name.
|
||||
*
|
||||
* LIMITS: the function names, the callee's parameter count and types, the
|
||||
* global types and the flag's meaning are hypotheses reconstructed from the
|
||||
* disassembly. Only the compiled bytes are evidence. Whether the first call's
|
||||
* argument is `&D_80122034` or a byte offset from a base is not observable;
|
||||
* only the `addiu a0,gp,1788` encoding is.
|
||||
*/
|
||||
|
||||
extern char D_80122034;
|
||||
extern char D_8012203A;
|
||||
|
||||
char *func_800B6BDC(int);
|
||||
void func_80026560(char *, char *);
|
||||
void func_800B6894(int, int, int);
|
||||
|
||||
void func_800B6C60(int key) {
|
||||
char *r = func_800B6BDC(key);
|
||||
|
||||
if (r != 0) {
|
||||
func_80026560(&D_80122034, r);
|
||||
D_8012203A = 1;
|
||||
func_800B6894(key, 2, 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,78 @@
|
||||
/*
|
||||
* func_80105148 — 132 bytes at 0x80105148..0x801051CC
|
||||
*
|
||||
* Byte-identical reconstruction of a framed three-way dispatch on a byte field
|
||||
* of the object passed in: each arm calls a different routine with a different
|
||||
* argument drawn from the same object, and all three fall to one return.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* addiu sp,sp,-24
|
||||
* sw ra,16(sp)
|
||||
* lbu v1,70(a0) v1 = p[70]
|
||||
* li v0,3
|
||||
* beq v1,v0,0x8010519C case 3
|
||||
* slti v0,v1,4 (delay slot)
|
||||
* beqz v0,0x80105178
|
||||
* li v0,2 (delay slot)
|
||||
* beq v1,v0,0x8010518C case 2
|
||||
* nop
|
||||
* j 0x801051BC default
|
||||
* nop
|
||||
* li v0,4
|
||||
* beq v1,v0,0x801051B0 case 4
|
||||
* nop
|
||||
* j 0x801051BC default
|
||||
* nop
|
||||
* jal 0x80105B54 case 2
|
||||
* nop
|
||||
* j 0x801051BC
|
||||
* nop
|
||||
* lbu a1,228(a0) case 3
|
||||
* jal 0x80105B68
|
||||
* nop
|
||||
* j 0x801051BC
|
||||
* nop
|
||||
* lbu a1,71(a0) case 4
|
||||
* jal 0x80105BA8
|
||||
* nop
|
||||
* lw ra,16(sp)
|
||||
* addiu sp,sp,24
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* The frame is 24 bytes: the 16-byte o32 outgoing argument area plus `ra` at
|
||||
* 16(sp). The decision tree is the balanced form cc1 emits for the three case
|
||||
* values 2/3/4 — test the middle value first, then split on `< 4` — and the
|
||||
* default arm is a jump to the shared epilogue rather than a fall-through,
|
||||
* because every arm ends at the same return. The `lbu` loads (rather than
|
||||
* `lb`) are why the field is `unsigned char`: plain `char` is unsigned on this
|
||||
* target (cookbook finding 7).
|
||||
*
|
||||
* LIMITS: the function name, the callees' arities and parameter types, the
|
||||
* field offsets and the case meanings are hypotheses reconstructed from the
|
||||
* disassembly. Only the compiled bytes are evidence. Whether the switch
|
||||
* selector is a struct field or an array element is not observable; only
|
||||
* `lbu v1,70(a0)` is. The two `lbu`s target `a1`, not `a0`, and `a0` still
|
||||
* holds the incoming object pointer at both call sites — so each arm passes
|
||||
* the object as the FIRST argument and the loaded byte as the second, and the
|
||||
* case-2 call passes the object too (which costs no instruction, since `a0`
|
||||
* is already loaded).
|
||||
*/
|
||||
|
||||
void func_80105B54(unsigned char *);
|
||||
void func_80105B68(unsigned char *, unsigned char);
|
||||
void func_80105BA8(unsigned char *, unsigned char);
|
||||
|
||||
void func_80105148(unsigned char *p) {
|
||||
switch (p[70]) {
|
||||
case 2:
|
||||
func_80105B54(p);
|
||||
break;
|
||||
case 3:
|
||||
func_80105B68(p, p[228]);
|
||||
break;
|
||||
case 4:
|
||||
func_80105BA8(p, p[71]);
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,113 @@
|
||||
--- a/maspsx/__init__.py
|
||||
+++ b/maspsx/__init__.py
|
||||
@@ -78,6 +78,38 @@
|
||||
return line.strip()
|
||||
|
||||
|
||||
+def line_jumps_via_reg(line: str, r_source: str) -> bool:
|
||||
+ """True if `line` is a register jump whose target is `r_source`.
|
||||
+
|
||||
+ Cookbook finding 27's gap: the delay-nop predicate (`line_loads_from_reg`)
|
||||
+ recognises loads and branches but not `jr`/`jalr`, and `jr`/`jalr` are not in
|
||||
+ `jump_mnemonics` either, so a load feeding a register jump got no delay nop.
|
||||
+
|
||||
+ This is OPT-IN (`--nop-on-reg-read`): the default must stay byte-identical for
|
||||
+ the corpus already matched against it. Local patch to a pinned vendored tool;
|
||||
+ see docs/SETUP.md for provenance.
|
||||
+ """
|
||||
+ line = strip_comments(line)
|
||||
+
|
||||
+ # escape dollar
|
||||
+ r_source = r_source.replace("$", r"\$")
|
||||
+
|
||||
+ if match := re.match(r"^([A-z][A-z0-9]*)\s+(.*)$", line):
|
||||
+ op, rest = match.group(1, 2)
|
||||
+ else:
|
||||
+ return False
|
||||
+
|
||||
+ if op in ("jr", "jalr"):
|
||||
+ # jr $31
|
||||
+ if re.match(rf"^{r_source}$", rest):
|
||||
+ return True
|
||||
+ # jalr $2,$31 (destination first, target last)
|
||||
+ if re.match(rf"^.*,\s*{r_source}\s*$", rest):
|
||||
+ return True
|
||||
+
|
||||
+ return False
|
||||
+
|
||||
+
|
||||
def line_loads_from_reg(line: str, r_source: str, loads_to_reg=False) -> bool:
|
||||
"""
|
||||
NOTE: Returns True even if line might use $at expansion
|
||||
@@ -448,6 +480,8 @@
|
||||
gp_allow_la=False,
|
||||
use_comm_section=False,
|
||||
use_comm_for_lcomm=False,
|
||||
+ no_jump_slot_nop=False,
|
||||
+ nop_on_reg_read=False,
|
||||
):
|
||||
self.lines = [x.strip() for x in lines]
|
||||
|
||||
@@ -460,6 +494,15 @@
|
||||
self.nop_mflo_mfhi = nop_mflo_mfhi
|
||||
self.nop_lw_lw = nop_lw_lw
|
||||
|
||||
+ # Local opt-in modes (Phase 10, developer-authorised). Both default off so
|
||||
+ # the matched corpus reproduces byte-identically.
|
||||
+ # no_jump_slot_nop: suppress the unconditional reorder nop after a
|
||||
+ # branch/jump, so GNU `as` can fill the slot itself (worker C's R1).
|
||||
+ # nop_on_reg_read: additionally treat a following `jr`/`jalr` that uses
|
||||
+ # the loaded register as needing the delay nop (worker C's R2).
|
||||
+ self.no_jump_slot_nop = no_jump_slot_nop
|
||||
+ self.nop_on_reg_read = nop_on_reg_read
|
||||
+
|
||||
self.sltu_at = sltu_at
|
||||
self.addiu_at = addiu_at
|
||||
self.div_uses_tge = div_uses_tge
|
||||
@@ -696,7 +739,14 @@
|
||||
) -> List[str]:
|
||||
res: List[str] = []
|
||||
|
||||
- if line_loads_from_reg(next_instruction, r_dest, loads_to_reg=self.nop_lw_lw):
|
||||
+ reuse = line_loads_from_reg(next_instruction, r_dest, loads_to_reg=self.nop_lw_lw)
|
||||
+ if not reuse and self.nop_on_reg_read:
|
||||
+ # Cookbook finding 27's gap: the predicate above recognises loads and
|
||||
+ # branches but not `jr`/`jalr`, so a load feeding a register jump got
|
||||
+ # no delay nop. Opt-in, so the default path is unchanged.
|
||||
+ reuse = line_jumps_via_reg(next_instruction, r_dest)
|
||||
+
|
||||
+ if reuse:
|
||||
nop_required = False
|
||||
|
||||
if not uses_at(next_instruction):
|
||||
@@ -1102,7 +1152,7 @@
|
||||
|
||||
elif op in branch_mnemonics or op in jump_mnemonics:
|
||||
res.append(line)
|
||||
- if self.is_reorder:
|
||||
+ if self.is_reorder and not self.no_jump_slot_nop:
|
||||
res.append("nop # DEBUG: branch/jump")
|
||||
|
||||
elif op == "move":
|
||||
--- a/maspsx.py
|
||||
+++ b/maspsx.py
|
||||
@@ -62,6 +62,10 @@
|
||||
parser.add_argument("--passthrough", action="store_true")
|
||||
parser.add_argument("--use-comm-section", action="store_true")
|
||||
parser.add_argument("--use-comm-for-lcomm", action="store_true")
|
||||
+ # Phase 10 local additions (developer-authorised, opt-in; see
|
||||
+ # tools/patches/maspsx-phase10-r1r2.patch and docs/SETUP.md).
|
||||
+ parser.add_argument("--no-jump-slot-nop", action="store_true")
|
||||
+ parser.add_argument("--nop-on-reg-read", action="store_true")
|
||||
# decomp.me debugging
|
||||
parser.add_argument("--print-output", action="store_true")
|
||||
parser.add_argument("--print-input", action="store_true")
|
||||
@@ -152,6 +156,8 @@
|
||||
gp_allow_la=version_config.gp_allow_la,
|
||||
use_comm_section=args.use_comm_section,
|
||||
use_comm_for_lcomm=args.use_comm_for_lcomm,
|
||||
+ no_jump_slot_nop=args.no_jump_slot_nop,
|
||||
+ nop_on_reg_read=args.nop_on_reg_read,
|
||||
)
|
||||
|
||||
try:
|
||||
+39
-10
@@ -186,6 +186,7 @@ class Region:
|
||||
as_flags: tuple[str, ...] = ()
|
||||
no_gp: tuple[str, ...] = ()
|
||||
no_maspsx: bool = False
|
||||
maspsx_flags: tuple[str, ...] = ()
|
||||
|
||||
|
||||
# A region's optional fourth field: space-separated `key=value` overrides.
|
||||
@@ -198,13 +199,22 @@ class Region:
|
||||
# maspsx's unconditional `nop` for a jump destroys the delay-slot fill that GNU
|
||||
# `as` reorder mode performs on an expanded symbol store (0x80102B10, 0x800F8B6C,
|
||||
# 0x800F3160), while the ASPSX `la`/`addiu` form needs maspsx (func_8002D2BC).
|
||||
# The same key also takes Phase 10's opt-in maspsx modes, which are local additions
|
||||
# to the pinned vendored tool (tools/patches/maspsx-phase10-r1r2.patch):
|
||||
# `maspsx=noreordernop` suppresses maspsx's unconditional reorder `nop` after a
|
||||
# branch/jump so GNU `as` can fill the slot itself (worker C's R1).
|
||||
# `maspsx=regread` additionally treats a following `jr`/`jalr` that uses the
|
||||
# loaded register as needing the load-delay `nop` (worker C's R2).
|
||||
# Both default off, so every region without them compiles exactly as before.
|
||||
_REGION_OPTION_KEYS = ("cc1", "as", "gp", "maspsx")
|
||||
_MASPSX_MODES = {"off": "--off", "noreordernop": "--no-jump-slot-nop",
|
||||
"regread": "--nop-on-reg-read"}
|
||||
|
||||
|
||||
def parse_region_options(text: str, line_number: int) -> tuple[tuple[str, ...], tuple[str, ...], tuple[str, ...], bool]:
|
||||
def parse_region_options(text: str, line_number: int) -> tuple[tuple[str, ...], tuple[str, ...], tuple[str, ...], bool, tuple[str, ...]]:
|
||||
"""Parse the optional per-region override field.
|
||||
|
||||
Returns `(cc1_flags, as_flags, no_gp, no_maspsx)`. `gp=-NAME` names a symbol
|
||||
Returns `(cc1_flags, as_flags, no_gp, no_maspsx, maspsx_flags)`. `gp=-NAME` names a symbol
|
||||
the registry marks `gp` that this region accesses absolutely instead, and
|
||||
`maspsx=off` drops the ASPSX emulation stage for this region.
|
||||
"""
|
||||
@@ -239,13 +249,19 @@ def parse_region_options(text: str, line_number: int) -> tuple[tuple[str, ...],
|
||||
)
|
||||
no_gp.append(name[1:])
|
||||
no_maspsx = False
|
||||
maspsx_flags: list[str] = []
|
||||
for value in overrides.get("maspsx", ()):
|
||||
if value not in ("on", "off"):
|
||||
if value not in _MASPSX_MODES:
|
||||
raise ToolError(
|
||||
f"regions line {line_number}: expected 'maspsx=on' or 'maspsx=off', got {value!r}"
|
||||
f"regions line {line_number}: expected one of "
|
||||
f"{', '.join(sorted(_MASPSX_MODES))} for 'maspsx', got {value!r}"
|
||||
)
|
||||
no_maspsx = value == "off"
|
||||
return (overrides.get("cc1", ()), overrides.get("as", ()), tuple(no_gp), no_maspsx)
|
||||
if value == "off":
|
||||
no_maspsx = True
|
||||
else:
|
||||
maspsx_flags.append(_MASPSX_MODES[value])
|
||||
return (overrides.get("cc1", ()), overrides.get("as", ()), tuple(no_gp), no_maspsx,
|
||||
tuple(maspsx_flags))
|
||||
|
||||
|
||||
def parse_regions(text: str) -> list[Region]:
|
||||
@@ -263,10 +279,10 @@ def parse_regions(text: str) -> list[Region]:
|
||||
end = parse_address(fields[1])
|
||||
if not 0 <= start < end:
|
||||
raise ToolError(f"regions line {number}: invalid range")
|
||||
cc1_flags, as_flags, no_gp, no_maspsx = ((), (), (), False)
|
||||
cc1_flags, as_flags, no_gp, no_maspsx, maspsx_flags = ((), (), (), False, ())
|
||||
if len(fields) == 4:
|
||||
cc1_flags, as_flags, no_gp, no_maspsx = parse_region_options(fields[3], number)
|
||||
regions.append(Region(start, end, fields[2], cc1_flags, as_flags, no_gp, no_maspsx))
|
||||
cc1_flags, as_flags, no_gp, no_maspsx, maspsx_flags = parse_region_options(fields[3], number)
|
||||
regions.append(Region(start, end, fields[2], cc1_flags, as_flags, no_gp, no_maspsx, maspsx_flags))
|
||||
regions.sort(key=lambda region: region.start)
|
||||
for left, right in zip(regions, regions[1:]):
|
||||
if right.start < left.end:
|
||||
@@ -453,6 +469,7 @@ class Toolchain:
|
||||
defsyms: Sequence[str]
|
||||
gp_symbols: frozenset[str] = frozenset()
|
||||
maspsx: Path | None = None
|
||||
maspsx_flags: tuple[str, ...] = ()
|
||||
aspsx_version: str = DEFAULT_ASPSX_VERSION
|
||||
nm: Path | None = None
|
||||
|
||||
@@ -512,7 +529,8 @@ def compile_c(source: Path, out_object: Path, work: Path, tools: Toolchain) -> N
|
||||
transformed = work / (out_object.stem + ".maspsx.s")
|
||||
run_filter(
|
||||
[sys.executable, str(tools.maspsx),
|
||||
f"--aspsx-version={tools.aspsx_version}"],
|
||||
f"--aspsx-version={tools.aspsx_version}",
|
||||
*tools.maspsx_flags],
|
||||
assembly, transformed,
|
||||
)
|
||||
assembly = transformed
|
||||
@@ -696,6 +714,7 @@ def _build(args: argparse.Namespace) -> tuple[Path, Path]:
|
||||
as_flags=[*tools.as_flags, *region.as_flags],
|
||||
gp_symbols=tools.gp_symbols - frozenset(region.no_gp),
|
||||
maspsx=None if region.no_maspsx else tools.maspsx,
|
||||
maspsx_flags=region.maspsx_flags,
|
||||
)
|
||||
compile_c(item.source, object_path, work, region_tools)
|
||||
localize_symbols(object_path, tools)
|
||||
@@ -778,6 +797,12 @@ def add_toolchain_arguments(parser: argparse.ArgumentParser) -> None:
|
||||
help="ASPSX emulator run between cc1 and the assembler")
|
||||
parser.add_argument("--no-maspsx", action="store_true",
|
||||
help="assemble cc1 output directly, without maspsx")
|
||||
parser.add_argument("--no-jump-slot-nop", action="store_true",
|
||||
help="maspsx mode: suppress the unconditional reorder nop after a "
|
||||
"branch/jump so GNU as can fill the slot (region: maspsx=noreordernop)")
|
||||
parser.add_argument("--nop-on-reg-read", action="store_true",
|
||||
help="maspsx mode: also treat a following jr/jalr that uses the loaded "
|
||||
"register as needing the load-delay nop (region: maspsx=regread)")
|
||||
parser.add_argument("--aspsx-version", default=DEFAULT_ASPSX_VERSION,
|
||||
help="ASPSX version for maspsx (default: the SDK's 2.81)")
|
||||
parser.add_argument("--cpp-flag", action="append", default=[],
|
||||
@@ -879,6 +904,10 @@ def resolve_toolchain(args: argparse.Namespace) -> Toolchain:
|
||||
defsyms=defsyms,
|
||||
gp_symbols=frozenset(gp_names - set(getattr(args, "no_gp", ()))),
|
||||
maspsx=maspsx,
|
||||
maspsx_flags=tuple(
|
||||
flag for flag, enabled in (
|
||||
("--no-jump-slot-nop", getattr(args, "no_jump_slot_nop", False)),
|
||||
("--nop-on-reg-read", getattr(args, "nop_on_reg_read", False))) if enabled),
|
||||
aspsx_version=args.aspsx_version,
|
||||
)
|
||||
|
||||
|
||||
@@ -203,20 +203,42 @@ class RegionOverrideTests(unittest.TestCase):
|
||||
"""The per-region override field: cc1/as flags, gp exclusions, maspsx stage."""
|
||||
|
||||
def test_cc1_and_as_flags(self) -> None:
|
||||
cc1, as_flags, no_gp, no_maspsx = sf3_match.parse_region_options("cc1=-O0 as=-G8", 1)
|
||||
cc1, as_flags, no_gp, no_maspsx, maspsx_flags = sf3_match.parse_region_options(
|
||||
"cc1=-O0 as=-G8", 1)
|
||||
self.assertEqual(cc1, ("-O0",))
|
||||
self.assertEqual(as_flags, ("-G8",))
|
||||
self.assertEqual(no_gp, ())
|
||||
self.assertFalse(no_maspsx)
|
||||
self.assertEqual(maspsx_flags, ())
|
||||
|
||||
def test_gp_exclusion_names(self) -> None:
|
||||
_cc1, _as_flags, no_gp, _no_maspsx = sf3_match.parse_region_options(
|
||||
_cc1, _as_flags, no_gp, _no_maspsx, _flags = sf3_match.parse_region_options(
|
||||
"gp=-D_80121F84,-D_80121F40", 1)
|
||||
self.assertEqual(no_gp, ("D_80121F84", "D_80121F40"))
|
||||
|
||||
def test_maspsx_off(self) -> None:
|
||||
_cc1, _as_flags, _no_gp, no_maspsx = sf3_match.parse_region_options("maspsx=off", 1)
|
||||
_cc1, _as_flags, _no_gp, no_maspsx, maspsx_flags = sf3_match.parse_region_options(
|
||||
"maspsx=off", 1)
|
||||
self.assertTrue(no_maspsx)
|
||||
self.assertEqual(maspsx_flags, ())
|
||||
|
||||
def test_maspsx_phase10_modes(self) -> None:
|
||||
"""The Phase 10 opt-in maspsx modes map to their CLI flags, and the
|
||||
legacy `on`/`off` spelling is gone."""
|
||||
_c, _a, _g, no_maspsx, flags = sf3_match.parse_region_options(
|
||||
"maspsx=noreordernop,regread", 1)
|
||||
self.assertFalse(no_maspsx)
|
||||
self.assertEqual(flags, ("--no-jump-slot-nop", "--nop-on-reg-read"))
|
||||
|
||||
def test_maspsx_modes_compose_with_off(self) -> None:
|
||||
_c, _a, _g, no_maspsx, flags = sf3_match.parse_region_options(
|
||||
"maspsx=off,noreordernop", 1)
|
||||
self.assertTrue(no_maspsx)
|
||||
self.assertEqual(flags, ("--no-jump-slot-nop",))
|
||||
|
||||
def test_default_region_has_no_maspsx_flags(self) -> None:
|
||||
regions = sf3_match.parse_regions("0x80010000 0x80010010 src/a.c\n")
|
||||
self.assertEqual(regions[0].maspsx_flags, ())
|
||||
|
||||
def test_rejects_a_gp_token_without_a_minus(self) -> None:
|
||||
with self.assertRaises(sf3_match.ToolError):
|
||||
|
||||
Reference in New Issue
Block a user