phase9: merge cycle 1 — 57 claims verified, 39 new regions (206 distinct bodies)
Coordinator-verified every staged claim with its own range runs (57/57 MATCH, one required the staged gp symbol g_80122158). Merged to a candidate, whole- binary gate MATCH at c_regions=215, differing_bytes=0, SHA-1 e173426c157384ebf1b6caf8c6fea18a85a14af9; promoted; make check green (regions=215 disagreements=0 result=AGREE, c_regions=215 MATCH, 223 tests). Workers: A 23, B 13, C 21 claims. New symbols: g_80122158/g_80122068/ g_8012277C/g_80122738 (gp-marked, worker C request). Override additions: 0x80104C38 maspsx=off (verified 1-byte DIFF with maspsx on at the loop back-edge, finding 17 family). gtemac.h GTE control map corrected after two workers' independent decodes and coordinator raw-word verification: the executable's control registers are the standard map UNshifted for 0..5 (rotation matrix, 0x80101CAC) and standard+7 from RBK onward (3 RBK .. 8 DQB); light matrix is 6..0, far colour 1..3, H 6, DQA 7, DQB 8. Cookbook finding 24's label for 3..5 was wrong; the numbers were always right. Added gte_ldH, gte_ldRT1RT2..gte_ldRT33; func_8001AE3C.c updated to the corrected RBK/GBK/BBK macros (re-verified MATCH); 0x80102FD4 and 0x80101CAC rewritten to macro form (re-verified MATCH). Recorded negatives: A 4 (cond-value-ifconv, alloc-scheduling, cc1-fold, strength-reduce+loop-rotate), B 7 (incl. exit-duplication, load-use-nop GTE alloc class, finding-4 sign-adjust trap), C 6 (incl. delay-slot-fill class, cc1-scheduler-bound). All deployed from src/.
This commit is contained in:
@@ -13,11 +13,18 @@
|
||||
0x80012780 0x8001278C src/func_80012780.c
|
||||
0x800127F0 0x8001281C src/func_800127F0.c
|
||||
0x8001281C 0x80012834 src/func_8001281C.c
|
||||
0x80012D8C 0x80012DBC src/func_80012D8C.c
|
||||
0x80012DBC 0x80012DE8 src/func_80012DBC.c
|
||||
0x80013C88 0x80013C90 src/func_80013C88.c
|
||||
0x800160E8 0x80016110 src/func_800160E8.c
|
||||
0x80016110 0x80016120 src/func_80016110.c
|
||||
0x80016120 0x80016158 src/func_80016120.c
|
||||
0x80016158 0x80016174 src/func_80016158.c
|
||||
0x80016174 0x80016198 src/func_80016174.c
|
||||
0x80016198 0x800161E0 src/func_80016198.c
|
||||
0x80016E50 0x80016E68 src/func_80016E50.c
|
||||
0x800171D8 0x80017200 src/func_800171D8.c
|
||||
0x8001761C 0x80017660 src/func_8001761C.c
|
||||
0x800179B8 0x800179CC src/func_800179B8.c
|
||||
0x800179CC 0x800179D4 src/func_800179CC.c
|
||||
0x800179D4 0x800179E0 src/func_800179D4.c
|
||||
@@ -27,6 +34,7 @@
|
||||
0x80017C60 0x80017C6C src/func_80017C60.c
|
||||
0x80017D1C 0x80017D48 src/func_80017D1C.c
|
||||
0x80017DD0 0x80017DF0 src/func_80017DD0.c
|
||||
0x800182D4 0x800182F4 src/func_800182D4.c
|
||||
0x80019B6C 0x80019B94 src/func_80019B6C.c
|
||||
0x80019C04 0x80019C10 src/func_80019C04.c
|
||||
0x8001AA9C 0x8001AAA8 src/func_8001AA9C.c
|
||||
@@ -36,6 +44,7 @@
|
||||
0x80021F88 0x80021FA8 src/func_80021F88.c
|
||||
0x80022A60 0x80022A98 src/func_80022A60.c
|
||||
0x80022FB8 0x80022FCC src/func_80022FB8.c
|
||||
0x800230E4 0x8002311C src/func_800230E4.c
|
||||
0x80024C14 0x80024C34 src/func_80024C14.c
|
||||
0x80026180 0x800261C0 src/func_80026180.c
|
||||
0x80026258 0x80026264 src/func_80026258.c
|
||||
@@ -48,11 +57,14 @@
|
||||
0x8002C7BC 0x8002C7EC src/func_8002C7BC.c
|
||||
0x8002C888 0x8002C894 src/func_8002C888.c
|
||||
0x8002C894 0x8002C8A4 src/func_8002C894.c
|
||||
0x8002D1F8 0x8002D22C src/func_8002D1F8.c
|
||||
0x8002D288 0x8002D2A0 src/func_8002D288.c gp=-D_80121F84
|
||||
0x8002D2A0 0x8002D2BC src/func_8002D2A0.c
|
||||
0x8002D2BC 0x8002D2D4 src/func_8002D2BC.c
|
||||
0x8002DEB4 0x8002DF1C src/func_8002DEB4.c
|
||||
0x8002E7C4 0x8002E7E4 src/func_8002E7C4.c
|
||||
0x8002F2F8 0x8002F300 src/func_8002F2F8.c
|
||||
0x8002F404 0x8002F450 src/func_8002F404.c
|
||||
0x80030358 0x80030390 src/func_80030358.c
|
||||
0x800321EC 0x800321F8 src/func_800321EC.c
|
||||
0x80036308 0x80036328 src/func_80036308.c
|
||||
@@ -64,12 +76,21 @@
|
||||
0x8003B2F0 0x8003B320 src/func_8003B2F0.c
|
||||
0x80042088 0x80042090 src/func_80042088.c
|
||||
0x80042D64 0x80042D88 src/func_80042D64.c
|
||||
0x80043D8C 0x80043DC4 src/func_80043D8C.c
|
||||
0x8004C060 0x8004C090 src/func_8004C060.c
|
||||
0x8004C090 0x8004C0AC src/func_8004C090.c
|
||||
0x8004C0AC 0x8004C0F0 src/func_8004C0AC.c
|
||||
0x8004C0F0 0x8004C110 src/func_8004C0F0.c
|
||||
0x8004CEEC 0x8004CF0C src/func_8004CEEC.c
|
||||
0x800516E0 0x800516FC src/func_800516E0.c
|
||||
0x8005182C 0x80051864 src/func_8005182C.c
|
||||
0x80052C98 0x80052CAC src/func_80052C98.c
|
||||
0x80057DFC 0x80057E04 src/func_80057DFC.c
|
||||
0x80058288 0x800582AC src/func_80058288.c
|
||||
0x8005E3D0 0x8005E3F4 src/func_8005E3D0.c
|
||||
0x80065B6C 0x80065B8C src/func_80065B6C.c
|
||||
0x800681A4 0x800681E0 src/func_800681A4.c
|
||||
0x800681E0 0x8006821C src/func_800681E0.c
|
||||
0x80068470 0x8006848C src/func_80068470.c
|
||||
0x80068F40 0x80068F6C src/func_80068F40.c
|
||||
0x80068F98 0x80068FA8 src/func_80068F98.c
|
||||
@@ -80,9 +101,12 @@
|
||||
0x8006F944 0x8006F95C src/func_8006F944.c
|
||||
0x8006FA78 0x8006FAA0 src/func_8006FA78.c
|
||||
0x8007049C 0x800704BC src/func_8007049C.c
|
||||
0x8007C4EC 0x8007C524 src/func_8007C4EC.c
|
||||
0x8007DC40 0x8007DC4C src/func_8007DC40.c
|
||||
0x8007DF00 0x8007DF34 src/func_8007DF00.c
|
||||
0x800827A8 0x800827C4 src/func_800827A8.c
|
||||
0x80082914 0x80082944 src/func_80082914.c
|
||||
0x80083440 0x80083470 src/func_80083440.c
|
||||
0x80083504 0x8008352C src/func_80083504.c
|
||||
0x8008352C 0x8008355C src/func_8008352C.c
|
||||
0x80085B80 0x80085B90 src/func_80085B80.c
|
||||
@@ -96,30 +120,44 @@
|
||||
0x8008B8E4 0x8008B8F4 src/func_8008B8E4.c
|
||||
0x8008B8F4 0x8008B910 src/func_8008B8F4.c
|
||||
0x8008F4A0 0x8008F4AC src/func_8008F4A0.c
|
||||
0x8008F4F4 0x8008F508 src/func_8008F4F4.c
|
||||
0x80090048 0x80090058 src/func_80090048.c
|
||||
0x80090B64 0x80090B7C src/func_80090B64.c
|
||||
0x800912D4 0x800912FC src/func_800912D4.c
|
||||
0x800912FC 0x8009132C src/func_800912FC.c
|
||||
0x80092088 0x800920BC src/func_80092088.c
|
||||
0x800920BC 0x800920DC src/func_800920BC.c
|
||||
0x800920DC 0x80092104 src/func_800920DC.c
|
||||
0x80092130 0x8009214C src/func_80092130.c
|
||||
0x800943C4 0x800943E0 src/func_800943C4.c
|
||||
0x80099E14 0x80099E34 src/func_80099E14.c
|
||||
0x8009D8A0 0x8009D8E0 src/func_8009D8A0.c
|
||||
0x8009E8D0 0x8009E95C src/func_8009E8D0.c
|
||||
0x8009F0E8 0x8009F120 src/func_8009F0E8.c
|
||||
0x800A2F20 0x800A2F44 src/func_800A2F20.c
|
||||
0x800A45E0 0x800A466C src/func_8009E8D0.c
|
||||
0x800A74BC 0x800A74D0 src/func_800A74BC.c
|
||||
0x800A8B48 0x800A8B8C src/func_800A8B48.c
|
||||
0x800AA56C 0x800AA59C src/func_800AA56C.c
|
||||
0x800ACC00 0x800ACC20 src/func_800ACC00.c
|
||||
0x800AE0F4 0x800AE10C src/func_800AE0F4.c
|
||||
0x800AF1FC 0x800AF20C src/func_800AF1FC.c
|
||||
0x800B0E30 0x800B0E64 src/func_800B0E30.c
|
||||
0x800B24EC 0x800B2534 src/func_800B24EC.c
|
||||
0x800B3474 0x800B34A4 src/func_800B3474.c
|
||||
0x800B5AF0 0x800B5AF8 src/func_80042088.c
|
||||
0x800B6BDC 0x800B6C14 src/func_800B6BDC.c
|
||||
0x800B7230 0x800B7264 src/func_800B7230.c
|
||||
0x800BBDEC 0x800BBDF8 src/func_800BBDEC.c
|
||||
0x800C5C84 0x800C5CBC src/func_800C5C84.c
|
||||
0x800F3160 0x800F316C src/func_800F3160.c maspsx=off
|
||||
0x800F3E70 0x800F3E88 src/func_800F3E70.c
|
||||
0x800F5200 0x800F521C src/func_800F5200.c
|
||||
0x800F5B40 0x800F5B70 src/func_800F5B40.c
|
||||
0x800F6570 0x800F6594 src/func_800F6570.c
|
||||
0x800F75D0 0x800F760C src/func_800F75D0.c
|
||||
0x800F7A84 0x800F7A94 src/func_800F7A84.c
|
||||
0x800F7A94 0x800F7AAC src/func_800F7A94.c
|
||||
0x800F7FB4 0x800F7FD8 src/func_800F7FB4.c
|
||||
0x800F8294 0x800F82A4 src/func_800F8294.c
|
||||
0x800F8ACC 0x800F8ADC src/func_800F8ACC.c
|
||||
@@ -141,30 +179,49 @@
|
||||
0x800FE86C 0x800FE878 src/func_800FE86C.c
|
||||
0x800FEEB8 0x800FEED0 src/func_800FEEB8.c
|
||||
0x800FEFFC 0x800FF008 src/func_800FEFFC.c
|
||||
0x800FF6A4 0x800FF6DC src/func_800FF6A4.c
|
||||
0x80100318 0x80100334 src/func_80100318.c
|
||||
0x80100964 0x8010097C src/func_80100964.c
|
||||
0x8010097C 0x80100998 src/func_8010097C.c
|
||||
0x80101244 0x8010128C src/func_80101244.c
|
||||
0x80101CAC 0x80101CDC src/func_80101CAC.c
|
||||
0x801027CC 0x801027F8 src/func_801027CC.c
|
||||
0x80102B10 0x80102B2C src/func_80102B10.c maspsx=off
|
||||
0x80102FD4 0x80102FE0 src/func_80102FD4.c
|
||||
0x80103A94 0x80103AA0 src/func_80103A94.c
|
||||
0x80103B54 0x80103B60 src/func_80103B54.c
|
||||
0x80103B60 0x80103B6C src/func_80103B60.c
|
||||
0x80103B6C 0x80103B8C src/func_80103B6C.c
|
||||
0x80103C7C 0x80103CA0 src/func_800F7FB4.c
|
||||
0x80103F84 0x80103FA8 src/func_800F7FB4.c
|
||||
0x80103FCC 0x80103FDC src/func_80103FCC.c
|
||||
0x80103FEC 0x80103FFC src/func_80103FEC.c
|
||||
0x80104C38 0x80104C60 src/func_80104C38.c maspsx=off
|
||||
0x80104C6C 0x80104CA0 src/func_80104C6C.c maspsx=off
|
||||
0x8010513C 0x80105148 src/func_8010513C.c
|
||||
0x8010543C 0x80105474 src/func_8010543C.c
|
||||
0x80105B34 0x80105B54 src/func_80105B34.c
|
||||
0x80105B54 0x80105B68 src/func_80105B54.c
|
||||
0x80105B68 0x80105B88 src/func_80105B68.c
|
||||
0x80105B88 0x80105BA8 src/func_80105B88.c
|
||||
0x80105BA8 0x80105BC8 src/func_80105BA8.c
|
||||
0x80105BC8 0x80105BDC src/func_80105BC8.c
|
||||
0x80107A94 0x80107AA0 src/func_80107A94.c
|
||||
0x80107AA0 0x80107AE0 src/func_80107AA0.c
|
||||
0x80107AE0 0x80107B20 src/func_80107AE0.c
|
||||
0x80107B20 0x80107B38 src/func_80107B20.c
|
||||
0x80107E68 0x80107E74 src/func_80107E68.c
|
||||
0x80107E74 0x80107E84 src/func_80107E74.c
|
||||
0x80107E84 0x80107E90 src/func_80107E84.c
|
||||
0x80107FAC 0x80107FDC src/func_80107FAC.c
|
||||
0x80107FDC 0x80108010 src/func_80107FDC.c
|
||||
0x80108024 0x80108034 src/func_80108024.c
|
||||
0x80108690 0x801086AC src/func_80108690.c
|
||||
0x80108710 0x80108734 src/func_80108710.c
|
||||
0x80109300 0x80109314 src/func_80109300.c
|
||||
0x80109314 0x80109338 src/func_800F8F9C.c
|
||||
0x80109778 0x80109790 src/func_80109778.c
|
||||
0x80109F38 0x80109F48 src/func_80109F38.c
|
||||
0x8010A888 0x8010A8B0 src/func_8010A888.c
|
||||
0x8010A8B0 0x8010A8D8 src/func_8010A8B0.c
|
||||
0x8010AAA0 0x8010AABC src/func_8010AAA0.c
|
||||
|
||||
|
@@ -377,4 +377,8 @@ func_80024668 0x80024668
|
||||
func_8002F404 0x8002F404
|
||||
func_800695D8 0x800695D8
|
||||
func_800F4098 0x800F4098
|
||||
g_80122068 0x80122068 gp
|
||||
g_80122158 0x80122158 gp
|
||||
g_80122354 0x80122354
|
||||
g_80122738 0x80122738 gp
|
||||
g_8012277C 0x8012277C gp
|
||||
|
||||
|
+47
-8
@@ -32,9 +32,11 @@
|
||||
* LIMITS. Only the operations listed below are covered, and only for the
|
||||
* register numbers observed here. The control-register numbering used by the
|
||||
* original is NOT the 0..31 "textbook" GTE control layout: this executable
|
||||
* writes DQB at $28, OFX/OFY at $24/$25, H at $26 and LR1LR2/LR3LG1/LG2LG3 at
|
||||
* $13/$14/$15, so the numbers below are the evidence, not a derivation. The
|
||||
* file is not an attempt to reconstruct Sony's gtemac.h.
|
||||
* writes DQB at $28, DQA at $27, OFX/OFY at $24/$25, H at $26, the far-colour
|
||||
* triple RFC/GFC/BFC at $21/$22/$23, and the light-direction triple
|
||||
* LR1LR2/LR3LG1/LG2LG3 at $13/$14/$15, so the numbers below are the evidence,
|
||||
* not a derivation. The file is not an attempt to reconstruct Sony's
|
||||
* gtemac.h.
|
||||
*/
|
||||
|
||||
#ifndef SF3_GTEMAC_H
|
||||
@@ -61,17 +63,54 @@
|
||||
|
||||
/* --- control registers (ctc2/cfc2) --------------------------------------- */
|
||||
|
||||
/* --- control registers (ctc2/cfc2) --------------------------------------- */
|
||||
|
||||
/* Rotation/translation matrix (UNSHIFTED standard map, 0x80101CAC: five
|
||||
* `ctc2`s from five loads). The executable uses the standard 0..5 numbers
|
||||
* here and standard+7 from RBK onward, so the 0x00-0x05 block is the only
|
||||
* unshifted control range. The four RT words hold the 3x3 rotation sorted
|
||||
* by the standard naming (RT1RT2 = row1col1,row1col2, ...):
|
||||
* $0 = RT1RT2 $1 = RT3RT21 $2 = RT22RT23
|
||||
* $3 = RT31RT32 $4 = RT33
|
||||
* $5 = TRX (TRY/TRZ unobserved; assumed from the standard order) */
|
||||
#define gte_ldRT1RT2(v) __asm__ volatile ("ctc2 %0,$0" : : "r"(v))
|
||||
#define gte_ldRT3RT21(v) __asm__ volatile ("ctc2 %0,$1" : : "r"(v))
|
||||
#define gte_ldRT22RT23(v) __asm__ volatile ("ctc2 %0,$2" : : "r"(v))
|
||||
#define gte_ldRT31RT32(v) __asm__ volatile ("ctc2 %0,$3" : : "r"(v))
|
||||
#define gte_ldRT33(v) __asm__ volatile ("ctc2 %0,$4" : : "r"(v))
|
||||
|
||||
/* Rotation/light source and colour-matrix control words.
|
||||
* $13/$14/$15 = LR1LR2 / LR3LG1 / LG2LG3 (0x8001AE3C, three `ctc2`s)
|
||||
* The executable's control-register numbering is the standard map shifted
|
||||
* by +7 (all byte-proved): RBK/GBK/BBK at $13/$14/$15, the light-matrix
|
||||
* triple plus LB1LB2/LB3 at $16..$20, the far-colour triple RFC/GFC/BFC at
|
||||
* $21/$22/$23, OFX/OFY at $24/$25, H at $26, DQA at $27, DQB at $28. The
|
||||
* data-register space is the standard map unshifted (see above).
|
||||
* $13/$14/$15 = RBK / GBK / BBK (0x8001AE3C, three `ctc2`s)
|
||||
* $16/$17/$18 = LR1LR2 / LR3LG1 / LG2LG3 (0x80102FE4, with $19/$20 =
|
||||
* $19/$20 = LB1LB2 / LB3 LB1LB2/LB3 — same body)
|
||||
* $21/$22/$23 = RFC / GFC / BFC (0x80103B6C, each shifted left
|
||||
* by 4 before the write)
|
||||
* $24/$25 = OFX / OFY (0x80109778, after `sll ...,16`)
|
||||
* $26 = H (0x80103A94, `cfc2 v0,$26`)
|
||||
* $26 = H (0x80103A94 `cfc2 v0,$26`,
|
||||
* 0x80102FD4 `ctc2 a0,$26`)
|
||||
* $27 = DQA (0x80103B54, `ctc2 a0,$27`)
|
||||
* $28 = DQB (0x80103B60, `ctc2 a0,$28`) */
|
||||
#define gte_ldLR1LR2(v) __asm__ volatile ("ctc2 %0,$13" : : "r"(v))
|
||||
#define gte_ldLR3LG1(v) __asm__ volatile ("ctc2 %0,$14" : : "r"(v))
|
||||
#define gte_ldLG2LG3(v) __asm__ volatile ("ctc2 %0,$15" : : "r"(v))
|
||||
#define gte_ldRBK(v) __asm__ volatile ("ctc2 %0,$13" : : "r"(v))
|
||||
#define gte_ldGBK(v) __asm__ volatile ("ctc2 %0,$14" : : "r"(v))
|
||||
#define gte_ldBBK(v) __asm__ volatile ("ctc2 %0,$15" : : "r"(v))
|
||||
#define gte_ldLR1LR2(v) __asm__ volatile ("ctc2 %0,$16" : : "r"(v))
|
||||
#define gte_ldLR3LG1(v) __asm__ volatile ("ctc2 %0,$17" : : "r"(v))
|
||||
#define gte_ldLG2LG3(v) __asm__ volatile ("ctc2 %0,$18" : : "r"(v))
|
||||
#define gte_ldLB1LB2(v) __asm__ volatile ("ctc2 %0,$19" : : "r"(v))
|
||||
#define gte_ldLB3(v) __asm__ volatile ("ctc2 %0,$20" : : "r"(v))
|
||||
#define gte_ldRFC(v) __asm__ volatile ("ctc2 %0,$21" : : "r"(v))
|
||||
#define gte_ldGFC(v) __asm__ volatile ("ctc2 %0,$22" : : "r"(v))
|
||||
#define gte_ldBFC(v) __asm__ volatile ("ctc2 %0,$23" : : "r"(v))
|
||||
#define gte_ldOFX(v) __asm__ volatile ("ctc2 %0,$24" : : "r"(v))
|
||||
#define gte_ldOFY(v) __asm__ volatile ("ctc2 %0,$25" : : "r"(v))
|
||||
#define gte_ldH(v) __asm__ volatile ("ctc2 %0,$26" : : "r"(v))
|
||||
#define gte_stH(r) __asm__ volatile ("cfc2 %0,$26" : "=r"(r))
|
||||
#define gte_ldDQA(v) __asm__ volatile ("ctc2 %0,$27" : : "r"(v))
|
||||
#define gte_ldDQB(v) __asm__ volatile ("ctc2 %0,$28" : : "r"(v))
|
||||
|
||||
/* --- commands ------------------------------------------------------------ */
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
/*
|
||||
* func_80012D8C — 48 bytes at 0x80012D8C..0x80012DBC
|
||||
*
|
||||
* The three-value sibling of func_80012DBC: same header shape, a different type
|
||||
* word and command word, and one more caller value stored past the end of the
|
||||
* other routine's layout.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* li v0,4
|
||||
* sh v0,6(a0) ; p->type_06 = 4
|
||||
* lui v0,0x300 ; 0x03000000
|
||||
* sw v0,8(a0) ; p->word_08 = 0x03000000
|
||||
* lui v0,0x4000 ; 0x40000000
|
||||
* or a1,a1,v0 ; flags |= 0x40000000
|
||||
* sw zero,0(a0) ; p->word_00 = 0
|
||||
* sh zero,4(a0) ; p->half_04 = 0
|
||||
* sw a1,12(a0) ; p->word_0c = flags
|
||||
* sw a2,16(a0) ; p->word_10 = value
|
||||
* jr ra
|
||||
* sw a3,20(a0) ; p->word_14 = extra (delay slot)
|
||||
*
|
||||
* LIMITS: the field offsets and widths are hypotheses read from the instruction
|
||||
* shape, as are the parameter types. The constants 4, 0x03000000 and 0x40000000
|
||||
* are facts about this executable, not about the compiler; what they mean is
|
||||
* unknown and is not guessed here. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
void func_80012D8C(char *base, int flags, int value, int extra) {
|
||||
*(short *)(base + 6) = 4;
|
||||
*(int *)(base + 8) = 0x03000000;
|
||||
flags |= 0x40000000;
|
||||
*(int *)(base + 0) = 0;
|
||||
*(short *)(base + 4) = 0;
|
||||
*(int *)(base + 12) = flags;
|
||||
*(int *)(base + 16) = value;
|
||||
*(int *)(base + 20) = extra;
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
/*
|
||||
* func_80012DBC — 44 bytes at 0x80012DBC..0x80012DE8
|
||||
*
|
||||
* Initialises a fixed-layout structure: a 16-bit type word, a 32-bit command
|
||||
* word, two zeroed header fields, a 32-bit flags word built by OR-ing a constant
|
||||
* into the caller's second argument, and one caller value. The store order below
|
||||
* is the order the original emits; the first two stores are hoisted ahead of the
|
||||
* header clear.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* li v0,4
|
||||
* sh v0,6(a0) ; p->type_06 = 4
|
||||
* lui v0,0x200 ; 0x02000000
|
||||
* sw v0,8(a0) ; p->word_08 = 0x02000000
|
||||
* lui v0,0x6800 ; 0x68000000
|
||||
* or a1,a1,v0 ; flags |= 0x68000000
|
||||
* sw zero,0(a0) ; p->word_00 = 0
|
||||
* sh zero,4(a0) ; p->half_04 = 0
|
||||
* sw a1,12(a0) ; p->word_0c = flags
|
||||
* jr ra
|
||||
* sw a2,16(a0) ; p->word_10 = value (delay slot)
|
||||
*
|
||||
* LIMITS: the field offsets and widths are hypotheses read from the instruction
|
||||
* shape, as are the parameter types. The constants 4, 0x02000000 and 0x68000000
|
||||
* are facts about this executable, not about the compiler; what they mean is
|
||||
* unknown and is not guessed here. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
void func_80012DBC(char *base, int flags, int value) {
|
||||
*(short *)(base + 6) = 4;
|
||||
*(int *)(base + 8) = 0x02000000;
|
||||
flags |= 0x68000000;
|
||||
*(int *)(base + 0) = 0;
|
||||
*(short *)(base + 4) = 0;
|
||||
*(int *)(base + 12) = flags;
|
||||
*(int *)(base + 16) = value;
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
/* func_800160E8 — 0x800160E8..0x80016110 (40 bytes).
|
||||
*
|
||||
* Original words:
|
||||
* 0x24020001 li v0,0x1
|
||||
* 0xAC820134 sw v0,0x134(a0)
|
||||
* 0x248200F8 addiu v0,a0,0xf8
|
||||
* 0xAC85009C sw a1,0x9c(a0)
|
||||
* 0xAC820138 sw v0,0x138(a0)
|
||||
* 0xAC800104 sw zero,0x104(a0)
|
||||
* 0x8CA20018 lw v0,0x18(a1)
|
||||
* 0xAC8500F8 sw a1,0xf8(a0)
|
||||
* 0x03E00008 jr ra
|
||||
* 0xACA200FC _sw v0,0xfc(a0) (delay slot)
|
||||
*
|
||||
* A list-node initialiser: it seeds a flag at 0x134, points 0x138 at its own
|
||||
* inline link field at 0xf8, stores the incoming node at 0x9c and at 0xf8,
|
||||
* clears 0x104, and copies the node's field 0x18 into its own 0xfc. The final
|
||||
* store lands in the `jr ra` delay slot.
|
||||
*
|
||||
* The `addiu v0,a0,0xf8` is emitted early and its store deferred until after
|
||||
* the 0x9c store, so the source order is not the emission order. Attempt 1 wrote
|
||||
* the two stores in ascending offset order (0x138 before 0x9c) and produced them
|
||||
* in that order, 12 differing bytes; swapping them to 0x9c-then-0x138 is what
|
||||
* matches. Reading `node->0x18` into a named local before the 0xf8 store is what
|
||||
* makes cc1 schedule that load above the store, as the original does — writing
|
||||
* the load inline in the final store leaves it after the store.
|
||||
*
|
||||
* LIMITS: every displacement is read from the bytes and the field names are
|
||||
* hypotheses; in particular that 0x138 points at 0xf8 (rather than merely
|
||||
* holding `self + 0xf8` as a value) is inferred from the self-relative
|
||||
* addiu and is not proved — a ring or list convention could explain it another
|
||||
* way. Nothing proves the two descriptors' node type is the same object type.
|
||||
*/
|
||||
|
||||
void func_800160E8(char *self, int *node)
|
||||
{
|
||||
int value;
|
||||
|
||||
*(int *)(self + 0x134) = 1;
|
||||
*(int *)(self + 0x9c) = (int)node;
|
||||
*(int *)(self + 0x138) = (int)(self + 0xf8);
|
||||
*(int *)(self + 0x104) = 0;
|
||||
value = *(int *)((char *)node + 0x18);
|
||||
*(int *)(self + 0xf8) = (int)node;
|
||||
*(int *)(self + 0xfc) = value;
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
/* func_80016120 — 0x80016120..0x80016158 (56 bytes).
|
||||
*
|
||||
* Original words:
|
||||
* lbu v0,0x0(a1) nop sb v0,0x10(a0) sb v0,0xec(a0)
|
||||
* lbu v0,0x1(a1) nop sb v0,0x11(a0) sb v0,0xed(a0)
|
||||
* lbu v0,0x2(a1) nop sb v0,0x12(a0) sb v0,0xee(a0)
|
||||
* jr ra
|
||||
* _clear v0 (delay slot)
|
||||
*
|
||||
* A fully unrolled three-byte copy into two parallel fields: each source byte
|
||||
* lands at 0x10+i and again at 0xec+i, with the second destination exactly 0xdc
|
||||
* bytes past the first in every case. The single `lbu` per byte is reused for
|
||||
* both stores rather than re-loaded.
|
||||
*
|
||||
* The unrolled shape is the evidence that the loop was written out (or was
|
||||
* unrolled by cc1 — which `-O2` alone does not do for three iterations, so the
|
||||
* explicit form is the one reproduced here). The `clear v0` in the `jr ra` delay
|
||||
* slot means the routine returns a zero constant rather than being `void`: a void
|
||||
* function would leave the last loaded byte in v0 (cookbook finding 13's scratch
|
||||
* rule, applied with the sign reversed).
|
||||
*
|
||||
* LIMITS: the two destination bases (0x10 and 0xec) and the three-byte count are
|
||||
* read from the displacements; that the two fields are parallel copies of one
|
||||
* three-byte value is inferred from the constant 0xdc stride and not proved.
|
||||
* The byte order of the source is assumed to be the natural one.
|
||||
*
|
||||
* ATTEMPT 1 (recorded): writing the load inline in both stores
|
||||
* (`dst[0x10] = src[0]; dst[0xec] = src[0];`) gives the original's *shape* —
|
||||
* `lbu` / `nop` / `sb` / `sb` per byte, the load-delay `nop` where the original
|
||||
* has it — but cc1 **re-loads** the byte for the second store (the store through
|
||||
* `dst` may alias `src`), so it is 80 bytes against 56: two `lbu` per byte.
|
||||
*
|
||||
* ATTEMPT 2 (recorded): declaring all three bytes as locals up front gives one
|
||||
* `lbu` per byte but **44 bytes against 56** — cc1 hoists all three loads and
|
||||
* interleaves the stores, which removes the load-use hazard and therefore the
|
||||
* three `nop`s the original carries.
|
||||
*
|
||||
* So the target needs one load per byte *and* a `nop` between load and first
|
||||
* store: the loads must not be hoisted, and the byte must not be re-loaded.
|
||||
* Attempt 3 reuses a single `c` local but re-assigns it between the byte groups,
|
||||
* so each load is a separate statement whose stores immediately follow.
|
||||
*/
|
||||
|
||||
int func_80016120(char *dst, char *src)
|
||||
{
|
||||
unsigned char c;
|
||||
|
||||
c = src[0];
|
||||
dst[0x10] = c;
|
||||
dst[0xec] = c;
|
||||
c = src[1];
|
||||
dst[0x11] = c;
|
||||
dst[0xed] = c;
|
||||
c = src[2];
|
||||
dst[0x12] = c;
|
||||
dst[0xee] = c;
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,52 @@
|
||||
/*
|
||||
* func_80016174 — 36 bytes at 0x80016174..0x80016198
|
||||
*
|
||||
* Clears a strided run of 32-bit slots: fourteen words starting at 0x80122908,
|
||||
* spaced 172 bytes apart, walking downwards. The loop bound is the byte offset
|
||||
* itself, counting from 2236 down to 0 inclusive.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* li v0,2236 ; offset = 0x8bc
|
||||
* L: lui at,0x8012 ; %hi of the symbol base
|
||||
* addu at,at,v0 ; at = base + offset (base first)
|
||||
* sw zero,10504(at) ; *(int *)(at + 0x2908) = 0 ; %lo as displacement
|
||||
* addiu v0,v0,-172 ; offset -= 0xac
|
||||
* bgez v0,0x80016174 ; loop while offset >= 0
|
||||
* nop ; no independent instruction for the slot
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* The base must be written as an address-named SYMBOL, not as a literal. This
|
||||
* was measured, not assumed — the three literal spellings below all emit
|
||||
* `addu at,v0,at` (variable first) and differ in exactly one byte, the register
|
||||
* field of the `addu`:
|
||||
*
|
||||
* *(int *)(0x80122908 + offset) -> addu at,v0,at (1 byte off)
|
||||
* *(int *)(offset + 0x80122908) -> addu at,v0,at (byte-identical to
|
||||
* the previous spelling: cc1
|
||||
* canonicalises `+` operand order,
|
||||
* so finding 22's swap lever does
|
||||
* NOT apply to a literal base)
|
||||
* *(int *)((char *)0x80122908 + offset)-> addu at,v0,at (1 byte off)
|
||||
*
|
||||
* With a symbol base, `lui %hi` + `addu at,at,v0` + `sw %lo(at)` is emitted and
|
||||
* the region matches. This is consistent with finding 22's stated limit — the
|
||||
* commutative-operand trigger does not generalise — and with finding 5's
|
||||
* literal-versus-symbol distinction; it is recorded as a refinement, not as a
|
||||
* new rule about the compiler.
|
||||
*
|
||||
* LIMITS: the base address, the start offset (2236), the stride (-172) and the
|
||||
* clear value are facts about this executable, not about the compiler. The
|
||||
* element type (`int`) is a hypothesis from the `sw` width, and the two `nop`s
|
||||
* are scheduling fills rather than source. What the cleared slots mean is
|
||||
* unknown and is not guessed here. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
extern char D_80122908[];
|
||||
|
||||
void func_80016174(void) {
|
||||
int offset;
|
||||
|
||||
for (offset = 2236; offset >= 0; offset -= 172)
|
||||
*(int *)((char *)D_80122908 + offset) = 0;
|
||||
}
|
||||
@@ -0,0 +1,73 @@
|
||||
/*
|
||||
* func_80016198 — 72 bytes at 0x80016198..0x800161E0
|
||||
*
|
||||
* Leaf routine that returns the address of the first free entry of a 14-entry
|
||||
* table of 0xac-byte records, or null when the table is full.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* addu a0,zero,zero 00002021 i = 0
|
||||
* lui a1,0x8012 3c058012 \
|
||||
* addiu a1,a1,0x2908 24a52908 / a1 = 0x80122908 (D_80122908)
|
||||
* addu v1,zero,zero 00001821 off = 0
|
||||
* 0x800161a8:
|
||||
* lui at,0x8012 3c018012 \
|
||||
* addu at,at,v1 00200821 / at = 0x80120000 + off
|
||||
* lw v0,0x2908(at) 8c220908 v0 = *(int *)(D_80122908 + off)
|
||||
* nop 00000000 load-delay slot
|
||||
* beq v0,zero,0x800161d8 10400007 if (v0 == 0) goto epilogue
|
||||
* move v0,a1 00a01021 v0 = a1 (delay slot)
|
||||
* addiu a1,a1,0xac 24a500ac a1 += 0xac
|
||||
* addiu a0,a0,0x1 24840001 i++
|
||||
* slti v0,a0,0xe 2882000e v0 = (i < 14) signed
|
||||
* bne v0,zero,0x800161a8 1440fff4 if (v0) loop
|
||||
* addiu v1,v1,0xac 246300ac off += 0xac (delay slot)
|
||||
* 0x800161d8:
|
||||
* jr ra 03e00008
|
||||
* nop 00000000 (delay slot)
|
||||
*
|
||||
* cc1 keeps **two** induction variables for the same index: the counter `i`
|
||||
* (`addiu a0,a0,1`, bound-tested by `slti` against 14) and the byte offset `off`
|
||||
* (`addiu v1,v1,0xac`, used for the address). The return value is a third: the
|
||||
* *address* variable a1, materialised once as `lui`+`addiu` and incremented by
|
||||
* 0xac per iteration, moved into v0 in the branch delay slot — so the found path
|
||||
* returns the element address.
|
||||
*
|
||||
* That separation is load-bearing. Deriving the read address from the same
|
||||
* expression as the returned pointer lets cc1 common-subexpression the two into a
|
||||
* single walked pointer, which loses both the explicit `lui at,%hi` / `addu` /
|
||||
* `lw %lo(at)` address form and the separate `off` (measured: 56 bytes, and a
|
||||
* variant that kept the pointer but folded the read gave 64 against the original's
|
||||
* 72). The source therefore reads through `D_80122908 + off` with a variable `off`
|
||||
* while `p` is walked independently, and the declaration order fixes the two
|
||||
* zeroing instructions: `i` is cleared before `off`.
|
||||
*
|
||||
* The table base is materialised as `lui`+`addiu` (the linker-resolved symbol
|
||||
* form, cookbook finding 4), and the indexed *read* uses the explicit
|
||||
* `lui at,%hi` / `addu at,at,index` / `lw %lo(at)` form because the index is added
|
||||
* between the halves. The 14-entry bound is a signed compare, so the counter is a
|
||||
* signed `int`.
|
||||
*
|
||||
* LIMITS: the function name, the record size, the table length of 14 and the claim
|
||||
* that a zero first word marks a free entry are hypotheses; only the bytes are
|
||||
* evidence. The element field read is 32-bit. The return type is written as
|
||||
* `int *` because the returned value is an address; whether the original used a
|
||||
* struct pointer is not recoverable.
|
||||
*/
|
||||
|
||||
extern char D_80122908[];
|
||||
|
||||
int *func_80016198(void)
|
||||
{
|
||||
int i = 0;
|
||||
int *p = (int *)D_80122908;
|
||||
int off = 0;
|
||||
|
||||
for (; i < 14; i++) {
|
||||
if (*(int *)(D_80122908 + off) == 0)
|
||||
return p;
|
||||
p = (int *)((char *)p + 0xac);
|
||||
off += 0xac;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
/*
|
||||
* func_8001761C — 68 bytes at 0x8001761C..0x80017660
|
||||
*
|
||||
* Leaf routine that scans a 60-entry table of 0x38-byte records and clears the
|
||||
* first word of every record whose second word matches the argument.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* addu a1,zero,zero 00002821 i = 0
|
||||
* addu v1,zero,zero 00001821 off = 0
|
||||
* 0x80017624:
|
||||
* lui at,0x8012 3c018012 \
|
||||
* addu at,at,v1 00200821 / at = 0x80120000 + off
|
||||
* lw v0,0x4cc4(at) 8c2204cc4 v0 = *(int *)(D_80124CC0 + off + 4)
|
||||
* nop 00000000 load-delay slot
|
||||
* bne v0,a0,0x80017648 14440007 if (v0 != a0) goto the increment
|
||||
* nop 00000000 (delay slot)
|
||||
* lui at,0x8012 3c018012 \
|
||||
* addu at,at,v1 00200821 / at = 0x80120000 + off
|
||||
* sw zero,0x4cc0(at) ac2004cc0 *(int *)(D_80124CC0 + off) = 0
|
||||
* 0x80017648:
|
||||
* addiu a1,a1,0x1 24a50001 i++
|
||||
* slti v0,a1,0x3c 28a2003c v0 = (i < 60) signed
|
||||
* bne v0,zero,0x80017624 1440fff6 if (v0) loop
|
||||
* addiu v1,v1,0x38 24630038 off += 0x38 (delay slot)
|
||||
* 0x80017658:
|
||||
* jr ra 03e00008
|
||||
* nop 00000000 (delay slot)
|
||||
*
|
||||
* The indexed symbol access is materialised as `lui at,%hi` / `addu at,at,index`
|
||||
* / `lw v0,%lo(at)` — the **explicit** address form, not the macro form (cookbook
|
||||
* finding 1's other mode), because the access has an index added between the high
|
||||
* and low halves. The low halves appear as the raw displacements 0x4cc4 and
|
||||
* 0x4cc0, and the two accesses each recompute the high half: that only happens if
|
||||
* the source names **two different symbols** rather than deriving one address from
|
||||
* the other. cc1 otherwise common-subexpressions the pair into a single walked
|
||||
* pointer (measured: 68 bytes with 40 differing bytes, first at 0x8001761D). So the
|
||||
* compare reads `D_80124CC4 + off` and the store writes `D_80124CC0 + off`, with a
|
||||
* shared `off`.
|
||||
*
|
||||
* cc1 keeps two induction variables: the counter `i` and the byte offset `off`,
|
||||
* incremented by 0x38 in the branch delay slot. The bound test is `slti` (signed),
|
||||
* so the counter is a signed `int` and the literal is 60. Both variables are
|
||||
* initialised to zero, and the **order** of the two zeroing instructions is
|
||||
* evidence: the original clears `a1` (the counter) before `v1` (the offset), which
|
||||
* is why the source initialises `i` before `off` and the `for` has an empty init.
|
||||
*
|
||||
* LIMITS: the function name, the record size, the table length of 60 and the field
|
||||
* meanings are hypotheses; only the bytes are evidence. Both accesses are 32-bit.
|
||||
* The two tables are written as `char[]` so the byte offsets stay explicit; no
|
||||
* array or struct type is claimed.
|
||||
*/
|
||||
|
||||
extern char D_80124CC0[];
|
||||
extern char D_80124CC4[];
|
||||
|
||||
void func_8001761C(int a0)
|
||||
{
|
||||
int i = 0;
|
||||
int off = 0;
|
||||
|
||||
for (; i < 60; i++) {
|
||||
if (*(int *)(D_80124CC4 + off) == a0)
|
||||
*(int *)(D_80124CC0 + off) = 0;
|
||||
off += 0x38;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
/* func_800182D4 — 0x800182D4..0x800182F4 (32 bytes).
|
||||
*
|
||||
* Original words:
|
||||
* 0x10800004 beq a0,zero,0x800182E8
|
||||
* 0x00001021 _clear v0 (delay slot)
|
||||
* 0x8C83000C lw v1,0xc(a0)
|
||||
* 0x080060BB j 0x800182EC
|
||||
* 0xAC830000 _sw v1,0x0(a1) (delay slot)
|
||||
* 0x24020001 li v0,0x1 <- 0x800182E8
|
||||
* 0x03E00008 jr ra <- 0x800182EC
|
||||
* 0x00000000 nop
|
||||
*
|
||||
* A guarded pointer copy with a boolean result. The success path clears v0 in
|
||||
* the branch delay slot and stores `src[3]` into `*dst`; the failure path
|
||||
* materialises 1 immediately before the shared return.
|
||||
*
|
||||
* The `sw` lands in the `j` delay slot and the `clear v0` in the `beq` slot
|
||||
* (cookbook finding 8). That v0 is cleared *before* the branch means the
|
||||
* success value is the fall-through constant, so `return 0` is the statement
|
||||
* after the store rather than an else-arm.
|
||||
*
|
||||
* LIMITS: `src[3]` (displacement 0xc) is a hypothesis about a 4-int header; the
|
||||
* routine may be copying any 4-byte field. Whether the failure result 1 means
|
||||
* "NULL argument" or something else is not observable from the bytes. Nothing
|
||||
* here proves the two pointers are different objects.
|
||||
*/
|
||||
|
||||
int func_800182D4(int *src, int *dst)
|
||||
{
|
||||
if (src == 0)
|
||||
return 1;
|
||||
*dst = src[3];
|
||||
return 0;
|
||||
}
|
||||
+17
-11
@@ -1,22 +1,28 @@
|
||||
/* func_8001AE3C — 0x8001AE3C..0x8001AE50 (20 bytes).
|
||||
*
|
||||
* Original words:
|
||||
* 0x48C46800 ctc2 a0,$13 LR1LR2
|
||||
* 0x48C57000 ctc2 a1,$14 LR3LG1
|
||||
* 0x48C67800 ctc2 a2,$15 LG2LG3
|
||||
* 0x48C46800 ctc2 a0,$13 RBK
|
||||
* 0x48C57000 ctc2 a1,$14 GBK
|
||||
* 0x48C67800 ctc2 a2,$15 BBK
|
||||
* 0x03E00008 jr ra
|
||||
* 0x00000000 nop
|
||||
*
|
||||
* Three GTE control-register writes through include/gtemac.h. Ghidra's PSX
|
||||
* loader collapses this trio into a single `ldbkdir a0,a1,a2` pseudo-instruction
|
||||
* and reports a 4-instruction body; the raw words above are the ground truth
|
||||
* (the object disassembles to exactly these three `ctc2`s).
|
||||
* Three GTE control-register writes through include/gtemac.h. Register numbers
|
||||
* $13/$14/$15 are the background-colour triple RBK/GBK/BBK under the
|
||||
* executable's control map (standard + 7, recorded in include/gtemac.h).
|
||||
* Ghidra's PSX loader collapses this trio into a single `ldbkdir a0,a1,a2`
|
||||
* pseudo-instruction and reports a 4-instruction body; the raw words above are
|
||||
* the ground truth (the object disassembles to exactly these three `ctc2`s).
|
||||
*
|
||||
* LIMITS: the parameter names are hypotheses; the register numbers are the
|
||||
* evidence, and the far-colour reading (RFC/GFC/BFC) was ruled out because
|
||||
* those sit at $21/$22/$23 (0x80103B6C).
|
||||
*/
|
||||
#include "../include/gtemac.h"
|
||||
|
||||
void func_8001AE3C(int lr1lr2, int lr3lg1, int lg2lg3)
|
||||
void func_8001AE3C(int rbk, int gbk, int bbk)
|
||||
{
|
||||
gte_ldLR1LR2(lr1lr2);
|
||||
gte_ldLR3LG1(lr3lg1);
|
||||
gte_ldLG2LG3(lg2lg3);
|
||||
gte_ldRBK(rbk);
|
||||
gte_ldGBK(gbk);
|
||||
gte_ldBBK(bbk);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
/*
|
||||
* func_800230E4 — 56 bytes at 0x800230E4..0x8002311C
|
||||
*
|
||||
* Copies three 32-bit values from one structure to another, scaling each left by
|
||||
* twelve bits on the way. The routine returns zero, which the original
|
||||
* materialises with `move v0,zero` in the `jr ra` delay slot.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lw v0,0(a0) / nop / sll v0,v0,0xc / sw v0,0(a1)
|
||||
* lw v0,4(a0) / nop / sll v0,v0,0xc / sw v0,4(a1)
|
||||
* lw v0,8(a0) / nop / sll v0,v0,0xc / sw v0,8(a1)
|
||||
* jr ra
|
||||
* move v0,zero ; return 0 (delay slot)
|
||||
*
|
||||
* The `nop` after each load is maspsx's load-delay fill.
|
||||
*
|
||||
* LIMITS: the element type (`int`) and the shift amount (12) are hypotheses read
|
||||
* from the instruction shape; the shift is a fact about this executable's fixed-
|
||||
* point convention, not about the compiler. The `return 0` is required — a `void`
|
||||
* body would not materialise `v0`. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
int func_800230E4(int *src, int *dst) {
|
||||
dst[0] = src[0] << 12;
|
||||
dst[1] = src[1] << 12;
|
||||
dst[2] = src[2] << 12;
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
/*
|
||||
* func_8002D1F8 — 52 bytes at 0x8002D1F8..0x8002D22C
|
||||
*
|
||||
* Leaf routine that indexes a global pointer table, null-checks the entry, and
|
||||
* reports whether the entry's field at +0xc is non-zero.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lui v0,0x8012 3c028012 \
|
||||
* lw v0,0x1c00(v0) 8c421c00 / v0 = *(int *)0x80121C00 (D_80121C00)
|
||||
* sll a0,a0,0x2 00042080 a0 *= 4
|
||||
* addu a0,a0,v0 00822021 a0 += table
|
||||
* lw a0,0x0(a0) 8c840000 a0 = table[a0]
|
||||
* nop 00000000 load-delay slot
|
||||
* beq a0,zero,0x8002d224 10800004 if (a0 == 0) goto epilogue
|
||||
* addu v0,zero,zero 00001021 v0 = 0 (delay slot)
|
||||
* lw v0,0xc(a0) 8c82000c v0 = *(int *)(a0 + 0xc)
|
||||
* nop 00000000 load-delay slot
|
||||
* sltu v0,zero,v0 0002102b v0 = (0 < v0)
|
||||
* 0x8002d224:
|
||||
* jr ra 03e00008
|
||||
* nop 00000000 (delay slot)
|
||||
*
|
||||
* The table base is a symbol load through the same register (cookbook finding
|
||||
* 2), so it is written as the named global `D_80121C00` and not as a literal
|
||||
* address (finding 5). The index is scaled by `sll` and added with the **index
|
||||
* first** (`addu a0,a0,v0`), which finding 22 records as the spelling
|
||||
* `*(int *)(base + index * 4)` written as one expression rather than as a named
|
||||
* pointer.
|
||||
*
|
||||
* The result is `sltu v0,zero,v0`, this compiler's form for `x != 0` producing
|
||||
* 0 or 1, so the return type is a boolean-valued `int` and the tail is a
|
||||
* comparison, not a bit test. The zero return for a null entry is hoisted into
|
||||
* the branch delay slot, and the load of the +0xc field is **not** hoisted into
|
||||
* it — so the source does not branch away to a separate zero return. It assigns a
|
||||
* result variable 0 first and overwrites it only when the entry is non-null,
|
||||
* which is the shape that keeps one shared epilogue: `int r = 0; if (p != 0)
|
||||
* r = ...; return r;`. The early-return spelling was measured and differs by 4
|
||||
* bytes, because cc1 then hoists the load into the branch delay slot and emits a
|
||||
* second jump.
|
||||
*
|
||||
* LIMITS: the function name, the claim that 0x80121C00 holds a table pointer and
|
||||
* the meaning of the +0xc field are hypotheses; only the bytes are evidence. The
|
||||
* index is assumed to be a 4-byte-strided index; whether the original source
|
||||
* used an array type or raw pointer arithmetic is not recoverable.
|
||||
*/
|
||||
|
||||
extern int D_80121C00;
|
||||
|
||||
int func_8002D1F8(int a0)
|
||||
{
|
||||
int p = *(int *)(D_80121C00 + a0 * 4);
|
||||
int r = 0;
|
||||
|
||||
if (p != 0)
|
||||
r = *(int *)(p + 0xc) != 0;
|
||||
|
||||
return r;
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
/*
|
||||
* func_8002E7C4 — 32 bytes at 0x8002E7C4..0x8002E7E4
|
||||
*
|
||||
* Writes two byte fields into a structure reached through a global pointer. The
|
||||
* pointer is reloaded for each store rather than cached in a local, so the C
|
||||
* keeps the two statements independent.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lui v0,0x8012 ; symbol load expands into v0 itself
|
||||
* lw v0,9236(v0) ; v0 = D_80122414 (a pointer)
|
||||
* nop ; maspsx load-delay nop
|
||||
* sb a0,45(v0) ; p->byte_2d = a0
|
||||
* lui v0,0x8012 ; reload for the second store
|
||||
* lw v0,9236(v0)
|
||||
* jr ra
|
||||
* sb a1,18(v0) ; p->byte_12 = a1 (delay slot)
|
||||
*
|
||||
* LIMITS: the symbol name D_80122414, the `char *` pointee type and both field
|
||||
* offsets (45 and 18 decimal) are hypotheses read from the instruction shape.
|
||||
* The parameter types are hypotheses too: the `sb` stores need no mask, so a
|
||||
* `char` or an `int` parameter would emit the same bytes here. Only the compiled
|
||||
* bytes are evidence.
|
||||
*/
|
||||
|
||||
extern char *D_80122414;
|
||||
|
||||
void func_8002E7C4(char first, char second) {
|
||||
D_80122414[45] = first;
|
||||
D_80122414[18] = second;
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
/*
|
||||
* func_8002F404 — 76 bytes at 0x8002F404..0x8002F450
|
||||
*
|
||||
* Leaf routine that sets or clears bit 0 of a 16-bit field in an object reached
|
||||
* through one pointer, reporting whether it found the object.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* beq a0,zero,0x8002f41c 10800005 if (a0 == 0) goto the zero result
|
||||
* li v1,0x1 24030001 v1 = 1 (delay slot)
|
||||
* lw a0,0x0(a0) 8c840000 a0 = *(int *)a0
|
||||
* nop 00000000 load-delay slot
|
||||
* bne a0,zero,0x8002f424 14800002 if (a0 != 0) goto the field update
|
||||
* andi v0,a1,0xff 30a200ff v0 = a1 & 0xff (delay slot)
|
||||
* 0x8002f41c:
|
||||
* j 0x8002f448 0800bd12 goto epilogue
|
||||
* addu v1,zero,zero 00001821 v1 = 0 (delay slot)
|
||||
* 0x8002f424:
|
||||
* bne v0,v1,0x8002f438 14400004 if ((a1 & 0xff) != 1) goto the clear path
|
||||
* nop 00000000 (delay slot)
|
||||
* lhu v0,0x6(a0) 94820006 v0 = *(unsigned short *)(a0 + 6)
|
||||
* j 0x8002f444 0800bd11 goto the store
|
||||
* ori v0,v0,0x1 34420001 v0 |= 1 (delay slot)
|
||||
* 0x8002f438:
|
||||
* lhu v0,0x6(a0) 94820006 v0 = *(unsigned short *)(a0 + 6)
|
||||
* nop 00000000 load-delay slot
|
||||
* andi v0,v0,0xfffe 3042fffe v0 &= ~1
|
||||
* 0x8002f444:
|
||||
* sh v0,0x6(a0) a4820006 *(unsigned short *)(a0 + 6) = v0
|
||||
* 0x8002f448:
|
||||
* jr ra 03e00008
|
||||
* move v0,v1 00601021 v0 = v1 (delay slot)
|
||||
*
|
||||
* The result is a **variable** v1, initialised to 1 in the first branch's delay
|
||||
* slot and zeroed on the shared failure path, then returned via `move v0,v1`. Both
|
||||
* null exits jump to one block that only sets v1 = 0, so the source assigns the
|
||||
* result variable rather than returning a literal in two places; writing it as two
|
||||
* `return 0;` statements gives a different layout.
|
||||
*
|
||||
* The second argument is masked with `andi ...,0xff` and compared against 1, so the
|
||||
* flag is a byte-wide value even though it arrives in a 32-bit register. The field
|
||||
* is 16-bit (`lhu`/`sh`) and is set with `| 1` or cleared with `& ~1` — the
|
||||
* assembler renders `~1` as 0xfffe.
|
||||
*
|
||||
* LIMITS: the function name, the pointer hop, the offset 0x6 and the meaning of the
|
||||
* flag byte are hypotheses; only the bytes are evidence. The comparison value 1 is
|
||||
* taken as the "set" selector because it is the value that produces the `ori`.
|
||||
*/
|
||||
|
||||
int func_8002F404(int a0, int a1)
|
||||
{
|
||||
int v1 = 1;
|
||||
|
||||
if (a0 == 0)
|
||||
v1 = 0;
|
||||
else {
|
||||
a0 = *(int *)a0;
|
||||
if (a0 == 0)
|
||||
v1 = 0;
|
||||
else if ((a1 & 0xff) == 1)
|
||||
*(unsigned short *)(a0 + 6) |= 1;
|
||||
else
|
||||
*(unsigned short *)(a0 + 6) &= ~1;
|
||||
}
|
||||
|
||||
return v1;
|
||||
}
|
||||
@@ -0,0 +1,57 @@
|
||||
/*
|
||||
* func_80043D8C — 56 bytes at 0x80043D8C..0x80043DC4
|
||||
*
|
||||
* Leaf routine that appends one word to a small counted array: it compares the
|
||||
* byte count against the byte capacity, and if there is room, increments the
|
||||
* count and stores the argument through the array's base pointer.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lbu v1,0x0(a0) 90830000 v1 = *(unsigned char *)(a0 + 0)
|
||||
* lbu v0,0x1(a0) 90820001 v0 = *(unsigned char *)(a0 + 1)
|
||||
* andi a2,v1,0xff 306200ff a2 = v1 & 0xff
|
||||
* sltu v0,a2,v0 0045102b v0 = (a2 < v0) unsigned
|
||||
* beq v0,zero,0x80043dbc 10400006 if (!(count < capacity)) goto epilogue
|
||||
* nop 00000000 (delay slot)
|
||||
* addiu v0,v1,0x1 24620001 v0 = v1 + 1
|
||||
* lw v1,0x4(a0) 8c830004 v1 = *(int *)(a0 + 4)
|
||||
* sb v0,0x0(a0) a0820000 *(char *)(a0 + 0) = v0
|
||||
* sll v0,a2,0x2 00021080 v0 = a2 * 4
|
||||
* addu v0,v0,v1 00431021 v0 = (index * 4) + base
|
||||
* sw a1,0x0(v0) ac450000 *(int *)v0 = a1
|
||||
* 0x80043dbc:
|
||||
* jr ra 03e00008
|
||||
* li v0,0x1 24020001 v0 = 1 (delay slot)
|
||||
*
|
||||
* Two byte loads with no load-delay `nop` in between: they target different
|
||||
* registers, and nothing reads them until the `andi`/`sltu` pair, so no stall is
|
||||
* needed. The `andi ...,0xff` before the comparison is the promotion mask this
|
||||
* ABI needs to compare two `unsigned char` values as unsigned ints (finding 7's
|
||||
* family), and the same masked value is reused as the array index — so the source
|
||||
* reads the count once into an `unsigned char` local.
|
||||
*
|
||||
* The capacity is at +1 and the base pointer at +4, so the object is a
|
||||
* `{ unsigned char count; unsigned char capacity; ... int *base; }` shape; the
|
||||
* pointer is loaded *after* the increment is computed but *before* the store to
|
||||
* the count, which is cc1's scheduling and is left to the compiler.
|
||||
*
|
||||
* The result is the constant 1 in the `jr ra` delay slot, so the function reports
|
||||
* success unconditionally (`return 1;`) — it does not report whether the append
|
||||
* happened. That is the shape of the bytes, not an inference about intent.
|
||||
*
|
||||
* LIMITS: the function name, the struct interpretation and the field meanings are
|
||||
* hypotheses; only the bytes are evidence. The array stride is 4 and the count is
|
||||
* an 8-bit field.
|
||||
*/
|
||||
|
||||
int func_80043D8C(unsigned char *a0, int a1)
|
||||
{
|
||||
unsigned char count = a0[0];
|
||||
unsigned char capacity = a0[1];
|
||||
|
||||
if (count < capacity) {
|
||||
a0[0] = count + 1;
|
||||
*(int *)(*(int *)(a0 + 4) + count * 4) = a1;
|
||||
}
|
||||
|
||||
return 1;
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
/*
|
||||
* func_8004C060 — 48 bytes at 0x8004C060..0x8004C090
|
||||
*
|
||||
* Leaf routine that reports whether a state field holds one of two values.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lw v0,0x20(a0) 8c820020 v0 = *(int *)(a0 + 0x20)
|
||||
* nop 00000000 load-delay slot
|
||||
* lw v1,0x21c(v0) 8c43021c v1 = *(int *)(v0 + 0x21c)
|
||||
* li v0,0x7 24020007 v0 = 7
|
||||
* beq v1,v0,0x8004c084 10620003 if (v1 == 7) goto set-one
|
||||
* addu a0,zero,zero 00002021 a0 = 0 (delay slot)
|
||||
* li v0,0x1 24020001 v0 = 1
|
||||
* bne v1,v0,0x8004c088 14620001 if (v1 != 1) goto epilogue
|
||||
* nop 00000000 (delay slot)
|
||||
* 0x8004c084:
|
||||
* li a0,0x1 24040001 a0 = 1
|
||||
* 0x8004c088:
|
||||
* jr ra 03e00008
|
||||
* move v0,a0 00801021 v0 = a0 (delay slot)
|
||||
*
|
||||
* The result lives in **a0**, not v0: a0's incoming pointer value is dead after
|
||||
* the two loads, so cc1 reused the argument register for the result and copies it
|
||||
* to v0 only in the `jr ra` delay slot. The zero initialisation is hoisted into
|
||||
* the first branch's delay slot, which is the tell that the source assigns the
|
||||
* result variable to 0 first and only overwrites it on a match.
|
||||
*
|
||||
* The two comparisons are `== 7` and `== 1` with the constants materialised into
|
||||
* v0 by `li` — SGIs have no branch-immediate form, so a comparison against a
|
||||
* constant always costs a register (the same reason func_80010418 materialises
|
||||
* its -1 sentinel).
|
||||
*
|
||||
* LIMITS: the function name and the meaning of the states 7 and 1 are
|
||||
* hypotheses; only the bytes are evidence. The two loads are 32-bit. The C uses
|
||||
* `||` for the pair because the two branches share one "set one" block; a nested
|
||||
* if or a `switch` would produce a different layout.
|
||||
*/
|
||||
|
||||
int func_8004C060(int a0)
|
||||
{
|
||||
int v1 = *(int *)(*(int *)(a0 + 0x20) + 0x21c);
|
||||
int r = 0;
|
||||
|
||||
if (v1 == 7 || v1 == 1)
|
||||
r = 1;
|
||||
|
||||
return r;
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
/*
|
||||
* func_8004C0AC — 68 bytes at 0x8004C0AC..0x8004C0F0
|
||||
*
|
||||
* Leaf routine that reports whether a state field holds one of four values,
|
||||
* three of them tested individually and two by a range check.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lw v0,0x20(a0) 8c820020 v0 = *(int *)(a0 + 0x20)
|
||||
* nop 00000000 load-delay slot
|
||||
* lw v1,0x21c(v0) 8c43021c v1 = *(int *)(v0 + 0x21c)
|
||||
* nop 00000000 load-delay slot
|
||||
* addiu v0,v1,-0x2 2462fffe v0 = v1 - 2
|
||||
* sltiu v0,v0,0x2 2c420002 v0 = ((unsigned)(v1 - 2) < 2) (v1 == 2 || 3)
|
||||
* bne v0,zero,0x8004c0e4 14400007 if (v0) goto set-one
|
||||
* addu a0,zero,zero 00002021 a0 = 0 (delay slot)
|
||||
* li v0,0x6 24020006 v0 = 6
|
||||
* beq v1,v0,0x8004c0e4 10620005 if (v1 == 6) goto set-one
|
||||
* nop 00000000 (delay slot)
|
||||
* li v0,0x4 24020004 v0 = 4
|
||||
* bne v1,v0,0x8004c0e8 14620001 if (v1 != 4) goto epilogue
|
||||
* nop 00000000 (delay slot)
|
||||
* 0x8004c0e4:
|
||||
* li a0,0x1 24040001 a0 = 1
|
||||
* 0x8004c0e8:
|
||||
* jr ra 03e00008
|
||||
* move v0,a0 00801021 v0 = a0 (delay slot)
|
||||
*
|
||||
* This is the same shape as func_8004C060 (matched): the result is accumulated in
|
||||
* **a0**, the dead argument register, and copied to v0 only in the `jr ra` delay
|
||||
* slot, with the zero initialisation hoisted into the first branch's delay slot.
|
||||
* That fixes the source as `int r = 0; if (cond) r = 1; return r;`.
|
||||
*
|
||||
* The first two values are tested by a single range check — `addiu` then `sltiu` —
|
||||
* which is cc1's form for `v1 == 2 || v1 == 3`. The remaining values are tested in
|
||||
* source order after it: 6 (taken branch to the shared set-one block) and then 4,
|
||||
* whose failure is the fall-through to the epilogue. So the condition is written
|
||||
* with 4 **last**, not in ascending order.
|
||||
*
|
||||
* LIMITS: the function name and the meaning of the states 2, 3, 4 and 6 are
|
||||
* hypotheses; only the bytes are evidence. The two loads are 32-bit. Whether the
|
||||
* original spelled the range as two `==` tests or as a range is not recoverable;
|
||||
* the `sltiu` shows cc1 saw a range either way.
|
||||
*/
|
||||
|
||||
int func_8004C0AC(int a0)
|
||||
{
|
||||
int v1 = *(int *)(*(int *)(a0 + 0x20) + 0x21c);
|
||||
int r = 0;
|
||||
|
||||
if (v1 == 2 || v1 == 3 || v1 == 6 || v1 == 4)
|
||||
r = 1;
|
||||
|
||||
return r;
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
/*
|
||||
* func_8004CEEC — 32 bytes at 0x8004CEEC..0x8004CF0C
|
||||
*
|
||||
* Leaf routine that initialises part of a structure: two zero fields, a
|
||||
* constant field, and a field copied from a global.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lui v1,0x8012 3c038012 \
|
||||
* lw v1,0x2354(v1) 8c632354 / v1 = *(int *)0x80122354 (D_80122354)
|
||||
* lui v0,0x8000 3c028000 v0 = 0x80000000
|
||||
* sh zero,0x3e(a0) a480003e *(short *)(a0 + 0x3e) = 0
|
||||
* sw zero,0xdc(a0) ac8000dc *(int *)(a0 + 0xdc) = 0
|
||||
* sw v0,0xe0(a0) ac8200e0 *(int *)(a0 + 0xe0) = 0x80000000
|
||||
* jr ra 03e00008
|
||||
* sw v1,0xe4(a0) ac8300e4 *(int *)(a0 + 0xe4) = v1 (delay slot)
|
||||
*
|
||||
* Two address forms appear side by side and both are the macro form: the
|
||||
* global is `lui`+`lw` through the same register (cookbook finding 2), and the
|
||||
* constant 0x80000000 is materialised by `lui v0,0x8000` alone — no `ori`,
|
||||
* because the low half is zero. 0x80122354 is outside the small-data window
|
||||
* and the original accesses it absolutely, so it is written as the named
|
||||
* global `D_80122354` with no registry row (the address-named symbol resolves
|
||||
* implicitly). The last store is independent and lands in the `jr ra` delay
|
||||
* slot.
|
||||
*
|
||||
* LIMITS: the function name, the structure and every field offset are
|
||||
* hypotheses read off the disassembly. The field widths are evidence: one
|
||||
* 16-bit store and three 32-bit stores. The constant is written as
|
||||
* `(int)0x80000000` because a bare `0x80000000` is unsigned in this dialect and
|
||||
* would make cc1 emit a different materialisation; the signed value is what the
|
||||
* `lui`-only sequence implies.
|
||||
*/
|
||||
|
||||
extern int D_80122354;
|
||||
|
||||
void func_8004CEEC(int a0)
|
||||
{
|
||||
int v1 = D_80122354;
|
||||
|
||||
*(short *)(a0 + 0x3e) = 0;
|
||||
*(int *)(a0 + 0xdc) = 0;
|
||||
*(int *)(a0 + 0xe0) = (int)0x80000000;
|
||||
*(int *)(a0 + 0xe4) = v1;
|
||||
}
|
||||
@@ -0,0 +1,32 @@
|
||||
/* func_800516E0 — 0x800516E0..0x800516FC (28 bytes).
|
||||
*
|
||||
* Original words:
|
||||
* 0x8C820020 lw v0,0x20(a0)
|
||||
* 0x00000000 nop
|
||||
* 0x8C4200F4 lw v0,0xf4(v0)
|
||||
* 0x00000000 nop
|
||||
* 0xAC4501FC sw a1,0x1fc(v0)
|
||||
* 0x03E00008 jr ra
|
||||
* 0xA0460200 sb a2,0x200(v0) (jr ra delay slot)
|
||||
*
|
||||
* A two-hop pointer walk ending in one word store and one byte store. The
|
||||
* second store lands in the `jr ra` delay slot (cookbook finding 8: the last
|
||||
* independent instruction fills the slot).
|
||||
*
|
||||
* The `nop` after each load is the load-delay fill the original carries; it
|
||||
* comes from maspsx/the assembler, not from the C.
|
||||
*
|
||||
* LIMITS: the struct layout is a hypothesis reconstructed from the two
|
||||
* displacements. Nothing proves that offsets 0x20, 0xf4, 0x1fc and 0x200 are
|
||||
* distinct fields of distinct objects, only that the compiled sequence walks
|
||||
* them in this order. The stored byte's signedness is not observable from `sb`.
|
||||
*/
|
||||
|
||||
void func_800516E0(int base, int value, int flag)
|
||||
{
|
||||
int inner = *(int *)(base + 0x20);
|
||||
|
||||
inner = *(int *)(inner + 0xf4);
|
||||
*(int *)(inner + 0x1fc) = value;
|
||||
*(char *)(inner + 0x200) = flag;
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
/*
|
||||
* func_8005182C — 56 bytes at 0x8005182C..0x80051864
|
||||
*
|
||||
* Two pointer hops, then a guarded range test. The result register is cleared
|
||||
* before the guard and the guard's body only runs when the flag byte is non-zero;
|
||||
* the `beqz`'s delay slot carries the `move v1,zero`, so the false path falls
|
||||
* through with the pre-cleared result.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lw v0,32(a0) ; q = p->ptr_20
|
||||
* nop
|
||||
* lw a0,244(v0) ; r = q->ptr_f4
|
||||
* nop
|
||||
* lbu v0,26(a0) ; r->byte_1a
|
||||
* nop
|
||||
* beqz v0,0x8005185C ; if (r->byte_1a == 0) return 0
|
||||
* move v1,zero ; result = 0 (delay slot)
|
||||
* lw v0,500(a0) ; r->word_1f4
|
||||
* nop
|
||||
* addiu v0,v0,-2 ; value - 2 <- stays in v0
|
||||
* sltiu v1,v0,2 ; result = (unsigned)(value - 2) < 2
|
||||
* 5C:jr ra
|
||||
* move v0,v1 ; return result (delay slot)
|
||||
*
|
||||
* The subtraction must stay in `v0` rather than being folded into the result
|
||||
* register: writing the comparison as one expression
|
||||
* (`result = (unsigned)(r->word - 2) < 2;`) makes `cc1` compute the subtraction
|
||||
* into `v1` and then compare in place, which is a 2-byte register-field diff.
|
||||
* Binding the subtraction to its own local reproduces the original's `addiu v0`.
|
||||
*
|
||||
* LIMITS: every offset, the pointer types and the `unsigned char` flag type are
|
||||
* hypotheses read from the instruction shape; the `lbu` is what shows the flag is
|
||||
* unsigned, and the `sltiu` is what shows the range test is unsigned. The
|
||||
* constants 2 and 2 are facts about this executable, not about the compiler.
|
||||
* Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
int func_8005182C(char *base) {
|
||||
int result = 0;
|
||||
char *q = *(char **)(base + 32);
|
||||
char *r = *(char **)(q + 244);
|
||||
|
||||
if (*(unsigned char *)(r + 26) != 0) {
|
||||
int value = *(int *)(r + 500) - 2;
|
||||
|
||||
result = (unsigned int)value < 2;
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
@@ -0,0 +1,31 @@
|
||||
/*
|
||||
* func_80052C98 — 20 bytes at 0x80052C98..0x80052CAC
|
||||
*
|
||||
* Leaf routine that walks two embedded pointers and clears one word.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lw v0,0x20(a0) 8c820020 v0 = *(int *)(a0 + 0x20)
|
||||
* nop 00000000 load-delay slot
|
||||
* lw v0,0xf4(v0) 8c4200f4 v0 = *(int *)(v0 + 0xf4)
|
||||
* jr ra 03e00008
|
||||
* sw zero,0x1f8(v0) ac4001f8 *(int *)(v0 + 0x1f8) = 0 (delay slot)
|
||||
*
|
||||
* The `nop` after the first load is the load-delay slot: the next instruction
|
||||
* reads the loaded register, so both cc1 and maspsx keep the stall visible.
|
||||
* The store is independent of the second load's result and is scheduled into
|
||||
* the `jr ra` delay slot.
|
||||
*
|
||||
* LIMITS: the function name and the interpretation of the offsets (a struct at
|
||||
* a0+0x20, a nested pointer at +0xf4, a flag word at +0x1f8) are hypotheses
|
||||
* read off the disassembly; no struct layout is claimed. The access widths are
|
||||
* evidence: 32-bit loads, 32-bit store. The pointer arithmetic is written with
|
||||
* `int *` and byte offsets added on the integer value, which is what makes cc1
|
||||
* emit plain `lw`/`sw` with a folded displacement.
|
||||
*/
|
||||
|
||||
void func_80052C98(int a0)
|
||||
{
|
||||
int v0 = *(int *)(a0 + 0x20);
|
||||
v0 = *(int *)(v0 + 0xf4);
|
||||
*(int *)(v0 + 0x1f8) = 0;
|
||||
}
|
||||
@@ -0,0 +1,40 @@
|
||||
/*
|
||||
* func_8005E3D0 — 36 bytes at 0x8005E3D0..0x8005E3F4
|
||||
*
|
||||
* Leaf routine that copies two fields out of a structure through two output
|
||||
* pointers and returns a third field.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lw v0,0xc4c(a0) 8c820c4c v0 = *(int *)(a0 + 0xc4c)
|
||||
* nop 00000000 load-delay slot
|
||||
* sw v0,0x0(a1) aca20000 *a1 = v0
|
||||
* lw v0,0xc50(a0) 8c820c50 v0 = *(int *)(a0 + 0xc50)
|
||||
* nop 00000000 load-delay slot
|
||||
* sw v0,0x0(a2) acc20000 *a2 = v0
|
||||
* lw v0,0xc58(a0) 8c820c58 v0 = *(int *)(a0 + 0xc58)
|
||||
* jr ra 03e00008
|
||||
* nop 00000000 (delay slot)
|
||||
*
|
||||
* The third load's result is never stored, so it is the return value: v0 is
|
||||
* live at the `jr ra`. That fixes the C return type as `int` (or a 32-bit
|
||||
* scalar) and makes the two output pointers `int *`.
|
||||
*
|
||||
* Both `nop`s sit between a load and a *store of the loaded register*. This is
|
||||
* exactly the case cookbook finding 27 records as maspsx's gap: maspsx inserts a
|
||||
* load-delay `nop` only when the next instruction *loads from* the loaded
|
||||
* register, and a store that merely reads it as a source slips through. If this
|
||||
* candidate comes back short by 4 or 8 bytes, the cause is that gap and not the
|
||||
* C, and the row is a second, demonstrated instance of the finding rather than a
|
||||
* source-shape problem.
|
||||
*
|
||||
* LIMITS: the function name, the field offsets and the claim that the structure
|
||||
* is a game object are hypotheses; only the bytes are evidence. All three
|
||||
* accesses are 32-bit.
|
||||
*/
|
||||
|
||||
int func_8005E3D0(int a0, int *a1, int *a2)
|
||||
{
|
||||
*a1 = *(int *)(a0 + 0xc4c);
|
||||
*a2 = *(int *)(a0 + 0xc50);
|
||||
return *(int *)(a0 + 0xc58);
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
/*
|
||||
* func_800681E0 — 60 bytes at 0x800681E0..0x8006821C
|
||||
*
|
||||
* Selects a table offset (a special case for index 1, otherwise index * 36),
|
||||
* loads a 32-bit word from a table and extracts a 5-bit field from its top. The
|
||||
* index scaling is `cc1`'s strength reduction of `* 36` into `*8`, `+self`, `*4`.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* li v0,1
|
||||
* beq a0,v0,0x800681FC ; if (index == 1) jump to the special case
|
||||
* nop
|
||||
* sll v0,a0,0x3 ; index * 8
|
||||
* addu v0,v0,a0 ; index * 9
|
||||
* j 0x80068200
|
||||
* sll v0,v0,0x2 ; index * 36 (delay slot)
|
||||
* FC: li v0,72 ; the special case, placed AFTER the general one
|
||||
* 00: lui at,0x8013 ; %hi of the table base
|
||||
* addu at,at,v0 ; table + offset (base first)
|
||||
* lw v0,11144(at) ; word = table[offset] (%lo as displacement)
|
||||
* nop
|
||||
* srl v0,v0,0x1a ; >> 26
|
||||
* jr ra
|
||||
* andi v0,v0,0x1f ; & 0x1f (delay slot)
|
||||
*
|
||||
* TWO source-shape levers are load-bearing here, both measured:
|
||||
* - The blocks are MIRRORED: the `index == 1` arm is emitted after the general
|
||||
* arm and reached by a forward branch, so the natural
|
||||
* `if (index == 1) ... else ...` spelling emits the inverted `bne` and the
|
||||
* wrong block order (20 differing bytes). Writing the guard as
|
||||
* `if (index != 1)` reproduces the original (cookbook finding 28).
|
||||
* - The base must be an address-shaped SYMBOL: with a literal base `cc1` emits
|
||||
* `addu at,v0,at` (offset first); the symbol spelling emits
|
||||
* `addu at,at,v0` and folds `%lo` into the load displacement. Same lever that
|
||||
* fixed 0x80016174.
|
||||
*
|
||||
* LIMITS: the table base 0x80132B88, the special-case offset (72), the element
|
||||
* stride (36) and the field extraction (>>26, &0x1f) are hypotheses read from the
|
||||
* instruction shape; what the table holds is unknown and is not guessed here. The
|
||||
* `srl` before the `andi` shows the load is read as UNSIGNED — a signed `>>`
|
||||
* would emit `sra`. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
extern int D_80132B88[];
|
||||
|
||||
int func_800681E0(int index) {
|
||||
int offset;
|
||||
|
||||
if (index != 1)
|
||||
offset = index * 36;
|
||||
else
|
||||
offset = 72;
|
||||
|
||||
return (*(unsigned int *)((char *)D_80132B88 + offset) >> 26) & 0x1F;
|
||||
}
|
||||
@@ -0,0 +1,56 @@
|
||||
/*
|
||||
* func_8007C4EC — 56 bytes at 0x8007C4EC..0x8007C524
|
||||
*
|
||||
* Leaf routine that copies three words from an argument array into a 16-byte
|
||||
* record selected by a byte index, and stores a fourth value in the record's
|
||||
* last word.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lui v1,0x8014 3c038014 \
|
||||
* addiu v1,v1,-0x79f8 24638608 / v1 = 0x80138608 (D_80138608)
|
||||
* lbu v0,0x25(a0) 90820025 v0 = *(unsigned char *)(a0 + 0x25)
|
||||
* lw a0,0x0(a1) 8ca40000 a0 = a1[0]
|
||||
* sll v0,v0,0x4 00021100 v0 = index * 16
|
||||
* addu v0,v0,v1 00431021 v0 = (index * 16) + base
|
||||
* sw a0,0x0(v0) ac440000 *(int *)(v0 + 0x0) = a1[0]
|
||||
* lw v1,0x4(a1) 8ca30004 v1 = a1[1]
|
||||
* nop 00000000 load-delay slot
|
||||
* sw v1,0x4(v0) ac430004 *(int *)(v0 + 0x4) = a1[1]
|
||||
* lw v1,0x8(a1) 8ca30008 v1 = a1[2]
|
||||
* sw a2,0xc(v0) ac42000c *(int *)(v0 + 0xc) = a2
|
||||
* jr ra 03e00008
|
||||
* sw v1,0x8(v0) ac430008 *(int *)(v0 + 0x8) = a1[2] (delay slot)
|
||||
*
|
||||
* The record base 0x80138608 is materialised as `lui`+`addiu`, the
|
||||
* linker-resolved symbol form (cookbook finding 4), so it is written as the named
|
||||
* symbol `D_80138608`; a literal address would give `lui`+`ori` (finding 5). The
|
||||
* index is a byte field at +0x25 of the first argument and is scaled by 16, so
|
||||
* each record is 16 bytes. The address arithmetic is stride first, base second
|
||||
* (`sll` then `addu`), which finding 22 identifies as `(index * 16) + symbol`.
|
||||
*
|
||||
* The `nop` between the second load and its store is a load-delay slot in front of
|
||||
* a *store of the loaded register* — the case cookbook finding 27 records as
|
||||
* maspsx's predicate gap, which does not bite here: the candidate came back with
|
||||
* the right length.
|
||||
*
|
||||
* The store order in the original is 0x0, 0x4, 0xc, 0x8, i.e. the record's third
|
||||
* word is stored last and lands in the `jr ra` delay slot; the C therefore writes
|
||||
* the four stores in address order and leaves the scheduling to cc1.
|
||||
*
|
||||
* LIMITS: the function name, the 16-byte record layout and the claim that a1 is a
|
||||
* three-word source array are hypotheses; only the bytes are evidence. Whether the
|
||||
* source used element stores or a struct assignment is not recoverable from the
|
||||
* bytes; the element form is used here.
|
||||
*/
|
||||
|
||||
extern char D_80138608[];
|
||||
|
||||
void func_8007C4EC(int a0, int *a1, int a2)
|
||||
{
|
||||
int *dst = (int *)(D_80138608 + *(unsigned char *)(a0 + 0x25) * 16);
|
||||
|
||||
dst[0] = a1[0];
|
||||
dst[1] = a1[1];
|
||||
dst[2] = a1[2];
|
||||
dst[3] = a2;
|
||||
}
|
||||
@@ -0,0 +1,29 @@
|
||||
/*
|
||||
* func_800827A8 — 28 bytes at 0x800827A8..0x800827C4
|
||||
*
|
||||
* Indexes a table of 16-byte elements through a global base pointer and returns
|
||||
* a masked 32-bit word from the element. The base is loaded through the symbol
|
||||
* macro and the `addu` takes the scaled index first and the base second, which
|
||||
* is the spelling recorded as cookbook finding 22 (direct expression form).
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lui v0,0x8012 ; symbol load expands into v0 itself
|
||||
* lw v0,8968(v0) ; v0 = D_80122308 (a base pointer)
|
||||
* sll a0,a0,4 ; index *= 16
|
||||
* addu a0,a0,v0 ; element = index + base (index first)
|
||||
* lw v0,0(a0) ; word = element->word_00
|
||||
* jr ra
|
||||
* andi v0,v0,0x7ff ; return word & 0x7ff (delay slot)
|
||||
*
|
||||
* LIMITS: the symbol name D_80122308, the element stride (16), the field offset
|
||||
* (0) and the mask (0x7ff) are hypotheses read from the instruction shape. The
|
||||
* operand order of the `addu` is a source-spelling artifact, not a compiler
|
||||
* fact; if the direct spelling fails, re-spell the addition before hunting for a
|
||||
* flag (finding 22). Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
extern int D_80122308;
|
||||
|
||||
int func_800827A8(int index) {
|
||||
return *(int *)(D_80122308 + index * 16) & 0x7FF;
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
/* func_80083440 — 0x80083440..0x80083470 (48 bytes).
|
||||
*
|
||||
* Original words:
|
||||
* lui a1,0x8012
|
||||
* lw a1,0x2308(a1) a1 = D_80122308
|
||||
* sll a0,a0,0x4 index *= 16
|
||||
* addu a1,a1,a0
|
||||
* lw v0,0x0(a1)
|
||||
* li v1,-0x3801 ~0x3800
|
||||
* and v0,v0,v1
|
||||
* ori v0,v0,0x2800
|
||||
* li v1,-0x4001 ~0x4000
|
||||
* and v0,v0,v1
|
||||
* jr ra
|
||||
* _sw v0,0x0(a1) (delay slot)
|
||||
*
|
||||
* A read-modify-write of one 16-byte-strided element in a table whose base is a
|
||||
* global *value* (the `lui`+`lw` pair is a symbol load into the same register,
|
||||
* cookbook finding 2, so 0x80122308 holds a pointer, not the table). The value
|
||||
* is cleared of bits 11-13 and bit 14, then bit 13 and bit 11 are set:
|
||||
* `(v & ~0x3800) | 0x2800`, then `& ~0x4000`.
|
||||
*
|
||||
* The masks are written as `~` of the set bits rather than as the raw
|
||||
* 0xFFFFC7FF/0xFFFFBFFF constants, which is what makes the `li` constants fall
|
||||
* out; the store lands in the `jr ra` delay slot (finding 8).
|
||||
*
|
||||
* LIMITS: the stride 16, the displaced base and the three mask bits are read
|
||||
* from the bytes; what the field means and why bits 11/13 are set while 12 and 14
|
||||
* are cleared is not observable. `D_80122308` is typed `int` because the only
|
||||
* proven use is as an integer base added to a scaled index — it may really be a
|
||||
* pointer, in which case the type is cosmetic here.
|
||||
*/
|
||||
|
||||
extern int D_80122308;
|
||||
|
||||
int func_80083440(int index)
|
||||
{
|
||||
int *element = (int *)(D_80122308 + (index << 4));
|
||||
int value = *element;
|
||||
|
||||
value &= ~0x3800;
|
||||
value |= 0x2800;
|
||||
value &= ~0x4000;
|
||||
*element = value;
|
||||
return value;
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
/* func_8008F4F4 — 0x8008F4F4..0x8008F508 (20 bytes).
|
||||
*
|
||||
* Original words:
|
||||
* 0x3C018012 lui at,0x8012
|
||||
* 0x00240821 addu at,at,a0
|
||||
* 0x80221ED8 lb v0,0x1ed8(at)
|
||||
* 0x03E00008 jr ra
|
||||
* 0x00000000 nop
|
||||
*
|
||||
* A signed byte read from a fixed table indexed by the argument. The
|
||||
* lui/addu/lb shape is the assembler's expansion of a symbol-plus-register
|
||||
* address (cookbook finding 1: this compiler emits the macro form by default),
|
||||
* so the base is a symbol at 0x80121ED8, not a literal — a literal would give
|
||||
* lui+ori (finding 5).
|
||||
*
|
||||
* `lb` not `lbu` is the discriminator for signedness: plain `char` is unsigned
|
||||
* on this target and would emit `lbu` (finding 7), so the element type is
|
||||
* explicitly `signed char`. That is a byte-visible choice, not cosmetic.
|
||||
*
|
||||
* LIMITS: the symbol name D_80121ED8 is the project's address-derived
|
||||
* convention. The array's length and the meaning of the index are unknown; the
|
||||
* body bounds nothing.
|
||||
*/
|
||||
|
||||
extern signed char D_80121ED8[];
|
||||
|
||||
signed char func_8008F4F4(int index)
|
||||
{
|
||||
return D_80121ED8[index];
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
/* func_80092088 — 0x80092088..0x800920BC (52 bytes).
|
||||
*
|
||||
* Original words:
|
||||
* beq a0,zero,0x800920B4
|
||||
* _clear v0 (delay slot)
|
||||
* lw a0,0xc(a0)
|
||||
* nop
|
||||
* bne a0,zero,0x800920A8
|
||||
* _li v0,0x1 (delay slot)
|
||||
* j 0x800920B4
|
||||
* _clear v0 (delay slot)
|
||||
* 0x800920A8: lw v1,0x160(a0)
|
||||
* nop
|
||||
* sw v1,0x0(a1)
|
||||
* 0x800920B4: jr ra
|
||||
* nop
|
||||
*
|
||||
* A doubly-guarded fetch-and-report: a null argument returns 0, a null inner
|
||||
* pointer returns 0, and otherwise the inner pointer's field 0x160 is published
|
||||
* through the out-parameter and the routine reports 1.
|
||||
*
|
||||
* Both null tests put their result constant in the branch delay slot, so v0 is
|
||||
* the return register throughout and no extra instruction materialises it
|
||||
* (cookbook finding 8). The success store is reached by the `bne` and then falls
|
||||
* into the shared `jr ra`, so v0 = 1 survives from the delay slot into the
|
||||
* return — that is why `li v0,1` sits where it does.
|
||||
*
|
||||
* LIMITS: the displacements 0xc and 0x160 are read from the bytes; the object
|
||||
* at 0xc is assumed to be a distinct node from the one at 0x160 and that is not
|
||||
* proved. `int` rather than a pointer type for the out-parameter keeps the
|
||||
* `sw` width honest but says nothing about what the published value is.
|
||||
*/
|
||||
|
||||
int func_80092088(int node, int *out)
|
||||
{
|
||||
if (node == 0)
|
||||
return 0;
|
||||
node = *(int *)(node + 0xc);
|
||||
if (node != 0) {
|
||||
*out = *(int *)(node + 0x160);
|
||||
return 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
/*
|
||||
* func_80092130 — 28 bytes at 0x80092130..0x8009214C
|
||||
*
|
||||
* Leaf routine that chases two pointers and returns a sign-extended halfword.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lw v0,0x0c(a0) 8c82000c v0 = *(int *)(a0 + 0x0c)
|
||||
* nop 00000000 load-delay slot
|
||||
* lw v0,0x160(v0) 8c420160 v0 = *(int *)(v0 + 0x160)
|
||||
* nop 00000000 load-delay slot
|
||||
* lh v0,0x0002(v0) 84420002 return *(short *)(v0 + 2)
|
||||
* jr ra 03e00008
|
||||
* nop 00000000 (delay slot)
|
||||
*
|
||||
* Both `nop`s are load-delay slots: each is followed by an instruction that
|
||||
* reads the register just loaded. This is the chained-dereference case that
|
||||
* maspsx does pad (cookbook finding 27's counterexample at 0x800C5C84 is the
|
||||
* same shape), so no `maspsx=off` override is needed. The final `nop` is the
|
||||
* `jr ra` delay slot: nothing independent is available to fill it.
|
||||
*
|
||||
* LIMITS: the function name, the struct layout behind a0+0x0c and the meaning
|
||||
* of the halfword at +2 are hypotheses from the disassembly. The load widths
|
||||
* and the sign-extending `lh` are evidence. `short` as the return type is what
|
||||
* makes cc1 emit `lh` rather than `lhu`; cookbook finding 7 (plain `char` is
|
||||
* unsigned here) does not apply to `short`.
|
||||
*/
|
||||
|
||||
short func_80092130(int a0)
|
||||
{
|
||||
int v0 = *(int *)(a0 + 0x0c);
|
||||
v0 = *(int *)(v0 + 0x160);
|
||||
return *(short *)(v0 + 2);
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
/* func_80099E14 — 0x80099E14..0x80099E34 (32 bytes).
|
||||
*
|
||||
* Original words:
|
||||
* 0x8C830008 lw v1,0x8(a0)
|
||||
* 0x00000000 nop
|
||||
* 0x9062000B lbu v0,0xb(v1)
|
||||
* 0x00000000 nop
|
||||
* 0x304200F9 andi v0,v0,0xf9
|
||||
* 0xA062000B sb v0,0xb(v1)
|
||||
* 0x03E00008 jr ra
|
||||
* 0x24020001 _li v0,0x1 (delay slot)
|
||||
*
|
||||
* Clears bits 1 and 2 of a byte field inside a structure reached through one
|
||||
* pointer hop, and reports success. `andi 0xf9` is a clear-mask (0xf9 = ~0x06),
|
||||
* so the stored value is `field & ~0x06`.
|
||||
*
|
||||
* `lbu` and not `lb` proves the field is read unsigned (cookbook finding 7:
|
||||
* plain `char` is unsigned here, so the element type is a plain `unsigned
|
||||
* char`). The `li v0,1` sits in the `jr ra` delay slot, so the return value is
|
||||
* computed last and costs no extra instruction.
|
||||
*
|
||||
* LIMITS: offsets 0x8 and 0xb are read from the displacements; the structure
|
||||
* they belong to is unknown. Which two bits 0x06 names is not observable. The
|
||||
* receiver of the pointer at 0x8 is not proven to be the same object whose
|
||||
* offset 0x80 other routines in this area touch.
|
||||
*/
|
||||
|
||||
int func_80099E14(int base)
|
||||
{
|
||||
unsigned char *p = *(unsigned char **)(base + 8);
|
||||
|
||||
p[0xb] &= 0xf9;
|
||||
return 1;
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
/*
|
||||
* func_8009D8A0 — 64 bytes at 0x8009D8A0..0x8009D8E0
|
||||
*
|
||||
* Splices a node into a list held by a structure reached through a pointer at
|
||||
* offset 8: takes the list head out of the structure, links the new node's back
|
||||
* pointer to it, puts the new node at the head, and fixes up the old head's back
|
||||
* pointer when there was one. Returns 1.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* sw zero,0(a1) ; out->word_00 = 0
|
||||
* lw v0,8(a0) ; p->ptr_08
|
||||
* nop
|
||||
* lw v0,40(v0) ; ->word_28 (the current list head)
|
||||
* nop
|
||||
* sw v0,4(a1) ; out->word_04 = old head
|
||||
* lw v0,8(a0) ; RELOAD p->ptr_08
|
||||
* nop
|
||||
* sw a1,40(v0) ; ->word_28 = out (new head)
|
||||
* lw v0,4(a1) ; out->word_04 (old head)
|
||||
* nop
|
||||
* beqz v0,0x8009D8D8 ; if (old head == 0) skip
|
||||
* nop
|
||||
* sw a1,0(v0) ; old head->word_00 = out
|
||||
* D8: jr ra
|
||||
* li v0,1 ; return 1 (delay slot)
|
||||
*
|
||||
* The `p->ptr_08` load appears TWICE in the original and the C must therefore
|
||||
* re-read it rather than cache it in a local — the two loads are the same
|
||||
* address, so caching would remove the second one and change the bytes.
|
||||
*
|
||||
* LIMITS: every offset and the parameter types are hypotheses read from the
|
||||
* instruction shape; what the structures mean is unknown and is not guessed here.
|
||||
* `out` is typed `int *` because its fields are stored as words and the back
|
||||
* pointer at offset 0 is compared against zero. Only the compiled bytes are
|
||||
* evidence.
|
||||
*/
|
||||
|
||||
int func_8009D8A0(char *p, int *out) {
|
||||
int old_head;
|
||||
|
||||
out[0] = 0;
|
||||
out[1] = *(int *)(*(char **)(p + 8) + 40);
|
||||
*(int *)(*(char **)(p + 8) + 40) = (int)out;
|
||||
|
||||
old_head = out[1];
|
||||
if (old_head != 0)
|
||||
*(int *)old_head = (int)out;
|
||||
|
||||
return 1;
|
||||
}
|
||||
@@ -0,0 +1,40 @@
|
||||
/*
|
||||
* func_8009F0E8 — 56 bytes at 0x8009F0E8..0x8009F120
|
||||
*
|
||||
* Reads three adjacent 16-bit fields out of a structure reached through a pointer
|
||||
* at offset 0x0c, writes them into a three-word output array, and adds a delta to
|
||||
* the middle one. The middle word is *reloaded* from the output array before the
|
||||
* add, because the output pointer may alias — so the C must read it back rather
|
||||
* than reuse a local.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lw v1,12(a0) ; s = p->ptr_0c
|
||||
* nop
|
||||
* lh v0,264(v1) ; s->half_108
|
||||
* nop
|
||||
* sw v0,0(a2) ; out[0] = s->half_108
|
||||
* lh v0,266(v1) ; s->half_10a
|
||||
* nop
|
||||
* sw v0,4(a2) ; out[1] = s->half_10a
|
||||
* lw v0,4(a2) ; reload out[1]
|
||||
* lh v1,268(v1) ; s->half_10c
|
||||
* addu v0,v0,a1 ; out[1] + delta
|
||||
* sw v1,8(a2) ; out[2] = s->half_10c
|
||||
* jr ra
|
||||
* sw v0,4(a2) ; out[1] += delta (delay slot)
|
||||
*
|
||||
* LIMITS: every offset and the `short` field type are hypotheses read from the
|
||||
* instruction shape — the `lh` is what shows the fields are signed 16-bit. The
|
||||
* reload of out[1] is a real observable in the original and is reproduced here by
|
||||
* using the `+=` form on the array element; binding it to a local first would
|
||||
* remove the reload and change the bytes. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
void func_8009F0E8(char *base, int delta, int *out) {
|
||||
short *s = *(short **)(base + 12);
|
||||
|
||||
out[0] = s[132];
|
||||
out[1] = s[133];
|
||||
out[2] = s[134];
|
||||
out[1] += delta;
|
||||
}
|
||||
@@ -0,0 +1,51 @@
|
||||
/* func_800A2F20 — 0x800A2F20..0x800A2F44 (36 bytes).
|
||||
*
|
||||
* Original words:
|
||||
* 0x24020258 li v0,0x258
|
||||
* 0x3C018014 lui at,0x8014 <- loop top
|
||||
* 0x00220821 addu at,at,v0
|
||||
* 0xA0209E01 sb zero,-0x61ff(at)
|
||||
* 0x2442FF9C addiu v0,v0,-0x64
|
||||
* 0x0441FFFB bgez v0,loop
|
||||
* 0x00000000 _nop (delay slot)
|
||||
* 0x03E00008 jr ra
|
||||
* 0x00000000 nop
|
||||
*
|
||||
* A descending byte-clear sweep. The count starts at 0x258 and steps down by
|
||||
* 0x64 (100), storing a zero byte each time, stopping once the counter goes
|
||||
* negative.
|
||||
*
|
||||
* The `lui`/`addu`/`sb` triple is the assembler's expansion of a
|
||||
* symbol-plus-register store (cookbook finding 1), so the target is a symbol
|
||||
* base indexed by the counter, not a moving pointer. `%hi` 0x8014 with `%lo`
|
||||
* -0x61ff places the base at 0x80139E01.
|
||||
*
|
||||
* The decrement is emitted *before* the `bgez` and the branch tests the
|
||||
* decremented value, with the first iteration reached by falling out of the
|
||||
* `li` — so the loop is bottom-tested. Spelling it as a `for` with the test
|
||||
* normally at the top relies on cc1 rotating the loop; if it does not, the
|
||||
* shape to try is an explicit `do`/`while` (cookbook finding 19's lever, whose
|
||||
* limit is that it applies only where the frame or rotation is the anomaly).
|
||||
*
|
||||
* LIMITS: the stride 0x64, the start 0x258 and the base 0x80139E01 are read
|
||||
* from the bytes; what the swept bytes mean is unknown. Note the base is
|
||||
* deliberately odd (…E01), consistent with a byte array whose real start is
|
||||
* elsewhere — the symbol name is the literal address and carries no meaning.
|
||||
*
|
||||
* ATTEMPT 1 (recorded): binding the base to a named `char *p` local and writing
|
||||
* `p[i] = 0` let cc1 strength-reduce the address into a second induction
|
||||
* variable — it emitted `lui`+`addiu` once, `sb zero,0(v0)`, and moved the
|
||||
* pointer instead of indexing, giving 16 differing bytes. The original keeps
|
||||
* the index form (`addu at,at,v0` every iteration), so the base must stay a
|
||||
* symbol in the index expression and no pointer local may be introduced.
|
||||
*/
|
||||
|
||||
extern char D_80139E01[];
|
||||
|
||||
void func_800A2F20(void)
|
||||
{
|
||||
int i;
|
||||
|
||||
for (i = 0x258; i >= 0; i -= 0x64)
|
||||
D_80139E01[i] = 0;
|
||||
}
|
||||
@@ -0,0 +1,70 @@
|
||||
/*
|
||||
* func_800A8B48 — 68 bytes at 0x800A8B48..0x800A8B8C
|
||||
*
|
||||
* Leaf routine that stores three bytes into a 24-byte record selected by an
|
||||
* index, with an out-of-range index selecting nothing.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* sltiu v0,a0,0x8 2c820008 v0 = (a0 < 8) unsigned
|
||||
* bne v0,zero,0x800a8b5c 14400003 if (in range) goto the address build
|
||||
* sll v0,a0,0x1 00042040 v0 = a0 * 2 (delay slot)
|
||||
* j 0x800a8b70 081002dc goto the null test
|
||||
* addu v0,zero,zero 00001021 v0 = 0 (delay slot)
|
||||
* 0x800a8b5c:
|
||||
* addu v0,v0,a0 00441021 v0 = a0 * 2 + a0 (= a0 * 3)
|
||||
* sll v0,v0,0x3 000420c0 v0 *= 8 (= a0 * 24)
|
||||
* lui v1,0x8014 3c038014 \
|
||||
* addiu v1,v1,-0x4088 2463bf78 / v1 = 0x8013BF78 (D_8013BF78)
|
||||
* addu v0,v0,v1 00431021 v0 = (index * 24) + base
|
||||
* 0x800a8b70:
|
||||
* beq v0,zero,0x800a8b84 10400004 if (p == 0) goto epilogue
|
||||
* nop 00000000 (delay slot)
|
||||
* sb a1,0x4(v0) a0450004 *(char *)(p + 4) = a1
|
||||
* sb a2,0x5(v0) a0460005 *(char *)(p + 5) = a2
|
||||
* sb a3,0x6(v0) a0470006 *(char *)(p + 6) = a3
|
||||
* 0x800a8b84:
|
||||
* jr ra 03e00008
|
||||
* nop 00000000 (delay slot)
|
||||
*
|
||||
* The `sltiu` bound test makes the index **unsigned**; the multiply is
|
||||
* strength-reduced to `sll`+`addu`+`sll` because 24 is not a power of two, which
|
||||
* also fixes the record stride as 24 bytes.
|
||||
*
|
||||
* The two arms are laid out **mirrored** from the natural spelling: the in-range
|
||||
* path is the branch **target** (`bne v0,zero,0x800a8b5c`) and the out-of-range
|
||||
* path is the fall-through, which materialises the null pointer in the delay slot
|
||||
* of a `j` to the merge point. Writing `if (a0 < 8) p = base + a0 * 24; else p = 0;`
|
||||
* produces the opposite layout (measured: 68 bytes, 25 differing bytes, first at
|
||||
* 0x800A8B4C); inverting the condition to `if (a0 >= 8) p = 0; else p = ...`
|
||||
* reproduces the original exactly. This is cookbook finding 28's mirrored-layout
|
||||
* lever, and its diagnostic signature held — the instruction count was already
|
||||
* correct and only the block order was wrong.
|
||||
*
|
||||
* The base 0x8013BF78 is materialised as `lui`+`addiu`, the linker-resolved symbol
|
||||
* form (cookbook finding 4), so it is written as the named symbol `D_8013BF78`
|
||||
* (a literal would give `lui`+`ori`, finding 5). The address arithmetic is stride
|
||||
* first, base second.
|
||||
*
|
||||
* LIMITS: the function name, the 24-byte record layout and the meaning of the
|
||||
* three stored bytes are hypotheses; only the bytes are evidence. The function
|
||||
* returns nothing observable (v0 holds the record pointer on one path and 0 on the
|
||||
* other, but the caller cannot rely on either), so the return type is `void`.
|
||||
*/
|
||||
|
||||
extern char D_8013BF78[];
|
||||
|
||||
void func_800A8B48(unsigned int a0, int a1, int a2, int a3)
|
||||
{
|
||||
char *p;
|
||||
|
||||
if (a0 >= 8)
|
||||
p = 0;
|
||||
else
|
||||
p = D_8013BF78 + a0 * 24;
|
||||
|
||||
if (p != 0) {
|
||||
p[4] = a1;
|
||||
p[5] = a2;
|
||||
p[6] = a3;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,68 @@
|
||||
/*
|
||||
* func_800B24EC — 72 bytes at 0x800B24EC..0x800B2534
|
||||
*
|
||||
* Leaf routine that follows two pointers with null checks and then copies a
|
||||
* four-byte value into an object at +0x194 and sets a flag byte at +0x197.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* beq a0,zero,0x800b252c 1080000f if (a0 == 0) goto epilogue
|
||||
* move t0,a1 00a04021 t0 = a1 (delay slot)
|
||||
* lw a0,0xc(a0) 8c84000c a0 = *(int *)(a0 + 0xc)
|
||||
* nop 00000000 load-delay slot
|
||||
* beq a0,zero,0x800b252c 1080000c if (a0 == 0) goto epilogue
|
||||
* nop 00000000 (delay slot)
|
||||
* lw a3,0x158(a0) 8c870158 a3 = *(int *)(a0 + 0x158)
|
||||
* nop 00000000 load-delay slot
|
||||
* beq a3,zero,0x800b252c 10e0000a if (a3 == 0) goto epilogue
|
||||
* nop 00000000 (delay slot)
|
||||
* lwl v0,0x3(t0) a9030003 \
|
||||
* lwr v0,0x0(t0) a9030000 / v0 = unaligned 4-byte load from a1
|
||||
* nop 00000000 load-delay slot
|
||||
* swl v0,0x197(a3) a8e20197 \
|
||||
* swr v0,0x194(a3) a8e20194 / unaligned 4-byte store to a3 + 0x194
|
||||
* sb a2,0x197(a3) a0e20197 *(char *)(a3 + 0x197) = a2
|
||||
* 0x800b252c:
|
||||
* jr ra 03e00008
|
||||
* nop 00000000 (delay slot)
|
||||
*
|
||||
* Three chained null checks, each exiting to one shared epilogue, with the second
|
||||
* argument copied to t0 in the first branch's delay slot — so the copy is written
|
||||
* as the source's own local and cc1 hoisted it.
|
||||
*
|
||||
* The `lwl`/`lwr` and `swl`/`swr` pairs are **unaligned** 32-bit accesses, which
|
||||
* this compiler emits only when the type's alignment is 1: a plain `*(int *)`
|
||||
* through a cast would give `lw`/`sw`. The source is therefore written as a
|
||||
* 4-byte `char[4]` structure copy, whose alignment is 1, rather than as an int
|
||||
* assignment. The byte at +0x197 is then written separately, overwriting the top
|
||||
* byte of the four just stored.
|
||||
*
|
||||
* LIMITS: the function name, the pointer chain, the offsets and the meaning of the
|
||||
* copied value and the flag byte are hypotheses; only the bytes are evidence. The
|
||||
* `char[4]` struct is a reconstruction of the access *alignment*, not evidence of
|
||||
* the original's types; whether the original used a packed struct, a char-array
|
||||
* copy or a byte-wise copy is not recoverable from these instructions alone.
|
||||
*/
|
||||
|
||||
struct func_800B24EC_quad {
|
||||
char b[4];
|
||||
};
|
||||
|
||||
void func_800B24EC(int a0, void *a1, int a2)
|
||||
{
|
||||
void *t0 = a1;
|
||||
int a3;
|
||||
|
||||
if (a0 == 0)
|
||||
return;
|
||||
|
||||
a0 = *(int *)(a0 + 0xc);
|
||||
if (a0 == 0)
|
||||
return;
|
||||
|
||||
a3 = *(int *)(a0 + 0x158);
|
||||
if (a3 == 0)
|
||||
return;
|
||||
|
||||
*(struct func_800B24EC_quad *)(a3 + 0x194) = *(struct func_800B24EC_quad *)t0;
|
||||
*(char *)(a3 + 0x197) = a2;
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
/*
|
||||
* func_800B3474 — 48 bytes at 0x800B3474..0x800B34A4
|
||||
*
|
||||
* Two chained pointer lookups through a table whose base lives in a global. The
|
||||
* index arrives as a 16-bit signed value, so the original sign-extends it with
|
||||
* `sll`+`sra` (the two shifts are fused into one `sra` by 14) and scales by four.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* sll a0,a0,0x10 ; index <<= 16
|
||||
* lui v0,0x8012 ; symbol load expands into v0 itself
|
||||
* lw v0,7168(v0) ; v0 = D_80121C00 (a table base)
|
||||
* sra a0,a0,0xe ; (short)index * 4
|
||||
* addu a0,a0,v0 ; table + index*4 (index first)
|
||||
* lw v0,0(a0) ; entry = table[index]
|
||||
* nop
|
||||
* lw v0,8(v0) ; entry->ptr_08
|
||||
* nop
|
||||
* lw v0,0(v0) ; *entry->ptr_08
|
||||
* jr ra
|
||||
* sltu v0,zero,v0 ; return that != 0 (delay slot)
|
||||
*
|
||||
* The `sll`/`sra` pair is what shows the parameter is a signed 16-bit value: an
|
||||
* `int` index would be used directly.
|
||||
*
|
||||
* LIMITS: the symbol name D_80121C00, the pointee types, the element stride (4)
|
||||
* and both field offsets are hypotheses read from the instruction shape; what the
|
||||
* structures mean is unknown and is not guessed here. The `addu` takes the scaled
|
||||
* index first and the base second, which is the direct-expression spelling
|
||||
* recorded as cookbook finding 22. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
extern int D_80121C00;
|
||||
|
||||
int func_800B3474(short index) {
|
||||
int *entry = *(int **)(D_80121C00 + index * 4);
|
||||
int *inner = *(int **)((char *)entry + 8);
|
||||
|
||||
return *inner != 0;
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
/*
|
||||
* func_800B7230 — 52 bytes at 0x800B7230..0x800B7264
|
||||
*
|
||||
* Walks a singly-linked list and clears three bits in a 16-bit field of every
|
||||
* node. The mask constant is loop-invariant, so `cc1` hoists its materialisation
|
||||
* out of the loop and keeps it in a register — which is why the original emits
|
||||
* `li v1,-15` once rather than an `andi` with an immediate.
|
||||
*
|
||||
* The mask cannot be an immediate, and the reason is the type of the temporary
|
||||
* rather than the field: `~14` is 0xFFFFFFF1, which `andi` cannot encode (it
|
||||
* zero-extends). Masking the 16-bit field directly
|
||||
* (`p->flags &= ~14;`) lets `cc1` narrow the operation to `andi v0,v0,0xfff1`
|
||||
* and loses 4 bytes; masking through an `int` temporary keeps the 32-bit `and`.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* beqz a0,0x800B725C ; if (p == 0) return
|
||||
* nop
|
||||
* li v1,-15 ; mask = ~14, hoisted out of the loop
|
||||
* 3C:lhu v0,26(a0) ; p->half_1a
|
||||
* nop
|
||||
* and v0,v0,v1 ; & ~14
|
||||
* sh v0,26(a0) ; p->half_1a = masked
|
||||
* lw a0,4(a0) ; p = p->next
|
||||
* nop
|
||||
* bnez a0,0x800B723C ; loop while p != 0
|
||||
* nop
|
||||
* 5C:jr ra
|
||||
* nop
|
||||
*
|
||||
* LIMITS: the struct layout (next pointer at 0x04, the masked field at 0x1a) and
|
||||
* the field's `unsigned short` type are hypotheses read from the instruction
|
||||
* shape; the `lhu` is what shows the field is unsigned, and the mask value is a
|
||||
* fact about this executable. The loop is a plain `while` — the test appears
|
||||
* twice because that is how `cc1` compiles one, not because the source repeats
|
||||
* it. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
typedef struct func_800B7230_node {
|
||||
int unknown_00;
|
||||
struct func_800B7230_node *next; /* offset 0x04 */
|
||||
char unknown_08[0x12];
|
||||
unsigned short flags; /* offset 0x1a */
|
||||
} func_800B7230_node;
|
||||
|
||||
void func_800B7230(func_800B7230_node *p) {
|
||||
while (p != 0) {
|
||||
int flags = p->flags;
|
||||
|
||||
flags &= ~14;
|
||||
p->flags = flags;
|
||||
p = p->next;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,40 @@
|
||||
/* func_800F5200 — 0x800F5200..0x800F521C (28 bytes).
|
||||
*
|
||||
* Original words:
|
||||
* 0x30A507FF andi a1,a1,0x7ff
|
||||
* 0x00052AC0 sll a1,a1,0xb
|
||||
* 0x308207FF andi v0,a0,0x7ff
|
||||
* 0x3C03E500 lui v1,0xe500
|
||||
* 0x00431025 or v0,v0,v1
|
||||
* 0x03E00008 jr ra
|
||||
* 0x00A21025 or v0,a1,v0 (jr ra delay slot)
|
||||
*
|
||||
* A packed 32-bit command word built from two 11-bit fields: the second
|
||||
* argument is masked and shifted to bit 11, the first is masked and OR-ed with
|
||||
* the 0xe5000000 tag, and the two halves are combined in the delay slot.
|
||||
*
|
||||
* The two halves are bound to named locals in the order the original emits
|
||||
* them. Writing the same expression as one nested `|` makes cc1 reassociate it:
|
||||
* `((field & 0x7ff) << 11) | ((value & 0x7ff) | 0xe5000000)` is `A | (B | C)`,
|
||||
* but cc1 canonicalises the associative operator to `(A | C) | B`, emitting
|
||||
* `lui` before the second `andi` and allocating the field to v0 — 16 differing
|
||||
* bytes at the right instruction count. Binding `shifted` first and `tag` second
|
||||
* pins both the emission order and the registers (the field stays in place in
|
||||
* a1, the tag lands in v0).
|
||||
*
|
||||
* LIMITS: `0xe5000000` is a literal from the original's `lui v1,0xe500`; whether
|
||||
* the real source named a constant is unobservable. The field widths (11 bits
|
||||
* each) and the mask values are read from the bytes. That the named locals are
|
||||
* what fixes the order is an observation about this compiler, not a rule — the
|
||||
* same lever failed on the constant-materialisation-order rows recorded in
|
||||
* cookbook finding 28. The function's meaning — a packed hardware command — is
|
||||
* a guess and is not evidence.
|
||||
*/
|
||||
|
||||
int func_800F5200(int value, int field)
|
||||
{
|
||||
int shifted = (field & 0x7ff) << 11;
|
||||
int tag = (value & 0x7ff) | 0xe5000000;
|
||||
|
||||
return shifted | tag;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
/*
|
||||
* func_800F6570 — 36 bytes at 0x800F6570..0x800F6594
|
||||
*
|
||||
* Fills a run of bytes with a caller-supplied value — a `memset` shape. The
|
||||
* count is tested before the first store and the counter is pre-decremented, so
|
||||
* the source uses an explicit guard plus a `do`/`while` (cookbook finding 19).
|
||||
*
|
||||
* The observed instructions are:
|
||||
* beqz a2,0x800F658C ; if (count == 0) skip the loop
|
||||
* addiu v0,a2,-1 ; counter = count - 1 (delay slot)
|
||||
* li v1,-1 ; loop sentinel
|
||||
* L: sb a1,0(a0) ; *p = value
|
||||
* addiu v0,v0,-1 ; --counter
|
||||
* bne v0,v1,0x800F657C ; loop while counter != -1
|
||||
* addiu a0,a0,1 ; ++p (delay slot)
|
||||
* jr ra
|
||||
* nop
|
||||
*
|
||||
* LIMITS: finding 19's scope applies — the explicit-guard spelling is used here
|
||||
* because it is the shape the original shows, not because it generalises; it was
|
||||
* recorded as ineffective on a row with no such frame. The parameter types
|
||||
* (`char *`, `int` value, `int` count) are hypotheses read from the instruction
|
||||
* shape. The trailing `nop` is a scheduling fill, not source. Only the compiled
|
||||
* bytes are evidence.
|
||||
*/
|
||||
|
||||
void func_800F6570(char *dest, int value, int count) {
|
||||
int remaining = count - 1;
|
||||
|
||||
if (count != 0)
|
||||
do {
|
||||
*dest++ = value;
|
||||
} while (--remaining != -1);
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
/*
|
||||
* func_800F7A94 — 24 bytes at 0x800F7A94..0x800F7AAC
|
||||
*
|
||||
* Reads a 16-bit value out of a heap/global structure, replaces it, and returns
|
||||
* the previous value. The original reloads the structure pointer through the
|
||||
* symbol macro and lets the store land in the `jr ra` delay slot.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lui v1,0x8012 ; symbol load expands into v1 itself
|
||||
* lw v1,-1196(v1) ; v1 = D_8011FB54 (a pointer)
|
||||
* nop ; maspsx load-delay nop
|
||||
* lhu v0,0(v1) ; v0 = *v1
|
||||
* jr ra
|
||||
* sh a0,0(v1) ; *v1 = value (delay slot)
|
||||
*
|
||||
* LIMITS: the symbol name D_8011FB54, the `unsigned short` element type and the
|
||||
* `unsigned short` parameter type are hypotheses read from the instruction
|
||||
* shape. The `nop` is maspsx's load-delay fill, not source. Only the compiled
|
||||
* bytes are evidence.
|
||||
*/
|
||||
|
||||
extern int D_8011FB54;
|
||||
|
||||
unsigned short func_800F7A94(unsigned short value) {
|
||||
unsigned short *p = (unsigned short *)D_8011FB54;
|
||||
unsigned short previous = *p;
|
||||
|
||||
*p = value;
|
||||
return previous;
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
/*
|
||||
* func_800FF6A4 — 56 bytes at 0x800FF6A4..0x800FF6DC
|
||||
*
|
||||
* Searches a singly-linked list for the node whose key field equals the argument
|
||||
* and returns that node (or null). The list head is a gp-relative global, so the
|
||||
* symbol needs a `gp` registry row; the routine itself sets up no frame.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lw v1,2080(gp) ; node = g_80122158 (list head)
|
||||
* nop
|
||||
* beqz v1,0x800FF6D4 ; if (node == 0) return 0
|
||||
* nop
|
||||
* B4: lw v0,8(v1) ; node->key_08
|
||||
* nop
|
||||
* beq v0,a0,0x800FF6D4 ; if (key == want) break
|
||||
* nop
|
||||
* lw v1,12(v1) ; node = node->next_0c
|
||||
* nop
|
||||
* bnez v1,0x800FF6B4 ; loop while node != 0
|
||||
* nop
|
||||
* D4: jr ra
|
||||
* move v0,v1 ; return node (delay slot)
|
||||
*
|
||||
* The `break` form (rather than two early `return`s) is what keeps a single
|
||||
* epilogue with the result already in `v1`; two returns would duplicate the
|
||||
* `jr ra`.
|
||||
*
|
||||
* LIMITS: the symbol name g_80122158, the offsets (key at 0x08, next at 0x0c) and
|
||||
* the parameter type are hypotheses read from the instruction shape; the offset
|
||||
* 2080 is a fact about this executable's gp layout (gp = 0x80121938, cookbook
|
||||
* findings 10 and 14). The `gp` marker on the symbol is required and is requested
|
||||
* from the coordinator, not edited into the tracked registry by this session.
|
||||
* Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
extern int g_80122158;
|
||||
|
||||
int func_800FF6A4(int wanted) {
|
||||
int node = g_80122158;
|
||||
|
||||
while (node != 0) {
|
||||
if (*(int *)(node + 8) == wanted)
|
||||
break;
|
||||
node = *(int *)(node + 12);
|
||||
}
|
||||
|
||||
return node;
|
||||
}
|
||||
@@ -0,0 +1,28 @@
|
||||
/*
|
||||
* func_8010097C — 28 bytes at 0x8010097C..0x80100998
|
||||
*
|
||||
* Clears one bit in a 16-bit field at offset 0x40 of a structure. The bit index
|
||||
* arrives in a register, so the mask is built with a variable shift (`sllv`)
|
||||
* rather than a constant one.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* li v0,1 ; mask = 1
|
||||
* sllv v0,v0,a1 ; mask <<= bit_index
|
||||
* lhu v1,64(a0) ; field = *(unsigned short *)(p + 0x40)
|
||||
* nor v0,zero,v0 ; ~mask
|
||||
* and v1,v1,v0 ; field &= ~mask
|
||||
* jr ra
|
||||
* sh v1,64(a0) ; *(unsigned short *)(p + 0x40) = field (delay slot)
|
||||
*
|
||||
* LIMITS: the parameter types (`char *` for the base, `int` for the bit index)
|
||||
* and the field's offset and width are hypotheses read from the instruction
|
||||
* shape. The return type is `void`: `v0` holds only the mask scratch, and no
|
||||
* value is live in it at the `jr ra`. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
void func_8010097C(char *base, int bit_index) {
|
||||
unsigned short field = *(unsigned short *)(base + 64);
|
||||
|
||||
field &= ~(1 << bit_index);
|
||||
*(unsigned short *)(base + 64) = field;
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
/*
|
||||
* func_80101244 — 72 bytes at 0x80101244..0x8010128C
|
||||
*
|
||||
* Leaf routine that decodes a variable-length 7-bit-per-byte integer and reports
|
||||
* how many bytes it consumed.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* move a2,a0 00803021 p = a0
|
||||
* lbu a0,0x0(a2) 90c40000 value = *p
|
||||
* nop 00000000 load-delay slot
|
||||
* andi v0,a0,0x80 30820080 v0 = value & 0x80
|
||||
* beq v0,zero,0x80101280 10400009 if (!v0) goto the store
|
||||
* li a3,0x1 24070001 count = 1 (delay slot)
|
||||
* 0x8010125c:
|
||||
* andi a0,a0,0x7f 3084007f value &= 0x7f
|
||||
* 0x80101260:
|
||||
* addiu a2,a2,0x1 24c60001 p++
|
||||
* addiu a3,a3,0x1 24e70001 count++
|
||||
* lbu v0,0x0(a2) 90c20000 v0 = *p
|
||||
* sll a0,a0,0x7 000421c0 value <<= 7
|
||||
* andi v1,v0,0x7f 3043007f v1 = *p & 0x7f
|
||||
* andi v0,v0,0x80 30420080 v0 = *p & 0x80
|
||||
* bne v0,zero,0x80101260 1440fffc if (*p & 0x80) loop
|
||||
* addu a0,a0,v1 00832021 value += v1 (delay slot)
|
||||
* 0x80101280:
|
||||
* sw a3,0x0(a1) aca70000 *a1 = count
|
||||
* jr ra 03e00008
|
||||
* move v0,a0 00801021 v0 = value (delay slot)
|
||||
*
|
||||
* The first byte's high bit decides whether the value continues, and the
|
||||
* `andi a0,a0,0x7f` that strips the continuation bit sits **outside** the loop's
|
||||
* back edge (the branch at 0x80101278 targets 0x80101260, past it), so the mask is
|
||||
* applied once for the first byte only. That makes the body a `do`/`while` whose
|
||||
* condition tests the byte just loaded, and the loop's accumulation
|
||||
* (`value = (value << 7) + (*p & 0x7f)`) has its add scheduled into the branch
|
||||
* delay slot.
|
||||
*
|
||||
* The `lbu` loads make the bytes unsigned, and the pointer walks one byte at a
|
||||
* time. The count is stored through the second argument before the value is
|
||||
* returned, and the `lbu` of the first byte is followed by a load-delay `nop`.
|
||||
*
|
||||
* LIMITS: the function name, the claim that this is a 7-bit continuation encoding
|
||||
* and the meaning of the second argument are hypotheses; only the bytes are
|
||||
* evidence. The accumulator is written as `int`; a wider or unsigned type is not
|
||||
* distinguishable from these instructions.
|
||||
*/
|
||||
|
||||
int func_80101244(unsigned char *a0, int *a1)
|
||||
{
|
||||
unsigned char *p = a0;
|
||||
int value = *p;
|
||||
int count = 1;
|
||||
|
||||
if (value & 0x80) {
|
||||
value &= 0x7f;
|
||||
do {
|
||||
p++;
|
||||
count++;
|
||||
value = (value << 7) + (*p & 0x7f);
|
||||
} while (*p & 0x80);
|
||||
}
|
||||
|
||||
*a1 = count;
|
||||
|
||||
return value;
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
/*
|
||||
* func_80101CAC — 48 bytes at 0x80101CAC..0x80101CDC
|
||||
*
|
||||
* Loads five consecutive words from a structure and writes them into the GTE's
|
||||
* control registers $0..$4 — the rotation matrix, packed two values per register
|
||||
* across five registers.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* 0x8C880000 lw t0,0(a0) ; src[0]
|
||||
* 0x8C890004 lw t1,4(a0) ; src[1]
|
||||
* 0x8C8A0008 lw t2,8(a0) ; src[2]
|
||||
* 0x8C8B000C lw t3,12(a0) ; src[3]
|
||||
* 0x8C8C0010 lw t4,16(a0) ; src[4]
|
||||
* 0x48C80000 ctc2 t0,$0 ; RT1RT2
|
||||
* 0x48C94800 ctc2 t1,$1 ; RT3RT21
|
||||
* 0x48CA5000 ctc2 t2,$2 ; RT22RT23
|
||||
* 0x48CB5800 ctc2 t3,$3 ; RT31RT32
|
||||
* 0x48CC6000 ctc2 t4,$4 ; RT33
|
||||
* 0x03E00008 jr ra
|
||||
* 0x00000000 nop
|
||||
*
|
||||
* The statements go through include/gtemac.h (`gte_ldRT1RT2` … `gte_ldRT33`),
|
||||
* the re-derived macro form the project convention requires; the loads stay in
|
||||
* C. The five register bindings are load-bearing and are NOT inline assembly: a
|
||||
* GNU C local register variable is a documented register-name binding, the same
|
||||
* mechanism `register int sp __asm__("$29")` uses for the stack accessor
|
||||
* (cookbook finding 24), which docs/MATCHING_CONVENTIONS.md states needs no
|
||||
* exemption. Without them `cc1` is free to allocate the five values to whatever
|
||||
* registers it likes AND to reorder the loads; the original's allocation is
|
||||
* t0..t4 in order.
|
||||
*
|
||||
* The register numbers are the evidence, not a derivation: the control-register
|
||||
* numbering this executable uses is the standard map unshifted for 0..5 (this
|
||||
* body) and standard+7 from RBK onward (cookbook findings 24 and the header's
|
||||
* map). $0..$4 holding a 3x3 rotation matrix packed two-per-register is the
|
||||
* standard reading of this exact sequence.
|
||||
*
|
||||
* LIMITS: the source is `int[5]` and the five register names are hypotheses read
|
||||
* from the instruction shape. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
#include "../include/gtemac.h"
|
||||
|
||||
void func_80101CAC(int *src) {
|
||||
register int m0 __asm__("$8");
|
||||
register int m1 __asm__("$9");
|
||||
register int m2 __asm__("$10");
|
||||
register int m3 __asm__("$11");
|
||||
register int m4 __asm__("$12");
|
||||
|
||||
m0 = src[0];
|
||||
m1 = src[1];
|
||||
m2 = src[2];
|
||||
m3 = src[3];
|
||||
m4 = src[4];
|
||||
|
||||
gte_ldRT1RT2(m0);
|
||||
gte_ldRT3RT21(m1);
|
||||
gte_ldRT22RT23(m2);
|
||||
gte_ldRT31RT32(m3);
|
||||
gte_ldRT33(m4);
|
||||
}
|
||||
@@ -0,0 +1,51 @@
|
||||
/*
|
||||
* func_801027CC — 44 bytes at 0x801027CC..0x801027F8
|
||||
*
|
||||
* Leaf routine that copies `n` words from one buffer to another, skipping the
|
||||
* loop entirely when `n` is zero.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* beq a2,zero,0x801027f0 10c00006 if (n == 0) goto the epilogue
|
||||
* addu v1,zero,zero 00001821 i = 0 (delay slot)
|
||||
* 0x801027d4:
|
||||
* lw v0,0x0(a1) 8ca20000 v0 = *src
|
||||
* addiu a1,a1,0x4 24a50004 src += 4
|
||||
* addiu v1,v1,0x1 24630001 i++
|
||||
* sw v0,0x0(a0) ac820000 *dst = v0
|
||||
* sltu v0,v1,a2 0066102b v0 = (i < n) unsigned
|
||||
* bne v0,zero,0x801027d4 1440fffa if (v0) loop
|
||||
* addiu a0,a0,0x4 24840004 dst += 4 (delay slot)
|
||||
* 0x801027f0:
|
||||
* jr ra 03e00008
|
||||
* nop 00000000 (delay slot)
|
||||
*
|
||||
* This is cookbook finding 19's shape — an explicit guard plus a `do`/`while`
|
||||
* loop, which is what keeps cc1 from emitting the unused `subu sp,sp,8` frame
|
||||
* the plain `for (i = n - 1; i != -1; i--)` spelling produces. Here the guard is
|
||||
* `if (n == 0) return;` and the counter is initialised to zero in the branch's
|
||||
* delay slot.
|
||||
*
|
||||
* The bound test is `sltu`, not `slt`, so the comparison is **unsigned**: the
|
||||
* count is taken as `unsigned int` (with the counter unsigned too, so the
|
||||
* comparison is between two unsigned values).
|
||||
*
|
||||
* LIMITS: the function name and the claim that this is a word-wise memcpy are
|
||||
* hypotheses; only the bytes are evidence. The four-byte stride is evidence
|
||||
* (`addiu ...,0x4`). The buffers are written as `int *` because every access is
|
||||
* 32-bit; no alignment or aliasing claim is made.
|
||||
*/
|
||||
|
||||
void func_801027CC(int a0, int a1, unsigned int a2)
|
||||
{
|
||||
unsigned int i = 0;
|
||||
|
||||
if (a2 == 0)
|
||||
return;
|
||||
|
||||
do {
|
||||
*(int *)a0 = *(int *)a1;
|
||||
a1 += 4;
|
||||
i++;
|
||||
a0 += 4;
|
||||
} while (i < a2);
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
/*
|
||||
* func_80102FD4 — 12 bytes at 0x80102FD4..0x80102FE0
|
||||
*
|
||||
* Leaf routine whose whole body is a single COP2 control-register write.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* ctc2 a0,$26 48c4d000 — GTE control register H
|
||||
* jr ra 03e00008
|
||||
* nop 00000000 (delay slot)
|
||||
*
|
||||
* The argument arrives in a0 and the instruction encoding's rt field is 4
|
||||
* ($a0), so the source takes the value as a parameter and the compiler is left
|
||||
* to place it. Register $26 is the GTE H register in this executable's
|
||||
* numbering (include/gtemac.h records that map, proved by 0x80103A94's
|
||||
* `cfc2 v0,$26`); the statement goes through the header's re-derived
|
||||
* `gte_ldH` macro, the form the project convention requires for coprocessor
|
||||
* instructions. Nothing else in this file is asm.
|
||||
*
|
||||
* LIMITS: the function name, the parameter name and the claim that the caller
|
||||
* passes a projection-plane distance are hypotheses. Only the compiled bytes
|
||||
* are evidence. The routine writes no memory and returns nothing observable —
|
||||
* v0 is untouched at the `jr ra`, so the C return type is void.
|
||||
*/
|
||||
|
||||
#include "../include/gtemac.h"
|
||||
|
||||
void func_80102FD4(int h)
|
||||
{
|
||||
gte_ldH(h);
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
/* func_80103B54 — 0x80103B54..0x80103B60 (12 bytes).
|
||||
*
|
||||
* Original words:
|
||||
* 0x48C4D800 ctc2 a0,$27
|
||||
* 0x03E00008 jr ra
|
||||
* 0x00000000 nop
|
||||
*
|
||||
* A GTE control-register write, the sibling of the registered func_80103B60
|
||||
* (`ctc2 a0,$28`). Control register $27 is DQA in this executable's numbering
|
||||
* (proved by the sibling's $28 = DQB, whose number is the standard one).
|
||||
*
|
||||
* The COP2 instruction is stated through include/gtemac.h (`gte_ldDQA`), the
|
||||
* re-derived macro form the project convention requires; the epilogue is
|
||||
* ordinary compiler output.
|
||||
*
|
||||
* LIMITS: the register number $27 is read straight from the original word
|
||||
* 0x48C4D800 (rd field = 0b11011). The parameter name is a hypothesis; only the
|
||||
* bytes are evidence.
|
||||
*/
|
||||
|
||||
#include "../include/gtemac.h"
|
||||
|
||||
void func_80103B54(int dqa)
|
||||
{
|
||||
gte_ldDQA(dqa);
|
||||
}
|
||||
@@ -0,0 +1,51 @@
|
||||
/* func_80103B6C — 0x80103B6C..0x80103B8C (32 bytes).
|
||||
*
|
||||
* Original words:
|
||||
* 0x00042100 sll a0,a0,4
|
||||
* 0x00052900 sll a1,a1,4
|
||||
* 0x00063100 sll a2,a2,4
|
||||
* 0x48C4A800 ctc2 a0,$21
|
||||
* 0x48C5B000 ctc2 a1,$22
|
||||
* 0x48C6B800 ctc2 a2,$23
|
||||
* 0x03E00008 jr ra
|
||||
* 0x00000000 nop
|
||||
*
|
||||
* Ghidra renders the three consecutive ctc2 words as one `ldfcdir a0,a1,a2`
|
||||
* pseudo-op (cookbook finding 25: the PSX loader collapses COP2 sequences, so a
|
||||
* Ghidra instruction count is not a size). The raw words above are the
|
||||
* authority and the extent is 32 bytes, not 20.
|
||||
*
|
||||
* The rd fields are 0b10101, 0b10110, 0b10111 = $21/$22/$23, the far-color
|
||||
* registers RFC/GFC/BFC — NOT the $13/$14/$15 that cookbook finding 24 records
|
||||
* for the light-direction triple (a different body). Each register is scaled by
|
||||
* 4 before the write.
|
||||
*
|
||||
* The three `sll` are C expressions, not asm: the shift is arithmetic and the
|
||||
* project convention extends to coprocessor instructions only. The `ctc2`s go
|
||||
* through include/gtemac.h (`gte_ldRFC/gte_ldGFC/gte_ldBFC`).
|
||||
*
|
||||
* Each shift is bound to a named local BEFORE the asm statements. Writing the
|
||||
* shift directly as the asm operand instead (`"r"(rfc << 4)`) emits the shifts
|
||||
* and the `ctc2`s interleaved — sll,ctc2,sll,ctc2,sll,ctc2, 10 differing bytes
|
||||
* with the instruction count still 32. Naming the locals first is what makes
|
||||
* cc1 emit all three shifts ahead of the first `ctc2`, which is the original's
|
||||
* order. This is a source-shape lever of the finding-28 family: it applies
|
||||
* because the count was already right and only the order was wrong.
|
||||
*
|
||||
* LIMITS: the register numbers are read from the words above; the parameter
|
||||
* names are hypotheses. That naming the locals reorders the emission is an
|
||||
* observation about this compiler at -O2, not a rule.
|
||||
*/
|
||||
|
||||
#include "../include/gtemac.h"
|
||||
|
||||
void func_80103B6C(int rfc, int gfc, int bfc)
|
||||
{
|
||||
int r = rfc << 4;
|
||||
int g = gfc << 4;
|
||||
int b = bfc << 4;
|
||||
|
||||
gte_ldRFC(r);
|
||||
gte_ldGFC(g);
|
||||
gte_ldBFC(b);
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
/* func_80104C38 — 0x80104C38..0x80104C60 (40 bytes).
|
||||
*
|
||||
* Original words:
|
||||
* 0x3C038012 lui v1,0x8012
|
||||
* 0x8C631E74 lw v1,-0x18c(v1)
|
||||
* 0x00000000 nop
|
||||
* 0x94620004 lhu v0,0x4(v1) <- loop top
|
||||
* 0x00000000 nop
|
||||
* 0x30420002 andi v0,v0,0x2
|
||||
* 0x1040FFFC beq v0,zero,loop
|
||||
* 0x00000000 _nop (delay slot)
|
||||
* 0x03E00008 jr ra
|
||||
* 0x00000000 nop
|
||||
*
|
||||
* A busy-wait on a hardware status bit. The polled register is reached through
|
||||
* a global pointer, read as an unsigned halfword at displacement 4, and the
|
||||
* loop spins while bit 1 is clear.
|
||||
*
|
||||
* The `lui`/`lw` pair is the macro expansion of a symbol load into the same
|
||||
* register (cookbook finding 2), so the global holds the pointer. Its address is
|
||||
* **0x8011FE74**, not 0x80121E74: `lui v1,0x8012` with the displacement the
|
||||
* disassembler prints as -396 is the sign-adjusted `%lo` split (finding 4), so
|
||||
* the effective address is 0x80120000 - 0x18C. Reading the `%lo` as a positive
|
||||
* halfword and naming the symbol 0x80121E74 is a mistake made on attempt 1; the
|
||||
* two candidate symbols are 8 KB apart and the wrong one gave 2 differing bytes.
|
||||
*
|
||||
* The pointer load is emitted **once, before the loop**, which is the compiler
|
||||
* correctly hoisting it: the pointer variable is not volatile, so its value
|
||||
* cannot change and re-reading it would be redundant. Only the halfword it
|
||||
* points at is re-read every iteration, which is what makes the pointee
|
||||
* `volatile`.
|
||||
*
|
||||
* ATTEMPT 2 (recorded): binding the global to a named `volatile unsigned
|
||||
* short *status` local leaves exactly **one** differing byte at 0x80104C50 — the
|
||||
* loop back-edge displacement. The sequence and the load-delay `nop` are both in
|
||||
* the right place, but the candidate branches to the `nop` at 0x80104C40 while
|
||||
* the original branches past it to the `lhu` at 0x80104C44, so the local spelling
|
||||
* puts the loop's top label one instruction early. Attempt 3 below drops the
|
||||
* local and indexes the global directly, which changes where cc1 places that
|
||||
* label.
|
||||
*
|
||||
* LIMITS: the symbol is named for its address and its type is inferred from the
|
||||
* single use. Whether it is a pointer to one register or to the base of a
|
||||
* structure whose word 1 is the status register is not observable. Nothing here
|
||||
* proves the loop terminates, and the routine has no timeout — a fact about the
|
||||
* original, not a defect in this reconstruction.
|
||||
*/
|
||||
|
||||
extern volatile unsigned short *D_8011FE74;
|
||||
|
||||
void func_80104C38(void)
|
||||
{
|
||||
while ((D_8011FE74[2] & 2) == 0)
|
||||
;
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
/*
|
||||
* func_8010543C — 56 bytes at 0x8010543C..0x80105474
|
||||
*
|
||||
* Leaf routine that computes a byte size from three fields of an object: a
|
||||
* half-rounded byte count, a rounded-up 5-byte-stride field, and a 16-bit
|
||||
* length.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lbu v0,0xe3(a0) 908200e3 v0 = *(unsigned char *)(a0 + 0xe3)
|
||||
* lbu a1,0xe9(a0) 908500e9 a1 = *(unsigned char *)(a0 + 0xe9)
|
||||
* lhu a0,0xec(a0) 948400ec a0 = *(unsigned short *)(a0 + 0xec)
|
||||
* addiu v0,v0,0x1 24420001 v0 += 1
|
||||
* sra v0,v0,0x1 00021043 v0 >>= 1 arithmetic
|
||||
* sll v0,v0,0x2 00021080 v0 *= 4
|
||||
* sll v1,a1,0x2 00051880 v1 = a1 * 4
|
||||
* addu v1,v1,a1 00651821 v1 = a1 * 4 + a1 (= a1 * 5)
|
||||
* addiu v1,v1,0x3 24630003 v1 += 3
|
||||
* andi v1,v1,0xffc 30630ffc v1 &= 0xffc
|
||||
* addiu v1,v1,0x4 24630004 v1 += 4
|
||||
* addu v0,v0,v1 00431021 v0 += v1
|
||||
* jr ra 03e00008
|
||||
* addu v0,v0,a0 00441021 v0 += a0 (delay slot)
|
||||
*
|
||||
* Three loads with no stalls in between: they target different registers and
|
||||
* nothing reads them until the arithmetic, so no load-delay `nop` is needed.
|
||||
*
|
||||
* The `sra` is the tell for the first term: `(x + 1) >> 1` is emitted as an
|
||||
* **arithmetic** shift, which this compiler uses for a signed division by two, so
|
||||
* the value is held in a signed `int` even though it arrives through `lbu` (an
|
||||
* unsigned byte load). The `andi ...,0xffc` and the `addiu ...,4` show the source
|
||||
* masks and then offsets the field explicitly, and `a1 * 5` is strength-reduced
|
||||
* to `sll`+`addu` because five is not a power of two.
|
||||
*
|
||||
* LIMITS: the function name and the meaning of the three fields are hypotheses;
|
||||
* only the bytes are evidence. The field at +0xec is a 16-bit load, the other two
|
||||
* are 8-bit. Whether the original wrote the mask as `& 0xffc` on an int or as a
|
||||
* bitfield access is not recoverable; the mask is written literally here because
|
||||
* that is what `andi` shows.
|
||||
*/
|
||||
|
||||
int func_8010543C(char *a0)
|
||||
{
|
||||
int x = *(unsigned char *)(a0 + 0xe3);
|
||||
int y = *(unsigned char *)(a0 + 0xe9);
|
||||
int z = *(unsigned short *)(a0 + 0xec);
|
||||
int a = ((x + 1) / 2) * 4;
|
||||
int b = ((y * 5 + 3) & 0xffc) + 4;
|
||||
|
||||
return a + b + z;
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
/*
|
||||
* func_80105B68 — 32 bytes at 0x80105B68..0x80105B88
|
||||
*
|
||||
* Initialises four fields of a structure: a length byte, a self-referential
|
||||
* interior pointer, a tag byte taken from the caller, and a small status byte.
|
||||
* The final constant is materialised once into `v0` and stored from there in the
|
||||
* `jr ra` delay slot, so the function returns nothing — see cookbook finding 13
|
||||
* (a store-only function leaves its constant in `v0` as scratch, and writing
|
||||
* `return 1` would cost an extra instruction).
|
||||
*
|
||||
* The observed instructions are:
|
||||
* li v0,76 ; 0x4c
|
||||
* sb v0,55(a0) ; p->byte_37 = 76
|
||||
* addiu v0,a0,36 ; v0 = p + 36
|
||||
* sw v0,44(a0) ; p->ptr_2c = p + 36
|
||||
* li v0,1
|
||||
* sb a1,36(a0) ; p->byte_24 = tag
|
||||
* jr ra
|
||||
* sb v0,54(a0) ; p->byte_36 = 1 (delay slot)
|
||||
*
|
||||
* LIMITS: the parameter types (`char *` base, `char` tag) and every field offset
|
||||
* are hypotheses read from the instruction shape; the tag type is a hypothesis
|
||||
* in particular, since an `sb` of a parameter needs no mask and an `int` would
|
||||
* emit the same bytes. The meaning of the constants 76 and 1 is unknown and is
|
||||
* not guessed here. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
void func_80105B68(char *base, char tag) {
|
||||
base[55] = 76;
|
||||
*(int *)(base + 44) = (int)(base + 36);
|
||||
base[36] = tag;
|
||||
base[54] = 1;
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
/*
|
||||
* func_80105BA8 — 32 bytes at 0x80105BA8..0x80105BC8
|
||||
*
|
||||
* Leaf routine that initialises a small object: a tag byte, a self-pointer,
|
||||
* a caller-supplied byte, and a one-byte flag.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* li v0,0x47 24020047 v0 = 0x47
|
||||
* sb v0,0x37(a0) a0820037 *(char *)(a0 + 0x37) = 0x47
|
||||
* addiu v0,a0,0x24 24820024 v0 = a0 + 0x24
|
||||
* sw v0,0x2c(a0) ac82002c *(int *)(a0 + 0x2c) = a0 + 0x24
|
||||
* li v0,0x1 24020001 v0 = 1
|
||||
* sb a1,0x24(a0) a0850024 *(char *)(a0 + 0x24) = a1
|
||||
* jr ra 03e00008
|
||||
* sb v0,0x36(a0) a0820036 *(char *)(a0 + 0x36) = 1 (delay slot)
|
||||
*
|
||||
* The second argument is stored with `sb` and is **not** masked first, so it is
|
||||
* not a `char` parameter: cookbook finding 7 says an unsigned `char` argument
|
||||
* would carry an `andi ...,0xff` (as func_80017AD4 does). It is therefore taken
|
||||
* as `int` and truncated by the store itself.
|
||||
*
|
||||
* The self-pointer at +0x2c is materialised as `addiu v0,a0,0x24` and stored —
|
||||
* an address computed from the argument, not a symbol, so no registry row.
|
||||
*
|
||||
* LIMITS: the function name, the structure and every offset are hypotheses read
|
||||
* off the disassembly. The widths are evidence: two 8-bit stores of constants,
|
||||
* one 8-bit store of the argument, one 32-bit self-pointer store. `0x47` is a
|
||||
* tag value with no established meaning.
|
||||
*/
|
||||
|
||||
void func_80105BA8(int a0, int a1)
|
||||
{
|
||||
*(char *)(a0 + 0x37) = 0x47;
|
||||
*(int *)(a0 + 0x2c) = a0 + 0x24;
|
||||
*(char *)(a0 + 0x24) = a1;
|
||||
*(char *)(a0 + 0x36) = 1;
|
||||
}
|
||||
@@ -0,0 +1,45 @@
|
||||
/*
|
||||
* func_80107AA0 — 64 bytes at 0x80107AA0..0x80107AE0
|
||||
*
|
||||
* Clears or sets the low bit of a 16-bit field on a global structure, depending
|
||||
* on a flag argument.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* bnez a0,0x80107AC0 ; if (set != 0) take the OR arm
|
||||
* nop
|
||||
* lui v1,0x8012 ; symbol load expands into v1 itself
|
||||
* lw v1,4168(v1) ; v1 = D_80121048 (a structure pointer)
|
||||
* nop
|
||||
* lhu v0,426(v1) ; v0 = ->half_1aa
|
||||
* j 0x80107AD8
|
||||
* andi v0,v0,0xfffe ; v0 &= ~1 (delay slot)
|
||||
* C0: lui v1,0x8012 ; RELOAD the pointer
|
||||
* lw v1,4168(v1)
|
||||
* nop
|
||||
* lhu v0,426(v1)
|
||||
* nop
|
||||
* ori v0,v0,0x1 ; v0 |= 1
|
||||
* D8: jr ra
|
||||
* sh v0,426(v1) ; ->half_1aa = v0 (delay slot)
|
||||
*
|
||||
* The single shared store at the end is a `cc1` CROSS-JUMP, not a shared
|
||||
* statement: the source stores inside each arm (`&= ` in one, `|= ` in the
|
||||
* other) and the jump optimiser merges the two identical `sh` into one. Writing
|
||||
* the source the other way round — computing a `value` local in both arms and
|
||||
* storing once — makes the store reload the global pointer a THIRD time and
|
||||
* costs 4 bytes. So the merged store must be produced by the optimiser rather
|
||||
* than written by hand.
|
||||
*
|
||||
* LIMITS: the symbol name D_80121048, the field offset (426) and the field's
|
||||
* `unsigned short` type are hypotheses read from the instruction shape; the `lhu`
|
||||
* is what shows the field is unsigned. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
extern int D_80121048;
|
||||
|
||||
void func_80107AA0(int set) {
|
||||
if (set == 0)
|
||||
*(unsigned short *)((char *)D_80121048 + 426) &= 0xFFFE;
|
||||
else
|
||||
*(unsigned short *)((char *)D_80121048 + 426) |= 1;
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
/*
|
||||
* func_80107AE0 — 64 bytes at 0x80107AE0..0x80107B20
|
||||
*
|
||||
* Leaf routine that sets or clears bit 2 of a 16-bit field in a global
|
||||
* structure, returning the new value.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* bne a0,zero,0x80107b00 14800007 if (a0 != 0) goto the set path
|
||||
* nop 00000000 (delay slot)
|
||||
* lui v1,0x8012 3c038012 \
|
||||
* lw v1,0x1048(v1) 8c631048 / v1 = *(int *)0x80121048 (D_80121048)
|
||||
* nop 00000000 load-delay slot
|
||||
* lhu v0,0x1aa(v1) 946201aa v0 = *(unsigned short *)(v1 + 0x1aa)
|
||||
* j 0x80107b18 08101ec6 goto the store
|
||||
* andi v0,v0,0xfffb 3042fffb v0 &= ~4 (delay slot)
|
||||
* 0x80107b00:
|
||||
* lui v1,0x8012 3c038012 \
|
||||
* lw v1,0x1048(v1) 8c631048 / v1 = *(int *)0x80121048
|
||||
* nop 00000000 load-delay slot
|
||||
* lhu v0,0x1aa(v1) 946201aa v0 = *(unsigned short *)(v1 + 0x1aa)
|
||||
* nop 00000000 load-delay slot
|
||||
* ori v0,v0,0x4 34420004 v0 |= 4
|
||||
* 0x80107b18:
|
||||
* jr ra 03e00008
|
||||
* sh v0,0x1aa(v1) a46201aa *(unsigned short *)(v1 + 0x1aa) = v0 (delay slot)
|
||||
*
|
||||
* The base load is **duplicated in both arms** and the store after the merge uses
|
||||
* the arm's own v1, so the source reads the global inside each branch rather than
|
||||
* once before the condition. The address is the `lui`+`lw`-through-the-same-
|
||||
* register symbol form (cookbook finding 2), so it is written as the named global
|
||||
* `D_80121048`; 0x80121048 is outside the small-data window and is accessed
|
||||
* absolutely, so no `gp` marker is needed.
|
||||
*
|
||||
* The clear is an `andi` with `~4` and the set is an `ori` with `4`, so the source
|
||||
* writes `& ~4` and `| 4` on a 16-bit field; `~4` is spelled as 0xfffb by the
|
||||
* assembler's 16-bit immediate.
|
||||
*
|
||||
* LIMITS: the function name, the claim that the global is a pointer to a structure
|
||||
* and the meaning of bit 2 are hypotheses; only the bytes are evidence. The field
|
||||
* is 16-bit (`lhu`/`sh`). The duplicated base load is an observation about the
|
||||
* original's source shape, not a requirement of the C.
|
||||
*/
|
||||
|
||||
extern int D_80121048;
|
||||
|
||||
int func_80107AE0(int a0)
|
||||
{
|
||||
unsigned short v0;
|
||||
int v1;
|
||||
|
||||
if (a0 == 0) {
|
||||
v1 = D_80121048;
|
||||
v0 = *(unsigned short *)(v1 + 0x1aa) & ~4;
|
||||
} else {
|
||||
v1 = D_80121048;
|
||||
v0 = *(unsigned short *)(v1 + 0x1aa) | 4;
|
||||
}
|
||||
|
||||
*(unsigned short *)(v1 + 0x1aa) = v0;
|
||||
|
||||
return v0;
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
/*
|
||||
* func_80107FAC — 48 bytes at 0x80107FAC..0x80107FDC
|
||||
*
|
||||
* ORs a table word selected by a masked index into a global structure's flags
|
||||
* word, and returns whether the index is below three. The index is truncated to
|
||||
* 16 bits before use, and the table lookup is an array index (so the scaling is a
|
||||
* `sll` by 2 rather than a separate multiply).
|
||||
*
|
||||
* The observed instructions are:
|
||||
* andi v0,a0,0xffff ; i = index & 0xffff
|
||||
* sll a0,v0,0x2 ; i * 4, into the now-dead argument register
|
||||
* lui a1,0x8012 ; symbol load expands into a1 itself
|
||||
* lw a1,3312(a1) ; a1 = D_80120CF0 (a structure pointer)
|
||||
* lui at,0x8012 ; %hi of the table base
|
||||
* addu at,at,a0 ; table + i*4 (base first)
|
||||
* lw a0,3320(at) ; a0 = table[i] (%lo as displacement)
|
||||
* lw v1,4(a1) ; a1->word_04
|
||||
* slti v0,v0,3 ; return i < 3
|
||||
* or v1,v1,a0 ; a1->word_04 | table[i]
|
||||
* jr ra
|
||||
* sw v1,4(a1) ; a1->word_04 = result (delay slot)
|
||||
*
|
||||
* Two source-shape requirements are load-bearing here. The global pointer must be
|
||||
* materialised BEFORE the table element is read (the original loads a1 first),
|
||||
* and the table element must be consumed inside the OR expression rather than
|
||||
* bound to a named local — a named local makes `cc1` hold the scaled index in a
|
||||
* different register and costs 2 bytes in the `sll` and `addu` register fields.
|
||||
*
|
||||
* Note the two different base forms in one function: the table base at
|
||||
* 0x80120CF8 is a literal written with `%hi` + `addu` + `%lo`-as-displacement,
|
||||
* while D_80120CF0 is loaded as a pointer value through the symbol macro. The
|
||||
* `addu` operand order here is base-first, the same order the
|
||||
* address-shaped-symbol spelling produced at 0x80016174.
|
||||
*
|
||||
* LIMITS: the symbol name D_80120CF0, the table base 0x80120CF8, the element
|
||||
* stride (4), the field offset (4) and the comparison constant (3) are hypotheses
|
||||
* read from the instruction shape; what the structures mean is unknown and is not
|
||||
* guessed here. Only the compiled bytes are evidence.
|
||||
*/
|
||||
|
||||
extern int D_80120CF0;
|
||||
extern int D_80120CF8[];
|
||||
|
||||
int func_80107FAC(int index) {
|
||||
int i = index & 0xFFFF;
|
||||
int *p = (int *)D_80120CF0;
|
||||
|
||||
p[1] = p[1] | D_80120CF8[i];
|
||||
return i < 3;
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
/*
|
||||
* func_80107FDC — 52 bytes at 0x80107FDC..0x80108010
|
||||
*
|
||||
* Leaf routine that clears one 16-byte record when its index is in range, and
|
||||
* reports whether it did so.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* andi v1,a0,0xffff 3084ffff v1 = a0 & 0xffff
|
||||
* slti v0,v1,0x3 28620003 v0 = (v1 < 3) signed
|
||||
* beq v0,zero,0x80108004 10400007 if (!(v1 < 3)) goto the zero return
|
||||
* li v0,0x1 24020001 v0 = 1 (delay slot)
|
||||
* lui a0,0x8012 3c048012 \
|
||||
* lw a0,0xcf4(a0) 8c840cf4 / a0 = *(int *)0x80120CF4 (D_80120CF4)
|
||||
* sll v1,v1,0x4 00042100 v1 *= 16
|
||||
* addu v1,v1,a0 00641821 v1 = (index * 16) + base
|
||||
* j 0x80108008 08100002 goto epilogue
|
||||
* sh zero,0x0(v1) a4600000 *(short *)v1 = 0 (delay slot)
|
||||
* 0x80108004:
|
||||
* addu v0,zero,zero 00001021 v0 = 0
|
||||
* 0x80108008:
|
||||
* jr ra 03e00008
|
||||
* nop 00000000 (delay slot)
|
||||
*
|
||||
* The `andi ...,0xffff` on entry is a mask the source applies explicitly to a
|
||||
* 32-bit parameter: the bound test is `slti` — a **signed** compare — and an
|
||||
* `unsigned short` parameter would instead produce `sltiu` (measured: that
|
||||
* spelling differs by exactly the one byte of the `slti`/`sltiu` opcode). So the
|
||||
* parameter is `int` and the mask is written in the C.
|
||||
*
|
||||
* The record base is a symbol load through the same register (finding 2), so it
|
||||
* is written as `D_80120CF4`, not as a literal address. The address arithmetic is
|
||||
* **stride first, base second** (`sll` then `addu v1,v1,a0`), the spelling
|
||||
* finding 22 identifies as `(index * 16) + symbol` rather than
|
||||
* `symbol + index * 16`; the rule's stated scope (a symbol or `gp` base) applies
|
||||
* here.
|
||||
*
|
||||
* The `1` result is materialised in the branch delay slot, so the source order is
|
||||
* the early guard `if (index >= 3) return 0;` followed by the work and
|
||||
* `return 1;`.
|
||||
*
|
||||
* LIMITS: the function name, the record size of 16 bytes and the meaning of the
|
||||
* table are hypotheses; only the bytes are evidence. The cleared field is 16-bit
|
||||
* (`sh`), so the record is assumed to start with a halfword.
|
||||
*/
|
||||
|
||||
extern int D_80120CF4;
|
||||
|
||||
int func_80107FDC(int a0)
|
||||
{
|
||||
int i = a0 & 0xffff;
|
||||
|
||||
if (i >= 3)
|
||||
return 0;
|
||||
|
||||
*(short *)((i * 16) + D_80120CF4) = 0;
|
||||
|
||||
return 1;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
/*
|
||||
* func_80108024 — 16 bytes at 0x80108024..0x80108034
|
||||
*
|
||||
* Saves the current `gp` register value into a global. This is one member of a
|
||||
* small gp save/restore family in the CRT region: 0x80108024 saves gp to
|
||||
* D_80120D10, 0x80108034 saves gp to D_80120D14 and then restores gp from
|
||||
* D_80120D10, and 0x8010804C restores gp from D_80120D14.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lui at,0x8012 ; symbol store expands through $at
|
||||
* sw gp,0x0d10(at) ; D_80120D10 = gp
|
||||
* jr ra
|
||||
* nop ; delay slot left empty by cc1, filled by maspsx
|
||||
*
|
||||
* The stored value is the `gp` register itself, which plain C cannot name. The
|
||||
* source therefore uses a GNU C global register variable — a documented
|
||||
* register-name binding, not inline assembly (see docs/MATCHING_CONVENTIONS.md
|
||||
* and cookbook finding 24, where `register int sp __asm__("$29")` reproduces the
|
||||
* stack accessor with no exemption needed).
|
||||
*
|
||||
* LIMITS: the symbol name D_80120D10 and the `int` type are hypotheses read from
|
||||
* the instruction shape; only the compiled bytes are evidence. The offset 0x0d10
|
||||
* is a fact about this executable's globals, not about the compiler. The
|
||||
* surrounding family members (0x80108034, 0x8010804C) are excluded from the
|
||||
* worklist by name and are not claimed here.
|
||||
*/
|
||||
|
||||
register int gp __asm__("$28");
|
||||
|
||||
extern int D_80120D10;
|
||||
|
||||
void func_80108024(void) {
|
||||
D_80120D10 = gp;
|
||||
}
|
||||
@@ -0,0 +1,41 @@
|
||||
/*
|
||||
* func_80108690 — 28 bytes at 0x80108690..0x801086AC
|
||||
*
|
||||
* Leaf routine that masks two arguments to 15 bits and stores them as two
|
||||
* adjacent halfwords in a global structure reached through a fixed pointer.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* andi a0,a0,0x7fff 30847fff a0 &= 0x7fff
|
||||
* lui v0,0x8012 3c028012 \
|
||||
* lw v0,0x1048(v0) 8c421048 / v0 = *(int *)0x80121048 (D_80121048)
|
||||
* andi a1,a1,0x7fff 30a57fff a1 &= 0x7fff
|
||||
* sh a0,0x180(v0) a4440180 *(short *)(v0 + 0x180) = a0
|
||||
* jr ra 03e00008
|
||||
* sh a1,0x182(v0) a4450182 *(short *)(v0 + 0x182) = a1 (delay slot)
|
||||
*
|
||||
* The address is materialised as `lui`+`lw` with the *same* register twice,
|
||||
* which is this toolchain's symbol-load macro form (cookbook finding 2) — so
|
||||
* the base is written as the named global `D_80121048`, not as a literal
|
||||
* address (finding 5: a literal would give `lui`+`ori`). 0x80121048 is
|
||||
* 0x8F0 below gp (0x80121938), i.e. outside the small-data window, and the
|
||||
* original accesses it absolutely, so no `gp` marker and no registry row is
|
||||
* needed; the address-named symbol resolves implicitly.
|
||||
*
|
||||
* The first `andi` is hoisted above the load and the second is placed after
|
||||
* it; that ordering is left to cc1's scheduler rather than forced.
|
||||
*
|
||||
* LIMITS: the function name, the global's type and the meaning of the two
|
||||
* halfwords at +0x180/+0x182 are hypotheses; only the bytes are evidence. The
|
||||
* masks are written as `& 0x7fff` on `int` arguments, which is what makes cc1
|
||||
* emit `andi` rather than a truncating `sh` of the full register.
|
||||
*/
|
||||
|
||||
extern int D_80121048;
|
||||
|
||||
void func_80108690(int a0, int a1)
|
||||
{
|
||||
int v0 = D_80121048;
|
||||
|
||||
*(short *)(v0 + 0x180) = a0 & 0x7fff;
|
||||
*(short *)(v0 + 0x182) = a1 & 0x7fff;
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
/*
|
||||
* func_8010A888 — 40 bytes at 0x8010A888..0x8010A8B0
|
||||
*
|
||||
* Leaf routine that sets one bit in a 32-bit control word reached through a
|
||||
* fixed global pointer.
|
||||
*
|
||||
* The observed instructions are:
|
||||
* lui a0,0x8012 3c048012 \
|
||||
* lw a0,0x105c(a0) 8c84105c / a0 = *(int *)0x8012105C (D_8012105C)
|
||||
* lui v1,0xf0ff 3c03f0ff \
|
||||
* lw v0,0x0(a0) 8c820000 / v0 = *a0
|
||||
* ori v1,v1,0xffff 3463ffff v1 = 0xf0ffffff
|
||||
* and v0,v0,v1 00431024 v0 &= 0xf0ffffff
|
||||
* lui v1,0x2000 3c032000 v1 = 0x20000000
|
||||
* or v0,v0,v1 00431025 v0 |= 0x20000000
|
||||
* jr ra 03e00008
|
||||
* sw v0,0x0(a0) ac820000 *a0 = v0 (delay slot)
|
||||
*
|
||||
* The address is materialised as `lui`+`lw` through the same register, this
|
||||
* toolchain's symbol-load macro form (cookbook finding 2), so the base is the
|
||||
* named global `D_8012105C` and not a literal address (finding 5). 0x8012105C
|
||||
* is outside the small-data window and the original accesses it absolutely, so
|
||||
* no `gp` marker and no registry row are needed; the address-named symbol
|
||||
* resolves implicitly. The mask 0xf0ffffff is built as `lui 0xf0ff` + `ori
|
||||
* 0xffff` and the set bit as `lui 0x2000` alone (zero low half), which is what a
|
||||
* plain `&`/`|` on the loaded word produces.
|
||||
*
|
||||
* LIMITS: the function name and the meaning of the bit (a hardware control
|
||||
* word, since 0x8012105C is read as a pointer) are hypotheses; only the bytes
|
||||
* are evidence. The C is written with the mask and the OR value as literals
|
||||
* because the original materialises them as literals; whether the original
|
||||
* source spelled them as named constants is unknowable from the code.
|
||||
*/
|
||||
|
||||
extern int D_8012105C;
|
||||
|
||||
void func_8010A888(void)
|
||||
{
|
||||
int *p = (int *)D_8012105C;
|
||||
|
||||
*p = (*p & 0xf0ffffff) | 0x20000000;
|
||||
}
|
||||
Reference in New Issue
Block a user