diff --git a/config/regions.tsv b/config/regions.tsv index 6d9c3c7..15c5be8 100644 --- a/config/regions.tsv +++ b/config/regions.tsv @@ -13,11 +13,18 @@ 0x80012780 0x8001278C src/func_80012780.c 0x800127F0 0x8001281C src/func_800127F0.c 0x8001281C 0x80012834 src/func_8001281C.c +0x80012D8C 0x80012DBC src/func_80012D8C.c +0x80012DBC 0x80012DE8 src/func_80012DBC.c 0x80013C88 0x80013C90 src/func_80013C88.c +0x800160E8 0x80016110 src/func_800160E8.c 0x80016110 0x80016120 src/func_80016110.c +0x80016120 0x80016158 src/func_80016120.c 0x80016158 0x80016174 src/func_80016158.c +0x80016174 0x80016198 src/func_80016174.c +0x80016198 0x800161E0 src/func_80016198.c 0x80016E50 0x80016E68 src/func_80016E50.c 0x800171D8 0x80017200 src/func_800171D8.c +0x8001761C 0x80017660 src/func_8001761C.c 0x800179B8 0x800179CC src/func_800179B8.c 0x800179CC 0x800179D4 src/func_800179CC.c 0x800179D4 0x800179E0 src/func_800179D4.c @@ -27,6 +34,7 @@ 0x80017C60 0x80017C6C src/func_80017C60.c 0x80017D1C 0x80017D48 src/func_80017D1C.c 0x80017DD0 0x80017DF0 src/func_80017DD0.c +0x800182D4 0x800182F4 src/func_800182D4.c 0x80019B6C 0x80019B94 src/func_80019B6C.c 0x80019C04 0x80019C10 src/func_80019C04.c 0x8001AA9C 0x8001AAA8 src/func_8001AA9C.c @@ -36,6 +44,7 @@ 0x80021F88 0x80021FA8 src/func_80021F88.c 0x80022A60 0x80022A98 src/func_80022A60.c 0x80022FB8 0x80022FCC src/func_80022FB8.c +0x800230E4 0x8002311C src/func_800230E4.c 0x80024C14 0x80024C34 src/func_80024C14.c 0x80026180 0x800261C0 src/func_80026180.c 0x80026258 0x80026264 src/func_80026258.c @@ -48,11 +57,14 @@ 0x8002C7BC 0x8002C7EC src/func_8002C7BC.c 0x8002C888 0x8002C894 src/func_8002C888.c 0x8002C894 0x8002C8A4 src/func_8002C894.c +0x8002D1F8 0x8002D22C src/func_8002D1F8.c 0x8002D288 0x8002D2A0 src/func_8002D288.c gp=-D_80121F84 0x8002D2A0 0x8002D2BC src/func_8002D2A0.c 0x8002D2BC 0x8002D2D4 src/func_8002D2BC.c 0x8002DEB4 0x8002DF1C src/func_8002DEB4.c +0x8002E7C4 0x8002E7E4 src/func_8002E7C4.c 0x8002F2F8 0x8002F300 src/func_8002F2F8.c +0x8002F404 0x8002F450 src/func_8002F404.c 0x80030358 0x80030390 src/func_80030358.c 0x800321EC 0x800321F8 src/func_800321EC.c 0x80036308 0x80036328 src/func_80036308.c @@ -64,12 +76,21 @@ 0x8003B2F0 0x8003B320 src/func_8003B2F0.c 0x80042088 0x80042090 src/func_80042088.c 0x80042D64 0x80042D88 src/func_80042D64.c +0x80043D8C 0x80043DC4 src/func_80043D8C.c +0x8004C060 0x8004C090 src/func_8004C060.c 0x8004C090 0x8004C0AC src/func_8004C090.c +0x8004C0AC 0x8004C0F0 src/func_8004C0AC.c 0x8004C0F0 0x8004C110 src/func_8004C0F0.c +0x8004CEEC 0x8004CF0C src/func_8004CEEC.c +0x800516E0 0x800516FC src/func_800516E0.c +0x8005182C 0x80051864 src/func_8005182C.c +0x80052C98 0x80052CAC src/func_80052C98.c 0x80057DFC 0x80057E04 src/func_80057DFC.c 0x80058288 0x800582AC src/func_80058288.c +0x8005E3D0 0x8005E3F4 src/func_8005E3D0.c 0x80065B6C 0x80065B8C src/func_80065B6C.c 0x800681A4 0x800681E0 src/func_800681A4.c +0x800681E0 0x8006821C src/func_800681E0.c 0x80068470 0x8006848C src/func_80068470.c 0x80068F40 0x80068F6C src/func_80068F40.c 0x80068F98 0x80068FA8 src/func_80068F98.c @@ -80,9 +101,12 @@ 0x8006F944 0x8006F95C src/func_8006F944.c 0x8006FA78 0x8006FAA0 src/func_8006FA78.c 0x8007049C 0x800704BC src/func_8007049C.c +0x8007C4EC 0x8007C524 src/func_8007C4EC.c 0x8007DC40 0x8007DC4C src/func_8007DC40.c 0x8007DF00 0x8007DF34 src/func_8007DF00.c +0x800827A8 0x800827C4 src/func_800827A8.c 0x80082914 0x80082944 src/func_80082914.c +0x80083440 0x80083470 src/func_80083440.c 0x80083504 0x8008352C src/func_80083504.c 0x8008352C 0x8008355C src/func_8008352C.c 0x80085B80 0x80085B90 src/func_80085B80.c @@ -96,30 +120,44 @@ 0x8008B8E4 0x8008B8F4 src/func_8008B8E4.c 0x8008B8F4 0x8008B910 src/func_8008B8F4.c 0x8008F4A0 0x8008F4AC src/func_8008F4A0.c +0x8008F4F4 0x8008F508 src/func_8008F4F4.c 0x80090048 0x80090058 src/func_80090048.c 0x80090B64 0x80090B7C src/func_80090B64.c 0x800912D4 0x800912FC src/func_800912D4.c 0x800912FC 0x8009132C src/func_800912FC.c +0x80092088 0x800920BC src/func_80092088.c 0x800920BC 0x800920DC src/func_800920BC.c 0x800920DC 0x80092104 src/func_800920DC.c +0x80092130 0x8009214C src/func_80092130.c 0x800943C4 0x800943E0 src/func_800943C4.c +0x80099E14 0x80099E34 src/func_80099E14.c +0x8009D8A0 0x8009D8E0 src/func_8009D8A0.c 0x8009E8D0 0x8009E95C src/func_8009E8D0.c +0x8009F0E8 0x8009F120 src/func_8009F0E8.c +0x800A2F20 0x800A2F44 src/func_800A2F20.c 0x800A45E0 0x800A466C src/func_8009E8D0.c 0x800A74BC 0x800A74D0 src/func_800A74BC.c +0x800A8B48 0x800A8B8C src/func_800A8B48.c 0x800AA56C 0x800AA59C src/func_800AA56C.c 0x800ACC00 0x800ACC20 src/func_800ACC00.c 0x800AE0F4 0x800AE10C src/func_800AE0F4.c 0x800AF1FC 0x800AF20C src/func_800AF1FC.c 0x800B0E30 0x800B0E64 src/func_800B0E30.c +0x800B24EC 0x800B2534 src/func_800B24EC.c +0x800B3474 0x800B34A4 src/func_800B3474.c 0x800B5AF0 0x800B5AF8 src/func_80042088.c 0x800B6BDC 0x800B6C14 src/func_800B6BDC.c +0x800B7230 0x800B7264 src/func_800B7230.c 0x800BBDEC 0x800BBDF8 src/func_800BBDEC.c 0x800C5C84 0x800C5CBC src/func_800C5C84.c 0x800F3160 0x800F316C src/func_800F3160.c maspsx=off 0x800F3E70 0x800F3E88 src/func_800F3E70.c +0x800F5200 0x800F521C src/func_800F5200.c 0x800F5B40 0x800F5B70 src/func_800F5B40.c +0x800F6570 0x800F6594 src/func_800F6570.c 0x800F75D0 0x800F760C src/func_800F75D0.c 0x800F7A84 0x800F7A94 src/func_800F7A84.c +0x800F7A94 0x800F7AAC src/func_800F7A94.c 0x800F7FB4 0x800F7FD8 src/func_800F7FB4.c 0x800F8294 0x800F82A4 src/func_800F8294.c 0x800F8ACC 0x800F8ADC src/func_800F8ACC.c @@ -141,30 +179,49 @@ 0x800FE86C 0x800FE878 src/func_800FE86C.c 0x800FEEB8 0x800FEED0 src/func_800FEEB8.c 0x800FEFFC 0x800FF008 src/func_800FEFFC.c +0x800FF6A4 0x800FF6DC src/func_800FF6A4.c 0x80100318 0x80100334 src/func_80100318.c 0x80100964 0x8010097C src/func_80100964.c +0x8010097C 0x80100998 src/func_8010097C.c +0x80101244 0x8010128C src/func_80101244.c +0x80101CAC 0x80101CDC src/func_80101CAC.c +0x801027CC 0x801027F8 src/func_801027CC.c 0x80102B10 0x80102B2C src/func_80102B10.c maspsx=off +0x80102FD4 0x80102FE0 src/func_80102FD4.c 0x80103A94 0x80103AA0 src/func_80103A94.c +0x80103B54 0x80103B60 src/func_80103B54.c 0x80103B60 0x80103B6C src/func_80103B60.c +0x80103B6C 0x80103B8C src/func_80103B6C.c 0x80103C7C 0x80103CA0 src/func_800F7FB4.c 0x80103F84 0x80103FA8 src/func_800F7FB4.c 0x80103FCC 0x80103FDC src/func_80103FCC.c 0x80103FEC 0x80103FFC src/func_80103FEC.c +0x80104C38 0x80104C60 src/func_80104C38.c maspsx=off 0x80104C6C 0x80104CA0 src/func_80104C6C.c maspsx=off 0x8010513C 0x80105148 src/func_8010513C.c +0x8010543C 0x80105474 src/func_8010543C.c 0x80105B34 0x80105B54 src/func_80105B34.c 0x80105B54 0x80105B68 src/func_80105B54.c +0x80105B68 0x80105B88 src/func_80105B68.c 0x80105B88 0x80105BA8 src/func_80105B88.c +0x80105BA8 0x80105BC8 src/func_80105BA8.c 0x80105BC8 0x80105BDC src/func_80105BC8.c 0x80107A94 0x80107AA0 src/func_80107A94.c +0x80107AA0 0x80107AE0 src/func_80107AA0.c +0x80107AE0 0x80107B20 src/func_80107AE0.c 0x80107B20 0x80107B38 src/func_80107B20.c 0x80107E68 0x80107E74 src/func_80107E68.c 0x80107E74 0x80107E84 src/func_80107E74.c 0x80107E84 0x80107E90 src/func_80107E84.c +0x80107FAC 0x80107FDC src/func_80107FAC.c +0x80107FDC 0x80108010 src/func_80107FDC.c +0x80108024 0x80108034 src/func_80108024.c +0x80108690 0x801086AC src/func_80108690.c 0x80108710 0x80108734 src/func_80108710.c 0x80109300 0x80109314 src/func_80109300.c 0x80109314 0x80109338 src/func_800F8F9C.c 0x80109778 0x80109790 src/func_80109778.c 0x80109F38 0x80109F48 src/func_80109F38.c +0x8010A888 0x8010A8B0 src/func_8010A888.c 0x8010A8B0 0x8010A8D8 src/func_8010A8B0.c 0x8010AAA0 0x8010AABC src/func_8010AAA0.c diff --git a/config/symbols.tsv b/config/symbols.tsv index 79565a0..8dd9a33 100644 --- a/config/symbols.tsv +++ b/config/symbols.tsv @@ -377,4 +377,8 @@ func_80024668 0x80024668 func_8002F404 0x8002F404 func_800695D8 0x800695D8 func_800F4098 0x800F4098 +g_80122068 0x80122068 gp +g_80122158 0x80122158 gp g_80122354 0x80122354 +g_80122738 0x80122738 gp +g_8012277C 0x8012277C gp diff --git a/include/gtemac.h b/include/gtemac.h index aae6362..6176906 100644 --- a/include/gtemac.h +++ b/include/gtemac.h @@ -32,9 +32,11 @@ * LIMITS. Only the operations listed below are covered, and only for the * register numbers observed here. The control-register numbering used by the * original is NOT the 0..31 "textbook" GTE control layout: this executable - * writes DQB at $28, OFX/OFY at $24/$25, H at $26 and LR1LR2/LR3LG1/LG2LG3 at - * $13/$14/$15, so the numbers below are the evidence, not a derivation. The - * file is not an attempt to reconstruct Sony's gtemac.h. + * writes DQB at $28, DQA at $27, OFX/OFY at $24/$25, H at $26, the far-colour + * triple RFC/GFC/BFC at $21/$22/$23, and the light-direction triple + * LR1LR2/LR3LG1/LG2LG3 at $13/$14/$15, so the numbers below are the evidence, + * not a derivation. The file is not an attempt to reconstruct Sony's + * gtemac.h. */ #ifndef SF3_GTEMAC_H @@ -61,17 +63,54 @@ /* --- control registers (ctc2/cfc2) --------------------------------------- */ +/* --- control registers (ctc2/cfc2) --------------------------------------- */ + +/* Rotation/translation matrix (UNSHIFTED standard map, 0x80101CAC: five + * `ctc2`s from five loads). The executable uses the standard 0..5 numbers + * here and standard+7 from RBK onward, so the 0x00-0x05 block is the only + * unshifted control range. The four RT words hold the 3x3 rotation sorted + * by the standard naming (RT1RT2 = row1col1,row1col2, ...): + * $0 = RT1RT2 $1 = RT3RT21 $2 = RT22RT23 + * $3 = RT31RT32 $4 = RT33 + * $5 = TRX (TRY/TRZ unobserved; assumed from the standard order) */ +#define gte_ldRT1RT2(v) __asm__ volatile ("ctc2 %0,$0" : : "r"(v)) +#define gte_ldRT3RT21(v) __asm__ volatile ("ctc2 %0,$1" : : "r"(v)) +#define gte_ldRT22RT23(v) __asm__ volatile ("ctc2 %0,$2" : : "r"(v)) +#define gte_ldRT31RT32(v) __asm__ volatile ("ctc2 %0,$3" : : "r"(v)) +#define gte_ldRT33(v) __asm__ volatile ("ctc2 %0,$4" : : "r"(v)) + /* Rotation/light source and colour-matrix control words. - * $13/$14/$15 = LR1LR2 / LR3LG1 / LG2LG3 (0x8001AE3C, three `ctc2`s) + * The executable's control-register numbering is the standard map shifted + * by +7 (all byte-proved): RBK/GBK/BBK at $13/$14/$15, the light-matrix + * triple plus LB1LB2/LB3 at $16..$20, the far-colour triple RFC/GFC/BFC at + * $21/$22/$23, OFX/OFY at $24/$25, H at $26, DQA at $27, DQB at $28. The + * data-register space is the standard map unshifted (see above). + * $13/$14/$15 = RBK / GBK / BBK (0x8001AE3C, three `ctc2`s) + * $16/$17/$18 = LR1LR2 / LR3LG1 / LG2LG3 (0x80102FE4, with $19/$20 = + * $19/$20 = LB1LB2 / LB3 LB1LB2/LB3 — same body) + * $21/$22/$23 = RFC / GFC / BFC (0x80103B6C, each shifted left + * by 4 before the write) * $24/$25 = OFX / OFY (0x80109778, after `sll ...,16`) - * $26 = H (0x80103A94, `cfc2 v0,$26`) + * $26 = H (0x80103A94 `cfc2 v0,$26`, + * 0x80102FD4 `ctc2 a0,$26`) + * $27 = DQA (0x80103B54, `ctc2 a0,$27`) * $28 = DQB (0x80103B60, `ctc2 a0,$28`) */ -#define gte_ldLR1LR2(v) __asm__ volatile ("ctc2 %0,$13" : : "r"(v)) -#define gte_ldLR3LG1(v) __asm__ volatile ("ctc2 %0,$14" : : "r"(v)) -#define gte_ldLG2LG3(v) __asm__ volatile ("ctc2 %0,$15" : : "r"(v)) +#define gte_ldRBK(v) __asm__ volatile ("ctc2 %0,$13" : : "r"(v)) +#define gte_ldGBK(v) __asm__ volatile ("ctc2 %0,$14" : : "r"(v)) +#define gte_ldBBK(v) __asm__ volatile ("ctc2 %0,$15" : : "r"(v)) +#define gte_ldLR1LR2(v) __asm__ volatile ("ctc2 %0,$16" : : "r"(v)) +#define gte_ldLR3LG1(v) __asm__ volatile ("ctc2 %0,$17" : : "r"(v)) +#define gte_ldLG2LG3(v) __asm__ volatile ("ctc2 %0,$18" : : "r"(v)) +#define gte_ldLB1LB2(v) __asm__ volatile ("ctc2 %0,$19" : : "r"(v)) +#define gte_ldLB3(v) __asm__ volatile ("ctc2 %0,$20" : : "r"(v)) +#define gte_ldRFC(v) __asm__ volatile ("ctc2 %0,$21" : : "r"(v)) +#define gte_ldGFC(v) __asm__ volatile ("ctc2 %0,$22" : : "r"(v)) +#define gte_ldBFC(v) __asm__ volatile ("ctc2 %0,$23" : : "r"(v)) #define gte_ldOFX(v) __asm__ volatile ("ctc2 %0,$24" : : "r"(v)) #define gte_ldOFY(v) __asm__ volatile ("ctc2 %0,$25" : : "r"(v)) +#define gte_ldH(v) __asm__ volatile ("ctc2 %0,$26" : : "r"(v)) #define gte_stH(r) __asm__ volatile ("cfc2 %0,$26" : "=r"(r)) +#define gte_ldDQA(v) __asm__ volatile ("ctc2 %0,$27" : : "r"(v)) #define gte_ldDQB(v) __asm__ volatile ("ctc2 %0,$28" : : "r"(v)) /* --- commands ------------------------------------------------------------ */ diff --git a/src/func_80012D8C.c b/src/func_80012D8C.c new file mode 100644 index 0000000..15358e7 --- /dev/null +++ b/src/func_80012D8C.c @@ -0,0 +1,37 @@ +/* + * func_80012D8C — 48 bytes at 0x80012D8C..0x80012DBC + * + * The three-value sibling of func_80012DBC: same header shape, a different type + * word and command word, and one more caller value stored past the end of the + * other routine's layout. + * + * The observed instructions are: + * li v0,4 + * sh v0,6(a0) ; p->type_06 = 4 + * lui v0,0x300 ; 0x03000000 + * sw v0,8(a0) ; p->word_08 = 0x03000000 + * lui v0,0x4000 ; 0x40000000 + * or a1,a1,v0 ; flags |= 0x40000000 + * sw zero,0(a0) ; p->word_00 = 0 + * sh zero,4(a0) ; p->half_04 = 0 + * sw a1,12(a0) ; p->word_0c = flags + * sw a2,16(a0) ; p->word_10 = value + * jr ra + * sw a3,20(a0) ; p->word_14 = extra (delay slot) + * + * LIMITS: the field offsets and widths are hypotheses read from the instruction + * shape, as are the parameter types. The constants 4, 0x03000000 and 0x40000000 + * are facts about this executable, not about the compiler; what they mean is + * unknown and is not guessed here. Only the compiled bytes are evidence. + */ + +void func_80012D8C(char *base, int flags, int value, int extra) { + *(short *)(base + 6) = 4; + *(int *)(base + 8) = 0x03000000; + flags |= 0x40000000; + *(int *)(base + 0) = 0; + *(short *)(base + 4) = 0; + *(int *)(base + 12) = flags; + *(int *)(base + 16) = value; + *(int *)(base + 20) = extra; +} diff --git a/src/func_80012DBC.c b/src/func_80012DBC.c new file mode 100644 index 0000000..61ed396 --- /dev/null +++ b/src/func_80012DBC.c @@ -0,0 +1,37 @@ +/* + * func_80012DBC — 44 bytes at 0x80012DBC..0x80012DE8 + * + * Initialises a fixed-layout structure: a 16-bit type word, a 32-bit command + * word, two zeroed header fields, a 32-bit flags word built by OR-ing a constant + * into the caller's second argument, and one caller value. The store order below + * is the order the original emits; the first two stores are hoisted ahead of the + * header clear. + * + * The observed instructions are: + * li v0,4 + * sh v0,6(a0) ; p->type_06 = 4 + * lui v0,0x200 ; 0x02000000 + * sw v0,8(a0) ; p->word_08 = 0x02000000 + * lui v0,0x6800 ; 0x68000000 + * or a1,a1,v0 ; flags |= 0x68000000 + * sw zero,0(a0) ; p->word_00 = 0 + * sh zero,4(a0) ; p->half_04 = 0 + * sw a1,12(a0) ; p->word_0c = flags + * jr ra + * sw a2,16(a0) ; p->word_10 = value (delay slot) + * + * LIMITS: the field offsets and widths are hypotheses read from the instruction + * shape, as are the parameter types. The constants 4, 0x02000000 and 0x68000000 + * are facts about this executable, not about the compiler; what they mean is + * unknown and is not guessed here. Only the compiled bytes are evidence. + */ + +void func_80012DBC(char *base, int flags, int value) { + *(short *)(base + 6) = 4; + *(int *)(base + 8) = 0x02000000; + flags |= 0x68000000; + *(int *)(base + 0) = 0; + *(short *)(base + 4) = 0; + *(int *)(base + 12) = flags; + *(int *)(base + 16) = value; +} diff --git a/src/func_800160E8.c b/src/func_800160E8.c new file mode 100644 index 0000000..7ed5ad8 --- /dev/null +++ b/src/func_800160E8.c @@ -0,0 +1,46 @@ +/* func_800160E8 — 0x800160E8..0x80016110 (40 bytes). + * + * Original words: + * 0x24020001 li v0,0x1 + * 0xAC820134 sw v0,0x134(a0) + * 0x248200F8 addiu v0,a0,0xf8 + * 0xAC85009C sw a1,0x9c(a0) + * 0xAC820138 sw v0,0x138(a0) + * 0xAC800104 sw zero,0x104(a0) + * 0x8CA20018 lw v0,0x18(a1) + * 0xAC8500F8 sw a1,0xf8(a0) + * 0x03E00008 jr ra + * 0xACA200FC _sw v0,0xfc(a0) (delay slot) + * + * A list-node initialiser: it seeds a flag at 0x134, points 0x138 at its own + * inline link field at 0xf8, stores the incoming node at 0x9c and at 0xf8, + * clears 0x104, and copies the node's field 0x18 into its own 0xfc. The final + * store lands in the `jr ra` delay slot. + * + * The `addiu v0,a0,0xf8` is emitted early and its store deferred until after + * the 0x9c store, so the source order is not the emission order. Attempt 1 wrote + * the two stores in ascending offset order (0x138 before 0x9c) and produced them + * in that order, 12 differing bytes; swapping them to 0x9c-then-0x138 is what + * matches. Reading `node->0x18` into a named local before the 0xf8 store is what + * makes cc1 schedule that load above the store, as the original does — writing + * the load inline in the final store leaves it after the store. + * + * LIMITS: every displacement is read from the bytes and the field names are + * hypotheses; in particular that 0x138 points at 0xf8 (rather than merely + * holding `self + 0xf8` as a value) is inferred from the self-relative + * addiu and is not proved — a ring or list convention could explain it another + * way. Nothing proves the two descriptors' node type is the same object type. + */ + +void func_800160E8(char *self, int *node) +{ + int value; + + *(int *)(self + 0x134) = 1; + *(int *)(self + 0x9c) = (int)node; + *(int *)(self + 0x138) = (int)(self + 0xf8); + *(int *)(self + 0x104) = 0; + value = *(int *)((char *)node + 0x18); + *(int *)(self + 0xf8) = (int)node; + *(int *)(self + 0xfc) = value; +} diff --git a/src/func_80016120.c b/src/func_80016120.c new file mode 100644 index 0000000..4a991c6 --- /dev/null +++ b/src/func_80016120.c @@ -0,0 +1,58 @@ +/* func_80016120 — 0x80016120..0x80016158 (56 bytes). + * + * Original words: + * lbu v0,0x0(a1) nop sb v0,0x10(a0) sb v0,0xec(a0) + * lbu v0,0x1(a1) nop sb v0,0x11(a0) sb v0,0xed(a0) + * lbu v0,0x2(a1) nop sb v0,0x12(a0) sb v0,0xee(a0) + * jr ra + * _clear v0 (delay slot) + * + * A fully unrolled three-byte copy into two parallel fields: each source byte + * lands at 0x10+i and again at 0xec+i, with the second destination exactly 0xdc + * bytes past the first in every case. The single `lbu` per byte is reused for + * both stores rather than re-loaded. + * + * The unrolled shape is the evidence that the loop was written out (or was + * unrolled by cc1 — which `-O2` alone does not do for three iterations, so the + * explicit form is the one reproduced here). The `clear v0` in the `jr ra` delay + * slot means the routine returns a zero constant rather than being `void`: a void + * function would leave the last loaded byte in v0 (cookbook finding 13's scratch + * rule, applied with the sign reversed). + * + * LIMITS: the two destination bases (0x10 and 0xec) and the three-byte count are + * read from the displacements; that the two fields are parallel copies of one + * three-byte value is inferred from the constant 0xdc stride and not proved. + * The byte order of the source is assumed to be the natural one. + * + * ATTEMPT 1 (recorded): writing the load inline in both stores + * (`dst[0x10] = src[0]; dst[0xec] = src[0];`) gives the original's *shape* — + * `lbu` / `nop` / `sb` / `sb` per byte, the load-delay `nop` where the original + * has it — but cc1 **re-loads** the byte for the second store (the store through + * `dst` may alias `src`), so it is 80 bytes against 56: two `lbu` per byte. + * + * ATTEMPT 2 (recorded): declaring all three bytes as locals up front gives one + * `lbu` per byte but **44 bytes against 56** — cc1 hoists all three loads and + * interleaves the stores, which removes the load-use hazard and therefore the + * three `nop`s the original carries. + * + * So the target needs one load per byte *and* a `nop` between load and first + * store: the loads must not be hoisted, and the byte must not be re-loaded. + * Attempt 3 reuses a single `c` local but re-assigns it between the byte groups, + * so each load is a separate statement whose stores immediately follow. + */ + +int func_80016120(char *dst, char *src) +{ + unsigned char c; + + c = src[0]; + dst[0x10] = c; + dst[0xec] = c; + c = src[1]; + dst[0x11] = c; + dst[0xed] = c; + c = src[2]; + dst[0x12] = c; + dst[0xee] = c; + return 0; +} diff --git a/src/func_80016174.c b/src/func_80016174.c new file mode 100644 index 0000000..66520dd --- /dev/null +++ b/src/func_80016174.c @@ -0,0 +1,52 @@ +/* + * func_80016174 — 36 bytes at 0x80016174..0x80016198 + * + * Clears a strided run of 32-bit slots: fourteen words starting at 0x80122908, + * spaced 172 bytes apart, walking downwards. The loop bound is the byte offset + * itself, counting from 2236 down to 0 inclusive. + * + * The observed instructions are: + * li v0,2236 ; offset = 0x8bc + * L: lui at,0x8012 ; %hi of the symbol base + * addu at,at,v0 ; at = base + offset (base first) + * sw zero,10504(at) ; *(int *)(at + 0x2908) = 0 ; %lo as displacement + * addiu v0,v0,-172 ; offset -= 0xac + * bgez v0,0x80016174 ; loop while offset >= 0 + * nop ; no independent instruction for the slot + * jr ra + * nop + * + * The base must be written as an address-named SYMBOL, not as a literal. This + * was measured, not assumed — the three literal spellings below all emit + * `addu at,v0,at` (variable first) and differ in exactly one byte, the register + * field of the `addu`: + * + * *(int *)(0x80122908 + offset) -> addu at,v0,at (1 byte off) + * *(int *)(offset + 0x80122908) -> addu at,v0,at (byte-identical to + * the previous spelling: cc1 + * canonicalises `+` operand order, + * so finding 22's swap lever does + * NOT apply to a literal base) + * *(int *)((char *)0x80122908 + offset)-> addu at,v0,at (1 byte off) + * + * With a symbol base, `lui %hi` + `addu at,at,v0` + `sw %lo(at)` is emitted and + * the region matches. This is consistent with finding 22's stated limit — the + * commutative-operand trigger does not generalise — and with finding 5's + * literal-versus-symbol distinction; it is recorded as a refinement, not as a + * new rule about the compiler. + * + * LIMITS: the base address, the start offset (2236), the stride (-172) and the + * clear value are facts about this executable, not about the compiler. The + * element type (`int`) is a hypothesis from the `sw` width, and the two `nop`s + * are scheduling fills rather than source. What the cleared slots mean is + * unknown and is not guessed here. Only the compiled bytes are evidence. + */ + +extern char D_80122908[]; + +void func_80016174(void) { + int offset; + + for (offset = 2236; offset >= 0; offset -= 172) + *(int *)((char *)D_80122908 + offset) = 0; +} diff --git a/src/func_80016198.c b/src/func_80016198.c new file mode 100644 index 0000000..9af80a2 --- /dev/null +++ b/src/func_80016198.c @@ -0,0 +1,73 @@ +/* + * func_80016198 — 72 bytes at 0x80016198..0x800161E0 + * + * Leaf routine that returns the address of the first free entry of a 14-entry + * table of 0xac-byte records, or null when the table is full. + * + * The observed instructions are: + * addu a0,zero,zero 00002021 i = 0 + * lui a1,0x8012 3c058012 \ + * addiu a1,a1,0x2908 24a52908 / a1 = 0x80122908 (D_80122908) + * addu v1,zero,zero 00001821 off = 0 + * 0x800161a8: + * lui at,0x8012 3c018012 \ + * addu at,at,v1 00200821 / at = 0x80120000 + off + * lw v0,0x2908(at) 8c220908 v0 = *(int *)(D_80122908 + off) + * nop 00000000 load-delay slot + * beq v0,zero,0x800161d8 10400007 if (v0 == 0) goto epilogue + * move v0,a1 00a01021 v0 = a1 (delay slot) + * addiu a1,a1,0xac 24a500ac a1 += 0xac + * addiu a0,a0,0x1 24840001 i++ + * slti v0,a0,0xe 2882000e v0 = (i < 14) signed + * bne v0,zero,0x800161a8 1440fff4 if (v0) loop + * addiu v1,v1,0xac 246300ac off += 0xac (delay slot) + * 0x800161d8: + * jr ra 03e00008 + * nop 00000000 (delay slot) + * + * cc1 keeps **two** induction variables for the same index: the counter `i` + * (`addiu a0,a0,1`, bound-tested by `slti` against 14) and the byte offset `off` + * (`addiu v1,v1,0xac`, used for the address). The return value is a third: the + * *address* variable a1, materialised once as `lui`+`addiu` and incremented by + * 0xac per iteration, moved into v0 in the branch delay slot — so the found path + * returns the element address. + * + * That separation is load-bearing. Deriving the read address from the same + * expression as the returned pointer lets cc1 common-subexpression the two into a + * single walked pointer, which loses both the explicit `lui at,%hi` / `addu` / + * `lw %lo(at)` address form and the separate `off` (measured: 56 bytes, and a + * variant that kept the pointer but folded the read gave 64 against the original's + * 72). The source therefore reads through `D_80122908 + off` with a variable `off` + * while `p` is walked independently, and the declaration order fixes the two + * zeroing instructions: `i` is cleared before `off`. + * + * The table base is materialised as `lui`+`addiu` (the linker-resolved symbol + * form, cookbook finding 4), and the indexed *read* uses the explicit + * `lui at,%hi` / `addu at,at,index` / `lw %lo(at)` form because the index is added + * between the halves. The 14-entry bound is a signed compare, so the counter is a + * signed `int`. + * + * LIMITS: the function name, the record size, the table length of 14 and the claim + * that a zero first word marks a free entry are hypotheses; only the bytes are + * evidence. The element field read is 32-bit. The return type is written as + * `int *` because the returned value is an address; whether the original used a + * struct pointer is not recoverable. + */ + +extern char D_80122908[]; + +int *func_80016198(void) +{ + int i = 0; + int *p = (int *)D_80122908; + int off = 0; + + for (; i < 14; i++) { + if (*(int *)(D_80122908 + off) == 0) + return p; + p = (int *)((char *)p + 0xac); + off += 0xac; + } + + return 0; +} diff --git a/src/func_8001761C.c b/src/func_8001761C.c new file mode 100644 index 0000000..abdb849 --- /dev/null +++ b/src/func_8001761C.c @@ -0,0 +1,66 @@ +/* + * func_8001761C — 68 bytes at 0x8001761C..0x80017660 + * + * Leaf routine that scans a 60-entry table of 0x38-byte records and clears the + * first word of every record whose second word matches the argument. + * + * The observed instructions are: + * addu a1,zero,zero 00002821 i = 0 + * addu v1,zero,zero 00001821 off = 0 + * 0x80017624: + * lui at,0x8012 3c018012 \ + * addu at,at,v1 00200821 / at = 0x80120000 + off + * lw v0,0x4cc4(at) 8c2204cc4 v0 = *(int *)(D_80124CC0 + off + 4) + * nop 00000000 load-delay slot + * bne v0,a0,0x80017648 14440007 if (v0 != a0) goto the increment + * nop 00000000 (delay slot) + * lui at,0x8012 3c018012 \ + * addu at,at,v1 00200821 / at = 0x80120000 + off + * sw zero,0x4cc0(at) ac2004cc0 *(int *)(D_80124CC0 + off) = 0 + * 0x80017648: + * addiu a1,a1,0x1 24a50001 i++ + * slti v0,a1,0x3c 28a2003c v0 = (i < 60) signed + * bne v0,zero,0x80017624 1440fff6 if (v0) loop + * addiu v1,v1,0x38 24630038 off += 0x38 (delay slot) + * 0x80017658: + * jr ra 03e00008 + * nop 00000000 (delay slot) + * + * The indexed symbol access is materialised as `lui at,%hi` / `addu at,at,index` + * / `lw v0,%lo(at)` — the **explicit** address form, not the macro form (cookbook + * finding 1's other mode), because the access has an index added between the high + * and low halves. The low halves appear as the raw displacements 0x4cc4 and + * 0x4cc0, and the two accesses each recompute the high half: that only happens if + * the source names **two different symbols** rather than deriving one address from + * the other. cc1 otherwise common-subexpressions the pair into a single walked + * pointer (measured: 68 bytes with 40 differing bytes, first at 0x8001761D). So the + * compare reads `D_80124CC4 + off` and the store writes `D_80124CC0 + off`, with a + * shared `off`. + * + * cc1 keeps two induction variables: the counter `i` and the byte offset `off`, + * incremented by 0x38 in the branch delay slot. The bound test is `slti` (signed), + * so the counter is a signed `int` and the literal is 60. Both variables are + * initialised to zero, and the **order** of the two zeroing instructions is + * evidence: the original clears `a1` (the counter) before `v1` (the offset), which + * is why the source initialises `i` before `off` and the `for` has an empty init. + * + * LIMITS: the function name, the record size, the table length of 60 and the field + * meanings are hypotheses; only the bytes are evidence. Both accesses are 32-bit. + * The two tables are written as `char[]` so the byte offsets stay explicit; no + * array or struct type is claimed. + */ + +extern char D_80124CC0[]; +extern char D_80124CC4[]; + +void func_8001761C(int a0) +{ + int i = 0; + int off = 0; + + for (; i < 60; i++) { + if (*(int *)(D_80124CC4 + off) == a0) + *(int *)(D_80124CC0 + off) = 0; + off += 0x38; + } +} diff --git a/src/func_800182D4.c b/src/func_800182D4.c new file mode 100644 index 0000000..c3b7858 --- /dev/null +++ b/src/func_800182D4.c @@ -0,0 +1,34 @@ +/* func_800182D4 — 0x800182D4..0x800182F4 (32 bytes). + * + * Original words: + * 0x10800004 beq a0,zero,0x800182E8 + * 0x00001021 _clear v0 (delay slot) + * 0x8C83000C lw v1,0xc(a0) + * 0x080060BB j 0x800182EC + * 0xAC830000 _sw v1,0x0(a1) (delay slot) + * 0x24020001 li v0,0x1 <- 0x800182E8 + * 0x03E00008 jr ra <- 0x800182EC + * 0x00000000 nop + * + * A guarded pointer copy with a boolean result. The success path clears v0 in + * the branch delay slot and stores `src[3]` into `*dst`; the failure path + * materialises 1 immediately before the shared return. + * + * The `sw` lands in the `j` delay slot and the `clear v0` in the `beq` slot + * (cookbook finding 8). That v0 is cleared *before* the branch means the + * success value is the fall-through constant, so `return 0` is the statement + * after the store rather than an else-arm. + * + * LIMITS: `src[3]` (displacement 0xc) is a hypothesis about a 4-int header; the + * routine may be copying any 4-byte field. Whether the failure result 1 means + * "NULL argument" or something else is not observable from the bytes. Nothing + * here proves the two pointers are different objects. + */ + +int func_800182D4(int *src, int *dst) +{ + if (src == 0) + return 1; + *dst = src[3]; + return 0; +} diff --git a/src/func_8001AE3C.c b/src/func_8001AE3C.c index e91ddee..4afc4f5 100644 --- a/src/func_8001AE3C.c +++ b/src/func_8001AE3C.c @@ -1,22 +1,28 @@ /* func_8001AE3C — 0x8001AE3C..0x8001AE50 (20 bytes). * * Original words: - * 0x48C46800 ctc2 a0,$13 LR1LR2 - * 0x48C57000 ctc2 a1,$14 LR3LG1 - * 0x48C67800 ctc2 a2,$15 LG2LG3 + * 0x48C46800 ctc2 a0,$13 RBK + * 0x48C57000 ctc2 a1,$14 GBK + * 0x48C67800 ctc2 a2,$15 BBK * 0x03E00008 jr ra * 0x00000000 nop * - * Three GTE control-register writes through include/gtemac.h. Ghidra's PSX - * loader collapses this trio into a single `ldbkdir a0,a1,a2` pseudo-instruction - * and reports a 4-instruction body; the raw words above are the ground truth - * (the object disassembles to exactly these three `ctc2`s). + * Three GTE control-register writes through include/gtemac.h. Register numbers + * $13/$14/$15 are the background-colour triple RBK/GBK/BBK under the + * executable's control map (standard + 7, recorded in include/gtemac.h). + * Ghidra's PSX loader collapses this trio into a single `ldbkdir a0,a1,a2` + * pseudo-instruction and reports a 4-instruction body; the raw words above are + * the ground truth (the object disassembles to exactly these three `ctc2`s). + * + * LIMITS: the parameter names are hypotheses; the register numbers are the + * evidence, and the far-colour reading (RFC/GFC/BFC) was ruled out because + * those sit at $21/$22/$23 (0x80103B6C). */ #include "../include/gtemac.h" -void func_8001AE3C(int lr1lr2, int lr3lg1, int lg2lg3) +void func_8001AE3C(int rbk, int gbk, int bbk) { - gte_ldLR1LR2(lr1lr2); - gte_ldLR3LG1(lr3lg1); - gte_ldLG2LG3(lg2lg3); + gte_ldRBK(rbk); + gte_ldGBK(gbk); + gte_ldBBK(bbk); } diff --git a/src/func_800230E4.c b/src/func_800230E4.c new file mode 100644 index 0000000..cb9a8b1 --- /dev/null +++ b/src/func_800230E4.c @@ -0,0 +1,28 @@ +/* + * func_800230E4 — 56 bytes at 0x800230E4..0x8002311C + * + * Copies three 32-bit values from one structure to another, scaling each left by + * twelve bits on the way. The routine returns zero, which the original + * materialises with `move v0,zero` in the `jr ra` delay slot. + * + * The observed instructions are: + * lw v0,0(a0) / nop / sll v0,v0,0xc / sw v0,0(a1) + * lw v0,4(a0) / nop / sll v0,v0,0xc / sw v0,4(a1) + * lw v0,8(a0) / nop / sll v0,v0,0xc / sw v0,8(a1) + * jr ra + * move v0,zero ; return 0 (delay slot) + * + * The `nop` after each load is maspsx's load-delay fill. + * + * LIMITS: the element type (`int`) and the shift amount (12) are hypotheses read + * from the instruction shape; the shift is a fact about this executable's fixed- + * point convention, not about the compiler. The `return 0` is required — a `void` + * body would not materialise `v0`. Only the compiled bytes are evidence. + */ + +int func_800230E4(int *src, int *dst) { + dst[0] = src[0] << 12; + dst[1] = src[1] << 12; + dst[2] = src[2] << 12; + return 0; +} diff --git a/src/func_8002D1F8.c b/src/func_8002D1F8.c new file mode 100644 index 0000000..3d4afd3 --- /dev/null +++ b/src/func_8002D1F8.c @@ -0,0 +1,58 @@ +/* + * func_8002D1F8 — 52 bytes at 0x8002D1F8..0x8002D22C + * + * Leaf routine that indexes a global pointer table, null-checks the entry, and + * reports whether the entry's field at +0xc is non-zero. + * + * The observed instructions are: + * lui v0,0x8012 3c028012 \ + * lw v0,0x1c00(v0) 8c421c00 / v0 = *(int *)0x80121C00 (D_80121C00) + * sll a0,a0,0x2 00042080 a0 *= 4 + * addu a0,a0,v0 00822021 a0 += table + * lw a0,0x0(a0) 8c840000 a0 = table[a0] + * nop 00000000 load-delay slot + * beq a0,zero,0x8002d224 10800004 if (a0 == 0) goto epilogue + * addu v0,zero,zero 00001021 v0 = 0 (delay slot) + * lw v0,0xc(a0) 8c82000c v0 = *(int *)(a0 + 0xc) + * nop 00000000 load-delay slot + * sltu v0,zero,v0 0002102b v0 = (0 < v0) + * 0x8002d224: + * jr ra 03e00008 + * nop 00000000 (delay slot) + * + * The table base is a symbol load through the same register (cookbook finding + * 2), so it is written as the named global `D_80121C00` and not as a literal + * address (finding 5). The index is scaled by `sll` and added with the **index + * first** (`addu a0,a0,v0`), which finding 22 records as the spelling + * `*(int *)(base + index * 4)` written as one expression rather than as a named + * pointer. + * + * The result is `sltu v0,zero,v0`, this compiler's form for `x != 0` producing + * 0 or 1, so the return type is a boolean-valued `int` and the tail is a + * comparison, not a bit test. The zero return for a null entry is hoisted into + * the branch delay slot, and the load of the +0xc field is **not** hoisted into + * it — so the source does not branch away to a separate zero return. It assigns a + * result variable 0 first and overwrites it only when the entry is non-null, + * which is the shape that keeps one shared epilogue: `int r = 0; if (p != 0) + * r = ...; return r;`. The early-return spelling was measured and differs by 4 + * bytes, because cc1 then hoists the load into the branch delay slot and emits a + * second jump. + * + * LIMITS: the function name, the claim that 0x80121C00 holds a table pointer and + * the meaning of the +0xc field are hypotheses; only the bytes are evidence. The + * index is assumed to be a 4-byte-strided index; whether the original source + * used an array type or raw pointer arithmetic is not recoverable. + */ + +extern int D_80121C00; + +int func_8002D1F8(int a0) +{ + int p = *(int *)(D_80121C00 + a0 * 4); + int r = 0; + + if (p != 0) + r = *(int *)(p + 0xc) != 0; + + return r; +} diff --git a/src/func_8002E7C4.c b/src/func_8002E7C4.c new file mode 100644 index 0000000..6c9cdda --- /dev/null +++ b/src/func_8002E7C4.c @@ -0,0 +1,30 @@ +/* + * func_8002E7C4 — 32 bytes at 0x8002E7C4..0x8002E7E4 + * + * Writes two byte fields into a structure reached through a global pointer. The + * pointer is reloaded for each store rather than cached in a local, so the C + * keeps the two statements independent. + * + * The observed instructions are: + * lui v0,0x8012 ; symbol load expands into v0 itself + * lw v0,9236(v0) ; v0 = D_80122414 (a pointer) + * nop ; maspsx load-delay nop + * sb a0,45(v0) ; p->byte_2d = a0 + * lui v0,0x8012 ; reload for the second store + * lw v0,9236(v0) + * jr ra + * sb a1,18(v0) ; p->byte_12 = a1 (delay slot) + * + * LIMITS: the symbol name D_80122414, the `char *` pointee type and both field + * offsets (45 and 18 decimal) are hypotheses read from the instruction shape. + * The parameter types are hypotheses too: the `sb` stores need no mask, so a + * `char` or an `int` parameter would emit the same bytes here. Only the compiled + * bytes are evidence. + */ + +extern char *D_80122414; + +void func_8002E7C4(char first, char second) { + D_80122414[45] = first; + D_80122414[18] = second; +} diff --git a/src/func_8002F404.c b/src/func_8002F404.c new file mode 100644 index 0000000..a9c2c20 --- /dev/null +++ b/src/func_8002F404.c @@ -0,0 +1,66 @@ +/* + * func_8002F404 — 76 bytes at 0x8002F404..0x8002F450 + * + * Leaf routine that sets or clears bit 0 of a 16-bit field in an object reached + * through one pointer, reporting whether it found the object. + * + * The observed instructions are: + * beq a0,zero,0x8002f41c 10800005 if (a0 == 0) goto the zero result + * li v1,0x1 24030001 v1 = 1 (delay slot) + * lw a0,0x0(a0) 8c840000 a0 = *(int *)a0 + * nop 00000000 load-delay slot + * bne a0,zero,0x8002f424 14800002 if (a0 != 0) goto the field update + * andi v0,a1,0xff 30a200ff v0 = a1 & 0xff (delay slot) + * 0x8002f41c: + * j 0x8002f448 0800bd12 goto epilogue + * addu v1,zero,zero 00001821 v1 = 0 (delay slot) + * 0x8002f424: + * bne v0,v1,0x8002f438 14400004 if ((a1 & 0xff) != 1) goto the clear path + * nop 00000000 (delay slot) + * lhu v0,0x6(a0) 94820006 v0 = *(unsigned short *)(a0 + 6) + * j 0x8002f444 0800bd11 goto the store + * ori v0,v0,0x1 34420001 v0 |= 1 (delay slot) + * 0x8002f438: + * lhu v0,0x6(a0) 94820006 v0 = *(unsigned short *)(a0 + 6) + * nop 00000000 load-delay slot + * andi v0,v0,0xfffe 3042fffe v0 &= ~1 + * 0x8002f444: + * sh v0,0x6(a0) a4820006 *(unsigned short *)(a0 + 6) = v0 + * 0x8002f448: + * jr ra 03e00008 + * move v0,v1 00601021 v0 = v1 (delay slot) + * + * The result is a **variable** v1, initialised to 1 in the first branch's delay + * slot and zeroed on the shared failure path, then returned via `move v0,v1`. Both + * null exits jump to one block that only sets v1 = 0, so the source assigns the + * result variable rather than returning a literal in two places; writing it as two + * `return 0;` statements gives a different layout. + * + * The second argument is masked with `andi ...,0xff` and compared against 1, so the + * flag is a byte-wide value even though it arrives in a 32-bit register. The field + * is 16-bit (`lhu`/`sh`) and is set with `| 1` or cleared with `& ~1` — the + * assembler renders `~1` as 0xfffe. + * + * LIMITS: the function name, the pointer hop, the offset 0x6 and the meaning of the + * flag byte are hypotheses; only the bytes are evidence. The comparison value 1 is + * taken as the "set" selector because it is the value that produces the `ori`. + */ + +int func_8002F404(int a0, int a1) +{ + int v1 = 1; + + if (a0 == 0) + v1 = 0; + else { + a0 = *(int *)a0; + if (a0 == 0) + v1 = 0; + else if ((a1 & 0xff) == 1) + *(unsigned short *)(a0 + 6) |= 1; + else + *(unsigned short *)(a0 + 6) &= ~1; + } + + return v1; +} diff --git a/src/func_80043D8C.c b/src/func_80043D8C.c new file mode 100644 index 0000000..bf4df07 --- /dev/null +++ b/src/func_80043D8C.c @@ -0,0 +1,57 @@ +/* + * func_80043D8C — 56 bytes at 0x80043D8C..0x80043DC4 + * + * Leaf routine that appends one word to a small counted array: it compares the + * byte count against the byte capacity, and if there is room, increments the + * count and stores the argument through the array's base pointer. + * + * The observed instructions are: + * lbu v1,0x0(a0) 90830000 v1 = *(unsigned char *)(a0 + 0) + * lbu v0,0x1(a0) 90820001 v0 = *(unsigned char *)(a0 + 1) + * andi a2,v1,0xff 306200ff a2 = v1 & 0xff + * sltu v0,a2,v0 0045102b v0 = (a2 < v0) unsigned + * beq v0,zero,0x80043dbc 10400006 if (!(count < capacity)) goto epilogue + * nop 00000000 (delay slot) + * addiu v0,v1,0x1 24620001 v0 = v1 + 1 + * lw v1,0x4(a0) 8c830004 v1 = *(int *)(a0 + 4) + * sb v0,0x0(a0) a0820000 *(char *)(a0 + 0) = v0 + * sll v0,a2,0x2 00021080 v0 = a2 * 4 + * addu v0,v0,v1 00431021 v0 = (index * 4) + base + * sw a1,0x0(v0) ac450000 *(int *)v0 = a1 + * 0x80043dbc: + * jr ra 03e00008 + * li v0,0x1 24020001 v0 = 1 (delay slot) + * + * Two byte loads with no load-delay `nop` in between: they target different + * registers, and nothing reads them until the `andi`/`sltu` pair, so no stall is + * needed. The `andi ...,0xff` before the comparison is the promotion mask this + * ABI needs to compare two `unsigned char` values as unsigned ints (finding 7's + * family), and the same masked value is reused as the array index — so the source + * reads the count once into an `unsigned char` local. + * + * The capacity is at +1 and the base pointer at +4, so the object is a + * `{ unsigned char count; unsigned char capacity; ... int *base; }` shape; the + * pointer is loaded *after* the increment is computed but *before* the store to + * the count, which is cc1's scheduling and is left to the compiler. + * + * The result is the constant 1 in the `jr ra` delay slot, so the function reports + * success unconditionally (`return 1;`) — it does not report whether the append + * happened. That is the shape of the bytes, not an inference about intent. + * + * LIMITS: the function name, the struct interpretation and the field meanings are + * hypotheses; only the bytes are evidence. The array stride is 4 and the count is + * an 8-bit field. + */ + +int func_80043D8C(unsigned char *a0, int a1) +{ + unsigned char count = a0[0]; + unsigned char capacity = a0[1]; + + if (count < capacity) { + a0[0] = count + 1; + *(int *)(*(int *)(a0 + 4) + count * 4) = a1; + } + + return 1; +} diff --git a/src/func_8004C060.c b/src/func_8004C060.c new file mode 100644 index 0000000..84dc8a9 --- /dev/null +++ b/src/func_8004C060.c @@ -0,0 +1,48 @@ +/* + * func_8004C060 — 48 bytes at 0x8004C060..0x8004C090 + * + * Leaf routine that reports whether a state field holds one of two values. + * + * The observed instructions are: + * lw v0,0x20(a0) 8c820020 v0 = *(int *)(a0 + 0x20) + * nop 00000000 load-delay slot + * lw v1,0x21c(v0) 8c43021c v1 = *(int *)(v0 + 0x21c) + * li v0,0x7 24020007 v0 = 7 + * beq v1,v0,0x8004c084 10620003 if (v1 == 7) goto set-one + * addu a0,zero,zero 00002021 a0 = 0 (delay slot) + * li v0,0x1 24020001 v0 = 1 + * bne v1,v0,0x8004c088 14620001 if (v1 != 1) goto epilogue + * nop 00000000 (delay slot) + * 0x8004c084: + * li a0,0x1 24040001 a0 = 1 + * 0x8004c088: + * jr ra 03e00008 + * move v0,a0 00801021 v0 = a0 (delay slot) + * + * The result lives in **a0**, not v0: a0's incoming pointer value is dead after + * the two loads, so cc1 reused the argument register for the result and copies it + * to v0 only in the `jr ra` delay slot. The zero initialisation is hoisted into + * the first branch's delay slot, which is the tell that the source assigns the + * result variable to 0 first and only overwrites it on a match. + * + * The two comparisons are `== 7` and `== 1` with the constants materialised into + * v0 by `li` — SGIs have no branch-immediate form, so a comparison against a + * constant always costs a register (the same reason func_80010418 materialises + * its -1 sentinel). + * + * LIMITS: the function name and the meaning of the states 7 and 1 are + * hypotheses; only the bytes are evidence. The two loads are 32-bit. The C uses + * `||` for the pair because the two branches share one "set one" block; a nested + * if or a `switch` would produce a different layout. + */ + +int func_8004C060(int a0) +{ + int v1 = *(int *)(*(int *)(a0 + 0x20) + 0x21c); + int r = 0; + + if (v1 == 7 || v1 == 1) + r = 1; + + return r; +} diff --git a/src/func_8004C0AC.c b/src/func_8004C0AC.c new file mode 100644 index 0000000..b61aca6 --- /dev/null +++ b/src/func_8004C0AC.c @@ -0,0 +1,54 @@ +/* + * func_8004C0AC — 68 bytes at 0x8004C0AC..0x8004C0F0 + * + * Leaf routine that reports whether a state field holds one of four values, + * three of them tested individually and two by a range check. + * + * The observed instructions are: + * lw v0,0x20(a0) 8c820020 v0 = *(int *)(a0 + 0x20) + * nop 00000000 load-delay slot + * lw v1,0x21c(v0) 8c43021c v1 = *(int *)(v0 + 0x21c) + * nop 00000000 load-delay slot + * addiu v0,v1,-0x2 2462fffe v0 = v1 - 2 + * sltiu v0,v0,0x2 2c420002 v0 = ((unsigned)(v1 - 2) < 2) (v1 == 2 || 3) + * bne v0,zero,0x8004c0e4 14400007 if (v0) goto set-one + * addu a0,zero,zero 00002021 a0 = 0 (delay slot) + * li v0,0x6 24020006 v0 = 6 + * beq v1,v0,0x8004c0e4 10620005 if (v1 == 6) goto set-one + * nop 00000000 (delay slot) + * li v0,0x4 24020004 v0 = 4 + * bne v1,v0,0x8004c0e8 14620001 if (v1 != 4) goto epilogue + * nop 00000000 (delay slot) + * 0x8004c0e4: + * li a0,0x1 24040001 a0 = 1 + * 0x8004c0e8: + * jr ra 03e00008 + * move v0,a0 00801021 v0 = a0 (delay slot) + * + * This is the same shape as func_8004C060 (matched): the result is accumulated in + * **a0**, the dead argument register, and copied to v0 only in the `jr ra` delay + * slot, with the zero initialisation hoisted into the first branch's delay slot. + * That fixes the source as `int r = 0; if (cond) r = 1; return r;`. + * + * The first two values are tested by a single range check — `addiu` then `sltiu` — + * which is cc1's form for `v1 == 2 || v1 == 3`. The remaining values are tested in + * source order after it: 6 (taken branch to the shared set-one block) and then 4, + * whose failure is the fall-through to the epilogue. So the condition is written + * with 4 **last**, not in ascending order. + * + * LIMITS: the function name and the meaning of the states 2, 3, 4 and 6 are + * hypotheses; only the bytes are evidence. The two loads are 32-bit. Whether the + * original spelled the range as two `==` tests or as a range is not recoverable; + * the `sltiu` shows cc1 saw a range either way. + */ + +int func_8004C0AC(int a0) +{ + int v1 = *(int *)(*(int *)(a0 + 0x20) + 0x21c); + int r = 0; + + if (v1 == 2 || v1 == 3 || v1 == 6 || v1 == 4) + r = 1; + + return r; +} diff --git a/src/func_8004CEEC.c b/src/func_8004CEEC.c new file mode 100644 index 0000000..f01e444 --- /dev/null +++ b/src/func_8004CEEC.c @@ -0,0 +1,44 @@ +/* + * func_8004CEEC — 32 bytes at 0x8004CEEC..0x8004CF0C + * + * Leaf routine that initialises part of a structure: two zero fields, a + * constant field, and a field copied from a global. + * + * The observed instructions are: + * lui v1,0x8012 3c038012 \ + * lw v1,0x2354(v1) 8c632354 / v1 = *(int *)0x80122354 (D_80122354) + * lui v0,0x8000 3c028000 v0 = 0x80000000 + * sh zero,0x3e(a0) a480003e *(short *)(a0 + 0x3e) = 0 + * sw zero,0xdc(a0) ac8000dc *(int *)(a0 + 0xdc) = 0 + * sw v0,0xe0(a0) ac8200e0 *(int *)(a0 + 0xe0) = 0x80000000 + * jr ra 03e00008 + * sw v1,0xe4(a0) ac8300e4 *(int *)(a0 + 0xe4) = v1 (delay slot) + * + * Two address forms appear side by side and both are the macro form: the + * global is `lui`+`lw` through the same register (cookbook finding 2), and the + * constant 0x80000000 is materialised by `lui v0,0x8000` alone — no `ori`, + * because the low half is zero. 0x80122354 is outside the small-data window + * and the original accesses it absolutely, so it is written as the named + * global `D_80122354` with no registry row (the address-named symbol resolves + * implicitly). The last store is independent and lands in the `jr ra` delay + * slot. + * + * LIMITS: the function name, the structure and every field offset are + * hypotheses read off the disassembly. The field widths are evidence: one + * 16-bit store and three 32-bit stores. The constant is written as + * `(int)0x80000000` because a bare `0x80000000` is unsigned in this dialect and + * would make cc1 emit a different materialisation; the signed value is what the + * `lui`-only sequence implies. + */ + +extern int D_80122354; + +void func_8004CEEC(int a0) +{ + int v1 = D_80122354; + + *(short *)(a0 + 0x3e) = 0; + *(int *)(a0 + 0xdc) = 0; + *(int *)(a0 + 0xe0) = (int)0x80000000; + *(int *)(a0 + 0xe4) = v1; +} diff --git a/src/func_800516E0.c b/src/func_800516E0.c new file mode 100644 index 0000000..d47c515 --- /dev/null +++ b/src/func_800516E0.c @@ -0,0 +1,32 @@ +/* func_800516E0 — 0x800516E0..0x800516FC (28 bytes). + * + * Original words: + * 0x8C820020 lw v0,0x20(a0) + * 0x00000000 nop + * 0x8C4200F4 lw v0,0xf4(v0) + * 0x00000000 nop + * 0xAC4501FC sw a1,0x1fc(v0) + * 0x03E00008 jr ra + * 0xA0460200 sb a2,0x200(v0) (jr ra delay slot) + * + * A two-hop pointer walk ending in one word store and one byte store. The + * second store lands in the `jr ra` delay slot (cookbook finding 8: the last + * independent instruction fills the slot). + * + * The `nop` after each load is the load-delay fill the original carries; it + * comes from maspsx/the assembler, not from the C. + * + * LIMITS: the struct layout is a hypothesis reconstructed from the two + * displacements. Nothing proves that offsets 0x20, 0xf4, 0x1fc and 0x200 are + * distinct fields of distinct objects, only that the compiled sequence walks + * them in this order. The stored byte's signedness is not observable from `sb`. + */ + +void func_800516E0(int base, int value, int flag) +{ + int inner = *(int *)(base + 0x20); + + inner = *(int *)(inner + 0xf4); + *(int *)(inner + 0x1fc) = value; + *(char *)(inner + 0x200) = flag; +} diff --git a/src/func_8005182C.c b/src/func_8005182C.c new file mode 100644 index 0000000..ff46349 --- /dev/null +++ b/src/func_8005182C.c @@ -0,0 +1,50 @@ +/* + * func_8005182C — 56 bytes at 0x8005182C..0x80051864 + * + * Two pointer hops, then a guarded range test. The result register is cleared + * before the guard and the guard's body only runs when the flag byte is non-zero; + * the `beqz`'s delay slot carries the `move v1,zero`, so the false path falls + * through with the pre-cleared result. + * + * The observed instructions are: + * lw v0,32(a0) ; q = p->ptr_20 + * nop + * lw a0,244(v0) ; r = q->ptr_f4 + * nop + * lbu v0,26(a0) ; r->byte_1a + * nop + * beqz v0,0x8005185C ; if (r->byte_1a == 0) return 0 + * move v1,zero ; result = 0 (delay slot) + * lw v0,500(a0) ; r->word_1f4 + * nop + * addiu v0,v0,-2 ; value - 2 <- stays in v0 + * sltiu v1,v0,2 ; result = (unsigned)(value - 2) < 2 + * 5C:jr ra + * move v0,v1 ; return result (delay slot) + * + * The subtraction must stay in `v0` rather than being folded into the result + * register: writing the comparison as one expression + * (`result = (unsigned)(r->word - 2) < 2;`) makes `cc1` compute the subtraction + * into `v1` and then compare in place, which is a 2-byte register-field diff. + * Binding the subtraction to its own local reproduces the original's `addiu v0`. + * + * LIMITS: every offset, the pointer types and the `unsigned char` flag type are + * hypotheses read from the instruction shape; the `lbu` is what shows the flag is + * unsigned, and the `sltiu` is what shows the range test is unsigned. The + * constants 2 and 2 are facts about this executable, not about the compiler. + * Only the compiled bytes are evidence. + */ + +int func_8005182C(char *base) { + int result = 0; + char *q = *(char **)(base + 32); + char *r = *(char **)(q + 244); + + if (*(unsigned char *)(r + 26) != 0) { + int value = *(int *)(r + 500) - 2; + + result = (unsigned int)value < 2; + } + + return result; +} diff --git a/src/func_80052C98.c b/src/func_80052C98.c new file mode 100644 index 0000000..7e2f4cc --- /dev/null +++ b/src/func_80052C98.c @@ -0,0 +1,31 @@ +/* + * func_80052C98 — 20 bytes at 0x80052C98..0x80052CAC + * + * Leaf routine that walks two embedded pointers and clears one word. + * + * The observed instructions are: + * lw v0,0x20(a0) 8c820020 v0 = *(int *)(a0 + 0x20) + * nop 00000000 load-delay slot + * lw v0,0xf4(v0) 8c4200f4 v0 = *(int *)(v0 + 0xf4) + * jr ra 03e00008 + * sw zero,0x1f8(v0) ac4001f8 *(int *)(v0 + 0x1f8) = 0 (delay slot) + * + * The `nop` after the first load is the load-delay slot: the next instruction + * reads the loaded register, so both cc1 and maspsx keep the stall visible. + * The store is independent of the second load's result and is scheduled into + * the `jr ra` delay slot. + * + * LIMITS: the function name and the interpretation of the offsets (a struct at + * a0+0x20, a nested pointer at +0xf4, a flag word at +0x1f8) are hypotheses + * read off the disassembly; no struct layout is claimed. The access widths are + * evidence: 32-bit loads, 32-bit store. The pointer arithmetic is written with + * `int *` and byte offsets added on the integer value, which is what makes cc1 + * emit plain `lw`/`sw` with a folded displacement. + */ + +void func_80052C98(int a0) +{ + int v0 = *(int *)(a0 + 0x20); + v0 = *(int *)(v0 + 0xf4); + *(int *)(v0 + 0x1f8) = 0; +} diff --git a/src/func_8005E3D0.c b/src/func_8005E3D0.c new file mode 100644 index 0000000..61afe78 --- /dev/null +++ b/src/func_8005E3D0.c @@ -0,0 +1,40 @@ +/* + * func_8005E3D0 — 36 bytes at 0x8005E3D0..0x8005E3F4 + * + * Leaf routine that copies two fields out of a structure through two output + * pointers and returns a third field. + * + * The observed instructions are: + * lw v0,0xc4c(a0) 8c820c4c v0 = *(int *)(a0 + 0xc4c) + * nop 00000000 load-delay slot + * sw v0,0x0(a1) aca20000 *a1 = v0 + * lw v0,0xc50(a0) 8c820c50 v0 = *(int *)(a0 + 0xc50) + * nop 00000000 load-delay slot + * sw v0,0x0(a2) acc20000 *a2 = v0 + * lw v0,0xc58(a0) 8c820c58 v0 = *(int *)(a0 + 0xc58) + * jr ra 03e00008 + * nop 00000000 (delay slot) + * + * The third load's result is never stored, so it is the return value: v0 is + * live at the `jr ra`. That fixes the C return type as `int` (or a 32-bit + * scalar) and makes the two output pointers `int *`. + * + * Both `nop`s sit between a load and a *store of the loaded register*. This is + * exactly the case cookbook finding 27 records as maspsx's gap: maspsx inserts a + * load-delay `nop` only when the next instruction *loads from* the loaded + * register, and a store that merely reads it as a source slips through. If this + * candidate comes back short by 4 or 8 bytes, the cause is that gap and not the + * C, and the row is a second, demonstrated instance of the finding rather than a + * source-shape problem. + * + * LIMITS: the function name, the field offsets and the claim that the structure + * is a game object are hypotheses; only the bytes are evidence. All three + * accesses are 32-bit. + */ + +int func_8005E3D0(int a0, int *a1, int *a2) +{ + *a1 = *(int *)(a0 + 0xc4c); + *a2 = *(int *)(a0 + 0xc50); + return *(int *)(a0 + 0xc58); +} diff --git a/src/func_800681E0.c b/src/func_800681E0.c new file mode 100644 index 0000000..af0f9be --- /dev/null +++ b/src/func_800681E0.c @@ -0,0 +1,54 @@ +/* + * func_800681E0 — 60 bytes at 0x800681E0..0x8006821C + * + * Selects a table offset (a special case for index 1, otherwise index * 36), + * loads a 32-bit word from a table and extracts a 5-bit field from its top. The + * index scaling is `cc1`'s strength reduction of `* 36` into `*8`, `+self`, `*4`. + * + * The observed instructions are: + * li v0,1 + * beq a0,v0,0x800681FC ; if (index == 1) jump to the special case + * nop + * sll v0,a0,0x3 ; index * 8 + * addu v0,v0,a0 ; index * 9 + * j 0x80068200 + * sll v0,v0,0x2 ; index * 36 (delay slot) + * FC: li v0,72 ; the special case, placed AFTER the general one + * 00: lui at,0x8013 ; %hi of the table base + * addu at,at,v0 ; table + offset (base first) + * lw v0,11144(at) ; word = table[offset] (%lo as displacement) + * nop + * srl v0,v0,0x1a ; >> 26 + * jr ra + * andi v0,v0,0x1f ; & 0x1f (delay slot) + * + * TWO source-shape levers are load-bearing here, both measured: + * - The blocks are MIRRORED: the `index == 1` arm is emitted after the general + * arm and reached by a forward branch, so the natural + * `if (index == 1) ... else ...` spelling emits the inverted `bne` and the + * wrong block order (20 differing bytes). Writing the guard as + * `if (index != 1)` reproduces the original (cookbook finding 28). + * - The base must be an address-shaped SYMBOL: with a literal base `cc1` emits + * `addu at,v0,at` (offset first); the symbol spelling emits + * `addu at,at,v0` and folds `%lo` into the load displacement. Same lever that + * fixed 0x80016174. + * + * LIMITS: the table base 0x80132B88, the special-case offset (72), the element + * stride (36) and the field extraction (>>26, &0x1f) are hypotheses read from the + * instruction shape; what the table holds is unknown and is not guessed here. The + * `srl` before the `andi` shows the load is read as UNSIGNED — a signed `>>` + * would emit `sra`. Only the compiled bytes are evidence. + */ + +extern int D_80132B88[]; + +int func_800681E0(int index) { + int offset; + + if (index != 1) + offset = index * 36; + else + offset = 72; + + return (*(unsigned int *)((char *)D_80132B88 + offset) >> 26) & 0x1F; +} diff --git a/src/func_8007C4EC.c b/src/func_8007C4EC.c new file mode 100644 index 0000000..d77f735 --- /dev/null +++ b/src/func_8007C4EC.c @@ -0,0 +1,56 @@ +/* + * func_8007C4EC — 56 bytes at 0x8007C4EC..0x8007C524 + * + * Leaf routine that copies three words from an argument array into a 16-byte + * record selected by a byte index, and stores a fourth value in the record's + * last word. + * + * The observed instructions are: + * lui v1,0x8014 3c038014 \ + * addiu v1,v1,-0x79f8 24638608 / v1 = 0x80138608 (D_80138608) + * lbu v0,0x25(a0) 90820025 v0 = *(unsigned char *)(a0 + 0x25) + * lw a0,0x0(a1) 8ca40000 a0 = a1[0] + * sll v0,v0,0x4 00021100 v0 = index * 16 + * addu v0,v0,v1 00431021 v0 = (index * 16) + base + * sw a0,0x0(v0) ac440000 *(int *)(v0 + 0x0) = a1[0] + * lw v1,0x4(a1) 8ca30004 v1 = a1[1] + * nop 00000000 load-delay slot + * sw v1,0x4(v0) ac430004 *(int *)(v0 + 0x4) = a1[1] + * lw v1,0x8(a1) 8ca30008 v1 = a1[2] + * sw a2,0xc(v0) ac42000c *(int *)(v0 + 0xc) = a2 + * jr ra 03e00008 + * sw v1,0x8(v0) ac430008 *(int *)(v0 + 0x8) = a1[2] (delay slot) + * + * The record base 0x80138608 is materialised as `lui`+`addiu`, the + * linker-resolved symbol form (cookbook finding 4), so it is written as the named + * symbol `D_80138608`; a literal address would give `lui`+`ori` (finding 5). The + * index is a byte field at +0x25 of the first argument and is scaled by 16, so + * each record is 16 bytes. The address arithmetic is stride first, base second + * (`sll` then `addu`), which finding 22 identifies as `(index * 16) + symbol`. + * + * The `nop` between the second load and its store is a load-delay slot in front of + * a *store of the loaded register* — the case cookbook finding 27 records as + * maspsx's predicate gap, which does not bite here: the candidate came back with + * the right length. + * + * The store order in the original is 0x0, 0x4, 0xc, 0x8, i.e. the record's third + * word is stored last and lands in the `jr ra` delay slot; the C therefore writes + * the four stores in address order and leaves the scheduling to cc1. + * + * LIMITS: the function name, the 16-byte record layout and the claim that a1 is a + * three-word source array are hypotheses; only the bytes are evidence. Whether the + * source used element stores or a struct assignment is not recoverable from the + * bytes; the element form is used here. + */ + +extern char D_80138608[]; + +void func_8007C4EC(int a0, int *a1, int a2) +{ + int *dst = (int *)(D_80138608 + *(unsigned char *)(a0 + 0x25) * 16); + + dst[0] = a1[0]; + dst[1] = a1[1]; + dst[2] = a1[2]; + dst[3] = a2; +} diff --git a/src/func_800827A8.c b/src/func_800827A8.c new file mode 100644 index 0000000..5ae21cc --- /dev/null +++ b/src/func_800827A8.c @@ -0,0 +1,29 @@ +/* + * func_800827A8 — 28 bytes at 0x800827A8..0x800827C4 + * + * Indexes a table of 16-byte elements through a global base pointer and returns + * a masked 32-bit word from the element. The base is loaded through the symbol + * macro and the `addu` takes the scaled index first and the base second, which + * is the spelling recorded as cookbook finding 22 (direct expression form). + * + * The observed instructions are: + * lui v0,0x8012 ; symbol load expands into v0 itself + * lw v0,8968(v0) ; v0 = D_80122308 (a base pointer) + * sll a0,a0,4 ; index *= 16 + * addu a0,a0,v0 ; element = index + base (index first) + * lw v0,0(a0) ; word = element->word_00 + * jr ra + * andi v0,v0,0x7ff ; return word & 0x7ff (delay slot) + * + * LIMITS: the symbol name D_80122308, the element stride (16), the field offset + * (0) and the mask (0x7ff) are hypotheses read from the instruction shape. The + * operand order of the `addu` is a source-spelling artifact, not a compiler + * fact; if the direct spelling fails, re-spell the addition before hunting for a + * flag (finding 22). Only the compiled bytes are evidence. + */ + +extern int D_80122308; + +int func_800827A8(int index) { + return *(int *)(D_80122308 + index * 16) & 0x7FF; +} diff --git a/src/func_80083440.c b/src/func_80083440.c new file mode 100644 index 0000000..d9be984 --- /dev/null +++ b/src/func_80083440.c @@ -0,0 +1,46 @@ +/* func_80083440 — 0x80083440..0x80083470 (48 bytes). + * + * Original words: + * lui a1,0x8012 + * lw a1,0x2308(a1) a1 = D_80122308 + * sll a0,a0,0x4 index *= 16 + * addu a1,a1,a0 + * lw v0,0x0(a1) + * li v1,-0x3801 ~0x3800 + * and v0,v0,v1 + * ori v0,v0,0x2800 + * li v1,-0x4001 ~0x4000 + * and v0,v0,v1 + * jr ra + * _sw v0,0x0(a1) (delay slot) + * + * A read-modify-write of one 16-byte-strided element in a table whose base is a + * global *value* (the `lui`+`lw` pair is a symbol load into the same register, + * cookbook finding 2, so 0x80122308 holds a pointer, not the table). The value + * is cleared of bits 11-13 and bit 14, then bit 13 and bit 11 are set: + * `(v & ~0x3800) | 0x2800`, then `& ~0x4000`. + * + * The masks are written as `~` of the set bits rather than as the raw + * 0xFFFFC7FF/0xFFFFBFFF constants, which is what makes the `li` constants fall + * out; the store lands in the `jr ra` delay slot (finding 8). + * + * LIMITS: the stride 16, the displaced base and the three mask bits are read + * from the bytes; what the field means and why bits 11/13 are set while 12 and 14 + * are cleared is not observable. `D_80122308` is typed `int` because the only + * proven use is as an integer base added to a scaled index — it may really be a + * pointer, in which case the type is cosmetic here. + */ + +extern int D_80122308; + +int func_80083440(int index) +{ + int *element = (int *)(D_80122308 + (index << 4)); + int value = *element; + + value &= ~0x3800; + value |= 0x2800; + value &= ~0x4000; + *element = value; + return value; +} diff --git a/src/func_8008F4F4.c b/src/func_8008F4F4.c new file mode 100644 index 0000000..7eb060e --- /dev/null +++ b/src/func_8008F4F4.c @@ -0,0 +1,30 @@ +/* func_8008F4F4 — 0x8008F4F4..0x8008F508 (20 bytes). + * + * Original words: + * 0x3C018012 lui at,0x8012 + * 0x00240821 addu at,at,a0 + * 0x80221ED8 lb v0,0x1ed8(at) + * 0x03E00008 jr ra + * 0x00000000 nop + * + * A signed byte read from a fixed table indexed by the argument. The + * lui/addu/lb shape is the assembler's expansion of a symbol-plus-register + * address (cookbook finding 1: this compiler emits the macro form by default), + * so the base is a symbol at 0x80121ED8, not a literal — a literal would give + * lui+ori (finding 5). + * + * `lb` not `lbu` is the discriminator for signedness: plain `char` is unsigned + * on this target and would emit `lbu` (finding 7), so the element type is + * explicitly `signed char`. That is a byte-visible choice, not cosmetic. + * + * LIMITS: the symbol name D_80121ED8 is the project's address-derived + * convention. The array's length and the meaning of the index are unknown; the + * body bounds nothing. + */ + +extern signed char D_80121ED8[]; + +signed char func_8008F4F4(int index) +{ + return D_80121ED8[index]; +} diff --git a/src/func_80092088.c b/src/func_80092088.c new file mode 100644 index 0000000..88d8bd8 --- /dev/null +++ b/src/func_80092088.c @@ -0,0 +1,44 @@ +/* func_80092088 — 0x80092088..0x800920BC (52 bytes). + * + * Original words: + * beq a0,zero,0x800920B4 + * _clear v0 (delay slot) + * lw a0,0xc(a0) + * nop + * bne a0,zero,0x800920A8 + * _li v0,0x1 (delay slot) + * j 0x800920B4 + * _clear v0 (delay slot) + * 0x800920A8: lw v1,0x160(a0) + * nop + * sw v1,0x0(a1) + * 0x800920B4: jr ra + * nop + * + * A doubly-guarded fetch-and-report: a null argument returns 0, a null inner + * pointer returns 0, and otherwise the inner pointer's field 0x160 is published + * through the out-parameter and the routine reports 1. + * + * Both null tests put their result constant in the branch delay slot, so v0 is + * the return register throughout and no extra instruction materialises it + * (cookbook finding 8). The success store is reached by the `bne` and then falls + * into the shared `jr ra`, so v0 = 1 survives from the delay slot into the + * return — that is why `li v0,1` sits where it does. + * + * LIMITS: the displacements 0xc and 0x160 are read from the bytes; the object + * at 0xc is assumed to be a distinct node from the one at 0x160 and that is not + * proved. `int` rather than a pointer type for the out-parameter keeps the + * `sw` width honest but says nothing about what the published value is. + */ + +int func_80092088(int node, int *out) +{ + if (node == 0) + return 0; + node = *(int *)(node + 0xc); + if (node != 0) { + *out = *(int *)(node + 0x160); + return 1; + } + return 0; +} diff --git a/src/func_80092130.c b/src/func_80092130.c new file mode 100644 index 0000000..581fa61 --- /dev/null +++ b/src/func_80092130.c @@ -0,0 +1,33 @@ +/* + * func_80092130 — 28 bytes at 0x80092130..0x8009214C + * + * Leaf routine that chases two pointers and returns a sign-extended halfword. + * + * The observed instructions are: + * lw v0,0x0c(a0) 8c82000c v0 = *(int *)(a0 + 0x0c) + * nop 00000000 load-delay slot + * lw v0,0x160(v0) 8c420160 v0 = *(int *)(v0 + 0x160) + * nop 00000000 load-delay slot + * lh v0,0x0002(v0) 84420002 return *(short *)(v0 + 2) + * jr ra 03e00008 + * nop 00000000 (delay slot) + * + * Both `nop`s are load-delay slots: each is followed by an instruction that + * reads the register just loaded. This is the chained-dereference case that + * maspsx does pad (cookbook finding 27's counterexample at 0x800C5C84 is the + * same shape), so no `maspsx=off` override is needed. The final `nop` is the + * `jr ra` delay slot: nothing independent is available to fill it. + * + * LIMITS: the function name, the struct layout behind a0+0x0c and the meaning + * of the halfword at +2 are hypotheses from the disassembly. The load widths + * and the sign-extending `lh` are evidence. `short` as the return type is what + * makes cc1 emit `lh` rather than `lhu`; cookbook finding 7 (plain `char` is + * unsigned here) does not apply to `short`. + */ + +short func_80092130(int a0) +{ + int v0 = *(int *)(a0 + 0x0c); + v0 = *(int *)(v0 + 0x160); + return *(short *)(v0 + 2); +} diff --git a/src/func_80099E14.c b/src/func_80099E14.c new file mode 100644 index 0000000..e25995f --- /dev/null +++ b/src/func_80099E14.c @@ -0,0 +1,34 @@ +/* func_80099E14 — 0x80099E14..0x80099E34 (32 bytes). + * + * Original words: + * 0x8C830008 lw v1,0x8(a0) + * 0x00000000 nop + * 0x9062000B lbu v0,0xb(v1) + * 0x00000000 nop + * 0x304200F9 andi v0,v0,0xf9 + * 0xA062000B sb v0,0xb(v1) + * 0x03E00008 jr ra + * 0x24020001 _li v0,0x1 (delay slot) + * + * Clears bits 1 and 2 of a byte field inside a structure reached through one + * pointer hop, and reports success. `andi 0xf9` is a clear-mask (0xf9 = ~0x06), + * so the stored value is `field & ~0x06`. + * + * `lbu` and not `lb` proves the field is read unsigned (cookbook finding 7: + * plain `char` is unsigned here, so the element type is a plain `unsigned + * char`). The `li v0,1` sits in the `jr ra` delay slot, so the return value is + * computed last and costs no extra instruction. + * + * LIMITS: offsets 0x8 and 0xb are read from the displacements; the structure + * they belong to is unknown. Which two bits 0x06 names is not observable. The + * receiver of the pointer at 0x8 is not proven to be the same object whose + * offset 0x80 other routines in this area touch. + */ + +int func_80099E14(int base) +{ + unsigned char *p = *(unsigned char **)(base + 8); + + p[0xb] &= 0xf9; + return 1; +} diff --git a/src/func_8009D8A0.c b/src/func_8009D8A0.c new file mode 100644 index 0000000..80c0f8f --- /dev/null +++ b/src/func_8009D8A0.c @@ -0,0 +1,50 @@ +/* + * func_8009D8A0 — 64 bytes at 0x8009D8A0..0x8009D8E0 + * + * Splices a node into a list held by a structure reached through a pointer at + * offset 8: takes the list head out of the structure, links the new node's back + * pointer to it, puts the new node at the head, and fixes up the old head's back + * pointer when there was one. Returns 1. + * + * The observed instructions are: + * sw zero,0(a1) ; out->word_00 = 0 + * lw v0,8(a0) ; p->ptr_08 + * nop + * lw v0,40(v0) ; ->word_28 (the current list head) + * nop + * sw v0,4(a1) ; out->word_04 = old head + * lw v0,8(a0) ; RELOAD p->ptr_08 + * nop + * sw a1,40(v0) ; ->word_28 = out (new head) + * lw v0,4(a1) ; out->word_04 (old head) + * nop + * beqz v0,0x8009D8D8 ; if (old head == 0) skip + * nop + * sw a1,0(v0) ; old head->word_00 = out + * D8: jr ra + * li v0,1 ; return 1 (delay slot) + * + * The `p->ptr_08` load appears TWICE in the original and the C must therefore + * re-read it rather than cache it in a local — the two loads are the same + * address, so caching would remove the second one and change the bytes. + * + * LIMITS: every offset and the parameter types are hypotheses read from the + * instruction shape; what the structures mean is unknown and is not guessed here. + * `out` is typed `int *` because its fields are stored as words and the back + * pointer at offset 0 is compared against zero. Only the compiled bytes are + * evidence. + */ + +int func_8009D8A0(char *p, int *out) { + int old_head; + + out[0] = 0; + out[1] = *(int *)(*(char **)(p + 8) + 40); + *(int *)(*(char **)(p + 8) + 40) = (int)out; + + old_head = out[1]; + if (old_head != 0) + *(int *)old_head = (int)out; + + return 1; +} diff --git a/src/func_8009F0E8.c b/src/func_8009F0E8.c new file mode 100644 index 0000000..6e7e0b6 --- /dev/null +++ b/src/func_8009F0E8.c @@ -0,0 +1,40 @@ +/* + * func_8009F0E8 — 56 bytes at 0x8009F0E8..0x8009F120 + * + * Reads three adjacent 16-bit fields out of a structure reached through a pointer + * at offset 0x0c, writes them into a three-word output array, and adds a delta to + * the middle one. The middle word is *reloaded* from the output array before the + * add, because the output pointer may alias — so the C must read it back rather + * than reuse a local. + * + * The observed instructions are: + * lw v1,12(a0) ; s = p->ptr_0c + * nop + * lh v0,264(v1) ; s->half_108 + * nop + * sw v0,0(a2) ; out[0] = s->half_108 + * lh v0,266(v1) ; s->half_10a + * nop + * sw v0,4(a2) ; out[1] = s->half_10a + * lw v0,4(a2) ; reload out[1] + * lh v1,268(v1) ; s->half_10c + * addu v0,v0,a1 ; out[1] + delta + * sw v1,8(a2) ; out[2] = s->half_10c + * jr ra + * sw v0,4(a2) ; out[1] += delta (delay slot) + * + * LIMITS: every offset and the `short` field type are hypotheses read from the + * instruction shape — the `lh` is what shows the fields are signed 16-bit. The + * reload of out[1] is a real observable in the original and is reproduced here by + * using the `+=` form on the array element; binding it to a local first would + * remove the reload and change the bytes. Only the compiled bytes are evidence. + */ + +void func_8009F0E8(char *base, int delta, int *out) { + short *s = *(short **)(base + 12); + + out[0] = s[132]; + out[1] = s[133]; + out[2] = s[134]; + out[1] += delta; +} diff --git a/src/func_800A2F20.c b/src/func_800A2F20.c new file mode 100644 index 0000000..a569938 --- /dev/null +++ b/src/func_800A2F20.c @@ -0,0 +1,51 @@ +/* func_800A2F20 — 0x800A2F20..0x800A2F44 (36 bytes). + * + * Original words: + * 0x24020258 li v0,0x258 + * 0x3C018014 lui at,0x8014 <- loop top + * 0x00220821 addu at,at,v0 + * 0xA0209E01 sb zero,-0x61ff(at) + * 0x2442FF9C addiu v0,v0,-0x64 + * 0x0441FFFB bgez v0,loop + * 0x00000000 _nop (delay slot) + * 0x03E00008 jr ra + * 0x00000000 nop + * + * A descending byte-clear sweep. The count starts at 0x258 and steps down by + * 0x64 (100), storing a zero byte each time, stopping once the counter goes + * negative. + * + * The `lui`/`addu`/`sb` triple is the assembler's expansion of a + * symbol-plus-register store (cookbook finding 1), so the target is a symbol + * base indexed by the counter, not a moving pointer. `%hi` 0x8014 with `%lo` + * -0x61ff places the base at 0x80139E01. + * + * The decrement is emitted *before* the `bgez` and the branch tests the + * decremented value, with the first iteration reached by falling out of the + * `li` — so the loop is bottom-tested. Spelling it as a `for` with the test + * normally at the top relies on cc1 rotating the loop; if it does not, the + * shape to try is an explicit `do`/`while` (cookbook finding 19's lever, whose + * limit is that it applies only where the frame or rotation is the anomaly). + * + * LIMITS: the stride 0x64, the start 0x258 and the base 0x80139E01 are read + * from the bytes; what the swept bytes mean is unknown. Note the base is + * deliberately odd (…E01), consistent with a byte array whose real start is + * elsewhere — the symbol name is the literal address and carries no meaning. + * + * ATTEMPT 1 (recorded): binding the base to a named `char *p` local and writing + * `p[i] = 0` let cc1 strength-reduce the address into a second induction + * variable — it emitted `lui`+`addiu` once, `sb zero,0(v0)`, and moved the + * pointer instead of indexing, giving 16 differing bytes. The original keeps + * the index form (`addu at,at,v0` every iteration), so the base must stay a + * symbol in the index expression and no pointer local may be introduced. + */ + +extern char D_80139E01[]; + +void func_800A2F20(void) +{ + int i; + + for (i = 0x258; i >= 0; i -= 0x64) + D_80139E01[i] = 0; +} diff --git a/src/func_800A8B48.c b/src/func_800A8B48.c new file mode 100644 index 0000000..c3de75c --- /dev/null +++ b/src/func_800A8B48.c @@ -0,0 +1,70 @@ +/* + * func_800A8B48 — 68 bytes at 0x800A8B48..0x800A8B8C + * + * Leaf routine that stores three bytes into a 24-byte record selected by an + * index, with an out-of-range index selecting nothing. + * + * The observed instructions are: + * sltiu v0,a0,0x8 2c820008 v0 = (a0 < 8) unsigned + * bne v0,zero,0x800a8b5c 14400003 if (in range) goto the address build + * sll v0,a0,0x1 00042040 v0 = a0 * 2 (delay slot) + * j 0x800a8b70 081002dc goto the null test + * addu v0,zero,zero 00001021 v0 = 0 (delay slot) + * 0x800a8b5c: + * addu v0,v0,a0 00441021 v0 = a0 * 2 + a0 (= a0 * 3) + * sll v0,v0,0x3 000420c0 v0 *= 8 (= a0 * 24) + * lui v1,0x8014 3c038014 \ + * addiu v1,v1,-0x4088 2463bf78 / v1 = 0x8013BF78 (D_8013BF78) + * addu v0,v0,v1 00431021 v0 = (index * 24) + base + * 0x800a8b70: + * beq v0,zero,0x800a8b84 10400004 if (p == 0) goto epilogue + * nop 00000000 (delay slot) + * sb a1,0x4(v0) a0450004 *(char *)(p + 4) = a1 + * sb a2,0x5(v0) a0460005 *(char *)(p + 5) = a2 + * sb a3,0x6(v0) a0470006 *(char *)(p + 6) = a3 + * 0x800a8b84: + * jr ra 03e00008 + * nop 00000000 (delay slot) + * + * The `sltiu` bound test makes the index **unsigned**; the multiply is + * strength-reduced to `sll`+`addu`+`sll` because 24 is not a power of two, which + * also fixes the record stride as 24 bytes. + * + * The two arms are laid out **mirrored** from the natural spelling: the in-range + * path is the branch **target** (`bne v0,zero,0x800a8b5c`) and the out-of-range + * path is the fall-through, which materialises the null pointer in the delay slot + * of a `j` to the merge point. Writing `if (a0 < 8) p = base + a0 * 24; else p = 0;` + * produces the opposite layout (measured: 68 bytes, 25 differing bytes, first at + * 0x800A8B4C); inverting the condition to `if (a0 >= 8) p = 0; else p = ...` + * reproduces the original exactly. This is cookbook finding 28's mirrored-layout + * lever, and its diagnostic signature held — the instruction count was already + * correct and only the block order was wrong. + * + * The base 0x8013BF78 is materialised as `lui`+`addiu`, the linker-resolved symbol + * form (cookbook finding 4), so it is written as the named symbol `D_8013BF78` + * (a literal would give `lui`+`ori`, finding 5). The address arithmetic is stride + * first, base second. + * + * LIMITS: the function name, the 24-byte record layout and the meaning of the + * three stored bytes are hypotheses; only the bytes are evidence. The function + * returns nothing observable (v0 holds the record pointer on one path and 0 on the + * other, but the caller cannot rely on either), so the return type is `void`. + */ + +extern char D_8013BF78[]; + +void func_800A8B48(unsigned int a0, int a1, int a2, int a3) +{ + char *p; + + if (a0 >= 8) + p = 0; + else + p = D_8013BF78 + a0 * 24; + + if (p != 0) { + p[4] = a1; + p[5] = a2; + p[6] = a3; + } +} diff --git a/src/func_800B24EC.c b/src/func_800B24EC.c new file mode 100644 index 0000000..11295e6 --- /dev/null +++ b/src/func_800B24EC.c @@ -0,0 +1,68 @@ +/* + * func_800B24EC — 72 bytes at 0x800B24EC..0x800B2534 + * + * Leaf routine that follows two pointers with null checks and then copies a + * four-byte value into an object at +0x194 and sets a flag byte at +0x197. + * + * The observed instructions are: + * beq a0,zero,0x800b252c 1080000f if (a0 == 0) goto epilogue + * move t0,a1 00a04021 t0 = a1 (delay slot) + * lw a0,0xc(a0) 8c84000c a0 = *(int *)(a0 + 0xc) + * nop 00000000 load-delay slot + * beq a0,zero,0x800b252c 1080000c if (a0 == 0) goto epilogue + * nop 00000000 (delay slot) + * lw a3,0x158(a0) 8c870158 a3 = *(int *)(a0 + 0x158) + * nop 00000000 load-delay slot + * beq a3,zero,0x800b252c 10e0000a if (a3 == 0) goto epilogue + * nop 00000000 (delay slot) + * lwl v0,0x3(t0) a9030003 \ + * lwr v0,0x0(t0) a9030000 / v0 = unaligned 4-byte load from a1 + * nop 00000000 load-delay slot + * swl v0,0x197(a3) a8e20197 \ + * swr v0,0x194(a3) a8e20194 / unaligned 4-byte store to a3 + 0x194 + * sb a2,0x197(a3) a0e20197 *(char *)(a3 + 0x197) = a2 + * 0x800b252c: + * jr ra 03e00008 + * nop 00000000 (delay slot) + * + * Three chained null checks, each exiting to one shared epilogue, with the second + * argument copied to t0 in the first branch's delay slot — so the copy is written + * as the source's own local and cc1 hoisted it. + * + * The `lwl`/`lwr` and `swl`/`swr` pairs are **unaligned** 32-bit accesses, which + * this compiler emits only when the type's alignment is 1: a plain `*(int *)` + * through a cast would give `lw`/`sw`. The source is therefore written as a + * 4-byte `char[4]` structure copy, whose alignment is 1, rather than as an int + * assignment. The byte at +0x197 is then written separately, overwriting the top + * byte of the four just stored. + * + * LIMITS: the function name, the pointer chain, the offsets and the meaning of the + * copied value and the flag byte are hypotheses; only the bytes are evidence. The + * `char[4]` struct is a reconstruction of the access *alignment*, not evidence of + * the original's types; whether the original used a packed struct, a char-array + * copy or a byte-wise copy is not recoverable from these instructions alone. + */ + +struct func_800B24EC_quad { + char b[4]; +}; + +void func_800B24EC(int a0, void *a1, int a2) +{ + void *t0 = a1; + int a3; + + if (a0 == 0) + return; + + a0 = *(int *)(a0 + 0xc); + if (a0 == 0) + return; + + a3 = *(int *)(a0 + 0x158); + if (a3 == 0) + return; + + *(struct func_800B24EC_quad *)(a3 + 0x194) = *(struct func_800B24EC_quad *)t0; + *(char *)(a3 + 0x197) = a2; +} diff --git a/src/func_800B3474.c b/src/func_800B3474.c new file mode 100644 index 0000000..49f1089 --- /dev/null +++ b/src/func_800B3474.c @@ -0,0 +1,39 @@ +/* + * func_800B3474 — 48 bytes at 0x800B3474..0x800B34A4 + * + * Two chained pointer lookups through a table whose base lives in a global. The + * index arrives as a 16-bit signed value, so the original sign-extends it with + * `sll`+`sra` (the two shifts are fused into one `sra` by 14) and scales by four. + * + * The observed instructions are: + * sll a0,a0,0x10 ; index <<= 16 + * lui v0,0x8012 ; symbol load expands into v0 itself + * lw v0,7168(v0) ; v0 = D_80121C00 (a table base) + * sra a0,a0,0xe ; (short)index * 4 + * addu a0,a0,v0 ; table + index*4 (index first) + * lw v0,0(a0) ; entry = table[index] + * nop + * lw v0,8(v0) ; entry->ptr_08 + * nop + * lw v0,0(v0) ; *entry->ptr_08 + * jr ra + * sltu v0,zero,v0 ; return that != 0 (delay slot) + * + * The `sll`/`sra` pair is what shows the parameter is a signed 16-bit value: an + * `int` index would be used directly. + * + * LIMITS: the symbol name D_80121C00, the pointee types, the element stride (4) + * and both field offsets are hypotheses read from the instruction shape; what the + * structures mean is unknown and is not guessed here. The `addu` takes the scaled + * index first and the base second, which is the direct-expression spelling + * recorded as cookbook finding 22. Only the compiled bytes are evidence. + */ + +extern int D_80121C00; + +int func_800B3474(short index) { + int *entry = *(int **)(D_80121C00 + index * 4); + int *inner = *(int **)((char *)entry + 8); + + return *inner != 0; +} diff --git a/src/func_800B7230.c b/src/func_800B7230.c new file mode 100644 index 0000000..67b4ac4 --- /dev/null +++ b/src/func_800B7230.c @@ -0,0 +1,53 @@ +/* + * func_800B7230 — 52 bytes at 0x800B7230..0x800B7264 + * + * Walks a singly-linked list and clears three bits in a 16-bit field of every + * node. The mask constant is loop-invariant, so `cc1` hoists its materialisation + * out of the loop and keeps it in a register — which is why the original emits + * `li v1,-15` once rather than an `andi` with an immediate. + * + * The mask cannot be an immediate, and the reason is the type of the temporary + * rather than the field: `~14` is 0xFFFFFFF1, which `andi` cannot encode (it + * zero-extends). Masking the 16-bit field directly + * (`p->flags &= ~14;`) lets `cc1` narrow the operation to `andi v0,v0,0xfff1` + * and loses 4 bytes; masking through an `int` temporary keeps the 32-bit `and`. + * + * The observed instructions are: + * beqz a0,0x800B725C ; if (p == 0) return + * nop + * li v1,-15 ; mask = ~14, hoisted out of the loop + * 3C:lhu v0,26(a0) ; p->half_1a + * nop + * and v0,v0,v1 ; & ~14 + * sh v0,26(a0) ; p->half_1a = masked + * lw a0,4(a0) ; p = p->next + * nop + * bnez a0,0x800B723C ; loop while p != 0 + * nop + * 5C:jr ra + * nop + * + * LIMITS: the struct layout (next pointer at 0x04, the masked field at 0x1a) and + * the field's `unsigned short` type are hypotheses read from the instruction + * shape; the `lhu` is what shows the field is unsigned, and the mask value is a + * fact about this executable. The loop is a plain `while` — the test appears + * twice because that is how `cc1` compiles one, not because the source repeats + * it. Only the compiled bytes are evidence. + */ + +typedef struct func_800B7230_node { + int unknown_00; + struct func_800B7230_node *next; /* offset 0x04 */ + char unknown_08[0x12]; + unsigned short flags; /* offset 0x1a */ +} func_800B7230_node; + +void func_800B7230(func_800B7230_node *p) { + while (p != 0) { + int flags = p->flags; + + flags &= ~14; + p->flags = flags; + p = p->next; + } +} diff --git a/src/func_800F5200.c b/src/func_800F5200.c new file mode 100644 index 0000000..10ecdde --- /dev/null +++ b/src/func_800F5200.c @@ -0,0 +1,40 @@ +/* func_800F5200 — 0x800F5200..0x800F521C (28 bytes). + * + * Original words: + * 0x30A507FF andi a1,a1,0x7ff + * 0x00052AC0 sll a1,a1,0xb + * 0x308207FF andi v0,a0,0x7ff + * 0x3C03E500 lui v1,0xe500 + * 0x00431025 or v0,v0,v1 + * 0x03E00008 jr ra + * 0x00A21025 or v0,a1,v0 (jr ra delay slot) + * + * A packed 32-bit command word built from two 11-bit fields: the second + * argument is masked and shifted to bit 11, the first is masked and OR-ed with + * the 0xe5000000 tag, and the two halves are combined in the delay slot. + * + * The two halves are bound to named locals in the order the original emits + * them. Writing the same expression as one nested `|` makes cc1 reassociate it: + * `((field & 0x7ff) << 11) | ((value & 0x7ff) | 0xe5000000)` is `A | (B | C)`, + * but cc1 canonicalises the associative operator to `(A | C) | B`, emitting + * `lui` before the second `andi` and allocating the field to v0 — 16 differing + * bytes at the right instruction count. Binding `shifted` first and `tag` second + * pins both the emission order and the registers (the field stays in place in + * a1, the tag lands in v0). + * + * LIMITS: `0xe5000000` is a literal from the original's `lui v1,0xe500`; whether + * the real source named a constant is unobservable. The field widths (11 bits + * each) and the mask values are read from the bytes. That the named locals are + * what fixes the order is an observation about this compiler, not a rule — the + * same lever failed on the constant-materialisation-order rows recorded in + * cookbook finding 28. The function's meaning — a packed hardware command — is + * a guess and is not evidence. + */ + +int func_800F5200(int value, int field) +{ + int shifted = (field & 0x7ff) << 11; + int tag = (value & 0x7ff) | 0xe5000000; + + return shifted | tag; +} diff --git a/src/func_800F6570.c b/src/func_800F6570.c new file mode 100644 index 0000000..c3a6227 --- /dev/null +++ b/src/func_800F6570.c @@ -0,0 +1,34 @@ +/* + * func_800F6570 — 36 bytes at 0x800F6570..0x800F6594 + * + * Fills a run of bytes with a caller-supplied value — a `memset` shape. The + * count is tested before the first store and the counter is pre-decremented, so + * the source uses an explicit guard plus a `do`/`while` (cookbook finding 19). + * + * The observed instructions are: + * beqz a2,0x800F658C ; if (count == 0) skip the loop + * addiu v0,a2,-1 ; counter = count - 1 (delay slot) + * li v1,-1 ; loop sentinel + * L: sb a1,0(a0) ; *p = value + * addiu v0,v0,-1 ; --counter + * bne v0,v1,0x800F657C ; loop while counter != -1 + * addiu a0,a0,1 ; ++p (delay slot) + * jr ra + * nop + * + * LIMITS: finding 19's scope applies — the explicit-guard spelling is used here + * because it is the shape the original shows, not because it generalises; it was + * recorded as ineffective on a row with no such frame. The parameter types + * (`char *`, `int` value, `int` count) are hypotheses read from the instruction + * shape. The trailing `nop` is a scheduling fill, not source. Only the compiled + * bytes are evidence. + */ + +void func_800F6570(char *dest, int value, int count) { + int remaining = count - 1; + + if (count != 0) + do { + *dest++ = value; + } while (--remaining != -1); +} diff --git a/src/func_800F7A94.c b/src/func_800F7A94.c new file mode 100644 index 0000000..66d2249 --- /dev/null +++ b/src/func_800F7A94.c @@ -0,0 +1,30 @@ +/* + * func_800F7A94 — 24 bytes at 0x800F7A94..0x800F7AAC + * + * Reads a 16-bit value out of a heap/global structure, replaces it, and returns + * the previous value. The original reloads the structure pointer through the + * symbol macro and lets the store land in the `jr ra` delay slot. + * + * The observed instructions are: + * lui v1,0x8012 ; symbol load expands into v1 itself + * lw v1,-1196(v1) ; v1 = D_8011FB54 (a pointer) + * nop ; maspsx load-delay nop + * lhu v0,0(v1) ; v0 = *v1 + * jr ra + * sh a0,0(v1) ; *v1 = value (delay slot) + * + * LIMITS: the symbol name D_8011FB54, the `unsigned short` element type and the + * `unsigned short` parameter type are hypotheses read from the instruction + * shape. The `nop` is maspsx's load-delay fill, not source. Only the compiled + * bytes are evidence. + */ + +extern int D_8011FB54; + +unsigned short func_800F7A94(unsigned short value) { + unsigned short *p = (unsigned short *)D_8011FB54; + unsigned short previous = *p; + + *p = value; + return previous; +} diff --git a/src/func_800FF6A4.c b/src/func_800FF6A4.c new file mode 100644 index 0000000..2139a32 --- /dev/null +++ b/src/func_800FF6A4.c @@ -0,0 +1,48 @@ +/* + * func_800FF6A4 — 56 bytes at 0x800FF6A4..0x800FF6DC + * + * Searches a singly-linked list for the node whose key field equals the argument + * and returns that node (or null). The list head is a gp-relative global, so the + * symbol needs a `gp` registry row; the routine itself sets up no frame. + * + * The observed instructions are: + * lw v1,2080(gp) ; node = g_80122158 (list head) + * nop + * beqz v1,0x800FF6D4 ; if (node == 0) return 0 + * nop + * B4: lw v0,8(v1) ; node->key_08 + * nop + * beq v0,a0,0x800FF6D4 ; if (key == want) break + * nop + * lw v1,12(v1) ; node = node->next_0c + * nop + * bnez v1,0x800FF6B4 ; loop while node != 0 + * nop + * D4: jr ra + * move v0,v1 ; return node (delay slot) + * + * The `break` form (rather than two early `return`s) is what keeps a single + * epilogue with the result already in `v1`; two returns would duplicate the + * `jr ra`. + * + * LIMITS: the symbol name g_80122158, the offsets (key at 0x08, next at 0x0c) and + * the parameter type are hypotheses read from the instruction shape; the offset + * 2080 is a fact about this executable's gp layout (gp = 0x80121938, cookbook + * findings 10 and 14). The `gp` marker on the symbol is required and is requested + * from the coordinator, not edited into the tracked registry by this session. + * Only the compiled bytes are evidence. + */ + +extern int g_80122158; + +int func_800FF6A4(int wanted) { + int node = g_80122158; + + while (node != 0) { + if (*(int *)(node + 8) == wanted) + break; + node = *(int *)(node + 12); + } + + return node; +} diff --git a/src/func_8010097C.c b/src/func_8010097C.c new file mode 100644 index 0000000..604572c --- /dev/null +++ b/src/func_8010097C.c @@ -0,0 +1,28 @@ +/* + * func_8010097C — 28 bytes at 0x8010097C..0x80100998 + * + * Clears one bit in a 16-bit field at offset 0x40 of a structure. The bit index + * arrives in a register, so the mask is built with a variable shift (`sllv`) + * rather than a constant one. + * + * The observed instructions are: + * li v0,1 ; mask = 1 + * sllv v0,v0,a1 ; mask <<= bit_index + * lhu v1,64(a0) ; field = *(unsigned short *)(p + 0x40) + * nor v0,zero,v0 ; ~mask + * and v1,v1,v0 ; field &= ~mask + * jr ra + * sh v1,64(a0) ; *(unsigned short *)(p + 0x40) = field (delay slot) + * + * LIMITS: the parameter types (`char *` for the base, `int` for the bit index) + * and the field's offset and width are hypotheses read from the instruction + * shape. The return type is `void`: `v0` holds only the mask scratch, and no + * value is live in it at the `jr ra`. Only the compiled bytes are evidence. + */ + +void func_8010097C(char *base, int bit_index) { + unsigned short field = *(unsigned short *)(base + 64); + + field &= ~(1 << bit_index); + *(unsigned short *)(base + 64) = field; +} diff --git a/src/func_80101244.c b/src/func_80101244.c new file mode 100644 index 0000000..e776b8d --- /dev/null +++ b/src/func_80101244.c @@ -0,0 +1,66 @@ +/* + * func_80101244 — 72 bytes at 0x80101244..0x8010128C + * + * Leaf routine that decodes a variable-length 7-bit-per-byte integer and reports + * how many bytes it consumed. + * + * The observed instructions are: + * move a2,a0 00803021 p = a0 + * lbu a0,0x0(a2) 90c40000 value = *p + * nop 00000000 load-delay slot + * andi v0,a0,0x80 30820080 v0 = value & 0x80 + * beq v0,zero,0x80101280 10400009 if (!v0) goto the store + * li a3,0x1 24070001 count = 1 (delay slot) + * 0x8010125c: + * andi a0,a0,0x7f 3084007f value &= 0x7f + * 0x80101260: + * addiu a2,a2,0x1 24c60001 p++ + * addiu a3,a3,0x1 24e70001 count++ + * lbu v0,0x0(a2) 90c20000 v0 = *p + * sll a0,a0,0x7 000421c0 value <<= 7 + * andi v1,v0,0x7f 3043007f v1 = *p & 0x7f + * andi v0,v0,0x80 30420080 v0 = *p & 0x80 + * bne v0,zero,0x80101260 1440fffc if (*p & 0x80) loop + * addu a0,a0,v1 00832021 value += v1 (delay slot) + * 0x80101280: + * sw a3,0x0(a1) aca70000 *a1 = count + * jr ra 03e00008 + * move v0,a0 00801021 v0 = value (delay slot) + * + * The first byte's high bit decides whether the value continues, and the + * `andi a0,a0,0x7f` that strips the continuation bit sits **outside** the loop's + * back edge (the branch at 0x80101278 targets 0x80101260, past it), so the mask is + * applied once for the first byte only. That makes the body a `do`/`while` whose + * condition tests the byte just loaded, and the loop's accumulation + * (`value = (value << 7) + (*p & 0x7f)`) has its add scheduled into the branch + * delay slot. + * + * The `lbu` loads make the bytes unsigned, and the pointer walks one byte at a + * time. The count is stored through the second argument before the value is + * returned, and the `lbu` of the first byte is followed by a load-delay `nop`. + * + * LIMITS: the function name, the claim that this is a 7-bit continuation encoding + * and the meaning of the second argument are hypotheses; only the bytes are + * evidence. The accumulator is written as `int`; a wider or unsigned type is not + * distinguishable from these instructions. + */ + +int func_80101244(unsigned char *a0, int *a1) +{ + unsigned char *p = a0; + int value = *p; + int count = 1; + + if (value & 0x80) { + value &= 0x7f; + do { + p++; + count++; + value = (value << 7) + (*p & 0x7f); + } while (*p & 0x80); + } + + *a1 = count; + + return value; +} diff --git a/src/func_80101CAC.c b/src/func_80101CAC.c new file mode 100644 index 0000000..97dff7b --- /dev/null +++ b/src/func_80101CAC.c @@ -0,0 +1,62 @@ +/* + * func_80101CAC — 48 bytes at 0x80101CAC..0x80101CDC + * + * Loads five consecutive words from a structure and writes them into the GTE's + * control registers $0..$4 — the rotation matrix, packed two values per register + * across five registers. + * + * The observed instructions are: + * 0x8C880000 lw t0,0(a0) ; src[0] + * 0x8C890004 lw t1,4(a0) ; src[1] + * 0x8C8A0008 lw t2,8(a0) ; src[2] + * 0x8C8B000C lw t3,12(a0) ; src[3] + * 0x8C8C0010 lw t4,16(a0) ; src[4] + * 0x48C80000 ctc2 t0,$0 ; RT1RT2 + * 0x48C94800 ctc2 t1,$1 ; RT3RT21 + * 0x48CA5000 ctc2 t2,$2 ; RT22RT23 + * 0x48CB5800 ctc2 t3,$3 ; RT31RT32 + * 0x48CC6000 ctc2 t4,$4 ; RT33 + * 0x03E00008 jr ra + * 0x00000000 nop + * + * The statements go through include/gtemac.h (`gte_ldRT1RT2` … `gte_ldRT33`), + * the re-derived macro form the project convention requires; the loads stay in + * C. The five register bindings are load-bearing and are NOT inline assembly: a + * GNU C local register variable is a documented register-name binding, the same + * mechanism `register int sp __asm__("$29")` uses for the stack accessor + * (cookbook finding 24), which docs/MATCHING_CONVENTIONS.md states needs no + * exemption. Without them `cc1` is free to allocate the five values to whatever + * registers it likes AND to reorder the loads; the original's allocation is + * t0..t4 in order. + * + * The register numbers are the evidence, not a derivation: the control-register + * numbering this executable uses is the standard map unshifted for 0..5 (this + * body) and standard+7 from RBK onward (cookbook findings 24 and the header's + * map). $0..$4 holding a 3x3 rotation matrix packed two-per-register is the + * standard reading of this exact sequence. + * + * LIMITS: the source is `int[5]` and the five register names are hypotheses read + * from the instruction shape. Only the compiled bytes are evidence. + */ + +#include "../include/gtemac.h" + +void func_80101CAC(int *src) { + register int m0 __asm__("$8"); + register int m1 __asm__("$9"); + register int m2 __asm__("$10"); + register int m3 __asm__("$11"); + register int m4 __asm__("$12"); + + m0 = src[0]; + m1 = src[1]; + m2 = src[2]; + m3 = src[3]; + m4 = src[4]; + + gte_ldRT1RT2(m0); + gte_ldRT3RT21(m1); + gte_ldRT22RT23(m2); + gte_ldRT31RT32(m3); + gte_ldRT33(m4); +} \ No newline at end of file diff --git a/src/func_801027CC.c b/src/func_801027CC.c new file mode 100644 index 0000000..d26eb34 --- /dev/null +++ b/src/func_801027CC.c @@ -0,0 +1,51 @@ +/* + * func_801027CC — 44 bytes at 0x801027CC..0x801027F8 + * + * Leaf routine that copies `n` words from one buffer to another, skipping the + * loop entirely when `n` is zero. + * + * The observed instructions are: + * beq a2,zero,0x801027f0 10c00006 if (n == 0) goto the epilogue + * addu v1,zero,zero 00001821 i = 0 (delay slot) + * 0x801027d4: + * lw v0,0x0(a1) 8ca20000 v0 = *src + * addiu a1,a1,0x4 24a50004 src += 4 + * addiu v1,v1,0x1 24630001 i++ + * sw v0,0x0(a0) ac820000 *dst = v0 + * sltu v0,v1,a2 0066102b v0 = (i < n) unsigned + * bne v0,zero,0x801027d4 1440fffa if (v0) loop + * addiu a0,a0,0x4 24840004 dst += 4 (delay slot) + * 0x801027f0: + * jr ra 03e00008 + * nop 00000000 (delay slot) + * + * This is cookbook finding 19's shape — an explicit guard plus a `do`/`while` + * loop, which is what keeps cc1 from emitting the unused `subu sp,sp,8` frame + * the plain `for (i = n - 1; i != -1; i--)` spelling produces. Here the guard is + * `if (n == 0) return;` and the counter is initialised to zero in the branch's + * delay slot. + * + * The bound test is `sltu`, not `slt`, so the comparison is **unsigned**: the + * count is taken as `unsigned int` (with the counter unsigned too, so the + * comparison is between two unsigned values). + * + * LIMITS: the function name and the claim that this is a word-wise memcpy are + * hypotheses; only the bytes are evidence. The four-byte stride is evidence + * (`addiu ...,0x4`). The buffers are written as `int *` because every access is + * 32-bit; no alignment or aliasing claim is made. + */ + +void func_801027CC(int a0, int a1, unsigned int a2) +{ + unsigned int i = 0; + + if (a2 == 0) + return; + + do { + *(int *)a0 = *(int *)a1; + a1 += 4; + i++; + a0 += 4; + } while (i < a2); +} diff --git a/src/func_80102FD4.c b/src/func_80102FD4.c new file mode 100644 index 0000000..cfbd0a5 --- /dev/null +++ b/src/func_80102FD4.c @@ -0,0 +1,30 @@ +/* + * func_80102FD4 — 12 bytes at 0x80102FD4..0x80102FE0 + * + * Leaf routine whose whole body is a single COP2 control-register write. + * + * The observed instructions are: + * ctc2 a0,$26 48c4d000 — GTE control register H + * jr ra 03e00008 + * nop 00000000 (delay slot) + * + * The argument arrives in a0 and the instruction encoding's rt field is 4 + * ($a0), so the source takes the value as a parameter and the compiler is left + * to place it. Register $26 is the GTE H register in this executable's + * numbering (include/gtemac.h records that map, proved by 0x80103A94's + * `cfc2 v0,$26`); the statement goes through the header's re-derived + * `gte_ldH` macro, the form the project convention requires for coprocessor + * instructions. Nothing else in this file is asm. + * + * LIMITS: the function name, the parameter name and the claim that the caller + * passes a projection-plane distance are hypotheses. Only the compiled bytes + * are evidence. The routine writes no memory and returns nothing observable — + * v0 is untouched at the `jr ra`, so the C return type is void. + */ + +#include "../include/gtemac.h" + +void func_80102FD4(int h) +{ + gte_ldH(h); +} \ No newline at end of file diff --git a/src/func_80103B54.c b/src/func_80103B54.c new file mode 100644 index 0000000..a78cc68 --- /dev/null +++ b/src/func_80103B54.c @@ -0,0 +1,26 @@ +/* func_80103B54 — 0x80103B54..0x80103B60 (12 bytes). + * + * Original words: + * 0x48C4D800 ctc2 a0,$27 + * 0x03E00008 jr ra + * 0x00000000 nop + * + * A GTE control-register write, the sibling of the registered func_80103B60 + * (`ctc2 a0,$28`). Control register $27 is DQA in this executable's numbering + * (proved by the sibling's $28 = DQB, whose number is the standard one). + * + * The COP2 instruction is stated through include/gtemac.h (`gte_ldDQA`), the + * re-derived macro form the project convention requires; the epilogue is + * ordinary compiler output. + * + * LIMITS: the register number $27 is read straight from the original word + * 0x48C4D800 (rd field = 0b11011). The parameter name is a hypothesis; only the + * bytes are evidence. + */ + +#include "../include/gtemac.h" + +void func_80103B54(int dqa) +{ + gte_ldDQA(dqa); +} diff --git a/src/func_80103B6C.c b/src/func_80103B6C.c new file mode 100644 index 0000000..3c3ac66 --- /dev/null +++ b/src/func_80103B6C.c @@ -0,0 +1,51 @@ +/* func_80103B6C — 0x80103B6C..0x80103B8C (32 bytes). + * + * Original words: + * 0x00042100 sll a0,a0,4 + * 0x00052900 sll a1,a1,4 + * 0x00063100 sll a2,a2,4 + * 0x48C4A800 ctc2 a0,$21 + * 0x48C5B000 ctc2 a1,$22 + * 0x48C6B800 ctc2 a2,$23 + * 0x03E00008 jr ra + * 0x00000000 nop + * + * Ghidra renders the three consecutive ctc2 words as one `ldfcdir a0,a1,a2` + * pseudo-op (cookbook finding 25: the PSX loader collapses COP2 sequences, so a + * Ghidra instruction count is not a size). The raw words above are the + * authority and the extent is 32 bytes, not 20. + * + * The rd fields are 0b10101, 0b10110, 0b10111 = $21/$22/$23, the far-color + * registers RFC/GFC/BFC — NOT the $13/$14/$15 that cookbook finding 24 records + * for the light-direction triple (a different body). Each register is scaled by + * 4 before the write. + * + * The three `sll` are C expressions, not asm: the shift is arithmetic and the + * project convention extends to coprocessor instructions only. The `ctc2`s go + * through include/gtemac.h (`gte_ldRFC/gte_ldGFC/gte_ldBFC`). + * + * Each shift is bound to a named local BEFORE the asm statements. Writing the + * shift directly as the asm operand instead (`"r"(rfc << 4)`) emits the shifts + * and the `ctc2`s interleaved — sll,ctc2,sll,ctc2,sll,ctc2, 10 differing bytes + * with the instruction count still 32. Naming the locals first is what makes + * cc1 emit all three shifts ahead of the first `ctc2`, which is the original's + * order. This is a source-shape lever of the finding-28 family: it applies + * because the count was already right and only the order was wrong. + * + * LIMITS: the register numbers are read from the words above; the parameter + * names are hypotheses. That naming the locals reorders the emission is an + * observation about this compiler at -O2, not a rule. + */ + +#include "../include/gtemac.h" + +void func_80103B6C(int rfc, int gfc, int bfc) +{ + int r = rfc << 4; + int g = gfc << 4; + int b = bfc << 4; + + gte_ldRFC(r); + gte_ldGFC(g); + gte_ldBFC(b); +} diff --git a/src/func_80104C38.c b/src/func_80104C38.c new file mode 100644 index 0000000..24ce501 --- /dev/null +++ b/src/func_80104C38.c @@ -0,0 +1,55 @@ +/* func_80104C38 — 0x80104C38..0x80104C60 (40 bytes). + * + * Original words: + * 0x3C038012 lui v1,0x8012 + * 0x8C631E74 lw v1,-0x18c(v1) + * 0x00000000 nop + * 0x94620004 lhu v0,0x4(v1) <- loop top + * 0x00000000 nop + * 0x30420002 andi v0,v0,0x2 + * 0x1040FFFC beq v0,zero,loop + * 0x00000000 _nop (delay slot) + * 0x03E00008 jr ra + * 0x00000000 nop + * + * A busy-wait on a hardware status bit. The polled register is reached through + * a global pointer, read as an unsigned halfword at displacement 4, and the + * loop spins while bit 1 is clear. + * + * The `lui`/`lw` pair is the macro expansion of a symbol load into the same + * register (cookbook finding 2), so the global holds the pointer. Its address is + * **0x8011FE74**, not 0x80121E74: `lui v1,0x8012` with the displacement the + * disassembler prints as -396 is the sign-adjusted `%lo` split (finding 4), so + * the effective address is 0x80120000 - 0x18C. Reading the `%lo` as a positive + * halfword and naming the symbol 0x80121E74 is a mistake made on attempt 1; the + * two candidate symbols are 8 KB apart and the wrong one gave 2 differing bytes. + * + * The pointer load is emitted **once, before the loop**, which is the compiler + * correctly hoisting it: the pointer variable is not volatile, so its value + * cannot change and re-reading it would be redundant. Only the halfword it + * points at is re-read every iteration, which is what makes the pointee + * `volatile`. + * + * ATTEMPT 2 (recorded): binding the global to a named `volatile unsigned + * short *status` local leaves exactly **one** differing byte at 0x80104C50 — the + * loop back-edge displacement. The sequence and the load-delay `nop` are both in + * the right place, but the candidate branches to the `nop` at 0x80104C40 while + * the original branches past it to the `lhu` at 0x80104C44, so the local spelling + * puts the loop's top label one instruction early. Attempt 3 below drops the + * local and indexes the global directly, which changes where cc1 places that + * label. + * + * LIMITS: the symbol is named for its address and its type is inferred from the + * single use. Whether it is a pointer to one register or to the base of a + * structure whose word 1 is the status register is not observable. Nothing here + * proves the loop terminates, and the routine has no timeout — a fact about the + * original, not a defect in this reconstruction. + */ + +extern volatile unsigned short *D_8011FE74; + +void func_80104C38(void) +{ + while ((D_8011FE74[2] & 2) == 0) + ; +} diff --git a/src/func_8010543C.c b/src/func_8010543C.c new file mode 100644 index 0000000..4977e9a --- /dev/null +++ b/src/func_8010543C.c @@ -0,0 +1,50 @@ +/* + * func_8010543C — 56 bytes at 0x8010543C..0x80105474 + * + * Leaf routine that computes a byte size from three fields of an object: a + * half-rounded byte count, a rounded-up 5-byte-stride field, and a 16-bit + * length. + * + * The observed instructions are: + * lbu v0,0xe3(a0) 908200e3 v0 = *(unsigned char *)(a0 + 0xe3) + * lbu a1,0xe9(a0) 908500e9 a1 = *(unsigned char *)(a0 + 0xe9) + * lhu a0,0xec(a0) 948400ec a0 = *(unsigned short *)(a0 + 0xec) + * addiu v0,v0,0x1 24420001 v0 += 1 + * sra v0,v0,0x1 00021043 v0 >>= 1 arithmetic + * sll v0,v0,0x2 00021080 v0 *= 4 + * sll v1,a1,0x2 00051880 v1 = a1 * 4 + * addu v1,v1,a1 00651821 v1 = a1 * 4 + a1 (= a1 * 5) + * addiu v1,v1,0x3 24630003 v1 += 3 + * andi v1,v1,0xffc 30630ffc v1 &= 0xffc + * addiu v1,v1,0x4 24630004 v1 += 4 + * addu v0,v0,v1 00431021 v0 += v1 + * jr ra 03e00008 + * addu v0,v0,a0 00441021 v0 += a0 (delay slot) + * + * Three loads with no stalls in between: they target different registers and + * nothing reads them until the arithmetic, so no load-delay `nop` is needed. + * + * The `sra` is the tell for the first term: `(x + 1) >> 1` is emitted as an + * **arithmetic** shift, which this compiler uses for a signed division by two, so + * the value is held in a signed `int` even though it arrives through `lbu` (an + * unsigned byte load). The `andi ...,0xffc` and the `addiu ...,4` show the source + * masks and then offsets the field explicitly, and `a1 * 5` is strength-reduced + * to `sll`+`addu` because five is not a power of two. + * + * LIMITS: the function name and the meaning of the three fields are hypotheses; + * only the bytes are evidence. The field at +0xec is a 16-bit load, the other two + * are 8-bit. Whether the original wrote the mask as `& 0xffc` on an int or as a + * bitfield access is not recoverable; the mask is written literally here because + * that is what `andi` shows. + */ + +int func_8010543C(char *a0) +{ + int x = *(unsigned char *)(a0 + 0xe3); + int y = *(unsigned char *)(a0 + 0xe9); + int z = *(unsigned short *)(a0 + 0xec); + int a = ((x + 1) / 2) * 4; + int b = ((y * 5 + 3) & 0xffc) + 4; + + return a + b + z; +} diff --git a/src/func_80105B68.c b/src/func_80105B68.c new file mode 100644 index 0000000..f9a1bc4 --- /dev/null +++ b/src/func_80105B68.c @@ -0,0 +1,33 @@ +/* + * func_80105B68 — 32 bytes at 0x80105B68..0x80105B88 + * + * Initialises four fields of a structure: a length byte, a self-referential + * interior pointer, a tag byte taken from the caller, and a small status byte. + * The final constant is materialised once into `v0` and stored from there in the + * `jr ra` delay slot, so the function returns nothing — see cookbook finding 13 + * (a store-only function leaves its constant in `v0` as scratch, and writing + * `return 1` would cost an extra instruction). + * + * The observed instructions are: + * li v0,76 ; 0x4c + * sb v0,55(a0) ; p->byte_37 = 76 + * addiu v0,a0,36 ; v0 = p + 36 + * sw v0,44(a0) ; p->ptr_2c = p + 36 + * li v0,1 + * sb a1,36(a0) ; p->byte_24 = tag + * jr ra + * sb v0,54(a0) ; p->byte_36 = 1 (delay slot) + * + * LIMITS: the parameter types (`char *` base, `char` tag) and every field offset + * are hypotheses read from the instruction shape; the tag type is a hypothesis + * in particular, since an `sb` of a parameter needs no mask and an `int` would + * emit the same bytes. The meaning of the constants 76 and 1 is unknown and is + * not guessed here. Only the compiled bytes are evidence. + */ + +void func_80105B68(char *base, char tag) { + base[55] = 76; + *(int *)(base + 44) = (int)(base + 36); + base[36] = tag; + base[54] = 1; +} diff --git a/src/func_80105BA8.c b/src/func_80105BA8.c new file mode 100644 index 0000000..8dfe9db --- /dev/null +++ b/src/func_80105BA8.c @@ -0,0 +1,37 @@ +/* + * func_80105BA8 — 32 bytes at 0x80105BA8..0x80105BC8 + * + * Leaf routine that initialises a small object: a tag byte, a self-pointer, + * a caller-supplied byte, and a one-byte flag. + * + * The observed instructions are: + * li v0,0x47 24020047 v0 = 0x47 + * sb v0,0x37(a0) a0820037 *(char *)(a0 + 0x37) = 0x47 + * addiu v0,a0,0x24 24820024 v0 = a0 + 0x24 + * sw v0,0x2c(a0) ac82002c *(int *)(a0 + 0x2c) = a0 + 0x24 + * li v0,0x1 24020001 v0 = 1 + * sb a1,0x24(a0) a0850024 *(char *)(a0 + 0x24) = a1 + * jr ra 03e00008 + * sb v0,0x36(a0) a0820036 *(char *)(a0 + 0x36) = 1 (delay slot) + * + * The second argument is stored with `sb` and is **not** masked first, so it is + * not a `char` parameter: cookbook finding 7 says an unsigned `char` argument + * would carry an `andi ...,0xff` (as func_80017AD4 does). It is therefore taken + * as `int` and truncated by the store itself. + * + * The self-pointer at +0x2c is materialised as `addiu v0,a0,0x24` and stored — + * an address computed from the argument, not a symbol, so no registry row. + * + * LIMITS: the function name, the structure and every offset are hypotheses read + * off the disassembly. The widths are evidence: two 8-bit stores of constants, + * one 8-bit store of the argument, one 32-bit self-pointer store. `0x47` is a + * tag value with no established meaning. + */ + +void func_80105BA8(int a0, int a1) +{ + *(char *)(a0 + 0x37) = 0x47; + *(int *)(a0 + 0x2c) = a0 + 0x24; + *(char *)(a0 + 0x24) = a1; + *(char *)(a0 + 0x36) = 1; +} diff --git a/src/func_80107AA0.c b/src/func_80107AA0.c new file mode 100644 index 0000000..707d5d8 --- /dev/null +++ b/src/func_80107AA0.c @@ -0,0 +1,45 @@ +/* + * func_80107AA0 — 64 bytes at 0x80107AA0..0x80107AE0 + * + * Clears or sets the low bit of a 16-bit field on a global structure, depending + * on a flag argument. + * + * The observed instructions are: + * bnez a0,0x80107AC0 ; if (set != 0) take the OR arm + * nop + * lui v1,0x8012 ; symbol load expands into v1 itself + * lw v1,4168(v1) ; v1 = D_80121048 (a structure pointer) + * nop + * lhu v0,426(v1) ; v0 = ->half_1aa + * j 0x80107AD8 + * andi v0,v0,0xfffe ; v0 &= ~1 (delay slot) + * C0: lui v1,0x8012 ; RELOAD the pointer + * lw v1,4168(v1) + * nop + * lhu v0,426(v1) + * nop + * ori v0,v0,0x1 ; v0 |= 1 + * D8: jr ra + * sh v0,426(v1) ; ->half_1aa = v0 (delay slot) + * + * The single shared store at the end is a `cc1` CROSS-JUMP, not a shared + * statement: the source stores inside each arm (`&= ` in one, `|= ` in the + * other) and the jump optimiser merges the two identical `sh` into one. Writing + * the source the other way round — computing a `value` local in both arms and + * storing once — makes the store reload the global pointer a THIRD time and + * costs 4 bytes. So the merged store must be produced by the optimiser rather + * than written by hand. + * + * LIMITS: the symbol name D_80121048, the field offset (426) and the field's + * `unsigned short` type are hypotheses read from the instruction shape; the `lhu` + * is what shows the field is unsigned. Only the compiled bytes are evidence. + */ + +extern int D_80121048; + +void func_80107AA0(int set) { + if (set == 0) + *(unsigned short *)((char *)D_80121048 + 426) &= 0xFFFE; + else + *(unsigned short *)((char *)D_80121048 + 426) |= 1; +} diff --git a/src/func_80107AE0.c b/src/func_80107AE0.c new file mode 100644 index 0000000..3e93ef1 --- /dev/null +++ b/src/func_80107AE0.c @@ -0,0 +1,62 @@ +/* + * func_80107AE0 — 64 bytes at 0x80107AE0..0x80107B20 + * + * Leaf routine that sets or clears bit 2 of a 16-bit field in a global + * structure, returning the new value. + * + * The observed instructions are: + * bne a0,zero,0x80107b00 14800007 if (a0 != 0) goto the set path + * nop 00000000 (delay slot) + * lui v1,0x8012 3c038012 \ + * lw v1,0x1048(v1) 8c631048 / v1 = *(int *)0x80121048 (D_80121048) + * nop 00000000 load-delay slot + * lhu v0,0x1aa(v1) 946201aa v0 = *(unsigned short *)(v1 + 0x1aa) + * j 0x80107b18 08101ec6 goto the store + * andi v0,v0,0xfffb 3042fffb v0 &= ~4 (delay slot) + * 0x80107b00: + * lui v1,0x8012 3c038012 \ + * lw v1,0x1048(v1) 8c631048 / v1 = *(int *)0x80121048 + * nop 00000000 load-delay slot + * lhu v0,0x1aa(v1) 946201aa v0 = *(unsigned short *)(v1 + 0x1aa) + * nop 00000000 load-delay slot + * ori v0,v0,0x4 34420004 v0 |= 4 + * 0x80107b18: + * jr ra 03e00008 + * sh v0,0x1aa(v1) a46201aa *(unsigned short *)(v1 + 0x1aa) = v0 (delay slot) + * + * The base load is **duplicated in both arms** and the store after the merge uses + * the arm's own v1, so the source reads the global inside each branch rather than + * once before the condition. The address is the `lui`+`lw`-through-the-same- + * register symbol form (cookbook finding 2), so it is written as the named global + * `D_80121048`; 0x80121048 is outside the small-data window and is accessed + * absolutely, so no `gp` marker is needed. + * + * The clear is an `andi` with `~4` and the set is an `ori` with `4`, so the source + * writes `& ~4` and `| 4` on a 16-bit field; `~4` is spelled as 0xfffb by the + * assembler's 16-bit immediate. + * + * LIMITS: the function name, the claim that the global is a pointer to a structure + * and the meaning of bit 2 are hypotheses; only the bytes are evidence. The field + * is 16-bit (`lhu`/`sh`). The duplicated base load is an observation about the + * original's source shape, not a requirement of the C. + */ + +extern int D_80121048; + +int func_80107AE0(int a0) +{ + unsigned short v0; + int v1; + + if (a0 == 0) { + v1 = D_80121048; + v0 = *(unsigned short *)(v1 + 0x1aa) & ~4; + } else { + v1 = D_80121048; + v0 = *(unsigned short *)(v1 + 0x1aa) | 4; + } + + *(unsigned short *)(v1 + 0x1aa) = v0; + + return v0; +} diff --git a/src/func_80107FAC.c b/src/func_80107FAC.c new file mode 100644 index 0000000..75b4b3b --- /dev/null +++ b/src/func_80107FAC.c @@ -0,0 +1,50 @@ +/* + * func_80107FAC — 48 bytes at 0x80107FAC..0x80107FDC + * + * ORs a table word selected by a masked index into a global structure's flags + * word, and returns whether the index is below three. The index is truncated to + * 16 bits before use, and the table lookup is an array index (so the scaling is a + * `sll` by 2 rather than a separate multiply). + * + * The observed instructions are: + * andi v0,a0,0xffff ; i = index & 0xffff + * sll a0,v0,0x2 ; i * 4, into the now-dead argument register + * lui a1,0x8012 ; symbol load expands into a1 itself + * lw a1,3312(a1) ; a1 = D_80120CF0 (a structure pointer) + * lui at,0x8012 ; %hi of the table base + * addu at,at,a0 ; table + i*4 (base first) + * lw a0,3320(at) ; a0 = table[i] (%lo as displacement) + * lw v1,4(a1) ; a1->word_04 + * slti v0,v0,3 ; return i < 3 + * or v1,v1,a0 ; a1->word_04 | table[i] + * jr ra + * sw v1,4(a1) ; a1->word_04 = result (delay slot) + * + * Two source-shape requirements are load-bearing here. The global pointer must be + * materialised BEFORE the table element is read (the original loads a1 first), + * and the table element must be consumed inside the OR expression rather than + * bound to a named local — a named local makes `cc1` hold the scaled index in a + * different register and costs 2 bytes in the `sll` and `addu` register fields. + * + * Note the two different base forms in one function: the table base at + * 0x80120CF8 is a literal written with `%hi` + `addu` + `%lo`-as-displacement, + * while D_80120CF0 is loaded as a pointer value through the symbol macro. The + * `addu` operand order here is base-first, the same order the + * address-shaped-symbol spelling produced at 0x80016174. + * + * LIMITS: the symbol name D_80120CF0, the table base 0x80120CF8, the element + * stride (4), the field offset (4) and the comparison constant (3) are hypotheses + * read from the instruction shape; what the structures mean is unknown and is not + * guessed here. Only the compiled bytes are evidence. + */ + +extern int D_80120CF0; +extern int D_80120CF8[]; + +int func_80107FAC(int index) { + int i = index & 0xFFFF; + int *p = (int *)D_80120CF0; + + p[1] = p[1] | D_80120CF8[i]; + return i < 3; +} diff --git a/src/func_80107FDC.c b/src/func_80107FDC.c new file mode 100644 index 0000000..c4a3aac --- /dev/null +++ b/src/func_80107FDC.c @@ -0,0 +1,58 @@ +/* + * func_80107FDC — 52 bytes at 0x80107FDC..0x80108010 + * + * Leaf routine that clears one 16-byte record when its index is in range, and + * reports whether it did so. + * + * The observed instructions are: + * andi v1,a0,0xffff 3084ffff v1 = a0 & 0xffff + * slti v0,v1,0x3 28620003 v0 = (v1 < 3) signed + * beq v0,zero,0x80108004 10400007 if (!(v1 < 3)) goto the zero return + * li v0,0x1 24020001 v0 = 1 (delay slot) + * lui a0,0x8012 3c048012 \ + * lw a0,0xcf4(a0) 8c840cf4 / a0 = *(int *)0x80120CF4 (D_80120CF4) + * sll v1,v1,0x4 00042100 v1 *= 16 + * addu v1,v1,a0 00641821 v1 = (index * 16) + base + * j 0x80108008 08100002 goto epilogue + * sh zero,0x0(v1) a4600000 *(short *)v1 = 0 (delay slot) + * 0x80108004: + * addu v0,zero,zero 00001021 v0 = 0 + * 0x80108008: + * jr ra 03e00008 + * nop 00000000 (delay slot) + * + * The `andi ...,0xffff` on entry is a mask the source applies explicitly to a + * 32-bit parameter: the bound test is `slti` — a **signed** compare — and an + * `unsigned short` parameter would instead produce `sltiu` (measured: that + * spelling differs by exactly the one byte of the `slti`/`sltiu` opcode). So the + * parameter is `int` and the mask is written in the C. + * + * The record base is a symbol load through the same register (finding 2), so it + * is written as `D_80120CF4`, not as a literal address. The address arithmetic is + * **stride first, base second** (`sll` then `addu v1,v1,a0`), the spelling + * finding 22 identifies as `(index * 16) + symbol` rather than + * `symbol + index * 16`; the rule's stated scope (a symbol or `gp` base) applies + * here. + * + * The `1` result is materialised in the branch delay slot, so the source order is + * the early guard `if (index >= 3) return 0;` followed by the work and + * `return 1;`. + * + * LIMITS: the function name, the record size of 16 bytes and the meaning of the + * table are hypotheses; only the bytes are evidence. The cleared field is 16-bit + * (`sh`), so the record is assumed to start with a halfword. + */ + +extern int D_80120CF4; + +int func_80107FDC(int a0) +{ + int i = a0 & 0xffff; + + if (i >= 3) + return 0; + + *(short *)((i * 16) + D_80120CF4) = 0; + + return 1; +} diff --git a/src/func_80108024.c b/src/func_80108024.c new file mode 100644 index 0000000..18305e3 --- /dev/null +++ b/src/func_80108024.c @@ -0,0 +1,34 @@ +/* + * func_80108024 — 16 bytes at 0x80108024..0x80108034 + * + * Saves the current `gp` register value into a global. This is one member of a + * small gp save/restore family in the CRT region: 0x80108024 saves gp to + * D_80120D10, 0x80108034 saves gp to D_80120D14 and then restores gp from + * D_80120D10, and 0x8010804C restores gp from D_80120D14. + * + * The observed instructions are: + * lui at,0x8012 ; symbol store expands through $at + * sw gp,0x0d10(at) ; D_80120D10 = gp + * jr ra + * nop ; delay slot left empty by cc1, filled by maspsx + * + * The stored value is the `gp` register itself, which plain C cannot name. The + * source therefore uses a GNU C global register variable — a documented + * register-name binding, not inline assembly (see docs/MATCHING_CONVENTIONS.md + * and cookbook finding 24, where `register int sp __asm__("$29")` reproduces the + * stack accessor with no exemption needed). + * + * LIMITS: the symbol name D_80120D10 and the `int` type are hypotheses read from + * the instruction shape; only the compiled bytes are evidence. The offset 0x0d10 + * is a fact about this executable's globals, not about the compiler. The + * surrounding family members (0x80108034, 0x8010804C) are excluded from the + * worklist by name and are not claimed here. + */ + +register int gp __asm__("$28"); + +extern int D_80120D10; + +void func_80108024(void) { + D_80120D10 = gp; +} diff --git a/src/func_80108690.c b/src/func_80108690.c new file mode 100644 index 0000000..f069b0a --- /dev/null +++ b/src/func_80108690.c @@ -0,0 +1,41 @@ +/* + * func_80108690 — 28 bytes at 0x80108690..0x801086AC + * + * Leaf routine that masks two arguments to 15 bits and stores them as two + * adjacent halfwords in a global structure reached through a fixed pointer. + * + * The observed instructions are: + * andi a0,a0,0x7fff 30847fff a0 &= 0x7fff + * lui v0,0x8012 3c028012 \ + * lw v0,0x1048(v0) 8c421048 / v0 = *(int *)0x80121048 (D_80121048) + * andi a1,a1,0x7fff 30a57fff a1 &= 0x7fff + * sh a0,0x180(v0) a4440180 *(short *)(v0 + 0x180) = a0 + * jr ra 03e00008 + * sh a1,0x182(v0) a4450182 *(short *)(v0 + 0x182) = a1 (delay slot) + * + * The address is materialised as `lui`+`lw` with the *same* register twice, + * which is this toolchain's symbol-load macro form (cookbook finding 2) — so + * the base is written as the named global `D_80121048`, not as a literal + * address (finding 5: a literal would give `lui`+`ori`). 0x80121048 is + * 0x8F0 below gp (0x80121938), i.e. outside the small-data window, and the + * original accesses it absolutely, so no `gp` marker and no registry row is + * needed; the address-named symbol resolves implicitly. + * + * The first `andi` is hoisted above the load and the second is placed after + * it; that ordering is left to cc1's scheduler rather than forced. + * + * LIMITS: the function name, the global's type and the meaning of the two + * halfwords at +0x180/+0x182 are hypotheses; only the bytes are evidence. The + * masks are written as `& 0x7fff` on `int` arguments, which is what makes cc1 + * emit `andi` rather than a truncating `sh` of the full register. + */ + +extern int D_80121048; + +void func_80108690(int a0, int a1) +{ + int v0 = D_80121048; + + *(short *)(v0 + 0x180) = a0 & 0x7fff; + *(short *)(v0 + 0x182) = a1 & 0x7fff; +} diff --git a/src/func_8010A888.c b/src/func_8010A888.c new file mode 100644 index 0000000..373f11d --- /dev/null +++ b/src/func_8010A888.c @@ -0,0 +1,42 @@ +/* + * func_8010A888 — 40 bytes at 0x8010A888..0x8010A8B0 + * + * Leaf routine that sets one bit in a 32-bit control word reached through a + * fixed global pointer. + * + * The observed instructions are: + * lui a0,0x8012 3c048012 \ + * lw a0,0x105c(a0) 8c84105c / a0 = *(int *)0x8012105C (D_8012105C) + * lui v1,0xf0ff 3c03f0ff \ + * lw v0,0x0(a0) 8c820000 / v0 = *a0 + * ori v1,v1,0xffff 3463ffff v1 = 0xf0ffffff + * and v0,v0,v1 00431024 v0 &= 0xf0ffffff + * lui v1,0x2000 3c032000 v1 = 0x20000000 + * or v0,v0,v1 00431025 v0 |= 0x20000000 + * jr ra 03e00008 + * sw v0,0x0(a0) ac820000 *a0 = v0 (delay slot) + * + * The address is materialised as `lui`+`lw` through the same register, this + * toolchain's symbol-load macro form (cookbook finding 2), so the base is the + * named global `D_8012105C` and not a literal address (finding 5). 0x8012105C + * is outside the small-data window and the original accesses it absolutely, so + * no `gp` marker and no registry row are needed; the address-named symbol + * resolves implicitly. The mask 0xf0ffffff is built as `lui 0xf0ff` + `ori + * 0xffff` and the set bit as `lui 0x2000` alone (zero low half), which is what a + * plain `&`/`|` on the loaded word produces. + * + * LIMITS: the function name and the meaning of the bit (a hardware control + * word, since 0x8012105C is read as a pointer) are hypotheses; only the bytes + * are evidence. The C is written with the mask and the OR value as literals + * because the original materialises them as literals; whether the original + * source spelled them as named constants is unknowable from the code. + */ + +extern int D_8012105C; + +void func_8010A888(void) +{ + int *p = (int *)D_8012105C; + + *p = (*p & 0xf0ffffff) | 0x20000000; +}