feat(decomp): parallel gate — 31 fns across 23 binaries (10 workers)

md_MAIN_044    func_800CD2EC
  ov_SC02_026    func_8017D508
  md_SC07_003    func_801A079C
  ov_SC01_084    func_80180000
  ov_SC02_005    func_8017F898
  ov_SC01_080    func_80181410 func_80181D98
  ov_SC03_007    func_801850B4
  ov_SC03_117    func_8017FED8
  ov_SC03_105    func_8017E180 func_8018423C
  ov_SC04_012    func_8017D4CC
  ov_SC05_005    func_8017EAEC
  ov_SC05_008    func_8017E6A4 func_8017F288
  ov_SC04_011    func_80186020
  ov_SC05_011    func_8017D610
  ov_SC05_018    func_801810B0
  ov_SC06_006    func_8017EEA4
  ov_SC06_008    func_8017E37C
  ov_SC07_001    func_8018088C
  ov_SC07_000    func_8017E658 func_8018006C func_801805C0
  ov_SC07_006    func_801826E0
  ov_SC06_032    func_8017DABC func_80183940 func_8018FB5C
  ov_SC07_007    func_801827E0
  ov_SC07_010    func_8017E5B8 func_8017F860
This commit is contained in:
Drew T
2026-09-01 23:11:26 -06:00
parent 69d6c76a3b
commit 027f4f2d2b
27 changed files with 2226 additions and 36 deletions
+2 -1
View File
@@ -809,6 +809,7 @@ build/src/ov_SC01_084/ov_SC01_084_jr_8015AE2C.o: JTBL_PADS := 0,4 # §8e pads (
build/src/ov_SC01_084/ov_SC01_084_jr_8016AB6C.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20
build/src/ov_SC01_084/ov_SC01_084_jr_80178D40.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x178
build/src/ov_SC01_084/ov_SC01_084_jr_8017CA80.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20
build/src/ov_SC01_084/ov_SC01_084_jr_8017F690.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x14
build/src/ov_SC01_084/ov_SC01_084_o0c.o: JTBL_PADS := 0,4,4,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x38,+0xa8,+0x118
ov_SC01_084_CHECK_SHA := config/check.ov_SC01_084.sha
ov_SC01_084_SYMBOLS := config/symbols.ov_SC01_084.txt
@@ -3935,7 +3936,7 @@ build/src/ov_SC05_011/ov_SC05_011_jr_80159C84.o: JTBL_PADS := 0,4 # §8e pads (
build/src/ov_SC05_011/ov_SC05_011_jr_8015AE2C.o: JTBL_PADS := 0,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20
build/src/ov_SC05_011/ov_SC05_011_jr_8016AB6C.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20
build/src/ov_SC05_011/ov_SC05_011_jr_80178D40.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x178
build/src/ov_SC05_011/ov_SC05_011_jr_8017BEBC.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20
build/src/ov_SC05_011/ov_SC05_011_jr_8017BEBC.o: JTBL_PADS := 0,0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20,+0x40
build/src/ov_SC05_011/ov_SC05_011_o0c.o: JTBL_PADS := 0,4,4,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x38,+0xa8,+0x118
ov_SC05_011_CHECK_SHA := config/check.ov_SC05_011.sha
ov_SC05_011_SYMBOLS := config/symbols.ov_SC05_011.txt
+1 -1
View File
@@ -166,7 +166,7 @@ segments:
- [0x9df28, .rodata, ov_SC01_084_jr_8017CA80] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
- [0x9df60, data, tail18]
- [0x9df70, .rodata, ov_SC01_084_jr_8017F690] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
- [0x9df84, data, tail19]
- [0x9df98, data, tail19]
- [0x9FA9C, bin, trailing] # final 3 bytes (EOF not 4-aligned; spimdisasm drops a partial word)
- [0x9FA9F] # EOF marker = the 0.4.dec byte length
# @TRAILING@ (above) is replaced by tools/new_overlay.sh: for a non-4-aligned overlay it becomes
+1 -1
View File
@@ -163,7 +163,7 @@ segments:
- [0x73b58, data, tail17]
- [0x73b5c, .rodata, ov_SC05_011_jr_8017AE2C] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
- [0x73b70, .rodata, ov_SC05_011_jr_8017BEBC] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
- [0x73bb0, data, tail18]
- [0x73bd0, data, tail18]
- [0x75084, bin, trailing] # final 3 bytes (EOF not 4-aligned; spimdisasm drops a partial word)
- [0x75087] # EOF marker = the 0.4.dec byte length
# @TRAILING@ (above) is replaced by tools/new_overlay.sh: for a non-4-aligned overlay it becomes
+66 -1
View File
@@ -302,7 +302,72 @@ void func_800CD1D4(s32 *s1) {
}
INCLUDE_ASM("asm/md_MAIN_044/nonmatchings/md_MAIN_044", func_800CD2EC);
extern u16 D_800B99DA;
extern s16 currentLocationId;
extern s32 func_80146E98(s32 a0);
extern void func_800CD57C(void *arg0);
extern s32 func_80146A6C(s32 a0, void *a1, s32 a2, s32 a3, s32 a4, s32 a5, s32 a6);
extern void func_800CD848(s32 param_1);
extern void func_800CD780(s32 param_1);
extern void func_800CD7A8(s32 a0, s32 a1);
extern void func_80163194(s32 a0, s32 a1, s32 a2, s32 a3, s32 arg4);
extern void func_80162FC0(s32 *a0);
extern void func_800CD758(s32 param_1);
extern void func_800CD63C(s32 a0, s32 a1);
extern s32 func_800CD894(void *a0);
extern void func_80146DE8(s32 *a0, s32 a1, s32 a2, s32 a3);
extern void func_80146E90(s32 *a0, s32 a1);
extern void func_80146C98(s32 *a0, s16 a1);
extern void func_80147324(s32 a0);
extern s32 func_80163408(s32 a0, s32 a1, s32 a2, s32 a3);
extern void func_80163328();
extern s32 func_801632F0();
extern s32 func_801632E0();
extern s32 func_80146CA0(void *a0);
void func_800CD2EC(s32 *s1) {
u8 buf[0x40];
u8 *p;
s32 s2;
s32 v0;
s2 = *(s32 *)(s1 + 8);
if (func_80146E98((s32)s1) == 0) {
if ((*(u16 *)&D_800B99DA & 3) == 0) {
func_800CD57C(s1);
}
func_80146A6C(0x26, s1, *(s16 *)((s32)s1 + 6), *(s16 *)((s32)s1 + 0xA),
*(s16 *)((s32)s1 + 0xE), *(s16 *)(s2 + 0x12), 0);
func_800CD848((s32)s1);
func_800CD780((s32)s1);
p = buf + 0x20;
func_800CD7A8((s32)s1, (s32)p);
func_80163194((s32)s1, 0, -0x1000, 0x14000, (s32)p);
func_80162FC0(s1);
func_800CD758((s32)s1);
func_800CD63C((s32)s1, (s32)p);
if (currentLocationId == 0x3067) {
if (func_800CD894(s1) != 0) {
func_80146DE8(s1, 0, 0xFFF40000, 0x20000);
func_80146E90(s1, 0x10);
*(s32 *)(s2 + 4) |= 0x50000000;
func_80146C98(s1, 3);
func_80147324(0x990);
return;
}
}
func_80163408((s32)s1, 0x23, 0x80, 8);
func_80163328(s1);
v0 = func_801632F0(s1);
if ((v0 & 1) != 0) {
func_801632E0(s1);
} else if ((v0 & 6) == 0) {
return;
}
}
func_80146CA0(s1);
}
extern s16 D_800CE210[];
extern u16 D_800CE212[];
+144 -1
View File
@@ -334,7 +334,150 @@ void func_801A072C(s32 arg0) {
}
INCLUDE_ASM("asm/md_SC07_003/nonmatchings/md_SC07_003", func_801A079C);
#include "common.h"
/*
* func_801A079C (md_SC07_003, 0x801A079C, 105 ins) == MATCH, byte-exact.
*
* SC07 actor tick: nudge the sprite's 0x12 field, kick the two 0x94-state
* cutscene hooks (0xF / 0x35), run the per-frame update, then pick the next
* state (offset 0x2) from the 0xE8 counter and the func_8012BD14 distance.
*
* LEVERS (each verified by flipping it back and re-scoring with match_one):
*
* 1. THE `0xF` STORE FOR `s1 <= 0x10000` IS AN **EARLY BLOCK**, NOT A TRAILING
* `else`. Written as `if (s1 > 0x10000) { ...body... } else { store 0xF; }`
* the else-block lands LAST, so it is the block that falls into the
* epilogue. gcc's cross_jump then merges every other `sh $v0,0x2($s0)`
* into it (cookbook §5a/§193-C: the SURVIVING copy is the later one, and
* `find_cross_jump` pairs a jump's block with `prev_real_insn(JUMP_LABEL)`
* — i.e. whatever falls through into the epilogue). Result: 100 ins, four
* stores lost, LENGTH-DRIFT −5 at closeness 56.
* With the early-return form the block that falls into the epilogue is
* `sw $v0,0x1C($s0)` instead, which matches NO `sh` block, so all five
* `sh $v0,0x2($s0)` survive — exactly the target. This one edit took the
* draft from 100/56 to 105/8. No §34 asm barrier is needed: choosing the
* fall-through block IS the barrier here.
*
* 2. BRANCH-SENSE / ARM ORDER is read off the target's `slt`+`beqz`/`bnez`
* pairs (§3-T4): `if (s1 <= 0x24000) {BDBC arm} else {0xC4000 arm}` and
* `if (s1 <= 0x64000) {store 0xF} else {store 0x80}`. Writing either the
* other way round emits the complementary branch and swaps the two blocks.
*
* 3. `unused[2]` IS LOAD-BEARING — DO NOT DELETE. The target's frame is 0x28
* with $s0/$s1/$ra at 0x18/0x1C/0x20; without an 8-byte aggregate local the
* frame is 0x20 (regs at 0x10/0x14/0x18) and eight instructions carry the
* wrong immediates. gcc-2.7.2 gives an aggregate a stack slot at expand
* time and never reclaims it, so an unreferenced 8-byte local costs zero
* instructions and buys the frame. (s16[4] and a 2xs32 struct match too —
* only the SIZE matters.)
*
* 4. `*(u16 *)(*(s32 *)(arg0 + 0x20) + 0x12) += func_8012BA10(arg0, 0x20);`
* is the TU's own house form (func_801A61C4) — the call is emitted first,
* then the pointer is reloaded, then `sh` rides the next jal's delay slot.
*
* 5. func_8012BD14 and func_8012CBA4 are declared `void` (the TU's canonical
* spelling, reconciled in S54) and their return values are read through the
* TU's `((s32 (*)(s32))f)(x)` fn-ptr cast — byte-neutral, see the note above
* func_801A1E30.
*
* SYMBOL AUDIT (law 1c, done after MATCH — match_one masks jal/HI16/LO16).
* Every name re-checked against the relocation lines of
* asm/md_SC07_003/nonmatchings/md_SC07_003/func_801A079C.s; the .s's symbol set
* and this draft's are identical (12 calls + 3 data):
* func_8012BA10(arg0,0x20) / func_8012B178(arg0,0xFFFB0000) /
* func_801A2658(arg0,&D_801A68BC) [state 0xF] and (arg0,&D_801A68C4) [0x35] /
* func_8013C9C4(D_80186F68) / func_8002D4C8(0xB53,0) / func_8012CBA4(arg0) /
* func_8012ADE4(arg0) / func_801A24E8(arg0) / rand /
* func_8012BD14(arg0) / func_8012BDBC(arg0,0x180) / func_8012BEE8(arg0).
* D_801A68C4 is its OWN relocation in this .s (it is &D_801A68BC[8], which
* func_801A23FC spells as `D_801A68BC + 8`) — law 1 says spell it as the .s
* does, so it gets its own extern here.
*
* BANK NOTE (law 2): src/md_SC07_003/md_SC07_003.c already declares eleven of
* these; every spelling below is copied verbatim from that file (rand:93,
* func_8012BEE8:282, func_8012BD14:410, func_8012BA10:412, func_8002D4C8:366,
* func_8012B178:1001, func_8012CBA4:1029, func_8012ADE4:1028, func_801A24E8:1032,
* func_8013C9C4:1340, func_801A2658:1341, D_801A68BC:1345, func_8012BDBC:4025).
* D_80186F68 and D_801A68C4 are absent from the TU, so they are free-standing;
* D_80186F68 follows the TU's own precedent for a func_8013C9C4 argument
* (`extern u16 D_80186F44[]` at line 1344) rather than the fleet's
* function-pointer-array spelling, which lives in other TUs only.
*/
extern s32 rand(void);
extern s32 func_8012BA10(s32 a0, s32 a1);
extern void func_8012B178(s32 a0, s32 a1);
extern void func_801A2658(s32 a0, s32 a1);
extern void func_8013C9C4(void *a0);
extern void func_8002D4C8(s32 a0, s32 a1);
extern void func_8012CBA4(s32 a0); /* canonical void; return read via fn-ptr cast */
extern void func_8012ADE4(u8 *a0);
extern void func_801A24E8(s32 a0);
extern void func_8012BD14(s32 a0); /* canonical void; return read via fn-ptr cast */
extern s32 func_8012BDBC(s32 a0, s32 a1);
extern s32 func_8012BEE8(s32 a0);
extern u8 D_80186F68[];
extern u8 D_801A68BC[];
extern u8 D_801A68C4[];
void func_801A079C(s32 arg0) {
s32 s1;
s32 unused[2]; /* LOAD-BEARING: buys the target's 0x28 frame (lever 3) */
s32 v1;
*(u16 *)(*(s32 *)(arg0 + 0x20) + 0x12) += func_8012BA10(arg0, 0x20);
func_8012B178(arg0, 0xFFFB0000);
v1 = *(s32 *)(arg0 + 0x94);
if (v1 == 0xF) {
func_801A2658(arg0, (s32)D_801A68BC);
func_8013C9C4(D_80186F68);
func_8002D4C8(0xB53, 0);
} else if (v1 == 0x35) {
func_801A2658(arg0, (s32)D_801A68C4);
func_8013C9C4(D_80186F68);
func_8002D4C8(0xB53, 0);
}
if ((((s32 (*)(s32))func_8012CBA4)(arg0) & 0x2000) == 0) {
func_8012ADE4((u8 *)arg0);
}
func_801A24E8(arg0);
if (*(s32 *)(arg0 + 0xE8) > 0xFFFF) {
if (rand() & 1) {
*(s16 *)(arg0 + 2) = 0xD;
} else {
*(s16 *)(arg0 + 2) = 0xB;
}
return;
}
s1 = ((s32 (*)(s32))func_8012BD14)(arg0);
if (s1 <= 0x10000) {
*(s16 *)(arg0 + 2) = 0xF;
return;
}
if (s1 <= 0x24000) {
if (func_8012BDBC(arg0, 0x180) != 0) {
*(s16 *)(arg0 + 2) = 0x11;
return;
}
} else if (s1 > 0xC4000 && *(s32 *)(arg0 + 0x94) == 0x4E) {
*(s16 *)(arg0 + 2) = 9;
return;
}
if (func_8012BEE8(arg0) != 0) {
if (s1 <= 0x64000) {
*(s16 *)(arg0 + 2) = 0xF;
} else {
*(s32 *)(arg0 + 0x1C) = 0x80;
}
}
}
extern void func_801A28AC(s32 a0);
extern void func_8012A828(s32 a0, void *a1);
+144 -2
View File
@@ -6522,7 +6522,70 @@ void func_801813D4(void *a0) {
}
INCLUDE_ASM("asm/ov_SC01_080/nonmatchings/ov_SC01_080_jr_8017AE2C", func_80181410);
#include "common.h"
struct vec;
extern s32 rand(void);
extern void RotMatrixYXZ(void *a0, void *a1);
extern void func_800D20C0(void *a0, void *a1, s32 a2);
extern void func_800D23D0(void *a0);
extern void func_801292C8(u8 *a0);
extern void func_8012931C(struct vec *a0);
extern void func_801696D8(s32 a0, s32 a1);
extern s32 func_80135168(u16 a0, u16 *a1, u16 *a2);
void func_80181410(s32 param_1)
{
u16 sp10[3];
u16 sp18[3];
u16 sp20[4];
u16 sp28[4];
u16 sp30[16];
u16 *p1;
u16 *p2;
u16 c;
u16 t;
sp20[0] = *(u16 *)(param_1 + 6);
sp20[1] = *(u16 *)(param_1 + 0xA);
sp20[2] = *(u16 *)(param_1 + 0xE);
p1 = sp28;
func_800D20C0(sp20, p1, 1);
p1 = 0;
p2 = sp28;
func_800D23D0(p2);
p2 = 0;
sp28[2] = sp28[1] * 2;
RotMatrixYXZ(sp28, sp30);
if (*(s32 *)(param_1 + 0x1C) & 1) {
*(s32 *)(param_1 + 0x2C) = (rand() & 0x7FF) + 0x600;
} else {
*(s32 *)(param_1 + 0x2C) = 0x500;
}
func_801696D8(param_1, (s32)sp30);
if (--*(s32 *)(param_1 + 0x1C) != -1) {
sp10[0] = *(u16 *)(param_1 + 6);
sp10[1] = *(u16 *)(param_1 + 0xA);
sp10[2] = *(u16 *)(param_1 + 0xE);
func_8012931C((struct vec *)param_1);
sp18[0] = *(u16 *)(param_1 + 6);
sp18[1] = *(u16 *)(param_1 + 0xA);
sp18[2] = *(u16 *)(param_1 + 0xE);
if (func_80135168(1, sp10, sp18) != 0) {
t = *(u16 *)(param_1 + 2);
*(u16 *)(param_1 + 6) = sp18[0];
*(u16 *)(param_1 + 0xA) = sp18[1];
c = sp18[2];
*(s32 *)(param_1 + 0x1C) = 0xC;
*(u16 *)(param_1 + 2) = t + 1;
*(u16 *)(param_1 + 0xE) = c;
}
} else {
func_801292C8((u8 *)param_1);
}
}
extern void (*D_8018A258[])(void);
@@ -6899,7 +6962,86 @@ void func_80181CA4(void)
}
INCLUDE_ASM("asm/ov_SC01_080/nonmatchings/ov_SC01_080_jr_8017AE2C", func_80181D98);
#include "common.h"
/* Table at D_8018A278: {u16 dist; s16 kind;} pairs, -1-terminated (see
* asm/ov_SC01_080/data/tail.data.s). `dist` is read three ways -- lh for the
* sentinel (combine folds (s16)u16 -> lh), lhu+sll/sra for the slt against
* D_801270D0, and lhu again for the +0x500 store (the call kills the CSE). */
typedef struct {
u16 dist;
s16 kind;
} SpotDef_8018A278;
extern s32 D_80126D50;
extern u16 D_80126B66;
extern s32 D_80126B9C;
extern s32 D_801270CC;
extern s32 D_801270D0;
extern s32 D_80127188;
extern s32 D_801C7548;
extern s32 D_801C754C;
extern SpotDef_8018A278 D_8018A278[];
extern void func_80180174(void);
extern void func_80181CA4(void);
extern s32 func_8012C588(s32 a0, s32 a1);
extern s32 func_8012C658(s32 a0, s32 a1, s32 a2);
/* $s0/$s1 pinned (§17): the natural priority order hands `p` $s0 and `t` $s1,
* the target has them the other way round. `s` holds D_80126D50 so the
* unconditional `*p = 1` lands in the bnez delay slot. */
void func_80181D98(void *a0) {
register s32 *p __asm__("$17");
register SpotDef_8018A278 *t __asm__("$16");
s32 e;
s32 s;
u16 v;
p = &D_801270CC;
switch (*p) {
case 0:
*p = 1;
D_801C7548 = 0;
D_801C754C = 1;
break;
case 1:
s = D_80126D50;
*p = 1;
if (s == 0) {
if (D_801C7548 == 0x34) {
func_80180174();
}
D_801C7548++;
if (D_801C7548 >= 0x3C && (s16)D_80126B66 >= -0x500 &&
(D_80126B9C & 0x8000000) != 0) {
D_801C7548 = 0;
func_8012C588(0x36, 0);
}
}
break;
case 2:
func_80181CA4();
p += 15;
t = D_8018A278;
if (D_80127188 == 4) {
while ((s16)t->dist != -1) {
if (*p == 0 && D_801270D0 >= (s16)t->dist) {
e = func_8012C658(0x2C, t->kind, 0);
if (e != 0) {
v = t->dist;
*(s32 *)(e + 0xCC) = (s32)p;
*(s16 *)(e + 0xFC) = v + 0x500;
}
*p = 1;
}
t++;
p++;
}
}
break;
}
}
void func_80181F5C(void *arg0)
{
+92 -1
View File
@@ -3157,7 +3157,98 @@ void func_8017FF68(s32 param_1)
}
INCLUDE_ASM("asm/ov_SC01_084/nonmatchings/ov_SC01_084_jr_8017F690", func_80180000);
#include "common.h"
// @class: dispatch-topology
// @stuck: none — MATCH (111/111 ins, match_one confirmed), 2nd compile.
//
// Levers that carry the byte match:
// (1) §2 of func_8018270C / §250 — the two `func_80028620` record bases are bound by
// ASM-INITIALISATION, `__asm__("la %0, SYM" : "=r"(p))`. This is what makes the
// base-relative accesses fold into `off($reg)` while the neighbouring members stay
// DIRECT `lui $at; sw %lo(sym)` globals (§216: a separate `lui` per adjacent byte
// means DISTINCT scalar objects, which is why D_800A5E94/95/96/EA4/EA5 are spelled as
// independent `extern u8`s and not as one struct).
// - Block A anchors at D_800A5E88 (offset 0 store + `base+0x10` for the 2nd record) -> $s0.
// - Block B anchors at D_800A5E90 and reaches the record base as `rec - 8`, which is
// exactly the target's `addiu $a1, $v1, -0x8`. A plain `&D_800A5E88` there const-folds
// to its own lui/addiu pair and loses the -8.
// Two SEPARATE locals (one per arm) — sharing one would drag block B's base into $s0.
// (2) The guard is the short-circuit decrement `x != 0 && --x == 0`: `beqz` on the loaded
// value with the `addiu -1` in its delay slot, then `bnez` with the `sw` in ITS delay slot.
// (3) D_80126B62 is the TU's `u16` house spelling (law 2) read as `(s32)(s16)` so combine
// folds sign_extend(zero_extend(mem:HI)) back into a single `lh`. The three-way ladder is
// written with `>=` (not `<`) so do_jump emits `slti`+`bnez`-to-the-else, and cross-jumping
// tail-merges the 0x1E and -0x1E stores onto the shared `sw $v0, %lo(D_800A5E8C)($at)`.
// (4) §164-54 / §193-G DISPATCH-TOPOLOGY ORACLE — the construct, not the body density, picks
// the compare shape. With only `case 2/4/6` gcc emitted a 3-way compare chain (closeness
// 22, +1 ins). Spelling the EMPTY `case 3: case 5: break;` makes 5 labels over the range
// 2..6 and gcc builds jtbl_801C60DC, whose 3rd and 5th words are the default target.
// (5) §20-style address materialisation for D_801270D8: `((struct { s32 w; } *)&D_801270D8)->w`
// keeps the symbol address in $a0 and reloads through it (the compare load, then the
// switch's own reload) — copied verbatim from func_8017F690 in this same TU.
extern s32 rand(void);
extern void func_80028620(s32, void *);
extern void func_8012C218(void *a0);
extern u16 D_80126B62;
extern s32 D_801270D8;
extern s32 D_800A5E88;
extern s32 D_800A5E8C;
extern s32 D_800A5E90;
extern u8 D_800A5E94;
extern u8 D_800A5E95;
extern u8 D_800A5E96;
extern u8 D_800A5EA4;
extern u8 D_800A5EA5;
void func_80180000(s32 param_1) {
if (*(s32 *)(param_1 + 0xE0) != 0 && --*(s32 *)(param_1 + 0xE0) == 0) {
u8 *base;
__asm__("la %0, D_800A5E88" : "=r"(base));
*(s32 *)base = 0;
D_800A5E8C = 0x1E;
D_800A5E94 = 0x99;
D_800A5E90 = 0;
D_800A5E95 = 0xB2;
D_800A5E96 = 0xB2;
func_80028620(0, base);
D_800A5EA4 = 0x19;
D_800A5EA5 = 0x19;
func_80028620(1, base + 0x10);
func_8012C218((void *)param_1);
} else {
if (--*(s32 *)(param_1 + 0xE4) == 0) {
u8 *rec;
D_800A5E88 = (rand() - 0x4000) >> 10;
if ((s32)(s16)D_80126B62 >= -0x2A0) {
D_800A5E8C = 0x1E;
} else if ((s32)(s16)D_80126B62 >= -0x4A0) {
D_800A5E8C = 0;
} else {
D_800A5E8C = -0x1E;
}
__asm__("la %0, D_800A5E90" : "=r"(rec));
*(s32 *)rec = (rand() - 0x4000) >> 10;
func_80028620(0, rec - 8);
*(s32 *)(param_1 + 0xE4) = (rand() & 1) + 1;
}
if (*(s32 *)(param_1 + 0xDC) != ((struct { s32 w; } *)&D_801270D8)->w) {
*(s32 *)(param_1 + 0xDC) = ((struct { s32 w; } *)&D_801270D8)->w;
switch (((struct { s32 w; } *)&D_801270D8)->w) {
case 2:
case 4:
case 6:
*(s32 *)(param_1 + 0xE0) = 0x1E;
break;
case 3:
case 5:
break;
}
}
}
}
extern void (*D_8018A90C[])(void);
+33 -1
View File
@@ -4376,7 +4376,39 @@ void func_8017F7CC(s32 param_1)
}
INCLUDE_ASM("asm/ov_SC02_005/nonmatchings/ov_SC02_005_jr_8017CF90", func_8017F898);
extern s32 D_801270C8;
extern u8 D_800D5C6C[];
extern s32 D_80197244;
extern void func_8017E190(void);
extern void func_8002D4C8(s32 a0, s32 a1);
extern void func_80154274(s32 *a0, s32 a1);
extern void func_8014706C(void *a0);
extern s32 func_8013767C(s32 a0);
void func_8017F898(s32 arg0) {
s32 s0 = arg0;
s16 sp10[2];
/* Non-volatile memory clobber (cookbook L1835 / §31 sched S7): the two
* dead s16 stack stores and the D_801270C8 load are all constant-address
* MEMs, so sched2 finds no memory dependence and its potential_hazard rule
* promotes the prologue `sw $ra` over the ALU candidates, sinking it to
* just above the branch. The clobber gives `sw $ra` a successor, so it is
* only ready after the load is picked and lands back in the prologue. */
__asm__("" : : : "memory");
sp10[1] = -0x40;
sp10[0] = 0;
if (D_801270C8 == 0xA) {
func_8017E190();
func_8002D4C8(0x510, 0x107F);
func_80154274((s32 *)s0, (s32)&D_800D5C6C);
func_8014706C((void *)s0);
*(s32 *)(s0 + 0x198) = func_8013767C((s32)&D_80197244);
*(u8 *)(s0 + 0x214) = *(u8 *)(s0 + 0x214) + 1;
}
}
extern s32 func_801399F0(s32 a0);
extern void func_80139914(s32 a0);
+73 -1
View File
@@ -3378,7 +3378,79 @@ void func_8017D4CC(void *a0) {
}
INCLUDE_ASM("asm/ov_SC02_026/nonmatchings/ov_SC02_026_jr_8017C180", func_8017D508);
/* 8 bytes, align 2 -> lwl/lwr/swl/swr copy (cookbook §48-C2) */
extern u16 func_80148800(s32 *a0);
extern void func_8017D780(s32 param_1, s16 *param_2);
// @class: schedule
// @stuck: none -- MATCH (114 ins). Four arms written out in full (§396-b: let gcc cross-jump the
// 0x2AA arms itself; hand-merging loses the register layout). The 0x71/0x600/-0x50 body is
// DUPLICATED in arms 1 and 3 and the target keeps BOTH copies, so it needs the §5a zero-byte
// cross-jump barrier -- but §336's "put it at the BOTTOM of the twin" is one statement too far
// here: below the last store it also blocks reorg's backward scan, the `j`'s delay slot steals
// `sh zero,0x32` from .L8017D67C instead of `sh $v0,0x30` (near 34, +1 ins). Placing it one
// statement HIGHER -- between the 0x2E store and the 0x30 store -- leaves a 1-instruction common
// suffix, which is below find_cross_jump's 2-insn minimum, so the merge still bails AND the
// delay slot still fills. That pins `li -0x50` below the asm (SCHEDULE-REORDER/3); birthing it
// as a local `k` ABOVE the barrier lets sched1 hoist it back to its target slot. All three
// lines (the local, the barrier, its position) are load-bearing -- do not tidy them.
void func_8017D508(s32 a0) {
typedef struct { s16 v[4]; } Blk8_80126940_8017D508;
extern s32 D_80126B58;
extern s16 D_80189BEC[];
extern Blk8_80126940_8017D508 D_80126940;
Blk8_80126940_8017D508 sp10;
u8 t;
if (func_80148800(&D_80126B58) & 3) {
t = (*(u8 *)(a0 + 5) + 1) & 1;
*(u8 *)(a0 + 5) = t;
*(s32 *)(a0 + 0x14) = D_80189BEC[t];
}
sp10 = D_80126940;
if (((u32)((u16)sp10.v[0] - 0x2C1) < 0x2BF) && (sp10.v[1] >= -0x35F) &&
((u32)((u16)sp10.v[2] - 0x581) < 0x77F)) {
*(s16 *)(a0 + 0x20) = 0x71;
*(s16 *)(a0 + 0x22) = 0x600;
*(s16 *)(a0 + 0x24) = 0;
*(s16 *)(a0 + 0x2E) = 0;
*(s16 *)(a0 + 0x30) = -0x50;
} else if (((u16)((u16)sp10.v[0] + 0x5FF) < 0x5FF) &&
((u32)((u16)sp10.v[2] - 0x681) < 0x3FC)) {
*(s16 *)(a0 + 0x20) = 0x2AA;
*(s16 *)(a0 + 0x22) = 0x800;
*(s16 *)(a0 + 0x24) = 0;
*(s16 *)(a0 + 0x2E) = 0;
*(s16 *)(a0 + 0x30) = 0;
} else if (sp10.v[2] < 0x380) {
s16 k = -0x50; /* §5a: born ABOVE the barrier so sched1 can hoist the li */
*(s16 *)(a0 + 0x20) = 0x71;
*(s16 *)(a0 + 0x22) = 0x600;
*(s16 *)(a0 + 0x24) = 0;
*(s16 *)(a0 + 0x2E) = 0;
__asm__ __volatile__(""); /* §5a/§336 cross-jump barrier - LOAD-BEARING, see above */
*(s16 *)(a0 + 0x30) = k;
} else {
*(s16 *)(a0 + 0x20) = 0x2AA;
*(s16 *)(a0 + 0x22) = 0x600;
*(s16 *)(a0 + 0x24) = 0;
*(s16 *)(a0 + 0x2E) = 0;
*(s16 *)(a0 + 0x30) = 0;
}
*(s16 *)(a0 + 0x32) = 0;
if (sp10.v[0] < -0x400) {
sp10.v[0] = -0x400;
}
if (sp10.v[1] >= -0x101) {
sp10.v[1] = -0x102;
}
func_8017D780(a0, sp10.v);
}
/* 8 bytes, align 2 -> lwl/lwr/swl/swr copy (cookbook §48-C2) */
+112 -1
View File
@@ -4227,7 +4227,118 @@ zero:
}
INCLUDE_ASM("asm/ov_SC03_007/nonmatchings/ov_SC03_007_jr_80183894", func_801850B4);
#include "common.h"
extern void func_80185B48(s32 a0);
extern void func_8012A828(s32 a0, void *a1);
extern s32 func_8012BD3C(s32 a0, s32 a1, s32 a2);
extern s32 func_8012D624(void *a0, s32 a1, s32 a2);
extern void func_80185E30(void *a0);
extern void (*D_8018C588[])(void);
/*
* Four load-bearing details (each single-axis A/B'd against match_one; the
* function is 95/95 byte-exact only with all four):
*
* 1. `s32 pad[4];` -- cookbook §162i1/§164-53 dead BLKmode local reserving the
* target's vars area, the same device func_80184E8C uses above in this TU.
* vars = 0x38 - ROUND8(args 0x10) - ROUND8(4*5 saved regs = 0x18) = 0x10.
* Without it the frame is -0x30 and every sw/lw offset is wrong.
*
* 2. `__asm__ __volatile__("" : "=r"(cc) : "0"(cc));` after the first
* func_8012D624 call. It is zero bytes and does BOTH jobs the target needs:
* (a) §5a cross-jump barrier -- find_cross_jump bails on a volatile asm, so
* gcc keeps BOTH copies of the `func_8012D624(a0,0xC0,0x30)` tail
* instead of tail-merging them (that merge costs 6 instructions);
* (b) it re-SETS cc, which kills the jump equivalence cse recorded at the
* `bne` above, so the deliberately-redundant `beq $s1,$s2` survives
* instead of folding to an unconditional `j`. That surviving beq is
* also what keeps cc live across the call -> cc earns $s1, the saved
* set grows to s0-s3+ra, and the frame reaches 0x38.
* Laundering `one` instead of `cc` here emits `beq $s2,$s1` (operands
* swapped, closeness 1). Laundering at the join instead of inside the arm
* loses the jump-threading (closeness 1 the other way).
*
* 3. `if (cc != one) goto second;` -- the explicit goto reproduces jump1's
* thread_jumps redirect (the bne skips PAST the redundant beq to
* .L8018512C). Written as a plain `if (cc == one) { ... }` the bne lands on
* the beq instead: same length, one wrong branch word.
*
* 4. `__asm__ __volatile__("");` before `one = 1;` -- zero-byte sched1 fence
* (§194-A) that keeps the `addiu $s2,$zero,1` from floating above the call
* and its sll/sra. `one` is pinned to $18 because otherwise the allocator
* hands cc/$s2 and one/$s1, i.e. the pair swapped.
*
* Every read-modify-write below is written in-place (`t = load; t += K;`)
* rather than `t = load + K;` -- §219: the in-place form reuses the load's
* register (`addiu $v0,$v0,0x200`), the other allocates a fresh one.
* q/u are separate locals from p/t on purpose: sharing them puts the head
* block's pointer in $v1 and its value in $v0, the reverse of the target.
*/
void func_801850B4(s32 a0) {
s32 pad[4];
s32 p;
s32 q;
s32 u;
s32 t;
s32 hold;
register s32 one __asm__("$18");
s32 cc;
func_80185B48(a0);
if (*(s32 *)(a0 + 0x1C) != 0) {
q = *(s32 *)(a0 + 0x20);
u = *(u16 *)(q + 0x12);
hold = u + 0x100;
u -= 0x380;
*(u16 *)(q + 0x12) = u;
cc = (s16)func_8012BD3C(a0, 0x100, 0x9000);
__asm__ __volatile__("");
one = 1;
if (cc != one) {
goto second;
}
func_8012D624((void *)a0, 0xC0, 0x30);
__asm__ __volatile__("" : "=r"(cc) : "0"(cc));
if (cc == one) {
goto rejoin;
}
second:
p = *(s32 *)(a0 + 0x20);
*(u16 *)(p + 0x12) = *(u16 *)(p + 0x12) + 0x800;
if (func_8012BD3C(a0, 0x100, 0x9000) == one) {
func_8012D624((void *)a0, 0xC0, 0x30);
}
rejoin:
*(u16 *)(*(s32 *)(a0 + 0x20) + 0x12) = hold;
if (*(s32 *)(a0 + 0x1C) >= 5) {
p = *(s32 *)(a0 + 0x20);
t = *(u16 *)(p + 0x1C);
t += 0x200;
*(u16 *)(p + 0x1C) = t;
*(u16 *)(p + 0x18) = t;
p = *(s32 *)(a0 + 0x20);
t = *(u16 *)(p + 0x1A);
t -= 0x100;
} else {
p = *(s32 *)(a0 + 0x20);
t = *(u16 *)(p + 0x1C);
t -= 0x600;
*(u16 *)(p + 0x1C) = t;
*(u16 *)(p + 0x18) = t;
p = *(s32 *)(a0 + 0x20);
t = *(u16 *)(p + 0x1A);
t += 0x300;
}
*(u16 *)(p + 0x1A) = t;
*(s32 *)(a0 + 0x1C) = *(s32 *)(a0 + 0x1C) - 1;
} else {
func_8012A828(a0, (void *)D_8018C588);
*(s32 *)(a0 + 0x1C) = 0;
func_80185E30((void *)a0);
}
}
extern u8 D_80126B5C;
extern s32 D_80126B64;
+144 -2
View File
@@ -3580,7 +3580,66 @@ void func_8017E170(s32 a0) {
}
INCLUDE_ASM("asm/ov_SC03_105/nonmatchings/ov_SC03_105_jr_8017C8D0", func_8017E180);
extern s32 func_8012E544(s32);
extern void func_80015978(s32, s32*);
extern void func_8017E33C(s32, s16*);
extern Blk8_80126940 D_80126940;
extern s32 D_801BA588;
void func_8017E180(s32 param_1) {
Blk8_80126940 arr[3];
s32 r;
s32 a0;
s32 a1;
s32 v1;
s32 t;
arr[0] = D_80126940;
r = func_8012E544(0x14B);
if (r != 0) {
func_80015978(r + 4, (s32*)&arr[1]);
if (*(s16*)&arr[1] < -0x200) {
*(s16*)&arr[1] = -0x200;
}
if (*(s16*)&arr[1] > 0x200) {
*(s16*)&arr[1] = 0x200;
}
} else {
arr[1] = D_80126940;
}
if (*(s16*)&arr[0] < -0x108) {
*(s16*)&arr[0] = -0x108;
}
if (*(s16*)&arr[0] > 0x108) {
*(s16*)&arr[0] = 0x108;
}
a1 = *(s16*)&arr[0];
a0 = *(s16*)&arr[1];
*(s16*)((s32)&arr[0] + 2) = -0x102;
*(s16*)((s32)&arr[0] + 4) = -0x200;
v1 = a1 - a0;
if (v1 < 0) {
v1 = a0 - a1;
}
__asm__ __volatile__("" ::: "memory");
*(s32*)(param_1 + 0x14) = (v1 * 0x280) / 0x400 + 0x320;
t = *(s16*)&arr[0];
*(s16*)&arr[0] = t - t * (s16)(*(s32*)(param_1 + 0x10) - 0x320) / 640;
switch (D_801BA588) {
case 0:
*(s16*)(param_1 + 0x20) = 0;
*(s16*)(param_1 + 0x30) = -0xB0;
break;
case 1:
*(s16*)(param_1 + 0x20) = 0xE3;
*(s16*)(param_1 + 0x30) = -0x40;
break;
}
func_8017E33C(param_1, (s16*)arr);
}
// @class: schedule
@@ -5801,7 +5860,90 @@ void func_80183F84(void *a0, void *a1) {
}
INCLUDE_ASM("asm/ov_SC03_105/nonmatchings/ov_SC03_105_jr_8017C8D0", func_8018423C);
#include "common.h"
/* 8 bytes, align 2 -> the `pos = vec` / `out = vec` assignments become the
* lwl/lwr/swl/swr movstrsi_internal pair (align != UNITS_PER_WORD). */
typedef struct {
u16 f0;
s16 f1;
u16 f2;
u16 f3;
} Vec4s16_8018423C;
extern s32 rand(void);
extern u16 D_800B99DA;
extern s32 D_8018E59C[];
extern s32 func_801850D8();
extern void func_8012F214(s32 a0, s32 a1, s32 a2);
extern s32 func_80135004(s32 a0, void *a1, s32 a2);
/* Four LOAD-BEARING details, each worth 1-2 instructions (Opus, S70y):
*
* 1. `tmp` is DECLARED AND UNUSED ON PURPOSE. The locals region is 0x20..0x3F
* (frame 0x58 = 0x20 outgoing args + 0x20 locals + 5 saved regs), i.e. FOUR
* 8-byte slots, and `out` sits at 0x38 -- so a fourth aggregate must be
* declared third to consume 0x30..0x37. expand_decl assigns slots upward
* from STARTING_FRAME_OFFSET in declaration order regardless of use.
* Deleting `tmp` moves `out` to 0x30 and shrinks the frame to 0x50.
*
* 2. `vec.f2 = 0;` BEFORE `vec.f0 = 0;`, and both AFTER `pos.f2`. The two
* `sh $zero` are the sched1 filler for the 0xE load-use gap; source order
* is what puts 0x2C ahead of 0x28 (§3-T2). The reversed order was the
* final 2-instruction residual.
*
* 3. `s32 r` with an EXPLICIT (s16) cast on the value, not `s16 r`. A `s16 r`
* defers the sll/sra 16 pair to the USE, which lands it after the second
* rand() where it coalesces straight into $a1 (-1 instruction). Casting at
* the definition puts sll/sra before/in the jal delay slot, keeps the value
* in the call-saved $s0 and re-materialises `addu $a1, $s0, $zero`.
*
* 4. `k++, step += 0x10` as the for-increment (k FIRST), and `j = i;` as the
* last statement of a do/while, not `if (i >= 3) break; j = i;`. The latter
* orders the test ahead of the copy, so reorg fills the back-branch delay
* slot with `addu $v1,$s0,$zero` instead of duplicating `addiu $s0,$v1,1`
* from the loop head (-1 instruction).
*/
void func_8018423C(s32 arg0)
{
Vec4s16_8018423C pos;
Vec4s16_8018423C vec;
Vec4s16_8018423C tmp;
Vec4s16_8018423C out;
s32 i;
s32 j;
s32 k;
s32 step;
s32 r;
pos.f0 = *(u16 *)(arg0 + 6);
pos.f1 = *(u16 *)(arg0 + 0xA);
pos.f2 = *(u16 *)(arg0 + 0xE);
vec.f2 = 0;
vec.f0 = 0;
j = 0;
do {
i = j + 1;
vec.f1 = -(i << 7);
func_8012F214(arg0, (s32)&vec, (s32)&vec);
if (func_80135004(1, &pos, (s32)&vec) != 0) {
if ((D_800B99DA & 7) == 0) {
step = -0x40;
out = vec;
for (k = 0; k < 8; k++, step += 0x10) {
out.f0 = vec.f0 + step;
r = (s16)(rand() % 0x1000 + 0x2000);
func_801850D8(7, r, 0, 0, &out, 0, D_8018E59C[rand() & 1], 0);
}
}
return;
}
pos = vec;
j = i;
} while (i < 3);
}
extern u8 D_801202A0[];
+62 -1
View File
@@ -3743,7 +3743,68 @@ void func_8017FEA4(s32 *a0) {
}
INCLUDE_ASM("asm/ov_SC03_117/nonmatchings/ov_SC03_117_jr_8017E6EC", func_8017FED8);
extern s32 func_8012BEE8(s32 a0);
extern void func_80015978(s32 a0, s32 *a1);
extern s32 rand(void);
extern u8 *func_801290DC(s32 arg0, u8 *arg1);
extern void func_8012BF4C(s32 *a0, s16 a1);
/* func_8017FED8 — jitter the caller's world position by a random signed offset
* and spawn entity 0x52 there.
*
* Two byte-load-bearing spellings (both cost zero instructions):
*
* 1. `s16 adj` — NOT s32. Cookbook §194-B/§209: an s32 second local lets
* local-alloc coalesce `adj = m` away (nop in the beqz delay slot); the
* HImode declaration makes it a mode-changing set that cprop cannot fold,
* so the `addu $a0,$v1,$zero` copy survives in the delay slot and `negu`
* writes $a0, not $v1, in place.
*
* 2. `m = rand(); m = m % K;` — NOT `m = rand() % K;`. expmed.c:2745
* (expand_divmod) zeroes `target` when `rem_flag && reg_mentioned_p(target,
* op0)`, i.e. only when the destination variable also appears in the
* DIVIDEND. The one-statement form lets gcc use m's own pseudo as the
* quotient temp too, stretching m's live range across the whole magic-number
* expansion so it conflicts with hard $v1 and gets evicted to $a0 (and the
* changed $a0 liveness then lets reorg steal `li $a0,0x52` into the
* `beqz` delay slot). Splitting the statement gives the quotient its own
* block-local pseudo -> quotient in $a0, m in $v1, adj in $a0.
*/
void func_8017FED8(s32 a0) {
u16 sp10[4];
u8 *ent;
s32 m1;
s16 adj1;
s32 m2;
s16 adj2;
if (func_8012BEE8(a0)) {
func_80015978(a0 + 4, (s32 *)sp10);
if (*(s32 *)(a0 + 0xDC) != 0) {
m1 = rand();
m1 = m1 % 352;
adj1 = m1;
if (m1 & 1) {
adj1 = -m1;
}
sp10[0] = sp10[0] + adj1;
m2 = rand();
m2 = m2 % 256;
adj2 = m2;
if (m2 & 1) {
adj2 = -m2;
}
sp10[2] = sp10[2] + adj2;
}
ent = func_801290DC(0x52, (u8 *)sp10);
if (ent != 0) {
func_8012BF4C((s32 *)a0, (rand() & 0x3F) + 0x20);
*(u16 *)(ent + 0xA) -= 0x200;
*(u16 *)(ent + 0x2E) = *(u16 *)(a0 + 0xFC);
}
}
}
+46 -1
View File
@@ -8942,7 +8942,52 @@ void func_80185FF4(void *a0)
}
INCLUDE_ASM("asm/ov_SC04_011/nonmatchings/ov_SC04_011_jr_8017D494", func_80186020);
extern u16 D_801EFD40;
extern u16 D_801EFD44;
extern u16 D_801EFD48;
extern u16 D_801EFD4C;
void func_80186020(void *a0) {
s32 p;
u16 x;
/* $v0 pin (cookbook "register pin" lever): without it sched1's birthing boost
hands $v0 to the wrong temp and the two lhu's swap registers (residual 12). */
register u16 y __asm__("$2");
if ((D_801EFD40 & 2) != 0) {
if ((s16)D_801EFD48 >= 0x120) {
*(s16 *)&D_801EFD4C = -0x60;
} else if ((s16)D_801EFD48 < -0x5F) {
*(s16 *)&D_801EFD4C = 0x60;
}
x = D_801EFD48;
y = D_801EFD4C;
p = *(s32 *)((s32)a0 + 0x20);
x = x + y;
y = D_801EFD44;
D_801EFD48 = x;
y = y + x;
*(u16 *)(p + 0x10) = y;
} else {
if ((D_801EFD40 & 1) != 0) {
if ((s16)D_801EFD48 >= 0x120) {
return;
}
y = (s16)D_801EFD48 + 0x60;
} else {
if ((s16)D_801EFD48 < -0x5F) {
return;
}
y = (s16)D_801EFD48 - 0x60;
}
x = D_801EFD44;
p = *(s32 *)((s32)a0 + 0x20);
D_801EFD48 = y;
x = x + y;
*(u16 *)(p + 0x10) = x;
}
}
void func_80186110(s32 arg0)
{
+101 -1
View File
@@ -4056,7 +4056,107 @@ void func_8017D3E0(s32 a0) {
}
INCLUDE_ASM("asm/ov_SC04_012/nonmatchings/ov_SC04_012_jr_8017AE2C", func_8017D4CC);
/* func_8017D4CC — 115 ins, MATCH.
*
* Three levers over the backlog draft (which sat at closeness 23):
*
* 1. §dbr "steal the join's first insn" — `sVar4 = 3;` belongs AFTER both inner
* `if`s, ONCE. dbr fills the `bne`'s delay slot with a COPY of the join's
* first insn and retargets the branch past it, leaving the original for the
* fall-through: that is the duplicated `addiu $v0,$zero,3` at 8017D66C /
* 8017D67C. The old draft wrote it both before and inside the `if`; gcc
* deleted the redundant one => exactly one instruction short (nins 114).
*
* 2. regalloc.md K3 (first-fit over the birth..death window): the else-arm's
* `sVar4 = 2;` must live INSIDE the `if (uVar6 < 2)` arm, not before the
* compare. Written before it, sVar4 is live across the `sltiu` temp, so the
* two conflict and everything shifts up a register ($v0->$v1->$a0). Written
* inside the arm it is born after the temp dies and shares $v0 with it —
* which is why the target can hold uVar6 in $v1 and sVar4 in $v0.
*
* 3. regalloc.md K3 again, for the 0x60000000 flag store: the or-chain must be
* written out in EACH arm, not through a shared accumulator behind a `goto`.
* In one block the three values (address, 0x40000000, 0x20000000) are all
* block-LOCAL qtys, so local-alloc's first-fit gives the address $v0 and the
* two constants $v1 (they die immediately); cross_jump then merges the
* identical `or/lui/or/sw` suffix back into one copy and dbr lifts the
* surviving `lui $v1,0x4000` into the `j`'s delay slot. Held in a shared
* local across the join it becomes a GLOBAL allocno instead, local-alloc
* hands the constants $v0 first, and the whole chain comes out swapped.
*/
extern s32 func_8012C354(s32 a0, s32 a1);
extern void func_8001C214(s32 a0, s32 a1);
extern s32 func_80029504(void);
extern u8 D_8018224C[];
extern u8 D_8018F198[];
extern u8 D_8018222C[];
extern u8 D_8018F210[];
extern u8 D_8018F288[];
extern u8 D_8018223C[];
void func_8017D4CC(s32 param_1) {
s32 iVar2;
u32 uVar5;
s16 sVar4;
u16 uVar6;
iVar2 = func_8012C354(param_1, (s32)D_8018224C);
if (iVar2 == 0) {
return;
}
sVar4 = *(s16 *)(param_1 + 0x70);
if (sVar4 == 1) goto CASE1;
if (sVar4 < 2) goto SKIP;
if (sVar4 == 2) goto CASE2;
if (sVar4 == 3) goto CASE3;
goto SKIP;
CASE1:
func_8001C214(*(s32 *)(param_1 + 0x20), (s32)D_8018F198);
*(u32 *)(param_1 + 0x58) = (u32)D_8018222C | 0x40000000 | 0x20000000;
goto SKIP;
CASE2:
func_8001C214(*(s32 *)(param_1 + 0x20), (s32)D_8018F210);
*(u32 *)(param_1 + 0x58) = (u32)D_8018223C | 0x40000000 | 0x20000000;
goto SKIP;
CASE3:
func_8001C214(*(s32 *)(param_1 + 0x20), (s32)D_8018F288);
*(u32 *)(param_1 + 0x58) = (u32)D_8018223C | 0x40000000 | 0x20000000;
SKIP:
*(u8 *)(param_1 + 0xc0) = 1;
*(u8 *)(param_1 + 0x75) = 2;
*(s16 *)(param_1 + 0xae) = -1;
*(s16 *)(param_1 + 2) = 1;
uVar5 = func_80029504();
if (uVar5 >= 0x2f0) {
if (*(s16 *)(param_1 + 0x70) == 0) {
*(s16 *)(*(s32 *)(param_1 + 0x20) + 0x12) = 0xd00;
}
if (*(s16 *)(param_1 + 0x70) == 1) {
*(s16 *)(*(s32 *)(param_1 + 0x20) + 0x12) = 0xd00;
}
if (*(s16 *)(param_1 + 0x70) == 2) {
*(s16 *)(*(s32 *)(param_1 + 0x20) + 0x12) = 0x300;
}
if (*(s16 *)(param_1 + 0x70) == 3) {
*(s16 *)(*(s32 *)(param_1 + 0x20) + 0x12) = 0x300;
}
} else {
uVar6 = *(u16 *)(param_1 + 0x70);
if (uVar6 < 2) {
sVar4 = 2;
} else {
if ((s16)uVar6 == 2) {
*(s16 *)(*(s32 *)(param_1 + 0x20) + 0x12) = 0x100;
}
if (*(s16 *)(param_1 + 0x70) == 3) {
*(s16 *)(*(s32 *)(param_1 + 0x20) + 0x12) = 0x100;
}
sVar4 = 3;
}
*(s16 *)(param_1 + 2) = sVar4;
}
}
extern void (*D_80182280[])(void);
+76 -1
View File
@@ -3328,7 +3328,82 @@ void func_8017EAB0(void *a0) {
}
INCLUDE_ASM("asm/ov_SC05_005/nonmatchings/ov_SC05_005_jr_8017D898", func_8017EAEC);
#include "common.h"
/* func_8017EAEC — ov_SC05_005 / ov_SC05_005_jr_8017D898 (107 ins, MATCH)
*
* Per-frame camera driver for this scene: every 4th tick (func_80148800 & 3)
* it flips the 1-bit phase at +5 and reloads the +0x14 field from the 2-entry
* table D_801860AC, then copies the 8-byte camera aggregate D_80126940 onto the
* stack, clamps it, offsets a second copy by sin/cos of the player angle, asks
* func_8017EE64 for a yaw and hands both to func_8017EC98.
*
* Levers:
* §48-C2 — D_80126940 is an 8-byte, 2-BYTE-ALIGNED aggregate, so the plain
* struct assign `sp10 = D_80126940;` is what emits the lwl/lwr + swl/swr
* block copy (same typedef shape the rest of this family uses).
* base+offset — `lw $v0, 0x20($s1)` reads D_80126B78 through the SAME $s1 that
* holds &D_80126B58 for the func_80148800 call, so it must be spelled off
* that base (`base[8]`), never as its own %hi(D_80126B78) fold.
* /128 — `bgez / addiu 0x7F / sra 7` is a signed divide by 128, not `>> 7`.
* §286 — THE ONLY RESIDUAL (closeness 4). Written as a plain statement,
* `*(s16 *)(a0 + 0xA0) = r;` is emitted BEFORE the argument setup, so the
* `li $a3,1` wins the jal's delay slot and the sh lands 4 insns early.
* Folding the store into a comma-expression in an ARGUMENT position places
* its RTL after the last arg move, and the sh takes the delay slot: 4 -> 0.
*/
/* 8 bytes, align 2 -> lwl/lwr/swl/swr copy (cookbook §48-C2) */
typedef struct { s16 v[4]; } Blk8_80126940_8017D6D0_8017EAEC;
extern u16 func_80148800(s32 *a0);
extern s32 func_8004787C(s32 a0);
extern s32 func_80047948(s32 a0);
extern s32 func_80012DBC(s32 a0, s32 a1, s32 a2, s32 a3);
extern s32 func_8017EE64(u8 *param_1, s16 *param_2, s16 *param_3);
extern void func_8017EC98(s32 param_1, s32 param_2, s16 *param_3);
extern s32 D_80126B58;
void func_8017EAEC(s32 a0) {
extern s16 D_801860AC[];
extern u8 D_80186058[];
extern s16 D_801B1EB8;
extern Blk8_80126940_8017D6D0_8017EAEC D_80126940;
Blk8_80126940_8017D6D0_8017EAEC sp10;
Blk8_80126940_8017D6D0_8017EAEC sp18;
s32 *base;
u8 t;
s32 r;
base = &D_80126B58;
if (func_80148800(base) & 3) {
t = (*(u8 *)(a0 + 5) + 1) & 1;
*(u8 *)(a0 + 5) = t;
*(s32 *)(a0 + 0x14) = D_801860AC[t];
}
sp10 = D_80126940;
if (sp10.v[0] < -0x1440) {
sp10.v[0] = -0x1440;
}
if (sp10.v[1] > -0x200) {
sp10.v[1] = -0x200;
}
if (sp10.v[0] > 0x1100) {
if (sp10.v[2] > -0x80) {
sp10.v[2] = -0x80;
}
}
sp18.v[0] = sp10.v[0] - func_8004787C(((s16 *)base[8])[9]) / 128;
sp18.v[2] = sp10.v[2] - func_80047948(((s16 *)base[8])[9]) / 128;
sp18.v[1] = sp10.v[1];
r = func_8017EE64(D_80186058, sp10.v, sp18.v);
/* §286: the store must ride the jal's delay slot -- see header. */
D_801B1EB8 = func_80012DBC(D_801B1EB8, (s16)r, 4, (*(s16 *)(a0 + 0xA0) = r, 1));
func_8017EC98(a0, D_801B1EB8, sp10.v);
}
extern s32 func_80012C6C(s32 a0, s32 a1, s32 a2);
+114 -3
View File
@@ -4583,7 +4583,7 @@ extern s32 D_801A108C;
extern s32 D_801A1090;
extern s32 D_801A1094;
extern void func_8017E840(void *a0);
extern void func_8017E6A4(void *a0);
extern void func_8017E6A4();
extern void func_8017EAE4(s32);
extern s32 func_8012AD50(void *a0);
@@ -4631,7 +4631,64 @@ void func_8017E354(s32 param_1) {
INCLUDE_ASM("asm/ov_SC05_008/nonmatchings/ov_SC05_008_jr_8017AE2C", func_8017E464);
INCLUDE_ASM("asm/ov_SC05_008/nonmatchings/ov_SC05_008_jr_8017AE2C", func_8017E6A4);
// @class: align-1 block move + sched1 load-delay fence
// The 8-byte sp+0x10 -> D_80126BE0 copy is an ALIGN-1 STRUCT ASSIGN (cookbook §160a /
// §48-C2): emit_block_move lowers it to lwl/lwr + swl/swr with ZERO memcpy-symbol
// reference, so this TU's file-scope `extern memcpy` (line 85, which disables the
// builtin and turns a memcpy() draft into a CALL -> LENGTH-DRIFT/-5) cannot drift it.
// Without the trailing zero-byte §194-A fence, sched1 hoists the `lhu D_80126B5E`
// above the block move and the pair eats its load-delay nop (cookbook-index L30: an
// lwl/lwr+swl/swr pair is a delay-slot SPONGE) -> LENGTH-DRIFT/-1. The fence keeps the
// nop and costs no bytes.
typedef struct { u8 b[8]; } Blk8_8017E6A4;
extern s32 func_80012F74(s32 a0, s32 a1, s32 a2, s32 a3);
extern void func_8012B370(s32);
extern void func_8012F214(s32 a0, s32 a1, s32 a2);
extern void func_80015954(s32 a0, s32 a1);
void func_8017E6A4(s32 param_1)
{
extern s32 D_801A12E4;
extern s32 D_8012699C;
extern s16 D_80126C88;
extern s16 D_80126C8A;
extern void (*D_80186544[])(void);
u8 buf[8];
u16 *rec;
s32 *b78;
s32 t;
rec = (u16 *)(D_8012697C + (D_80126980 + 8) * 6);
b78 = D_80126B78;
t = D_801A12E4;
*(u16 *)(param_1 + 6) = rec[0];
*(u16 *)(param_1 + 0xA) = rec[1];
*(u16 *)(param_1 + 0xE) = rec[2];
*(s16 *)((s32)b78 + 0x14) = -*(u16 *)(*(s32 *)(param_1 + 0x20) + 0x14) & 0xFFF;
*(s16 *)&D_80126C88 = -*(u16 *)(*(s32 *)(param_1 + 0x20) + 0x10) & 0xFFF;
*(s16 *)&D_80126C8A = (*(u16 *)(*(s32 *)(param_1 + 0x20) + 0x12) + 0x800) & 0xFFF;
D_8012699C = (func_80012F74((D_8012699C << 14) >> 16, (t << 29) >> 16, 10, 1) << 16) >> 14;
func_8012B370(param_1);
func_8012F214(param_1, (s32)D_80186544, (s32)&buf[0]);
func_80015954((s32)&buf[0], (s32)&D_80126B5C);
*(Blk8_8017E6A4 *)D_80126BE0 = *(Blk8_8017E6A4 *)&buf[0];
__asm__ __volatile__("");
*(s16 *)((s32)b78 + 8) = D_80126B5E;
*(s32 *)((s32)b78 + 0x48) = (s32)(s16)D_80126B5E;
*(s16 *)((s32)b78 + 0xA) = D_80126B62;
*(s32 *)((s32)b78 + 0x4C) = (s32)(s16)D_80126B62;
*(s16 *)((s32)b78 + 0xC) = D_80126B66;
*(s32 *)((s32)b78 + 0x50) = (s32)(s16)D_80126B66;
}
void func_8017E840(void *a0) {
extern s32 D_801A12E4;
@@ -4996,7 +5053,61 @@ void func_8017F24C(void *a0) {
}
INCLUDE_ASM("asm/ov_SC05_008/nonmatchings/ov_SC05_008_jr_8017AE2C", func_8017F288);
extern s32 func_8012C588(s32 a0, s32 a1);
extern s32 func_8012AD50(void *a0);
void func_8017F288(s32 param_1) {
extern s32 D_801151D4;
extern s32 D_801A1274;
extern void (*D_801865B8[])(void);
s32 wp;
s32 s1;
s32 v0;
s16 key;
s16 key2;
s16 i;
s16 lim;
void (**tbl)(void);
wp = D_801151D4;
s1 = D_801A1274;
key = *(u16 *)(*(s32 *)(wp + 0x34) + *(u16 *)(wp + 0x38) * 6 + 4);
if (*(s32 *)(s1 + 0xC) != 0) {
do {
if (*(s16 *)(s1 + 4) >= key) {
goto done1;
}
s1 += 0x10;
} while (*(u32 *)(s1 + 0xC) != 0);
}
done1:
i = *(u16 *)(wp + 0x38);
lim = i + 0x50;
if (i < lim) {
tbl = D_801865B8;
do {
key2 = *(u16 *)(*(s32 *)(wp + 0x34) + i * 6 + 4);
if (*(s32 *)(s1 + 0xC) != 0) {
do {
if (*(s16 *)(s1 + 4) >= key2) {
goto next;
}
v0 = func_8012C588((s32)tbl[*(u32 *)(s1 + 0xC)], param_1);
if (v0 != 0) {
*(s32 *)(v0 + 0xCC) = s1;
*(s32 *)(v0 + 0xDC) = i;
}
s1 += 0x10;
} while (*(u32 *)(s1 + 0xC) != 0);
}
next:
i++;
} while (i < lim);
}
D_801A1274 = s1;
func_8012AD50((void *)param_1);
}
extern s32 func_8012C588(s32 a0, s32 a1);
+67 -1
View File
@@ -3534,7 +3534,73 @@ void func_8017D4A4(void *arg0) {
}
INCLUDE_ASM("asm/ov_SC05_011/nonmatchings/ov_SC05_011_jr_8017BEBC", func_8017D610);
/* Twin of the banked same-TU neighbour func_8017D4A4 (§194-E): three
* "if (obj->f90 == &G && obj->f98 == 0)" guards whose f90 loads CSE together,
* three (func_80029178(id) & 0xFF) flag drains, then the neighbours own
* switch ((s16)(*(u16 *)(obj + 0x108))--) with case bodies ordered 0,1,5,3/7
* so cross-jumping folds the three func_800183E0 tails into one. */
extern s32 func_8012A828();
extern void func_80029124();
extern s32 func_80029178();
extern void func_800183E0();
extern s32 rand();
extern s32 D_80181DE8;
extern s32 D_80181FB8;
extern s32 D_80182138;
extern s32 D_80181D48;
extern s32 D_801820B0;
extern u8 D_80181E40[];
extern u8 D_8019A07C[];
extern u8 D_8019A61C[];
extern u8 D_8019A34C[];
extern s32 jtbl_8019BD08[];
void func_8017D610(void *arg0) {
if (*(s32 *)((s32)arg0 + 0x90) == (s32)&D_80181DE8) {
if (*(s16 *)((s32)arg0 + 0x98) == 0) {
func_8012A828(arg0, D_80181E40);
}
}
if (*(s32 *)((s32)arg0 + 0x90) == (s32)&D_80181FB8) {
if (*(s16 *)((s32)arg0 + 0x98) == 0) {
func_8012A828(arg0, &D_801820B0);
}
}
if (*(s32 *)((s32)arg0 + 0x90) == (s32)&D_80182138) {
if (*(s16 *)((s32)arg0 + 0x98) == 0) {
func_8012A828(arg0, &D_801820B0);
}
}
if (func_80029178(0x128) & 0xFF) {
func_80029124(0x128, 0);
func_8012A828(arg0, &D_80181DE8);
}
if (func_80029178(0x12A) & 0xFF) {
func_80029124(0x12A, 0);
func_8012A828(arg0, &D_80181D48);
}
if (func_80029178(0x12C) & 0xFF) {
func_80029124(0x12C, 0);
func_8012A828(arg0, &D_80182138);
}
switch ((s16)(*(u16 *)((s32)arg0 + 0x108))--) {
case 0:
*(u16 *)((s32)arg0 + 0x108) = (rand() & 0x3F) + 0x3C;
break;
case 1:
func_800183E0((s32)D_8019A07C);
break;
case 5:
func_800183E0((s32)D_8019A61C);
break;
case 3:
case 7:
func_800183E0((s32)D_8019A34C);
break;
}
}
extern void (*D_801823B0[])(void);
+85 -2
View File
@@ -4823,7 +4823,7 @@ extern void func_80016714(void *a0, s32 a1);
extern void func_8012C218(void *a0);
extern s32 func_80181294(u16 a0, s16 a1);
extern void func_8018124C(void);
extern void func_801810B0(void *a0);
extern void func_801810B0();
void func_80180DBC(void *a0) {
Ent_801E650C *p;
@@ -4885,7 +4885,90 @@ void func_80180DBC(void *a0) {
}
INCLUDE_ASM("asm/ov_SC05_018/nonmatchings/ov_SC05_018_jr_8017D604", func_801810B0);
/* func_801810B0 — POLY_FT4 (0x28 bytes) centred sprite emit.
*
* Levers that closed this one (previous attempt sat at closeness=47):
*
* 1. D_800A651C IS NOT `s32[]`. The tail computes the index with
* `sll 2; addu; sll 2` = *20, i.e. a 5-word record. Declaring it
* `extern struct { s32 a; s32 b[4]; } D_800A651C[];` at BLOCK scope
* (the proven src/800.c:3377/:5326 form — the card's fleet type
* ('s32','') is the scalar spelling engine_core.h's DEFINE_ macros use)
* supplied the two missing instructions and fixed idx 67..78.
*
* 2. The tpage store lands in the `beqz` DELAY SLOT only if the flag load
* PRECEDES it in source. With `*(u16*)(prim+0x16) = tpage;` written
* before the `if`, sched1 cannot prove the store does not alias the
* global, so a store->load dependence pins the store first and reorg
* steals `addiu 0x2E` from the arm instead. Hoisting the read into a
* local (`flag = D_801E664C;`) turns that into a load->store anti-dep:
* lui/lh/nop/beqz emit first and the `sh` sinks into the slot.
*
* 3. The UV and XY blocks are plain libgpu MACRO ORDER — setUV4
* (u0,v0,u1,v1,u2,v2,u3,v3) and setXY4 (x0,y0,x1,y1,x2,y2,x3,y3).
* Writing the stores in the order the *target emits* them (all v's
* then all u's / all y's then all x's) is the trap: that is the
* SCHEDULER's output, not the source. Source = macro order; sched1
* regroups by shared constant/register on its own. The declaration
* order of x/y/half is byte-irrelevant (all 6 permutations MATCH).
*/
void func_801810B0(s32 arg0)
{
extern void *func_80010A08(s32 arg0);
extern s32 GetClut(s32 a0, s32 a1);
extern s32 GetTPage(s32 a0, s32 a1, s32 a2, s32 a3);
extern s32 AddPrim(s32 a0, void *a1);
extern struct { s32 a; s32 b[4]; } D_800A651C[]; /* block scope: stride 0x14 (§ src/800.c:3377) */
s32 prim;
s32 clut;
s32 tpage;
s32 flag;
s32 x;
s32 y;
s32 half;
if (*(s16 *)(arg0 + 0xA) <= 0) {
return;
}
prim = (s32)func_80010A08(0x28);
clut = GetClut(0x160, 0x140);
*(u16 *)(prim + 0xE) = clut;
tpage = GetTPage(0, 1, 0x2C0, 0x100);
*(s32 *)(prim + 4) = 0x808080;
*(u8 *)(prim + 3) = 9;
*(u8 *)(prim + 7) = 0x2C;
flag = D_801E664C;
*(u16 *)(prim + 0x16) = tpage;
if (flag != 0) {
*(u8 *)(prim + 7) = 0x2E;
}
*(u8 *)(prim + 0xC) = 0x90;
*(u8 *)(prim + 0xD) = 0x50;
*(u8 *)(prim + 0x14) = 0xCF;
*(u8 *)(prim + 0x15) = 0x50;
*(u8 *)(prim + 0x1C) = 0x90;
*(u8 *)(prim + 0x1D) = 0x8F;
*(u8 *)(prim + 0x24) = 0xCF;
*(u8 *)(prim + 0x25) = 0x8F;
half = (s16)*(u16 *)(arg0 + 0xA) / 2;
x = *(s16 *)(arg0 + 4);
y = *(s16 *)(arg0 + 6);
*(s16 *)(prim + 8) = x - half;
*(s16 *)(prim + 0xA) = y - half;
*(s16 *)(prim + 0x10) = x + half;
*(s16 *)(prim + 0x12) = y - half;
*(s16 *)(prim + 0x18) = x - half;
*(s16 *)(prim + 0x1A) = y + half;
*(s16 *)(prim + 0x20) = x + half;
*(s16 *)(prim + 0x22) = y + half;
AddPrim(D_800A651C[*(u16 *)&D_800B9A02].a + 0x28, (void *)prim);
}
void func_80181204(void) {
+92 -1
View File
@@ -3638,7 +3638,98 @@ u32 func_8017ED24(u32 param_1, u32 param_2)
}
INCLUDE_ASM("asm/ov_SC06_006/nonmatchings/ov_SC06_006_jr_8017DB90", func_8017EEA4);
/* func_8017EEA4 — ov_SC06_006 actor step (116 ins).
*
* @class: INTEGRATION-ONLY. The body was already byte-true (match_one MATCH); the
* whole-binary gate refused it on SIX declaration conflicts with its own TU, all
* of which are fixed below. Do not redraft the body.
*
* LEVERS (each verified by removing it; only #2 is byte-visible):
* 1. `Blk8` is ALREADY a typedef in src/shared/engine_types.h (`struct {u8 b[8];}`),
* so a file-scope `typedef ... Blk8;` is a redefinition. Both block types are
* declared at BLOCK scope instead — legal shadowing, and match_one's standalone
* context has neither `Blk8` nor this TU's `Col4`. Byte-neutral.
* 2. §37 asm-label alias for func_8017ED24. The TU DEFINES `u32 func_8017ED24(u32,
* u32)` above our stub, so re-declaring it `(void*, s32)` is a hard conflict —
* but calling it through the u32 prototype costs a callee-saved register
* (work/$s0 gets rematerialised, $s2 disappears: 113 ins, LENGTH-DRIFT/-3).
* A distinct C name bound to the same asm symbol keeps the byte-true pointer
* form at zero conflict. This is the ONE edit of the six that moves bytes.
* 3. func_8017F4C0: adopt the TU's own spelling (def @3920 / decl @4819) whose 3rd
* param is s32, and put the (u16) cast at the CALL SITE — that is what emits the
* target's `andi $a2, $v0, 0xFFFF`. Byte-identical to a u16 parameter.
* 4. func_8017F258 is defined below us as `void func_8017F258(void)` (it reads its
* argument through a register-$4 pin, §42). An unprototyped decl is C89-
* compatible with that definition AND still passes $a0 in the delay slot.
* 5. §37 again for D_80185FA0: the TU declares it `Rec_80180AE8 []` AFTER our stub,
* and the typedef is not in scope here (duplicating it would be a redefinition).
* 6. §124 DEFINITION-side alias. The banked neighbour func_80180AE8 carries a
* block-scope `extern s32 func_8017EEA4();`, which conflicts with our required
* `void` return. Returning s32 instead is NOT free: $v0 becomes live-out, gcc
* can no longer sink `addiu $v0,$a0,1` into the case-0 branch delay slot, and the
* function grows to 117 ins. fix_arity_callers cannot help — it strips parameter
* lists, and this conflict is on the RETURN axis. So alias the definition.
*
* @stuck: none — match_one MATCH (116/116) AND recover_integration --probe-only
* reports `static: none / real cc1 MATCH 116 ins` compiling inside the real TU.
*/
extern s32 rand(void);
extern void func_8017F4C0(void *a0, u16 a1, s32 a2);
extern void aF8017ED24(void *, s32) __asm__("func_8017ED24");
extern void func_8017F258();
extern u8 D_80185FA0_recs[] __asm__("D_80185FA0");
extern u8 D_80185F94[];
void aF8017EEA4(void *arg) __asm__("func_8017EEA4");
void aF8017EEA4(void *arg)
{
typedef struct { u8 b[4]; } Blk4;
typedef struct { u8 b[8]; } Blk8;
u8 *work = (u8 *)arg + 0xC;
u8 *rec;
Blk4 t1;
Blk4 t2;
s16 phase;
(*(u16 *)((u8 *)arg + 6))++;
rec = &D_80185FA0_recs[*(s16 *)((u8 *)arg + 0x4C) * 20];
func_8017F4C0(work, *(u16 *)(rec + 2), (u16)((rand() & 0x7F) - 0x3F));
aF8017ED24(&t1, *(s32 *)(rec + 0xC));
*(Blk4 *)((u8 *)arg + 0xC) = t1;
aF8017ED24(&t2, *(s32 *)(rec + 0x10));
*(Blk4 *)((u8 *)arg + 0x10) = t2;
*(Blk8 *)(*(u8 **)((u8 *)arg + 8) + 8) = *(Blk8 *)D_80185F94;
phase = *(s16 *)((u8 *)arg + 4);
switch (phase)
{
case 0:
if (*(s16 *)((u8 *)arg + 6) < *(u8 *)(rec + 1))
break;
(*(u16 *)((u8 *)arg + 4))++;
*(u16 *)((u8 *)arg + 6) = 0;
break;
case 1:
*(s32 *)(*(u8 **)((u8 *)arg + 8) + 4) &= 0x7FFFFFFF;
*(s16 *)(*(u8 **)((u8 *)arg + 8) + 0x18) = *(s16 *)(*(u8 **)((u8 *)arg + 8) + 0x1A) =
*(u16 *)((u8 *)arg + 6) * *(u16 *)(rec + 6);
if (*(s16 *)((u8 *)arg + 6) < *(s16 *)(rec + 4))
break;
(*(u16 *)((u8 *)arg + 4))++;
*(u16 *)((u8 *)arg + 6) = 0;
break;
case 2:
if (*(s16 *)((u8 *)arg + 6) < *(s16 *)(rec + 8))
break;
(*(u16 *)((u8 *)arg + 6) = 0, (*(u16 *)((u8 *)arg + 4))++);
break;
case 3:
func_8017F258(arg);
break;
}
}
#include "common.h"
+70 -1
View File
@@ -3908,7 +3908,76 @@ next3:
}
INCLUDE_ASM("asm/ov_SC06_008/nonmatchings/ov_SC06_008_jr_8017C294", func_8017E37C);
#include "common.h"
/* ---- decls (TU house style: ratan2/func_80021174/rand copied verbatim from
* the existing decls in src/ov_SC06_008/ov_SC06_008_jr_8017C294.c;
* func_801290DC is left unprototyped exactly as the neighbour
* func_8017EBAC uses it; func_8017EBAC matches its definition below in
* the same TU. D_80189570 / D_801895B4 are new to this TU.) ---- */
extern u16 *D_80189570[];
extern s32 D_801895B4;
extern s32 func_80021174(s32 a0, s32 a1);
extern s32 ratan2(s32 dx, s32 dy);
extern s32 rand(void);
extern void func_8017EBAC(int param_1);
extern s32 func_801290DC();
/* 0x10-byte spawn record built on the stack at sp+0x10 */
typedef struct {
s16 x; /* 0x00 */
s16 y; /* 0x02 */
s16 z; /* 0x04 */
u16 ang; /* 0x06 */
s16 f8; /* 0x08 */
s16 fA; /* 0x0A */
s16 fC; /* 0x0C */
s16 fE; /* 0x0E */
} Cfg_8017E37C;
/* 0x10-byte VECTOR at sp+0x20 (pad needed: it fixes s0/ra at 0x30/0x34) */
typedef struct {
s32 vx, vy, vz, pad;
} Vec_8017E37C;
s32 func_8017E37C(void *a0, s32 a1, s32 a2)
{
Cfg_8017E37C sp10;
Vec_8017E37C sp20;
u16 *p;
s32 q;
/* §219: the base load and the stride add MUST be two statements —
* folding them into one expression schedules the a1*6 chain first and
* lands the base in $v1 instead of accumulating into $s0. */
p = D_80189570[*(s16 *)((s32)a0 + 0x70)];
p += (s16)a1 * 3;
sp10.x = p[0];
sp20.vx = sp10.x;
sp10.y = p[1] + a2;
sp20.vy = sp10.y;
sp10.z = p[2];
sp20.vz = sp10.z;
sp10.f8 = p[3];
if (sp10.f8 == 0x7FFF) {
return 1;
}
if (func_80021174(D_801895B4, (s32)&sp20) != 1) {
return 0;
}
sp10.fC = p[5];
sp10.ang = ratan2(sp10.f8 - sp10.x, sp10.fC - sp10.z);
q = func_801290DC(0x68, &sp10);
if (q != 0) {
*(s16 *)(q + 0x2C) = sp10.ang;
}
if ((rand() & 7) == 0) {
func_8017EBAC((int)&sp10);
}
return 0;
}
extern void (*D_80189604[])(void);
+116 -1
View File
@@ -3567,7 +3567,122 @@ void func_8017D940(int param_1)
}
INCLUDE_ASM("asm/ov_SC06_032/nonmatchings/ov_SC06_032_jr_8017C24C", func_8017DABC);
extern void func_80015978(s32 a0, s32 *a1);
extern void func_800178EC(s32 a0);
/* func_8017DABC — builds one POLY_FT4 (0x48 bytes) on the stack and hands it to
* func_800178EC. Line-for-line relative of the banked twin ov_SC06_018:func_8017DB20
* (§193-A); adapted, not copied — every literal and both jal symbols were re-read
* off this .s. Only two relocations exist here (func_80015978, func_800178EC);
* there are no %hi/%lo or data symbols.
*
* Two load-bearing deviations from the twin:
*
* 1. param_3/param_4 are s32 here, NOT the twin's s16. This TU already carries
* `extern void func_8017DABC();` at ov_SC06_032_jr_8017C24C.c:3522, and an
* empty parameter-name-list declaration cannot match a definition with an
* argument type that has a default promotion — cc1 rejects the s16 spelling
* with "conflicting types for `func_8017DABC'" (verified: the s16 body is a
* clean standalone MATCH but does not compile in this TU's decl environment,
* the §376/§378 bank-failure class). The twin's TU has NO prior declaration
* at all, which is why s16 was legal there. Widening to s32 is byte-inert:
* both operands of `v1val + param_3` promote to int either way, so the
* prologue keeps its plain `addu $s0,$a2,$zero` and the store still truncates.
*
* 2. `register Quad_8017DABC *p asm("$16")` pins the primitive pointer to $s0,
* reproducing `addiu $s0,$sp,0x10` and the $s0-based tail; the vertex block
* is written through the object `q` so those stores stay $sp-relative, which
* is what the .s does (sh at 0x10/0x12/0x18/0x1A/0x20/0x22/0x28/0x2A($sp)
* against sh 0x4($s0) and the whole 0x30..0x44 tail off $s0).
*
* The two zero-byte `__asm__ __volatile__("")` barriers are scheduling fences:
* the first keeps `p->vz0 = 0` from sinking past the rgb block, the second keeps
* the rgb stores ahead of the first `lw 0xE4($s2)`. func_800178EC takes ONE
* argument (fleet-wide `extern void func_800178EC(s32 a0);`, 7 TUs); $a1 is live
* at the jal only because it still holds the 0x13F used by the v2/v3 stores.
*
* match_one: MATCH, closeness 0, 78/78 instructions.
*/
typedef struct {
s16 vx0, vy0, vz0, pad0;
s16 vx1, vy1, vz1, pad1;
s16 vx2, vy2, vz2, pad2;
s16 vx3, vy3, vz3, pad3;
s16 u0, v0;
s16 u1, v1;
s16 u2, v2;
s16 u3, v3;
s32 rgb0, rgb1, rgb2, rgb3;
s32 flags;
u8 clut;
} Quad_8017DABC;
void func_8017DABC(s32 param_1, s32 param_2, s32 param_3, s32 param_4, s32 param_5, s32 param_6)
{
Quad_8017DABC q;
register Quad_8017DABC *p asm("$16");
s32 buf[2];
u16 v1val;
u16 v0val;
s16 yhi;
s16 ylo;
s16 xsel;
s16 t_vx0;
s16 t_vx1;
func_80015978(param_1 + 4, buf);
v1val = *(u16 *)buf;
v0val = *((u16 *)buf + 1);
t_vx0 = v1val + param_3;
p = &q;
yhi = v0val + 0x80;
t_vx1 = v1val + param_4;
ylo = v0val - 0x80;
q.vx0 = t_vx0;
q.vy0 = yhi;
q.vx1 = t_vx1;
q.vy1 = ylo;
switch (param_2) {
case 0:
xsel = v1val - 0xA0;
break;
case 1:
xsel = v1val + 0xA0;
break;
default:
goto skip_store;
}
q.vx2 = xsel;
q.vy2 = yhi;
q.vx3 = xsel;
q.vy3 = ylo;
skip_store:
p->vz0 = 0;
__asm__ __volatile__("");
p->rgb0 = p->rgb1 = param_5;
p->rgb2 = p->rgb3 = param_6;
__asm__ __volatile__("");
p->u0 = *(s32 *)(param_1 + 0xE4) + 0xC00;
p->v0 = 0x100;
p->u1 = *(s32 *)(param_1 + 0xE4) + 0xC3F;
p->v1 = 0x100;
p->u2 = *(s32 *)(param_1 + 0xE4) + 0xC00;
p->v2 = 0x13F;
p->u3 = *(s32 *)(param_1 + 0xE4) + 0xC3F;
p->v3 = 0x13F;
p->clut = 0x36;
p->flags = 0x50000000;
func_800178EC((s32)p);
}
s32 func_8017DBF4(void) {
+210 -2
View File
@@ -3493,7 +3493,167 @@ void func_801838A4(s32 p) {
}
INCLUDE_ASM("asm/ov_SC06_032/nonmatchings/ov_SC06_032_jr_80182890", func_80183940);
/* func_80183940 @ ov_SC06_032 (subseg ov_SC06_032_jr_80182890) - 110 ins. MATCH.
*
* GATE: .venv/bin/python tools/match_one.py func_80183940 \
* --c .run/S70y_1/opus/func_80183940.c \
* --asm-subdir asm/ov_SC06_032/nonmatchings/ov_SC06_032_jr_80182890
*
* --------------------------------------------------------------- what it is
* "Sweep-check the object's forward arc, then either punt or hand off."
* 1. Build a rotation matrix from the owner's angle pair
* (*(s32*)(self+0x20))[0x10], [0x12] via func_80049CAC, load it into the
* GTE, and load the owner's translation (that same object + 0x34) as the
* trans matrix.
* 2. RotTransSV the two-entry local SVECTOR table at D_801A6964 -- this
* overlay's own tail.data (asm/ov_SC06_032/data/tail.data.s:23827), 16
* bytes = {0,0,-20,0} then {0,0,+20,0}, i.e. a back point and a front
* point 20 units either side -- into world space (b, c).
* 3. func_80135888(D_80126B78, D_80126B90, &b, &c) is the fleet's global
* collision probe. Blocked -> punt to func_8012F568(1, 1, self->0xFE,
* 0x50, &c, D_801152A8). Clear -> wind the owner's 0x14 angle back
* 0x100 and try the two handoffs (func_8012CBF4, else func_8012BEE8).
* 4. On the func_8012CBF4 path, re-register via func_80146A6C(6, self, ...)
* and stamp 0x2000 into the func_80132EF4(self, 0x22) record's 0x34.
* 5. Every arm except "func_8012BEE8 returned 0" falls into the shared
* func_8012C218(self) tail -- that is the `j .L80183AD0` at 80183A54 and
* 80183AB8, and the early-out is the beqz at 80183AC8.
*
* ------------------------------------------------------- how it was matched
* §193-A twin remap. The h_norm twin in ov_SC06_018 (subseg _jr_80187AEC,
* 110 ins, banked) is line-for-line identical in shape; the ONLY per-overlay
* symbol is the two-entry SVECTOR table, re-pointed here at this overlay's
* D_801A6964. Every other symbol -- func_80049CAC, RotTransSV, func_80135888,
* func_8012F568, func_8012CBF4, func_80146A6C, func_80132EF4, func_8012BEE8,
* func_8012C218, D_801152A8, D_80126B78, D_80126B90 -- is RESIDENT (fixed VA),
* so the remap is otherwise identity. Symbol set re-checked instruction by
* instruction against this target's own relocation lines (law 1c: match_one
* masks jal/HI16/LO16, so a MATCH proves shape, never identity).
*
* ---------------------------------------------------------------- the levers
* L1 LOCAL DECL ORDER IS THE FRAME LAYOUT. Frame 0x78; the 0x18-byte
* outgoing-arg area that func_80146A6C's 7 args force ends at 0x20, so
* mat lands at sp+0x20 (0x20 bytes), then sv 0x40, b 0x48, c 0x50,
* flag 0x58, with NO gap. Reordering these five shifts every sp
* displacement. `flag` is RotTransSV's shared 3rd/"flag" out-arg and is
* passed to both calls, which is why it is one local and not two.
*
* L2 ZERO FILE-SCOPE FOOTPRINT. This TU already carries, at file scope
* ABOVE the L3496 INCLUDE_ASM site: func_80146A6C L133, func_80135888
* L578, D_801152A8 L587/L622, func_80049CAC L2621/L2745, RotTransSV
* L2624, func_8012C218 L2710, func_80132EF4 L2744, func_8012BEE8 L2748,
* func_8012CBF4 L2759 -- every spelling below is that same spelling, so
* the block-scope copies are composite-compatible and add nothing.
* func_8012F568 (L3633) and the two s32* pointer externs (L3639-L3640,
* L3848) are declared only BELOW the site and only at block scope, so
* they are kept at block scope here too rather than hoisted.
*
* L3 THE GTE OPS ARE WRITTEN AS RAW INLINE ASM, NOT AS
* `#define gte_SetRotMatrix` / `#define gte_SetTransMatrix`. The
* expansion is token-identical to this TU's OWN macro bodies (L6488 /
* L6502) so codegen is unchanged and match_one confirms 110/110 -- but
* defining those two macros here would plant a second definition ~3000
* lines AHEAD of the existing pair, giving the draft a file-scope blast
* radius over already-banked code for no codegen benefit. Zero-footprint
* form: the same tokens, no macro.
*
* L4 THE TYPEDEFS ARE BLOCK-SCOPE AND §120-UNIQUIFIED. MTX_80183940 and
* SVEC_80183940 are layout-identical to engine_types.h `MATRIX`
* ({short m[3][3]; long t[3];}, 0x20) and `SVECTOR`
* ({short vx,vy,vz,pad;}, 8). They are written out rather than used by
* name so this draft compiles standalone under match_one, whose context
* does not carry engine_types.h; both names are absent from the TU
* (grep: 0 hits), and at block scope they cannot collide at all.
*/
void func_80183940(s32 param_1)
{
typedef struct { short vx, vy, vz, pad; } SVEC_80183940; /* == SVECTOR */
typedef struct { short m[3][3]; long t[3]; } MTX_80183940; /* == MATRIX */
extern s32 *D_80126B78;
extern s32 *D_80126B90;
extern u8 D_801152A8[];
extern SVEC_80183940 D_801A6964[];
extern void func_80049CAC(s32 a0, s32 a1);
extern void RotTransSV(void *a0, void *a1, void *a2);
extern s32 func_80135888(s32 a0, s32 a1, s32 a2, s32 a3);
extern void func_8012F568(s32 a0, s32 a1, s32 a2, s32 a3, s32 a4, s32 a5);
extern s32 func_8012CBF4(s32 a0);
extern s32 func_80146A6C(s32 a0, void *a1, s32 a2, s32 a3, s32 a4, s32 a5, s32 a6);
extern s32 func_80132EF4(s32 a0, s32 a1);
extern s32 func_8012BEE8(s32 a0);
extern void func_8012C218(void *a0);
MTX_80183940 mat; /* sp+0x20 */
SVEC_80183940 sv; /* sp+0x40 */
SVEC_80183940 b; /* sp+0x48 */
SVEC_80183940 c; /* sp+0x50 */
SVEC_80183940 flag; /* sp+0x58 */
s32 iVar;
sv.vx = *(u16 *)(*(s32 *)(param_1 + 0x20) + 0x10);
sv.vy = *(u16 *)(*(s32 *)(param_1 + 0x20) + 0x12);
sv.vz = 0;
func_80049CAC((s32)&sv, (s32)&mat);
/* gte_SetRotMatrix(&mat) -- expansion is byte-for-byte this TU's L6488 */
__asm__ volatile (
"lw $12, 0( %0 );"
"lw $13, 4( %0 );"
"ctc2 $12, $0;"
"ctc2 $13, $1;"
"lw $12, 8( %0 );"
"lw $13, 12( %0 );"
"lw $14, 16( %0 );"
"ctc2 $12, $2;"
"ctc2 $13, $3;"
"ctc2 $14, $4"
:
: "r"( &mat )
: "$12", "$13", "$14" );
/* gte_SetTransMatrix(*(s32 *)(param_1 + 0x20) + 0x34) -- this TU's L6502 */
__asm__ volatile (
"lw $12, 20( %0 );"
"lw $13, 24( %0 );"
"ctc2 $12, $5;"
"lw $14, 28( %0 );"
"ctc2 $13, $6;"
"ctc2 $14, $7"
:
: "r"( *(s32 *)(param_1 + 0x20) + 0x34 )
: "$12", "$13", "$14" );
RotTransSV(&D_801A6964[0], &b, &flag);
RotTransSV(&D_801A6964[1], &c, &flag);
if (func_80135888((s32)D_80126B78, (s32)D_80126B90, (s32)&b, (s32)&c) != 0) {
s16 sVar1 = *(s16 *)(param_1 + 0xFE);
func_8012F568(1, 1, sVar1, 0x50, (s32)&c, (s32)D_801152A8);
} else {
*(u16 *)(*(s32 *)(param_1 + 0x20) + 0x14) =
*(u16 *)(*(s32 *)(param_1 + 0x20) + 0x14) - 0x100;
iVar = func_8012CBF4(param_1);
if (iVar != 0) {
s16 sVar1 = *(s16 *)(param_1 + 6);
s16 sVar2 = *(s16 *)(param_1 + 0xA);
s16 sVar3 = *(s16 *)(param_1 + 0xE);
func_80146A6C(6, (void *)param_1, sVar1, sVar2, sVar3, 0, 0);
iVar = func_80132EF4(param_1, 0x22);
if (iVar != 0) {
*(u16 *)(iVar + 0x34) = 0x2000;
}
} else {
iVar = func_8012BEE8(param_1);
if (iVar == 0) {
return;
}
}
}
func_8012C218((void *)param_1);
}
@@ -11427,4 +11587,52 @@ void func_8018F3E4(void)
}
INCLUDE_ASM("asm/ov_SC06_032/nonmatchings/ov_SC06_032_jr_80182890", func_8018FB5C);
void func_8018FB5C(s32 arg0, s32 arg1, s32 arg2)
{
typedef struct { s32 a; s32 b[4]; } OtBlk_80182890_8018FB5C; /* == engine_types.h OtBlk (0x14); BLOCK scope per §94 type-carry */
extern OtBlk_80182890_8018FB5C D_800A651C_v[] __asm__("D_800A651C"); /* §200 alias: TU spells it `extern u8 D_800A651C[]` */
extern void *func_80010A08(s32);
extern void func_8004914C(void *);
extern void func_800491AC(void *);
extern s32 RotTransPers(s32, s32, s32 *, s32 *);
extern u8 D_800AF648;
extern u8 D_800A6518[];
extern short D_800B9A02;
extern void func_80016638(void *a0, s32 a1, s32 a2);
s32 sp10;
s32 sp14;
s32 temp_v0_2;
void *temp_v0;
s32 ot;
s32 depth4;
register u16 *bidx __asm__("$8");
register u32 mask1 __asm__("$7");
register s32 rgb __asm__("$16");
register u32 tag0 __asm__("$4");
rgb = arg2;
__asm__("" : "=r"(rgb) : "0"(rgb)); /* zero-byte 2nd SET: kills the sched1 birthing boost */
temp_v0 = func_80010A08(0x10);
*(u8 *)((u8 *)temp_v0 + 3) = 3;
*(s32 *)((u8 *)temp_v0 + 4) = rgb;
*(u8 *)((u8 *)temp_v0 + 7) = 0x42;
func_8004914C(&D_800AF648);
func_800491AC(&D_800AF648);
temp_v0_2 = RotTransPers(arg0, temp_v0 + 8, &sp10, &sp14);
if ((temp_v0_2 > 0) && (sp14 >= 0) &&
(RotTransPers(arg1, temp_v0 + 0xC, &sp10, &sp14) > 0) && (sp14 >= 0)) {
/* addPrim(otp, p) == setaddr(p, getaddr(otp)), setaddr(otp, p) */
mask1 = 0xFFFFFF;
bidx = (u16 *)&D_800B9A02;
depth4 = temp_v0_2 * 4;
tag0 = *(u32 *)temp_v0;
*(u32 *)temp_v0 = (tag0 & 0xFF000000) |
(*(u32 *)(depth4 + D_800A651C_v[*bidx].a) & mask1);
ot = D_800A651C_v[*bidx].a;
*(u32 *)(depth4 + ot) =
(*(u32 *)(depth4 + ot) & 0xFF000000) | ((u32)temp_v0 & mask1);
func_80016638(&D_800A6518[*bidx * 20], temp_v0_2, 1);
}
}
+131 -3
View File
@@ -4174,7 +4174,51 @@ extern void func_8012B200(u8 *a0);
}
INCLUDE_ASM("asm/ov_SC07_000/nonmatchings/ov_SC07_000_jr_8017BEBC", func_8017E658);
extern u16 D_80126B66;
extern u16 D_801865DA;
extern u16 D_801865DC;
extern void func_8012AD80(s32 arg0);
extern void func_8017E804(s32 *a0, s32 a1, s32 a2);
extern void func_8017E7A0(s32 arg0);
void func_8017E658(void *a0) {
s32 d;
s32 t;
s32 want;
if (*(s16 *)((u8 *)a0 + 0xE) >= 0x2401) {
if (*(s32 *)((u8 *)a0 + 0x18) != 0) {
*(s32 *)((u8 *)a0 + 0x18) -= 0x2000;
if (*(s32 *)((u8 *)a0 + 0x18) < 0) {
*(s32 *)((u8 *)a0 + 0x18) = 0;
}
}
} else {
d = (s16)(D_80126B66 - *(s16 *)((u8 *)a0 + 0xE));
if (d < 0x100) {
t = *(s32 *)((u8 *)a0 + 0x18);
if (t > 0x20000) {
*(s32 *)((u8 *)a0 + 0x18) = t - 0x4000;
}
} else if (d < 0x1F0) {
func_8017E804((s32 *)a0, 0x70000, 0x4000);
} else {
func_8017E804((s32 *)a0, 0xA0000, 0x4000);
}
}
want = (s16)(*(u16 *)((u8 *)&D_801865DA + (*(s16 *)((u8 *)a0 + 0xFE) << 3)) - 0x50);
if (*(s16 *)((u8 *)a0 + 0xA) < want) {
*(s16 *)((u8 *)a0 + 0xA) = *(s16 *)((u8 *)a0 + 0xA) + 1;
} else if (want < *(s16 *)((u8 *)a0 + 0xA)) {
*(s16 *)((u8 *)a0 + 0xA) = *(s16 *)((u8 *)a0 + 0xA) - 1;
}
func_8012AD80((s32)a0);
if (*(s16 *)((u8 *)&D_801865DC + (*(s16 *)((u8 *)a0 + 0xFE) << 3)) < *(s16 *)((u8 *)a0 + 0xE) + 0x180) {
*(s16 *)((u8 *)a0 + 0xFE) = *(s16 *)((u8 *)a0 + 0xFE) + 1;
}
func_8017E7A0((s32)a0);
}
extern s32 func_8004787C(s32 a0);
@@ -4995,7 +5039,58 @@ void func_8018000C(s32 arg0) {
}
INCLUDE_ASM("asm/ov_SC07_000/nonmatchings/ov_SC07_000_jr_8017BEBC", func_8018006C);
#include "common.h"
extern void func_8017DF94(s32 arg0, s16 *arg1, s32 arg2);
extern void func_801805C0();
extern void func_8012C218(void *arg0);
extern s32 func_8004787C(s32 a0);
extern void func_80180200(void *a0);
extern void func_801802E4(void *a0);
extern void func_8012E014(s32 a0);
extern u16 D_801865DA;
void func_8018006C(void *a0) {
struct { s16 x; s16 y; s16 z; } sp;
s32 idx;
u16 y;
if (*(s16 *)((u8 *)a0 + 0xE) - 0x80 < *(s16 *)(*(s32 *)((u8 *)a0 + 0x64) + 0xE)) {
func_8017DF94(*(s16 *)((u8 *)a0 + 0x70), (s16 *)&sp, 0);
func_801805C0(a0, *(s16 *)((u8 *)a0 + 0x102));
func_8012C218(a0);
return;
}
idx = *(s16 *)((u8 *)a0 + 0x70);
y = *(u16 *)((u8 *)a0 + 0xA);
if (*(s16 *)((u8 *)a0 + 0x100) >= 0x400) {
if ((s16)y < -0x1FF) {
*(s16 *)((u8 *)a0 + 0xA) = y + 8;
} else {
func_8017DF94(idx, (s16 *)&sp, 0);
func_8012C218(a0);
return;
}
} else {
*(s16 *)((u8 *)a0 + 0x100) = *(s16 *)((u8 *)a0 + 0x100) + 0x10;
*(s16 *)((u8 *)a0 + 0xA) = *(u16 *)((u8 *)a0 + 0xA) + (func_8004787C(*(s16 *)((u8 *)a0 + 0x100)) >> 9);
}
if (*(s16 *)((u8 *)a0 + 0x106) != 0) {
*(s16 *)((u8 *)a0 + 0x106) = *(s16 *)((u8 *)a0 + 0x106) - 1;
if ((*(s16 *)((u8 *)a0 + 0x106) & 3) == 0) {
func_80180200(a0);
}
}
func_801802E4(a0);
sp.z = 0;
sp.x = 0;
sp.y = *(u16 *)((u8 *)a0 + 0xA) - *(u16 *)((u8 *)&D_801865DA + (*(s16 *)((u8 *)a0 + 0x70) << 3));
func_8017DF94(*(s16 *)((u8 *)a0 + 0x70), (s16 *)&sp, 1);
if (*(u8 *)((u8 *)a0 + 0x74) != 0) {
func_8012E014((s32)a0);
}
}
extern void (*D_801867DC[])(void);
@@ -5120,7 +5215,40 @@ void func_80180584(void *a0) {
}
INCLUDE_ASM("asm/ov_SC07_000/nonmatchings/ov_SC07_000_jr_8017BEBC", func_801805C0);
#include "common.h"
extern s32 D_801867E8[];
extern s32 D_80186820[];
extern void *D_80186830[];
extern u16 D_80186848[];
extern u16 D_80186850[];
extern s32 func_8012C658(s32 a0, s32 a1, s32 a2);
extern void func_8018091C(void *a0, void *a1);
extern void func_80180704(void *a0);
extern void func_80180790(s32 self);
extern void func_8002D4C8(s32 a0, s32 a1);
void func_801805C0(s32 arg0, s32 arg1) {
s32 idx;
s32 i;
s32 rec;
char *base;
idx = D_801867E8[arg1];
base = (char *)D_80186830[idx];
for (i = 0; i < D_80186820[idx]; i++) {
rec = func_8012C658(0x3C4, (idx << 8) + i, arg0);
if (rec != 0) {
func_8018091C((void *)rec, base + i * 0xC);
*(u16 *)(rec + 0xA) += *(u16 *)(arg0 + 0xA) - D_80186848[idx];
*(u16 *)(rec + 0xE) += *(u16 *)(arg0 + 0xE) - D_80186850[idx];
func_80180704((void *)rec);
func_80180790(rec);
}
}
func_8002D4C8(0xC02, 0);
}
extern s32 rand(void);
+63 -1
View File
@@ -5518,7 +5518,69 @@ void func_80180850(void *a0) {
}
INCLUDE_ASM("asm/ov_SC07_001/nonmatchings/ov_SC07_001_jr_8017BEBC", func_8018088C);
// @class: loop-invariant sign-extension order
// @stuck: none — MATCH (104 ins, match_one).
// The whole residual was a 4-instruction REGALLOC-PERM in the loop preheader:
// the two hoisted `sll/sra 16` pairs (sign-extending `base` and `limit`) came out
// in the wrong order, taking $s5/$s4 with them. -dS shows BOTH sign-extensions are
// created by loop.c `move_movables` (insns 218-221, hoisted out of the loop), NOT by
// the pre-loop assignments — so `base`/`limit` stay HImode pseudos and NO statement
// permutation outside the loop can reorder them (verified: 5 orderings, all closeness 4;
// an `asm("")` fence costs a nop, register pins push the extends back INTO the loop).
// The order is the order `move_movables` FINDS them, i.e. the operand order of the
// loop's own comparison. Writing the guard as `limit > sum` instead of `sum < limit`
// expands `limit` first, hoists its extend first, and closes 4 -> 0. Same slt, same
// 104 instructions. (Extends §49/§167-34: the LUID dial reached through the COMPARISON
// OPERAND ORDER when the movable is loop-hoisted.)
//
// `base` must stay `u16` (the raw lhu is what feeds `addu $v0,$v0,$s3` at 0xA($s0));
// `(s16)base` in the guard is the separate $s4. `D_800B99DA % 5` uses the UNSIGNED
// 0xCCCCCCCD magic + `andi 0xFFFF` because gcc-2.7.2 c-typeck shortens `unsigned short
// % const` back to unsigned short (build_binary_op's TRUNC_MOD_EXPR `shorten`).
extern s32 D_80185678[];
extern u16 D_80185688[];
extern s16 D_801AAD6E[];
extern u16 D_800B99DA;
extern s32 func_8012913C(s32 a0);
extern s32 rand(void);
void func_8018088C(s32 a0)
{
s32 idx;
s32 rec;
u16 base;
s16 limit;
s32 ent;
idx = *(s16 *)(a0 + 0x70);
rec = D_80185678[idx];
if (rec == 0) {
return;
}
if (D_800B99DA % 5 != 0) {
return;
}
if (*(s32 *)(a0 + 0x3C) == *(s32 *)(a0 + 8)) {
return;
}
base = *(u16 *)&D_801AAD6E[idx * 4];
limit = D_80185688[idx];
while (*(s16 *)(rec + 6) != -1) {
if (limit > *(s16 *)(rec + 2) + (s16)base) {
ent = func_8012913C(0x22);
if (ent != 0) {
*(u16 *)(ent + 6) = *(u16 *)rec;
*(u16 *)(ent + 0xA) = *(u16 *)(rec + 2) + base;
*(u16 *)(ent + 0xE) = *(u16 *)(rec + 4);
*(u16 *)(ent + 0x34) = (rand() % 4095 + 0x6000) & 0xFFF0;
*(u16 *)(*(s32 *)(ent + 0x20) + 0x2C) = 0xC00C;
}
}
rec += 8;
}
}
extern s32 func_8012C1B8(void);
extern void func_8012CAE4(void *a0);
+44 -1
View File
@@ -5154,7 +5154,50 @@ void func_801823F8(s32 t)
}
INCLUDE_ASM("asm/ov_SC07_006/nonmatchings/ov_SC07_006_jr_8017BEBC", func_801826E0);
#include "common.h"
extern s32 D_801BFCE4;
extern u8 D_8018E4D4[];
extern u8 D_8018DF10;
extern void func_8017D8A4(s32 a0, void *a1, void *a2, s32 a3, s32 a4);
extern void func_80143C74(s32 a0, s32 a1);
extern void func_80128EA8(s32 a0, s32 a1, s32 a2);
extern s32 func_8012C658(s32 a0, s32 a1, s32 a2);
extern s32 rand(void);
void func_801826E0(s32 param_1, s32 param_2) {
s32 *q;
s32 work;
s32 rnd;
s32 t;
s32 u;
q = &D_801BFCE4;
func_8017D8A4(*q + 0xC, D_8018E4D4, D_8018E4D4 + 0x24, param_2, 1);
func_8017D8A4(*q + 0x18, D_8018E4D4 + 0xC, D_8018E4D4 + 0x30, param_2, 1);
func_8017D8A4(*q + 0x84, D_8018E4D4 - 0xC, D_8018E4D4 + 0x18, param_2, 1);
work = ((s32 (*)())func_80143C74)(param_1, 0);
if (work != 0) {
*(u16 *)(work + 0xE) = *(u16 *)(work + 0xE) - *(u16 *)(D_801BFCE4 + 0x84);
rnd = rand();
t = *(u16 *)(D_801BFCE4 + 0x86) - 0x20;
*(u16 *)(work + 0xA) = *(u16 *)(work + 0xA) + (t + (rnd & 0x3F));
rnd = rand();
u = *(u16 *)(D_801BFCE4 + 0x88) - 0x20;
*(u16 *)(work + 6) = *(u16 *)(work + 6) + (u + (rnd & 0x3F));
rnd = rand() & 0xFF;
*(u32 *)(work + 0x10) = (rnd - 0x80) << 10;
rnd = rand() & 3;
*(u16 *)(work + 0x16) = -(rnd + 4);
*(u32 *)(work + 0x18) = -((rand() & 0xFF) << 11);
func_80128EA8(*(u32 *)(work + 0x20), (void *)(work + 0xD0), &D_8018DF10);
func_8012C658(0x3AC, 6, work);
func_8012C658(0x3AC, 6, work);
func_8012C658(0x3AC, 6, work);
func_8012C658(0x3AC, 6, work);
}
}
/* func_801828A4 — banked from the S40 wave-1 draft.
+35 -1
View File
@@ -7929,7 +7929,41 @@ void func_801827A4(void *a0) {
}
INCLUDE_ASM("asm/ov_SC07_007/nonmatchings/ov_SC07_007_jr_8017BEBC", func_801827E0);
extern u8 D_80199AC4[];
extern void func_8002D844(s32 a0);
extern void func_8002D4C8(s32 a0, s32 a1);
/* Copies the 0x944B-byte SEQ/MIDI blob at 0x80199D4C (== D_80199AC4 + 0x288,
* starts "MThd") up to the 0x801F0000 scratch buffer, then opens + plays it.
*
* Two levers were needed over the numeric warm-start body:
* (1) the loop test is SIGNED (slt, not sltu): 0x801F944A is an `unsigned int`
* in C89, so a bare `p <= 0x801F944A` converts p and emits sltu. Cast it.
* (2) the `lui $at / addu $at,$at,$v1 / lbu %lo($at)` triple is an ASPSX
* big-offset macro expansion, and its operand order is a TELL for how the
* offset was spelled in the source (maspsx __init__.py, load path):
* numeric offset -> addu $at,<base>,$at
* SYMBOLIC addend -> addu $at,$at,<base>
* Fleet census: 26/26 numeric-`lui $at` sites use the first order, 569/569
* `%hi(sym)` sites use the second. This target has a numeric-looking `lui
* $at,0xFFFB` but the SECOND order — the lone outlier fleet-wide — because
* the addend is a symbol MINUS a constant, which splat cannot name. So the
* source indexes a real data symbol; `*(u8 *)(p - 0x562B4)` cannot emit it.
* gcc folds `sym[p - K]` into `%hi(sym+(-K))`, giving 0x80199AC4 + 0x288 -
* 0x801F0000 = -0x562B4 -> lui 0xFFFB / lbu -0x62B4. Byte-verified with the
* symbol resolved: all 23 words identical, relocations included. */
void func_801827E0(void) {
s32 p = 0x801F0000;
do {
*(u8 *)p = D_80199AC4[(p - 0x801F0000) + 0x288];
p += 1;
} while (p <= (s32)0x801F944A);
func_8002D844(0x801F0000);
func_8002D4C8(0x18E, 0);
}
extern void func_8001ABBC(s32 a0, s32 a1, void *a2, s32 a3, s32 sp10);
extern CdFileLoc cdFileLocTable[];
+102 -2
View File
@@ -4819,7 +4819,39 @@ void func_8017E520(s32 a0) {
}
INCLUDE_ASM("asm/ov_SC07_010/nonmatchings/ov_SC07_010_jr_8017AE2C", func_8017E5B8);
void func_8017E5B8(s32 a0, s32 a1) {
struct P8 { s16 unk0; u8 pad[6]; };
struct P4 { u16 unk0; u16 pad; };
struct EntA { u8 pad[0xDC]; s32 unkDC; u8 pad2[0x1C]; u16 unkFC; };
extern struct P8 D_80185D28[];
extern struct P8 D_80185D2A[];
extern struct P8 D_80185D2C[];
extern struct P8 D_80185D2E[];
extern struct P4 D_80185D5A[];
extern s32 func_80146A6C(s32, void*, s32, s32, s32, s32, s32);
extern void func_80015954(s32 a0, s32 a1);
extern s32 func_8012C588(s32 a0, s32 a1);
struct EntA *p;
s32 i;
func_80146A6C(0x1E, (void *)a0, D_80185D28[a1].unk0, D_80185D2A[a1].unk0,
D_80185D2C[a1].unk0, 0, 0);
for (i = 0; i < 4; i++) {
p = (struct EntA *)func_8012C588(0x3A6, 0);
if (p != 0) {
func_80015954((s32)&D_80185D28[a1], (s32)p + 4);
p->unkFC = D_80185D5A[D_80185D2E[a1].unk0].unk0 + (i * 170 - 341);
p->unkDC = 0;
}
p = (struct EntA *)func_8012C588(0x3A6, 0);
if (p != 0) {
func_80015954((s32)&D_80185D28[a1], (s32)p + 4);
p->unkFC = D_80185D5A[D_80185D2E[a1].unk0].unk0 + (i * 170 - 227);
p->unkDC = 1;
}
}
}
extern void func_80015954(s32 a0, s32 a1);
extern s32 func_8012C588(s32 a0, s32 a1);
@@ -5589,7 +5621,75 @@ void func_8017F798(void *arg0) {
}
INCLUDE_ASM("asm/ov_SC07_010/nonmatchings/ov_SC07_010_jr_8017AE2C", func_8017F860);
typedef struct {
s16 state; /* 0x00 */
s16 timer; /* 0x02 */
s16 unk4; /* 0x04 */
s16 unk6; /* 0x06 */
s16 unk8; /* 0x08 */
s16 unkA; /* 0x0A */
s16 unkC; /* 0x0C */
s16 unkE; /* 0x0E */
s16 unk10; /* 0x10 */
s16 unk12; /* 0x12 */
s32 unk14; /* 0x14 */
u16 unk18; /* 0x18 */
u16 unk1A; /* 0x1A */
} Blk1C;
extern void func_8017FA24(void *a0);
extern void func_8017FA50(s32 a0, void *s0);
void func_8017F860(s32 arg0) {
extern u8 D_801A7F8C[];
Blk1C *rec;
s32 i;
for (i = 0; i < 0x10; i++) {
rec = &((Blk1C *)D_801A7F8C)[i];
rec->unk6 = (rec->unk6 - 0x2D) & 0xFFF;
switch (rec->state) {
case 0:
if (rec->timer != 0) {
rec->timer--;
if (rec->timer == 0) {
rec->unk4 = -0x155;
rec->state++;
}
}
break;
case 1:
rec->unkC += 0x100;
if (rec->unkC > 0x1000) {
rec->unkC = 0x1000;
}
rec->unkE = rec->unk10 = rec->unkC;
func_8017FA50(arg0, rec);
if (rec->unkC == 0x1000) {
rec->timer = 0x1E;
rec->state++;
}
break;
case 2:
func_8017FA24(rec);
func_8017FA50(arg0, rec);
rec->timer--;
if (rec->timer == -1) {
rec->state++;
}
break;
case 3:
func_8017FA24(rec);
func_8017FA50(arg0, rec);
if (rec->unkC == 0) {
rec->timer = 0;
rec->state = 0;
}
break;
}
}
}
void func_8017FA24(void *a0) {
extern s32 D_80185C70[];