diff --git a/config/overlays.mk b/config/overlays.mk index ae2b47799..aa234adeb 100644 --- a/config/overlays.mk +++ b/config/overlays.mk @@ -21,7 +21,7 @@ ov_SC01_077_ELF := $(ov_SC01_077_OUT).elf ov_SC01_077_MAPFILE := $(ov_SC01_077_OUT).map ov_SC01_077_LD_SCRIPT := $(ov_SC01_077_OUT).ld ov_SC01_077_SPLAT_YAML := config/splat.ov_SC01_077.yaml -ov_SC01_077_JTBL_INTERLEAVE := --order tail.data.o,ov_SC01_077_a.o,ov_SC01_077_jr_8012ACE0.o,tail2.data.o,ov_SC01_077_jr_80135888.o,tail3.data.o,ov_SC01_077_jr_80135A4C.o,tail4.data.o,ov_SC01_077_jr_80135D20.o,tail5.data.o,ov_SC01_077_jr_801380E0.o,ov_SC01_077_o0.o,tail6.data.o,ov_SC01_077.o,tail7.data.o,ov_SC01_077_jr_8015444C.o,ov_SC01_077_jr_80154C24.o,ov_SC01_077_jr_801588CC.o,ov_SC01_077_jr_80159C84.o,tail8.data.o,ov_SC01_077_jr_8015A3C8.o,tail9.data.o,ov_SC01_077_jr_8015AE2C.o,tail10.data.o,ov_SC01_077_jr_8015C32C.o,tail11.data.o,ov_SC01_077_jr_8016AB6C.o,tail12.data.o,ov_SC01_077_jr_80171B4C.o,ov_SC01_077_jr_801734BC.o,tail13.data.o,ov_SC01_077_jr_801789AC.o,ov_SC01_077_jr_80178D40.o,tail14.data.o,ov_SC01_077_jr_8017A4AC.o,tail15.data.o,ov_SC01_077_jr_8017AE2C.o,tail16.data.o,ov_SC01_077_jr_80180B64.o,tail17.data.o,ov_SC01_077_jr_8018103C.o,tail18.data.o,ov_SC01_077_jr_80181BE4.o,tail19.data.o,ov_SC01_077_jr_801820DC.o,ov_SC01_077_jr_80182268.o,ov_SC01_077_jr_80182E7C.o,tail20.data.o,ov_SC01_077_jr_80183324.o,tail21.data.o,ov_SC01_077_jr_80183AF0.o,ov_SC01_077_jr_80183BAC.o,ov_SC01_077_jr_80183CF4.o,tail22.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve +ov_SC01_077_JTBL_INTERLEAVE := --order tail.data.o,ov_SC01_077_a.o,ov_SC01_077_jr_8012ACE0.o,tail2.data.o,ov_SC01_077_jr_80135888.o,tail3.data.o,ov_SC01_077_jr_80135A4C.o,tail4.data.o,ov_SC01_077_jr_80135D20.o,tail5.data.o,ov_SC01_077_jr_801380E0.o,ov_SC01_077_o0.o,tail6.data.o,ov_SC01_077.o,tail7.data.o,ov_SC01_077_jr_8015444C.o,ov_SC01_077_jr_80154C24.o,ov_SC01_077_jr_801588CC.o,ov_SC01_077_jr_80159C84.o,tail8.data.o,ov_SC01_077_jr_8015A3C8.o,tail9.data.o,ov_SC01_077_jr_8015AE2C.o,tail10.data.o,ov_SC01_077_jr_8015C32C.o,tail11.data.o,ov_SC01_077_jr_8016AB6C.o,tail12.data.o,ov_SC01_077_jr_80171B4C.o,ov_SC01_077_jr_801734BC.o,tail13.data.o,ov_SC01_077_jr_801789AC.o,ov_SC01_077_jr_80178D40.o,tail14.data.o,ov_SC01_077_jr_8017A4AC.o,tail15.data.o,ov_SC01_077_jr_8017AE2C.o,tail16.data.o,ov_SC01_077_jr_80180B64.o,ov_SC01_077_jr_8018103C.o,tail17.data.o,ov_SC01_077_jr_80181BE4.o,tail18.data.o,ov_SC01_077_jr_801820DC.o,ov_SC01_077_jr_80182268.o,ov_SC01_077_jr_80182E7C.o,ov_SC01_077_jr_80183324.o,tail19.data.o,ov_SC01_077_jr_80183AF0.o,ov_SC01_077_jr_80183BAC.o,ov_SC01_077_jr_80183CF4.o,tail20.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve build/src/ov_SC01_077/ov_SC01_077.o: JTBL_PADS := 0,0,4,0,4,0,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20,+0x38,+0x58,+0x70,+0x90,+0xa8 build/src/ov_SC01_077/ov_SC01_077_a.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x14 build/src/ov_SC01_077/ov_SC01_077_jr_8012ACE0.o: JTBL_PADS := 0,0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0xcc,+0xe0 @@ -34,8 +34,10 @@ build/src/ov_SC01_077/ov_SC01_077_jr_8015AE2C.o: JTBL_PADS := 0,4 # §8e pads ( build/src/ov_SC01_077/ov_SC01_077_jr_8016AB6C.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20 build/src/ov_SC01_077/ov_SC01_077_jr_80178D40.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x178 build/src/ov_SC01_077/ov_SC01_077_jr_8017AE2C.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x14 +build/src/ov_SC01_077/ov_SC01_077_jr_80180B64.o: JTBL_PADS := 0,0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20,+0x40 build/src/ov_SC01_077/ov_SC01_077_jr_8018103C.o: JTBL_PADS := 0,4,0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x18,+0x38,+0x58 build/src/ov_SC01_077/ov_SC01_077_jr_80182268.o: JTBL_PADS := 0,0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20,+0x40 +build/src/ov_SC01_077/ov_SC01_077_jr_80182E7C.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20 build/src/ov_SC01_077/ov_SC01_077_jr_80183324.o: JTBL_PADS := 0,0,0,0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20,+0x40,+0x60,+0x80 build/src/ov_SC01_077/ov_SC01_077_jr_80183BAC.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20 build/src/ov_SC01_077/ov_SC01_077_o0.o: JTBL_PADS := 0,4,4,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x38,+0xa8,+0x118 @@ -1520,7 +1522,7 @@ ov_SC03_014_ELF := $(ov_SC03_014_OUT).elf ov_SC03_014_MAPFILE := $(ov_SC03_014_OUT).map ov_SC03_014_LD_SCRIPT := $(ov_SC03_014_OUT).ld ov_SC03_014_SPLAT_YAML := config/splat.ov_SC03_014.yaml -ov_SC03_014_JTBL_INTERLEAVE := --order tail.data.o,ov_SC03_014.o,ov_SC03_014_jr_8012ACE0.o,tail2.data.o,ov_SC03_014_jr_80135888.o,tail3.data.o,ov_SC03_014_jr_80135A4C.o,tail4.data.o,ov_SC03_014_jr_80135D20.o,tail5.data.o,ov_SC03_014_jr_801380E0.o,ov_SC03_014_o0e.o,tail6.data.o,ov_SC03_014_jr_8013F350.o,tail7.data.o,ov_SC03_014_jr_8013FFD8.o,tail8.data.o,ov_SC03_014_jr_80140608.o,tail9.data.o,ov_SC03_014_jr_8015444C.o,ov_SC03_014_jr_80154C24.o,ov_SC03_014_jr_801588CC.o,ov_SC03_014_jr_80159C84.o,tail10.data.o,ov_SC03_014_jr_8015A3C8.o,tail11.data.o,ov_SC03_014_jr_8015AE2C.o,tail12.data.o,ov_SC03_014_jr_8015C32C.o,tail13.data.o,ov_SC03_014_jr_8016AB6C.o,tail14.data.o,ov_SC03_014_jr_80171B4C.o,ov_SC03_014_jr_801734BC.o,tail15.data.o,ov_SC03_014_jr_801789AC.o,ov_SC03_014_jr_80178D40.o,tail16.data.o,ov_SC03_014_jr_8017A4AC.o,tail17.data.o,ov_SC03_014_jr_8017AE2C.o,tail18.data.o,ov_SC03_014_jr_8017EB7C.o,tail19.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve +ov_SC03_014_JTBL_INTERLEAVE := --order tail.data.o,ov_SC03_014.o,ov_SC03_014_jr_8012ACE0.o,tail2.data.o,ov_SC03_014_jr_80135888.o,tail3.data.o,ov_SC03_014_jr_80135A4C.o,tail4.data.o,ov_SC03_014_jr_80135D20.o,tail5.data.o,ov_SC03_014_jr_801380E0.o,ov_SC03_014_o0e.o,tail6.data.o,ov_SC03_014_jr_8013F350.o,tail7.data.o,ov_SC03_014_jr_8013FFD8.o,tail8.data.o,ov_SC03_014_jr_80140608.o,tail9.data.o,ov_SC03_014_jr_8015444C.o,ov_SC03_014_jr_80154C24.o,ov_SC03_014_jr_801588CC.o,ov_SC03_014_jr_80159C84.o,tail10.data.o,ov_SC03_014_jr_8015A3C8.o,tail11.data.o,ov_SC03_014_jr_8015AE2C.o,tail12.data.o,ov_SC03_014_jr_8015C32C.o,tail13.data.o,ov_SC03_014_jr_8016AB6C.o,tail14.data.o,ov_SC03_014_jr_80171B4C.o,ov_SC03_014_jr_801734BC.o,tail15.data.o,ov_SC03_014_jr_801789AC.o,ov_SC03_014_jr_80178D40.o,tail16.data.o,ov_SC03_014_jr_8017A4AC.o,tail17.data.o,ov_SC03_014_jr_8017AE2C.o,tail18.data.o,ov_SC03_014_jr_8017EB7C.o,tail19.data.o,ov_SC03_014_jr_801848E4.o,tail20.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve build/src/ov_SC03_014/ov_SC03_014.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x14 build/src/ov_SC03_014/ov_SC03_014_jr_8012ACE0.o: JTBL_PADS := 0,0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0xcc,+0xe0 build/src/ov_SC03_014/ov_SC03_014_jr_80135D20.o: JTBL_PADS := 0,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x18 diff --git a/config/splat.ov_SC01_077.yaml b/config/splat.ov_SC01_077.yaml index f3b7e2ba4..aabf23060 100644 --- a/config/splat.ov_SC01_077.yaml +++ b/config/splat.ov_SC01_077.yaml @@ -144,21 +144,19 @@ segments: - [0xb0f64, .rodata, ov_SC01_077_jr_8017AE2C] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - [0xb0f98, data, tail16] - [0xb0fa0, .rodata, ov_SC01_077_jr_80180B64] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0xb0fc0, data, tail17] - [0xb1000, .rodata, ov_SC01_077_jr_8018103C] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0xb1078, data, tail18] + - [0xb1078, data, tail17] - [0xb1098, .rodata, ov_SC01_077_jr_80181BE4] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0xb10ac, data, tail19] + - [0xb10ac, data, tail18] - [0xb1128, .rodata, ov_SC01_077_jr_801820DC] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - [0xb1148, .rodata, ov_SC01_077_jr_80182268] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - [0xb11a8, .rodata, ov_SC01_077_jr_80182E7C] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0xb11c8, data, tail20] - [0xb11e8, .rodata, ov_SC01_077_jr_80183324] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0xb1288, data, tail21] + - [0xb1288, data, tail19] - [0xb12a8, .rodata, ov_SC01_077_jr_80183AF0] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - [0xb12c8, .rodata, ov_SC01_077_jr_80183BAC] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - [0xb1308, .rodata, ov_SC01_077_jr_80183CF4] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0xb1328, data, tail22] + - [0xb1328, data, tail20] - [0xB29D4, bin, trailing] # final 3 bytes (EOF 0xB29D7 not word-aligned; spimdisasm drops # a trailing partial word and a <4-byte `data` carve emits nothing, # so use `bin` = raw .incbin, byte-exact). Word-aligned overlays omit this. diff --git a/config/splat.ov_SC03_014.yaml b/config/splat.ov_SC03_014.yaml index 13b6c24fb..55f960ec5 100644 --- a/config/splat.ov_SC03_014.yaml +++ b/config/splat.ov_SC03_014.yaml @@ -169,6 +169,8 @@ segments: - [0xc1184, data, tail18] - [0xc11b0, .rodata, ov_SC03_014_jr_8017EB7C] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - [0xc11d0, data, tail19] + - [0xc1338, .rodata, ov_SC03_014_jr_801848E4] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) + - [0xc1358, data, tail20] - [0xC2EE4, bin, trailing] # final 3 bytes (EOF not 4-aligned; spimdisasm drops a partial word) - [0xC2EE7] # EOF marker = the 0.4.dec byte length # @TRAILING@ (above) is replaced by tools/new_overlay.sh: for a non-4-aligned overlay it becomes diff --git a/src/ov_SC01_077/ov_SC01_077_jr_80180B64.c b/src/ov_SC01_077/ov_SC01_077_jr_80180B64.c index 25129ae7b..b2f2a9cd4 100644 --- a/src/ov_SC01_077/ov_SC01_077_jr_80180B64.c +++ b/src/ov_SC01_077/ov_SC01_077_jr_80180B64.c @@ -3099,7 +3099,110 @@ void func_80180B64(int param_1) } -INCLUDE_ASM("asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_80180B64", func_80180C90); +#include "common.h" + +extern int rand(void); +extern void func_8012B2CC(); +extern void func_8012B23C(); +extern void func_8012B14C(); +extern void func_8012A828(); +extern void func_80182968(); +extern u8 D_8018AA8C[]; +extern u8 D_801AFF18[]; +extern s16 D_8018AACC; +extern s16 D_8018AAD0; +extern s32 ratan2(s32, s32); + +void func_80180C90(int param_1) +{ + s32 angle; + s32 r; + + if (*(s16 *)(param_1 + 6) >= 0x201) { + *(u8 *)(param_1 + 0xc1) = 0; + *(s16 *)(param_1 + 2) = 1; + *(s16 *)(param_1 + 0x34) = 0; + *(s32 *)(param_1 + 0x1c) = (rand() & 0x1f) + 0x46; + *(s16 *)(param_1 + 0xe0) = 0; + *(u16 *)(param_1 + 0xe2) = *(u16 *)(*(s32 *)(param_1 + 0x20) + 0x12); + switch ((s32)((u32)*(u16 *)(param_1 + 0x70) << 0x10) >> 0x18) { + case 0: + case 1: + case 2: + case 3: + case 7: + if (*(u16 *)(param_1 + 0xfe) & 2) { + *(s16 *)(param_1 + 0xe2) = 0; + } + goto L_defA; + case 5: + func_80182968(param_1); + break; + case 4: + case 6: + default: + L_defA: + *(s16 *)(param_1 + 0xe4) = 0x1e; + *(s16 *)(param_1 + 0x5e) = 0; + *(u16 *)(param_1 + 0x5c) = 0xaa10; + func_8012B2CC(param_1); + func_8012B23C(param_1); + func_8012B14C(param_1, D_8018AA8C); + func_8012A828(param_1, D_801AFF18); + break; + } + *(s16 *)(param_1 + 0x34) = 4; + angle = (ratan2((s32)*(s16 *)(param_1 + 0xe) - (s32)D_8018AAD0, + (s32)D_8018AACC - (s32)*(s16 *)(param_1 + 6)) - + 0x400) & + 0xfff; + } else { + r = (rand() & 0x7ff) - 0x400; + *(u8 *)(param_1 + 0xc1) = 0; + *(s16 *)(param_1 + 2) = 1; + *(s16 *)(param_1 + 0x34) = 0; + *(s32 *)(param_1 + 0x1c) = (rand() & 0x1f) + 0x46; + *(s16 *)(param_1 + 0xe0) = 0; + *(u16 *)(param_1 + 0xe2) = *(u16 *)(*(s32 *)(param_1 + 0x20) + 0x12); + switch ((s32)((u32)*(u16 *)(param_1 + 0x70) << 0x10) >> 0x18) { + case 0: + case 1: + case 2: + case 3: + case 7: + if (*(u16 *)(param_1 + 0xfe) & 2) { + *(s16 *)(param_1 + 0xe2) = 0; + } + goto L_defB; + case 5: + func_80182968(param_1); + break; + case 4: + case 6: + default: + L_defB: + *(s16 *)(param_1 + 0xe4) = 0x1e; + *(s16 *)(param_1 + 0x5e) = 0; + *(u16 *)(param_1 + 0x5c) = 0xaa10; + func_8012B2CC(param_1); + func_8012B23C(param_1); + func_8012B14C(param_1, D_8018AA8C); + func_8012A828(param_1, D_801AFF18); + break; + } + *(s16 *)(param_1 + 0x34) = 3; + angle = ((ratan2((s32)*(s16 *)(param_1 + 0xe) - (s32)D_8018AAD0, + (s32)D_8018AACC - (s32)*(s16 *)(param_1 + 6)) - + 0x400) & + 0xfff) + + r; + } + *(u16 *)(param_1 + 0xe2) = angle; + *(s16 *)(param_1 + 0xe0) = 0; + *(u16 *)(param_1 + 0xe4) = (rand() & 0x1f) + 0x1e; + *(s32 *)(param_1 + 0x1c) = 0x10; +} + extern s16 D_8018AACC; extern s16 D_8018AAD0; diff --git a/src/ov_SC01_077/ov_SC01_077_jr_80182E7C.c b/src/ov_SC01_077/ov_SC01_077_jr_80182E7C.c index 0fb54c0f6..2e00c1e52 100644 --- a/src/ov_SC01_077/ov_SC01_077_jr_80182E7C.c +++ b/src/ov_SC01_077/ov_SC01_077_jr_80182E7C.c @@ -3180,4 +3180,123 @@ void func_8018301C(s32 param_1) } -INCLUDE_ASM("asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_80182E7C", func_8018308C); +#include "common.h" + +extern s32 D_80126B58; +extern s32 func_8012E544(s32 a0); +extern void func_8012A828(s32, void *); +extern void func_8002D4C8(s32 a0, s32 a1); +extern void func_8012B178(s32 a0, s32 a1); +extern s32 func_80143B6C(s32 a0, s32 a1); +extern s32 func_8012CBA4(s32 a0); +extern void func_8012B2CC(s32 a0); +extern void func_8012B23C(s32 a0); +extern void func_8012B14C(s32 a0, s32 a1); +extern void func_80182968(void); +extern s32 rand(void); +extern s16 D_8018AB94; +extern s16 D_8018AB96; +extern unsigned char D_8018AA8C[]; +extern unsigned char D_801AFF18[]; +extern unsigned char D_801B09D0; + +void func_8018308C(s32 param_1) +{ + u8 *base = (u8 *)&D_80126B58; + s32 flag; + s32 cond; + s32 t; + s32 d; + s32 y; + u16 mode; + s32 p; + + mode = *(u16 *)(param_1 + 0x70); + cond = 1; + if ((u32)(mode - 0x505) >= 2) { + cond = ((mode & 0xff00) == 0x700); + } + if (cond == 0) { + flag = 0; + } else { + p = func_8012E544(0x298); + if (p != 0) { + if (*(u16 *)(p + 2) != 2) { + flag = 0; + } else { + d = *(s16 *)(p + 0xe) - *(s16 *)(param_1 + 0xe); + if (d < 0) { + d = -d; + } + flag = d < 0x20; + } + } + } + + if (flag != 0) { + *(s16 *)(param_1 + 2) = 0x10; + *(s16 *)(param_1 + 0x34) = 0; + *(s16 *)(param_1 + 0x5c) = 0; + *(s32 *)(param_1 + 0x1c) = 0; + ((void (*)(s32, u8 *))func_8012A828)(param_1, ((u8 *)&D_801B09D0)); + *(s16 *)(param_1 + 0x98) = 0; + *(u16 *)(*(s32 *)(param_1 + 0x20) + 0x2c) |= 0x10; + *(s16 *)(*(s32 *)(param_1 + 0x20) + 0x1a) = 0x100; + func_8002D4C8(0x436, 0); + } else { + switch (*(u16 *)(param_1 + 0x34)) { + case 0: + if (*(s16 *)(base + 0xe) > D_8018AB94) { + *(u16 *)(param_1 + 0x34) = 1; + func_8012B178(param_1, 0xfff60000); + } + break; + case 1: + t = *(u16 *)(param_1 + 0xfc) - 1; + *(u16 *)(param_1 + 0xfc) = t; + if ((t << 0x10) <= 0) { + func_80143B6C(param_1, 0); + *(u16 *)(param_1 + 0xfc) = 8; + } + func_8012CBA4(param_1); + y = *(s16 *)(param_1 + 0xe); + if ((y < D_8018AB96) || ((y - *(s16 *)(base + 0xe)) < 0x80)) { + *(u16 *)(param_1 + 0x70) = 0; + *(u8 *)(param_1 + 0xc1) = 0; + *(u16 *)(param_1 + 0x2) = 1; + *(u16 *)(param_1 + 0x34) = 0; + *(s32 *)(param_1 + 0x1c) = (rand() & 0x1f) + 0x46; + *(u16 *)(param_1 + 0xe0) = 0; + *(u16 *)(param_1 + 0xe2) = *(u16 *)(*(s32 *)(param_1 + 0x20) + 0x12); + switch ((s32)((u32)*(u16 *)(param_1 + 0x70) << 0x10) >> 0x18) { + case 0: + case 1: + case 2: + case 3: + case 7: + if ((*(u16 *)(param_1 + 0xfe) & 2) != 0) { + *(u16 *)(param_1 + 0xe2) = 0; + } + goto seq; + case 5: + ((void (*)(s32))func_80182968)(param_1); + break; + case 4: + case 6: + default: + seq: + *(u16 *)(param_1 + 0xe4) = 0x1e; + *(u16 *)(param_1 + 0x5e) = 0; + *(u16 *)(param_1 + 0x5c) = 0xaa10; + func_8012B2CC(param_1); + func_8012B23C(param_1); + func_8012B14C(param_1, (s32)&D_8018AA8C); + func_8012A828(param_1, &D_801AFF18); + break; + } + } + break; + } + } +} + diff --git a/src/ov_SC02_000/ov_SC02_000_jr_8018173C.c b/src/ov_SC02_000/ov_SC02_000_jr_8018173C.c index 49d39578a..d95fdb79e 100644 --- a/src/ov_SC02_000/ov_SC02_000_jr_8018173C.c +++ b/src/ov_SC02_000/ov_SC02_000_jr_8018173C.c @@ -3874,7 +3874,177 @@ void func_80189304(void *a0) { } -INCLUDE_ASM("asm/ov_SC02_000/nonmatchings/ov_SC02_000_jr_8018173C", func_80189340); +#include "common.h" + +/* func_80189340 -- ov_SC02_000 (jr_8018173C). Projects a 8-node "tail"/streamer chain + * (D_8018F6E4 = a u16[3] offset table, one row per node; the per-node byte deltas live in + * the object at +0x100 walking DOWNWARD) and, for every node whose bit is set in the + * s16 mask at obj+0xA8, emits 4 semi-transparent gouraud quads (POLY_G4, len=8, code=0x3A) + * that fan the previous->current screen segment out by +-radius in x (j=0,1) or y (j=2,3). + * + * Byte-verified levers (match_one MATCH, 260 ins): + * L1 FRAME (sp+0xE8, out-arg area 0x18 for func_8005A600's 5th arg). Declared-local slots + * are handed out in DECLARATION order; everything at/above sp+0xA0 is a reload SPILL + * slot (8-aligned, size rounded to 8 -- assign_stack_local(align == -1)), which is why + * prim/ot/mask/ptrC land on 0xA0/0xA8/0xB0/0xB8 with holes between. So the declared set + * must be exactly: tags[28] (0x18), sv[4] (0x88), xy[4] (0x90), rgb (0x98), flag (0x9C). + * `rgb` and `flag` MUST be plain address-taken scalars (4-aligned, no hole); folding them + * into the xy aggregate would 8-align them and shift the frame. + * L2 tags[1..27] are genuinely DEAD stores in the original -- keep them. The loop body must + * read `tags[0]` (not the `prim` pseudo): tags[] is an ARRAY_REF, so the store tags[i]= + * invalidates it every iteration and re-emits `lw $v0,0x18($sp)`. + * L3 ONE variable for BOTH the tags counter and the outer node counter. Two separate + * counters split the allocno; the merged one out-ranks the ptrC giv, takes $s4, and + * pushes ptrC into the 0xB8 spill -- which is also what forces the target's TWO separate + * `la D_8018F6E4` materialisations in the preheader (257 -> 260 ins). + * L4 `ret * 4` is written TWICE (ot, and the divisor). Binding it to a local computes it + * once and loses a `sll`; CSE cannot merge the two because a loop back-edge separates + * their extended basic blocks. + * L5 The 14 prim stores go through STRUCT types (MEM_IN_STRUCT_P). cse.c's true_dependence + * drops the dependence of a varying in-struct store on a fixed NON-struct scalar, so the + * plain scalar `rgb` survives across them (one `lw 0x98($sp)` feeding both colour stores) + * while the ARRAY_REF xy[] reads do NOT (each `sh` gets its own `lhu`). Plain + * `*(s32 *)(pp + 0x04)` casts would re-load rgb twice and lose the match. + * L6 `s32 c = rgb;` -- a block-scoped temp read at the TOP of the j body. It is what hoists + * `lw $v1,0x98($sp)` above `addiu $v0,$zero,8` and hands rgb $v1 (not $v0), which in turn + * lets the 0x3A constant float up. Reading `rgb` directly at the two colour stores is an + * 8-instruction schedule miss; it must NOT be hoisted out of the j loop (AddPrim clobbers + * memory, so loop.c cannot treat the load as invariant). + * L7 `xy[0]=xy[2]; xy[1]=xy[3];` sit AFTER the three sv[] stores (any earlier placement is + * 12-50 off), and `q` is a single walking `*q--` cursor seeded at obj+0x103 -- gcc folds + * the three peeled decrements into the one `addiu $s3,$s6,0x100`. + * + * Integration surface (host TU src/ov_SC02_000/ov_SC02_000_jr_8018173C.c): + * AGREES at file scope -- D_800AF648 (u8[]), D_80126950 (s32), D_800B9A02 (s16, matched to + * the TU canon; the u16 spelling also byte-matches but would CONFLICT), func_8004914C / + * func_800491AC (void(void*)). NOT declared anywhere in the TU or in include/ -- + * D_8018F6E4, D_800A651C, RotTransPers, func_80010A08, GetTPage, func_8005A600, AddPrim. + * D_8018F6E4 is overlay-local data: the ~2 family sibling needs its own symbol remapped. + */ + +extern u8 D_800AF648[]; +extern u16 D_8018F6E4[]; +extern s32 D_80126950; +extern s16 D_800B9A02; +extern s32 D_800A651C; + +extern void func_8004914C(void *a0); +extern void func_800491AC(void *a0); +extern s32 RotTransPers(void *a0, void *a1, s32 *a2, s32 *a3); +extern void *func_80010A08(s32 a0); +extern s32 GetTPage(s32 a0, s32 a1, s32 a2, s32 a3); +extern s32 func_8005A600(s32 a0, s32 a1, s32 a2, s32 a3, s32 a4); +extern s32 AddPrim(s32 a0, void *a1); + +void func_80189340(s32 p) +{ + typedef struct { u8 pad0[3]; u8 len; u8 pad1[3]; u8 code; } PHdr_80189340; + typedef struct { + u32 tag; /* 0x00 */ + u32 c0; /* 0x04 */ + u16 x0, y0; /* 0x08, 0x0A */ + u32 c1; /* 0x0C */ + u16 x1, y1; /* 0x10, 0x12 */ + u32 c2; /* 0x14 */ + u16 x2, y2; /* 0x18, 0x1A */ + u32 c3; /* 0x1C */ + u16 x3, y3; /* 0x20, 0x22 */ + } PG4_80189340; /* 0x24 */ + + s32 tags[28]; /* sp+0x18 */ + s16 sv[4]; /* sp+0x88 */ + u16 xy[4]; /* sp+0x90 */ + s32 rgb; /* sp+0x98 */ + s32 flag; /* sp+0x9C */ + + u8 *q; + u8 *pp; + u8 *prim; + s32 ret; + s32 ot; + s32 radius; + s32 tp; + s32 i, j; + u8 mask; + + func_8004914C(D_800AF648); + func_800491AC(D_800AF648); + + mask = 1; + q = (u8 *)(p + 0x103); + sv[0] = D_8018F6E4[0] + *q--; + sv[1] = D_8018F6E4[1] + *q--; + sv[2] = D_8018F6E4[2] + *q--; + ret = RotTransPers(sv, &xy[2], &rgb, &flag); + if (ret > 0 && flag >= 0) { + ot = *(s32 *)((s8 *)&D_800A651C + ((u16)D_800B9A02 * 0x14)) + ret * 4; + prim = (u8 *)func_80010A08(0x3FC); + if (prim != 0) { + tp = GetTPage(0, 1, 0, 0); + func_8005A600((s32)prim, 0, 0, (u16)tp, 0); + + tags[0] = (s32)(prim + 0xC); + for (i = 1; i < 28; i++) { + tags[i] = tags[0] + i * 0x24; + } + + radius = ((D_80126950 + 0x1F4) * 8) / (ret * 4); + pp = (u8 *)tags[0]; + + for (i = 1; i < 8; i++) { + sv[0] = D_8018F6E4[i * 3] + *q--; + sv[1] = D_8018F6E4[i * 3 + 1] + *q--; + sv[2] = D_8018F6E4[i * 3 + 2] + *q--; + xy[0] = xy[2]; + xy[1] = xy[3]; + RotTransPers(sv, &xy[2], &rgb, &flag); + if ((*(s16 *)(p + 0xA8) & mask) != 0) { + rgb = *(s32 *)(p + 0x1C) << 6; + for (j = 0; j < 4; j++) { + s32 c = rgb; + ((PHdr_80189340 *)pp)->len = 8; + ((PG4_80189340 *)pp)->c1 = 0; + ((PG4_80189340 *)pp)->c3 = 0; + ((PG4_80189340 *)pp)->c0 = c; + ((PG4_80189340 *)pp)->c2 = c; + ((PHdr_80189340 *)pp)->code = 0x3A; + ((PG4_80189340 *)pp)->x0 = xy[0]; + ((PG4_80189340 *)pp)->x1 = xy[0]; + ((PG4_80189340 *)pp)->x2 = xy[2]; + ((PG4_80189340 *)pp)->x3 = xy[2]; + ((PG4_80189340 *)pp)->y0 = xy[1]; + ((PG4_80189340 *)pp)->y1 = xy[1]; + ((PG4_80189340 *)pp)->y2 = xy[3]; + ((PG4_80189340 *)pp)->y3 = xy[3]; + switch (j) { + case 0: + ((PG4_80189340 *)pp)->x1 = ((PG4_80189340 *)pp)->x1 + radius; + ((PG4_80189340 *)pp)->x3 = ((PG4_80189340 *)pp)->x3 + radius; + break; + case 1: + ((PG4_80189340 *)pp)->x1 = ((PG4_80189340 *)pp)->x1 - radius; + ((PG4_80189340 *)pp)->x3 = ((PG4_80189340 *)pp)->x3 - radius; + break; + case 2: + ((PG4_80189340 *)pp)->y1 = ((PG4_80189340 *)pp)->y1 + radius; + ((PG4_80189340 *)pp)->y3 = ((PG4_80189340 *)pp)->y3 + radius; + break; + case 3: + ((PG4_80189340 *)pp)->y1 = ((PG4_80189340 *)pp)->y1 - radius; + ((PG4_80189340 *)pp)->y3 = ((PG4_80189340 *)pp)->y3 - radius; + break; + } + AddPrim(ot, pp); + pp += 0x24; + } + } + mask = mask << 1; + } + AddPrim(ot, prim); + } + } +} + INCLUDE_ASM("asm/ov_SC02_000/nonmatchings/ov_SC02_000_jr_8018173C", func_80189750); diff --git a/src/ov_SC02_011/ov_SC02_011_jr_8017AE2C.c b/src/ov_SC02_011/ov_SC02_011_jr_8017AE2C.c index d55c44bd1..9aaadc0ba 100644 --- a/src/ov_SC02_011/ov_SC02_011_jr_8017AE2C.c +++ b/src/ov_SC02_011/ov_SC02_011_jr_8017AE2C.c @@ -9250,7 +9250,88 @@ void func_8018CC6C(void *a0) *((s16 *) (((s32) v1) + 0x2C)) |= 0x10; } -INCLUDE_ASM("asm/ov_SC02_011/nonmatchings/ov_SC02_011_jr_8017AE2C", func_8018CC9C); +#include "common.h" + +extern s32 func_8012BEE8(s32 a0); +extern s32 func_8004787C(s32 a0); +extern s32 func_80047948(s32 a0); +extern s32 func_80143BDC(u16 *a0); +extern void func_8018D8B8(s32 a0); + +void func_8018CC9C(s32 param_1) +{ + s32 iVar3; + s32 uVar1; + s32 sVar2; + u16 sp10[3]; + + switch (*(u16 *)(param_1 + 0x34)) { + case 0: + if (func_8012BEE8(param_1) != 0) { + *(s16 *)(param_1 + 0x34) = 1; + *(s32 *)(param_1 + 0x1C) = 0x1E; + *(u32 *)(*(s32 *)(param_1 + 0x20) + 4) = + *(u32 *)(*(s32 *)(param_1 + 0x20) + 4) | 0x80000000; + *(s16 *)(param_1 + 0xE4) = 3; + } else { + uVar1 = func_8004787C((*(s32 *)(param_1 + 0x1C) << 10) >> 3); + iVar3 = *(s32 *)(param_1 + 0x20); + *(s16 *)(iVar3 + 0x1C) = uVar1; + *(s16 *)(iVar3 + 0x18) = uVar1; + sVar2 = func_80047948((*(s32 *)(param_1 + 0x1C) << 10) >> 3); + *(s16 *)(*(s32 *)(param_1 + 0x20) + 0x1A) = sVar2 * 2 + 0x1000; + } + break; + case 1: + uVar1 = *(u16 *)(param_1 + 0xE4) - 1; + *(u16 *)(param_1 + 0xE4) = uVar1; + if ((s16)uVar1 == 0) { + sp10[0] = *(u16 *)(param_1 + 0x6); + sp10[1] = *(u16 *)(param_1 + 0xA) - 0x38; + sp10[2] = *(u16 *)(param_1 + 0xE); + { + s32 e; + s32 q; + + e = func_80143BDC(sp10); + if (e != 0) { + q = *(s32 *)(e + 0xCC); + if (q != 0) { + *(s16 *)(q + 0x1A) = 0x5000; + *(s16 *)(q + 0x18) = 0x5000; + } + } + } + *(s16 *)(param_1 + 0x34) = 2; + } + /* fallthrough */ + case 2: + if (func_8012BEE8(param_1) != 0) { + *(s16 *)(param_1 + 0x34) = 3; + *(u32 *)(*(s32 *)(param_1 + 0x20) + 4) = + *(u32 *)(*(s32 *)(param_1 + 0x20) + 4) & 0x7FFFFFFF; + func_8018D8B8(param_1); + *(s32 *)(param_1 + 0x1C) = 8; + *(u16 *)(param_1 + 0x5C) = 0xAA10; + } + break; + case 3: + if (func_8012BEE8(param_1) != 0) { + *(u16 *)(*(s32 *)(param_1 + 0x20) + 0x2C) = + *(u16 *)(*(s32 *)(param_1 + 0x20) + 0x2C) & 0xFFEF; + *(s16 *)(param_1 + 0x2) = 0xB; + } else { + sVar2 = func_80047948((*(s32 *)(param_1 + 0x1C) << 10) >> 3); + iVar3 = *(s32 *)(param_1 + 0x20); + *(s16 *)(iVar3 + 0x1C) = sVar2; + *(s16 *)(iVar3 + 0x18) = sVar2; + uVar1 = func_8004787C((*(s32 *)(param_1 + 0x1C) << 10) >> 3); + *(s16 *)(*(s32 *)(param_1 + 0x20) + 0x1A) = uVar1 * 2 + 0x1000; + } + break; + } +} + INCLUDE_ASM("asm/ov_SC02_011/nonmatchings/ov_SC02_011_jr_8017AE2C", func_8018CEB0); @@ -9390,7 +9471,224 @@ void func_8018E780(s32 param_1) } -INCLUDE_ASM("asm/ov_SC02_011/nonmatchings/ov_SC02_011_jr_8017AE2C", func_8018E8A0); +#include "common.h" + +/* func_8018E8A0 — ov_SC02_011 / ov_SC02_011_jr_8017AE2C, 211 ins. + * STATUS: MATCH (relocation-masked, 211/211; frame 0xC0 / vars=136 / regs=9). + * + * Five load-bearing spellings, each byte-proven against gcc-2.7.2 source: + * + * 1. TWO INDUCTION REGISTERS, $s6=obj+0xCC (biv) and $s4=obj+0xD4 (reduced giv). + * `q = zb + k * 3` makes q a giv of a COUNTER biv, so its add_val is the + * invariant pseudo `zb`, not a CONST_INT. simplify_giv_expr (loop.c) refuses + * to fold `(reg) + (const)` into an add_val (both invariant, not CONSTANT_P + * => returns 0), so the four MEMs q[-1] q[0] ((s16*)q)[-1] ((s16*)q)[1] are + * NOT givs at all and keep their -4/-2/0/+2 offsets off $s4. Spelling the + * same addresses off the pointer biv `p` instead gives CONST_INT add_vals, + * combine_givs merges all six into ONE register based at 0xD6, and the two + * registers collapse (measured: 210 ins, wrong offsets). + * + * 2. `zb = obj+0xD4` is assigned INSIDE the loop so LICM hoists it into the + * preheader; the giv init then coalesces with it into a single + * `addiu $s4,$s3,0xD4` after `addiu $s7,$sp,0x58`. Assigned before the loop + * it stays in the entry block and costs `addiu $v1,$s3,212` + `move $s4,$v1`. + * + * 3. THE SECOND COUNTER `k` IS REQUIRED. loop.c emits a reduced giv's update + * with emit_insn_before(..., biv_increment_insn), i.e. immediately BEFORE the + * biv it derives from. The target's increment order is [i++][$s4+=12][$s6+=12], + * so the giv must hang off a biv that increments between i and p. `k` is a + * dead counter that loop.c deletes after reduction, leaving exactly that order + * (and letting reorg steal `i++` into the `bne $s5,$v0` delay slot). + * + * 4. MEM_IN_STRUCT_P ASYMMETRY unblocks the entry-block schedule. sched.c's + * true_dependence() drops a store->load dependence only when the LOAD is + * MEM_IN_STRUCT_P at a varying address and the STORE is neither. Reading the + * trail count through a struct type (Cnt_8018E8A0) and writing prim.c[] through + * plain `*(u32 *)` casts is what lets the loop-bound `lh 0x108($s3)` float up + * past the colour stores; `prim.code` deliberately STAYS a struct member so it + * still pins the lh behind it. With the natural spelling the lh is chained + * behind all five sp stores and lands next to the blez with a nop. + * + * 5. THE $a0 PIN TRIO. `d = obj->x1C - i*4` is a global allocno; global.c marks + * REG_DEAD before the store, so the output freely reuses the dying $v1 and gcc + * picks `subu $v1,$v0,$v1`. Pinning d alone makes gcc fold the load into $a0 + * (`lw $a0` / `subu $a0,$a0,$v0`); pinning only the operands leaves d on $v1. + * All three pins together are the minimum that reproduces + * `lw $v0` / `sll $v1` / `subu $a0,$v0,$v1`. + */ + +/* ---- integration surface (§161c), checked against + * src/ov_SC02_011/ov_SC02_011_jr_8017AE2C.c (func_8018E8A0 is the + * INCLUDE_ASM at line 9393): + * func_8012C218(void *a0) TU:4256 verbatim (also TU:4984 (void*)) + * func_8004787C(s32 a0) TU:2197 verbatim (also TU:6679) + * func_80047948(s32 a0) TU:2196 verbatim (also TU:6678) + * ratan2(s32, s32) TU:243 verbatim (TU:519 same types) + * func_80017758(void *, void *) TU:1755 verbatim (also TU:6125) + * func_80049CAC(s32, s32) TU:3113 verbatim -> call site casts, + * exactly the TU's own idiom at TU:3143 + * RotTransSV(void *, void *, void *) TU:3116 verbatim + * func_8012B414(int a0) NOT declared in this TU; this is the + * fleet-canonical form (every other TU + * uses `int`), so the call site casts. + * + * The GTE macros are named SRM_/STM_8018E8A0, NOT gte_SetRotMatrix / + * gte_SetTransMatrix: the host TU already defines those two names twice + * (TU:9781 and TU:10513) and a third definition would collide. + * + * DATA (family remap must re-point these per overlay): D_801DF7F8 (u32[8] + * colour table) and D_801DF85C (u8[], SVECTOR quad source) have NO existing + * declaration anywhere in src/ or include/. + */ +extern void func_8012B414(int a0); +extern void func_8012C218(void *a0); +extern s32 func_8004787C(s32 a0); +extern s32 func_80047948(s32 a0); +extern s32 ratan2(s32 a0, s32 a1); +extern void func_80049CAC(s32 a0, s32 a1); +extern void RotTransSV(void *a0, void *a1, void *a2); +extern s32 func_80017758(void *a0, void *a1); + +extern u32 D_801DF7F8[]; +extern u8 D_801DF85C[]; + +#define SRM_8018E8A0(r0) __asm__ volatile ( \ + "lw $12, 0( %0 );" \ + "lw $13, 4( %0 );" \ + "ctc2 $12, $0;" \ + "ctc2 $13, $1;" \ + "lw $12, 8( %0 );" \ + "lw $13, 12( %0 );" \ + "lw $14, 16( %0 );" \ + "ctc2 $12, $2;" \ + "ctc2 $13, $3;" \ + "ctc2 $14, $4" \ + : \ + : "r"( r0 ) \ + : "$12", "$13", "$14" ) +#define STM_8018E8A0(r0) __asm__ volatile ( \ + "lw $12, 20( %0 );" \ + "lw $13, 24( %0 );" \ + "ctc2 $12, $5;" \ + "lw $14, 28( %0 );" \ + "ctc2 $13, $6;" \ + "ctc2 $14, $7" \ + : \ + : "r"( r0 ) \ + : "$12", "$13", "$14" ) + +typedef struct { s16 vx, vy, vz, pad; } SVec8_8018E8A0; /* 0x08 */ +typedef struct { s32 vx, vy, vz, pad; } Vec16_8018E8A0; /* 0x10 */ +typedef struct { s16 m[3][3]; s16 pad; s32 t[3]; } Mtx_8018E8A0; /* 0x20 */ +typedef struct { s16 f[0x86]; } Cnt_8018E8A0; +typedef struct { + SVec8_8018E8A0 v[4]; /* 0x00 */ + u32 c[4]; /* 0x20 */ + s32 code; /* 0x30 */ + s32 pad; /* 0x34 */ +} Prim_8018E8A0; /* 0x38 */ + +void func_8018E8A0(void *a0) +{ + Vec16_8018E8A0 mv; /* sp+0x10 */ + Prim_8018E8A0 prim; /* sp+0x20 */ + Mtx_8018E8A0 mtx; /* sp+0x58 */ + SVec8_8018E8A0 rot; /* sp+0x78 */ + s32 flag; /* sp+0x80 */ + s32 *p; + s32 *q; + s32 *zb; + SVec8_8018E8A0 *src; + SVec8_8018E8A0 *dst; + register s32 d __asm__("$4"); + s32 i, j, k, base, ang, t; + u32 col; + + *(u16 *)(*(s32 *)((s32)a0 + 0x20) + 0x14) += 0x40; + func_8012B414((s32)a0); + + mv.vz = -0x40000; + rot.vz = 0; + col = D_801DF7F8[(*(u16 *)((s32)a0 + 0x10A))++ & 7]; + prim.code = 0x50000000; + *(u32 *)((s32)&prim + 0x20) = col; + *(u32 *)((s32)&prim + 0x24) = col; + *(u32 *)((s32)&prim + 0x28) = col; + *(u32 *)((s32)&prim + 0x2C) = col; + + p = (s32 *)((s32)a0 + 0xCC); + for (i = 0, k = 0; i < ((Cnt_8018E8A0 *)a0)->f[0x84]; i++, k++, p += 3) { + zb = (s32 *)((s32)a0 + 0xD4); + { register s32 c1 __asm__("$2"); register s32 c2 __asm__("$3"); + c1 = *(s32 *)((s32)a0 + 0x1C); c2 = i * 4; d = c1 - c2; } + if (d > 0x40) { + if (i == 4) { + func_8012C218(a0); + return; + } + continue; + } + base = d << 6; + ang = base & 0xFFF; + flag = (func_8004787C(ang) << 4) >> 12; + + switch (*(s16 *)((s32)a0 + 0x70)) { + case 0: + t = ang; + goto horiz; + case 1: + t = base + 0x800; + t &= 0xFC0; + horiz: + mv.vy = (func_80047948(t) << 4) * flag; + mv.vx = 0; + rot.vx = -ratan2(mv.vy, mv.vz); + rot.vy = 0; + break; + case 2: + t = ang; + goto vert; + case 3: + t = base + 0x800; + t &= 0xFC0; + vert: + mv.vx = (func_80047948(t) << 4) * flag; + mv.vy = 0; + rot.vx = 0; + rot.vy = ratan2(-mv.vz, mv.vx) - 0x400; + break; + } + ((void (*)(void *, void *))func_80049CAC)(&rot, &mtx); + + q = zb + k * 3; + p[0] += mv.vx; + q[-1] += mv.vy; + q[0] += mv.vz; + mtx.t[0] = ((s16 *)p)[1]; + mtx.t[1] = ((s16 *)q)[-1]; + mtx.t[2] = ((s16 *)q)[1]; + + src = (SVec8_8018E8A0 *)(D_801DF85C + + ((s32)(s16)*(u16 *)((s32)a0 + 0x70) / 2) * 0x20); + dst = prim.v; + + SRM_8018E8A0(&mtx); + STM_8018E8A0(&mtx); + + for (j = 0; j < 4; j++, src++, dst++) { + RotTransSV(src, dst, &flag); + } + func_80017758(&prim, (void *)(*(s32 *)((s32)a0 + 0x20) + 0x34)); + } + + *(s32 *)((s32)a0 + 0x1C) += 1; + if ((*(s32 *)((s32)a0 + 0x1C) & 3) == 0) { + if (((Cnt_8018E8A0 *)a0)->f[0x84] < 5) { + ((Cnt_8018E8A0 *)a0)->f[0x84] = ((Cnt_8018E8A0 *)a0)->f[0x84] + 1; + } + } +} + extern void (*D_801DF89C[])(void); diff --git a/src/ov_SC02_015/ov_SC02_015_jr_8017AE2C.c b/src/ov_SC02_015/ov_SC02_015_jr_8017AE2C.c index 788f38ec0..fcb7bb22e 100644 --- a/src/ov_SC02_015/ov_SC02_015_jr_8017AE2C.c +++ b/src/ov_SC02_015/ov_SC02_015_jr_8017AE2C.c @@ -4016,7 +4016,64 @@ void func_8017D500(void *a0) { INCLUDE_ASM("asm/ov_SC02_015/nonmatchings/ov_SC02_015_jr_8017AE2C", func_8017D53C); -INCLUDE_ASM("asm/ov_SC02_015/nonmatchings/ov_SC02_015_jr_8017AE2C", func_8017D5C8); +#include "common.h" + +/* Local address-suffixed clones of the PSX MATRIX/SVECTOR layouts (cookbook: match_one's isolated + * compile only has -Iinclude, so "../shared/engine_core.h" can't resolve from its scratch dir -- + * the host TU (src/ov_SC02_015/ov_SC02_015_jr_8017AE2C.c) already includes engine_core.h and + * therefore already has the real MATRIX/SVECTOR in scope; these local names exist ONLY to let this + * file compile standalone under match_one and carry zero risk of colliding with the host's globals + * at integration time). Layout: m[3][3] (18B) + 2B pad + t[3] s32 (12B) = 0x20; vx/vy/vz/pad s16 = 8B. */ +typedef struct { s16 m[3][3]; s32 t[3]; } MATRIX_8017D5C8; +typedef struct { s16 vx, vy, vz, pad; } SVECTOR_8017D5C8; + +extern s32 D_80126B58; +extern s16 D_80181CEC[]; +extern u16 func_80148800(s32 *a0); +extern s32 func_80012C6C(s32 a0, s32 a1, s32 a2); +extern s32 func_80012ABC(s32 a0, s32 a1, s32 a2); +extern void func_80049CAC(s32 a0, s32 a1); +extern void func_8012F14C(s32 a0, s32 a1, s32 a2); + +void func_8017D5C8(s32 param_1, s16 *param_2) { + MATRIX_8017D5C8 m1; + SVECTOR_8017D5C8 svec_in; + SVECTOR_8017D5C8 svec_out; + u8 t; + + if (func_80148800(&D_80126B58) & 3) { + t = (*(u8 *)(param_1 + 5) + 1) & 1; + *(u8 *)(param_1 + 5) = t; + *(s32 *)(param_1 + 0x14) = D_80181CEC[t]; + } + + *(s32 *)(param_1 + 0x8) = (s16)func_80012C6C((s32)*(s16 *)(param_1 + 0x8), (s32)*(s16 *)(param_1 + 0xC), 4); + *(s32 *)(param_1 + 0x10) = (s16)func_80012C6C((s32)*(s16 *)(param_1 + 0x10), (s32)*(s16 *)(param_1 + 0x14), 4); + *(s16 *)(param_1 + 0x18) = func_80012ABC((s32)*(s16 *)(param_1 + 0x18), (s32)*(s16 *)(param_1 + 0x20), 4); + *(s16 *)(param_1 + 0x1A) = func_80012ABC((s32)*(s16 *)(param_1 + 0x1A), (s32)*(s16 *)(param_1 + 0x22), 4); + *(s16 *)(param_1 + 0x1C) = func_80012ABC((s32)*(s16 *)(param_1 + 0x1C), (s32)*(s16 *)(param_1 + 0x24), 4); + *(s16 *)(param_1 + 0x28) = func_80012C6C((s32)*(s16 *)(param_1 + 0x28), (s32)*(s16 *)(param_1 + 0x2E), 0x10); + *(s16 *)(param_1 + 0x2A) = func_80012C6C((s32)*(s16 *)(param_1 + 0x2A), (s32)*(s16 *)(param_1 + 0x30), 0x10); + *(s16 *)(param_1 + 0x2C) = func_80012C6C((s32)*(s16 *)(param_1 + 0x2C), (s32)*(s16 *)(param_1 + 0x32), 0x10); + + *(s32 *)(param_1 + 0x48) = (s32)*(s16 *)(param_1 + 0x28) + (s32)param_2[0]; + *(s32 *)(param_1 + 0x4C) = (s32)*(s16 *)(param_1 + 0x2A) + (s32)param_2[1]; + *(s32 *)(param_1 + 0x50) = (s32)*(s16 *)(param_1 + 0x2C) + (s32)param_2[2]; + func_80049CAC(param_1 + 0x18, (s32)&m1); + + m1.t[0] = *(s16 *)(param_1 + 0x28) + param_2[0]; + m1.t[1] = *(s16 *)(param_1 + 0x2A) + param_2[1]; + m1.t[2] = *(s16 *)(param_1 + 0x2C) + param_2[2]; + svec_in.vx = 0; + svec_in.vy = 0; + svec_in.vz = *(s32 *)(param_1 + 0x10); + ((void (*)(s32, s32, s32))func_8012F14C)((s32)&m1, (s32)&svec_in, (s32)&svec_out); + + *(s32 *)(param_1 + 0x3C) = (s32)svec_out.vx; + *(s32 *)(param_1 + 0x40) = (s32)svec_out.vy; + *(s32 *)(param_1 + 0x44) = (s32)svec_out.vz; +} + extern void (*D_80181D80[])(void); diff --git a/src/ov_SC02_017/ov_SC02_017_jr_8017DF34.c b/src/ov_SC02_017/ov_SC02_017_jr_8017DF34.c index 996eeceb4..6fe660c78 100644 --- a/src/ov_SC02_017/ov_SC02_017_jr_8017DF34.c +++ b/src/ov_SC02_017/ov_SC02_017_jr_8017DF34.c @@ -4273,7 +4273,51 @@ void func_8018501C(s32 a0) } -INCLUDE_ASM("asm/ov_SC02_017/nonmatchings/ov_SC02_017_jr_8017DF34", func_80185064); +#include "common.h" + +extern s32 func_8012B8E4(s32 arg0, s32 arg1); +extern s32 func_8012BEE8(s32 a0); +extern s32 func_8012BD3C(s32 a0, s32 a1, s32 a2); +extern void func_80185AF0(s32 a0); +extern void func_801859C4(s32 a0); +extern void func_80185B5C(void); +extern void func_80185CA4(s32 a0); + +void func_80185064(s32 a0) { + s32 v; + s32 flags; + + v = func_8012B8E4(a0, 0xA); + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x12) += v; + if ((*(s32 *)(a0 + 0xDC) & 0x20) != 0) { + if ((u32)(*(s32 *)(a0 + 0x1C) - 10) < 0xD) { + func_80185AF0(a0); + } + } else { + if (*(s32 *)(a0 + 0x1C) == 0x16) { + func_801859C4(a0); + } + if (*(s32 *)(a0 + 0x1C) < 0x10) { + ((void (*)(s32))func_80185B5C)(a0); + } + } + if (func_8012BEE8(a0) != 0) { + flags = *(s32 *)(a0 + 0xDC); + if (flags & 1) { + *(s32 *)(a0 + 0xDC) = flags & 0xFFFFFFFE; + *(s16 *)(a0 + 2) = 3; + } else if (flags & 0x10) { + *(s32 *)(a0 + 0xDC) = flags & 0xFFFFFFEF; + *(s16 *)(a0 + 2) = 3; + } else if (func_8012BD3C(a0, 0x400, 0x4000) == 0) { + *(s16 *)(a0 + 2) = 5; + } else { + *(s32 *)(a0 + 0x1C) = 0x1E; + func_80185CA4(a0); + } + } +} + #include "common.h" diff --git a/src/ov_SC02_027/ov_SC02_027_jr_8017D898.c b/src/ov_SC02_027/ov_SC02_027_jr_8017D898.c index b367832ef..31bb0d441 100644 --- a/src/ov_SC02_027/ov_SC02_027_jr_8017D898.c +++ b/src/ov_SC02_027/ov_SC02_027_jr_8017D898.c @@ -5223,7 +5223,171 @@ INCLUDE_ASM("asm/ov_SC02_027/nonmatchings/ov_SC02_027_jr_8017D898", func_801870F INCLUDE_ASM("asm/ov_SC02_027/nonmatchings/ov_SC02_027_jr_8017D898", func_80187288); -INCLUDE_ASM("asm/ov_SC02_027/nonmatchings/ov_SC02_027_jr_8017D898", func_80187400); +#include "common.h" + +/* func_80187400 — ov_SC02_027 / ov_SC02_027_jr_8017D898.c MATCH (249 ins) + * + * Per-frame tick for a 4-state actor: first a proximity sweep over the live entity + * list (retarget every type-0x16F entity within 0x1900 to state 0x1D), then a + * switch on the state byte at +0xC2. case 0 FALLS THROUGH into case 1 (that is + * why the `*(u8*)(a0+0xC2) = 1` store is unconditional and why case 0's block sits + * physically above the shared tail); the `func_8012C218` bail of case 0/1 and of + * case 2 are textually identical, so jump.c cross-jumps them into one copy at + * .L801876C4 (the LOWER site is the one redirected — §162 cross-jump direction). + * + * Declaration provenance (§161c — every decl checked against the whole host TU): + * func_8012CEB0(s32,s32,s32) — TU col-0 decl at L5276 (and fn-scope L4404): identical + * func_80143B6C(s32,s32) — TU col-0 decl at L4482: identical + * func_8012C218(void *) — TU col-0 decls at L2665/L3877/L4481: identical + * func_8012DE2C / func_8012DDA4 / func_80013350 — NOT in the TU; forms are byte-copies + * of the canonical set in src/shared/engine_core.h + * (L24522/24523/24524, DEFINE_func_8012DBD0) + * func_8012CC64 / func_8012CBF4 — NOT in the TU; canonical returns are `void` + * (engine_core.h L5361 / L8375) and the asm USES $v0, so + * the read is a call-site cast (codegen-neutral, the + * codebase's own idiom — cf. func_80131340). + * D_801DA780 — real dlabel, asm/ov_SC02_027/data/tail18.data.s:5345 + * (0x801DA780, 8 bytes); declared nowhere else in src/. + * The TU instantiates NO DEFINE_ macro, so none of the above can be redefined behind us. + * + * Codegen notes (each closed a residual — do not "clean up"): + * - sp10/18/20/28 are four 8-byte, align-2 vectors. align(2) < 4 is what makes the + * aggregate copies unaligned lwl/lwr + swl/swr block moves, and `sp28 = sp20` is a + * DEAD copy that gcc-2.7.2 KEEPS (no aggregate DSE) — load-bearing, not dead code. + * Byte-proven twin: func_80131340 in ov_SC02_027_jr_8012ACE0.c. + * - The 0x1D constant is hoisted to $s2 in the loop preheader by loop.c because it is + * used TWICE in the loop (the 0x5E compare and the 0x5E store); 0x16F/0xA are used + * once each and stay inside. Writing the literal twice is what produces that. + * - Two scheduling levers, both alias-analysis (sched.c true_dependence), see inline. + */ + +extern s32 func_8012DE2C(s32 a0); +extern s32 func_8012DDA4(void); +extern void func_80013350(s32 a0, void *a1); +extern s32 func_8012CEB0(s32 a0, s32 a1, s32 a2); +extern void func_8012CC64(s32 a0, s32 a1); +extern void func_8012CBF4(s32 a0); +extern s32 func_80143B6C(s32 a0, s32 a1); +extern void func_8012C218(void *a0); + +extern s32 D_801DA780; + +void func_80187400(s32 a0) +{ + /* 8-byte, align-2 vector — align < 4 is what makes the aggregate copies + * unaligned lwl/lwr + swl/swr block moves (§ sibling func_80131340). */ + struct V8 { + u16 vx, vy, vz, pad; + }; + struct Cnt { + s32 c; + }; + + struct V8 sp10; + struct V8 sp18; + struct V8 sp20; + struct V8 sp28; + s32 p; + + sp10.vz = 0; + sp10.vx = 0; + sp10.vy = 0x30; + + p = func_8012DE2C(a0); + while (p != 0) { + if (*(u16 *)p == 0x16F && (*(u16 *)(p + 0x5C) & 0x8000) != 0 && + *(u16 *)(p + 0x5E) != 0x1D && + ((s32 (*)(s32, s32))func_80013350)(a0 + 4, p + 4) < 0x1900) { + *(u16 *)(p + 0x60) = 0xA; + *(u16 *)(p + 0x5C) |= 1; + /* The 0x5E store must be SOURCE-ORDERED after the a0+0x20 loads: the + * anti-dependence (different base regs -> memrefs_conflict_p cannot + * disambiguate) pins it below them, so the scheduler can only place it + * in the lhu's load-delay stall. Written before them it floats up into + * the lhu 0x5C shadow instead and costs a nop. */ + *(u16 *)(p + 0x62) = *(u16 *)(*(s32 *)(a0 + 0x20) + 0x12) + 0x800; + *(u16 *)(p + 0x5E) = 0x1D; + } + p = func_8012DDA4(); + } + + switch (*(u8 *)(a0 + 0xC2)) { + case 0: + *(u8 *)(a0 + 0xC2) = 1; + sp18.vx = *(u16 *)(a0 + 0x3A); + sp18.vy = *(u16 *)(a0 + 0x3E); + sp18.vz = *(u16 *)(a0 + 0x42); + sp20 = sp18; + sp20.vx += sp10.vx; + sp20.vy += sp10.vy; + sp20.vz += sp10.vz; + sp28 = sp20; /* load-bearing dead aggregate copy — no aggregate DSE in 2.7.2 */ + func_8012CEB0((s32)&sp18, (s32)&sp20, 1); + sp20.vx -= sp10.vx; + sp20.vy -= sp10.vy; + sp20.vz -= sp10.vz; + *(u16 *)(a0 + 0x3A) = sp20.vx; + *(u16 *)(a0 + 0x3E) = sp20.vy; + *(u16 *)(a0 + 0x42) = sp20.vz; + /* fallthrough */ + case 1: + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x10) += *(u16 *)(a0 + 0xFE); + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x14) += *(u16 *)(a0 + 0x100); + *(s32 *)(a0 + 0x1C) += 1; + if (*(s32 *)(a0 + 0x1C) >= 0x29) { + func_8012C218((void *)a0); + return; + } + D_801DA780 = ((s32 (*)(s32, s32))func_8012CC64)(a0, (s32)&sp10); + if (D_801DA780 & 0x2000) { + *(u8 *)(a0 + 0xC2) = 2; + func_80143B6C(a0, 1); + *(s32 *)(a0 + 0x14) = 0xFFF30000; + *(s32 *)(a0 + 0x1C) = 0; + } + break; + + case 2: + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x10) += *(u16 *)(a0 + 0xFE); + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x14) += *(u16 *)(a0 + 0x100); + *(s32 *)(a0 + 0x1C) += 1; + if (*(s32 *)(a0 + 0x1C) >= 0x29) { + func_8012C218((void *)a0); + return; + } + D_801DA780 = ((s32 (*)(s32, s32))func_8012CC64)(a0, (s32)&sp10); + if (D_801DA780 & 0x2000) { + func_80143B6C(a0, 1); + *(u8 *)(a0 + 0xC2) = 3; + *(s32 *)(a0 + 0x1C) = 0; + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x14) = 0; + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x10) &= 0xFFF; + if (*(s16 *)(*(s32 *)(a0 + 0x20) + 0x10) < 0x800) { + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x10) = 0x400; + } else { + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x10) = 0xC00; + } + } + break; + + case 3: + *(s32 *)(a0 + 0x10) = *(s32 *)(a0 + 0x10) * 15 / 16; + *(s32 *)(a0 + 0x18) = *(s32 *)(a0 + 0x18) * 15 / 16; + D_801DA780 = ((s32 (*)(s32))func_8012CBF4)(a0); + /* MEM_IN_STRUCT_P lever (sched.c:837 true_dependence escape): a struct-member + * ref at a VARYING address never conflicts with a non-struct ref at a FIXED + * address, so this load may hoist above the D_801DA780 store and fill the jal + * shadow. Written as `*(s32 *)(a0 + 0x1C)` the ref is a plain INDIRECT_REF, + * the escape does not fire, and the load stalls one nop below the store. */ + ((struct Cnt *)(a0 + 0x1C))->c += 1; + if (((struct Cnt *)(a0 + 0x1C))->c >= 0x11) { + *(u16 *)(a0 + 0x2) = 3; + *(s32 *)(a0 + 0x1C) = 0x1E; + } + break; + } +} + extern void (*D_801B45BC[])(void); diff --git a/src/ov_SC02_041/ov_SC02_041_jr_8017BEBC.c b/src/ov_SC02_041/ov_SC02_041_jr_8017BEBC.c index b4c488fc5..d63c90606 100644 --- a/src/ov_SC02_041/ov_SC02_041_jr_8017BEBC.c +++ b/src/ov_SC02_041/ov_SC02_041_jr_8017BEBC.c @@ -4908,7 +4908,94 @@ INCLUDE_ASM("asm/ov_SC02_041/nonmatchings/ov_SC02_041_jr_8017BEBC", func_80180A5 INCLUDE_ASM("asm/ov_SC02_041/nonmatchings/ov_SC02_041_jr_8017BEBC", func_80180ACC); -INCLUDE_ASM("asm/ov_SC02_041/nonmatchings/ov_SC02_041_jr_8017BEBC", func_80180C40); +#include "common.h" + +extern s32 func_8014CB7C(void); +extern s32 func_8012BE54(s32 a0); +extern s32 func_8012B8A4(s32 a0); +extern void func_8012B2CC(s32 a0); +extern void func_8012B1B4(s32 a0, void *a1); +extern void func_8012A828(s32 a0, void *a1); +extern void func_8001C214(s32 a0, void *a1); +extern void func_8002D4C8(s32 a0, s32 a1); +extern s32 func_8012C658(s32 arg0, s32 arg1, s32 arg2); +extern s32 func_8012D624(s32 a0, s32 a1, s32 a2); +extern int rand(void); + +extern u16 D_80126B62; +extern s16 D_80126B96; +extern u8 D_801ABF64; +extern u8 D_801ABF7C; +extern u8 D_801AC734; + +void func_80180C40(s32 a0) { + s32 i; + s32 e; + s32 lim; + s16 d; + + if (func_8014CB7C() != 0) { + if (func_8012BE54(a0) < 0x1000) { + lim = 0x40; + if (*(s16 *)(a0 + 0x70) != 0) { + lim = 0x80; + } + d = *(u16 *)(a0 + 0xA) - D_80126B62; + if ((d >= 0) && (d <= lim)) { + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x12) = func_8012B8A4(a0); + *(u16 *)(a0 + 0xFC) = (rand() & 0xFF) - 0x80; + *(u16 *)(a0 + 0xFE) = (rand() & 0xFF) - 0x80; + *(u16 *)(a0 + 0x100) = (rand() & 0xFF) - 0x80; + func_8012B2CC(a0); + func_8012B1B4(a0, &D_801ABF64); + *(s16 *)(a0 + 0x16) = -0x18; + *(s32 *)(a0 + 0x1C) = 0x3C; + *(s16 *)(a0 + 0x5C) = 0; + *(u16 *)(a0 + 0x2) = *(u16 *)(a0 + 0x2) + 1; + if (*(s16 *)(a0 + 0x70) != 0) { + func_8012A828(a0, &D_801ABF7C); + *(u16 *)(a0 + 0xA) = *(u16 *)(a0 + 0xA) - 0x40; + func_8002D4C8(0xB41, 0); + } else { + func_8001C214(*(s32 *)(a0 + 0x20), &D_801AC734); + i = 0; + do { + e = func_8012C658(0x244, 3, a0); + if (e != 0) { + s32 r = rand(); + s32 b = *(u16 *)(*(s32 *)(*(s32 *)(e + 0x64) + 0x20) + 0x12) - 0x200; + *(u16 *)(*(s32 *)(e + 0x20) + 0x12) = b + (r & 0x3FF); + *(u16 *)(e + 0xFC) = (rand() & 0xFF) - 0x80; + *(u16 *)(e + 0xFE) = (rand() & 0xFF) - 0x80; + *(u16 *)(e + 0x100) = (rand() & 0xFF) - 0x80; + func_8012B2CC(e); + func_8012B1B4(e, &D_801ABF64); + *(s16 *)(e + 0x16) = -0x18; + *(s32 *)(e + 0x1C) = 0x3C; + func_8002D4C8(0xB42, 0); + } + i++; + } while (i < 3); + } + } + } + } else { + if (*(s16 *)(a0 + 0x70) != 0) { + if (*(u8 *)(a0 + 0x74) != 0) { + if (func_8012D624(a0, 0x100, 0xA) != 0) { + D_80126B96 = 0x4007; + } + } + } else { + *(u16 *)(a0 + 0xA) = *(u16 *)(a0 + 0xA) - 0x40; + if (func_8012D624(a0, 0x18, 0xA) != 0) { + D_80126B96 = 0x4007; + } + *(u16 *)(a0 + 0xA) = *(u16 *)(a0 + 0xA) + 0x40; + } + } +} + INCLUDE_ASM("asm/ov_SC02_041/nonmatchings/ov_SC02_041_jr_8017BEBC", func_80180ED4); @@ -5371,7 +5458,77 @@ INCLUDE_ASM("asm/ov_SC02_041/nonmatchings/ov_SC02_041_jr_8017BEBC", func_80181D3 INCLUDE_ASM("asm/ov_SC02_041/nonmatchings/ov_SC02_041_jr_8017BEBC", func_80181DD4); -INCLUDE_ASM("asm/ov_SC02_041/nonmatchings/ov_SC02_041_jr_8017BEBC", func_80181E18); +#include "common.h" + +extern s32 func_8012BCCC(s32 a0); +extern s32 func_8012BEE8(s32 a0); +extern s32 func_8012B608(s32 a0, s32 a1, s32 a2); +extern void func_8012B178(s32 a0, s32 a1); +extern void func_8012CBA4(s32 a0); +extern s32 func_8012B70C(s16 *a0, s16 *a1); +extern void func_8012ADE4(u8 *a0); +extern int rand(void); +extern u8 D_800D3918[]; +extern s16 D_801152B0; + +void func_80181E18(s32 a0) { + s16 sp10[4]; + s32 c; + + if (func_8012BCCC(a0) <= 0x24000) { + *(s16 *)(a0 + 2) = 1; + return; + } + if (*(s32 *)(a0 + 0x14) == 0) { + *(s32 *)(a0 + 0x14) = 0x100000; + } + if (*(u16 *)(a0 + 0x34) == 0) { + if (func_8012BEE8(a0)) { + s32 r; + s32 base; + s32 v; + *(s32 *)(a0 + 0x1C) = 0x1E; + *(s16 *)(a0 + 0x34) = *(u16 *)(a0 + 0x34) + 1; + r = rand() % 0x400; + base = *(s16 *)(*(s32 *)(a0 + 0x20) + 0x12); + if (rand() & 1) { + v = base + r; + } else { + v = base - r; + } + *(s32 *)(a0 + 0xE4) = v; + } + } else { + s32 d = func_8012B608(*(s16 *)(*(s32 *)(a0 + 0x20) + 0x12), + *(s32 *)(a0 + 0xE4), 8); + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x12) = + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x12) + d; + if (func_8012BEE8(a0)) { + *(s16 *)(a0 + 0x34) = 0; + *(s32 *)(a0 + 0x1C) = 0x5A; + } + } + + func_8012B178(a0, 0xFFFC0000); + c = ((s32 (*)(s32))func_8012CBA4)(a0); + if ((c & 0x8000) && *(s32 *)(a0 + 0xE8) == 0) { + s32 t1; + sp10[0] = *(s32 *)(a0 + 0x10) >> 8; + sp10[2] = *(s32 *)(a0 + 0x18) >> 8; + t1 = func_8012B70C((s16 *)D_800D3918, &D_801152B0); + *(s32 *)(a0 + 0xE4) = + (t1 * 2 - func_8012B70C(sp10, (s16 *)D_800D3918)) & 0xFFF; + *(s16 *)(a0 + 0x34) = 1; + *(s32 *)(a0 + 0x1C) = 0x1E; + *(s32 *)(a0 + 0xE8) = 0x10; + } else if ((c & 0x2000) == 0) { + func_8012ADE4((u8 *)a0); + } + if (*(s32 *)(a0 + 0xE8) != 0) { + *(s32 *)(a0 + 0xE8) = *(s32 *)(a0 + 0xE8) - 1; + } +} + INCLUDE_ASM("asm/ov_SC02_041/nonmatchings/ov_SC02_041_jr_8017BEBC", func_80181FFC); @@ -5541,7 +5698,160 @@ extern void func_8012A828(s32, void*); } -INCLUDE_ASM("asm/ov_SC02_041/nonmatchings/ov_SC02_041_jr_8017BEBC", func_80182784); +/* + * func_80182784 -- ov_SC02_041 / ov_SC02_041_jr_8017BEBC.c MATCH (124 ins) + * Family reach x4 (zero-crack sibling family exemplar). + * + * WHAT IT DOES + * Entity per-frame step for state 6 (spawn/attach of a sub-object): + * (1) state == 6 -> rotate the local offset (0,0,0x20) by the entity's object matrix + * (obj+0x34), fetch bone/joint 7's matrix via func_8012EC04, project it through + * func_8012F14C with the shared D_800D3918 offset vector, then build a 0x14-byte + * spawn record {pos.xyz, id=0x1FA, 0, 0, 0x7FFF, gfxid=obj->0x12, 0} and hand it to + * func_8012C51C; on success cache the child at +0xCC, swap the object's anim + * (func_8001C924 D_801AF8D0), push the entity to script D_801AD280 and set flag 1. + * (2) once flags 1|2 are both set and func_80013350(entity+4, child+4) < 0x1801 + * (distance gate), free the child, restore anim D_801AF898, push script D_801B5F44, + * state = 7, clear the child ptr and flags 0|1. + * (3) if the current script IS D_801B5F44 and entity flag 0x4000 at +0x72 is set, + * push script D_801B5BC4, sub-state = 2, func_8012B23C + func_80182F20. + * + * SIBLING-FIRST provenance (STEP 0 / cookbook 160g -- all of it from the DESTINATION TU): + * - the 0x14-byte spawn record IS `struct S80190C84` (engine_types.h:1337), lifted from + * the TU's own banked func_80184D20 (TU:5537-5566), which builds the same record + * (f6 = id, f8 = fA = 0, fC = 0x7FFF, fE = gfx, f10 = 0) and calls func_8012C51C(&sp, a0). + * - `s32 buf[8]` + `func_8012EC04(p, n, buf)` + `func_8012F14C(buf, , out)` is the + * TU's banked func_80184C60 (TU:6372-6383) verbatim, including the cast-call spelling + * that keeps the TU's `(s32,s32,s32)` prototype for func_8012F14C. + * - `ApplyMatrixSV((void *)(*(s32 *)(a0 + 0x20) + 0x34), v, v)` is engine_core.h + * DEFINE_func_80168070() (engine_core.h:21265) verbatim. + * - the `&D_801B5F44` script-pointer compare uses the TU's own `extern short D_801B5F44;` + * (TU:5537) -- taking its address, NOT re-spelling it as an array. + * + * IDIOMS THAT ARE LOAD-BEARING (byte-measured this session) + * 1. The 0xE0 flag gate MUST be three NESTED ifs, not `(f&1) && (f&2) && ...`. + * The C front end's fold_truthop rewrites the `&&` of two bit tests on the same + * operand into `(f & 3) == 3` -- one `andi 3` + `bne v0,v1` instead of the target's + * `andi 1 / beqz / andi 2 / beqz`. Nesting is not an && expression, so fold never + * sees it; cse still collapses the two `lw 0xE0` into one, which is what the target + * does. (-2 ins / 52 diffs in v1.) NEW LAW candidate -- see notes. + * 2. `sv[1] = 0; sv[0] = 0; sv[2] = 0x20;` -- the 1-before-0 order is real source order, + * not scheduling; 0,1,2 emits `sh 0x50` before `sh 0x52`. + * 3. `u16 out[4]` / `u16 sv[4]` (NOT s16): the target loads every component with `lhu`. + * s16 arrays give `lh` on all five reads. Same for `*(u16 *)(obj + 0x12)`. + * 4. Local declaration order struct -> buf -> out -> sv reproduces the frame exactly + * (0x10 / 0x28 / 0x48 / 0x50). The 4-byte hole at sp+0x24 is not a missing local: + * assign_stack_local gives every BLKmode local BIGGEST_ALIGNMENT (8) on MIPS, so the + * 0x14-byte struct rounds 0x24 -> 0x28 for buf. + * 5. `temp = func_8012C51C(...); *(s32*)(a0+0xCC) = temp; if (temp != 0)` -- the store + * lands in the beqz delay slot (cookbook 162: non-void return visible in delay slots). + * + * DECLARATION SURFACE (audited against the WHOLE destination TU; it expands ZERO DEFINE_ + * macros, so engine_core.h contributes no file-scope declaration): + * ApplyMatrixSV TU:170 `extern void ApplyMatrixSV(void*,void*,void*);` <- AGREES verbatim + * func_8012F14C TU:330 `extern void func_8012F14C(s32,s32,s32);` <- AGREES verbatim + * func_8012EC04 TU:6369 `extern void func_8012EC04(s32,s32,s32*);` <- AGREES verbatim + * func_8012C51C TU:6410 `extern s32 func_8012C51C(void*,s32);` <- AGREES verbatim + * func_8012C218 TU:5092 `extern void func_8012C218(void *a0);` <- AGREES verbatim + * func_8012A828 TU:4738 `extern void func_8012A828(s32 a0, void *a1);` <- AGREES verbatim + * func_80182F20 TU:5438 `extern s32 func_80182F20(s32 a0);` <- AGREES verbatim + * D_800D3918 TU:634 `extern u8 D_800D3918[];` <- AGREES verbatim + * D_801B5F44 TU:5537 `extern short D_801B5F44;` <- AGREES verbatim + * func_8012B23C not in the TU; fleet-dominant spelling (1576x) used verbatim. + * func_8001C924 not in the TU; fleet-dominant spelling (70x) used verbatim. + * func_80013350 not in the TU. The fleet-dominant spelling is `extern void + * func_80013350(s32, void*);` (1379x) but this call site NEEDS the s32 + * return (`slti $v0,$v0,0x1801`), so `extern s32 func_80013350(s32 a0, + * void *a1);` is used. No conflict in THIS TU -- but if a later splice + * ever drops a `void` spelling into it, keep this one and cast at the + * use site. + * D_801AF8D0 / D_801AD280 / D_801AF898 / D_801B5BC4 -- declared NOWHERE in the TU + * (D_801B5BC4 exists only in ov_SC05_001, a different overlay), so the + * block-scope spellings here are free. + * St_80182784 -- fresh typedef name, zero hits across src/ and include/. It is + * layout-identical to `struct S80190C84` (engine_types.h:1337), which IS + * visible in the TU via ../shared/engine_core.h; on banking it SHOULD be + * replaced by `struct S80190C84 s;` and the typedef dropped (byte-neutral + * -- the local tag exists only so the standalone match_one compile works, + * since engine_core.h is not on its -I path). + */ + +#include "common.h" + +/* Layout-identical to `struct S80190C84` (src/shared/engine_types.h:1337), which IS + * visible in the destination TU via ../shared/engine_core.h; a fresh tag is used here + * only so the standalone match_one compile (no engine_core.h on its -I path) works. */ +typedef struct { s16 f0, f2, f4, f6, f8, fA, fC, fE; s32 f10; } St_80182784; + +extern void ApplyMatrixSV(void *a0, void *a1, void *a2); +extern void func_8012EC04(s32 param_1, s32 param_2, s32 *param_3); +extern void func_8012F14C(s32 a0, s32 a1, s32 a2); +extern s32 func_8012C51C(void *a0, s32 a1); +extern void func_8012C218(void *a0); +extern void func_8012A828(s32 a0, void *a1); +extern void func_8012B23C(s32 a0); +extern s32 func_80013350(s32 a0, void *a1); +extern void func_8001C924(s32 a0, void *a1); +extern s32 func_80182F20(s32 a0); +extern u8 D_800D3918[]; +extern short D_801B5F44; + +void func_80182784(s32 a0) { + + extern u8 D_801AF8D0[]; + extern u8 D_801AD280[]; + extern u8 D_801AF898[]; + extern u8 D_801B5BC4[]; + St_80182784 s; + s32 buf[8]; + u16 out[4]; + u16 sv[4]; + s32 temp; + + if (*(s32 *)(a0 + 0x94) == 6) { + sv[1] = 0; + sv[0] = 0; + sv[2] = 0x20; + ApplyMatrixSV((void *)(*(s32 *)(a0 + 0x20) + 0x34), sv, sv); + func_8012EC04(a0, 7, buf); + ((void (*)(s32 *, u8 *, u16 *))func_8012F14C)(buf, D_800D3918, out); + s.f0 = out[0] + sv[0]; + s.f2 = out[1]; + s.f4 = out[2] + sv[2]; + s.f6 = 0x1FA; + s.f8 = 0; + s.fA = 0; + s.fC = 0x7FFF; + s.fE = *(u16 *)(*(s32 *)(a0 + 0x20) + 0x12); + s.f10 = 0; + temp = func_8012C51C(&s, a0); + *(s32 *)(a0 + 0xCC) = temp; + if (temp != 0) { + func_8001C924(*(s32 *)(a0 + 0x20), D_801AF8D0); + func_8012A828(a0, D_801AD280); + *(s32 *)(a0 + 0xE0) |= 1; + } + } + if (*(s32 *)(a0 + 0xE0) & 1) { + if (*(s32 *)(a0 + 0xE0) & 2) { + if (func_80013350(a0 + 4, (void *)(*(s32 *)(a0 + 0xCC) + 4)) < 0x1801) { + func_8012C218(*(void **)(a0 + 0xCC)); + func_8001C924(*(s32 *)(a0 + 0x20), D_801AF898); + func_8012A828(a0, &D_801B5F44); + *(s32 *)(a0 + 0x94) = 7; + *(s32 *)(a0 + 0xCC) = 0; + *(s32 *)(a0 + 0xE0) &= ~3; + } + } + } + if (*(s32 *)(a0 + 0x90) == (s32)&D_801B5F44 && (*(u16 *)(a0 + 0x72) & 0x4000)) { + func_8012A828(a0, D_801B5BC4); + *(s16 *)(a0 + 0x2) = 2; + func_8012B23C(a0); + func_80182F20(a0); + } +} + INCLUDE_ASM("asm/ov_SC02_041/nonmatchings/ov_SC02_041_jr_8017BEBC", func_80182974); @@ -5615,7 +5925,94 @@ void func_8018354C(void *a0) { } -INCLUDE_ASM("asm/ov_SC02_041/nonmatchings/ov_SC02_041_jr_8017BEBC", func_80183560); +#include "common.h" + +/* func_80183560 — ov_SC02_041, TU ov_SC02_041_jr_8017BEBC.c (132 ins). + * + * Family of 4 (zero-crack exemplar): func_80183560 (ov_SC02_041) / func_80186B84 (ov_SC04_002) / + * func_80185F54 (ov_SC04_005) / func_801825C0 (ov_SC04_007) — all four `asm/.../jr_8017BEBC`, + * all 141-line .s, all still INCLUDE_ASM. + * + * Per-frame step of an entity's ground marker. a0+0x64 is the owner entity; the two SVECTORs at + * +0xFC and +0x104 are copied from it (vx,vy,vz only — the +0x102/+0x10A pad words are NOT copied, + * which is why this is six scalar assignments and not two struct copies). The marker's own two + * SVECTORs at +0xEC and +0xF4 are then set to the midpoints of (+0xFC,+0xDC) and (+0x104,+0xE4); + * +0xEC -= +0xF4 turns the first into the edge vector, VectorNormalSS normalises it in place, and + * the result is divided by the speed taken from the 2-entry s16 table D_801AD38C (0x0074/0x003C, + * asm/ov_SC02_041/data/tail.data.s:46622) indexed by the low nibble of the u16 at +0x70, then added + * back onto +0xF4 to give the advanced position. +0x1C = 0x10 is the colour/intensity that + * func_801837B0 (same TU, already matched) reads, and the u16 at +0x2 is the state counter. + * + * DECLARATION SURFACE (whole-TU grep of ov_SC02_041_jr_8017BEBC.c, §161c): + * * VectorNormalSS — TU line 1848 `extern s32 VectorNormalSS(void *a0, void *a1);`. The decl + * below is CHARACTER-IDENTICAL, so it is a legal repeat, not a conflict. + * * D_801AD38C — not declared anywhere in ov_SC02_041; the TU's nearest neighbour is + * `extern void (*D_801AD37C[])(void);` at line 5604, a DIFFERENT symbol (the func_801834B4 + * dispatch table). `extern s16 D_801AD38C[]` is new and conflict-free. It must stay `s16`: + * the target reads it with `lh`, and an `s16` LOCAL would instead give lhu+sll+sra (see below). + * * func_80183560 itself has no prototype anywhere in the overlay — only the INCLUDE_ASM at + * line 5618 that this definition replaces. + * + * CODEGEN NOTES (what the .s pins — every one of these was byte-measured): + * * `d` MUST be s32, not s16. An `s16 d` is HImode: gcc loads the table entry with the movhi + * pattern (`lhu`) and then sign-extends at the division with `sll 16; sra 16` — two extra + * instructions and 134 ins total. An s32 local makes the load a plain + * `(sign_extend:SI (mem:HI ...))` = the target's single `lh`. + * * The owner pointer is re-read from a0+0x64 for EVERY one of the six copies (six `lw`s in the + * target). Caching it in a local collapses them to one load. + * * The reads are all `s16`: the `sra $v0,$v0,1` in the averaging block needs the signed sum, and + * the `lhu`s elsewhere are combine's own force_to_mode rewrite of a dead-high-half sign_extend + * — do NOT chase them with `u16` casts in the source. + * * >>> THE WHOLE FUNCTION TURNS ON `(*(s16 *)(a0 + 0x2))++;` <<< Spelled `+= 1` instead, the + * draft is 132/132 instructions with the SAME multiset but 25 mismatched: the three div + * quotients come out $a0/$a1/$a2 instead of $a1/$a2/$a3 and the tail block schedules + * add/store/add/store instead of add,add,add,store,store,store. `x += 1` on a memory lvalue + * expands to one read-modify-write chain; postincrement expands via an explicit temp + * (`t = *p; *p = t + 1;`), which is one more RTL insn on that chain. That extra insn changes + * the dependence depth the pre-RA scheduler ranks by, so the 0x2 load is hoisted above the + * `0x1C = 0x10` store, its value and the 0x10 constant become simultaneously live, and the + * resulting pressure pushes the whole tail onto the target's registers. `(*p)++` and + * `t = *p + 1; *p = t;` both match; `+= 1`, `*p = *p + 1` and `t = *p; *p = t + 1;` do not. + */ + +extern s32 VectorNormalSS(void *a0, void *a1); +extern s16 D_801AD38C[]; + +void func_80183560(s32 a0) { + s16 *v; + s32 d; + + *(s16 *)(a0 + 0xFC) = *(s16 *)(*(s32 *)(a0 + 0x64) + 0xFC); + *(s16 *)(a0 + 0xFE) = *(s16 *)(*(s32 *)(a0 + 0x64) + 0xFE); + *(s16 *)(a0 + 0x100) = *(s16 *)(*(s32 *)(a0 + 0x64) + 0x100); + *(s16 *)(a0 + 0x104) = *(s16 *)(*(s32 *)(a0 + 0x64) + 0x104); + *(s16 *)(a0 + 0x106) = *(s16 *)(*(s32 *)(a0 + 0x64) + 0x106); + *(s16 *)(a0 + 0x108) = *(s16 *)(*(s32 *)(a0 + 0x64) + 0x108); + + *(s16 *)(a0 + 0xEC) = (*(s16 *)(a0 + 0xFC) + *(s16 *)(a0 + 0xDC)) >> 1; + *(s16 *)(a0 + 0xEE) = (*(s16 *)(a0 + 0xFE) + *(s16 *)(a0 + 0xDE)) >> 1; + *(s16 *)(a0 + 0xF0) = (*(s16 *)(a0 + 0x100) + *(s16 *)(a0 + 0xE0)) >> 1; + *(s16 *)(a0 + 0xF4) = (*(s16 *)(a0 + 0x104) + *(s16 *)(a0 + 0xE4)) >> 1; + *(s16 *)(a0 + 0xF6) = (*(s16 *)(a0 + 0x106) + *(s16 *)(a0 + 0xE6)) >> 1; + *(s16 *)(a0 + 0xF8) = (*(s16 *)(a0 + 0x108) + *(s16 *)(a0 + 0xE8)) >> 1; + + *(s16 *)(a0 + 0xEC) = *(s16 *)(a0 + 0xEC) - *(s16 *)(a0 + 0xF4); + *(s16 *)(a0 + 0xEE) = *(s16 *)(a0 + 0xEE) - *(s16 *)(a0 + 0xF6); + *(s16 *)(a0 + 0xF0) = *(s16 *)(a0 + 0xF0) - *(s16 *)(a0 + 0xF8); + + v = (s16 *)(a0 + 0xEC); + VectorNormalSS(v, v); + + d = D_801AD38C[*(u16 *)(a0 + 0x70) & 0xF]; + + *(s32 *)(a0 + 0x1C) = 0x10; + (*(s16 *)(a0 + 0x2))++; + + *(s16 *)(a0 + 0xEC) = *(s16 *)(a0 + 0xF4) + *(s16 *)(a0 + 0xEC) / d; + *(s16 *)(a0 + 0xEE) = *(s16 *)(a0 + 0xF6) + *(s16 *)(a0 + 0xEE) / d; + *(s16 *)(a0 + 0xF0) = *(s16 *)(a0 + 0xF8) + *(s16 *)(a0 + 0xF0) / d; +} + extern void func_801837B0(s32 a0); extern s32 func_8012BEE8(s32 a0); @@ -6117,7 +6514,115 @@ void func_80183FE0(s32 a0, s32 a1, s32 a2) INCLUDE_ASM("asm/ov_SC02_041/nonmatchings/ov_SC02_041_jr_8017BEBC", func_8018428C); -INCLUDE_ASM("asm/ov_SC02_041/nonmatchings/ov_SC02_041_jr_8017BEBC", func_801843D8); +#include "common.h" + +typedef struct { u32 addr : 24; u32 len : 8; } PTag_801843D8; + +extern u8 *D_800A5E60; +extern u8 D_800A6610[]; +extern short D_800B9A02; + +#define gte_ldv3_801843D8(r0, r1, r2) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 0( %1 );" \ + "lwc2 $3, 4( %1 );" \ + "lwc2 $4, 0( %2 );" \ + "lwc2 $5, 4( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) ) + +#define gte_ldv0_801843D8(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_rtpt_801843D8() __asm__ volatile ("nop;nop;rtpt") +#define gte_rtps_801843D8() __asm__ volatile ("nop;nop;rtps") +#define gte_avsz4_801843D8() __asm__ volatile ("nop;nop;avsz4") + +#define gte_stsxy3_801843D8(r0, r1, r2) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 0( %1 );" \ + "swc2 $14, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsxy_801843D8(r0) __asm__ volatile ( \ + "swc2 $14, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stotz_801843D8(r0) __asm__ volatile ( \ + "swc2 $7, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stflg_801843D8(r0) __asm__ volatile ( \ + "cfc2 $12, $31;" \ + "nop;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +void func_801843D8(void *a0) +{ + u8 *pkt; + s32 flag1, flag2, otz; + + pkt = D_800A5E60; + D_800A5E60 = pkt + 0x24; + + *(u32 *)(pkt + 4) = *(u32 *)((u8 *)a0 + 0x20); + *(u32 *)(pkt + 0xC) = *(u32 *)((u8 *)a0 + 0x24); + *(u32 *)(pkt + 0x14) = *(u32 *)((u8 *)a0 + 0x28); + __asm__ volatile (""); + *(u32 *)(pkt + 0x1C) = *(u32 *)((u8 *)a0 + 0x2C); + pkt[3] = 8; + pkt[7] = 0x3A; + + gte_ldv3_801843D8(a0, (u8 *)a0 + 0x8, (u8 *)a0 + 0x10); + gte_rtpt_801843D8(); + gte_stflg_801843D8(&flag1); + gte_stsxy3_801843D8(pkt + 8, pkt + 0x10, pkt + 0x18); + + gte_ldv0_801843D8((u8 *)a0 + 0x18); + gte_rtps_801843D8(); + gte_stflg_801843D8(&flag2); + flag1 = flag1 | flag2; + gte_stsxy_801843D8(pkt + 0x20); + gte_avsz4_801843D8(); + gte_stotz_801843D8(&otz); + + if ((flag1 & 0xFFFFEFFF) == 0) { + u8 *pkt2; + u32 *otp; + s32 idx; + u8 *ot; + + ot = &D_800A6610[(*(u16 *)&D_800B9A02) << 14]; + idx = otz + 1; + if (idx >= 0x1000) idx = 0xFFF; + otp = (u32 *)((idx << 2) + (u32)ot); + + ((PTag_801843D8 *)pkt)->addr = ((PTag_801843D8 *)otp)->addr; + ((PTag_801843D8 *)otp)->addr = (u32)pkt; + + pkt2 = D_800A5E60; + D_800A5E60 = pkt2 + 8; + pkt2[3] = 1; + *(u32 *)(pkt2 + 4) = 0xE100002A; + + ((PTag_801843D8 *)pkt2)->addr = ((PTag_801843D8 *)otp)->addr; + ((PTag_801843D8 *)otp)->addr = (u32)pkt2; + } +} + extern void (*D_801AD3D4[])(void); diff --git a/src/ov_SC03_014/ov_SC03_014_jr_8017AE2C.c b/src/ov_SC03_014/ov_SC03_014_jr_8017AE2C.c index f15328948..6f63f556c 100644 --- a/src/ov_SC03_014/ov_SC03_014_jr_8017AE2C.c +++ b/src/ov_SC03_014/ov_SC03_014_jr_8017AE2C.c @@ -3832,7 +3832,93 @@ void func_8017CE54(void *a0) { } -INCLUDE_ASM("asm/ov_SC03_014/nonmatchings/ov_SC03_014_jr_8017AE2C", func_8017CE90); +#include "common.h" + +/* callee-set / sibling search (cookbook §160g): all callees resolved from the DESTINATION TU + * itself, src/ov_SC03_014/ov_SC03_014_jr_8017AE2C.c, whose immediately-preceding matched sibling + * func_8017C6A0 (same TU, same family) uses the identical: + * - func_801465C0 loose void(void) file-scope decl, fought via a zero-arg cast-at-callsite + * (§161c) -- `((void *(*)(void))func_801465C0)()`. + * - func_80146C3C(void) called on the failure path with the object pointer cast in. + * - the returned object (named s3 there, s1 here) has s16 fields at +0x18/+0x1A/+0x2C, matching + * this function's `*(s16*)(s1+0x18)`, `+0x1A`, `+0x2C` stores exactly. + * func_8001CD04(void*,void*) is defined in src/800.c (INCLUDE_ASM, shared engine) -- + * asm/nonmatchings/800/func_8001CD04.s shows it stores its 2nd arg into (a0+0x20), confirming a0 + * here is the newly-allocated object and a1 is an address it retains -- i.e. func_8001CD04(s1, s2) + * installs s2 (&D_801EA840) onto the object. + * The `lwl/lwr`+`swl/swr` pair at D_8018EF84[idx] / D_801EA840 is the §160a align-1 struct-copy + * idiom; D_8018EF84 is already `extern M2C_UNK D_8018EF84;` (address-only) in many ov_SC05_017 + * TUs, and the shared align-1 4-byte struct type `B4` (src/shared/engine_types.h:713) is already + * used repo-wide for this exact shape. The SAME address (&D_801EA840) also takes a plain ALIGNED + * `sw` of 0xFFFFFF on the other branch -- an explicit `(u32*)` cast forces that, while the direct + * `B4` struct assignment on the table-lookup branch keeps the align-1 `swl/swr` -- §160a's "read + * the move width off the target, not off the data's apparent type", both branches hitting the same + * lvalue. + * D_801EA840/844/845/846 are owned by asm/ov_SC03_014/data/tail19.data.s (a data-only splat unit, + * §160c) -- not bundled with this function's .s, so plain `extern` declarations are correct, not a + * definition. D_801EA840 is one `.word` (4 bytes); D_801EA844 and D_801EA845 are each one `.byte`; + * D_801EA846 opens a much larger multi-byte block in that data file, but this function only stores + * a zero to its FIRST byte, so a scalar `extern u8 D_801EA846;` is sufficient here. + * + * TYPE: match_one's standalone `common.h` does NOT reach src/shared/engine_types.h (only the real + * TU's `#include "../shared/engine_core.h"` does), so `B4` isn't visible here. Per the + * ov_SC06_032_jr_8018FCE8.c precedent ("cc1 errors on the redefinition" of an identical-name + * typedef), a fresh, function-scoped name is used instead -- collides with nothing; the integrator + * may DROP this local typedef and reuse the TU's own `B4` when banking into the host TU. + */ + +typedef struct { u8 d[4]; } Blk4_8017CE90; + +extern void func_801465C0(void); +extern void func_8001CD04(void *a0, void *a1); +extern void func_800233CC(void *a0, u16 a1); +extern void func_80146C3C(void); + +extern Blk4_8017CE90 D_8018EF84[]; +extern Blk4_8017CE90 D_801EA840; +extern u8 D_801EA844; +extern u8 D_801EA845; +extern u8 D_801EA846; + +void func_8017CE90(void *a0) +{ + void *s1; + Blk4_8017CE90 *s2; + s32 pad[10]; + + s1 = ((void *(*)(void))func_801465C0)(); + *(void **)((u8 *)a0 + 0x20) = s1; + + if (s1 != NULL) { + s2 = &D_801EA840; + func_8001CD04(s1, s2); + + *(s32 *)((u8 *)s1 + 4) |= 0x50000000; + func_800233CC(s2, 0x50); + + *(s32 *)((u8 *)a0 + 0x10) = 0; + if (*(s32 *)((u8 *)a0 + 0x30) & 0x80) { + *(u32 *)s2 = 0xFFFFFF; + *(s16 *)((u8 *)s1 + 0x1A) = 0x80; + *(s16 *)((u8 *)s1 + 0x18) = 0x80; + *(s32 *)((u8 *)a0 + 0x14) = 0x80; + } else { + *s2 = D_8018EF84[*(s32 *)((u8 *)a0 + 0x2C)]; + *(s32 *)((u8 *)a0 + 0x14) = 0x20; + } + + D_801EA844 = 0; + D_801EA845 = 0; + D_801EA846 = 0; + + *(s32 *)((u8 *)a0 + 0x30) = *(s32 *)((u8 *)a0 + 0x30) & 0x7F; + *(s16 *)((u8 *)s1 + 0x2C) = *(u16 *)((u8 *)a0 + 0xE) + 0x10; + *(u16 *)((u8 *)a0 + 0x2) = *(u16 *)((u8 *)a0 + 0x2) + 1; + } else { + ((void (*)(void *))func_80146C3C)(a0); + } +} + INCLUDE_ASM("asm/ov_SC03_014/nonmatchings/ov_SC03_014_jr_8017AE2C", func_8017CFBC); diff --git a/src/ov_SC03_014/ov_SC03_014_jr_801848E4.c b/src/ov_SC03_014/ov_SC03_014_jr_801848E4.c index 0b3be252e..9717685ad 100644 --- a/src/ov_SC03_014/ov_SC03_014_jr_801848E4.c +++ b/src/ov_SC03_014/ov_SC03_014_jr_801848E4.c @@ -2893,7 +2893,145 @@ s32 func_801859B0(void *a0) INCLUDE_ASM("asm/ov_SC03_014/nonmatchings/ov_SC03_014_jr_801848E4", func_80185A0C); -INCLUDE_ASM("asm/ov_SC03_014/nonmatchings/ov_SC03_014_jr_801848E4", func_80185B44); +#include "common.h" + +/* ---- integration surface ---------------------------------------------- + * Decls conform to the host TU src/ov_SC03_014/ov_SC03_014_jr_801848E4.c: + * L3535 extern s32 func_8012C354(s32 a0, s32 a1); + * L2524 extern s32 func_8012C658(s32 arg0, s32 arg1, s32 arg2); + * L2916 extern void func_8012A828(s32 a0, void *a1); + * L3500 extern void func_8001C214(int, int); + * func_8012B23C is only block-scope in the host TU (L4008, same signature); + * func_80143970 / func_8001CA1C are not declared there at all. + * None of the D_* globals are declared in the host TU. + * -------------------------------------------------------------------- */ +extern s32 func_8012C354(s32 a0, s32 a1); +extern s32 func_8012C658(s32 arg0, s32 arg1, s32 arg2); +extern void func_8012A828(s32 a0, void *a1); +extern void func_8012B23C(s32 a0); +extern s32 func_80143970(s32 a0); /* canonical: DEFINE_func_80143970() in src/shared/engine_core.h returns s32 */ +extern void func_8001CA1C(s32 a0, s32 a1); +extern void func_8001C214(int, int); + +extern u16 D_80190490[]; +extern u8 D_801904C4[]; +extern u8 D_801E2170[]; +extern u8 D_80190590[]; +extern u8 D_8019059C[]; +extern u8 D_801905A4[]; +extern u8 D_801905B4[]; +extern u8 D_801905E4[]; +extern u8 D_801905F0[]; +extern u8 D_80190720[]; +extern u8 D_80190780[]; +extern u8 D_801C97C8[]; +extern u8 D_80191838[]; +extern u8 D_801E732C[]; +extern u8 D_801E7B50[]; +extern u8 D_801D6ED0[]; +extern s32 D_801EAC80; + +void func_80185B44(s32 a0) { + u16 t; + s32 c; + + if (func_8012C354(a0, (s32)D_80190490) == 0) { + return; + } + + switch (*(s16 *)(a0 + 0x70)) { + case 0: + *(u8 *)(a0 + 0xC0) = 1; + *(s16 *)(a0 + 0x2) = 0xE; + func_8012A828(a0, D_801E2170); + func_8012B23C(a0); + func_80143970(a0); + *(s32 *)(a0 + 0xB4) = -0x2032; + *(s32 *)(a0 + 0xBC) = (s32)D_801904C4; + *(s16 *)(a0 + 0xAE) = -1; + *(u8 *)(a0 + 0x75) = 0; + *(u32 *)(a0 + 0xC4) |= 1; + t = D_80190490[0]; + D_801EAC80 = 0; + *(s16 *)(a0 + 0xFE) = 1; + *(u16 *)(a0 + 0x76) = t; + *(s32 *)(a0 + 0x6C) = func_8012C658(0x265, 7, a0); + break; + + case 1: + func_8001CA1C(*(s32 *)(a0 + 0x20), (s32)D_80190590); + func_8012A828(a0, D_8019059C); + *(s32 *)(a0 + 0x58) = (s32)D_801905A4 | 0x40000000 | 0x20000000; + *(u16 *)(a0 + 0x5C) = 0x8000; + *(u32 *)(*(s32 *)(a0 + 0x20) + 4) |= 0x50000000; + *(u32 *)(*(s32 *)(a0 + 0x20) + 4) |= 0x8000000; + *(u8 *)(a0 + 0xC0) = 1; + *(s16 *)(a0 + 0x2) = 3; + *(s32 *)(a0 + 0xBC) = (s32)D_801905B4; + *(u8 *)(a0 + 0x75) = 0; + *(s32 *)(a0 + 0xB4) = 0; + *(s32 *)(a0 + 0x1C) = 0x5A; + *(u32 *)(a0 + 0xC4) |= 2; + break; + + case 2: + func_8001CA1C(*(s32 *)(a0 + 0x20), (s32)D_801905E4); + func_8012A828(a0, D_801905F0); + *(u32 *)(*(s32 *)(a0 + 0x20) + 4) |= 0x50000000; + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x2C) |= 0x10; + *(s16 *)(*(s32 *)(a0 + 0x20) + 0x18) = 0x2000; + *(s16 *)(*(s32 *)(a0 + 0x20) + 0x1A) = 0x2000; + *(s16 *)(a0 + 0x2) = 4; + *(u8 *)(a0 + 0x75) = 0; + c = D_801EAC80; + *(s32 *)(a0 + 0x1C) = 0x1E; + D_801EAC80 = c + 1; + break; + + case 3: + func_8001CA1C(*(s32 *)(a0 + 0x20), (s32)D_80190720); + func_8012A828(a0, D_80190780); + *(u32 *)(*(s32 *)(a0 + 0x20) + 4) |= 0x50000000; + *(u32 *)(*(s32 *)(a0 + 0x20) + 4) |= 0x8000000; + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x2C) |= 0x10; + *(s16 *)(a0 + 0x2) = 5; + *(u8 *)(a0 + 0x75) = 0; + *(s32 *)(a0 + 0x1C) = 0x5A; + break; + + case 4: + func_8001C214(*(s32 *)(a0 + 0x20), (s32)D_801C97C8); + *(u32 *)(*(s32 *)(a0 + 0x20) + 4) |= 0x50000000; + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x2C) |= 0x10; + *(s16 *)(a0 + 0x2) = 7; + *(u8 *)(a0 + 0x75) = 0; + *(s32 *)(a0 + 0x1C) = 0x5A; + break; + + case 5: + *(s32 *)(a0 + 0x58) = (s32)D_80191838 | 0x40000000 | 0x20000000; + *(s16 *)(a0 + 0x5C) = 0x800; + *(u32 *)(*(s32 *)(a0 + 0x20) + 4) |= 0x80000000; + *(s16 *)(a0 + 0x2) = 0xD; + break; + + case 6: + func_8001C214(*(s32 *)(a0 + 0x20), (s32)D_801E732C); + func_8012A828(a0, D_801E7B50); + *(s16 *)(a0 + 0x5C) = 0x800; + *(s16 *)(a0 + 0x2) = 0xF; + func_8012B23C(a0); + break; + + case 7: + func_8001C214(*(s32 *)(a0 + 0x20), (s32)D_801D6ED0); + *(s16 *)(a0 + 0x5C) = 0; + *(s16 *)(a0 + 0x2) = 0x10; + func_8012B23C(a0); + break; + } +} + #include "common.h" @@ -3321,7 +3459,61 @@ INCLUDE_ASM("asm/ov_SC03_014/nonmatchings/ov_SC03_014_jr_801848E4", func_8018882 INCLUDE_ASM("asm/ov_SC03_014/nonmatchings/ov_SC03_014_jr_801848E4", func_801889B4); -INCLUDE_ASM("asm/ov_SC03_014/nonmatchings/ov_SC03_014_jr_801848E4", func_80188A80); +extern void func_8001CA1C(s32 a0, s32 a1); +extern u8 D_80191928[]; +extern s32 func_80143970(s32 a0); +extern void func_8012AD44(s32 *a0, s16 a1); +extern void func_8018A31C(s32 a0); +extern void func_80189104(void *a0); +extern void func_8018AD5C(void); +extern void func_8018AD80(s32 a0); +extern void func_8002D4C8(s32 a0, s32 a1); + +void func_80188A80(s32 a0) { + register s32 obj __asm__("$16"); /* $s0 */ + register s32 self __asm__("$17") = a0; /* $s1 */ + s32 t0; + s32 t1; + s32 t2; + + obj = *(s32 *)(self + 0x20); + func_8001CA1C(obj, (s32)D_80191928); + + *(u16 *)(self + 0xA) = *(u16 *)(self + 0xA) - 0xA0; + *(u16 *)(self + 0xE) = *(u16 *)(self + 0xE) - 0x10; + *(u16 *)(obj + 0x18) = 0x2C00; + *(u16 *)(obj + 0x1A) = 0x1C00; + *(u32 *)(obj + 4) = *(u32 *)(obj + 4) | 0x1000000; + + t0 = *(s32 *)(self + 0xCC); + *(u16 *)(t0) = 0; + *(s32 *)(self + 0xCC) = func_80143970(self); + + *(s32 *)(self + 0x14) = 0xFFF80000; + *(u16 *)(self + 0x5C) = 0; + *(s32 *)(self + 0x1C) = 8; + func_8012AD44((s32 *)self, 3); + + obj = *(s32 *)(self + 0x64); + t1 = *(s32 *)(obj + 0x20); + *(u32 *)(t1 + 4) = *(u32 *)(t1 + 4) | 0x80000000; + func_8018A31C(obj); + + *(u16 *)(obj + 0x5C) = 0; + func_8012AD44((s32 *)obj, 2); + + t2 = *(s32 *)(self + 0xD0); + if (t2 != 0) { + *(u16 *)(t2 + 0x2C) = 0; + func_80189104((void *)t2); + } + + func_8018AD5C(); + func_8018AD80(self + 4); + + func_8002D4C8(0x87B, 0); +} + INCLUDE_ASM("asm/ov_SC03_014/nonmatchings/ov_SC03_014_jr_801848E4", func_80188B94); diff --git a/src/ov_SC03_093/ov_SC03_093_jr_8017D898.c b/src/ov_SC03_093/ov_SC03_093_jr_8017D898.c index 8ef713fd9..36f3241a9 100644 --- a/src/ov_SC03_093/ov_SC03_093_jr_8017D898.c +++ b/src/ov_SC03_093/ov_SC03_093_jr_8017D898.c @@ -4599,7 +4599,84 @@ void func_80181308(s32 a0) { INCLUDE_ASM("asm/ov_SC03_093/nonmatchings/ov_SC03_093_jr_8017D898", func_801813E0); -INCLUDE_ASM("asm/ov_SC03_093/nonmatchings/ov_SC03_093_jr_8017D898", func_80181470); +#include "common.h" + +extern s32 func_8012C354(s32 a0, s32 a1); +extern void func_8012C1B8(void); +extern void func_8012CAE4(void *a0); +extern void func_8001CA1C(s32 a0, s32 a1); +extern s32 func_801820C0(s32 a0); +extern void func_80182144(void *a0); +extern s32 func_8012E504(s32 a0, s32 a1); + +extern u8 D_801A4564[]; +extern u8 D_801A453C[]; +extern u8 D_801A4638[]; +extern u8 D_801A4598[]; + +void func_80181470(s32 a0) { + s32 child; + s16 v2000; + u16 flags; + + func_8012C354(a0, (s32)D_801A4564); + + child = ((s32 (*)(void))func_8012C1B8)(); + if (child == 0) { + ((void (*)(s32))func_8012CAE4)(a0); + return; + } + + func_8001CA1C(child, (s32)D_801A453C); + + v2000 = 0x2000; + *(s32 *)(child + 0x80) = a0 + 0xFC; + *(u32 *)(child + 0x4) |= 0x50000000; + *(u16 *)(child + 0x2C) |= 0x90; + *(s16 *)(child + 0x1A) = v2000; + *(s16 *)(child + 0x18) = v2000; + + *(s32 *)(a0 + 0x20) = child; + *(s32 *)(a0 + 0xD4) = func_801820C0(a0); + + *(s16 *)(a0 + 0xAE) = v2000; + *(s16 *)(a0 + 0xFC) = 0x80; + *(s16 *)(a0 + 0xFE) = 0x80; + *(s16 *)(a0 + 0x100) = 0x80; + + *(s16 *)(a0 + 0x5C) = 0; + if ((*(u16 *)(a0 + 0x70) & 1) != 0) { + *(u8 *)(a0 + 0xC0) = 1; + *(s32 *)(a0 + 0xBC) = (s32)D_801A4638; + *(s32 *)(a0 + 0xB4) = -0x2003; + *(s32 *)(a0 + 0xC4) |= 1; + } + + flags = *(u16 *)(a0 + 0x70); + if ((flags & 0x4000) != 0) { + *(s16 *)(a0 + 0x2) = 5; + *(s16 *)(a0 + 0xFC) = 0; + *(s16 *)(a0 + 0xFE) = 0; + *(s16 *)(a0 + 0x100) = 0; + } else if ((flags & 0x8000) == 0) { + *(s16 *)(a0 + 0x2) = 3; + } else { + *(s16 *)(a0 + 0x2) = 1; + *(s16 *)(a0 + 0xFC) = 0; + *(s16 *)(a0 + 0xFE) = 0; + *(s16 *)(a0 + 0x100) = 0; + func_80182144((void *)*(s32 *)(a0 + 0xD4)); + } + + if (func_8012E504(a0, 0x1A2) != 0) { + return; + } + *(s32 *)(a0 + 0xE4) = 1; + *(s32 *)(a0 + 0xD0) = (s32)D_801A4598; + *(s32 *)(a0 + 0xCC) = (s32)D_801A4598; + *(s32 *)(a0 + 0xE0) = *(s32 *)D_801A4598; +} + INCLUDE_ASM("asm/ov_SC03_093/nonmatchings/ov_SC03_093_jr_8017D898", func_801815FC); diff --git a/src/ov_SC03_093/ov_SC03_093_jr_801825B8.c b/src/ov_SC03_093/ov_SC03_093_jr_801825B8.c index 30a117293..3e31f77ed 100644 --- a/src/ov_SC03_093/ov_SC03_093_jr_801825B8.c +++ b/src/ov_SC03_093/ov_SC03_093_jr_801825B8.c @@ -4735,7 +4735,109 @@ void func_80184938(s32 a0, s32 a1) { } -INCLUDE_ASM("asm/ov_SC03_093/nonmatchings/ov_SC03_093_jr_801825B8", func_80184F90); +#include "common.h" + +extern s32 func_8012DEB8(s32 a0, s32 a1, s32 a2); +extern s32 func_80133784(s32 a0, void *a1, s32 a2); +extern void func_80184938(s32 a0, s32 a1); + +extern u16 D_80126B96; + +typedef struct { s16 x, y, z, w; } V8_80184F90; + +extern V8_80184F90 D_801C5E68[]; +extern V8_80184F90 D_801C5E88[]; +extern V8_80184F90 D_801C5EA8[]; + +s32 func_80184F90(s32 a0) { + V8_80184F90 src; + V8_80184F90 dst; + s32 i; + s32 j; + + switch (*(s16 *)(a0 + 0xFC)) { + case 1: + src.z = 0x60; + dst.z = 0x80; + for (j = -0x50; j < 0x60; j += 0x20) { + src.y = j; + dst.y = j; + for (i = -0x50; i < 0x60; i += 0x20) { + src.x = i; + dst.x = i; + if (func_8012DEB8(a0, (s32)&src, (s32)&dst) == 1) { + D_80126B96 = 0x4004; + } + } + } + break; + + case 2: + src.z = -0x60; + dst.z = -0x80; + for (j = -0x50; j < 0x60; j += 0x20) { + src.y = j; + dst.y = j; + for (i = -0x50; i < 0x60; i += 0x20) { + src.x = i; + dst.x = i; + if (func_8012DEB8(a0, (s32)&src, (s32)&dst) == 1) { + D_80126B96 = 0x4004; + } + } + } + break; + + case 3: + src.x = 0x60; + dst.x = 0x80; + for (j = -0x50; j < 0x60; j += 0x20) { + src.y = j; + dst.y = j; + for (i = -0x50; i < 0x60; i += 0x20) { + src.z = i; + dst.z = i; + if (func_8012DEB8(a0, (s32)&src, (s32)&dst) == 1) { + D_80126B96 = 0x4004; + } + } + } + break; + + case 4: + src.x = -0x60; + dst.x = -0x80; + for (j = -0x50; j < 0x60; j += 0x20) { + src.y = j; + dst.y = j; + for (i = -0x50; i < 0x60; i += 0x20) { + src.z = i; + dst.z = i; + if (func_8012DEB8(a0, (s32)&src, (s32)&dst) == 1) { + D_80126B96 = 0x4004; + } + } + } + break; + + } + + src.x = *(u16 *)(a0 + 0x6) + D_801C5E68[*(s16 *)(a0 + 0xFC) - 1].x; + src.y = *(u16 *)(a0 + 0xA) + D_801C5E68[*(s16 *)(a0 + 0xFC) - 1].y; + src.z = *(u16 *)(a0 + 0xE) + D_801C5E68[*(s16 *)(a0 + 0xFC) - 1].z; + dst.x = src.x + D_801C5E88[*(s16 *)(a0 + 0xFC) - 1].x; + dst.y = src.y + D_801C5E88[*(s16 *)(a0 + 0xFC) - 1].y; + dst.z = src.z + D_801C5E88[*(s16 *)(a0 + 0xFC) - 1].z; + + if (func_80133784(0, &src, (s32)&dst) & 0x8000) { + *(s16 *)(a0 + 0x6) = (u16)dst.x - D_801C5EA8[*(s16 *)(a0 + 0xFC) - 1].x; + *(s16 *)(a0 + 0xE) = (u16)dst.z - D_801C5EA8[*(s16 *)(a0 + 0xFC) - 1].z; + func_80184938(a0, *(s16 *)(a0 + 0xFC)); + return 0; + } + return 1; +} + INCLUDE_ASM("asm/ov_SC03_093/nonmatchings/ov_SC03_093_jr_801825B8", func_80185344); diff --git a/tools/corpus.py b/tools/corpus.py index 03db63889..90a62d680 100644 --- a/tools/corpus.py +++ b/tools/corpus.py @@ -68,6 +68,9 @@ import json import os import re import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import cdecl # noqa: E402 — the ONE masking oracle (§134/R33) from collections import namedtuple REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) @@ -163,8 +166,23 @@ def stubs(binary): out, unparsed, unresolved, missing_s = {}, [], [], [] for p in src_files(binary): - for i, line in enumerate(open(p, errors="replace"), 1): - if not _INCLUDE_ASM_CAND.match(line): + # CANDIDACY IS DECIDED ON COMMENT-MASKED TEXT (P30 S48, byte-witnessed — the FOURTH instance + # of this class in one session, after jr_isolate_all's alias scan, overlay_src_split's + # def-proto scan and scope_data_externs' brace scan). `_INCLUDE_ASM_CAND` only skips a line + # that BEGINS with a comment marker, so a crack agent's decl annotated + # extern void func_801842DC(s32 a0); /* TU:4023 INCLUDE_ASM (no decl) */ + # is CODE followed by a comment whose prose contains `INCLUDE_ASM (` — the candidate filter + # fires, the strict parser finds no quoted path, and the whole binary's stub oracle REFUSES + # (correctly, per its own R32 contract). One such draft then broke `gate_stage` for every + # LATER binary in the run too, because they all walk the corpus. Mask first (§134/R33 — one + # masking oracle), then PARSE FROM THE ORIGINAL, since `_mask` also blanks string content + # and would erase the asm path. + raw = open(p, errors="replace").read() + masked = cdecl._mask(raw) + if len(masked) != len(raw): # length invariant broken -> do not mis-index; scan raw + masked = raw + for i, (line, mline) in enumerate(zip(raw.split("\n"), masked.split("\n")), 1): + if not _INCLUDE_ASM_CAND.match(mline): continue m = _INCLUDE_ASM.search(line) if not m: # candidate the strict parser cannot read