diff --git a/config/overlays.mk b/config/overlays.mk index 23190e336..90279f26b 100644 --- a/config/overlays.mk +++ b/config/overlays.mk @@ -809,6 +809,7 @@ build/src/ov_SC01_084/ov_SC01_084_jr_8015AE2C.o: JTBL_PADS := 0,4 # §8e pads ( build/src/ov_SC01_084/ov_SC01_084_jr_8016AB6C.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20 build/src/ov_SC01_084/ov_SC01_084_jr_80178D40.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x178 build/src/ov_SC01_084/ov_SC01_084_jr_8017CA80.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20 +build/src/ov_SC01_084/ov_SC01_084_jr_8017F690.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x14 build/src/ov_SC01_084/ov_SC01_084_o0c.o: JTBL_PADS := 0,4,4,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x38,+0xa8,+0x118 ov_SC01_084_CHECK_SHA := config/check.ov_SC01_084.sha ov_SC01_084_SYMBOLS := config/symbols.ov_SC01_084.txt @@ -3935,7 +3936,7 @@ build/src/ov_SC05_011/ov_SC05_011_jr_80159C84.o: JTBL_PADS := 0,4 # §8e pads ( build/src/ov_SC05_011/ov_SC05_011_jr_8015AE2C.o: JTBL_PADS := 0,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20 build/src/ov_SC05_011/ov_SC05_011_jr_8016AB6C.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20 build/src/ov_SC05_011/ov_SC05_011_jr_80178D40.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x178 -build/src/ov_SC05_011/ov_SC05_011_jr_8017BEBC.o: JTBL_PADS := 0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20 +build/src/ov_SC05_011/ov_SC05_011_jr_8017BEBC.o: JTBL_PADS := 0,0,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20,+0x40 build/src/ov_SC05_011/ov_SC05_011_o0c.o: JTBL_PADS := 0,4,4,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x38,+0xa8,+0x118 ov_SC05_011_CHECK_SHA := config/check.ov_SC05_011.sha ov_SC05_011_SYMBOLS := config/symbols.ov_SC05_011.txt diff --git a/config/splat.ov_SC01_084.yaml b/config/splat.ov_SC01_084.yaml index a1d89fb71..ca0de7181 100644 --- a/config/splat.ov_SC01_084.yaml +++ b/config/splat.ov_SC01_084.yaml @@ -166,7 +166,7 @@ segments: - [0x9df28, .rodata, ov_SC01_084_jr_8017CA80] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - [0x9df60, data, tail18] - [0x9df70, .rodata, ov_SC01_084_jr_8017F690] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x9df84, data, tail19] + - [0x9df98, data, tail19] - [0x9FA9C, bin, trailing] # final 3 bytes (EOF not 4-aligned; spimdisasm drops a partial word) - [0x9FA9F] # EOF marker = the 0.4.dec byte length # @TRAILING@ (above) is replaced by tools/new_overlay.sh: for a non-4-aligned overlay it becomes diff --git a/config/splat.ov_SC05_011.yaml b/config/splat.ov_SC05_011.yaml index 512c061af..402167706 100644 --- a/config/splat.ov_SC05_011.yaml +++ b/config/splat.ov_SC05_011.yaml @@ -163,7 +163,7 @@ segments: - [0x73b58, data, tail17] - [0x73b5c, .rodata, ov_SC05_011_jr_8017AE2C] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - [0x73b70, .rodata, ov_SC05_011_jr_8017BEBC] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x73bb0, data, tail18] + - [0x73bd0, data, tail18] - [0x75084, bin, trailing] # final 3 bytes (EOF not 4-aligned; spimdisasm drops a partial word) - [0x75087] # EOF marker = the 0.4.dec byte length # @TRAILING@ (above) is replaced by tools/new_overlay.sh: for a non-4-aligned overlay it becomes diff --git a/src/md_MAIN_044/md_MAIN_044.c b/src/md_MAIN_044/md_MAIN_044.c index 8cc228319..1f687e31b 100644 --- a/src/md_MAIN_044/md_MAIN_044.c +++ b/src/md_MAIN_044/md_MAIN_044.c @@ -302,7 +302,72 @@ void func_800CD1D4(s32 *s1) { } -INCLUDE_ASM("asm/md_MAIN_044/nonmatchings/md_MAIN_044", func_800CD2EC); +extern u16 D_800B99DA; +extern s16 currentLocationId; +extern s32 func_80146E98(s32 a0); +extern void func_800CD57C(void *arg0); +extern s32 func_80146A6C(s32 a0, void *a1, s32 a2, s32 a3, s32 a4, s32 a5, s32 a6); +extern void func_800CD848(s32 param_1); +extern void func_800CD780(s32 param_1); +extern void func_800CD7A8(s32 a0, s32 a1); +extern void func_80163194(s32 a0, s32 a1, s32 a2, s32 a3, s32 arg4); +extern void func_80162FC0(s32 *a0); +extern void func_800CD758(s32 param_1); +extern void func_800CD63C(s32 a0, s32 a1); +extern s32 func_800CD894(void *a0); +extern void func_80146DE8(s32 *a0, s32 a1, s32 a2, s32 a3); +extern void func_80146E90(s32 *a0, s32 a1); +extern void func_80146C98(s32 *a0, s16 a1); +extern void func_80147324(s32 a0); +extern s32 func_80163408(s32 a0, s32 a1, s32 a2, s32 a3); +extern void func_80163328(); +extern s32 func_801632F0(); +extern s32 func_801632E0(); +extern s32 func_80146CA0(void *a0); + +void func_800CD2EC(s32 *s1) { + u8 buf[0x40]; + u8 *p; + s32 s2; + s32 v0; + + s2 = *(s32 *)(s1 + 8); + if (func_80146E98((s32)s1) == 0) { + if ((*(u16 *)&D_800B99DA & 3) == 0) { + func_800CD57C(s1); + } + func_80146A6C(0x26, s1, *(s16 *)((s32)s1 + 6), *(s16 *)((s32)s1 + 0xA), + *(s16 *)((s32)s1 + 0xE), *(s16 *)(s2 + 0x12), 0); + func_800CD848((s32)s1); + func_800CD780((s32)s1); + p = buf + 0x20; + func_800CD7A8((s32)s1, (s32)p); + func_80163194((s32)s1, 0, -0x1000, 0x14000, (s32)p); + func_80162FC0(s1); + func_800CD758((s32)s1); + func_800CD63C((s32)s1, (s32)p); + if (currentLocationId == 0x3067) { + if (func_800CD894(s1) != 0) { + func_80146DE8(s1, 0, 0xFFF40000, 0x20000); + func_80146E90(s1, 0x10); + *(s32 *)(s2 + 4) |= 0x50000000; + func_80146C98(s1, 3); + func_80147324(0x990); + return; + } + } + func_80163408((s32)s1, 0x23, 0x80, 8); + func_80163328(s1); + v0 = func_801632F0(s1); + if ((v0 & 1) != 0) { + func_801632E0(s1); + } else if ((v0 & 6) == 0) { + return; + } + } + func_80146CA0(s1); +} + extern s16 D_800CE210[]; extern u16 D_800CE212[]; diff --git a/src/md_SC07_003/md_SC07_003.c b/src/md_SC07_003/md_SC07_003.c index 29aa2468e..f66fad114 100644 --- a/src/md_SC07_003/md_SC07_003.c +++ b/src/md_SC07_003/md_SC07_003.c @@ -334,7 +334,150 @@ void func_801A072C(s32 arg0) { } -INCLUDE_ASM("asm/md_SC07_003/nonmatchings/md_SC07_003", func_801A079C); +#include "common.h" + +/* + * func_801A079C (md_SC07_003, 0x801A079C, 105 ins) == MATCH, byte-exact. + * + * SC07 actor tick: nudge the sprite's 0x12 field, kick the two 0x94-state + * cutscene hooks (0xF / 0x35), run the per-frame update, then pick the next + * state (offset 0x2) from the 0xE8 counter and the func_8012BD14 distance. + * + * LEVERS (each verified by flipping it back and re-scoring with match_one): + * + * 1. THE `0xF` STORE FOR `s1 <= 0x10000` IS AN **EARLY BLOCK**, NOT A TRAILING + * `else`. Written as `if (s1 > 0x10000) { ...body... } else { store 0xF; }` + * the else-block lands LAST, so it is the block that falls into the + * epilogue. gcc's cross_jump then merges every other `sh $v0,0x2($s0)` + * into it (cookbook §5a/§193-C: the SURVIVING copy is the later one, and + * `find_cross_jump` pairs a jump's block with `prev_real_insn(JUMP_LABEL)` + * — i.e. whatever falls through into the epilogue). Result: 100 ins, four + * stores lost, LENGTH-DRIFT −5 at closeness 56. + * With the early-return form the block that falls into the epilogue is + * `sw $v0,0x1C($s0)` instead, which matches NO `sh` block, so all five + * `sh $v0,0x2($s0)` survive — exactly the target. This one edit took the + * draft from 100/56 to 105/8. No §34 asm barrier is needed: choosing the + * fall-through block IS the barrier here. + * + * 2. BRANCH-SENSE / ARM ORDER is read off the target's `slt`+`beqz`/`bnez` + * pairs (§3-T4): `if (s1 <= 0x24000) {BDBC arm} else {0xC4000 arm}` and + * `if (s1 <= 0x64000) {store 0xF} else {store 0x80}`. Writing either the + * other way round emits the complementary branch and swaps the two blocks. + * + * 3. `unused[2]` IS LOAD-BEARING — DO NOT DELETE. The target's frame is 0x28 + * with $s0/$s1/$ra at 0x18/0x1C/0x20; without an 8-byte aggregate local the + * frame is 0x20 (regs at 0x10/0x14/0x18) and eight instructions carry the + * wrong immediates. gcc-2.7.2 gives an aggregate a stack slot at expand + * time and never reclaims it, so an unreferenced 8-byte local costs zero + * instructions and buys the frame. (s16[4] and a 2xs32 struct match too — + * only the SIZE matters.) + * + * 4. `*(u16 *)(*(s32 *)(arg0 + 0x20) + 0x12) += func_8012BA10(arg0, 0x20);` + * is the TU's own house form (func_801A61C4) — the call is emitted first, + * then the pointer is reloaded, then `sh` rides the next jal's delay slot. + * + * 5. func_8012BD14 and func_8012CBA4 are declared `void` (the TU's canonical + * spelling, reconciled in S54) and their return values are read through the + * TU's `((s32 (*)(s32))f)(x)` fn-ptr cast — byte-neutral, see the note above + * func_801A1E30. + * + * SYMBOL AUDIT (law 1c, done after MATCH — match_one masks jal/HI16/LO16). + * Every name re-checked against the relocation lines of + * asm/md_SC07_003/nonmatchings/md_SC07_003/func_801A079C.s; the .s's symbol set + * and this draft's are identical (12 calls + 3 data): + * func_8012BA10(arg0,0x20) / func_8012B178(arg0,0xFFFB0000) / + * func_801A2658(arg0,&D_801A68BC) [state 0xF] and (arg0,&D_801A68C4) [0x35] / + * func_8013C9C4(D_80186F68) / func_8002D4C8(0xB53,0) / func_8012CBA4(arg0) / + * func_8012ADE4(arg0) / func_801A24E8(arg0) / rand / + * func_8012BD14(arg0) / func_8012BDBC(arg0,0x180) / func_8012BEE8(arg0). + * D_801A68C4 is its OWN relocation in this .s (it is &D_801A68BC[8], which + * func_801A23FC spells as `D_801A68BC + 8`) — law 1 says spell it as the .s + * does, so it gets its own extern here. + * + * BANK NOTE (law 2): src/md_SC07_003/md_SC07_003.c already declares eleven of + * these; every spelling below is copied verbatim from that file (rand:93, + * func_8012BEE8:282, func_8012BD14:410, func_8012BA10:412, func_8002D4C8:366, + * func_8012B178:1001, func_8012CBA4:1029, func_8012ADE4:1028, func_801A24E8:1032, + * func_8013C9C4:1340, func_801A2658:1341, D_801A68BC:1345, func_8012BDBC:4025). + * D_80186F68 and D_801A68C4 are absent from the TU, so they are free-standing; + * D_80186F68 follows the TU's own precedent for a func_8013C9C4 argument + * (`extern u16 D_80186F44[]` at line 1344) rather than the fleet's + * function-pointer-array spelling, which lives in other TUs only. + */ + +extern s32 rand(void); +extern s32 func_8012BA10(s32 a0, s32 a1); +extern void func_8012B178(s32 a0, s32 a1); +extern void func_801A2658(s32 a0, s32 a1); +extern void func_8013C9C4(void *a0); +extern void func_8002D4C8(s32 a0, s32 a1); +extern void func_8012CBA4(s32 a0); /* canonical void; return read via fn-ptr cast */ +extern void func_8012ADE4(u8 *a0); +extern void func_801A24E8(s32 a0); +extern void func_8012BD14(s32 a0); /* canonical void; return read via fn-ptr cast */ +extern s32 func_8012BDBC(s32 a0, s32 a1); +extern s32 func_8012BEE8(s32 a0); + +extern u8 D_80186F68[]; +extern u8 D_801A68BC[]; +extern u8 D_801A68C4[]; + +void func_801A079C(s32 arg0) { + s32 s1; + s32 unused[2]; /* LOAD-BEARING: buys the target's 0x28 frame (lever 3) */ + s32 v1; + + *(u16 *)(*(s32 *)(arg0 + 0x20) + 0x12) += func_8012BA10(arg0, 0x20); + func_8012B178(arg0, 0xFFFB0000); + + v1 = *(s32 *)(arg0 + 0x94); + if (v1 == 0xF) { + func_801A2658(arg0, (s32)D_801A68BC); + func_8013C9C4(D_80186F68); + func_8002D4C8(0xB53, 0); + } else if (v1 == 0x35) { + func_801A2658(arg0, (s32)D_801A68C4); + func_8013C9C4(D_80186F68); + func_8002D4C8(0xB53, 0); + } + + if ((((s32 (*)(s32))func_8012CBA4)(arg0) & 0x2000) == 0) { + func_8012ADE4((u8 *)arg0); + } + func_801A24E8(arg0); + + if (*(s32 *)(arg0 + 0xE8) > 0xFFFF) { + if (rand() & 1) { + *(s16 *)(arg0 + 2) = 0xD; + } else { + *(s16 *)(arg0 + 2) = 0xB; + } + return; + } + + s1 = ((s32 (*)(s32))func_8012BD14)(arg0); + if (s1 <= 0x10000) { + *(s16 *)(arg0 + 2) = 0xF; + return; + } + if (s1 <= 0x24000) { + if (func_8012BDBC(arg0, 0x180) != 0) { + *(s16 *)(arg0 + 2) = 0x11; + return; + } + } else if (s1 > 0xC4000 && *(s32 *)(arg0 + 0x94) == 0x4E) { + *(s16 *)(arg0 + 2) = 9; + return; + } + if (func_8012BEE8(arg0) != 0) { + if (s1 <= 0x64000) { + *(s16 *)(arg0 + 2) = 0xF; + } else { + *(s32 *)(arg0 + 0x1C) = 0x80; + } + } +} + extern void func_801A28AC(s32 a0); extern void func_8012A828(s32 a0, void *a1); diff --git a/src/ov_SC01_080/ov_SC01_080_jr_8017AE2C.c b/src/ov_SC01_080/ov_SC01_080_jr_8017AE2C.c index 16201135a..22ba9fac2 100644 --- a/src/ov_SC01_080/ov_SC01_080_jr_8017AE2C.c +++ b/src/ov_SC01_080/ov_SC01_080_jr_8017AE2C.c @@ -6522,7 +6522,70 @@ void func_801813D4(void *a0) { } -INCLUDE_ASM("asm/ov_SC01_080/nonmatchings/ov_SC01_080_jr_8017AE2C", func_80181410); +#include "common.h" + +struct vec; + +extern s32 rand(void); +extern void RotMatrixYXZ(void *a0, void *a1); +extern void func_800D20C0(void *a0, void *a1, s32 a2); +extern void func_800D23D0(void *a0); +extern void func_801292C8(u8 *a0); +extern void func_8012931C(struct vec *a0); +extern void func_801696D8(s32 a0, s32 a1); +extern s32 func_80135168(u16 a0, u16 *a1, u16 *a2); + +void func_80181410(s32 param_1) +{ + u16 sp10[3]; + u16 sp18[3]; + u16 sp20[4]; + u16 sp28[4]; + u16 sp30[16]; + u16 *p1; + u16 *p2; + u16 c; + u16 t; + + sp20[0] = *(u16 *)(param_1 + 6); + sp20[1] = *(u16 *)(param_1 + 0xA); + sp20[2] = *(u16 *)(param_1 + 0xE); + p1 = sp28; + func_800D20C0(sp20, p1, 1); + p1 = 0; + p2 = sp28; + func_800D23D0(p2); + p2 = 0; + sp28[2] = sp28[1] * 2; + RotMatrixYXZ(sp28, sp30); + if (*(s32 *)(param_1 + 0x1C) & 1) { + *(s32 *)(param_1 + 0x2C) = (rand() & 0x7FF) + 0x600; + } else { + *(s32 *)(param_1 + 0x2C) = 0x500; + } + func_801696D8(param_1, (s32)sp30); + if (--*(s32 *)(param_1 + 0x1C) != -1) { + sp10[0] = *(u16 *)(param_1 + 6); + sp10[1] = *(u16 *)(param_1 + 0xA); + sp10[2] = *(u16 *)(param_1 + 0xE); + func_8012931C((struct vec *)param_1); + sp18[0] = *(u16 *)(param_1 + 6); + sp18[1] = *(u16 *)(param_1 + 0xA); + sp18[2] = *(u16 *)(param_1 + 0xE); + if (func_80135168(1, sp10, sp18) != 0) { + t = *(u16 *)(param_1 + 2); + *(u16 *)(param_1 + 6) = sp18[0]; + *(u16 *)(param_1 + 0xA) = sp18[1]; + c = sp18[2]; + *(s32 *)(param_1 + 0x1C) = 0xC; + *(u16 *)(param_1 + 2) = t + 1; + *(u16 *)(param_1 + 0xE) = c; + } + } else { + func_801292C8((u8 *)param_1); + } +} + extern void (*D_8018A258[])(void); @@ -6899,7 +6962,86 @@ void func_80181CA4(void) } -INCLUDE_ASM("asm/ov_SC01_080/nonmatchings/ov_SC01_080_jr_8017AE2C", func_80181D98); +#include "common.h" + +/* Table at D_8018A278: {u16 dist; s16 kind;} pairs, -1-terminated (see + * asm/ov_SC01_080/data/tail.data.s). `dist` is read three ways -- lh for the + * sentinel (combine folds (s16)u16 -> lh), lhu+sll/sra for the slt against + * D_801270D0, and lhu again for the +0x500 store (the call kills the CSE). */ +typedef struct { + u16 dist; + s16 kind; +} SpotDef_8018A278; + +extern s32 D_80126D50; +extern u16 D_80126B66; +extern s32 D_80126B9C; +extern s32 D_801270CC; +extern s32 D_801270D0; +extern s32 D_80127188; +extern s32 D_801C7548; +extern s32 D_801C754C; +extern SpotDef_8018A278 D_8018A278[]; +extern void func_80180174(void); +extern void func_80181CA4(void); +extern s32 func_8012C588(s32 a0, s32 a1); +extern s32 func_8012C658(s32 a0, s32 a1, s32 a2); + +/* $s0/$s1 pinned (§17): the natural priority order hands `p` $s0 and `t` $s1, + * the target has them the other way round. `s` holds D_80126D50 so the + * unconditional `*p = 1` lands in the bnez delay slot. */ +void func_80181D98(void *a0) { + register s32 *p __asm__("$17"); + register SpotDef_8018A278 *t __asm__("$16"); + s32 e; + s32 s; + u16 v; + + p = &D_801270CC; + switch (*p) { + case 0: + *p = 1; + D_801C7548 = 0; + D_801C754C = 1; + break; + case 1: + s = D_80126D50; + *p = 1; + if (s == 0) { + if (D_801C7548 == 0x34) { + func_80180174(); + } + D_801C7548++; + if (D_801C7548 >= 0x3C && (s16)D_80126B66 >= -0x500 && + (D_80126B9C & 0x8000000) != 0) { + D_801C7548 = 0; + func_8012C588(0x36, 0); + } + } + break; + case 2: + func_80181CA4(); + p += 15; + t = D_8018A278; + if (D_80127188 == 4) { + while ((s16)t->dist != -1) { + if (*p == 0 && D_801270D0 >= (s16)t->dist) { + e = func_8012C658(0x2C, t->kind, 0); + if (e != 0) { + v = t->dist; + *(s32 *)(e + 0xCC) = (s32)p; + *(s16 *)(e + 0xFC) = v + 0x500; + } + *p = 1; + } + t++; + p++; + } + } + break; + } +} + void func_80181F5C(void *arg0) { diff --git a/src/ov_SC01_084/ov_SC01_084_jr_8017F690.c b/src/ov_SC01_084/ov_SC01_084_jr_8017F690.c index e79028c6a..6d44bf269 100644 --- a/src/ov_SC01_084/ov_SC01_084_jr_8017F690.c +++ b/src/ov_SC01_084/ov_SC01_084_jr_8017F690.c @@ -3157,7 +3157,98 @@ void func_8017FF68(s32 param_1) } -INCLUDE_ASM("asm/ov_SC01_084/nonmatchings/ov_SC01_084_jr_8017F690", func_80180000); +#include "common.h" + +// @class: dispatch-topology +// @stuck: none — MATCH (111/111 ins, match_one confirmed), 2nd compile. +// +// Levers that carry the byte match: +// (1) §2 of func_8018270C / §250 — the two `func_80028620` record bases are bound by +// ASM-INITIALISATION, `__asm__("la %0, SYM" : "=r"(p))`. This is what makes the +// base-relative accesses fold into `off($reg)` while the neighbouring members stay +// DIRECT `lui $at; sw %lo(sym)` globals (§216: a separate `lui` per adjacent byte +// means DISTINCT scalar objects, which is why D_800A5E94/95/96/EA4/EA5 are spelled as +// independent `extern u8`s and not as one struct). +// - Block A anchors at D_800A5E88 (offset 0 store + `base+0x10` for the 2nd record) -> $s0. +// - Block B anchors at D_800A5E90 and reaches the record base as `rec - 8`, which is +// exactly the target's `addiu $a1, $v1, -0x8`. A plain `&D_800A5E88` there const-folds +// to its own lui/addiu pair and loses the -8. +// Two SEPARATE locals (one per arm) — sharing one would drag block B's base into $s0. +// (2) The guard is the short-circuit decrement `x != 0 && --x == 0`: `beqz` on the loaded +// value with the `addiu -1` in its delay slot, then `bnez` with the `sw` in ITS delay slot. +// (3) D_80126B62 is the TU's `u16` house spelling (law 2) read as `(s32)(s16)` so combine +// folds sign_extend(zero_extend(mem:HI)) back into a single `lh`. The three-way ladder is +// written with `>=` (not `<`) so do_jump emits `slti`+`bnez`-to-the-else, and cross-jumping +// tail-merges the 0x1E and -0x1E stores onto the shared `sw $v0, %lo(D_800A5E8C)($at)`. +// (4) §164-54 / §193-G DISPATCH-TOPOLOGY ORACLE — the construct, not the body density, picks +// the compare shape. With only `case 2/4/6` gcc emitted a 3-way compare chain (closeness +// 22, +1 ins). Spelling the EMPTY `case 3: case 5: break;` makes 5 labels over the range +// 2..6 and gcc builds jtbl_801C60DC, whose 3rd and 5th words are the default target. +// (5) §20-style address materialisation for D_801270D8: `((struct { s32 w; } *)&D_801270D8)->w` +// keeps the symbol address in $a0 and reloads through it (the compare load, then the +// switch's own reload) — copied verbatim from func_8017F690 in this same TU. + +extern s32 rand(void); +extern void func_80028620(s32, void *); +extern void func_8012C218(void *a0); +extern u16 D_80126B62; +extern s32 D_801270D8; +extern s32 D_800A5E88; +extern s32 D_800A5E8C; +extern s32 D_800A5E90; +extern u8 D_800A5E94; +extern u8 D_800A5E95; +extern u8 D_800A5E96; +extern u8 D_800A5EA4; +extern u8 D_800A5EA5; + +void func_80180000(s32 param_1) { + if (*(s32 *)(param_1 + 0xE0) != 0 && --*(s32 *)(param_1 + 0xE0) == 0) { + u8 *base; + __asm__("la %0, D_800A5E88" : "=r"(base)); + *(s32 *)base = 0; + D_800A5E8C = 0x1E; + D_800A5E94 = 0x99; + D_800A5E90 = 0; + D_800A5E95 = 0xB2; + D_800A5E96 = 0xB2; + func_80028620(0, base); + D_800A5EA4 = 0x19; + D_800A5EA5 = 0x19; + func_80028620(1, base + 0x10); + func_8012C218((void *)param_1); + } else { + if (--*(s32 *)(param_1 + 0xE4) == 0) { + u8 *rec; + D_800A5E88 = (rand() - 0x4000) >> 10; + if ((s32)(s16)D_80126B62 >= -0x2A0) { + D_800A5E8C = 0x1E; + } else if ((s32)(s16)D_80126B62 >= -0x4A0) { + D_800A5E8C = 0; + } else { + D_800A5E8C = -0x1E; + } + __asm__("la %0, D_800A5E90" : "=r"(rec)); + *(s32 *)rec = (rand() - 0x4000) >> 10; + func_80028620(0, rec - 8); + *(s32 *)(param_1 + 0xE4) = (rand() & 1) + 1; + } + if (*(s32 *)(param_1 + 0xDC) != ((struct { s32 w; } *)&D_801270D8)->w) { + *(s32 *)(param_1 + 0xDC) = ((struct { s32 w; } *)&D_801270D8)->w; + switch (((struct { s32 w; } *)&D_801270D8)->w) { + case 2: + case 4: + case 6: + *(s32 *)(param_1 + 0xE0) = 0x1E; + break; + case 3: + case 5: + break; + } + } + } +} + extern void (*D_8018A90C[])(void); diff --git a/src/ov_SC02_005/ov_SC02_005_jr_8017CF90.c b/src/ov_SC02_005/ov_SC02_005_jr_8017CF90.c index 62789ba36..f2722da1b 100644 --- a/src/ov_SC02_005/ov_SC02_005_jr_8017CF90.c +++ b/src/ov_SC02_005/ov_SC02_005_jr_8017CF90.c @@ -4376,7 +4376,39 @@ void func_8017F7CC(s32 param_1) } -INCLUDE_ASM("asm/ov_SC02_005/nonmatchings/ov_SC02_005_jr_8017CF90", func_8017F898); +extern s32 D_801270C8; +extern u8 D_800D5C6C[]; +extern s32 D_80197244; +extern void func_8017E190(void); +extern void func_8002D4C8(s32 a0, s32 a1); +extern void func_80154274(s32 *a0, s32 a1); +extern void func_8014706C(void *a0); +extern s32 func_8013767C(s32 a0); + +void func_8017F898(s32 arg0) { + s32 s0 = arg0; + s16 sp10[2]; + + /* Non-volatile memory clobber (cookbook L1835 / §31 sched S7): the two + * dead s16 stack stores and the D_801270C8 load are all constant-address + * MEMs, so sched2 finds no memory dependence and its potential_hazard rule + * promotes the prologue `sw $ra` over the ALU candidates, sinking it to + * just above the branch. The clobber gives `sw $ra` a successor, so it is + * only ready after the load is picked and lands back in the prologue. */ + __asm__("" : : : "memory"); + + sp10[1] = -0x40; + sp10[0] = 0; + if (D_801270C8 == 0xA) { + func_8017E190(); + func_8002D4C8(0x510, 0x107F); + func_80154274((s32 *)s0, (s32)&D_800D5C6C); + func_8014706C((void *)s0); + *(s32 *)(s0 + 0x198) = func_8013767C((s32)&D_80197244); + *(u8 *)(s0 + 0x214) = *(u8 *)(s0 + 0x214) + 1; + } +} + extern s32 func_801399F0(s32 a0); extern void func_80139914(s32 a0); diff --git a/src/ov_SC02_026/ov_SC02_026_jr_8017C180.c b/src/ov_SC02_026/ov_SC02_026_jr_8017C180.c index aff767fe3..2b0de2f1f 100644 --- a/src/ov_SC02_026/ov_SC02_026_jr_8017C180.c +++ b/src/ov_SC02_026/ov_SC02_026_jr_8017C180.c @@ -3378,7 +3378,79 @@ void func_8017D4CC(void *a0) { } -INCLUDE_ASM("asm/ov_SC02_026/nonmatchings/ov_SC02_026_jr_8017C180", func_8017D508); +/* 8 bytes, align 2 -> lwl/lwr/swl/swr copy (cookbook §48-C2) */ + +extern u16 func_80148800(s32 *a0); +extern void func_8017D780(s32 param_1, s16 *param_2); + +// @class: schedule +// @stuck: none -- MATCH (114 ins). Four arms written out in full (§396-b: let gcc cross-jump the +// 0x2AA arms itself; hand-merging loses the register layout). The 0x71/0x600/-0x50 body is +// DUPLICATED in arms 1 and 3 and the target keeps BOTH copies, so it needs the §5a zero-byte +// cross-jump barrier -- but §336's "put it at the BOTTOM of the twin" is one statement too far +// here: below the last store it also blocks reorg's backward scan, the `j`'s delay slot steals +// `sh zero,0x32` from .L8017D67C instead of `sh $v0,0x30` (near 34, +1 ins). Placing it one +// statement HIGHER -- between the 0x2E store and the 0x30 store -- leaves a 1-instruction common +// suffix, which is below find_cross_jump's 2-insn minimum, so the merge still bails AND the +// delay slot still fills. That pins `li -0x50` below the asm (SCHEDULE-REORDER/3); birthing it +// as a local `k` ABOVE the barrier lets sched1 hoist it back to its target slot. All three +// lines (the local, the barrier, its position) are load-bearing -- do not tidy them. + +void func_8017D508(s32 a0) { + + typedef struct { s16 v[4]; } Blk8_80126940_8017D508; + + extern s32 D_80126B58; + extern s16 D_80189BEC[]; + extern Blk8_80126940_8017D508 D_80126940; + Blk8_80126940_8017D508 sp10; + u8 t; + + if (func_80148800(&D_80126B58) & 3) { + t = (*(u8 *)(a0 + 5) + 1) & 1; + *(u8 *)(a0 + 5) = t; + *(s32 *)(a0 + 0x14) = D_80189BEC[t]; + } + sp10 = D_80126940; + if (((u32)((u16)sp10.v[0] - 0x2C1) < 0x2BF) && (sp10.v[1] >= -0x35F) && + ((u32)((u16)sp10.v[2] - 0x581) < 0x77F)) { + *(s16 *)(a0 + 0x20) = 0x71; + *(s16 *)(a0 + 0x22) = 0x600; + *(s16 *)(a0 + 0x24) = 0; + *(s16 *)(a0 + 0x2E) = 0; + *(s16 *)(a0 + 0x30) = -0x50; + } else if (((u16)((u16)sp10.v[0] + 0x5FF) < 0x5FF) && + ((u32)((u16)sp10.v[2] - 0x681) < 0x3FC)) { + *(s16 *)(a0 + 0x20) = 0x2AA; + *(s16 *)(a0 + 0x22) = 0x800; + *(s16 *)(a0 + 0x24) = 0; + *(s16 *)(a0 + 0x2E) = 0; + *(s16 *)(a0 + 0x30) = 0; + } else if (sp10.v[2] < 0x380) { + s16 k = -0x50; /* §5a: born ABOVE the barrier so sched1 can hoist the li */ + *(s16 *)(a0 + 0x20) = 0x71; + *(s16 *)(a0 + 0x22) = 0x600; + *(s16 *)(a0 + 0x24) = 0; + *(s16 *)(a0 + 0x2E) = 0; + __asm__ __volatile__(""); /* §5a/§336 cross-jump barrier - LOAD-BEARING, see above */ + *(s16 *)(a0 + 0x30) = k; + } else { + *(s16 *)(a0 + 0x20) = 0x2AA; + *(s16 *)(a0 + 0x22) = 0x600; + *(s16 *)(a0 + 0x24) = 0; + *(s16 *)(a0 + 0x2E) = 0; + *(s16 *)(a0 + 0x30) = 0; + } + *(s16 *)(a0 + 0x32) = 0; + if (sp10.v[0] < -0x400) { + sp10.v[0] = -0x400; + } + if (sp10.v[1] >= -0x101) { + sp10.v[1] = -0x102; + } + func_8017D780(a0, sp10.v); +} + /* 8 bytes, align 2 -> lwl/lwr/swl/swr copy (cookbook §48-C2) */ diff --git a/src/ov_SC03_007/ov_SC03_007_jr_80183894.c b/src/ov_SC03_007/ov_SC03_007_jr_80183894.c index 243654368..10e1e28b2 100644 --- a/src/ov_SC03_007/ov_SC03_007_jr_80183894.c +++ b/src/ov_SC03_007/ov_SC03_007_jr_80183894.c @@ -4227,7 +4227,118 @@ zero: } -INCLUDE_ASM("asm/ov_SC03_007/nonmatchings/ov_SC03_007_jr_80183894", func_801850B4); +#include "common.h" + +extern void func_80185B48(s32 a0); +extern void func_8012A828(s32 a0, void *a1); +extern s32 func_8012BD3C(s32 a0, s32 a1, s32 a2); +extern s32 func_8012D624(void *a0, s32 a1, s32 a2); +extern void func_80185E30(void *a0); +extern void (*D_8018C588[])(void); + +/* + * Four load-bearing details (each single-axis A/B'd against match_one; the + * function is 95/95 byte-exact only with all four): + * + * 1. `s32 pad[4];` -- cookbook §162i1/§164-53 dead BLKmode local reserving the + * target's vars area, the same device func_80184E8C uses above in this TU. + * vars = 0x38 - ROUND8(args 0x10) - ROUND8(4*5 saved regs = 0x18) = 0x10. + * Without it the frame is -0x30 and every sw/lw offset is wrong. + * + * 2. `__asm__ __volatile__("" : "=r"(cc) : "0"(cc));` after the first + * func_8012D624 call. It is zero bytes and does BOTH jobs the target needs: + * (a) §5a cross-jump barrier -- find_cross_jump bails on a volatile asm, so + * gcc keeps BOTH copies of the `func_8012D624(a0,0xC0,0x30)` tail + * instead of tail-merging them (that merge costs 6 instructions); + * (b) it re-SETS cc, which kills the jump equivalence cse recorded at the + * `bne` above, so the deliberately-redundant `beq $s1,$s2` survives + * instead of folding to an unconditional `j`. That surviving beq is + * also what keeps cc live across the call -> cc earns $s1, the saved + * set grows to s0-s3+ra, and the frame reaches 0x38. + * Laundering `one` instead of `cc` here emits `beq $s2,$s1` (operands + * swapped, closeness 1). Laundering at the join instead of inside the arm + * loses the jump-threading (closeness 1 the other way). + * + * 3. `if (cc != one) goto second;` -- the explicit goto reproduces jump1's + * thread_jumps redirect (the bne skips PAST the redundant beq to + * .L8018512C). Written as a plain `if (cc == one) { ... }` the bne lands on + * the beq instead: same length, one wrong branch word. + * + * 4. `__asm__ __volatile__("");` before `one = 1;` -- zero-byte sched1 fence + * (§194-A) that keeps the `addiu $s2,$zero,1` from floating above the call + * and its sll/sra. `one` is pinned to $18 because otherwise the allocator + * hands cc/$s2 and one/$s1, i.e. the pair swapped. + * + * Every read-modify-write below is written in-place (`t = load; t += K;`) + * rather than `t = load + K;` -- §219: the in-place form reuses the load's + * register (`addiu $v0,$v0,0x200`), the other allocates a fresh one. + * q/u are separate locals from p/t on purpose: sharing them puts the head + * block's pointer in $v1 and its value in $v0, the reverse of the target. + */ +void func_801850B4(s32 a0) { + s32 pad[4]; + s32 p; + s32 q; + s32 u; + s32 t; + s32 hold; + register s32 one __asm__("$18"); + s32 cc; + + func_80185B48(a0); + if (*(s32 *)(a0 + 0x1C) != 0) { + q = *(s32 *)(a0 + 0x20); + u = *(u16 *)(q + 0x12); + hold = u + 0x100; + u -= 0x380; + *(u16 *)(q + 0x12) = u; + cc = (s16)func_8012BD3C(a0, 0x100, 0x9000); + __asm__ __volatile__(""); + one = 1; + if (cc != one) { + goto second; + } + func_8012D624((void *)a0, 0xC0, 0x30); + __asm__ __volatile__("" : "=r"(cc) : "0"(cc)); + if (cc == one) { + goto rejoin; + } + second: + p = *(s32 *)(a0 + 0x20); + *(u16 *)(p + 0x12) = *(u16 *)(p + 0x12) + 0x800; + if (func_8012BD3C(a0, 0x100, 0x9000) == one) { + func_8012D624((void *)a0, 0xC0, 0x30); + } + rejoin: + *(u16 *)(*(s32 *)(a0 + 0x20) + 0x12) = hold; + if (*(s32 *)(a0 + 0x1C) >= 5) { + p = *(s32 *)(a0 + 0x20); + t = *(u16 *)(p + 0x1C); + t += 0x200; + *(u16 *)(p + 0x1C) = t; + *(u16 *)(p + 0x18) = t; + p = *(s32 *)(a0 + 0x20); + t = *(u16 *)(p + 0x1A); + t -= 0x100; + } else { + p = *(s32 *)(a0 + 0x20); + t = *(u16 *)(p + 0x1C); + t -= 0x600; + *(u16 *)(p + 0x1C) = t; + *(u16 *)(p + 0x18) = t; + p = *(s32 *)(a0 + 0x20); + t = *(u16 *)(p + 0x1A); + t += 0x300; + } + *(u16 *)(p + 0x1A) = t; + *(s32 *)(a0 + 0x1C) = *(s32 *)(a0 + 0x1C) - 1; + } else { + func_8012A828(a0, (void *)D_8018C588); + *(s32 *)(a0 + 0x1C) = 0; + func_80185E30((void *)a0); + } +} + extern u8 D_80126B5C; extern s32 D_80126B64; diff --git a/src/ov_SC03_105/ov_SC03_105_jr_8017C8D0.c b/src/ov_SC03_105/ov_SC03_105_jr_8017C8D0.c index 313c79cbc..a5006b90f 100644 --- a/src/ov_SC03_105/ov_SC03_105_jr_8017C8D0.c +++ b/src/ov_SC03_105/ov_SC03_105_jr_8017C8D0.c @@ -3580,7 +3580,66 @@ void func_8017E170(s32 a0) { } -INCLUDE_ASM("asm/ov_SC03_105/nonmatchings/ov_SC03_105_jr_8017C8D0", func_8017E180); +extern s32 func_8012E544(s32); +extern void func_80015978(s32, s32*); +extern void func_8017E33C(s32, s16*); + + +extern Blk8_80126940 D_80126940; +extern s32 D_801BA588; + +void func_8017E180(s32 param_1) { + Blk8_80126940 arr[3]; + s32 r; + s32 a0; + s32 a1; + s32 v1; + s32 t; + + arr[0] = D_80126940; + r = func_8012E544(0x14B); + if (r != 0) { + func_80015978(r + 4, (s32*)&arr[1]); + if (*(s16*)&arr[1] < -0x200) { + *(s16*)&arr[1] = -0x200; + } + if (*(s16*)&arr[1] > 0x200) { + *(s16*)&arr[1] = 0x200; + } + } else { + arr[1] = D_80126940; + } + if (*(s16*)&arr[0] < -0x108) { + *(s16*)&arr[0] = -0x108; + } + if (*(s16*)&arr[0] > 0x108) { + *(s16*)&arr[0] = 0x108; + } + a1 = *(s16*)&arr[0]; + a0 = *(s16*)&arr[1]; + *(s16*)((s32)&arr[0] + 2) = -0x102; + *(s16*)((s32)&arr[0] + 4) = -0x200; + v1 = a1 - a0; + if (v1 < 0) { + v1 = a0 - a1; + } + __asm__ __volatile__("" ::: "memory"); + *(s32*)(param_1 + 0x14) = (v1 * 0x280) / 0x400 + 0x320; + t = *(s16*)&arr[0]; + *(s16*)&arr[0] = t - t * (s16)(*(s32*)(param_1 + 0x10) - 0x320) / 640; + switch (D_801BA588) { + case 0: + *(s16*)(param_1 + 0x20) = 0; + *(s16*)(param_1 + 0x30) = -0xB0; + break; + case 1: + *(s16*)(param_1 + 0x20) = 0xE3; + *(s16*)(param_1 + 0x30) = -0x40; + break; + } + func_8017E33C(param_1, (s16*)arr); +} + // @class: schedule @@ -5801,7 +5860,90 @@ void func_80183F84(void *a0, void *a1) { } -INCLUDE_ASM("asm/ov_SC03_105/nonmatchings/ov_SC03_105_jr_8017C8D0", func_8018423C); +#include "common.h" + +/* 8 bytes, align 2 -> the `pos = vec` / `out = vec` assignments become the + * lwl/lwr/swl/swr movstrsi_internal pair (align != UNITS_PER_WORD). */ +typedef struct { + u16 f0; + s16 f1; + u16 f2; + u16 f3; +} Vec4s16_8018423C; + +extern s32 rand(void); +extern u16 D_800B99DA; +extern s32 D_8018E59C[]; +extern s32 func_801850D8(); +extern void func_8012F214(s32 a0, s32 a1, s32 a2); +extern s32 func_80135004(s32 a0, void *a1, s32 a2); + +/* Four LOAD-BEARING details, each worth 1-2 instructions (Opus, S70y): + * + * 1. `tmp` is DECLARED AND UNUSED ON PURPOSE. The locals region is 0x20..0x3F + * (frame 0x58 = 0x20 outgoing args + 0x20 locals + 5 saved regs), i.e. FOUR + * 8-byte slots, and `out` sits at 0x38 -- so a fourth aggregate must be + * declared third to consume 0x30..0x37. expand_decl assigns slots upward + * from STARTING_FRAME_OFFSET in declaration order regardless of use. + * Deleting `tmp` moves `out` to 0x30 and shrinks the frame to 0x50. + * + * 2. `vec.f2 = 0;` BEFORE `vec.f0 = 0;`, and both AFTER `pos.f2`. The two + * `sh $zero` are the sched1 filler for the 0xE load-use gap; source order + * is what puts 0x2C ahead of 0x28 (§3-T2). The reversed order was the + * final 2-instruction residual. + * + * 3. `s32 r` with an EXPLICIT (s16) cast on the value, not `s16 r`. A `s16 r` + * defers the sll/sra 16 pair to the USE, which lands it after the second + * rand() where it coalesces straight into $a1 (-1 instruction). Casting at + * the definition puts sll/sra before/in the jal delay slot, keeps the value + * in the call-saved $s0 and re-materialises `addu $a1, $s0, $zero`. + * + * 4. `k++, step += 0x10` as the for-increment (k FIRST), and `j = i;` as the + * last statement of a do/while, not `if (i >= 3) break; j = i;`. The latter + * orders the test ahead of the copy, so reorg fills the back-branch delay + * slot with `addu $v1,$s0,$zero` instead of duplicating `addiu $s0,$v1,1` + * from the loop head (-1 instruction). + */ +void func_8018423C(s32 arg0) +{ + Vec4s16_8018423C pos; + Vec4s16_8018423C vec; + Vec4s16_8018423C tmp; + Vec4s16_8018423C out; + s32 i; + s32 j; + s32 k; + s32 step; + s32 r; + + pos.f0 = *(u16 *)(arg0 + 6); + pos.f1 = *(u16 *)(arg0 + 0xA); + pos.f2 = *(u16 *)(arg0 + 0xE); + vec.f2 = 0; + vec.f0 = 0; + + j = 0; + do { + i = j + 1; + vec.f1 = -(i << 7); + func_8012F214(arg0, (s32)&vec, (s32)&vec); + if (func_80135004(1, &pos, (s32)&vec) != 0) { + if ((D_800B99DA & 7) == 0) { + step = -0x40; + out = vec; + for (k = 0; k < 8; k++, step += 0x10) { + out.f0 = vec.f0 + step; + r = (s16)(rand() % 0x1000 + 0x2000); + func_801850D8(7, r, 0, 0, &out, 0, D_8018E59C[rand() & 1], 0); + } + } + return; + } + pos = vec; + j = i; + } while (i < 3); +} + extern u8 D_801202A0[]; diff --git a/src/ov_SC03_117/ov_SC03_117_jr_8017E6EC.c b/src/ov_SC03_117/ov_SC03_117_jr_8017E6EC.c index 51f77d9d1..cc4cba107 100644 --- a/src/ov_SC03_117/ov_SC03_117_jr_8017E6EC.c +++ b/src/ov_SC03_117/ov_SC03_117_jr_8017E6EC.c @@ -3743,7 +3743,68 @@ void func_8017FEA4(s32 *a0) { } -INCLUDE_ASM("asm/ov_SC03_117/nonmatchings/ov_SC03_117_jr_8017E6EC", func_8017FED8); +extern s32 func_8012BEE8(s32 a0); +extern void func_80015978(s32 a0, s32 *a1); +extern s32 rand(void); +extern u8 *func_801290DC(s32 arg0, u8 *arg1); +extern void func_8012BF4C(s32 *a0, s16 a1); + +/* func_8017FED8 — jitter the caller's world position by a random signed offset + * and spawn entity 0x52 there. + * + * Two byte-load-bearing spellings (both cost zero instructions): + * + * 1. `s16 adj` — NOT s32. Cookbook §194-B/§209: an s32 second local lets + * local-alloc coalesce `adj = m` away (nop in the beqz delay slot); the + * HImode declaration makes it a mode-changing set that cprop cannot fold, + * so the `addu $a0,$v1,$zero` copy survives in the delay slot and `negu` + * writes $a0, not $v1, in place. + * + * 2. `m = rand(); m = m % K;` — NOT `m = rand() % K;`. expmed.c:2745 + * (expand_divmod) zeroes `target` when `rem_flag && reg_mentioned_p(target, + * op0)`, i.e. only when the destination variable also appears in the + * DIVIDEND. The one-statement form lets gcc use m's own pseudo as the + * quotient temp too, stretching m's live range across the whole magic-number + * expansion so it conflicts with hard $v1 and gets evicted to $a0 (and the + * changed $a0 liveness then lets reorg steal `li $a0,0x52` into the + * `beqz` delay slot). Splitting the statement gives the quotient its own + * block-local pseudo -> quotient in $a0, m in $v1, adj in $a0. + */ +void func_8017FED8(s32 a0) { + u16 sp10[4]; + u8 *ent; + s32 m1; + s16 adj1; + s32 m2; + s16 adj2; + + if (func_8012BEE8(a0)) { + func_80015978(a0 + 4, (s32 *)sp10); + if (*(s32 *)(a0 + 0xDC) != 0) { + m1 = rand(); + m1 = m1 % 352; + adj1 = m1; + if (m1 & 1) { + adj1 = -m1; + } + sp10[0] = sp10[0] + adj1; + m2 = rand(); + m2 = m2 % 256; + adj2 = m2; + if (m2 & 1) { + adj2 = -m2; + } + sp10[2] = sp10[2] + adj2; + } + ent = func_801290DC(0x52, (u8 *)sp10); + if (ent != 0) { + func_8012BF4C((s32 *)a0, (rand() & 0x3F) + 0x20); + *(u16 *)(ent + 0xA) -= 0x200; + *(u16 *)(ent + 0x2E) = *(u16 *)(a0 + 0xFC); + } + } +} + diff --git a/src/ov_SC04_011/ov_SC04_011_jr_8017D494.c b/src/ov_SC04_011/ov_SC04_011_jr_8017D494.c index 90786b791..75e49f9f7 100644 --- a/src/ov_SC04_011/ov_SC04_011_jr_8017D494.c +++ b/src/ov_SC04_011/ov_SC04_011_jr_8017D494.c @@ -8942,7 +8942,52 @@ void func_80185FF4(void *a0) } -INCLUDE_ASM("asm/ov_SC04_011/nonmatchings/ov_SC04_011_jr_8017D494", func_80186020); +extern u16 D_801EFD40; +extern u16 D_801EFD44; +extern u16 D_801EFD48; +extern u16 D_801EFD4C; + +void func_80186020(void *a0) { + s32 p; + u16 x; + /* $v0 pin (cookbook "register pin" lever): without it sched1's birthing boost + hands $v0 to the wrong temp and the two lhu's swap registers (residual 12). */ + register u16 y __asm__("$2"); + + if ((D_801EFD40 & 2) != 0) { + if ((s16)D_801EFD48 >= 0x120) { + *(s16 *)&D_801EFD4C = -0x60; + } else if ((s16)D_801EFD48 < -0x5F) { + *(s16 *)&D_801EFD4C = 0x60; + } + x = D_801EFD48; + y = D_801EFD4C; + p = *(s32 *)((s32)a0 + 0x20); + x = x + y; + y = D_801EFD44; + D_801EFD48 = x; + y = y + x; + *(u16 *)(p + 0x10) = y; + } else { + if ((D_801EFD40 & 1) != 0) { + if ((s16)D_801EFD48 >= 0x120) { + return; + } + y = (s16)D_801EFD48 + 0x60; + } else { + if ((s16)D_801EFD48 < -0x5F) { + return; + } + y = (s16)D_801EFD48 - 0x60; + } + x = D_801EFD44; + p = *(s32 *)((s32)a0 + 0x20); + D_801EFD48 = y; + x = x + y; + *(u16 *)(p + 0x10) = x; + } +} + void func_80186110(s32 arg0) { diff --git a/src/ov_SC04_012/ov_SC04_012_jr_8017AE2C.c b/src/ov_SC04_012/ov_SC04_012_jr_8017AE2C.c index 813cd679a..a1b82522a 100644 --- a/src/ov_SC04_012/ov_SC04_012_jr_8017AE2C.c +++ b/src/ov_SC04_012/ov_SC04_012_jr_8017AE2C.c @@ -4056,7 +4056,107 @@ void func_8017D3E0(s32 a0) { } -INCLUDE_ASM("asm/ov_SC04_012/nonmatchings/ov_SC04_012_jr_8017AE2C", func_8017D4CC); +/* func_8017D4CC — 115 ins, MATCH. + * + * Three levers over the backlog draft (which sat at closeness 23): + * + * 1. §dbr "steal the join's first insn" — `sVar4 = 3;` belongs AFTER both inner + * `if`s, ONCE. dbr fills the `bne`'s delay slot with a COPY of the join's + * first insn and retargets the branch past it, leaving the original for the + * fall-through: that is the duplicated `addiu $v0,$zero,3` at 8017D66C / + * 8017D67C. The old draft wrote it both before and inside the `if`; gcc + * deleted the redundant one => exactly one instruction short (nins 114). + * + * 2. regalloc.md K3 (first-fit over the birth..death window): the else-arm's + * `sVar4 = 2;` must live INSIDE the `if (uVar6 < 2)` arm, not before the + * compare. Written before it, sVar4 is live across the `sltiu` temp, so the + * two conflict and everything shifts up a register ($v0->$v1->$a0). Written + * inside the arm it is born after the temp dies and shares $v0 with it — + * which is why the target can hold uVar6 in $v1 and sVar4 in $v0. + * + * 3. regalloc.md K3 again, for the 0x60000000 flag store: the or-chain must be + * written out in EACH arm, not through a shared accumulator behind a `goto`. + * In one block the three values (address, 0x40000000, 0x20000000) are all + * block-LOCAL qtys, so local-alloc's first-fit gives the address $v0 and the + * two constants $v1 (they die immediately); cross_jump then merges the + * identical `or/lui/or/sw` suffix back into one copy and dbr lifts the + * surviving `lui $v1,0x4000` into the `j`'s delay slot. Held in a shared + * local across the join it becomes a GLOBAL allocno instead, local-alloc + * hands the constants $v0 first, and the whole chain comes out swapped. + */ +extern s32 func_8012C354(s32 a0, s32 a1); +extern void func_8001C214(s32 a0, s32 a1); +extern s32 func_80029504(void); +extern u8 D_8018224C[]; +extern u8 D_8018F198[]; +extern u8 D_8018222C[]; +extern u8 D_8018F210[]; +extern u8 D_8018F288[]; +extern u8 D_8018223C[]; + +void func_8017D4CC(s32 param_1) { + s32 iVar2; + u32 uVar5; + s16 sVar4; + u16 uVar6; + + iVar2 = func_8012C354(param_1, (s32)D_8018224C); + if (iVar2 == 0) { + return; + } + sVar4 = *(s16 *)(param_1 + 0x70); + if (sVar4 == 1) goto CASE1; + if (sVar4 < 2) goto SKIP; + if (sVar4 == 2) goto CASE2; + if (sVar4 == 3) goto CASE3; + goto SKIP; +CASE1: + func_8001C214(*(s32 *)(param_1 + 0x20), (s32)D_8018F198); + *(u32 *)(param_1 + 0x58) = (u32)D_8018222C | 0x40000000 | 0x20000000; + goto SKIP; +CASE2: + func_8001C214(*(s32 *)(param_1 + 0x20), (s32)D_8018F210); + *(u32 *)(param_1 + 0x58) = (u32)D_8018223C | 0x40000000 | 0x20000000; + goto SKIP; +CASE3: + func_8001C214(*(s32 *)(param_1 + 0x20), (s32)D_8018F288); + *(u32 *)(param_1 + 0x58) = (u32)D_8018223C | 0x40000000 | 0x20000000; +SKIP: + *(u8 *)(param_1 + 0xc0) = 1; + *(u8 *)(param_1 + 0x75) = 2; + *(s16 *)(param_1 + 0xae) = -1; + *(s16 *)(param_1 + 2) = 1; + uVar5 = func_80029504(); + if (uVar5 >= 0x2f0) { + if (*(s16 *)(param_1 + 0x70) == 0) { + *(s16 *)(*(s32 *)(param_1 + 0x20) + 0x12) = 0xd00; + } + if (*(s16 *)(param_1 + 0x70) == 1) { + *(s16 *)(*(s32 *)(param_1 + 0x20) + 0x12) = 0xd00; + } + if (*(s16 *)(param_1 + 0x70) == 2) { + *(s16 *)(*(s32 *)(param_1 + 0x20) + 0x12) = 0x300; + } + if (*(s16 *)(param_1 + 0x70) == 3) { + *(s16 *)(*(s32 *)(param_1 + 0x20) + 0x12) = 0x300; + } + } else { + uVar6 = *(u16 *)(param_1 + 0x70); + if (uVar6 < 2) { + sVar4 = 2; + } else { + if ((s16)uVar6 == 2) { + *(s16 *)(*(s32 *)(param_1 + 0x20) + 0x12) = 0x100; + } + if (*(s16 *)(param_1 + 0x70) == 3) { + *(s16 *)(*(s32 *)(param_1 + 0x20) + 0x12) = 0x100; + } + sVar4 = 3; + } + *(s16 *)(param_1 + 2) = sVar4; + } +} + extern void (*D_80182280[])(void); diff --git a/src/ov_SC05_005/ov_SC05_005_jr_8017D898.c b/src/ov_SC05_005/ov_SC05_005_jr_8017D898.c index 3b32477e8..18ed84383 100644 --- a/src/ov_SC05_005/ov_SC05_005_jr_8017D898.c +++ b/src/ov_SC05_005/ov_SC05_005_jr_8017D898.c @@ -3328,7 +3328,82 @@ void func_8017EAB0(void *a0) { } -INCLUDE_ASM("asm/ov_SC05_005/nonmatchings/ov_SC05_005_jr_8017D898", func_8017EAEC); +#include "common.h" + +/* func_8017EAEC — ov_SC05_005 / ov_SC05_005_jr_8017D898 (107 ins, MATCH) + * + * Per-frame camera driver for this scene: every 4th tick (func_80148800 & 3) + * it flips the 1-bit phase at +5 and reloads the +0x14 field from the 2-entry + * table D_801860AC, then copies the 8-byte camera aggregate D_80126940 onto the + * stack, clamps it, offsets a second copy by sin/cos of the player angle, asks + * func_8017EE64 for a yaw and hands both to func_8017EC98. + * + * Levers: + * §48-C2 — D_80126940 is an 8-byte, 2-BYTE-ALIGNED aggregate, so the plain + * struct assign `sp10 = D_80126940;` is what emits the lwl/lwr + swl/swr + * block copy (same typedef shape the rest of this family uses). + * base+offset — `lw $v0, 0x20($s1)` reads D_80126B78 through the SAME $s1 that + * holds &D_80126B58 for the func_80148800 call, so it must be spelled off + * that base (`base[8]`), never as its own %hi(D_80126B78) fold. + * /128 — `bgez / addiu 0x7F / sra 7` is a signed divide by 128, not `>> 7`. + * §286 — THE ONLY RESIDUAL (closeness 4). Written as a plain statement, + * `*(s16 *)(a0 + 0xA0) = r;` is emitted BEFORE the argument setup, so the + * `li $a3,1` wins the jal's delay slot and the sh lands 4 insns early. + * Folding the store into a comma-expression in an ARGUMENT position places + * its RTL after the last arg move, and the sh takes the delay slot: 4 -> 0. + */ + +/* 8 bytes, align 2 -> lwl/lwr/swl/swr copy (cookbook §48-C2) */ +typedef struct { s16 v[4]; } Blk8_80126940_8017D6D0_8017EAEC; + +extern u16 func_80148800(s32 *a0); +extern s32 func_8004787C(s32 a0); +extern s32 func_80047948(s32 a0); +extern s32 func_80012DBC(s32 a0, s32 a1, s32 a2, s32 a3); +extern s32 func_8017EE64(u8 *param_1, s16 *param_2, s16 *param_3); +extern void func_8017EC98(s32 param_1, s32 param_2, s16 *param_3); +extern s32 D_80126B58; + +void func_8017EAEC(s32 a0) { + + extern s16 D_801860AC[]; + extern u8 D_80186058[]; + extern s16 D_801B1EB8; + extern Blk8_80126940_8017D6D0_8017EAEC D_80126940; + + Blk8_80126940_8017D6D0_8017EAEC sp10; + Blk8_80126940_8017D6D0_8017EAEC sp18; + s32 *base; + u8 t; + s32 r; + + base = &D_80126B58; + if (func_80148800(base) & 3) { + t = (*(u8 *)(a0 + 5) + 1) & 1; + *(u8 *)(a0 + 5) = t; + *(s32 *)(a0 + 0x14) = D_801860AC[t]; + } + sp10 = D_80126940; + if (sp10.v[0] < -0x1440) { + sp10.v[0] = -0x1440; + } + if (sp10.v[1] > -0x200) { + sp10.v[1] = -0x200; + } + if (sp10.v[0] > 0x1100) { + if (sp10.v[2] > -0x80) { + sp10.v[2] = -0x80; + } + } + sp18.v[0] = sp10.v[0] - func_8004787C(((s16 *)base[8])[9]) / 128; + sp18.v[2] = sp10.v[2] - func_80047948(((s16 *)base[8])[9]) / 128; + sp18.v[1] = sp10.v[1]; + r = func_8017EE64(D_80186058, sp10.v, sp18.v); + /* §286: the store must ride the jal's delay slot -- see header. */ + D_801B1EB8 = func_80012DBC(D_801B1EB8, (s16)r, 4, (*(s16 *)(a0 + 0xA0) = r, 1)); + func_8017EC98(a0, D_801B1EB8, sp10.v); +} + extern s32 func_80012C6C(s32 a0, s32 a1, s32 a2); diff --git a/src/ov_SC05_008/ov_SC05_008_jr_8017AE2C.c b/src/ov_SC05_008/ov_SC05_008_jr_8017AE2C.c index bde4742fd..3ca11cbfd 100644 --- a/src/ov_SC05_008/ov_SC05_008_jr_8017AE2C.c +++ b/src/ov_SC05_008/ov_SC05_008_jr_8017AE2C.c @@ -4583,7 +4583,7 @@ extern s32 D_801A108C; extern s32 D_801A1090; extern s32 D_801A1094; extern void func_8017E840(void *a0); -extern void func_8017E6A4(void *a0); +extern void func_8017E6A4(); extern void func_8017EAE4(s32); extern s32 func_8012AD50(void *a0); @@ -4631,7 +4631,64 @@ void func_8017E354(s32 param_1) { INCLUDE_ASM("asm/ov_SC05_008/nonmatchings/ov_SC05_008_jr_8017AE2C", func_8017E464); -INCLUDE_ASM("asm/ov_SC05_008/nonmatchings/ov_SC05_008_jr_8017AE2C", func_8017E6A4); +// @class: align-1 block move + sched1 load-delay fence +// The 8-byte sp+0x10 -> D_80126BE0 copy is an ALIGN-1 STRUCT ASSIGN (cookbook §160a / +// §48-C2): emit_block_move lowers it to lwl/lwr + swl/swr with ZERO memcpy-symbol +// reference, so this TU's file-scope `extern memcpy` (line 85, which disables the +// builtin and turns a memcpy() draft into a CALL -> LENGTH-DRIFT/-5) cannot drift it. +// Without the trailing zero-byte §194-A fence, sched1 hoists the `lhu D_80126B5E` +// above the block move and the pair eats its load-delay nop (cookbook-index L30: an +// lwl/lwr+swl/swr pair is a delay-slot SPONGE) -> LENGTH-DRIFT/-1. The fence keeps the +// nop and costs no bytes. + +typedef struct { u8 b[8]; } Blk8_8017E6A4; + +extern s32 func_80012F74(s32 a0, s32 a1, s32 a2, s32 a3); +extern void func_8012B370(s32); +extern void func_8012F214(s32 a0, s32 a1, s32 a2); +extern void func_80015954(s32 a0, s32 a1); + +void func_8017E6A4(s32 param_1) +{ + + extern s32 D_801A12E4; + extern s32 D_8012699C; + extern s16 D_80126C88; + extern s16 D_80126C8A; + extern void (*D_80186544[])(void); + u8 buf[8]; + u16 *rec; + s32 *b78; + s32 t; + + rec = (u16 *)(D_8012697C + (D_80126980 + 8) * 6); + b78 = D_80126B78; + t = D_801A12E4; + + *(u16 *)(param_1 + 6) = rec[0]; + *(u16 *)(param_1 + 0xA) = rec[1]; + *(u16 *)(param_1 + 0xE) = rec[2]; + + *(s16 *)((s32)b78 + 0x14) = -*(u16 *)(*(s32 *)(param_1 + 0x20) + 0x14) & 0xFFF; + *(s16 *)&D_80126C88 = -*(u16 *)(*(s32 *)(param_1 + 0x20) + 0x10) & 0xFFF; + *(s16 *)&D_80126C8A = (*(u16 *)(*(s32 *)(param_1 + 0x20) + 0x12) + 0x800) & 0xFFF; + + D_8012699C = (func_80012F74((D_8012699C << 14) >> 16, (t << 29) >> 16, 10, 1) << 16) >> 14; + + func_8012B370(param_1); + func_8012F214(param_1, (s32)D_80186544, (s32)&buf[0]); + func_80015954((s32)&buf[0], (s32)&D_80126B5C); + *(Blk8_8017E6A4 *)D_80126BE0 = *(Blk8_8017E6A4 *)&buf[0]; + __asm__ __volatile__(""); + + *(s16 *)((s32)b78 + 8) = D_80126B5E; + *(s32 *)((s32)b78 + 0x48) = (s32)(s16)D_80126B5E; + *(s16 *)((s32)b78 + 0xA) = D_80126B62; + *(s32 *)((s32)b78 + 0x4C) = (s32)(s16)D_80126B62; + *(s16 *)((s32)b78 + 0xC) = D_80126B66; + *(s32 *)((s32)b78 + 0x50) = (s32)(s16)D_80126B66; +} + void func_8017E840(void *a0) { extern s32 D_801A12E4; @@ -4996,7 +5053,61 @@ void func_8017F24C(void *a0) { } -INCLUDE_ASM("asm/ov_SC05_008/nonmatchings/ov_SC05_008_jr_8017AE2C", func_8017F288); +extern s32 func_8012C588(s32 a0, s32 a1); +extern s32 func_8012AD50(void *a0); + +void func_8017F288(s32 param_1) { + extern s32 D_801151D4; + extern s32 D_801A1274; + extern void (*D_801865B8[])(void); + s32 wp; + s32 s1; + s32 v0; + s16 key; + s16 key2; + s16 i; + s16 lim; + void (**tbl)(void); + + wp = D_801151D4; + s1 = D_801A1274; + key = *(u16 *)(*(s32 *)(wp + 0x34) + *(u16 *)(wp + 0x38) * 6 + 4); + if (*(s32 *)(s1 + 0xC) != 0) { + do { + if (*(s16 *)(s1 + 4) >= key) { + goto done1; + } + s1 += 0x10; + } while (*(u32 *)(s1 + 0xC) != 0); + } +done1: + i = *(u16 *)(wp + 0x38); + lim = i + 0x50; + if (i < lim) { + tbl = D_801865B8; + do { + key2 = *(u16 *)(*(s32 *)(wp + 0x34) + i * 6 + 4); + if (*(s32 *)(s1 + 0xC) != 0) { + do { + if (*(s16 *)(s1 + 4) >= key2) { + goto next; + } + v0 = func_8012C588((s32)tbl[*(u32 *)(s1 + 0xC)], param_1); + if (v0 != 0) { + *(s32 *)(v0 + 0xCC) = s1; + *(s32 *)(v0 + 0xDC) = i; + } + s1 += 0x10; + } while (*(u32 *)(s1 + 0xC) != 0); + } + next: + i++; + } while (i < lim); + } + D_801A1274 = s1; + func_8012AD50((void *)param_1); +} + extern s32 func_8012C588(s32 a0, s32 a1); diff --git a/src/ov_SC05_011/ov_SC05_011_jr_8017BEBC.c b/src/ov_SC05_011/ov_SC05_011_jr_8017BEBC.c index 195dff760..d39d26154 100644 --- a/src/ov_SC05_011/ov_SC05_011_jr_8017BEBC.c +++ b/src/ov_SC05_011/ov_SC05_011_jr_8017BEBC.c @@ -3534,7 +3534,73 @@ void func_8017D4A4(void *arg0) { } -INCLUDE_ASM("asm/ov_SC05_011/nonmatchings/ov_SC05_011_jr_8017BEBC", func_8017D610); +/* Twin of the banked same-TU neighbour func_8017D4A4 (§194-E): three + * "if (obj->f90 == &G && obj->f98 == 0)" guards whose f90 loads CSE together, + * three (func_80029178(id) & 0xFF) flag drains, then the neighbours own + * switch ((s16)(*(u16 *)(obj + 0x108))--) with case bodies ordered 0,1,5,3/7 + * so cross-jumping folds the three func_800183E0 tails into one. */ +extern s32 func_8012A828(); +extern void func_80029124(); +extern s32 func_80029178(); +extern void func_800183E0(); +extern s32 rand(); + +extern s32 D_80181DE8; +extern s32 D_80181FB8; +extern s32 D_80182138; +extern s32 D_80181D48; +extern s32 D_801820B0; +extern u8 D_80181E40[]; +extern u8 D_8019A07C[]; +extern u8 D_8019A61C[]; +extern u8 D_8019A34C[]; +extern s32 jtbl_8019BD08[]; + +void func_8017D610(void *arg0) { + if (*(s32 *)((s32)arg0 + 0x90) == (s32)&D_80181DE8) { + if (*(s16 *)((s32)arg0 + 0x98) == 0) { + func_8012A828(arg0, D_80181E40); + } + } + if (*(s32 *)((s32)arg0 + 0x90) == (s32)&D_80181FB8) { + if (*(s16 *)((s32)arg0 + 0x98) == 0) { + func_8012A828(arg0, &D_801820B0); + } + } + if (*(s32 *)((s32)arg0 + 0x90) == (s32)&D_80182138) { + if (*(s16 *)((s32)arg0 + 0x98) == 0) { + func_8012A828(arg0, &D_801820B0); + } + } + if (func_80029178(0x128) & 0xFF) { + func_80029124(0x128, 0); + func_8012A828(arg0, &D_80181DE8); + } + if (func_80029178(0x12A) & 0xFF) { + func_80029124(0x12A, 0); + func_8012A828(arg0, &D_80181D48); + } + if (func_80029178(0x12C) & 0xFF) { + func_80029124(0x12C, 0); + func_8012A828(arg0, &D_80182138); + } + switch ((s16)(*(u16 *)((s32)arg0 + 0x108))--) { + case 0: + *(u16 *)((s32)arg0 + 0x108) = (rand() & 0x3F) + 0x3C; + break; + case 1: + func_800183E0((s32)D_8019A07C); + break; + case 5: + func_800183E0((s32)D_8019A61C); + break; + case 3: + case 7: + func_800183E0((s32)D_8019A34C); + break; + } +} + extern void (*D_801823B0[])(void); diff --git a/src/ov_SC05_018/ov_SC05_018_jr_8017D604.c b/src/ov_SC05_018/ov_SC05_018_jr_8017D604.c index cfb204302..7136e5237 100644 --- a/src/ov_SC05_018/ov_SC05_018_jr_8017D604.c +++ b/src/ov_SC05_018/ov_SC05_018_jr_8017D604.c @@ -4823,7 +4823,7 @@ extern void func_80016714(void *a0, s32 a1); extern void func_8012C218(void *a0); extern s32 func_80181294(u16 a0, s16 a1); extern void func_8018124C(void); -extern void func_801810B0(void *a0); +extern void func_801810B0(); void func_80180DBC(void *a0) { Ent_801E650C *p; @@ -4885,7 +4885,90 @@ void func_80180DBC(void *a0) { } -INCLUDE_ASM("asm/ov_SC05_018/nonmatchings/ov_SC05_018_jr_8017D604", func_801810B0); +/* func_801810B0 — POLY_FT4 (0x28 bytes) centred sprite emit. + * + * Levers that closed this one (previous attempt sat at closeness=47): + * + * 1. D_800A651C IS NOT `s32[]`. The tail computes the index with + * `sll 2; addu; sll 2` = *20, i.e. a 5-word record. Declaring it + * `extern struct { s32 a; s32 b[4]; } D_800A651C[];` at BLOCK scope + * (the proven src/800.c:3377/:5326 form — the card's fleet type + * ('s32','') is the scalar spelling engine_core.h's DEFINE_ macros use) + * supplied the two missing instructions and fixed idx 67..78. + * + * 2. The tpage store lands in the `beqz` DELAY SLOT only if the flag load + * PRECEDES it in source. With `*(u16*)(prim+0x16) = tpage;` written + * before the `if`, sched1 cannot prove the store does not alias the + * global, so a store->load dependence pins the store first and reorg + * steals `addiu 0x2E` from the arm instead. Hoisting the read into a + * local (`flag = D_801E664C;`) turns that into a load->store anti-dep: + * lui/lh/nop/beqz emit first and the `sh` sinks into the slot. + * + * 3. The UV and XY blocks are plain libgpu MACRO ORDER — setUV4 + * (u0,v0,u1,v1,u2,v2,u3,v3) and setXY4 (x0,y0,x1,y1,x2,y2,x3,y3). + * Writing the stores in the order the *target emits* them (all v's + * then all u's / all y's then all x's) is the trap: that is the + * SCHEDULER's output, not the source. Source = macro order; sched1 + * regroups by shared constant/register on its own. The declaration + * order of x/y/half is byte-irrelevant (all 6 permutations MATCH). + */ +void func_801810B0(s32 arg0) +{ + extern void *func_80010A08(s32 arg0); + extern s32 GetClut(s32 a0, s32 a1); + extern s32 GetTPage(s32 a0, s32 a1, s32 a2, s32 a3); + extern s32 AddPrim(s32 a0, void *a1); + extern struct { s32 a; s32 b[4]; } D_800A651C[]; /* block scope: stride 0x14 (§ src/800.c:3377) */ + s32 prim; + s32 clut; + s32 tpage; + s32 flag; + s32 x; + s32 y; + s32 half; + + if (*(s16 *)(arg0 + 0xA) <= 0) { + return; + } + + prim = (s32)func_80010A08(0x28); + clut = GetClut(0x160, 0x140); + *(u16 *)(prim + 0xE) = clut; + tpage = GetTPage(0, 1, 0x2C0, 0x100); + *(s32 *)(prim + 4) = 0x808080; + *(u8 *)(prim + 3) = 9; + *(u8 *)(prim + 7) = 0x2C; + flag = D_801E664C; + *(u16 *)(prim + 0x16) = tpage; + if (flag != 0) { + *(u8 *)(prim + 7) = 0x2E; + } + + *(u8 *)(prim + 0xC) = 0x90; + *(u8 *)(prim + 0xD) = 0x50; + *(u8 *)(prim + 0x14) = 0xCF; + *(u8 *)(prim + 0x15) = 0x50; + *(u8 *)(prim + 0x1C) = 0x90; + *(u8 *)(prim + 0x1D) = 0x8F; + *(u8 *)(prim + 0x24) = 0xCF; + *(u8 *)(prim + 0x25) = 0x8F; + + half = (s16)*(u16 *)(arg0 + 0xA) / 2; + x = *(s16 *)(arg0 + 4); + y = *(s16 *)(arg0 + 6); + + *(s16 *)(prim + 8) = x - half; + *(s16 *)(prim + 0xA) = y - half; + *(s16 *)(prim + 0x10) = x + half; + *(s16 *)(prim + 0x12) = y - half; + *(s16 *)(prim + 0x18) = x - half; + *(s16 *)(prim + 0x1A) = y + half; + *(s16 *)(prim + 0x20) = x + half; + *(s16 *)(prim + 0x22) = y + half; + + AddPrim(D_800A651C[*(u16 *)&D_800B9A02].a + 0x28, (void *)prim); +} + void func_80181204(void) { diff --git a/src/ov_SC06_006/ov_SC06_006_jr_8017DB90.c b/src/ov_SC06_006/ov_SC06_006_jr_8017DB90.c index cdc07cf0a..1b4c9475e 100644 --- a/src/ov_SC06_006/ov_SC06_006_jr_8017DB90.c +++ b/src/ov_SC06_006/ov_SC06_006_jr_8017DB90.c @@ -3638,7 +3638,98 @@ u32 func_8017ED24(u32 param_1, u32 param_2) } -INCLUDE_ASM("asm/ov_SC06_006/nonmatchings/ov_SC06_006_jr_8017DB90", func_8017EEA4); +/* func_8017EEA4 — ov_SC06_006 actor step (116 ins). + * + * @class: INTEGRATION-ONLY. The body was already byte-true (match_one MATCH); the + * whole-binary gate refused it on SIX declaration conflicts with its own TU, all + * of which are fixed below. Do not redraft the body. + * + * LEVERS (each verified by removing it; only #2 is byte-visible): + * 1. `Blk8` is ALREADY a typedef in src/shared/engine_types.h (`struct {u8 b[8];}`), + * so a file-scope `typedef ... Blk8;` is a redefinition. Both block types are + * declared at BLOCK scope instead — legal shadowing, and match_one's standalone + * context has neither `Blk8` nor this TU's `Col4`. Byte-neutral. + * 2. §37 asm-label alias for func_8017ED24. The TU DEFINES `u32 func_8017ED24(u32, + * u32)` above our stub, so re-declaring it `(void*, s32)` is a hard conflict — + * but calling it through the u32 prototype costs a callee-saved register + * (work/$s0 gets rematerialised, $s2 disappears: 113 ins, LENGTH-DRIFT/-3). + * A distinct C name bound to the same asm symbol keeps the byte-true pointer + * form at zero conflict. This is the ONE edit of the six that moves bytes. + * 3. func_8017F4C0: adopt the TU's own spelling (def @3920 / decl @4819) whose 3rd + * param is s32, and put the (u16) cast at the CALL SITE — that is what emits the + * target's `andi $a2, $v0, 0xFFFF`. Byte-identical to a u16 parameter. + * 4. func_8017F258 is defined below us as `void func_8017F258(void)` (it reads its + * argument through a register-$4 pin, §42). An unprototyped decl is C89- + * compatible with that definition AND still passes $a0 in the delay slot. + * 5. §37 again for D_80185FA0: the TU declares it `Rec_80180AE8 []` AFTER our stub, + * and the typedef is not in scope here (duplicating it would be a redefinition). + * 6. §124 DEFINITION-side alias. The banked neighbour func_80180AE8 carries a + * block-scope `extern s32 func_8017EEA4();`, which conflicts with our required + * `void` return. Returning s32 instead is NOT free: $v0 becomes live-out, gcc + * can no longer sink `addiu $v0,$a0,1` into the case-0 branch delay slot, and the + * function grows to 117 ins. fix_arity_callers cannot help — it strips parameter + * lists, and this conflict is on the RETURN axis. So alias the definition. + * + * @stuck: none — match_one MATCH (116/116) AND recover_integration --probe-only + * reports `static: none / real cc1 MATCH 116 ins` compiling inside the real TU. + */ +extern s32 rand(void); +extern void func_8017F4C0(void *a0, u16 a1, s32 a2); +extern void aF8017ED24(void *, s32) __asm__("func_8017ED24"); +extern void func_8017F258(); +extern u8 D_80185FA0_recs[] __asm__("D_80185FA0"); +extern u8 D_80185F94[]; + +void aF8017EEA4(void *arg) __asm__("func_8017EEA4"); + +void aF8017EEA4(void *arg) +{ + typedef struct { u8 b[4]; } Blk4; + typedef struct { u8 b[8]; } Blk8; + u8 *work = (u8 *)arg + 0xC; + u8 *rec; + Blk4 t1; + Blk4 t2; + s16 phase; + + (*(u16 *)((u8 *)arg + 6))++; + rec = &D_80185FA0_recs[*(s16 *)((u8 *)arg + 0x4C) * 20]; + func_8017F4C0(work, *(u16 *)(rec + 2), (u16)((rand() & 0x7F) - 0x3F)); + aF8017ED24(&t1, *(s32 *)(rec + 0xC)); + *(Blk4 *)((u8 *)arg + 0xC) = t1; + aF8017ED24(&t2, *(s32 *)(rec + 0x10)); + *(Blk4 *)((u8 *)arg + 0x10) = t2; + *(Blk8 *)(*(u8 **)((u8 *)arg + 8) + 8) = *(Blk8 *)D_80185F94; + + phase = *(s16 *)((u8 *)arg + 4); + switch (phase) + { + case 0: + if (*(s16 *)((u8 *)arg + 6) < *(u8 *)(rec + 1)) + break; + (*(u16 *)((u8 *)arg + 4))++; + *(u16 *)((u8 *)arg + 6) = 0; + break; + case 1: + *(s32 *)(*(u8 **)((u8 *)arg + 8) + 4) &= 0x7FFFFFFF; + *(s16 *)(*(u8 **)((u8 *)arg + 8) + 0x18) = *(s16 *)(*(u8 **)((u8 *)arg + 8) + 0x1A) = + *(u16 *)((u8 *)arg + 6) * *(u16 *)(rec + 6); + if (*(s16 *)((u8 *)arg + 6) < *(s16 *)(rec + 4)) + break; + (*(u16 *)((u8 *)arg + 4))++; + *(u16 *)((u8 *)arg + 6) = 0; + break; + case 2: + if (*(s16 *)((u8 *)arg + 6) < *(s16 *)(rec + 8)) + break; + (*(u16 *)((u8 *)arg + 6) = 0, (*(u16 *)((u8 *)arg + 4))++); + break; + case 3: + func_8017F258(arg); + break; + } +} + #include "common.h" diff --git a/src/ov_SC06_008/ov_SC06_008_jr_8017C294.c b/src/ov_SC06_008/ov_SC06_008_jr_8017C294.c index df976120f..dc412e647 100644 --- a/src/ov_SC06_008/ov_SC06_008_jr_8017C294.c +++ b/src/ov_SC06_008/ov_SC06_008_jr_8017C294.c @@ -3908,7 +3908,76 @@ next3: } -INCLUDE_ASM("asm/ov_SC06_008/nonmatchings/ov_SC06_008_jr_8017C294", func_8017E37C); +#include "common.h" + +/* ---- decls (TU house style: ratan2/func_80021174/rand copied verbatim from + * the existing decls in src/ov_SC06_008/ov_SC06_008_jr_8017C294.c; + * func_801290DC is left unprototyped exactly as the neighbour + * func_8017EBAC uses it; func_8017EBAC matches its definition below in + * the same TU. D_80189570 / D_801895B4 are new to this TU.) ---- */ +extern u16 *D_80189570[]; +extern s32 D_801895B4; +extern s32 func_80021174(s32 a0, s32 a1); +extern s32 ratan2(s32 dx, s32 dy); +extern s32 rand(void); +extern void func_8017EBAC(int param_1); +extern s32 func_801290DC(); + +/* 0x10-byte spawn record built on the stack at sp+0x10 */ +typedef struct { + s16 x; /* 0x00 */ + s16 y; /* 0x02 */ + s16 z; /* 0x04 */ + u16 ang; /* 0x06 */ + s16 f8; /* 0x08 */ + s16 fA; /* 0x0A */ + s16 fC; /* 0x0C */ + s16 fE; /* 0x0E */ +} Cfg_8017E37C; + +/* 0x10-byte VECTOR at sp+0x20 (pad needed: it fixes s0/ra at 0x30/0x34) */ +typedef struct { + s32 vx, vy, vz, pad; +} Vec_8017E37C; + +s32 func_8017E37C(void *a0, s32 a1, s32 a2) +{ + Cfg_8017E37C sp10; + Vec_8017E37C sp20; + u16 *p; + s32 q; + + /* §219: the base load and the stride add MUST be two statements — + * folding them into one expression schedules the a1*6 chain first and + * lands the base in $v1 instead of accumulating into $s0. */ + p = D_80189570[*(s16 *)((s32)a0 + 0x70)]; + p += (s16)a1 * 3; + + sp10.x = p[0]; + sp20.vx = sp10.x; + sp10.y = p[1] + a2; + sp20.vy = sp10.y; + sp10.z = p[2]; + sp20.vz = sp10.z; + sp10.f8 = p[3]; + if (sp10.f8 == 0x7FFF) { + return 1; + } + if (func_80021174(D_801895B4, (s32)&sp20) != 1) { + return 0; + } + sp10.fC = p[5]; + sp10.ang = ratan2(sp10.f8 - sp10.x, sp10.fC - sp10.z); + q = func_801290DC(0x68, &sp10); + if (q != 0) { + *(s16 *)(q + 0x2C) = sp10.ang; + } + if ((rand() & 7) == 0) { + func_8017EBAC((int)&sp10); + } + return 0; +} + extern void (*D_80189604[])(void); diff --git a/src/ov_SC06_032/ov_SC06_032_jr_8017C24C.c b/src/ov_SC06_032/ov_SC06_032_jr_8017C24C.c index 1aeaebdc8..5adbafadc 100644 --- a/src/ov_SC06_032/ov_SC06_032_jr_8017C24C.c +++ b/src/ov_SC06_032/ov_SC06_032_jr_8017C24C.c @@ -3567,7 +3567,122 @@ void func_8017D940(int param_1) } -INCLUDE_ASM("asm/ov_SC06_032/nonmatchings/ov_SC06_032_jr_8017C24C", func_8017DABC); +extern void func_80015978(s32 a0, s32 *a1); +extern void func_800178EC(s32 a0); + +/* func_8017DABC — builds one POLY_FT4 (0x48 bytes) on the stack and hands it to + * func_800178EC. Line-for-line relative of the banked twin ov_SC06_018:func_8017DB20 + * (§193-A); adapted, not copied — every literal and both jal symbols were re-read + * off this .s. Only two relocations exist here (func_80015978, func_800178EC); + * there are no %hi/%lo or data symbols. + * + * Two load-bearing deviations from the twin: + * + * 1. param_3/param_4 are s32 here, NOT the twin's s16. This TU already carries + * `extern void func_8017DABC();` at ov_SC06_032_jr_8017C24C.c:3522, and an + * empty parameter-name-list declaration cannot match a definition with an + * argument type that has a default promotion — cc1 rejects the s16 spelling + * with "conflicting types for `func_8017DABC'" (verified: the s16 body is a + * clean standalone MATCH but does not compile in this TU's decl environment, + * the §376/§378 bank-failure class). The twin's TU has NO prior declaration + * at all, which is why s16 was legal there. Widening to s32 is byte-inert: + * both operands of `v1val + param_3` promote to int either way, so the + * prologue keeps its plain `addu $s0,$a2,$zero` and the store still truncates. + * + * 2. `register Quad_8017DABC *p asm("$16")` pins the primitive pointer to $s0, + * reproducing `addiu $s0,$sp,0x10` and the $s0-based tail; the vertex block + * is written through the object `q` so those stores stay $sp-relative, which + * is what the .s does (sh at 0x10/0x12/0x18/0x1A/0x20/0x22/0x28/0x2A($sp) + * against sh 0x4($s0) and the whole 0x30..0x44 tail off $s0). + * + * The two zero-byte `__asm__ __volatile__("")` barriers are scheduling fences: + * the first keeps `p->vz0 = 0` from sinking past the rgb block, the second keeps + * the rgb stores ahead of the first `lw 0xE4($s2)`. func_800178EC takes ONE + * argument (fleet-wide `extern void func_800178EC(s32 a0);`, 7 TUs); $a1 is live + * at the jal only because it still holds the 0x13F used by the v2/v3 stores. + * + * match_one: MATCH, closeness 0, 78/78 instructions. + */ + +typedef struct { + s16 vx0, vy0, vz0, pad0; + s16 vx1, vy1, vz1, pad1; + s16 vx2, vy2, vz2, pad2; + s16 vx3, vy3, vz3, pad3; + s16 u0, v0; + s16 u1, v1; + s16 u2, v2; + s16 u3, v3; + s32 rgb0, rgb1, rgb2, rgb3; + s32 flags; + u8 clut; +} Quad_8017DABC; + +void func_8017DABC(s32 param_1, s32 param_2, s32 param_3, s32 param_4, s32 param_5, s32 param_6) +{ + Quad_8017DABC q; + register Quad_8017DABC *p asm("$16"); + s32 buf[2]; + u16 v1val; + u16 v0val; + s16 yhi; + s16 ylo; + s16 xsel; + s16 t_vx0; + s16 t_vx1; + + func_80015978(param_1 + 4, buf); + v1val = *(u16 *)buf; + v0val = *((u16 *)buf + 1); + + t_vx0 = v1val + param_3; + p = &q; + yhi = v0val + 0x80; + t_vx1 = v1val + param_4; + ylo = v0val - 0x80; + q.vx0 = t_vx0; + q.vy0 = yhi; + q.vx1 = t_vx1; + q.vy1 = ylo; + + switch (param_2) { + case 0: + xsel = v1val - 0xA0; + break; + case 1: + xsel = v1val + 0xA0; + break; + default: + goto skip_store; + } + q.vx2 = xsel; + q.vy2 = yhi; + q.vx3 = xsel; + q.vy3 = ylo; +skip_store: + p->vz0 = 0; + __asm__ __volatile__(""); + + p->rgb0 = p->rgb1 = param_5; + p->rgb2 = p->rgb3 = param_6; + __asm__ __volatile__(""); + p->u0 = *(s32 *)(param_1 + 0xE4) + 0xC00; + p->v0 = 0x100; + + p->u1 = *(s32 *)(param_1 + 0xE4) + 0xC3F; + p->v1 = 0x100; + + p->u2 = *(s32 *)(param_1 + 0xE4) + 0xC00; + p->v2 = 0x13F; + + p->u3 = *(s32 *)(param_1 + 0xE4) + 0xC3F; + p->v3 = 0x13F; + p->clut = 0x36; + p->flags = 0x50000000; + + func_800178EC((s32)p); +} + s32 func_8017DBF4(void) { diff --git a/src/ov_SC06_032/ov_SC06_032_jr_80182890.c b/src/ov_SC06_032/ov_SC06_032_jr_80182890.c index 7014c5618..78e535cf6 100644 --- a/src/ov_SC06_032/ov_SC06_032_jr_80182890.c +++ b/src/ov_SC06_032/ov_SC06_032_jr_80182890.c @@ -3493,7 +3493,167 @@ void func_801838A4(s32 p) { } -INCLUDE_ASM("asm/ov_SC06_032/nonmatchings/ov_SC06_032_jr_80182890", func_80183940); +/* func_80183940 @ ov_SC06_032 (subseg ov_SC06_032_jr_80182890) - 110 ins. MATCH. + * + * GATE: .venv/bin/python tools/match_one.py func_80183940 \ + * --c .run/S70y_1/opus/func_80183940.c \ + * --asm-subdir asm/ov_SC06_032/nonmatchings/ov_SC06_032_jr_80182890 + * + * --------------------------------------------------------------- what it is + * "Sweep-check the object's forward arc, then either punt or hand off." + * 1. Build a rotation matrix from the owner's angle pair + * (*(s32*)(self+0x20))[0x10], [0x12] via func_80049CAC, load it into the + * GTE, and load the owner's translation (that same object + 0x34) as the + * trans matrix. + * 2. RotTransSV the two-entry local SVECTOR table at D_801A6964 -- this + * overlay's own tail.data (asm/ov_SC06_032/data/tail.data.s:23827), 16 + * bytes = {0,0,-20,0} then {0,0,+20,0}, i.e. a back point and a front + * point 20 units either side -- into world space (b, c). + * 3. func_80135888(D_80126B78, D_80126B90, &b, &c) is the fleet's global + * collision probe. Blocked -> punt to func_8012F568(1, 1, self->0xFE, + * 0x50, &c, D_801152A8). Clear -> wind the owner's 0x14 angle back + * 0x100 and try the two handoffs (func_8012CBF4, else func_8012BEE8). + * 4. On the func_8012CBF4 path, re-register via func_80146A6C(6, self, ...) + * and stamp 0x2000 into the func_80132EF4(self, 0x22) record's 0x34. + * 5. Every arm except "func_8012BEE8 returned 0" falls into the shared + * func_8012C218(self) tail -- that is the `j .L80183AD0` at 80183A54 and + * 80183AB8, and the early-out is the beqz at 80183AC8. + * + * ------------------------------------------------------- how it was matched + * §193-A twin remap. The h_norm twin in ov_SC06_018 (subseg _jr_80187AEC, + * 110 ins, banked) is line-for-line identical in shape; the ONLY per-overlay + * symbol is the two-entry SVECTOR table, re-pointed here at this overlay's + * D_801A6964. Every other symbol -- func_80049CAC, RotTransSV, func_80135888, + * func_8012F568, func_8012CBF4, func_80146A6C, func_80132EF4, func_8012BEE8, + * func_8012C218, D_801152A8, D_80126B78, D_80126B90 -- is RESIDENT (fixed VA), + * so the remap is otherwise identity. Symbol set re-checked instruction by + * instruction against this target's own relocation lines (law 1c: match_one + * masks jal/HI16/LO16, so a MATCH proves shape, never identity). + * + * ---------------------------------------------------------------- the levers + * L1 LOCAL DECL ORDER IS THE FRAME LAYOUT. Frame 0x78; the 0x18-byte + * outgoing-arg area that func_80146A6C's 7 args force ends at 0x20, so + * mat lands at sp+0x20 (0x20 bytes), then sv 0x40, b 0x48, c 0x50, + * flag 0x58, with NO gap. Reordering these five shifts every sp + * displacement. `flag` is RotTransSV's shared 3rd/"flag" out-arg and is + * passed to both calls, which is why it is one local and not two. + * + * L2 ZERO FILE-SCOPE FOOTPRINT. This TU already carries, at file scope + * ABOVE the L3496 INCLUDE_ASM site: func_80146A6C L133, func_80135888 + * L578, D_801152A8 L587/L622, func_80049CAC L2621/L2745, RotTransSV + * L2624, func_8012C218 L2710, func_80132EF4 L2744, func_8012BEE8 L2748, + * func_8012CBF4 L2759 -- every spelling below is that same spelling, so + * the block-scope copies are composite-compatible and add nothing. + * func_8012F568 (L3633) and the two s32* pointer externs (L3639-L3640, + * L3848) are declared only BELOW the site and only at block scope, so + * they are kept at block scope here too rather than hoisted. + * + * L3 THE GTE OPS ARE WRITTEN AS RAW INLINE ASM, NOT AS + * `#define gte_SetRotMatrix` / `#define gte_SetTransMatrix`. The + * expansion is token-identical to this TU's OWN macro bodies (L6488 / + * L6502) so codegen is unchanged and match_one confirms 110/110 -- but + * defining those two macros here would plant a second definition ~3000 + * lines AHEAD of the existing pair, giving the draft a file-scope blast + * radius over already-banked code for no codegen benefit. Zero-footprint + * form: the same tokens, no macro. + * + * L4 THE TYPEDEFS ARE BLOCK-SCOPE AND §120-UNIQUIFIED. MTX_80183940 and + * SVEC_80183940 are layout-identical to engine_types.h `MATRIX` + * ({short m[3][3]; long t[3];}, 0x20) and `SVECTOR` + * ({short vx,vy,vz,pad;}, 8). They are written out rather than used by + * name so this draft compiles standalone under match_one, whose context + * does not carry engine_types.h; both names are absent from the TU + * (grep: 0 hits), and at block scope they cannot collide at all. + */ + +void func_80183940(s32 param_1) +{ + typedef struct { short vx, vy, vz, pad; } SVEC_80183940; /* == SVECTOR */ + typedef struct { short m[3][3]; long t[3]; } MTX_80183940; /* == MATRIX */ + + extern s32 *D_80126B78; + extern s32 *D_80126B90; + extern u8 D_801152A8[]; + extern SVEC_80183940 D_801A6964[]; + + extern void func_80049CAC(s32 a0, s32 a1); + extern void RotTransSV(void *a0, void *a1, void *a2); + extern s32 func_80135888(s32 a0, s32 a1, s32 a2, s32 a3); + extern void func_8012F568(s32 a0, s32 a1, s32 a2, s32 a3, s32 a4, s32 a5); + extern s32 func_8012CBF4(s32 a0); + extern s32 func_80146A6C(s32 a0, void *a1, s32 a2, s32 a3, s32 a4, s32 a5, s32 a6); + extern s32 func_80132EF4(s32 a0, s32 a1); + extern s32 func_8012BEE8(s32 a0); + extern void func_8012C218(void *a0); + + MTX_80183940 mat; /* sp+0x20 */ + SVEC_80183940 sv; /* sp+0x40 */ + SVEC_80183940 b; /* sp+0x48 */ + SVEC_80183940 c; /* sp+0x50 */ + SVEC_80183940 flag; /* sp+0x58 */ + s32 iVar; + + sv.vx = *(u16 *)(*(s32 *)(param_1 + 0x20) + 0x10); + sv.vy = *(u16 *)(*(s32 *)(param_1 + 0x20) + 0x12); + sv.vz = 0; + func_80049CAC((s32)&sv, (s32)&mat); + + /* gte_SetRotMatrix(&mat) -- expansion is byte-for-byte this TU's L6488 */ + __asm__ volatile ( + "lw $12, 0( %0 );" + "lw $13, 4( %0 );" + "ctc2 $12, $0;" + "ctc2 $13, $1;" + "lw $12, 8( %0 );" + "lw $13, 12( %0 );" + "lw $14, 16( %0 );" + "ctc2 $12, $2;" + "ctc2 $13, $3;" + "ctc2 $14, $4" + : + : "r"( &mat ) + : "$12", "$13", "$14" ); + /* gte_SetTransMatrix(*(s32 *)(param_1 + 0x20) + 0x34) -- this TU's L6502 */ + __asm__ volatile ( + "lw $12, 20( %0 );" + "lw $13, 24( %0 );" + "ctc2 $12, $5;" + "lw $14, 28( %0 );" + "ctc2 $13, $6;" + "ctc2 $14, $7" + : + : "r"( *(s32 *)(param_1 + 0x20) + 0x34 ) + : "$12", "$13", "$14" ); + + RotTransSV(&D_801A6964[0], &b, &flag); + RotTransSV(&D_801A6964[1], &c, &flag); + + if (func_80135888((s32)D_80126B78, (s32)D_80126B90, (s32)&b, (s32)&c) != 0) { + s16 sVar1 = *(s16 *)(param_1 + 0xFE); + func_8012F568(1, 1, sVar1, 0x50, (s32)&c, (s32)D_801152A8); + } else { + *(u16 *)(*(s32 *)(param_1 + 0x20) + 0x14) = + *(u16 *)(*(s32 *)(param_1 + 0x20) + 0x14) - 0x100; + iVar = func_8012CBF4(param_1); + if (iVar != 0) { + s16 sVar1 = *(s16 *)(param_1 + 6); + s16 sVar2 = *(s16 *)(param_1 + 0xA); + s16 sVar3 = *(s16 *)(param_1 + 0xE); + func_80146A6C(6, (void *)param_1, sVar1, sVar2, sVar3, 0, 0); + iVar = func_80132EF4(param_1, 0x22); + if (iVar != 0) { + *(u16 *)(iVar + 0x34) = 0x2000; + } + } else { + iVar = func_8012BEE8(param_1); + if (iVar == 0) { + return; + } + } + } + func_8012C218((void *)param_1); +} + @@ -11427,4 +11587,52 @@ void func_8018F3E4(void) } -INCLUDE_ASM("asm/ov_SC06_032/nonmatchings/ov_SC06_032_jr_80182890", func_8018FB5C); +void func_8018FB5C(s32 arg0, s32 arg1, s32 arg2) +{ + typedef struct { s32 a; s32 b[4]; } OtBlk_80182890_8018FB5C; /* == engine_types.h OtBlk (0x14); BLOCK scope per §94 type-carry */ + extern OtBlk_80182890_8018FB5C D_800A651C_v[] __asm__("D_800A651C"); /* §200 alias: TU spells it `extern u8 D_800A651C[]` */ + extern void *func_80010A08(s32); + extern void func_8004914C(void *); + extern void func_800491AC(void *); + extern s32 RotTransPers(s32, s32, s32 *, s32 *); + extern u8 D_800AF648; + extern u8 D_800A6518[]; + extern short D_800B9A02; + extern void func_80016638(void *a0, s32 a1, s32 a2); + + s32 sp10; + s32 sp14; + s32 temp_v0_2; + void *temp_v0; + s32 ot; + s32 depth4; + register u16 *bidx __asm__("$8"); + register u32 mask1 __asm__("$7"); + register s32 rgb __asm__("$16"); + register u32 tag0 __asm__("$4"); + + rgb = arg2; + __asm__("" : "=r"(rgb) : "0"(rgb)); /* zero-byte 2nd SET: kills the sched1 birthing boost */ + temp_v0 = func_80010A08(0x10); + *(u8 *)((u8 *)temp_v0 + 3) = 3; + *(s32 *)((u8 *)temp_v0 + 4) = rgb; + *(u8 *)((u8 *)temp_v0 + 7) = 0x42; + func_8004914C(&D_800AF648); + func_800491AC(&D_800AF648); + temp_v0_2 = RotTransPers(arg0, temp_v0 + 8, &sp10, &sp14); + if ((temp_v0_2 > 0) && (sp14 >= 0) && + (RotTransPers(arg1, temp_v0 + 0xC, &sp10, &sp14) > 0) && (sp14 >= 0)) { + /* addPrim(otp, p) == setaddr(p, getaddr(otp)), setaddr(otp, p) */ + mask1 = 0xFFFFFF; + bidx = (u16 *)&D_800B9A02; + depth4 = temp_v0_2 * 4; + tag0 = *(u32 *)temp_v0; + *(u32 *)temp_v0 = (tag0 & 0xFF000000) | + (*(u32 *)(depth4 + D_800A651C_v[*bidx].a) & mask1); + ot = D_800A651C_v[*bidx].a; + *(u32 *)(depth4 + ot) = + (*(u32 *)(depth4 + ot) & 0xFF000000) | ((u32)temp_v0 & mask1); + func_80016638(&D_800A6518[*bidx * 20], temp_v0_2, 1); + } +} + diff --git a/src/ov_SC07_000/ov_SC07_000_jr_8017BEBC.c b/src/ov_SC07_000/ov_SC07_000_jr_8017BEBC.c index 97c843b0e..ac8443051 100644 --- a/src/ov_SC07_000/ov_SC07_000_jr_8017BEBC.c +++ b/src/ov_SC07_000/ov_SC07_000_jr_8017BEBC.c @@ -4174,7 +4174,51 @@ extern void func_8012B200(u8 *a0); } -INCLUDE_ASM("asm/ov_SC07_000/nonmatchings/ov_SC07_000_jr_8017BEBC", func_8017E658); +extern u16 D_80126B66; +extern u16 D_801865DA; +extern u16 D_801865DC; +extern void func_8012AD80(s32 arg0); +extern void func_8017E804(s32 *a0, s32 a1, s32 a2); +extern void func_8017E7A0(s32 arg0); + +void func_8017E658(void *a0) { + s32 d; + s32 t; + s32 want; + + if (*(s16 *)((u8 *)a0 + 0xE) >= 0x2401) { + if (*(s32 *)((u8 *)a0 + 0x18) != 0) { + *(s32 *)((u8 *)a0 + 0x18) -= 0x2000; + if (*(s32 *)((u8 *)a0 + 0x18) < 0) { + *(s32 *)((u8 *)a0 + 0x18) = 0; + } + } + } else { + d = (s16)(D_80126B66 - *(s16 *)((u8 *)a0 + 0xE)); + if (d < 0x100) { + t = *(s32 *)((u8 *)a0 + 0x18); + if (t > 0x20000) { + *(s32 *)((u8 *)a0 + 0x18) = t - 0x4000; + } + } else if (d < 0x1F0) { + func_8017E804((s32 *)a0, 0x70000, 0x4000); + } else { + func_8017E804((s32 *)a0, 0xA0000, 0x4000); + } + } + want = (s16)(*(u16 *)((u8 *)&D_801865DA + (*(s16 *)((u8 *)a0 + 0xFE) << 3)) - 0x50); + if (*(s16 *)((u8 *)a0 + 0xA) < want) { + *(s16 *)((u8 *)a0 + 0xA) = *(s16 *)((u8 *)a0 + 0xA) + 1; + } else if (want < *(s16 *)((u8 *)a0 + 0xA)) { + *(s16 *)((u8 *)a0 + 0xA) = *(s16 *)((u8 *)a0 + 0xA) - 1; + } + func_8012AD80((s32)a0); + if (*(s16 *)((u8 *)&D_801865DC + (*(s16 *)((u8 *)a0 + 0xFE) << 3)) < *(s16 *)((u8 *)a0 + 0xE) + 0x180) { + *(s16 *)((u8 *)a0 + 0xFE) = *(s16 *)((u8 *)a0 + 0xFE) + 1; + } + func_8017E7A0((s32)a0); +} + extern s32 func_8004787C(s32 a0); @@ -4995,7 +5039,58 @@ void func_8018000C(s32 arg0) { } -INCLUDE_ASM("asm/ov_SC07_000/nonmatchings/ov_SC07_000_jr_8017BEBC", func_8018006C); +#include "common.h" + +extern void func_8017DF94(s32 arg0, s16 *arg1, s32 arg2); +extern void func_801805C0(); +extern void func_8012C218(void *arg0); +extern s32 func_8004787C(s32 a0); +extern void func_80180200(void *a0); +extern void func_801802E4(void *a0); +extern void func_8012E014(s32 a0); +extern u16 D_801865DA; + +void func_8018006C(void *a0) { + struct { s16 x; s16 y; s16 z; } sp; + s32 idx; + u16 y; + + if (*(s16 *)((u8 *)a0 + 0xE) - 0x80 < *(s16 *)(*(s32 *)((u8 *)a0 + 0x64) + 0xE)) { + func_8017DF94(*(s16 *)((u8 *)a0 + 0x70), (s16 *)&sp, 0); + func_801805C0(a0, *(s16 *)((u8 *)a0 + 0x102)); + func_8012C218(a0); + return; + } + idx = *(s16 *)((u8 *)a0 + 0x70); + y = *(u16 *)((u8 *)a0 + 0xA); + if (*(s16 *)((u8 *)a0 + 0x100) >= 0x400) { + if ((s16)y < -0x1FF) { + *(s16 *)((u8 *)a0 + 0xA) = y + 8; + } else { + func_8017DF94(idx, (s16 *)&sp, 0); + func_8012C218(a0); + return; + } + } else { + *(s16 *)((u8 *)a0 + 0x100) = *(s16 *)((u8 *)a0 + 0x100) + 0x10; + *(s16 *)((u8 *)a0 + 0xA) = *(u16 *)((u8 *)a0 + 0xA) + (func_8004787C(*(s16 *)((u8 *)a0 + 0x100)) >> 9); + } + if (*(s16 *)((u8 *)a0 + 0x106) != 0) { + *(s16 *)((u8 *)a0 + 0x106) = *(s16 *)((u8 *)a0 + 0x106) - 1; + if ((*(s16 *)((u8 *)a0 + 0x106) & 3) == 0) { + func_80180200(a0); + } + } + func_801802E4(a0); + sp.z = 0; + sp.x = 0; + sp.y = *(u16 *)((u8 *)a0 + 0xA) - *(u16 *)((u8 *)&D_801865DA + (*(s16 *)((u8 *)a0 + 0x70) << 3)); + func_8017DF94(*(s16 *)((u8 *)a0 + 0x70), (s16 *)&sp, 1); + if (*(u8 *)((u8 *)a0 + 0x74) != 0) { + func_8012E014((s32)a0); + } +} + extern void (*D_801867DC[])(void); @@ -5120,7 +5215,40 @@ void func_80180584(void *a0) { } -INCLUDE_ASM("asm/ov_SC07_000/nonmatchings/ov_SC07_000_jr_8017BEBC", func_801805C0); +#include "common.h" + +extern s32 D_801867E8[]; +extern s32 D_80186820[]; +extern void *D_80186830[]; +extern u16 D_80186848[]; +extern u16 D_80186850[]; +extern s32 func_8012C658(s32 a0, s32 a1, s32 a2); +extern void func_8018091C(void *a0, void *a1); +extern void func_80180704(void *a0); +extern void func_80180790(s32 self); +extern void func_8002D4C8(s32 a0, s32 a1); + +void func_801805C0(s32 arg0, s32 arg1) { + s32 idx; + s32 i; + s32 rec; + char *base; + + idx = D_801867E8[arg1]; + base = (char *)D_80186830[idx]; + for (i = 0; i < D_80186820[idx]; i++) { + rec = func_8012C658(0x3C4, (idx << 8) + i, arg0); + if (rec != 0) { + func_8018091C((void *)rec, base + i * 0xC); + *(u16 *)(rec + 0xA) += *(u16 *)(arg0 + 0xA) - D_80186848[idx]; + *(u16 *)(rec + 0xE) += *(u16 *)(arg0 + 0xE) - D_80186850[idx]; + func_80180704((void *)rec); + func_80180790(rec); + } + } + func_8002D4C8(0xC02, 0); +} + extern s32 rand(void); diff --git a/src/ov_SC07_001/ov_SC07_001_jr_8017BEBC.c b/src/ov_SC07_001/ov_SC07_001_jr_8017BEBC.c index 33adc0230..9fc0f870e 100644 --- a/src/ov_SC07_001/ov_SC07_001_jr_8017BEBC.c +++ b/src/ov_SC07_001/ov_SC07_001_jr_8017BEBC.c @@ -5518,7 +5518,69 @@ void func_80180850(void *a0) { } -INCLUDE_ASM("asm/ov_SC07_001/nonmatchings/ov_SC07_001_jr_8017BEBC", func_8018088C); +// @class: loop-invariant sign-extension order +// @stuck: none — MATCH (104 ins, match_one). +// The whole residual was a 4-instruction REGALLOC-PERM in the loop preheader: +// the two hoisted `sll/sra 16` pairs (sign-extending `base` and `limit`) came out +// in the wrong order, taking $s5/$s4 with them. -dS shows BOTH sign-extensions are +// created by loop.c `move_movables` (insns 218-221, hoisted out of the loop), NOT by +// the pre-loop assignments — so `base`/`limit` stay HImode pseudos and NO statement +// permutation outside the loop can reorder them (verified: 5 orderings, all closeness 4; +// an `asm("")` fence costs a nop, register pins push the extends back INTO the loop). +// The order is the order `move_movables` FINDS them, i.e. the operand order of the +// loop's own comparison. Writing the guard as `limit > sum` instead of `sum < limit` +// expands `limit` first, hoists its extend first, and closes 4 -> 0. Same slt, same +// 104 instructions. (Extends §49/§167-34: the LUID dial reached through the COMPARISON +// OPERAND ORDER when the movable is loop-hoisted.) +// +// `base` must stay `u16` (the raw lhu is what feeds `addu $v0,$v0,$s3` at 0xA($s0)); +// `(s16)base` in the guard is the separate $s4. `D_800B99DA % 5` uses the UNSIGNED +// 0xCCCCCCCD magic + `andi 0xFFFF` because gcc-2.7.2 c-typeck shortens `unsigned short +// % const` back to unsigned short (build_binary_op's TRUNC_MOD_EXPR `shorten`). + +extern s32 D_80185678[]; +extern u16 D_80185688[]; +extern s16 D_801AAD6E[]; +extern u16 D_800B99DA; +extern s32 func_8012913C(s32 a0); +extern s32 rand(void); + +void func_8018088C(s32 a0) +{ + s32 idx; + s32 rec; + u16 base; + s16 limit; + s32 ent; + + idx = *(s16 *)(a0 + 0x70); + rec = D_80185678[idx]; + if (rec == 0) { + return; + } + if (D_800B99DA % 5 != 0) { + return; + } + if (*(s32 *)(a0 + 0x3C) == *(s32 *)(a0 + 8)) { + return; + } + base = *(u16 *)&D_801AAD6E[idx * 4]; + limit = D_80185688[idx]; + while (*(s16 *)(rec + 6) != -1) { + if (limit > *(s16 *)(rec + 2) + (s16)base) { + ent = func_8012913C(0x22); + if (ent != 0) { + *(u16 *)(ent + 6) = *(u16 *)rec; + *(u16 *)(ent + 0xA) = *(u16 *)(rec + 2) + base; + *(u16 *)(ent + 0xE) = *(u16 *)(rec + 4); + *(u16 *)(ent + 0x34) = (rand() % 4095 + 0x6000) & 0xFFF0; + *(u16 *)(*(s32 *)(ent + 0x20) + 0x2C) = 0xC00C; + } + } + rec += 8; + } +} + extern s32 func_8012C1B8(void); extern void func_8012CAE4(void *a0); diff --git a/src/ov_SC07_006/ov_SC07_006_jr_8017BEBC.c b/src/ov_SC07_006/ov_SC07_006_jr_8017BEBC.c index 1bd434051..99896ad7b 100644 --- a/src/ov_SC07_006/ov_SC07_006_jr_8017BEBC.c +++ b/src/ov_SC07_006/ov_SC07_006_jr_8017BEBC.c @@ -5154,7 +5154,50 @@ void func_801823F8(s32 t) } -INCLUDE_ASM("asm/ov_SC07_006/nonmatchings/ov_SC07_006_jr_8017BEBC", func_801826E0); +#include "common.h" + +extern s32 D_801BFCE4; +extern u8 D_8018E4D4[]; +extern u8 D_8018DF10; +extern void func_8017D8A4(s32 a0, void *a1, void *a2, s32 a3, s32 a4); +extern void func_80143C74(s32 a0, s32 a1); +extern void func_80128EA8(s32 a0, s32 a1, s32 a2); +extern s32 func_8012C658(s32 a0, s32 a1, s32 a2); +extern s32 rand(void); + +void func_801826E0(s32 param_1, s32 param_2) { + s32 *q; + s32 work; + s32 rnd; + s32 t; + s32 u; + + q = &D_801BFCE4; + func_8017D8A4(*q + 0xC, D_8018E4D4, D_8018E4D4 + 0x24, param_2, 1); + func_8017D8A4(*q + 0x18, D_8018E4D4 + 0xC, D_8018E4D4 + 0x30, param_2, 1); + func_8017D8A4(*q + 0x84, D_8018E4D4 - 0xC, D_8018E4D4 + 0x18, param_2, 1); + work = ((s32 (*)())func_80143C74)(param_1, 0); + if (work != 0) { + *(u16 *)(work + 0xE) = *(u16 *)(work + 0xE) - *(u16 *)(D_801BFCE4 + 0x84); + rnd = rand(); + t = *(u16 *)(D_801BFCE4 + 0x86) - 0x20; + *(u16 *)(work + 0xA) = *(u16 *)(work + 0xA) + (t + (rnd & 0x3F)); + rnd = rand(); + u = *(u16 *)(D_801BFCE4 + 0x88) - 0x20; + *(u16 *)(work + 6) = *(u16 *)(work + 6) + (u + (rnd & 0x3F)); + rnd = rand() & 0xFF; + *(u32 *)(work + 0x10) = (rnd - 0x80) << 10; + rnd = rand() & 3; + *(u16 *)(work + 0x16) = -(rnd + 4); + *(u32 *)(work + 0x18) = -((rand() & 0xFF) << 11); + func_80128EA8(*(u32 *)(work + 0x20), (void *)(work + 0xD0), &D_8018DF10); + func_8012C658(0x3AC, 6, work); + func_8012C658(0x3AC, 6, work); + func_8012C658(0x3AC, 6, work); + func_8012C658(0x3AC, 6, work); + } +} + /* func_801828A4 — banked from the S40 wave-1 draft. diff --git a/src/ov_SC07_007/ov_SC07_007_jr_8017BEBC.c b/src/ov_SC07_007/ov_SC07_007_jr_8017BEBC.c index 63ea2941b..5a3495ea3 100644 --- a/src/ov_SC07_007/ov_SC07_007_jr_8017BEBC.c +++ b/src/ov_SC07_007/ov_SC07_007_jr_8017BEBC.c @@ -7929,7 +7929,41 @@ void func_801827A4(void *a0) { } -INCLUDE_ASM("asm/ov_SC07_007/nonmatchings/ov_SC07_007_jr_8017BEBC", func_801827E0); +extern u8 D_80199AC4[]; +extern void func_8002D844(s32 a0); +extern void func_8002D4C8(s32 a0, s32 a1); + +/* Copies the 0x944B-byte SEQ/MIDI blob at 0x80199D4C (== D_80199AC4 + 0x288, + * starts "MThd") up to the 0x801F0000 scratch buffer, then opens + plays it. + * + * Two levers were needed over the numeric warm-start body: + * (1) the loop test is SIGNED (slt, not sltu): 0x801F944A is an `unsigned int` + * in C89, so a bare `p <= 0x801F944A` converts p and emits sltu. Cast it. + * (2) the `lui $at / addu $at,$at,$v1 / lbu %lo($at)` triple is an ASPSX + * big-offset macro expansion, and its operand order is a TELL for how the + * offset was spelled in the source (maspsx __init__.py, load path): + * numeric offset -> addu $at,,$at + * SYMBOLIC addend -> addu $at,$at, + * Fleet census: 26/26 numeric-`lui $at` sites use the first order, 569/569 + * `%hi(sym)` sites use the second. This target has a numeric-looking `lui + * $at,0xFFFB` but the SECOND order — the lone outlier fleet-wide — because + * the addend is a symbol MINUS a constant, which splat cannot name. So the + * source indexes a real data symbol; `*(u8 *)(p - 0x562B4)` cannot emit it. + * gcc folds `sym[p - K]` into `%hi(sym+(-K))`, giving 0x80199AC4 + 0x288 - + * 0x801F0000 = -0x562B4 -> lui 0xFFFB / lbu -0x62B4. Byte-verified with the + * symbol resolved: all 23 words identical, relocations included. */ +void func_801827E0(void) { + s32 p = 0x801F0000; + + do { + *(u8 *)p = D_80199AC4[(p - 0x801F0000) + 0x288]; + p += 1; + } while (p <= (s32)0x801F944A); + + func_8002D844(0x801F0000); + func_8002D4C8(0x18E, 0); +} + extern void func_8001ABBC(s32 a0, s32 a1, void *a2, s32 a3, s32 sp10); extern CdFileLoc cdFileLocTable[]; diff --git a/src/ov_SC07_010/ov_SC07_010_jr_8017AE2C.c b/src/ov_SC07_010/ov_SC07_010_jr_8017AE2C.c index 85e559b45..8c049ac3f 100644 --- a/src/ov_SC07_010/ov_SC07_010_jr_8017AE2C.c +++ b/src/ov_SC07_010/ov_SC07_010_jr_8017AE2C.c @@ -4819,7 +4819,39 @@ void func_8017E520(s32 a0) { } -INCLUDE_ASM("asm/ov_SC07_010/nonmatchings/ov_SC07_010_jr_8017AE2C", func_8017E5B8); +void func_8017E5B8(s32 a0, s32 a1) { + struct P8 { s16 unk0; u8 pad[6]; }; + struct P4 { u16 unk0; u16 pad; }; + struct EntA { u8 pad[0xDC]; s32 unkDC; u8 pad2[0x1C]; u16 unkFC; }; + extern struct P8 D_80185D28[]; + extern struct P8 D_80185D2A[]; + extern struct P8 D_80185D2C[]; + extern struct P8 D_80185D2E[]; + extern struct P4 D_80185D5A[]; + extern s32 func_80146A6C(s32, void*, s32, s32, s32, s32, s32); + extern void func_80015954(s32 a0, s32 a1); + extern s32 func_8012C588(s32 a0, s32 a1); + struct EntA *p; + s32 i; + + func_80146A6C(0x1E, (void *)a0, D_80185D28[a1].unk0, D_80185D2A[a1].unk0, + D_80185D2C[a1].unk0, 0, 0); + for (i = 0; i < 4; i++) { + p = (struct EntA *)func_8012C588(0x3A6, 0); + if (p != 0) { + func_80015954((s32)&D_80185D28[a1], (s32)p + 4); + p->unkFC = D_80185D5A[D_80185D2E[a1].unk0].unk0 + (i * 170 - 341); + p->unkDC = 0; + } + p = (struct EntA *)func_8012C588(0x3A6, 0); + if (p != 0) { + func_80015954((s32)&D_80185D28[a1], (s32)p + 4); + p->unkFC = D_80185D5A[D_80185D2E[a1].unk0].unk0 + (i * 170 - 227); + p->unkDC = 1; + } + } +} + extern void func_80015954(s32 a0, s32 a1); extern s32 func_8012C588(s32 a0, s32 a1); @@ -5589,7 +5621,75 @@ void func_8017F798(void *arg0) { } -INCLUDE_ASM("asm/ov_SC07_010/nonmatchings/ov_SC07_010_jr_8017AE2C", func_8017F860); +typedef struct { + s16 state; /* 0x00 */ + s16 timer; /* 0x02 */ + s16 unk4; /* 0x04 */ + s16 unk6; /* 0x06 */ + s16 unk8; /* 0x08 */ + s16 unkA; /* 0x0A */ + s16 unkC; /* 0x0C */ + s16 unkE; /* 0x0E */ + s16 unk10; /* 0x10 */ + s16 unk12; /* 0x12 */ + s32 unk14; /* 0x14 */ + u16 unk18; /* 0x18 */ + u16 unk1A; /* 0x1A */ +} Blk1C; + +extern void func_8017FA24(void *a0); +extern void func_8017FA50(s32 a0, void *s0); + +void func_8017F860(s32 arg0) { + extern u8 D_801A7F8C[]; + Blk1C *rec; + s32 i; + + for (i = 0; i < 0x10; i++) { + rec = &((Blk1C *)D_801A7F8C)[i]; + rec->unk6 = (rec->unk6 - 0x2D) & 0xFFF; + switch (rec->state) { + case 0: + if (rec->timer != 0) { + rec->timer--; + if (rec->timer == 0) { + rec->unk4 = -0x155; + rec->state++; + } + } + break; + case 1: + rec->unkC += 0x100; + if (rec->unkC > 0x1000) { + rec->unkC = 0x1000; + } + rec->unkE = rec->unk10 = rec->unkC; + func_8017FA50(arg0, rec); + if (rec->unkC == 0x1000) { + rec->timer = 0x1E; + rec->state++; + } + break; + case 2: + func_8017FA24(rec); + func_8017FA50(arg0, rec); + rec->timer--; + if (rec->timer == -1) { + rec->state++; + } + break; + case 3: + func_8017FA24(rec); + func_8017FA50(arg0, rec); + if (rec->unkC == 0) { + rec->timer = 0; + rec->state = 0; + } + break; + } + } +} + void func_8017FA24(void *a0) { extern s32 D_80185C70[];