From 7b256d1c35dbfd8987b8ea542e72686d957c6013 Mon Sep 17 00:00:00 2001 From: Drew T <50529377+Druthulu@users.noreply.github.com> Date: Wed, 2 Sep 2026 04:45:11 -0600 Subject: [PATCH] =?UTF-8?q?feat(decomp):=20parallel=20gate=20=E2=80=94=203?= =?UTF-8?q?=20fns=20across=203=20binaries=20(10=20workers)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ov_SC01_009 func_8017FFD0 ov_SC03_013 func_8017E6F4 ov_SC07_007 func_80182184 --- src/ov_SC01_009/ov_SC01_009_jr_8017E590.c | 188 +++++++++++++++++++++- src/ov_SC03_013/ov_SC03_013_jr_8017C730.c | 136 +++++++++++++++- src/ov_SC07_007/ov_SC07_007_jr_8017BEBC.c | 117 +++++++++++++- 3 files changed, 438 insertions(+), 3 deletions(-) diff --git a/src/ov_SC01_009/ov_SC01_009_jr_8017E590.c b/src/ov_SC01_009/ov_SC01_009_jr_8017E590.c index 9dee4dd28..6a5305569 100644 --- a/src/ov_SC01_009/ov_SC01_009_jr_8017E590.c +++ b/src/ov_SC01_009/ov_SC01_009_jr_8017E590.c @@ -4063,7 +4063,193 @@ s32 func_8017FED8(s32 a0) { } -INCLUDE_ASM("asm/ov_SC01_009/nonmatchings/ov_SC01_009_jr_8017E590", func_8017FFD0); +#include "common.h" + +/* func_8017FFD0 — ov_SC01_009 / ov_SC01_009_jr_8017E590, 261 ins, frame 0x38. + * NOT the ov_SC03_108 function of the same name (overlays share VRAM; §238 homonym trap — + * the earlier gate_lane batch for THIS binary carried the SC03_108 state-machine body, which is + * why it failed). Body recovered from .run/pool_1/opus (S71 attempt 1) and re-verified here: + * match_one MATCH 261/261; 90-symbol relocation order identical to the .s (law 1c); spliced into + * a copy of the destination TU it compiles clean at offset 0x1A40 / 0x414 bytes with 0 masked + * mismatches and identical relocations. Two law-2 points: func_80143994 is declared `void` by the + * neighbour func_8017FA14, so its value is taken through a cast function pointer exactly as that + * neighbour does for func_8012C1B8; D_801EDA4C is stored before D_801EDA6C so the D_801F32E0 / + * D_801F32DC loads come out in the target's symbol order (match_one's mask cannot see this). */ +s32 func_8017FFD0(s32 param_1) +{ + extern u8 D_801ED4B4[]; + extern s32 D_801ED4E4[]; + extern void func_8012CAE4(void *a0); + extern s32 func_8012C1B8(void); + extern void func_8001C214(s32 a0, s32 a1); + extern void func_8012A828(s32 a0, void *a1); + extern void func_8001C1E4(void *a0, s32 a1); + /* law 2: the TU (func_8017FA14) declares this void; it returns a value here, so go + * through a cast function pointer exactly as that neighbour does for func_8012C1B8. */ + extern void func_80143994(s32 a0, s32 a1); + extern void func_80181724(s32 a0, s32 a1); + extern s32 func_800291B4(s32 arg); + extern void func_8001C2C4(s32 a0); + extern s32 func_8018057C(s32 a0); + extern void MoveImage(void *a0, s32 a1, s32 a2); + extern u8 D_801ECCDC[]; + extern u8 D_801EB39C[]; + extern u8 D_801ED25C[]; + extern u8 D_801ED34C[]; + extern void (*D_80182CA4[])(void *); + extern s16 D_801EE880; + extern s16 D_801EE8D8; + extern s32 D_801EED54[]; + extern s32 D_801ED5EC[]; + extern s32 D_801ED5F0[]; + extern s32 D_801ED7F0[]; + extern s32 D_801EDAA8; + extern u16 D_801EE740[]; + extern u8 D_80183A40[]; + extern void (*D_8018670C[])(void *); + extern u8 *D_801ED8D8; + extern u16 D_801F32D0; + extern u16 D_801F32D4; + extern s32 D_801F32DC; + extern s32 D_801F32E0; + extern s32 D_801F32E4; + extern s32 *D_801F32E8; + extern s32 D_801EDA4C; + extern s32 D_801EDA6C; + + register s32 ret __asm__("$2"); + s32 ptr; + s32 i70a; + s32 i70b; + s32 tmp1; + s32 tmp2; + s16 t; + u16 *q; + u16 buf[6]; + + if (D_801ED4E4[D_801ED4B4[*(s16 *)(param_1 + 0x70)]] == 0) { + func_8012CAE4((void *)param_1); + return ret; + } + ptr = func_8012C1B8(); + *(s32 *)(param_1 + 0x20) = ptr; + if (ptr == 0) { + return ret; + } + + if (*(s16 *)(param_1 + 0x70) >= 0x28) { + func_8001C214(ptr, D_801ED4E4[*(s16 *)(param_1 + 0x70)]); + if (*(s16 *)(param_1 + 0x70) == 0x28) { + func_8012A828(param_1, D_801ECCDC); + *(s32 *)(param_1 + 0xCC) = (s32)&D_801EE880; + *(s32 *)(param_1 + 0xD8) = ((s32 (*)(s32, s32))func_80143994)(param_1, 0x3000); + ret = 7; + goto STORE; + } + if (*(s16 *)(param_1 + 0x70) == 0x29) { + D_801F32D4 = 0; + func_8012A828(param_1, D_801EB39C); + *(s32 *)(param_1 + 0xCC) = (s32)&D_801EE8D8; + *(s32 *)(param_1 + 0xD8) = ((s32 (*)(s32, s32))func_80143994)(param_1, 0x3000); + ret = 0xA; + goto STORE; + } + if (*(s16 *)(param_1 + 0x70) == 0x2A) { + func_8001C1E4((void *)*(s32 *)(param_1 + 0x20), + *(s32 *)(*(s32 *)(param_1 + 0x64) + 0x20)); + func_8012A828(param_1, D_80182CA4); + *(s32 *)(*(s32 *)(param_1 + 0x20) + 4) = 0x80000000; + __asm__ volatile(""); + ret = 0x16; + goto STORE; + } + __asm__ volatile(""); + if (*(s16 *)(param_1 + 0x70) == 0x2B) { + func_8012A828(param_1, D_801ED25C); + *(s32 *)(param_1 + 0xCC) = D_801EED54[*(s16 *)(param_1 + 0xFC)]; + *(s32 *)(param_1 + 0xD8) = ((s32 (*)(s32, s32))func_80143994)(param_1, 0x2000); + ret = 0x19; + goto STORE; + } + __asm__ volatile(""); + if (*(s16 *)(param_1 + 0x70) == 0x2C) { + func_8012A828(param_1, &D_80182CA4[0]); + *(s32 *)(param_1 + 0x1C) = 0x5A; + *(s32 *)(param_1 + 0xD8) = ((s32 (*)(s32, s32))func_80143994)(param_1, 0x1800); + ret = 0x1C; + goto STORE; + } + __asm__ volatile(""); + if (*(s16 *)(param_1 + 0x70) == 0x2D) { + func_8012A828(param_1, D_801ED34C); + *(s32 *)(param_1 + 0x1C) = 0x58; + ret = 0x1D; + goto STORE; + } + } + + i70a = *(s16 *)(param_1 + 0x70); + __asm__ volatile("" ::: "memory"); + i70b = *(s16 *)(param_1 + 0x70); + D_801F32D0 = 0; + { + u16 idx = D_801ED4B4[i70a]; + *(u16 *)(param_1 + 0x10A) = idx; + /* §137/§158 allocno-priority lever, from the cc1 -dl dump of this block: + * the six pseudos here form a CONFLICT PATH idx-reload-(idx<<3)-(r<<3)-tmp1-tmp2, + * so first-fit local-alloc 2-colours it and the ONLY question is which class takes + * $v0. All of {162,165,169,77,78} tied at pri 6666 (refs 2 / len 3), so the tie fell + * to qty number and the reload won -> whole path inverted ($v0<->$v1, 11 insns off). + * idx (reg 156) sat at 6000 (refs 3 / len 5). Folding an "r"(idx) input INTO the + * existing memory barrier adds one ref for ZERO emitted bytes -> pri 16000, so idx is + * allocated first, takes $v0, and the rest of the path falls out as the target. + * It must be MERGED into the barrier, not a second asm: a separate + * __asm__ volatile("" : : "r"(idx)) adds an insn to the sched1 stream and stays at 11. + */ + __asm__ volatile("" : : "r"(idx) : "memory"); + tmp1 = D_801ED5F0[idx * 2]; + tmp2 = D_801ED5EC[*(s16 *)(param_1 + 0x10A) * 2]; + } + D_801F32DC = tmp1; + D_801F32E0 = tmp2; + *(u16 *)(param_1 + 0x108) = + (func_800291B4(D_80183A40[i70b]) & 0xFF) - 1; + func_8001C2C4(*(s32 *)(param_1 + 0x20)); + if (*(s16 *)(param_1 + 0x108) == 0) { + D_801ED8D8 = (u8 *)D_8018670C; + func_80181724(*(s16 *)(param_1 + 0x70), 2); + *(u16 *)(param_1 + 0x108) = 1; + t = *(s16 *)(param_1 + 0x70); + if ((t == 9) || (t == 0x1B) || (t == 0x21)) { + D_801F32D0 = 1; + } + } + D_801F32E4 = D_801ED7F0[*(s16 *)(param_1 + 0x10A)]; + D_801F32E8 = &D_801EDAA8; + if (D_801F32E4 == 0) { + D_801F32E4 = func_8018057C(param_1); + } + D_801EDA4C = D_801F32E0; + D_801EDA6C = D_801F32DC; + D_801F32E8[1] = D_801F32E4; + q = &D_801EE740[*(s16 *)(param_1 + 0x10A) * 4]; + buf[0] = *q++; + buf[1] = *q++; + buf[2] = 8; + buf[3] = 0x28; + MoveImage(buf, 0x1C0, 0x1B8); + buf[0] = *q++; + buf[1] = *q++; + buf[2] = 0x10; + buf[3] = 1; + MoveImage(buf, 0x160, 0x1D0); + ret = *(u16 *)(param_1 + 2) + 1; + +STORE: + *(u16 *)(param_1 + 2) = ret; + return ret; +} + void func_801803E4(s32 param_1) { diff --git a/src/ov_SC03_013/ov_SC03_013_jr_8017C730.c b/src/ov_SC03_013/ov_SC03_013_jr_8017C730.c index 58b3e3a17..f4c4a9786 100644 --- a/src/ov_SC03_013/ov_SC03_013_jr_8017C730.c +++ b/src/ov_SC03_013/ov_SC03_013_jr_8017C730.c @@ -3821,7 +3821,141 @@ void func_8017E528(s32 param_1, s32 param_2, s16 *param_3) { } -INCLUDE_ASM("asm/ov_SC03_013/nonmatchings/ov_SC03_013_jr_8017C730", func_8017E6F4); +#include "common.h" + +// @class: cse-extended-block / jump1-collapse +// @stuck: none -- MATCH 182/182 (S71a fable). Three coupled levers, all byte-proven here: +// (1) A `register __asm__("$3")` pin on an if/else SELECT result blocks jump.c:728's +// `x = b; if (c) x = a;` collapse: expand emits the pinned arm as ior-into-pseudo + copy +// (2 insns), so prev_active_insn(x=a) is not the condjump and the diamond survives to +// reorg, which sinks the ori into the beq slot and inverts the branch (bne/ori/move vs the +// target's beq/addu/ori). Result pseudo must stay UNPINNED for the target shape. +// (2) With all three selects collapsed, every join is a 1-use label with no barrier, so cse's +// skip_blocks (AROUND) path walks from the first `D_80184D2C[idx]` read through the ratan2 +// calls to the second and merges the two `(set reg, symbol_ref)` pseudos -> the address +// lives across the calls in $s3 (+2 insns, frame 0x38). A real DIAMOND between the two uses +// ends the cse block (TAKEN follow falls into the join label): the vol clamp spelled as +// if/else (its else arm `v | 0x1000` is an IOR, so jump.c cannot collapse it). +// (3) The diamond's arms would tie vol to v's dying $v1; pinning vol to $5 puts both arms in +// $a1 (the arg register) and reorg fills the bgez slot from the mostly-true target thread. +// The abs step must be `if ((s16)w < 0) w = -w; e = w;` (single sext after the join; reorg +// copies the redundant-skipped `sra` into the slot). `e = w; if (e<0) e=-e;` cross-jumps here. + +extern u16 D_80126B5E; +extern u16 D_80126B62; +extern u16 D_80126B66; +extern u16 D_80126B96; +extern s16 D_80126B98; +extern u16 D_80184D1C[]; +extern u16 D_80184D1E[]; +extern s16 D_80184D2C[]; +extern u8 D_80184CEC; +extern s32 ratan2(s32 a0, s32 a1); +extern s32 func_800291B4(s32 arg); +extern s32 func_80132EF4(s32 a0, s32 a1); +extern void func_80129374(s32 a0, s32 a1); +extern void func_8017EBA0(s32 a0, s32 a1); +extern s32 func_8017EAB0(s32 a0); +extern s32 func_8017EA08(void *a0); +extern void func_8002D59C(s32 a0, u16 a1, s32 a2); + +void func_8017E6F4(s32 a0) { + s32 s0 = a0; + short d; + s16 t; + s32 p; + s32 ang1, ang2; + short e; + s32 tmp; + register s32 lo __asm__("$4"); /* pin: splits the CSE temp from the select result */ + register s32 vol __asm__("$5"); + s32 w; + s32 v; + s32 r; + u16 *p96; + + *(s16 *)(s0 + 0x104) = *(u16 *)(s0 + 0x102); + *(s16 *)(s0 + 0x102) = 0; + *(s32 *)(s0 + 0x1C) = *(s32 *)(s0 + 0x1C) + 1; + + d = D_80126B62 - *(u16 *)(s0 + 0xA); + if (d < 0) { + d = -d; + } + if (d >= 0x181) { + return; + } + + t = *(s16 *)(s0 + 0xFE); + if (t != 0) { + if ((func_800291B4(0xCE) & 0xFF) < t) { + *(s16 *)(s0 + 0x5C) = 0x800; + if (*(s16 *)(s0 + 0xFE) != 0) { + goto have_flag; + } + } else { + *(s16 *)(s0 + 0x5C) = 0; + return; + } + } + if (*(s32 *)(s0 + 0x1C) & 0x20) { + return; + } +have_flag: + + if (*(s32 *)(s0 + 0x1C) & 1) { + p = func_80132EF4(s0, 0x22); + if (p != 0) { + func_80129374(p, s0); + *(u16 *)(*(s32 *)(p + 0x20) + 0x18) = D_80184D1C[*(s16 *)(s0 + 0x100) * 2]; + *(u16 *)(*(s32 *)(p + 0x20) + 0x1A) = D_80184D1E[*(s16 *)(s0 + 0x100) * 2]; + } + } + + if (*(s16 *)(s0 + 0x100) == 2) { + func_8017EBA0(s0, (s32)&D_80184CEC); + } + + if (*(u16 *)&D_80184D2C[*(s16 *)(s0 + 0x70)] != 0) { + tmp = ratan2(-*(s16 *)&D_80126B66, *(s16 *)&D_80126B5E) - 0x400; + v = tmp & 0xFFF; + if (tmp & 0x800) { ang1 = v | 0xF000; } else { ang1 = v; } + tmp = ratan2(-*(s16 *)(s0 + 0xE), *(s16 *)(s0 + 6)) - 0x400; + lo = tmp & 0xFFF; + if (tmp & 0x800) { ang2 = lo | 0xF000; } else { ang2 = lo; } + tmp = ang2 - ang1; + lo = tmp & 0xFFF; + if (tmp & 0x800) { w = lo | 0xF000; } else { w = lo; } + if ((s16)w < 0) { + w = -w; + } + e = w; + if (e < 0x200) { + v = ((0x200 - e) >> 2) - (d >> 2); + if (v < 0) { vol = 0x1000; } else { vol = v | 0x1000; } + func_8002D59C(*(u16 *)&D_80184D2C[*(s16 *)(s0 + 0x70)], vol, (u16)*(s16 *)(s0 + 0x70)); + *(s16 *)(s0 + 0x102) = 1; + } + } + + t = *(s16 *)(s0 + 0xFE); + if (t == 0) { + if ((*(s32 *)(s0 + 0x1C) & 0x1F) < 8) { + return; + } + } + if (t == 2) { + r = func_8017EAB0(s0); + } else { + r = func_8017EA08((void *)s0); + } + if ((s16)r != 0) { + p96 = &D_80126B96; + D_80126B98 = 0xC; + *p96 |= 0x4000; + } +} + extern void (*D_80184D50[])(void); diff --git a/src/ov_SC07_007/ov_SC07_007_jr_8017BEBC.c b/src/ov_SC07_007/ov_SC07_007_jr_8017BEBC.c index 4fd4325c2..7687b1a1e 100644 --- a/src/ov_SC07_007/ov_SC07_007_jr_8017BEBC.c +++ b/src/ov_SC07_007/ov_SC07_007_jr_8017BEBC.c @@ -7951,7 +7951,122 @@ void func_80181F4C(s32 arg0, s32 arg1) { #undef F4C_IDX -INCLUDE_ASM("asm/ov_SC07_007/nonmatchings/ov_SC07_007_jr_8017BEBC", func_80182184); +/* func_80182184 — links TWO 0x18-byte double-buffered sprite packets + (D_801C78D4[n][D_800B9A02] and D_801C7934[n][D_800B9A02], n = arg1 & 0xFF) + into OT bank D_800A6618[D_800B9A02]. It is the two-block sibling of the + already-banked func_80181F4C, 20 lines up in this TU (§330 neighbour-shape). + + Levers (all six symbols spelled from this function's own .s relocations): + + §195-H D_800B9A02 must be the ARRAY spelling + a constant index — that is the + MEM_IN_STRUCT_P (/s) reload dial that produces the 26 per-statement + `lhu 0($t1)` reloads. Same __asm__-alias idiom func_80181F4C uses, so + the TU's own `extern s16 D_800B9A02;` is untouched (no C-level clash). + + §186b/§226 `s32 pad[6]` = 24 bytes of addressed locals; with the four saved + registers that is the target's `addiu $sp,$sp,-0x28` frame exactly. + + NEW (this function) — THE OT BASE IS TWO DIFFERENT ADDRESSING MODES, AND THE + SPELLING PICKS WHICH. The FIRST read of D_800A6618 is written with the + SYMBOL (`(u8 *)D_800A6618 + (idx << 14)`) and keeps the `lui $at / + addu / lw %lo($at)` form; every later access is written through the + pointer VARIABLE `ob` and gets the 1-instruction-shorter `addu $x,$t5 / + lw 0($x)` form. Writing the symbol in all four places is +1 instruction + (block 2's link read) — that single row was the whole length drift. + + §17/§325 pins — three, each fixing an independent local-alloc permutation that + no source reordering reaches (local-alloc's qty_compare priority, and + find_free_reg always takes the lowest free register): + r -> $6 : without it `r4` (2 refs) outranks `r` (13 refs) and takes + $a2, pushing D_801C7934 to $a3 (15 residual rows). + mLO-> $12 : the 0xFFFFFF mask, otherwise $t7 (15 rows in a 3-cycle + with 0x100 and 0xFF000000). A §137 zero-byte `"r"(K)` + ref is NOT available here — cse does not fold the extra + reference onto the existing constant pseudo, so each + probe cost a real +1/+2 instructions. + ob -> $13 : the D_800A6618 base, otherwise $t6 and 0xFF000000 $t5. + Pinning 0xFF000000 instead of `ob` also re-orders sched1 (the + `lui $t4,0xFF` moves from idx 72 to 63) — pin the POINTER, not the + second constant (§325's direction, replicated). + + Address shapes copied from func_80181F4C: `p4 = p + 4` first (a pointer VALUE) + yields `addiu $a2,$t2,4` + `addu` + `sw 0(..)`; every other access is written + `p + F184_IDX + K` so K stays in the memory operand. `q` gives the three 0xA/0x9/0x8 + byte stores one shared index computation. */ + +extern u16 aB9A02[1] __asm__("D_800B9A02"); +extern u8 D_801C78D4[]; +extern u8 D_801C78D7[]; +extern u8 D_801C7934[]; +extern u8 D_801C7937[]; +extern u32 D_800A6618[]; + +#define F184_IDX (aB9A02[0] * 0x18 + n * 0x30) + +void func_80182184(s32 arg0, s32 arg1) { + s32 n = arg1 & 0xFF; + u8 *p = D_801C78D4; + u8 *p4 = p + 4; + register u8 *r __asm__("$6"); + register u32 mLO __asm__("$12"); + u8 *r4; + u8 *q; + u32 *ot1; + u32 *ot2; + register u8 *ob __asm__("$13"); + s32 pad[6]; + + (void)&pad; + + *(u8 *)(D_801C78D7 + F184_IDX) = 5; + *(u32 *)(p4 + F184_IDX) = ((n * 2 + 0x8A) & 0x9FF) | 0xE1000000; + *(u8 *)(p + F184_IDX + 0xB) = 0x64; + q = p + (n * 0x30 + aB9A02[0] * 0x18); + q[0xA] = arg0; + q[0x9] = arg0; + q[0x8] = arg0; + *(s16 *)(p + F184_IDX + 0xC) = -0x118; + *(s16 *)(p + F184_IDX + 0xE) = -0xF0; + *(u8 *)(p + F184_IDX + 0x10) = 0; + *(u8 *)(p + F184_IDX + 0x11) = 0; + *(u16 *)(p + F184_IDX + 0x12) = ((n << 4) & 0x3F) | 0x7880; + *(s16 *)(p + F184_IDX + 0x14) = 0x100; + *(s16 *)(p + F184_IDX + 0x16) = 0x100; + + ob = (u8 *)D_800A6618; + mLO = 0xFFFFFF; + *(u32 *)(p + F184_IDX) = (*(u32 *)(p + F184_IDX) & 0xFF000000) | + (*(u32 *)((u8 *)D_800A6618 + (aB9A02[0] << 14)) & mLO); + + ot1 = (u32 *)(ob + (aB9A02[0] << 14)); + *ot1 = (*ot1 & 0xFF000000) | ((n * 0x30 + (aB9A02[0] * 0x18 + (u32)p)) & mLO); + + r = D_801C7934; + r4 = r + 4; + *(u8 *)(D_801C7937 + F184_IDX) = 5; + *(u32 *)(r4 + F184_IDX) = ((n * 2 + 0x9A) & 0x9FF) | 0xE1000000; + *(u8 *)(r + F184_IDX + 0xB) = 0x64; + q = r + (n * 0x30 + aB9A02[0] * 0x18); + q[0xA] = arg0; + q[0x9] = arg0; + q[0x8] = arg0; + *(s16 *)(r + F184_IDX + 0xC) = -0x118; + *(s16 *)(r + F184_IDX + 0xE) = 0x10; + *(u8 *)(r + F184_IDX + 0x10) = 0; + *(u8 *)(r + F184_IDX + 0x11) = 0; + *(u16 *)(r + F184_IDX + 0x12) = ((n << 4) & 0x3F) | 0x7880; + *(s16 *)(r + F184_IDX + 0x14) = 0x100; + *(s16 *)(r + F184_IDX + 0x16) = 0xE0; + + *(u32 *)(r + F184_IDX) = (*(u32 *)(r + F184_IDX) & 0xFF000000) | + (*(u32 *)(ob + (aB9A02[0] << 14)) & mLO); + + ot2 = (u32 *)(ob + (aB9A02[0] << 14)); + *ot2 = (*ot2 & 0xFF000000) | ((n * 0x30 + (aB9A02[0] * 0x18 + (u32)r)) & mLO); +} + +#undef F184_IDX +