diff --git a/src/ov_SC01_080/ov_SC01_080_jr_8017AE2C.c b/src/ov_SC01_080/ov_SC01_080_jr_8017AE2C.c index 5115d88efa..46403a926f 100644 --- a/src/ov_SC01_080/ov_SC01_080_jr_8017AE2C.c +++ b/src/ov_SC01_080/ov_SC01_080_jr_8017AE2C.c @@ -6433,7 +6433,124 @@ void func_801809E4(s32 param_1) { } -INCLUDE_ASM("asm/ov_SC01_080/nonmatchings/ov_SC01_080_jr_8017AE2C", func_80180D54); +/* func_80180D54 — MATCH (match_one closeness 0, 119/119). + * + * Provenance / declarations (all copied from this same TU, + * src/ov_SC01_080/ov_SC01_080_jr_8017AE2C.c, cookbook §135-8): + * D_800B99DA (u16) — TU col-0 decl, line 615 / 6703 + * D_801C7540 (s16) — TU col-0 decl, line 6158 / 6507 / 6704 + * D_801C5D40 (u8[]) — TU col-0 decl, line 6273 / 6319 / 6705 + * D_80127188 (s32) — TU col-0 decl, line 5477 / 6274 / 6320 + * func_800D20C0(void*,void*,s32), func_800D23D0(void*) — line 1711/1713, 6161/6162 + * func_8012913C(s32)->u8*, func_801291C0(void)->u8* — line 6317/6318 + * Every symbol re-checked against this .s's own relocation lines (law 1c); + * jal order in the .s is 800D20C0, 800D23D0, 801291C0, 801291C0, 8012913C, 8012913C, + * which is the source order below. + * + * NOT the same function as the same-address func_80180D54 banked in ov_SC02_035 / + * ov_SC03_097 — those overlays hold different code at 0x80180D54; discarded. + * + * Codegen notes (each closed a residual; twin of func_801815D4 at line 6711 of this TU): + * - The two `__asm__("" : "=r"(pN) : "0"(pN))` pins are LOAD-BEARING (house idiom, taken + * verbatim from func_801815D4). Without them gcc CSEs `sp10+4` into one pseudo that is + * live across func_800D20C0, so it takes a callee-saved register: the arg lands in $s1 + * instead of $s0, an extra `move $a1,$s0` appears in the prologue, and the second call's + * delay slot holds `move $a0,$s0` instead of `addiu $a0,$sp,0x18`. Measured: removing + * them costs +1 insn and 118 mismatches. + * - `cnt` is loaded BEFORE the D_801C7540 store in SOURCE order. The store aliases the + * load for true_dependence(), so with the natural order sched1 cannot hoist the + * `lw 0x1C($s0)` into the load-delay nop after `lhu 0x1A($sp)` (costs a nop, §3-T2). + * - `mask` is pinned to $s3 (§17/§148-C register pin). 0x3FF and 0x19 are both invariants + * of the do/while; only ONE of the pair gets hoisted by move_movables at this loop's + * insn_count (§148-A), and whichever is written as an explicit pre-loop local wins the + * allocno tiebreak and takes the LOWER hard reg. The target wants the opposite pairing + * ($s2=0x19 hoisted, $s3=0x3FF explicit). Making 0x19 the explicit local instead flips + * the registers correctly but frees a movable slot, so `&D_801C5D40` then hoists into + * $s4 as well (+4 insns, frame 0x40). The pin gets the register order with no side + * effect on the hoist budget. Measured: close 11 -> 3. + * - `q` must be a SEPARATE local from `p`: sharing one pointer variable makes the + * func_8012913C result inherit $a0 (the register `p` needs inside the loops), where the + * target uses $v1 because 1 is being materialized into $v0 for the delay slot. + * Measured: close 3 -> 0. + * - The `sll/sra` before the `slti 0xC0` is CSE store-forwarding: the `+= 1` keeps the + * HImode value, so the signed compare must sign-extend the just-stored register. + */ + +extern u16 D_800B99DA; +extern s16 D_801C7540; +extern u8 D_801C5D40[]; +extern s32 D_80127188; +extern void func_800D20C0(void *a0, void *a1, s32 a2); +extern void func_800D23D0(void *a0); +extern u8 *func_8012913C(s32 a0); +extern u8 *func_801291C0(void); + +void func_80180D54(s32 param_1) { + u16 sp10[6]; + u16 *p0; + u16 *p1; + register s16 mask __asm__("$19"); + s16 i; + u8 *p; + u8 *q; + s32 cnt; + + sp10[2] = 0; + sp10[1] = 0; + sp10[0] = 0; + p0 = sp10; + p1 = sp10 + 4; + __asm__("" : "=r"(p0) : "0"(p0)); + __asm__("" : "=r"(p1) : "0"(p1)); + func_800D20C0(p0, p1, 1); + func_800D23D0(sp10 + 4); + cnt = *(s32 *)(param_1 + 0x1C); + D_801C7540 = sp10[5] - 0x400; + if (cnt != 0) { + *(s32 *)(param_1 + 0x1C) = cnt - 1; + mask = 0x3FF; + i = 0; + do { + p = func_801291C0(); + if (p != 0) { + *(u16 *)p = 0x19; + *(u16 *)(p + 0x2C) = mask; + *(s32 *)(p + 0x34) = (s32)&D_801C5D40 + (*(s16 *)(param_1 + 0xDC) << 5); + *(u16 *)(param_1 + 0xDC) = *(u16 *)(param_1 + 0xDC) + 1; + if (*(s16 *)(param_1 + 0xDC) >= 0xC0) { + *(s16 *)(param_1 + 0xDC) = 0; + } + } + i = i + 1; + } while (i <= 0); + } else if (D_80127188 < 4) { + if ((D_800B99DA & 1) == 0) { + mask = 0x3FF; + i = 0; + do { + p = func_801291C0(); + if (p != 0) { + *(u16 *)p = 0x19; + *(u16 *)(p + 0x2C) = mask; + *(s32 *)(p + 0x34) = (s32)&D_801C5D40 + (*(s16 *)(param_1 + 0xDC) << 5); + *(u16 *)(param_1 + 0xDC) = *(u16 *)(param_1 + 0xDC) + 1; + if (*(s16 *)(param_1 + 0xDC) >= 0xC0) { + *(s16 *)(param_1 + 0xDC) = 0; + } + } + i = i + 1; + } while (i <= 0); + } + } else { + func_8012913C(0x1C); + q = func_8012913C(0x1C); + if (q != 0) { + *(u16 *)(q + 0x2C) = 1; + } + *(u16 *)(param_1 + 0x2) = *(u16 *)(param_1 + 0x2) + 1; + } +} + extern void (*D_8018A22C[])(void); diff --git a/src/ov_SC01_084/ov_SC01_084_jr_8017F690.c b/src/ov_SC01_084/ov_SC01_084_jr_8017F690.c index 6d44bf2692..0e4a3912ab 100644 --- a/src/ov_SC01_084/ov_SC01_084_jr_8017F690.c +++ b/src/ov_SC01_084/ov_SC01_084_jr_8017F690.c @@ -3833,7 +3833,63 @@ void func_801812F4(void *a0) { } -INCLUDE_ASM("asm/ov_SC01_084/nonmatchings/ov_SC01_084_jr_8017F690", func_80181310); +extern s16 D_801C7748; +extern s16 D_801C774C; +extern s32 D_801270C8; +extern s32 D_801270D0; +extern s16 D_8018A9EC[]; +extern s32 D_8018A9F0; +extern s32 func_8012C658(s32 arg0, s32 arg1, s32 arg2); + +void func_80181310(void) { + s16 *p; + s32 *r; + s32 *u; + s32 t; + + if (D_801C774C == 1) { + p = D_8018A9EC; + r = &D_801270C8; + *r = 0; + t = D_801C7748; + if (D_8018A9F0 != -1) { + do { + if (p[0] <= t && t < p[1]) { + switch (*(s32 *)(p + 2)) { + case 1: + u = &D_801270D0; + if (*u == 0) { + *u = 1; + func_8012C658(0x26, 0x100, 0); + func_8012C658(0x26, 0x101, 0); + func_8012C658(0x26, 0x102, 0); + } + break; + case 2: + /* (&D_801270C8)[3] == D_801270D4; written as an offset from the + D_801270C8 base so gcc reuses the base register ($a0 + 0xC). */ + if ((&D_801270C8)[3] == 0 && D_801C7748 < 0x2501) { + func_8012C658(0x1D, 0, 0); + (&D_801270C8)[3]++; + } + break; + case 3: + if ((&D_801270C8)[3] == 0 && D_801C7748 < 0x2501) { + func_8012C658(0x1D, 1, 0); + (&D_801270C8)[3]++; + } + break; + case 4: + D_801270C8 = 1; + break; + } + } + p += 4; + } while (*(s32 *)(p + 2) != -1); + } + } +} + extern void (*D_8018AA2C[])(void); extern s16 D_801C774C; diff --git a/src/ov_SC03_030/ov_SC03_030_jr_8017AE2C.c b/src/ov_SC03_030/ov_SC03_030_jr_8017AE2C.c index df564f4446..208dd6c9a0 100644 --- a/src/ov_SC03_030/ov_SC03_030_jr_8017AE2C.c +++ b/src/ov_SC03_030/ov_SC03_030_jr_8017AE2C.c @@ -6597,7 +6597,7 @@ void func_801814F4(s32 param_1) { extern s32 func_80181824(); extern void RotTransSV(void *a0, void *a1, void *a2); extern void RotMatrixZ(s32 a0, void *a1); -extern void func_80181A60(void *a0); +extern void func_80181A60(); extern u8 D_801869E0[]; extern u8 D_801869E8[]; @@ -6761,7 +6761,178 @@ s32 func_80181824(s32 param_1, s32 param_2) } -INCLUDE_ASM("asm/ov_SC03_030/nonmatchings/ov_SC03_030_jr_8017AE2C", func_80181A60); +/* func_80181A60 — ov_SC03_030 / ov_SC03_030_jr_8017AE2C (106 ins). + * + * Build one 0x10-byte white LINE_F2 GPU packet, project the caller's two + * SVECTORs through the GTE with the D_800AF648 rot/trans matrix, write the two + * screen (x,y) pairs into the packet, and — only when BOTH rtps calls come back + * with no flag bits other than 0x1000 — link the packet into the current OT + * (D_800A651C[D_800B9A02].a) at depth otz>>2 with the open-coded PSY-Q addPrim + * RMW pair. Sibling idiom: src/800.c func_80015D4C / func_80015F04 (both + * MATCHED) — same allocator + same addPrim asymmetry (first half inlines the + * OT-slot address, second half self-accumulates it into the otz*4 register). + * + * STATUS: match_one MATCH, 106/106 (S71 fable redraw, recovered from the opus + * closeness-2 body at .run/S71a_1/opus/scratch_func_80181A60/base.c). + * + * THREE ZERO-BYTE LEVERS: + * 1. `register u32 tag0 __asm__("$5")` — the packet tag word read into its own + * pinned local (cf. src/800.c func_80015D4C's "$4" pin). Unpinned, tag0 + * steals $v1 and the whole packet/mask/matrix register set shifts by one. + * 2. `register u32 mFF __asm__("$6")` — mFF and the CSE'd &D_800B9A02 base tie + * in global-alloc and land swapped ($a3/$a2) without the pin. + * 3. THE LAST 2 (a sched2 LUID tie): the target block reads lhu(ix), then + * lw tag0, then lw otz. At sched2 all three loads are priority 1 and the + * tie falls to LUID = sched1's OUTPUT order. sched1's birthing boost + * (single-set dest) sinks a boosted load to just before its consumer, so: + * - tag0 single-set -> boosted -> sinks below the otz load (LUID too high); + * - tag0 2-set, lhu single-set -> tag0 starves to the block TOP, above the + * still-boosted lhu (the "overshoot": lw before lhu); + * - BOTH 2-set -> both priority 1, the final tie is source order, and + * `ix = lhu; tag0 = lw;` gives exactly lhu < tag0 < otz. + * The lhu's dest gets its 2nd set from a VOLATILE dead re-tie + * `__asm__ volatile("" : "=r"(ix) : "0"(ix));` placed AFTER the first + * addPrim store: + * - volatile, because a non-volatile re-tie whose output is dead is deleted + * (vJ1: boost back, 2 off); + * - after the store, because a re-tie right after the lhu is a reorg + * stop_search_p wall in the bnez fall-through thread -> nop delay slot + * (A1: 39 off); + * - NOT a plain second `ix = *(u16 *)&D_800B9A02;` in the 2nd half: the + * long-lived 2-set pseudo swaps ix/base ($a3/$v1) in local-alloc (8 off). + * `u16 ix` cannot work (combine folds lhu+zext into fresh single-set SI + * temps) and a `register ... __asm__("$3")` ix cannot either (cse + * forward-substitutes the zero-extend temp past the hard-reg copy). + * + * Symbol audit (law 1c — match_one masks jal/HI16/LO16): the .s names exactly + * four relocated symbols and this draft names those four and no others — + * `jal func_80010A08` ($a0 = 0x10), `%hi/%lo(D_800AF648)` (address-of, GTE + * matrix base), `%hi/%lo(D_800B9A02)` (address-of, then `lhu` => u16 read), + * `%lo(D_800A651C)($at)` x2 (lw, index scaled *20 = sizeof{s32 a; s32 b[4]}). + * Both bnez targets are .L80181BF4 (the shared epilogue). + * + * TU CHECK vs src/ov_SC03_030/ov_SC03_030_jr_8017AE2C.c: + * - `extern s16 D_800B9A02;` (TU:2469) copied verbatim; `lhu` forced with + * `*(u16 *)&`, the idiom this TU already uses at line 3708. + * - `extern u8 D_800AF648[];` (TU:5292) copied verbatim. + * - `func_80010A08` has no file-scope decl in the TU; fleet decl (void *, (s32)). + * - D_800A651C declared at BLOCK scope (engine_core.h's DEFINE_ macros declare + * it scalar in their own bodies) — the src/800.c func_80015D4C dodge. + * - GTE macro names carry an _A60 suffix (the TU defines unsuffixed + * gte_SetRotMatrix/gte_stszotz etc. elsewhere). + * - BANK NOTE: TU:6600 forward-declares `extern void func_80181A60(void *a0);` + * -> §378 variant 1 (fix_arity_callers/cast_self_callers). Defining the + * function as `(void *arg0)` with a local `v = arg0` copy is NOT + * byte-equivalent (+$s1, 109 off) — keep the typed pointer parameter. + */ +#include "common.h" + +typedef struct { s16 vx, vy, vz, pad; } SVec_80181A60; +typedef struct { s32 a; s32 b[4]; } OtBlk_80181A60; +typedef struct { + u8 addr[3]; + u8 len; + u8 r0, g0, b0, code; + s16 x0, y0; + s16 x1, y1; +} LineF2_80181A60; + +/* PSY-Q GTE macros — same forms already used in this TU (func_80181708/func_80181824). */ +#define gte_SetRotTransMatrix_A60(r0) __asm__ volatile ( \ + "lw $12, 0( %0 );" \ + "lw $13, 4( %0 );" \ + "ctc2 $12, $0;" \ + "ctc2 $13, $1;" \ + "lw $12, 8( %0 );" \ + "lw $13, 12( %0 );" \ + "lw $14, 16( %0 );" \ + "ctc2 $12, $2;" \ + "ctc2 $13, $3;" \ + "ctc2 $14, $4;" \ + "lw $12, 20( %0 );" \ + "lw $13, 24( %0 );" \ + "ctc2 $12, $5;" \ + "lw $14, 28( %0 );" \ + "ctc2 $13, $6;" \ + "ctc2 $14, $7" \ + : \ + : "r"( r0 ) \ + : "$12", "$13", "$14" ) +#define gte_ldv0_A60(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 )" \ + : \ + : "r"( r0 ) ) +#define gte_rtps_A60() __asm__ volatile ("nop;nop;rtps") +#define gte_stsxy_A60(r0) __asm__ volatile ( \ + "swc2 $14, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) +#define gte_stflg_A60(r0) __asm__ volatile ( \ + "cfc2 $12, $31;" \ + "nop;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) +#define gte_stszotz_A60(r0) __asm__ volatile ( \ + "mfc2 $12, $19;" \ + "nop;" \ + "sra $12, $12, 2;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +extern void *func_80010A08(s32); + +void func_80181A60(SVec_80181A60 *v) +{ + extern OtBlk_80181A60 D_800A651C[]; + LineF2_80181A60 *p; + u8 *mx; + u32 m24; + register u32 mFF __asm__("$6"); + s32 flag; + s32 otz; + s32 sh; + register u32 tag0 __asm__("$5"); + u32 ix; + + p = (LineF2_80181A60 *)func_80010A08(0x10); + m24 = 0x00FFFFFF; + p->len = 3; + mx = D_800AF648; + *(u32 *)&p->r0 = m24; + p->code = 0x40; + + gte_SetRotTransMatrix_A60(mx); + gte_ldv0_A60(v); + gte_rtps_A60(); + gte_stsxy_A60(&p->x0); + gte_stflg_A60(&flag); + gte_stszotz_A60(&otz); + if ((flag & ~0x1000) == 0) { + v++; + gte_ldv0_A60(v); + gte_rtps_A60(); + gte_stsxy_A60(&p->x1); + gte_stflg_A60(&flag); + if ((flag & ~0x1000) == 0) { + mFF = 0xFF000000; + ix = *(u16 *)&D_800B9A02; + tag0 = *(u32 *)p; + tag0 = tag0 & mFF; + sh = otz * 4; + *(u32 *)p = tag0 | (*(u32 *)(sh + D_800A651C[ix].a) & m24); + __asm__ volatile("" : "=r"(ix) : "0"(ix)); + sh = sh + D_800A651C[*(u16 *)&D_800B9A02].a; + *(u32 *)sh = (*(u32 *)sh & mFF) | ((u32)p & m24); + } + } +} + extern void func_8012C218(void *a0); diff --git a/src/ov_SC06_000/ov_SC06_000_jr_8017AE2C.c b/src/ov_SC06_000/ov_SC06_000_jr_8017AE2C.c index 6ec4520920..64f7016c29 100644 --- a/src/ov_SC06_000/ov_SC06_000_jr_8017AE2C.c +++ b/src/ov_SC06_000/ov_SC06_000_jr_8017AE2C.c @@ -8135,7 +8135,63 @@ void func_80183338(void *a0) { } -INCLUDE_ASM("asm/ov_SC06_000/nonmatchings/ov_SC06_000_jr_8017AE2C", func_80183398); +void func_80183398(s32 a0) { + extern s32 D_801AECB0; + extern s16 func_801825D8(void); + extern s32 func_8012BEE8(s32 a0); + extern void func_8012B2CC(s32 a0); + extern void func_801823BC(); + extern void func_80182464(s32 a0, s32 a1); + s32 ptr; + s32 t; + u16 v; + s32 pad[2]; + + if (*(s16 *)(a0 + 0x70) != 0) { + if (D_801AECB0 != 0) { + if (*(u16 *)(a0 + 0x34) == 0) { + v = *(u16 *)(a0 + 0x100) - 4; + *(u16 *)(a0 + 0x100) = v; + if ((s16)v < -0x27F) { + *(u16 *)(a0 + 0x34) = *(u16 *)(a0 + 0x34) + 1; + } + } else { + v = *(u16 *)(a0 + 0x100) + 4; + *(u16 *)(a0 + 0x100) = v; + if ((s16)v >= 0) { + *(u16 *)(a0 + 0x34) = *(u16 *)(a0 + 0x34) - 1; + } + } + } + ptr = *(s32 *)(a0 + 0x20); + *(u16 *)(ptr + 0x12) = *(u16 *)(ptr + 0x12) - 4; + t = ((s32 (*)(void)) func_801825D8)(); + *(s16 *)(a0 + 0xA) = (*(u16 *)(a0 + 0xFE) + *(u16 *)(a0 + 0x100)) + t - 0x131; + func_80182464(*(s16 *)(*(s32 *)(a0 + 0x20) + 0x12), *(s16 *)(a0 + 0xA)); + } else { + if (D_801AECB0 != 0 && func_8012BEE8(a0) != 0) { + if (*(u16 *)(a0 + 0x34) == 0) { + v = *(u16 *)(a0 + 0x100) - 6; + *(u16 *)(a0 + 0x100) = v; + if ((s16)v < -0x3BF) { + *(u16 *)(a0 + 0x34) = *(u16 *)(a0 + 0x34) + 1; + } + } else { + v = *(u16 *)(a0 + 0x100) + 6; + *(u16 *)(a0 + 0x100) = v; + if ((s16)v >= 0) { + *(u16 *)(a0 + 0x34) = *(u16 *)(a0 + 0x34) - 1; + } + } + } + ptr = *(s32 *)(a0 + 0x20); + *(u16 *)(ptr + 0x12) = *(u16 *)(ptr + 0x12) + 4; + *(s16 *)(a0 + 0xA) = *(u16 *)(a0 + 0xFE) + *(u16 *)(a0 + 0x100); + func_801823BC(*(s16 *)(*(s32 *)(a0 + 0x20) + 0x12), *(s16 *)(a0 + 0xA)); + } + func_8012B2CC(a0); +} + extern void (*D_8018C734[])(void);