diff --git a/.gitignore b/.gitignore index 6a15c8be63..c7c4b41428 100644 --- a/.gitignore +++ b/.gitignore @@ -168,3 +168,12 @@ models/ unsloth_compiled_cache/ .venv-serve/ .env + +# Phase 29 SESSION-22 — the wave22 UNBANKED drafts. 13 of the 18 (6 NEAR with precise residual +# diagnoses + 7 that reached match_one MATCH but did not bank whole-binary) are the next session's +# named fuel and cost ~2.59M subagent tokens. They lived only on disk, i.e. one `git clean -fdx` +# from gone — the same exposure the .run/giants Fable5 cracks had before Phase 27 curated them. +!/.run/wave22/ +/.run/wave22/* +!/.run/wave22/*.c +!/.run/wave22_targets.json diff --git a/.run/wave22/_best.c b/.run/wave22/_best.c new file mode 100644 index 0000000000..db84e4783b --- /dev/null +++ b/.run/wave22/_best.c @@ -0,0 +1,47 @@ +extern void func_8012F038(); +extern void func_8012F14C(); +extern s32 func_80135888(s32, s32, s32, s32); +extern u16 D_80126B5E; +extern u16 D_80126B62; +extern u16 D_80126B66; +/* derived from the asm (%hi/%lo of D_80126BE0 taken as an address arg); matches the + `extern u8 D_80126BE0[];` decl already used across the sibling overlays. */ +extern u8 D_80126BE0[]; + +void func_80132F40(s32 arg0) +{ + typedef struct { u16 vx, vy, vz, pad; } Svec_80132F40; + + Svec_80132F40 src; + Svec_80132F40 dst; + s16 *p; + void *ps1; + void *ps2; + s32 w; + s32 h; + s32 m; + + src.vx = D_80126B5E; + src.vy = D_80126B62; + src.vz = D_80126B66; + dst = src; + + if (func_80135888(*(s32 *)(arg0 + 0x20), *(s32 *)(arg0 + 0x58), + (s32)D_80126BE0, (s32)&dst) != 0) { + p = (s16 *)((*(s32 *)(arg0 + 0x58) & 0x0FFFFFFF) | 0x80000000); + w = p[4]; + h = p[5]; + m = w; + if (h < w) { + m = h; + } + ps1 = &src; __asm__ __volatile__("" : "=r"(ps1) : "0"(ps1)); + func_8012F038(*(s32 *)(arg0 + 0x20) + 0x34, ps1, &dst); + dst.vy = m; + ps2 = &src; __asm__ __volatile__("" : "=r"(ps2) : "0"(ps2)); + func_8012F14C(*(s32 *)(arg0 + 0x20) + 0x34, &dst, ps2); + D_80126B5E = src.vx; + D_80126B62 = src.vy; + D_80126B66 = src.vz; + } +} diff --git a/.run/wave22/func_8012A328.c b/.run/wave22/func_8012A328.c new file mode 100644 index 0000000000..58e44f43a7 --- /dev/null +++ b/.run/wave22/func_8012A328.c @@ -0,0 +1,44 @@ +typedef signed char s8; +typedef unsigned char u8; +typedef short s16; +typedef unsigned short u16; +typedef int s32; +typedef unsigned int u32; + +extern s32 func_80012F74(s32, s32, s32, s32); +extern s8 D_801152C0; +extern s16 D_801152C2; +extern s16 D_80126940; +extern s16 D_80126942; +extern s16 D_80126944; +extern s32 D_80126B58; +extern s16 D_80127080; +extern s16 D_80126CAE; + +void func_8012A328(void) +{ + typedef struct { + u8 pad[0x154]; + u16 f154; + s16 f156; + u16 f158; + } Obj_80126B58; + Obj_80126B58 *p = (Obj_80126B58 *)&D_80126B58; + s16 *q; + s32 a, b; + if (*(u8 *)&D_801152C0 != 0) { + a = D_80127080; b = D_80126CAE; + if (D_801152C2 < ((a - b) < 0 ? (b - a) : (a - b))) { + q = &D_80126942; + *q = func_80012F74(*q, p->f156, 4, 1); + } else { + q = &D_80126942; + *q = func_80012F74(*q, D_80127080, 4, 1); + } + } else { + s16 *r = &D_80126942; + *r = func_80012F74(*r, D_80126CAE, 4, 1); + } + D_80126940 = p->f154; + D_80126944 = p->f158; +} diff --git a/.run/wave22/func_8012E014.c b/.run/wave22/func_8012E014.c new file mode 100644 index 0000000000..f2547b83bc --- /dev/null +++ b/.run/wave22/func_8012E014.c @@ -0,0 +1,49 @@ +typedef signed char s8; +typedef unsigned char u8; +typedef short s16; +typedef unsigned short u16; +typedef int s32; +typedef unsigned int u32; + +extern s32 D_80126B5C; +/* derived from asm %hi/%lo refs (not present in sig_hints) */ +extern s32 D_80126B60; +extern s32 D_80126B64; + +extern void func_80049CAC(s32, s32); +void func_8012F0BC(s32*, s32*, s32*); +void func_8012F1A4(s32*, s32, s32*); + +void func_8012E014(s32 arg0) { + typedef struct { s32 vx, vy, vz, pad; } Vec_8012E014; + Vec_8012E014 sp10; + Vec_8012E014 sp20; + Vec_8012E014 sp30; + s32 obj; + s16 t; + + sp10.vx = D_80126B5C; + sp10.vy = D_80126B60; + sp10.vz = D_80126B64; + func_8012F0BC((s32*)(*(s32*)(arg0 + 0x20) + 0x34), (s32*)&sp10, (s32*)&sp30); + obj = *(s32*)(arg0 + 0x20); + if (obj != 0) { + register s32 obj2 __asm__("$4"); + func_80049CAC(obj + 0x10, obj + 0x34); + obj2 = *(s32*)(arg0 + 0x20); + t = *(u16*)(arg0 + 6) + *(u16*)(arg0 + 0x50); + *(s16*)(obj2 + 8) = t; + *(s32*)(obj2 + 0x48) = t; + t = *(u16*)(arg0 + 0xA) + *(u16*)(arg0 + 0x52); + *(s16*)(obj2 + 0xA) = t; + *(s32*)(obj2 + 0x4C) = t; + t = *(u16*)(arg0 + 0xE) + *(u16*)(arg0 + 0x54); + *(s16*)(obj2 + 0xC) = t; + *(u16*)(obj2 + 0x2C) |= 1; + *(s32*)(obj2 + 0x50) = t; + } + func_8012F1A4((s32*)(*(s32*)(arg0 + 0x20) + 0x34), (s32)&sp30, (s32*)&sp20); + D_80126B5C = sp20.vx; + D_80126B60 = sp20.vy; + D_80126B64 = sp20.vz; +} diff --git a/.run/wave22/func_8012E364.c b/.run/wave22/func_8012E364.c new file mode 100644 index 0000000000..f56fbe470b --- /dev/null +++ b/.run/wave22/func_8012E364.c @@ -0,0 +1,89 @@ +/* func_8012E364 (ov_SC01_077_jr_8012ACE0) — NEAR: 67/67 ins, closeness 7 (reloc-masked). + * + * Structure is fully solved: identical instruction count, identical control flow, identical + * delay-slot fills, identical %hi/%lo pairs. The residual 7 slots are pure gcc-2.7.2 register + * allocation / list-scheduler tie-breaks (see "RESIDUAL" at the bottom). + * + * Symbols: sig_hints listed no callees and no data decls for this fn, so the three externs below + * are derived from the asm's %hi/%lo pairs: + * D_80126CE0 -> `lh` (0x8012E370) => s16 + * D_801D9498 -> `lw`/`sw` => s32 + * D_801D949C -> `lw`/`sw` => s32 + * D_80126CE0 is already declared `extern s16 D_80126CE0;` inside ov_SC03_099_jr_8016AB6C.c etc., + * so the s16 typing is consistent with the rest of the tree. + * + * LOAD-BEARING constructs (do not "simplify"): + * 1. `spd` — a local holding 0x1000 that is SET BEFORE the if/else chain. This is what makes gcc + * materialise the constant in a pseudo whose live range crosses the branch, giving + * `addiu $a3,$zero,0x1000` in the `blez` delay slot at 0x8012E3DC and the `sw $a3` / + * `addu $v1,$v1,$a3` register forms at L8012E410. Writing the literal 0x1000 inline instead + * costs 2 instructions (a separate `li` + an unfilled delay slot). + * 2. `a` — ONE variable reused for three roles (the D_80126CE0 value, the division result, and + * the entity pointer). That makes it a cross-block (global) allocno, so local-alloc's + * combine_regs cannot tie the division result to the `sra` temp — which is what puts the + * result in $a0 (`subu $a0,$v0,$v1`) instead of coalescing into $v0. Splitting `a` into + * x/t/ent regresses 2 slots. + * 3. The three register pins ($6 param, $5 prev, $2 flags) reproduce the target's allocation of + * the tail block; without them the whole a/t register profile shifts down one slot (and the + * leading `move $a2,$a0` disappears entirely, costing an instruction). + * + * RESIDUAL (7 slots, both clusters are compiler tie-breaks, not C-expressible): + * idx 43-45 the two block-head loads are swapped: gcc schedules `lw $a0,0x20($a2)` before + * `lw $a1,D_801D949C`; the target has them the other way round. Priority-driven + * (the ent->lhu->ori->sh chain outranks prev->subu->addu->sw), NOT source order — + * every permutation of the 7 tail statements gives the identical schedule. + * idx 59-62 the target keeps a redundant copy `addu $v0,$v1,$zero` (in the bgez delay slot) + * before `negu $v0,$v0`; gcc here coalesces d with v and emits `nop`/`negu $v1,$v1`. + * Forcing the copy (pin `v` to $3, compare on `v`) makes it appear but lands `d` in + * $a1/$a0 rather than $v0 — same 4-slot cost either way. + * ~2500 source-shape/pin combinations swept, plus a 400 s decomp-permuter run at -j10 + * (33 saved candidates, all score 7, never below). This is genuine regalloc hard tail. + */ +extern s16 D_80126CE0; +extern s32 D_801D9498; +extern s32 D_801D949C; + +void func_8012E364(s32 arg0_) +{ + register s32 arg0 __asm__("$6"); + register s32 prev __asm__("$5"); + register u16 flags __asm__("$2"); + s32 a; + s32 diff; + s32 v; + s32 d; + s32 spd; + + arg0 = arg0_; + *(s16 *)(arg0 + 0x5C) = 0; + a = D_80126CE0; + if (a == 0) { + D_801D9498 = 0x1000; + D_801D949C = 0x1000; + } + a = ((0x90 - a) << 12) / 0x90; + *(s32 *)(arg0 + 0x1C) += 1; + spd = 0x1000; + + diff = D_801D9498 - a; + if (diff > 0) { + D_801D9498 -= diff >> 2; + } else if (diff < 0) { + D_801D9498 += (-diff) / 4; + } + + prev = D_801D949C; + a = *(s32 *)(arg0 + 0x20); + v = D_801D9498 - prev + spd; + D_801D949C = spd; + flags = *(u16 *)(a + 0x2C); + D_801D9498 = v; + *(u16 *)(a + 0x2C) = flags | 0x10; + + a = *(s32 *)(arg0 + 0x20); + d = v; + if (d < 0) d = -d; + *(s16 *)(a + 0x1C) = d; + *(s16 *)(a + 0x18) = d; + *(s16 *)(*(s32 *)(arg0 + 0x20) + 0x1A) = 0x1000; +} diff --git a/.run/wave22/func_80132F40.c b/.run/wave22/func_80132F40.c new file mode 100644 index 0000000000..3c664c392b --- /dev/null +++ b/.run/wave22/func_80132F40.c @@ -0,0 +1,73 @@ +/* func_80132F40 — ov_SC01_077 (jr_8012ACE0 region), 72 ins, -O2. + * + * STATUS: NEAR — 72/72 instructions, closeness 6, klass OPCODE-MIXED(branch,width). + * Instructions 0..36 and 44..71 are byte-identical (relocation-masked); the whole residual + * is the 6-instruction min(w,h) block at idx 37..43. + * + * Canonical decls (wave22_targets.json sig_hints) verbatim. D_80126BE0 is NOT in sig_hints — + * it is derived from the asm (`lui/addiu %hi/%lo` passed as arg 3) and declared exactly as the + * 20+ sibling TUs already declare it: `extern u8 D_80126BE0[];`. + * + * Structure facts that were byte-forced (each verified by a compile): + * - locals are an SVECTOR-ish ARRAY of 4 (vars=32 in the target frame 0x40 with only 4 saved + * regs; v[0]@sp+0x10 and v[1]@sp+0x18 are used, v[2]/v[3] are dead but hold the frame). + * - `q` is a SOURCE-LEVEL pointer local for v[1] (cookbook §32#1): a base living in a + * callee-saved reg across calls/branches can only come from a source local. Pinning it to + * $17 pushes arg0 into $16, which reproduces the target's $s0=arg0 / $s1=&v[1] exactly. + * - `&v[0]` must be REMATERIALIZED at both call sites (`addiu $aN,$sp,0x10`). That only + * happens in a narrow whole-function CSE fork (§83d): w/h declared s16 (not s32). With + * s32 w/h, gcc's CSE commons the struct-copy's source-address pseudo (created by + * mips.c's block-move `copy_addr_to_reg`) into both call sites, hoisting sp+0x10 into a + * FIFTH callee-saved register (frame 0x48, +3 ins = 73). Verified across ~120 spellings: + * CFG shape, inline helpers, nested-block ptr + §2 output-only volatile kill, reg-tie + * barriers, packed/aligned copies and array/struct/pair layouts all leave the fork alone. + * + * REMAINING BLOCKER (idx 37..43): the target's min block is + * lh $v1,8($v0) / lh $v0,0xA($v0) / nop / addu $a0,$v0,$zero / + * slt $v0,$v0,$v1 / beqz / (delay) addu $s2,$v1,$zero / addu $s2,$a0,$zero + * i.e. form-b (`m = w; if (h < w) m = h;`) with (a) NO coalescing of m into w's register and + * (b) an explicit copy of h into $a0 because the slt overwrites h's $v0. Every C spelling + * that yields form-b semantics without the m/w coalescing also flips the CSE fork above and + * costs 3 instructions elsewhere; the only 72-ins state found keeps form-a and pins w to $3 + * as s16, whose HImode load is `lhu` + `sll/sra` (3 ins) instead of `lh` + `nop`/`addu` (3). + * Net: same length, 6 opcode/register mismatches. + */ +extern void func_8012F038(); +extern void func_8012F14C(); +extern s32 func_80135888(s32, s32, s32, s32); +extern u16 D_80126B5E; +extern u16 D_80126B62; +extern u16 D_80126B66; +extern u8 D_80126BE0[]; + +void func_80132F40(s32 arg0) +{ + typedef struct { u16 vx, vy, vz, pad; } Svec_80132F40; + + Svec_80132F40 v[4]; + register Svec_80132F40 *q __asm__("$17"); + s16 *p; + register s16 w __asm__("$3"); + register s32 h __asm__("$2"); + s32 m; + + q = &v[1]; + v[0].vx = D_80126B5E; + v[0].vy = D_80126B62; + v[0].vz = D_80126B66; + v[1] = v[0]; + + if (func_80135888(*(s32 *)(arg0 + 0x20), *(s32 *)(arg0 + 0x58), + (s32)D_80126BE0, (s32)q) != 0) { + p = (s16 *)((*(s32 *)(arg0 + 0x58) & 0x0FFFFFFF) | 0x80000000); + w = p[4]; + h = p[5]; + if (h < w) { m = h; } else { m = w; } + func_8012F038(*(s32 *)(arg0 + 0x20) + 0x34, &v[0], q); + v[1].vy = m; + func_8012F14C(*(s32 *)(arg0 + 0x20) + 0x34, q, &v[0]); + D_80126B5E = v[0].vx; + D_80126B62 = v[0].vy; + D_80126B66 = v[0].vz; + } +} diff --git a/.run/wave22/func_80133298.c b/.run/wave22/func_80133298.c new file mode 100644 index 0000000000..bdc2dba216 --- /dev/null +++ b/.run/wave22/func_80133298.c @@ -0,0 +1,37 @@ +typedef signed char s8; +typedef unsigned char u8; +typedef signed short s16; +typedef unsigned short u16; +typedef signed int s32; +typedef unsigned int u32; + +extern s32 D_80126B5C; +extern s32 D_80126B60; +extern s32 D_80126B64; + +void func_8012B2CC(s32); +void func_8012F0BC(s32*, s32*, s32*); +void func_8012F1A4(s32*, s32, s32*); +void func_8013339C(short*, short*); + +void func_80133298(s32 arg0) +{ + typedef struct { s32 vx, vy, vz, pad; } Vec_80133298; + typedef struct { s16 m[3][3]; s32 t[3]; } Mtx_80133298; + Vec_80133298 vin; + Vec_80133298 vout; + Vec_80133298 tmp; + Mtx_80133298 mtx; + + vin.vx = D_80126B5C; + vin.vy = D_80126B60; + vin.vz = D_80126B64; + func_8012F0BC((s32*)(*(s32*)(arg0 + 0x20) + 0x34), (s32*)&vin, (s32*)&tmp); + func_8012B2CC(arg0); + mtx = *(Mtx_80133298*)(*(s32*)(arg0 + 0x20) + 0x34); + func_8013339C((short*)&mtx, (short*)(*(s32*)(arg0 + 0x20) + 0x18)); + func_8012F1A4((s32*)&mtx, (s32)&tmp, (s32*)&vout); + D_80126B5C = vout.vx; + D_80126B60 = vout.vy; + D_80126B64 = vout.vz; +} diff --git a/.run/wave22/func_80135260.c b/.run/wave22/func_80135260.c new file mode 100644 index 0000000000..6489cc82a0 --- /dev/null +++ b/.run/wave22/func_80135260.c @@ -0,0 +1,126 @@ +/* func_80135260 — "is reachable/hittable" gate (136 ins, ov_SC01_077 jr_8012ACE0). + * + * Sibling of the already-banked func_80135D20 (src/ov_SC03_099/ov_SC03_099_jr_80135D20.c L659+): + * same func_80135480 switch (jtbl_801D8158, cases 0..4, default falls through with p/flag/q + * UNINITIALISED — reproduced by leaving the switch without a default), same peeled-probe + + * rotated list loop, different `elsepath`. + * + * Load-bearing details (all byte-proven by ablation against match_one): + * - func_80135480 is DEFINED later in this same TU (L3261) returning s16. Calling it through an + * `s32 (*)(...)` cast is REQUIRED: the target does NOT re-extend the callee result before the + * switch (`addu $v1,$v0,$zero` only); a plain s16-returning call costs an extra `sll/sra 16`. + * The cast is call-site-local, so func_80135480's own definition is untouched (no //@EDIT). + * - D_801870AC / D_801870B0 / D_801870B8 MUST be declared as 4-byte POINTERS + * (`extern u16 *D_801870Bx;`), not the canonical `extern u8` + `(*(u16**)&sym)` cast. With the + * u8 form gcc CSEs `&sym` into two callee-saved regs ($s1/$s5), which costs a 7th saved + * register ($s6 for arg3) and 3 extra instructions. This is exactly the retype already + * documented in this TU's @stuck note at L3127. The decls are BLOCK-SCOPE-compatible with the + * TU's other `extern u8 D_801870AC;` bodies (the TU already mixes both forms: L2461 vs L2587). + * - `u8 dead[56];` is a DEAD local that only sets the frame: without it the frame is 0x30, the + * target's is 0x68 (gcc-2.7.2 assign_stack_local runs at expand time, so an unreferenced + * aggregate still owns its slot). Body bytes are already exact at 136/136 without it — the 16 + * residual diffs are purely the sp/save-slot immediates. + * - explicit goto layout (loop / hit / body / elsepath) reproduces the target's block ORDER: the + * `hit` block sits physically between the loop test and the loop body, and its leading + * `move a0,flag` gets copied into all four branch delay slots. + * - the two D_801870xx compare fields are read `lhu` -> u16; the func_80134A74 args are `(s16)` + * casts of those u16 lvalues (combine folds the fresh ones into `lh`, and sll/sra the one that + * is still live in $a1 from the 0xFF80 compare). + * - func_80135D20's §5a cross-jump barrier is NOT needed here (ablated: MATCH either way). + */ +extern s16 func_80135480(void*, s32, s16*, s16*); +extern s32 func_80135EB0(void *arg0, s32 arg1_); +extern s32 func_80136A94(s32 a0, s32 a1, s32 a2, s32 a3); +extern int func_80134A74(int, s16, s16, int); + +s32 func_80135260(s32 arg0, s32 arg1, s16 *arg2, s16 *arg3) +{ + /* BLOCK scope on purpose: this TU already declares D_801870AC/B0/B8 as `extern u8` inside + * other function bodies (L2461-2463) and as `extern u16 *` inside others (L2587-2589). + * A FILE-scope pointer decl here would make those inner `extern u8` decls conflicting-types. + * The same reason keeps D_801D9504/D_801D9524 local (cf. the TU's own L3268-3270). */ + extern u16 *D_801870AC; + extern u16 *D_801870B0; + extern u16 *D_801870B8; + extern s32 D_801D9504; + extern s32 D_801D9524; + + u8 dead[56]; + s32 *p; + s32 flag; + s32 q; + u16 *pa; + u16 *pb; + u16 *pc; + u16 *pd; + u16 *pe; + u16 t; + + switch (((s32 (*)(void *, s32, s16 *, s16 *))func_80135480)((void *)arg0, arg1, arg2, arg3)) { + case 0: + return 0; + case 1: + q = arg0 + 0x34; + p = (s32 *)((arg1 & 0xFFFFFFF) | 0x80000000); + flag = 0; + break; + case 2: + q = arg0 + 0x34; + p = (s32 *)((arg1 & 0xFFFFFFF) | 0x80000000); + flag = 1; + break; + case 3: + q = arg0 + 0x34; + p = &D_801D9524; + flag = 0; + break; + case 4: + q = (s32)&D_801D9504; + p = &D_801D9524; + flag = 1; + break; + } + if (arg1 >= 0) { + goto elsepath; + } + if (func_80135EB0(p, 0) != 0) { + goto hit; + } + p = (s32 *)*p; + if (p == 0) { + return 0; + } +loop: + if (func_80135EB0(p, 0) == 0) { + goto body; + } +hit: + func_80136A94(flag, arg0, (s32)arg3, q); + return 1; +body: + p = (s32 *)*p; + if (p != 0) { + goto loop; + } + return 0; +elsepath: + pa = D_801870B0; + pb = D_801870AC; + pc = D_801870B8; + pc[0] = pa[0] - pb[0]; + pc[1] = pa[1] - pb[1]; + pc[2] = pa[2] - pb[2]; + if (func_80134A74(0, (s16)pb[0], (s16)pb[2], (int)p) != 0) { + goto hit; + } + pd = D_801870AC; + pe = D_801870B0; + t = pe[0]; + if (((pd[0] & 0xFF80) == (t & 0xFF80)) && ((pd[2] & 0xFF80) == (pe[2] & 0xFF80))) { + return 0; + } + if (func_80134A74(0, (s16)t, (s16)pe[2], (int)p) != 0) { + goto hit; + } + return 0; +} diff --git a/.run/wave22/func_80138C60.c b/.run/wave22/func_80138C60.c new file mode 100644 index 0000000000..72c7666084 --- /dev/null +++ b/.run/wave22/func_80138C60.c @@ -0,0 +1,49 @@ +extern s32 func_80139BE0(s32); +void func_80139C7C(u8*); +s32 func_8013A8FC(s32); + +s32 func_80138C60(s32 arg0) +{ + typedef struct { + u8 b0; + u8 b1; + u8 b2; + u8 b3; + } Q_80138C60; + extern Q_80138C60 D_801870EC[]; + u16 t; + u8 b; + s32 f; + s32 idx; + + if (*(s32 *)(arg0 + 8) & 0x2000) { + return 1; + } + + t = *(u8 *)(arg0 + 0x22) & 7; + *(s16 *)(arg0 + 0x18) = t; + if (t < 2) { + *(u8 *)(arg0 + 0x20) = 0; + } + + f = *(s32 *)(arg0 + 8); + if (f & 0x40) { + *(u8 *)(arg0 + 0x22) = (*(u8 *)(arg0 + 0x22) & 0x67) | (f & ~0x67); + } + + b = *(u8 *)(arg0 + 0x22); + if (b & 0x80) { + idx = (b & 0x18) >> 3; + } else { + idx = (b & 0x78) >> 3; + if (b & 0x60) { + *(s16 *)(arg0 + 0x1C) = 0; + } + } + + *(Q_80138C60 *)(arg0 + 0x24) = D_801870EC[idx]; + + func_80139BE0(arg0); + func_80139C7C((u8 *)arg0); + return func_8013A8FC(arg0); +} diff --git a/.run/wave22/func_8013B598.c b/.run/wave22/func_8013B598.c new file mode 100644 index 0000000000..09a87d0924 --- /dev/null +++ b/.run/wave22/func_8013B598.c @@ -0,0 +1,34 @@ +/* func_8013B598 (ov_SC01_077_o0, -O0): fills entry a0 of the 0x1C-stride table at + * D_801DAA08 from a 3-halfword source record. + * + * -O0 CONSTANT-OFFSET FOLD (the quirk that stalled this -O0 cluster in Phase 19): + * at -O0 gcc-2.7.2 folds a constant offset into the memory operand ONLY through a + * COMPONENT_REF / a real ARRAY_REF on an array-typed DECL: + * D_801DAA08[a0].f4 = ... -> lui $at,%hi(sym); addu $at,$at,idx; sh %lo(sym+4)($at) + * A CAST base defeats it (NOP_EXPR over the ADDR_EXPR), and so does pointer indexing: + * ((S *)D_801DAA08)[a0].f4 -> lui;addiu;addu;sh 4(reg) (+2 ins per store) + * a1[1] -> addiu $v1,$a0,2; lhu 0($v1) (+1 ins per load) + * So the table extern must carry the FULL struct type (not the pad-only canonical + * E_3B7AC), and the source record must be read as ->field, not as a[i]. Casting the + * incoming POINTER is harmless (a NOP on a register), only the base decl's type matters. + * + * Both types are body-scoped (cookbook §100) and the extern is declared inside the body, + * matching the established -O0 sibling pattern in src/ov_SC07_010/ov_SC07_010_o0.c + * (func_8013B7AC: `extern E_3B7AC D_801A9420[];` in-body). NOTE FOR BANKING: the canonical + * sig-layer decl `extern E_3B7AC D_801DAA08[];` (f0 + pad only) CANNOT be used here — it has + * no f4/f6/f8/fC members. The in-body decl below must be the one that survives; do not also + * emit a file-scope E_3B7AC decl for D_801DAA08 (conflicting types). + * + * 66/66 instructions byte-identical (match_one --o0). + */ +void func_8013B598(s32 a0, u16 *a1) { + typedef struct { u16 f0; u16 f2; u16 f4; } Src_8013B598; + typedef struct { s32 f0; u16 f4; u16 f6; u16 f8; u16 fA; s32 fC; u8 pad[0xC]; } Spr_8013B598; + extern Spr_8013B598 D_801DAA08[]; + + D_801DAA08[a0].f0 = 1; + D_801DAA08[a0].f4 = ((Src_8013B598 *)a1)->f0; + D_801DAA08[a0].f6 = ((Src_8013B598 *)a1)->f2; + D_801DAA08[a0].f8 = ((Src_8013B598 *)a1)->f4; + D_801DAA08[a0].fC = 0x100; +} diff --git a/.run/wave22/func_8013B6A0.c b/.run/wave22/func_8013B6A0.c new file mode 100644 index 0000000000..9e5b547da3 --- /dev/null +++ b/.run/wave22/func_8013B6A0.c @@ -0,0 +1,24 @@ +void func_8013B6A0(s32 idx, u16 *src, s32 val) +{ + typedef struct { + s32 f0; + u16 f4; + u16 f6; + u16 f8; + u16 fA; + s32 fC; + u8 pad[0xC]; + } Ent_8013B6A0; + typedef struct { + u16 f0; + u16 f2; + u16 f4; + } Src_8013B6A0; + extern Ent_8013B6A0 D_801DAA08[]; + + D_801DAA08[idx].f0 = 1; + D_801DAA08[idx].f4 = ((Src_8013B6A0 *)src)->f0; + D_801DAA08[idx].f6 = ((Src_8013B6A0 *)src)->f2; + D_801DAA08[idx].f8 = ((Src_8013B6A0 *)src)->f4; + D_801DAA08[idx].fC = val; +} diff --git a/.run/wave22/func_80140958.c b/.run/wave22/func_80140958.c new file mode 100644 index 0000000000..dd22eb5ce6 --- /dev/null +++ b/.run/wave22/func_80140958.c @@ -0,0 +1,180 @@ +/* func_80140958 - ov_SC01_077 @ 0x80140958, 260 ins, -O2. + * + * STATUS: NEAR - 260/260 instructions, 10 mismatched (match_one --json: + * {"status":"near","closeness":10,"nins":260,"verdict":{"klass":"ADDRESSING","profile":"cse"}}). + * + * Canonical sig_hints decls are used VERBATIM. Everything else is derived from the asm: + * func_80140D68 - jal target, 5 args (5th at 0x10($sp)); ret is the new OT ptr + * D_80115143 - lbu %lo(D_80115143) + * D_8018793C[] - Prim4, indexed by (s16)i (stride 4) + * D_80187E9C[] / D_80187EAC[] - s16 tables, indexed by (D_80115116 & 7) + * D_801879BC / D_801879BE - u16 scalars + * D_800AE7BC[] - Env (u32 *ot; u32 pad[4]) 0x14 stride; the addPrim OT array + * (same object resident.c calls Env_800D29F8) + * + * KEY FINDINGS (worth a cookbook entry): + * 1. The 0x80115110 block is reached through THREE held base pointers, not through + * separate externs: $s4=&D_80115110, $s5=$s4+0xA, $s7=$s4+6 (and $s6=&D_801879BE). + * Separate externs can never produce `addiu $s5,$s4,0xA` (no %hi CSE across symbols), + * so the source must own real pointer variables. + * 2. The `if (i < n) { ; do {...} while (i < n); }` shape is what puts + * the base-pointer setup AFTER the loop-guard `beqz` (a plain `while` puts explicit + * assignments in the entry block instead, shifting the whole prologue). + * 3. `t` must be `s16` and func_8014168C must be CALLED THROUGH an (s32(*)(s16)) cast: + * with the canonical `s16` return gcc emits `sll 16 / srl 15` instead of the target's + * single `sll $v0,$v0,1`, and the s16 truncation then lands at the AC8 merge point. + * 4. `((s8 *)D_80115158)[k * 2]` materialises the address into a register (and LICM hoists + * it). Hoisting `k*2` into its own local (`s32 k2 = k * 2;`) restores the target's + * in-loop `lb $v0, D_80115158($v0)` form. (cookbook: computed index vs. plain reg index.) + * 5. The SPRT biv must be anchored ONE ABOVE the highest field (ot+0x14, all-negative + * offsets) or gcc splits it into two bivs; it then re-anchors to ot+0x10 by itself. + * + * RESIDUAL (10 ins, all one root cause - contiguous idx 134..149): + * gcc constant-folds `m == 3` so `k = m` becomes `li $a3,3` and `m*4` becomes `li $t3,12`, + * where the target keeps `addu $a3,$v1,$zero` + `sll $t3,$v1,2`. Because $v1 then dies + * early, gcc also fills two delay slots (`li $v0,3`, `addiu $v1,$s3,1`) that the target + * leaves as nops. Writing `m == i` (or `m == c3`) defeats the fold but makes gcc coalesce + * j/k into the wrong pair ($a2 <-> $a3 swapped) - 16 mismatched instead of 10. Tried: + * pinning k/$a3, pinning j/$a2, pinning m/$v1, an "=r"/"0" opaque barrier on m, + * re-reading b[0], `m == i`, `m == c3`, `m == (s32)i`, volatile load. None reach 0. + */ +/* DRAFT-ONLY: Hw4/Prim4 come from engine_types.h in the real TU (via engine_core.h). + * They are reproduced here ONLY so the standalone match_one compile has them. + * DROP THESE TWO LINES WHEN BANKING. */ +typedef struct { s16 x; s16 y; } Hw4; +typedef struct { u16 f0; s16 f2; } Prim4; + +extern s16 func_8014168C(s16); + +extern short D_800B9A02; +extern u16 D_80115110; +extern u8 D_80115140[]; +extern s16 D_8011514E; +extern u8 D_80115158[]; +extern Hw4 D_8011516A[]; +extern Prim4 D_8018798C[]; +extern Prim4 *D_80187A80[]; + +/* derived from asm */ +extern u8 D_80115143; /* lbu %lo(D_80115143) */ +extern Prim4 D_8018793C[]; /* base + ((s16)i)*4 */ +extern s16 D_80187E9C[]; /* lh %lo + (f6&7)*2 */ +extern s16 D_80187EAC[]; /* lh %lo + (f6&7)*2 */ +extern u16 D_801879BC; /* lhu %lo */ +extern u16 D_801879BE; /* lhu 0($s6) - held ptr */ +/* jal target; 5 args, 5th on the stack at 0x10($sp) */ +extern s32 *func_80140D68(s32 *, Prim4 *, s32, s32, s32); + +s32 *func_80140958(ot, i, n) +s32 *ot; +s16 i; +s16 n; +{ + typedef struct { + u32 *ot; + u32 pad[4]; + } Env_80140958; + extern Env_80140958 D_800AE7BC[]; + + Prim4 *p; + s16 t; + s32 c3; + /* three held bases into the 0x80115110 block ($s4 / $s5=+0xA / $s7=+6) */ + register u16 *a __asm__("$20"); + register u16 *b __asm__("$21"); + register u16 *c __asm__("$23"); + register u16 *e __asm__("$22"); + register u16 *pb __asm__("$10"); + register u32 m24 __asm__("$8"); + register u32 mhi __asm__("$9"); + register s32 eight __asm__("$2"); + + if (i < n) { + c3 = 3; + a = &D_80115110; + b = a + 5; + c = a + 3; + e = &D_801879BE; + do { + if (a[0] == 0 && a[1] != c3 && a[1] < 6) { + if (i == a[5]) { + ot = func_80140D68(ot, &D_8018793C[i], i, D_80187E9C[a[3] & 7], 0); + } + } else { + p = D_80187A80[i]; + if (p != 0 && i == b[0] && i != 6) { + if (i == 2 && *(s16 *)(b + 7) != 0) { + p = D_8018798C; + } + if (i != c3) { + t = ((s32 (*)(s16))func_8014168C)(i) * 2; + } else { + t = (*(u8 *)&D_8011514E - D_80115143) * 2; + } + ot = func_80140D68(ot, p, i, D_80187EAC[c[0] & 7], t); + if (i == 2 && *(s16 *)(c + 9) == 1 && *(s16 *)(c + 12) != 0) { + ot = func_80140D68(ot, p, 2, 8, (*(s16 *)(c + 12) & 0xF) * 2); + } + } + } + if (i == c3) { + s32 m = b[0]; + + if (m == 3 && (b[-2] & 8) != 0) { + s32 k; + u8 *q = (u8 *)ot + 0x14; + s16 j; + s16 y; + + j = 0; + k = m; + pb = (u16 *)&D_800B9A02; + m24 = 0xFFFFFF; + mhi = 0xFF000000; + + for (; j < 2; j++) { + if (j == 0) { + if (D_80115140[k] == 0) { + continue; + } + q[-7] = 0x30; + y = *e - 4; + } else { + s32 k2 = k * 2; + + if (((s8 *)D_80115158)[k2] - ((s8 *)D_80115140)[k] < 2) { + continue; + } + q[-7] = 0x38; + y = *e + 3; + } + *(s16 *)(q - 10) = y; + __asm__("" ::: "memory"); + *(u32 *)ot = 0x4000000; + q[-8] = 0x78; + *(u32 *)(q - 0x10) = 0x64808080; + *(s16 *)(q - 6) = 0x4056; + *(s16 *)(q - 0xC) = D_801879BC + (u16)D_8011516A[m].x + 0x4A; + eight = 8; + *(s16 *)(q - 2) = eight; + *(s16 *)(q - 4) = eight; + + *(u32 *)ot = (*(u32 *)ot & mhi) | + (D_800AE7BC[*pb].ot[2] & m24); + { + register u32 *op __asm__("$4"); + + op = D_800AE7BC[*pb].ot; + + op[2] = (op[2] & mhi) | ((u32)ot & m24); + } + q += 0x14; + ot += 5; + } + } + } + i = i + 1; + } while (i < n); + } + return ot; +} diff --git a/.run/wave22/func_80140D68.c b/.run/wave22/func_80140D68.c new file mode 100644 index 0000000000..7edac54129 --- /dev/null +++ b/.run/wave22/func_80140D68.c @@ -0,0 +1,85 @@ +/* func_80140D68 — SPRT (0x14) primitive builder + PsyQ addPrim() into OT_800D29F8[2]. + * + * @class: schedule + * @status: NEAR 9/65 — 65/65 instructions, every instruction byte-correct, ONE positional drift. + * @stuck: the 0xFFFFFF mask register ($t1) builds ATOMICALLY here (lui idx2 / ori idx3); the target + * SPLITS the pair (lui idx2 ... ori idx13, between `sra $a2,$a2,14` and `addiu $a3,$a3,-0xD`). + * Everything from idx15 on is byte-identical; idx3..13 is a pure shift-by-one caused solely by + * that one `ori`. gcc-2.7.2 splits the large constant BEFORE sched2 (verified with + * -fno-schedule-insns2: cc1 emits `li $9,0x00ff0000` + `ori $9,$9,0xffff` as two insns), so the + * position is decided by the pre-sched2 RTL LUID of the `ori`, which is NOT reachable from the + * source in this shape. Explored and byte-measured (~200 compiles): + * - 126 header-statement orderings (q/dx/x/y/w/h permutations) -> all 9, ori pinned at idx3 + * - PTag bitfield layouts (u32:24+u32:8 / u32:24+u8 / u32:24) -> all 9 + * - plain-local and `register __asm__("$9")` mask + redundant `& mlo`, swept over 12 source + * positions -> 9 or 12, never 13 + * - `volatile short` decl instead of a cast at use -> 40+ + * The ONE family that moves the `ori` is: pointer-cast header (no Sprt struct) + a SHARED + * `u32 *ot` local + LITERAL masks — there the constant sinks in RTL to just after the volatile + * lhu and schedules at idx14 (one slot late). Best of that family is kept at + * .run/wave22/_a80140D68/x_pin2.c (closeness 12: idx13/14 ori<->addiu swap + a clean $v0/$v1 + * swap in the second addPrim half, because a single shared `ot` pseudo cannot take two registers + * while splitting it into two pseudos moves the `ori` back to idx23). Permuter-shaped. + * + * Signature (read off the asm): a0 = SPRT out, a1 = s16 *src, a2 = s16 idx (in-callee + * sll16/sra14 => K&R narrow param, cookbook §99), a3 = s32 dx, 0x10($sp) = s16 ofs (K&R narrow; + * ANSI `s16 ofs` yields `lh`+`sll 1` = 2 ins instead of the target's lw+sll16+sra15). + * + * Draft-local shims: these two typedefs already exist VERBATIM in src/shared/engine_types.h + * (Hw4 @833, Env_800D29F8). The guard makes the draft self-contained for match_one (which only + * prepends common.h) while collapsing to nothing once banked into a TU that includes the header. */ +#ifndef BFM_ENGINE_TYPES_H +typedef struct { s16 x; s16 y; } Hw4; +typedef struct { + u32 *ot; /* 0x00 */ + u32 pad[4]; /* 0x04..0x13 */ +} Env_800D29F8; /* 0x14 stride */ +#endif + +extern Hw4 D_8011516A[]; +extern short D_800B9A02; +extern Env_800D29F8 D_800AE7BC[]; + +#define OT8_80140D68 (*(u32 *)(D_800AE7BC[*(volatile u16 *)&D_800B9A02].ot + 2)) + +u32 *func_80140D68(out, src, idx, dx, ofs) + u32 *out; + s16 *src; + s16 idx; + s32 dx; + s16 ofs; +{ + typedef struct { + u32 tag; /* 0x00 */ + u32 code; /* 0x04 */ + s16 x, y; /* 0x08 */ + u8 u, v; /* 0x0C */ + u16 clut; /* 0x0E */ + s16 w, h; /* 0x10 */ + } Sprt_80140D68; /* 0x14 */ + typedef struct { u32 addr : 24; u32 len : 8; } PTag_80140D68; + + register u32 mhi __asm__("$8"); + Sprt_80140D68 *p; + s16 *q; + + p = (Sprt_80140D68 *)out; + p->tag = 0x04000000; + p->u = 0x70; + p->v = 0x10; + mhi = 0x64808080; + p->code = mhi; + p->clut = 0x4056; + + dx -= 0xD; + q = (s16 *)(ofs * 2 + (s32)src); + p->x = q[0] + D_8011516A[idx].x + dx; + p->y = q[1] - 4; + p->h = 0x10; + p->w = 0x10; + + ((PTag_80140D68 *)p)->addr = ((PTag_80140D68 *)&OT8_80140D68)->addr; + ((PTag_80140D68 *)&OT8_80140D68)->addr = (u32)p; + + return out + 5; +} diff --git a/.run/wave22/func_80148E54.c b/.run/wave22/func_80148E54.c new file mode 100644 index 0000000000..35bd16f18b --- /dev/null +++ b/.run/wave22/func_80148E54.c @@ -0,0 +1,25 @@ +extern s32 ratan2(s32, s32); +extern s32 D_801151D4; +/* derived from asm: lui/addu/lw %lo(D_80188694) indexed by (u16>>12)*4, then jalr with no args; + * the result is sign-extended from 16 bits => the table's functions return s16. */ +extern s16 (*D_80188694[])(); + +s32 func_80148E54(s32 arg0) { + register s32 tmp __asm__("$4") = (ratan2(*(s32 *)(D_801151D4 + 0x44) - *(s32 *)(D_801151D4 + 0x50), + *(s32 *)(D_801151D4 + 0x48) - *(s32 *)(D_801151D4 + 0x3C)) - 0x400) & 0xFFF; + s32 ang; + __asm__("" : "=r"(tmp) : "0"(tmp)); + ang = tmp; + + switch (*(u8 *)(arg0 + 0xA9)) { + case 0x41: + return D_80188694[*(u16 *)(arg0 + 0xAA) >> 12](); + case 0x53: + case 0x73: + if ((*(u16 *)(arg0 + 0xAE) & 0xFF) == 0x80 && (*(u16 *)(arg0 + 0xAE) >> 8) == 0x80) { + return -1; + } + return (ang + ratan2((*(u16 *)(arg0 + 0xAE) & 0xFF) - 0x80, + 0x80 - (*(u16 *)(arg0 + 0xAE) >> 8))) & 0xFFF; + } +} diff --git a/.run/wave22/func_8014A738.c b/.run/wave22/func_8014A738.c new file mode 100644 index 0000000000..0d4c808b64 --- /dev/null +++ b/.run/wave22/func_8014A738.c @@ -0,0 +1,65 @@ +typedef signed char s8; +typedef unsigned char u8; +typedef short s16; +typedef unsigned short u16; +typedef int s32; +typedef unsigned int u32; + +extern s32 func_80029178(s32); +extern u8 D_801202A0[]; + +s32 func_8014A738(void *arg0) { + typedef struct { + s16 vx; + s16 vy; + s16 vz; + s16 pad; + } Vec_8014A738; + typedef struct { + u8 pad0[6]; + u16 x; + u8 pad8[6]; + u16 z; + u8 pad10[0xB1]; + u8 state; + u8 padC2[0x4A]; + } Ent_801202A0; + Vec_8014A738 d; + u32 i; + Ent_801202A0 *base; + + if ((func_80029178(0x83) & 0xFF) == 0) { + return 0; + } + base = (Ent_801202A0 *)D_801202A0; + for (i = 0; i < 0x60; i++) { + if (base[i].state == 7) { + d.vx = ((Ent_801202A0 *)arg0)->x - base[i].x; + d.vz = ((Ent_801202A0 *)arg0)->z - base[i].z; + if (d.vx >= 0) { + if (d.vx < 0x40) { + goto zcheck; + } + } else { + if (-d.vx < 0x40) { + goto zcheck; + } + } + continue; + zcheck: + if (d.vz >= 0) { + if (d.vz < 0x40) { + goto found; + } + } else { + if (-d.vz < 0x40) { + goto found; + } + } + continue; + found: + return 1; + } + } + return 0; +} diff --git a/.run/wave22/func_80163534.c b/.run/wave22/func_80163534.c new file mode 100644 index 0000000000..878bffaa33 --- /dev/null +++ b/.run/wave22/func_80163534.c @@ -0,0 +1,51 @@ +/* canonical (sig_hints) — return type widened s32 so the `sw $v0` after the jal has a + source; the hint's `void` cannot express the store. */ +extern s32 func_80163664(s32, u16, u16, s32, s32, s32, s32, s32, s32, s32, u16, s32, s32); + +/* canonical (sig_hints) */ +extern s32 D_80115100; +extern s32 D_80115200; +extern u16 D_80126B18[]; + +/* derived from the asm: separate lui/%lo per symbol => distinct externs */ +extern s32 D_80115204; +extern s32 D_80115208; +extern u16 D_801270B0[]; +extern u16 D_801270B2; +extern u16 D_801270B4; +/* LOAD-BEARING: D_80126B1A must be declared/stored as an ARRAY, not a scalar. + gcc-2.7.2 true_dependence() lets an in-struct MEM with an unstable (register) + address bypass a not-in-struct MEM with a stable (symbol) address. As a scalar, + the `sh %lo(D_80126B1A)` store does NOT conflict with `a5[2]`, so the block-6 + `lhu $t0,4($t0)` gets hoisted into the D_80126B1A block's load-delay slot and the + whole filler queue shifts one slot early (75 ins, no `nop`). ARRAY_REF sets + MEM_IN_STRUCT_P on the store => the two MEMs conflict => the load stays put. */ +extern u16 D_80126B1A[]; +extern u16 D_80126B1C; +extern s32 D_80114EB0; +extern s32 D_80114EC8; +extern s32 D_8011DAF0; +extern s32 D_80115298; +extern s32 D_80126734; + +void func_80163534(s32 a0, u16 a1, u16 a2, s32 a3, u16 a4, u16 *a5) +{ + s32 *p = &D_80115200; + + *p = 0; + D_80115204 = 0; + D_80115208 = 0; + + D_801270B0[0] = *(u16 *)(a0 + 0x44) + a5[0]; + D_801270B2 = *(u16 *)(a0 + 0x46) + a5[1]; + D_801270B4 = *(u16 *)(a0 + 0x48) + a5[2]; + + D_80126B18[0] = *(u16 *)(a0 + 0x6) + a5[0]; + D_80126B1A[0] = *(u16 *)(a0 + 0xA) + a5[1]; + D_80126B1C = *(u16 *)(a0 + 0xE) + a5[2]; + + *p = func_80163664(a0, a1, a2, (s32)D_801270B0, (s32)D_80126B18, + (s32)&D_80114EB0, (s32)&D_80114EC8, (s32)&D_80115100, + (s32)&D_8011DAF0, a3, a4, (s32)&D_80115298, + (s32)&D_80126734); +} diff --git a/.run/wave22/func_80171B4C.c b/.run/wave22/func_80171B4C.c new file mode 100644 index 0000000000..524b357d63 --- /dev/null +++ b/.run/wave22/func_80171B4C.c @@ -0,0 +1,33 @@ +extern void func_80146D90(s32); +extern s32 ratan2(s32, s32); +extern s32 D_801151D4; + +s32 func_80171B4C(arg0, arg1) +s32 arg0; +u8 arg1; +{ + s32 p, ang, idx; + u8 c; + p = D_801151D4; + *(u8 *)(arg0 + 0xA9) = 0x41; + ang = ratan2(*(s32 *)(p + 0x68) - *(s32 *)(p + 0x5C), + *(s32 *)(p + 0x70) - *(s32 *)(p + 0x64)); + idx = ((*(s16 *)(*(s32 *)(arg0 + 0x20) + 0x12) - ((ang + 0x800) & 0xFFF)) + 0x100) & 0xE00; + switch (idx / 0x200) { + case 0: *(u16 *)(arg0 + 0xAA) = 0x1000; break; + case 1: *(u16 *)(arg0 + 0xAA) = 0x3000; break; + case 2: *(u16 *)(arg0 + 0xAA) = 0x2000; break; + case 3: *(u16 *)(arg0 + 0xAA) = 0x6000; break; + case 4: *(u16 *)(arg0 + 0xAA) = 0x4000; break; + case 5: *(u16 *)(arg0 + 0xAA) = 0xC000; break; + case 6: *(u16 *)(arg0 + 0xAA) = 0x8000; break; + case 7: *(u16 *)(arg0 + 0xAA) = 0x9000; break; + } + c = *(u8 *)(arg0 + 0x20C); + *(u8 *)(arg0 + 0x20C) = c + 1; + if (c != arg1) { + return 0; + } + func_80146D90(arg0); + return 1; +} diff --git a/.run/wave22/func_80176734.c b/.run/wave22/func_80176734.c new file mode 100644 index 0000000000..48bcfe3fa9 --- /dev/null +++ b/.run/wave22/func_80176734.c @@ -0,0 +1,275 @@ +/* func_80176734 — ov_SC01_077 HUD sync. NEAR: 366/371 ins, close=217, LENGTH-DRIFT -5. + * + * Base idiom = the matched sibling func_80176218 in the same TU: + * st=&D_8011F7A8 ($s3), cach=st+0x48 ($s1), flag=st+0xE0 ($s2), cur=D_80078E78 ($s4), + * param copy ($s6), chg ($s5), r/chg2 ($s0). Index expr written OFFSET-FIRST + * (((p<<16)>>14) + st + 0x28) so gcc emits `addu rd,off,st` like the target. + * `u8 zbuf[8]` is NOT dead weight: it is what makes get_frame_size() give the target's + * 0x40 frame (verified — without it the prologue is `addiu sp,sp,-56`). buf[4] works too. + * The 0x2E block writes the EQUAL case first so gcc emits `bne` (not `beq`) — byte-verified. + * The tail routes t+5 / t+9 through temps so gcc emits `addiu $a2,$v1,5; addu $v0,$v0,$a2` + * instead of reassociating to (lhu+5)+t. + * + * RESIDUAL — 5 instructions the target's gcc kept and this draft's gcc optimizes away. + * All 5 are "target gcc was LESS aggressive"; each was isolated and reproduced, but every + * known lever for them wrecks the register allocation (see below), so none are applied. + * b1 (+3): missing `andi $v0,$v1,0xFF` (an explicit QI->SI zero-extend of the D_8011F7B0 + * lbu that combine deletes here via nonzero_bits), missing `addiu $a0,$a1,0x3C` + * (cse folds `p[4]` into `sb 0x40($a1)`), missing the load-delay `nop` after + * `lh D_801152BA` (sched1 hoists the lh here, target keeps it at the branch). + * b2 (-3): `*(u8*)(st+8)` is const-folded to `lui/%lo(D_8011F7A8)+8` here; the target + * keeps `lbu/sb 0x8($s3)`. cse knows st's constant in that block for us and + * not for the target. This fold also costs `st` two references, which is why + * global-alloc gives st $s5 and chg $s3 here (target: st $s3, chg $s5). + * chg (+1)/switch(+1): the target keeps the uncoalesced copies `addu $s5,$v0,$zero` and + * `addu $v0,$s0,$zero` (one copy at .L80176AF4, gas duplicates it into 4 delay + * slots); here cse unifies the temp with the variable and the copy dies. + * end (+2): the target keeps the provably-redundant `beqz $v1,.L80176C84` (re-test of + * chg==0) and the `j .L80176C88`/`sb` split; cse's record_jump_equiv kills it here. + * + * PROVEN LEVERS (each reproduces its target instruction, all rejected as net-negative): + * __asm__("" : "=r"(p) : "0"(p)) before the store -> `addiu $a0,$a1,0x3C` appears (369 ins) + * __asm__("" : "=r"(st) : "0"(st)) before the RMW -> `lbu/sb 0x8($s3)` appears (b2 exact) + * __asm__("" : "=r"(rv) : "0"(r)) before `if (r)` -> switch region becomes 42/42 exact + * Each launder is also a scheduling/liveness barrier: it re-shuffles the call-saved + * allocation and drops aligned agreement from 138/371 to as low as 60/371. A real fix + * has to make cse weaker WITHOUT a barrier (i.e. find the source shape that ends cse's + * extended basic block at .L801767BC), not paper over each instruction. + */ + +extern void func_800183E0(s32); +extern s32 func_801619D0(void*); +extern s32 func_80161A00(s32); +extern s32 func_80161A30(s32); +extern s32 func_80161A60(s32); +void func_801775E0(s32, s32); + +extern u8 D_80078E78[]; +extern u8 D_800B9A13; +extern u8 D_800D43D4; +extern u8 D_800D4414; +extern u8 D_800D45D4; +extern u8 D_8011F7A8; +extern s32 D_80126B58; +extern s16 D_80126D20; +extern u16 D_8018A238; +extern u8 D_8018A2CC[]; +extern s32 D_8018A2E4[]; + +/* derived from the asm (not in sig_hints) */ +extern s16 D_801152BA; +extern u8 D_80115214; +extern u8 D_8011F7B0; +extern s16 D_80126CE0; +extern u16 D_8018A22A[]; +extern u8 D_8018A2D8[]; + +void func_80176734(s32 param_1) +{ + s32 st = (s32)&D_8011F7A8; + s32 cach = st + 0x48; + s32 flag = st + 0xE0; + s32 cur = (s32)D_80078E78; + s32 e; + s32 b; + s32 chg; + s32 chg2; + u16 g; + s32 r; + u8 fv; + u8 zbuf[8]; + + e = *(s32 *)(((param_1 << 16) >> 14) + st + 0x28); + if (D_801152BA != 0) { + u8 *p = (u8 *)(e + 0x3C); + s32 vv = D_8011F7B0 - 0x80; + if (D_8011F7B0 >= 0x80) { + vv = ~D_8011F7B0 - 0x80; + } + p[4] = vv; + *(u8 *)(e + 4) = vv; + *(u8 *)(st + 8) = *(u8 *)(st + 8) + D_80115214; + } else { + *(u8 *)(e + 0x40) = 0x80; + *(u8 *)(e + 4) = 0x80; + } + + fv = *(u8 *)(flag + 0x48); + if (fv != 0) { + b = ((param_1 << 16) >> 14) + st; + *(u8 *)(*(s32 *)(b + 0x28) + 0xD) = D_8018A2D8[fv]; + fv = *(u8 *)(flag + 0x48); + if (fv < 4) { + *(u8 *)(flag + 0x48) = fv + 1; + } else { + u8 cv = *(u8 *)(cach + 0x48); + s32 ee = *(s32 *)(b + 0x28); + if (cv & 0x80) { + *(u16 *)(ee + 0x20) = D_8018A238; + func_800183E0((s32)&D_800D45D4); + } else if (cv != 0) { + *(u16 *)(ee + 0x20) = D_8018A22A[cv]; + func_800183E0(D_8018A2E4[*(u8 *)(cach + 0x48)]); + } + fv = *(u8 *)(flag + 0x48); + if (fv == 5) { + if (*(u8 *)(cach + 0x48) == 0) { + *(u8 *)(flag + 0x48) = 0; + } else { + *(u8 *)(flag + 0x48) = fv + 1; + } + } else if (fv == 10) { + *(u8 *)(flag + 0x48) = 0; + } else { + *(u8 *)(flag + 0x48) = fv + 1; + } + } + } else { + u8 cv = *(u8 *)(cach + 0x48); + u8 sv = *(u8 *)(cur + 0x48); + if (cv != sv) { + if (cv == 0 && sv != 0) { + *(u8 *)(flag + 0x48) = 5; + } else { + *(u8 *)(flag + 0x48) = 0; + } + *(u8 *)(cach + 0x48) = *(u8 *)(cur + 0x48); + { + s32 bb2 = ((param_1 << 16) >> 14) + st; + u8 f2 = *(u8 *)(flag + 0x48); + s32 ee = *(s32 *)(bb2 + 0x28); + *(u8 *)(flag + 0x48) = f2 + 1; + *(u8 *)(ee + 0xD) = D_8018A2D8[f2]; + } + } + } + + { + s16 sv2 = *(s16 *)(cur + 0x2E); + if (*(s16 *)(cach + 0x2E) == sv2) { + if (*(s16 *)(flag + 0x2E) == 0) goto L9C0; + *(s16 *)(flag + 0x2E) = 0; + } else { + *(s16 *)(cach + 0x2E) = sv2; + *(s16 *)(flag + 0x2E) = 1; + } + } + { + s32 ee = *(s32 *)(((param_1 << 16) >> 14) + st + 0x28); + s32 t = (s32)(*(u16 *)(cach + 0x2E)) << 16; + if (t != 0) { + *(u8 *)(ee + 0x49) = D_8018A2CC[t >> 20]; + } else { + *(u8 *)(ee + 0x49) = 0xA0; + } + } +L9C0: + { + s32 fa = *(s16 *)(cach + 0x1E) & 0x8000; + s32 fb = *(s16 *)(cur + 0x1E) & 0x8000; + if (fa != fb) { + s32 arg; + if (fa != 0) { + arg = (s32)&D_800D43D4; + *(s16 *)(cach + 0x1E) = 0; + } else { + *(s16 *)(cach + 0x1E) = -0x8000; + arg = (s32)&D_800D4414; + } + func_800183E0(arg); + } + } + + { + u8 cc3 = D_800B9A13; + if (cc3 == 3) { + chg = 0; + } else { + chg = (*(u8 *)(st + 7) != cc3); + if (chg != 0) { + *(u8 *)(st + 7) = cc3; + } + } + } + + r = 0; + switch (*(u8 *)(cur + 0x48)) { + case 3: + if (func_801619D0(&D_80126B58) != 0) r = 0xFF; + break; + case 4: + if (func_80161A00((s32)&D_80126B58) != 0) r = 0xFF; + break; + case 5: + if (func_80161A30((s32)&D_80126B58) != 0) r = 0xFF; + break; + case 6: + if (func_80161A60((s32)&D_80126B58) != 0) r = 0xBA; + break; + } + + if (r != 0) { + *(u8 *)(flag + 0x47) = 1; + *(u8 *)(cach + 0x4B) = *(u8 *)(cur + 0x48) | 0xF0; + *(u8 *)(cach + 0x47) = ((s32)D_80126D20 << 7) / r; + goto LC88; + } + + if (*(u8 *)(cach + 0x4B) >= 0xF0) { + *(u8 *)(cach + 0x4B) = 0; + } + + if (D_80126CE0 != 0) { + g = *(u16 *)&D_80126CE0; + *(u8 *)(cach + 0x4B) = g; + chg2 = (*(u8 *)(cach + 0x47) != (u8)g); + } else { + g = *(u8 *)(cur + 0x47); + chg2 = 0; + if (*(u8 *)(cach + 0x47) != (u8)g || *(u8 *)(cach + 0x47) == 0x80) { + chg2 = 1; + } + if (*(u8 *)(cur + 0x47) != 0 && *(u8 *)(cach + 0x4B) != 0) { + *(u8 *)(cach + 0x4B) = 0; + *(u8 *)(cach + 0x47) = *(u8 *)(cur + 0x47); + } + } + + if (chg2 == 0 && chg == 0) { + if (*(u8 *)(flag + 0x47) == 0) goto LC88; + if (chg == 0) { + *(u8 *)(flag + 0x47) = 0; + goto LC88; + } + } + { + u8 c47 = *(u8 *)(cach + 0x47); + if (c47 < (s16)g) { + *(u8 *)(cach + 0x47) = g; + } else { + if ((s16)g != 0) { + *(u8 *)(cach + 0x47) = c47 - 3; + } else { + *(u8 *)(cach + 0x47) = c47 - 8; + } + c47 = *(u8 *)(cach + 0x47); + if (c47 == 0 || c47 > 0x80) { + *(u8 *)(cach + 0x47) = 0; + *(u8 *)(cach + 0x4B) = 0; + } else if (c47 < (s16)g) { + *(u8 *)(cach + 0x47) = g; + } + } + *(u8 *)(flag + 0x47) = 1; + } +LC88: + b = ((param_1 << 16) >> 14) + st; + { + s32 t = (*(u8 *)(st + 7) != 0) << 8; + s32 t5 = t + 5; + s32 t9 = t + 9; + *(s16 *)(*(s32 *)(b + 0x28) + 0x32) = *(u16 *)(st + 0x12) + t5; + func_801775E0(*(s32 *)(b + 0x28) + 0x64, + (s16)(*(u16 *)(st + 0x12) + t9)); + } +} diff --git a/.run/wave22/func_80177B5C.c b/.run/wave22/func_80177B5C.c new file mode 100644 index 0000000000..d8d0d53371 --- /dev/null +++ b/.run/wave22/func_80177B5C.c @@ -0,0 +1,149 @@ +/* func_80177B5C (ov_SC01_077) — NEAR: 147/147 ins, closeness 11 (match_one). + * GPU SPRT-chain builder: 2 lead sprites, a 3-iteration digit loop, 2 tail sprites. + * Registers ALL match; stack frame + all 6 callee saves match; TAIL region is byte-exact. + * Residual = 11 pure sched-order swaps in 3 spots: + * [18-21] prologue: gcc emits `lui s2,0x300` (c3) before the loop-mask `lui/ori t8`; target reversed. + * [24-27] prologue: `lui t0,0x300` (ca) lands 3 slots early (target puts it after the `lw` of arg5). + * [86-89] loop: gcc reassociates `cl | (X | 0x4000)` -> `(cl|0x4000) | X`, so it emits + * ori/sll/addiu/or instead of the target's sll/addiu/ori/or. Not defeatable via + * temps, tie-barriers, volatile barriers, operand swap, or << vs * (all tried). + * Levers that got it here (all byte-measured): 12 register pins; K&R-wide params with explicit + * (s16) casts; "raw copy then narrow in place" for arg3/arg5; mk1/cc1 explicit prologue constants; + * ca pinned to $t0 (this alone moved the whole t-reg file into place: 101 -> 75); + * value-temp + tie/volatile barrier + delayed store for each packet's word-2; + * keep-alive asm on tr/xr at the end (stops in-place clobber of $s5/$s1). + */ +typedef signed char s8; +typedef unsigned char u8; +typedef short s16; +typedef unsigned short u16; +typedef int s32; +typedef unsigned int u32; + +/* derived from asm %hi/%lo refs (sig_hints.data was empty): + lbu $t9, %lo(D_8018A300)($at) with $at = %hi(D_8018A300) + (s16)param3 */ +extern u8 D_8018A300[]; + +/* caller decl (src/ov_SC01_077/ov_SC01_077_jr_801734BC.c:3068): + extern u32 *func_80177B5C(u32 *a0, s32 a1, s32 a2, s32 a3, s32 a4); */ +u32 *func_80177B5C(p, bits, tbli, x, y) +u32 *p; +u32 bits; +s32 tbli; +s32 x; +s32 y; +{ + register u32 bb __asm__("$14"); + u32 *q; + register u32 v __asm__("$25"); + register u32 cl __asm__("$3"); + register u32 cs __asm__("$5"); + register u32 ca __asm__("$8"); + s16 i; + u32 mk1; + u32 cc1; + u32 flag; + u32 nn; + register u32 n __asm__("$7"); + register u32 t __asm__("$13"); + u32 col; + u32 uv; + u32 tt; + u32 nv; + u32 x1; + u32 x2; + u32 w; + u32 g; + u32 w3; + register u32 yr __asm__("$16"); + register u32 yt __asm__("$4"); + register u32 tr __asm__("$21"); + register u32 xr __asm__("$17"); + register u32 c3 __asm__("$18"); + register s32 ff __asm__("$19"); + register s32 two __asm__("$20"); + + yt = y; + tr = tbli; + __asm__("" : "=r"(tr) : "0"(tr)); + xr = x; + __asm__("" : "=r"(xr) : "0"(xr)); + mk1 = 0xFFFFFF; + cc1 = 0x74808080; + bb = bits; + t = x + 0xE; + flag = 0x1000000; + i = 0; + two = 2; + ff = 255; + c3 = 0x3000000; + ca = 0x3000000; + v = D_8018A300[(s16)tbli]; + + p[0] = ((u32)(p - 5) & mk1) | ca; + x1 = (x - 3) & 0xFFFF; + x2 = (x + 5) & 0xFFFF; + p[1] = cc1; + yr = yt; + __asm__("" : "=r"(yr) : "0"(yr)); + yt = (s16)yt; + cs = (yt + 1) << 16; + w = cs | x1; + __asm__("" : "=r"(w) : "0"(w)); + cl = ((v << 6) | 0x4016) << 16; + p[2] = w; + p[3] = cl | 0x1800; + p += 5; + p[0] = ((u32)(p - 5) & mk1) | ca; + p[1] = cc1; + p[2] = cs | x2; + p[3] = cl | 0x1808; + p += 5; + + q = p; + yt = yt << 16; + { + for (; i < 3; i++) { + nn = (bb << 16) >> 28; + n = nn; + if (((nn != 0) || (i == two)) || (i == ff)) { + flag = 0; + } + q[0] = ((u32)(q - 5) & 0xFFFFFF) | c3; + q[2] = (yt | (t & 0xFFFF)) | flag; + col = 0x74808080; + q[1] = col; + q[3] = cl | (((n * 8) + 8) | 0x4000); + q += 5; + t += 8; + bb <<= 4; + } + } + p = q; + + __asm__("" : "=r"(v) : "0"(v)); + g = ((u32)(p - 5) & 0xFFFFFF) | 0x3000000; + __asm__ __volatile__(""); + cs = yr << 16; + p[0] = g; + w3 = cs | ((xr + 0x2A) & 0xFFFF); + __asm__ __volatile__(""); + cl = ((v << 6) | 0x4016) << 16; + uv = ((s16)tr) << 4; + p[2] = w3; + tt = uv | 0x1000; + p[1] = col; + p[3] = cl | tt; + p += 5; + p[0] = ((u32)(p - 5) & 0xFFFFFF) | 0x3000000; + __asm__ __volatile__(""); + cs = cs | ((xr + 0x32) & 0xFFFF); + uv = uv | 0x1008; + cl = cl | uv; + p[1] = col; + p[2] = cs; + p[3] = cl; + p += 5; + __asm__("" :: "r"(tr), "r"(xr)); + return p; +} diff --git a/.run/wave22/func_80177DA8.c b/.run/wave22/func_80177DA8.c new file mode 100644 index 0000000000..27e6cb889f --- /dev/null +++ b/.run/wave22/func_80177DA8.c @@ -0,0 +1,53 @@ +extern u8 D_8018A300[]; + +void func_80177DA8(p, v, idx) +u8 *p; +u32 v; +s16 idx; +{ + u8 *r; + u16 c; + u16 flag; + u32 n; + u8 m; + s16 i; + u8 t; + u32 x; + u32 uv; + u32 w1; + u32 w2; + + flag = 0x100; + i = 0; + t = D_8018A300[idx]; + c = (t << 6) | 0x4016; + *(u16 *)(p + 0xE) = c; + p += 0x14; + *(u16 *)(p + 0xE) = c; + p += 0x14; + r = p; + do { + n = (v << 16) >> 28; + m = n; + if (n != 0 || i == 2 || i == 0xFF) { + flag = 0; + } + v <<= 4; + i++; + *(s16 *)(r + 0xA) = flag | (*(s16 *)(r + 0xA) & ~0x100); + *(u8 *)(r + 0xC) = m * 8 + 8; + r += 0x14; + } while (i < 3); + *(u16 *)(p + 0xE) = c; + p += 0x14; + *(u16 *)(p + 0xE) = c; + p += 0x14; + *(u16 *)(p + 0xE) = c; + p += 0x14; + x = ((t << 6) | 0x4016) << 16; + uv = idx << 4; + w1 = uv | 0x1000; + *(u32 *)(p + 0xC) = x | w1; + w2 = uv | 0x1008; + *(u32 *)(p + 0x20) = x | w2; +} diff --git a/.run/wave22_targets.json b/.run/wave22_targets.json new file mode 100644 index 0000000000..b4e6fb5687 --- /dev/null +++ b/.run/wave22_targets.json @@ -0,0 +1,351 @@ +[ + { + "fn": "func_80163534", + "kind": "fresh", + "stub": "src/ov_SC01_077/ov_SC01_077_jr_8015C32C.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_8015C32C", + "ghidra_c": ".run/ghidra_c/func_80163534.c", + "o0": false, + "sig_hints": { + "callees": { + "func_80163664": "void func_80163664(s32, u16, u16, s32, s32, s32, s32, s32, s32, s32, u16, s32, s32);" + }, + "data": { + "D_80115100": "extern s32 D_80115100;", + "D_80115200": "extern s32 D_80115200;", + "D_80126B18": "extern u16 D_80126B18[];" + } + }, + "n_callees": 1, + "n_data": 15 + }, + { + "fn": "func_8012E014", + "kind": "fresh", + "stub": "src/ov_SC01_077/ov_SC01_077_jr_8012ACE0.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_8012ACE0", + "ghidra_c": ".run/ghidra_c/func_8012E014.c", + "o0": false, + "sig_hints": { + "callees": { + "func_80049CAC": "extern void func_80049CAC(s32, s32);", + "func_8012F0BC": "void func_8012F0BC(s32*, s32*, s32*);", + "func_8012F1A4": "void func_8012F1A4(s32*, s32, s32*);" + }, + "data": { + "D_80126B5C": "extern u8 D_80126B5C;" + } + }, + "n_callees": 3, + "n_data": 3 + }, + { + "fn": "func_80132F40", + "kind": "fresh", + "stub": "src/ov_SC01_077/ov_SC01_077_jr_8012ACE0.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_8012ACE0", + "ghidra_c": ".run/ghidra_c/func_80132F40.c", + "o0": false, + "sig_hints": { + "callees": { + "func_8012F038": "extern void func_8012F038();", + "func_8012F14C": "extern void func_8012F14C();", + "func_80135888": "extern s32 func_80135888(s32, s32, s32, s32);" + }, + "data": { + "D_80126B5E": "extern u16 D_80126B5E;", + "D_80126B62": "extern u16 D_80126B62;", + "D_80126B66": "extern u16 D_80126B66;" + } + }, + "n_callees": 3, + "n_data": 6 + }, + { + "fn": "func_80171B4C", + "kind": "fresh", + "stub": "src/ov_SC01_077/ov_SC01_077_jr_8016AB6C.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_8016AB6C", + "ghidra_c": ".run/ghidra_c/func_80171B4C.c", + "o0": false, + "sig_hints": { + "callees": { + "func_80146D90": "extern void func_80146D90(s32);", + "ratan2": "extern s32 ratan2(s32, s32);" + }, + "data": { + "D_801151D4": "extern s32 D_801151D4;" + } + }, + "n_callees": 2, + "n_data": 2 + }, + { + "fn": "func_8012E364", + "kind": "fresh", + "stub": "src/ov_SC01_077/ov_SC01_077_jr_8012ACE0.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_8012ACE0", + "ghidra_c": null, + "o0": false, + "sig_hints": { + "callees": {}, + "data": {} + }, + "n_callees": 0, + "n_data": 3 + }, + { + "fn": "func_8013B6A0", + "kind": "fresh", + "stub": "src/ov_SC01_077/ov_SC01_077_o0.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_o0", + "ghidra_c": ".run/ghidra_c/func_8013B6A0.c", + "o0": true, + "sig_hints": { + "callees": {}, + "data": { + "D_801DAA08": "extern E_3B7AC D_801DAA08[];" + } + }, + "n_callees": 0, + "n_data": 5 + }, + { + "fn": "func_80148E54", + "kind": "fresh", + "stub": "src/ov_SC01_077/ov_SC01_077_after.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_after", + "ghidra_c": ".run/ghidra_c/func_80148E54.c", + "o0": false, + "sig_hints": { + "callees": { + "ratan2": "extern s32 ratan2(s32, s32);" + }, + "data": { + "D_801151D4": "extern s32 D_801151D4;" + } + }, + "n_callees": 1, + "n_data": 2 + }, + { + "fn": "func_8013B598", + "kind": "fresh", + "stub": "src/ov_SC01_077/ov_SC01_077_o0.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_o0", + "ghidra_c": ".run/ghidra_c/func_8013B598.c", + "o0": true, + "sig_hints": { + "callees": {}, + "data": { + "D_801DAA08": "extern E_3B7AC D_801DAA08[];" + } + }, + "n_callees": 0, + "n_data": 5 + }, + { + "fn": "func_80133298", + "kind": "fresh", + "stub": "src/ov_SC01_077/ov_SC01_077_jr_8012ACE0.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_8012ACE0", + "ghidra_c": ".run/ghidra_c/func_80133298.c", + "o0": false, + "sig_hints": { + "callees": { + "func_8012B2CC": "void func_8012B2CC(s32);", + "func_8012F0BC": "void func_8012F0BC(s32*, s32*, s32*);", + "func_8012F1A4": "void func_8012F1A4(s32*, s32, s32*);", + "func_8013339C": "void func_8013339C(short*, short*);" + }, + "data": { + "D_80126B5C": "extern u8 D_80126B5C;" + } + }, + "n_callees": 4, + "n_data": 3 + }, + { + "fn": "func_80140D68", + "kind": "fresh", + "stub": "src/ov_SC01_077/ov_SC01_077.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077", + "ghidra_c": ".run/ghidra_c/func_80140D68.c", + "o0": false, + "sig_hints": { + "callees": {}, + "data": { + "D_800B9A02": "extern short D_800B9A02;", + "D_8011516A": "extern Hw4 D_8011516A[];" + } + }, + "n_callees": 0, + "n_data": 3 + }, + { + "fn": "func_80177DA8", + "kind": "fresh", + "stub": "src/ov_SC01_077/ov_SC01_077_jr_801734BC.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_801734BC", + "ghidra_c": ".run/ghidra_c/func_80177DA8.c", + "o0": false, + "sig_hints": { + "callees": {}, + "data": {} + }, + "n_callees": 0, + "n_data": 1 + }, + { + "fn": "func_80138C60", + "kind": "fresh", + "stub": "src/ov_SC01_077/ov_SC01_077_jr_801380E0.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_801380E0", + "ghidra_c": null, + "o0": false, + "sig_hints": { + "callees": { + "func_80139C7C": "void func_80139C7C(u8*);", + "func_8013A8FC": "s32 func_8013A8FC(s32);" + }, + "data": {} + }, + "n_callees": 3, + "n_data": 2 + }, + { + "fn": "func_8014A738", + "kind": "fresh", + "stub": "src/ov_SC01_077/ov_SC01_077_after.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_after", + "ghidra_c": ".run/ghidra_c/func_8014A738.c", + "o0": false, + "sig_hints": { + "callees": { + "func_80029178": "extern s32 func_80029178(s32);" + }, + "data": { + "D_801202A0": "extern u8 D_801202A0[];" + } + }, + "n_callees": 1, + "n_data": 1 + }, + { + "fn": "func_8012A328", + "kind": "fresh", + "stub": "src/ov_SC01_077/ov_SC01_077_a.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_a", + "ghidra_c": ".run/ghidra_c/func_8012A328.c", + "o0": false, + "sig_hints": { + "callees": { + "func_80012F74": "extern s32 func_80012F74(s32, s32, s32, s32);" + }, + "data": { + "D_801152C0": "extern s8 D_801152C0;", + "D_801152C2": "extern s16 D_801152C2;", + "D_80126940": "extern s16 D_80126940;", + "D_80126942": "extern s16 D_80126942;", + "D_80126944": "extern s16 D_80126944;", + "D_80126B58": "extern s32 D_80126B58;", + "D_80127080": "extern s16 D_80127080;" + } + }, + "n_callees": 1, + "n_data": 8 + }, + { + "fn": "func_80176734", + "kind": "redraft", + "stub": "src/ov_SC01_077/ov_SC01_077_jr_801734BC.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_801734BC", + "ghidra_c": ".run/ghidra_c/func_80176734.c", + "o0": false, + "sig_hints": { + "callees": { + "func_800183E0": "extern void func_800183E0(s32);", + "func_801619D0": "extern s32 func_801619D0(void*);", + "func_80161A00": "extern s32 func_80161A00(s32);", + "func_80161A30": "extern s32 func_80161A30(s32);", + "func_80161A60": "extern s32 func_80161A60(s32);", + "func_801775E0": "void func_801775E0(s32, s32);" + }, + "data": { + "D_80078E78": "extern u8 D_80078E78[];", + "D_800B9A13": "extern u8 D_800B9A13;", + "D_800D43D4": "extern u8 D_800D43D4;", + "D_800D4414": "extern u8 D_800D4414;", + "D_800D45D4": "extern u8 D_800D45D4;", + "D_8011F7A8": "extern u8 D_8011F7A8;", + "D_80126B58": "extern s32 D_80126B58;", + "D_80126D20": "extern s16 D_80126D20;", + "D_8018A238": "extern u16 D_8018A238;", + "D_8018A2CC": "extern u8 D_8018A2CC[];", + "D_8018A2E4": "extern s32 D_8018A2E4[];" + } + }, + "n_callees": 6, + "n_data": 17 + }, + { + "fn": "func_80140958", + "kind": "redraft", + "stub": "src/ov_SC01_077/ov_SC01_077.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077", + "ghidra_c": ".run/ghidra_c/func_80140958.c", + "o0": false, + "sig_hints": { + "callees": { + "func_8014168C": "extern s16 func_8014168C(s16);" + }, + "data": { + "D_800B9A02": "extern short D_800B9A02;", + "D_80115110": "extern u16 D_80115110;", + "D_80115140": "extern u8 D_80115140[];", + "D_8011514E": "extern s16 D_8011514E;", + "D_80115158": "extern u8 D_80115158[];", + "D_8011516A": "extern Hw4 D_8011516A[];", + "D_8018798C": "extern Prim4 D_8018798C[];", + "D_80187A80": "extern Prim4 *D_80187A80[];" + } + }, + "n_callees": 2, + "n_data": 15 + }, + { + "fn": "func_80177B5C", + "kind": "redraft", + "stub": "src/ov_SC01_077/ov_SC01_077_jr_801734BC.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_801734BC", + "ghidra_c": ".run/ghidra_c/func_80177B5C.c", + "o0": false, + "sig_hints": { + "callees": {}, + "data": {} + }, + "n_callees": 0, + "n_data": 1 + }, + { + "fn": "func_80135260", + "kind": "redraft", + "stub": "src/ov_SC01_077/ov_SC01_077_jr_8012ACE0.c", + "asm_subdir": "asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_8012ACE0", + "ghidra_c": ".run/ghidra_c/func_80135260.c", + "o0": false, + "sig_hints": { + "callees": { + "func_80134A74": "int func_80134A74(int, s16, s16, int);", + "func_80135480": "s16 func_80135480(void*, s32, s16*, s16*);" + }, + "data": { + "D_801870AC": "extern u8 D_801870AC;", + "D_801870B0": "extern u8 D_801870B0;", + "D_801870B8": "extern u8 D_801870B8;" + } + }, + "n_callees": 4, + "n_data": 7 + } +] \ No newline at end of file diff --git a/docs/family-hseq.md b/docs/family-hseq.md index 372b0811de..3dd97a6a43 100644 --- a/docs/family-hseq.md +++ b/docs/family-hseq.md @@ -2,11 +2,11 @@ > Generated by `tools/family_hseq.py` from the 138 overlay sigs + per-overlay src stubs. Ranked by TEMPLATABLE byte-weight (PURE+IMM members × nins × 4). The byte-gate is the arbiter. -**Fleet (overlays):** 90.1% fn / 83.3% instr / 72.3% distinct-code matched. Unmatched: 34,902 instances / 2,184,164 ins (21,460 distinct classes). +**Fleet (overlays):** 90.5% fn / 85.1% instr / 75.3% distinct-code matched. Unmatched: 33,253 instances / 1,952,353 ins (20,550 distinct classes). -**Tail cross-check (Phase-25 close):** 21,142 tail fns / 964,724 ins → 562 h_seq families ≥2, **149 substantial (nins≥80) / 494,961 ins**. +**Tail cross-check (Phase-25 close):** 19,633 tail fns / 796,456 ins → 557 h_seq families ≥2, **145 substantial (nins≥80) / 335,768 ins**. -**Full frontier (all unmatched by h_seq):** 2665 target families (≥2 members or a matched sibling) + 3771 singletons (Step-D residue). Substantial: **521 families / 1,015,826 templatable ins**, 62 with a matched sibling (zero-crack). Substantial member classes: 5,674 PURE · 40 IMM · 6 STRUCT-excluded. +**Full frontier (all unmatched by h_seq):** 2655 target families (≥2 members or a matched sibling) + 3771 singletons (Step-D residue). Substantial: **515 families / 820,679 templatable ins**, 62 with a matched sibling (zero-crack). Substantial member classes: 4,580 PURE · 40 IMM · 6 STRUCT-excluded. ## Top substantial families (by templatable byte-weight) @@ -15,51 +15,51 @@ |--:|--:|--|--|--|--|--:|:-:|--|--:| | 1 | 371 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x80176734 draft-ov077 | 51,198 | | 2 | 329 | 137 (137/0/0) | 1/137 | per-location | PURE | 1 | Y | 0x8013c414 matched-ov077 | 45,073 | -| 3 | 327 | 137 (137/0/0) | 1/137 | per-location | PURE | 1 | · | 0x80176218 matched-ov077 | 44,799 | -| 4 | 289 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | Y | 0x80135eb0 draft-ov077 | 39,882 | -| 5 | 272 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | Y | 0x8013b83c draft-ov077 | 37,536 | -| 6 | 260 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x80140958 draft-ov077 | 35,880 | -| 7 | 231 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x80175da8 draft-ov077 | 31,878 | -| 8 | 198 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | Y | 0x8013bd74 draft-ov077 | 27,324 | -| 9 | 198 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x801412a8 draft-ov077 | 27,324 | -| 10 | 188 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x80175ab8 draft-ov077 | 25,944 | -| 11 | 183 | 137 (137/0/0) | 1/137 | per-location | PURE | 1 | Y | 0x8014032c matched-ov077 | 25,071 | -| 12 | 165 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x80178004 draft-ov077 | 22,770 | -| 13 | 154 | 137 (137/0/0) | 1/137 | per-location | PURE | 1 | Y | 0x8013c0f8 matched-ov077 | 21,098 | -| 14 | 154 | 136 (136/0/0) | 1/136 | per-location | PURE | 2 | · | 0x80144090 matched-ov077 | 20,944 | -| 15 | 147 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x80177b5c draft-ov077 | 20,286 | -| 16 | 136 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | Y | 0x80135260 draft-ov077 | 18,768 | -| 17 | 137 | 132 (132/0/0) | 1/132 | per-location | PURE | 6 | · | 0x80133ab0 matched-ov077 | 18,084 | -| 18 | 114 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x8015d1b8 draft-ov077 | 15,732 | -| 19 | 110 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x801330e0 draft-ov077 | 15,180 | -| 20 | 947 | 16 (16/0/0) | 6/16 | cross-address | PURE | 0 | Y | 0x8017c974 draft-ov077 | 15,152 | -| 21 | 91 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | Y | 0x801789ac draft-ov077 | 12,558 | -| 22 | 952 | 13 (7/6/0) | 5/13 | cross-address | IMM | 103 | Y | 0x8017bebc matched | 12,376 | -| 23 | 88 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x8016ec0c draft-ov077 | 12,144 | -| 24 | 82 | 137 (137/0/0) | 1/137 | per-location | PURE | 1 | · | 0x8014cf04 matched | 11,234 | -| 25 | 80 | 136 (136/0/0) | 1/136 | per-location | PURE | 2 | · | 0x80143d28 matched-ov077 | 10,880 | -| 26 | 195 | 36 (36/0/0) | 11/36 | scattered | PURE | 0 | · | 0x8017c218 modal | 7,020 | -| 27 | 299 | 20 (20/0/0) | 20/20 | scattered | PURE | 0 | Y | 0x8017fee0 modal | 5,980 | -| 28 | 227 | 26 (26/0/0) | 10/26 | scattered | PURE | 0 | · | 0x8017bfec modal | 5,902 | -| 29 | 236 | 20 (20/0/0) | 20/20 | scattered | PURE | 0 | · | 0x80180520 modal | 4,720 | -| 30 | 158 | 26 (26/0/0) | 10/26 | scattered | PURE | 0 | · | 0x8017c378 modal | 4,108 | -| 31 | 793 | 5 (5/0/0) | 2/5 | cross-address | PURE | 0 | · | 0x8017d174 modal | 3,965 | -| 32 | 246 | 16 (16/0/0) | 6/16 | cross-address | PURE | 0 | · | 0x8017c294 draft-ov077 | 3,936 | -| 33 | 766 | 5 (5/0/0) | 3/5 | cross-address | PURE | 0 | · | 0x8017df84 modal | 3,830 | -| 34 | 328 | 11 (11/0/0) | 11/11 | scattered | PURE | 0 | · | 0x801833f0 modal | 3,608 | -| 35 | 890 | 4 (4/0/0) | 1/4 | per-location | PURE | 134 | Y | 0x80178d40 matched-ov077 | 3,560 | -| 36 | 237 | 15 (15/0/0) | 15/15 | scattered | PURE | 0 | · | 0x801832a8 modal | 3,555 | -| 37 | 253 | 14 (14/0/0) | 10/14 | scattered | PURE | 0 | · | 0x8017c9bc modal | 3,542 | -| 38 | 491 | 7 (7/0/0) | 6/7 | cross-address | PURE | 0 | Y | 0x801863cc modal | 3,437 | -| 39 | 166 | 20 (20/0/0) | 20/20 | scattered | PURE | 0 | · | 0x8017fac0 modal | 3,320 | -| 40 | 92 | 36 (36/0/0) | 11/36 | scattered | PURE | 0 | · | 0x8017c910 modal | 3,312 | -| 41 | 90 | 36 (36/0/0) | 11/36 | scattered | PURE | 0 | · | 0x8017c064 modal | 3,240 | -| 42 | 293 | 11 (11/0/0) | 4/11 | cross-address | PURE | 0 | · | 0x8017c43c modal | 3,223 | -| 43 | 770 | 4 (4/0/0) | 1/4 | per-location | PURE | 134 | · | 0x80144b9c matched-ov077 | 3,080 | -| 44 | 611 | 5 (5/0/0) | 4/5 | cross-address | PURE | 0 | · | 0x80186e24 modal | 3,055 | -| 45 | 493 | 6 (6/0/0) | 1/6 | per-location | PURE | 132 | Y | 0x8015a3c8 matched-ov077 | 2,958 | -| 46 | 263 | 11 (11/0/0) | 11/11 | scattered | PURE | 0 | · | 0x80182fd4 modal | 2,893 | -| 47 | 557 | 5 (5/0/0) | 4/5 | cross-address | PURE | 0 | Y | 0x80186570 modal | 2,785 | -| 48 | 185 | 15 (15/0/0) | 13/15 | scattered | PURE | 0 | · | 0x8018b23c modal | 2,775 | -| 49 | 551 | 5 (5/0/0) | 4/5 | cross-address | PURE | 0 | Y | 0x80189540 modal | 2,755 | -| 50 | 125 | 22 (22/0/0) | 22/22 | scattered | PURE | 0 | Y | 0x80185440 modal | 2,750 | +| 3 | 272 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | Y | 0x8013b83c draft-ov077 | 37,536 | +| 4 | 260 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x80140958 draft-ov077 | 35,880 | +| 5 | 198 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | Y | 0x8013bd74 draft-ov077 | 27,324 | +| 6 | 198 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x801412a8 draft-ov077 | 27,324 | +| 7 | 183 | 137 (137/0/0) | 1/137 | per-location | PURE | 1 | Y | 0x8014032c matched-ov077 | 25,071 | +| 8 | 165 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x80178004 draft-ov077 | 22,770 | +| 9 | 154 | 137 (137/0/0) | 1/137 | per-location | PURE | 1 | Y | 0x8013c0f8 matched-ov077 | 21,098 | +| 10 | 154 | 136 (136/0/0) | 1/136 | per-location | PURE | 2 | · | 0x80144090 matched-ov077 | 20,944 | +| 11 | 147 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x80177b5c draft-ov077 | 20,286 | +| 12 | 136 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | Y | 0x80135260 draft-ov077 | 18,768 | +| 13 | 137 | 132 (132/0/0) | 1/132 | per-location | PURE | 6 | · | 0x80133ab0 matched-ov077 | 18,084 | +| 14 | 947 | 16 (16/0/0) | 6/16 | cross-address | PURE | 0 | Y | 0x8017c974 draft-ov077 | 15,152 | +| 15 | 952 | 13 (7/6/0) | 5/13 | cross-address | IMM | 103 | Y | 0x8017bebc matched | 12,376 | +| 16 | 88 | 138 (138/0/0) | 1/138 | per-location | PURE | 0 | · | 0x8016ec0c draft-ov077 | 12,144 | +| 17 | 80 | 136 (136/0/0) | 1/136 | per-location | PURE | 2 | · | 0x80143d28 matched-ov077 | 10,880 | +| 18 | 195 | 36 (36/0/0) | 11/36 | scattered | PURE | 0 | · | 0x8017c218 modal | 7,020 | +| 19 | 299 | 20 (20/0/0) | 20/20 | scattered | PURE | 0 | Y | 0x8017fee0 modal | 5,980 | +| 20 | 227 | 26 (26/0/0) | 10/26 | scattered | PURE | 0 | · | 0x8017bfec modal | 5,902 | +| 21 | 236 | 20 (20/0/0) | 20/20 | scattered | PURE | 0 | · | 0x80180520 modal | 4,720 | +| 22 | 158 | 26 (26/0/0) | 10/26 | scattered | PURE | 0 | · | 0x8017c378 modal | 4,108 | +| 23 | 793 | 5 (5/0/0) | 2/5 | cross-address | PURE | 0 | · | 0x8017d174 modal | 3,965 | +| 24 | 246 | 16 (16/0/0) | 6/16 | cross-address | PURE | 0 | · | 0x8017c294 draft-ov077 | 3,936 | +| 25 | 766 | 5 (5/0/0) | 3/5 | cross-address | PURE | 0 | · | 0x8017df84 modal | 3,830 | +| 26 | 328 | 11 (11/0/0) | 11/11 | scattered | PURE | 0 | · | 0x801833f0 modal | 3,608 | +| 27 | 890 | 4 (4/0/0) | 1/4 | per-location | PURE | 134 | Y | 0x80178d40 matched-ov077 | 3,560 | +| 28 | 237 | 15 (15/0/0) | 15/15 | scattered | PURE | 0 | · | 0x801832a8 modal | 3,555 | +| 29 | 253 | 14 (14/0/0) | 10/14 | scattered | PURE | 0 | · | 0x8017c9bc modal | 3,542 | +| 30 | 491 | 7 (7/0/0) | 6/7 | cross-address | PURE | 0 | Y | 0x801863cc modal | 3,437 | +| 31 | 166 | 20 (20/0/0) | 20/20 | scattered | PURE | 0 | · | 0x8017fac0 modal | 3,320 | +| 32 | 92 | 36 (36/0/0) | 11/36 | scattered | PURE | 0 | · | 0x8017c910 modal | 3,312 | +| 33 | 90 | 36 (36/0/0) | 11/36 | scattered | PURE | 0 | · | 0x8017c064 modal | 3,240 | +| 34 | 293 | 11 (11/0/0) | 4/11 | cross-address | PURE | 0 | · | 0x8017c43c modal | 3,223 | +| 35 | 770 | 4 (4/0/0) | 1/4 | per-location | PURE | 134 | · | 0x80144b9c matched-ov077 | 3,080 | +| 36 | 611 | 5 (5/0/0) | 4/5 | cross-address | PURE | 0 | · | 0x80186e24 modal | 3,055 | +| 37 | 493 | 6 (6/0/0) | 1/6 | per-location | PURE | 132 | Y | 0x8015a3c8 matched-ov077 | 2,958 | +| 38 | 263 | 11 (11/0/0) | 11/11 | scattered | PURE | 0 | · | 0x80182fd4 modal | 2,893 | +| 39 | 557 | 5 (5/0/0) | 4/5 | cross-address | PURE | 0 | Y | 0x80186570 modal | 2,785 | +| 40 | 185 | 15 (15/0/0) | 13/15 | scattered | PURE | 0 | · | 0x8018b23c modal | 2,775 | +| 41 | 551 | 5 (5/0/0) | 4/5 | cross-address | PURE | 0 | Y | 0x80189540 modal | 2,755 | +| 42 | 125 | 22 (22/0/0) | 22/22 | scattered | PURE | 0 | Y | 0x80185440 modal | 2,750 | +| 43 | 386 | 7 (7/0/0) | 6/7 | cross-address | PURE | 0 | · | 0x80186b78 modal | 2,702 | +| 44 | 83 | 32 (32/0/0) | 30/32 | scattered | PURE | 0 | · | 0x80189c7c modal | 2,656 | +| 45 | 177 | 15 (15/0/0) | 13/15 | scattered | PURE | 0 | · | 0x8018aa98 modal | 2,655 | +| 46 | 513 | 5 (5/0/0) | 4/5 | cross-address | PURE | 0 | Y | 0x801878e8 modal | 2,565 | +| 47 | 96 | 26 (26/0/0) | 10/26 | scattered | PURE | 0 | · | 0x8017c738 modal | 2,496 | +| 48 | 154 | 15 (15/0/0) | 15/15 | scattered | PURE | 0 | · | 0x8018389c modal | 2,310 | +| 49 | 164 | 14 (14/0/0) | 14/14 | scattered | PURE | 0 | · | 0x80190748 modal | 2,296 | +| 50 | 113 | 20 (20/0/0) | 10/20 | scattered | PURE | 13 | · | 0x8017bef8 matched-ov077 | 2,260 |