diff --git a/.run/near6/d68_blockscope.c b/.run/near6/d68_blockscope.c new file mode 100644 index 0000000000..e52c8f2a91 --- /dev/null +++ b/.run/near6/d68_blockscope.c @@ -0,0 +1,133 @@ +/* func_80140D68 — SPRT (0x14) primitive builder + PsyQ addPrim() into OT_800D29F8[2]. + * + * @class: schedule + * @status: MATCH 0/65 (match_one, asm-subdir asm/ov_SC01_077/nonmatchings/ov_SC01_077) + * + * --------------------------------------------------------------------------- + * HOW THE "ori pinned at idx 3" WALL FELL — three sched1 levers, all byte-measured + * against cc1 RTL dumps (-dS/-dR = .i.sched/.i.sched2). + * + * The 0xFFFFFF mask is ONE movsi that sched1's own `try_split` (sched.c:4830 -> + * mips.md large_int split) turns into `lui`+`ori` BOTH SETTING THE SAME PSEUDO. + * reg_n_sets becomes 2 (sched.c:4617 update_flow_info), so `birthing_insn_p` + * (sched.c:2498, needs REG_N_SETS==1) never boosts either half. sched1 schedules + * BACKWARD, so an un-boosted priority-1 ALU insn only wins a tick when NOTHING + * boosted and NOTHING on the memory unit is ready — otherwise it drifts to the + * block head. That is the whole residual: the `ori` floated to sched1 output + * position 3, sched2 inherited that as its LUID, and it re-floated to idx 3. + * + * For the target, sched2 must sort LUID(sra a2) < LUID(ori) < LUID(addiu a3), + * i.e. sched1 must EMIT the ori between them. Three things had to be true: + * + * L1 — POINTER STORES, NOT A `Sprt *` STRUCT (fixes `lw $v1,0x10($sp)` @ idx 3). + * `p->tag = ...` through a struct pointer sets MEM_IN_STRUCT_P (`/s`) on the + * store, and `anti_dependence()` then says the `/s` store does NOT alias the + * plain `(mem (sp+16))` incoming-arg load — so the lw became ready 3 ticks + * early and sched1 placed it at pos 6 instead of pos 3. Storing through + * `u32 *out` (no `/s`) restores the anti-dep and the lw lands at idx 3. + * (§30's store-vs-load `/s` flag, used in the *opposite* direction.) + * + * L2 — BITFIELD addPrim, INLINE (keeps the ori off the "empty ready list" hole). + * With literal masks (`out[0] = (out[0] & 0xff000000) | (ot[2] & 0xffffff)`) + * sched1 hits a tick where the ori is the ONLY ready insn and freezes it at + * output pos 32 -> idx 14 (measured: 12 off — that is what every + * `.run/wave22/_a80140D68/*.c` 12-scorer was). The `((PTag*)…)->addr` + * bitfield form (cookbook §31 "BITFIELD STORE = THE MASK-ORDER DECOUPLER") + * keeps that tick occupied so the ori keeps drifting. + * + * L3 — KILL THE BIRTHING BOOST ON THE FIRST `addu` (the actual crack). + * sched1 queues each `lhu` 2 cycles behind its consumer (r3000 load latency, + * mips.md:157), which opens exactly ONE gap tick above them. While + * `q[0]+D_8011516A[idx].x` was a fresh single-set pseudo it was boosted, took + * the tick above the loads, and `dx-0xD` (also boosted) took the gap — so the + * forced emission order was sra, lhu, lhu, dx with no ALU-only tick left + * for the ori anywhere between sra and dx. Writing the sum into a variable + * that is assigned AGAIN (`a = q[0]+D…; a += dx;`) gives it REG_N_SETS==2, + * kills its boost, lets the boosted `dx-0xD` take the higher tick, and leaves + * the gap tick to the ori: sched1 now emits sra, lhu, lhu, ori, dx (ori at + * output pos 20) — exactly the LUID order sched2 needs. 11 -> 0. + * + * NEW IDIOM (for cookbook §S2/§49): an un-boosted 2-insn constant that floats to the + * top of its block cannot be moved by its OWN source position (126 header orderings + + * 150 body permutations measured: no effect at all). It is moved by DELETING A BOOST + * from whatever insn currently owns the load-latency gap above it — + * `x = A + B; x += C;` instead of `x = A + B + C;` is a zero-byte boost-kill that + * hands that gap tick to the floater. + * + * Measured dead ends kept for the record (every number from a real match_one run): + * register-pinned / plain mask locals over 12 positions -> 9 or 12, never 13 + * 126 header orderings / 150 body permutations -> 12 (ori stuck at 14) + * two `u32 *ot` locals / inline OTP / `u32 ot` int form -> 26 / 56 / 12 + * `__asm__("":"=r"(v):"0"(v))` re-tie on the lhu temps -> 32 / 64 (hoists the lhu) + * the same re-tie placed AFTER last use -> deleted as dead, 11 + * reusing one s32 temp for both lhu (S12 fence) -> 30 (loads float up) + * dropping the $8 pin on mhi -> 8 (REGALLOC-PERM $t0/$t1) + * dropping the (u16) casts on the two lhu -> 2 (WIDTH lh != lhu) + * + * Signature (read off the asm): a0 = SPRT out, a1 = s16 *src, a2 = s16 idx (in-callee + * sll16/sra14 => K&R narrow param, cookbook §99), a3 = s32 dx, 0x10($sp) = s16 ofs + * (K&R narrow; ANSI `s16 ofs` yields `lh`+`sll 1` = 2 ins instead of lw+sll16+sra15). + * + * Draft-local shims: these two typedefs already exist VERBATIM in src/shared/engine_types.h + * (Hw4 @833, Env_800D29F8). The guard makes the draft self-contained for match_one (which + * only prepends common.h) while collapsing to nothing once banked into a TU that includes + * the header. */ +#ifndef BFM_ENGINE_TYPES_H +typedef struct { s16 x; s16 y; } Hw4; +typedef struct { u16 f0; s16 f2; } Prim4; +#endif +/* §94 TYPE-CARRY FIX: Env_800D29F8 is NOT in engine_types.h (verified: 0 hits), so it must NOT sit + behind the BFM_ENGINE_TYPES_H guard — in the real TU that guard is DEFINED, the typedef vanished, + and the next line failed to parse. Hw4 stays guarded because it genuinely IS in the shared header + (redefining it would conflict). Kept draft-local per cookbook §100, not lifted. */ + +extern Hw4 D_8011516A[]; +extern short D_800B9A02; + +#define OTP_80140D68 (D_800AE7BC[*(volatile u16 *)&D_800B9A02].ot) + +s32 *func_80140D68(out, src, idx, dx, ofs) + s32 *out; + Prim4 *src; /* conformed to the fleet prototype; used only as (s32)src */ + s16 idx; + s32 dx; + s16 ofs; +{ + /* §94 TYPE-CARRY: this typedef and its extern MUST be BLOCK-scope. extract_unit carries + file-scope externs and #defines into each remapped sibling (Phase-27 _carry_macros) but + NOT file-scope typedefs — so a file-scope Env_800D29F8 is silently dropped from every + sibling and the family sweeps 0/137. Body-local typedefs DO survive (PTag_80140D68 below + is the proof), so it lives here. */ + typedef struct { + u32 *ot; /* 0x00 */ + u32 pad[4]; /* 0x04..0x13 */ + } Env_800D29F8; /* 0x14 stride */ + extern Env_800D29F8 D_800AE7BC[]; + + typedef struct { u32 addr : 24; u32 len : 8; } PTag_80140D68; + + register u32 mhi __asm__("$8"); + s16 *q; + s32 a; + + out[0] = 0x04000000; + *((u8 *)out + 0xC) = 0x70; + *((u8 *)out + 0xD) = 0x10; + mhi = 0x64808080; + out[1] = mhi; + *(u16 *)((u8 *)out + 0xE) = 0x4056; + + dx -= 0xD; + q = (s16 *)(ofs * 2 + (s32)src); + a = (u16)q[0] + (u16)D_8011516A[idx].x; + a += dx; + *(s16 *)((u8 *)out + 0x8) = a; + *(s16 *)((u8 *)out + 0xA) = q[1] - 4; + *(s16 *)((u8 *)out + 0x12) = 0x10; + *(s16 *)((u8 *)out + 0x10) = 0x10; + + ((PTag_80140D68 *)out)->addr = ((PTag_80140D68 *)(OTP_80140D68 + 2))->addr; + ((PTag_80140D68 *)(OTP_80140D68 + 2))->addr = (u32)out; + + return out + 5; +} diff --git a/.run/near6/d68_final.c b/.run/near6/d68_final.c new file mode 100644 index 0000000000..18530755ab --- /dev/null +++ b/.run/near6/d68_final.c @@ -0,0 +1,127 @@ +/* func_80140D68 — SPRT (0x14) primitive builder + PsyQ addPrim() into OT_800D29F8[2]. + * + * @class: schedule + * @status: MATCH 0/65 (match_one, asm-subdir asm/ov_SC01_077/nonmatchings/ov_SC01_077) + * + * --------------------------------------------------------------------------- + * HOW THE "ori pinned at idx 3" WALL FELL — three sched1 levers, all byte-measured + * against cc1 RTL dumps (-dS/-dR = .i.sched/.i.sched2). + * + * The 0xFFFFFF mask is ONE movsi that sched1's own `try_split` (sched.c:4830 -> + * mips.md large_int split) turns into `lui`+`ori` BOTH SETTING THE SAME PSEUDO. + * reg_n_sets becomes 2 (sched.c:4617 update_flow_info), so `birthing_insn_p` + * (sched.c:2498, needs REG_N_SETS==1) never boosts either half. sched1 schedules + * BACKWARD, so an un-boosted priority-1 ALU insn only wins a tick when NOTHING + * boosted and NOTHING on the memory unit is ready — otherwise it drifts to the + * block head. That is the whole residual: the `ori` floated to sched1 output + * position 3, sched2 inherited that as its LUID, and it re-floated to idx 3. + * + * For the target, sched2 must sort LUID(sra a2) < LUID(ori) < LUID(addiu a3), + * i.e. sched1 must EMIT the ori between them. Three things had to be true: + * + * L1 — POINTER STORES, NOT A `Sprt *` STRUCT (fixes `lw $v1,0x10($sp)` @ idx 3). + * `p->tag = ...` through a struct pointer sets MEM_IN_STRUCT_P (`/s`) on the + * store, and `anti_dependence()` then says the `/s` store does NOT alias the + * plain `(mem (sp+16))` incoming-arg load — so the lw became ready 3 ticks + * early and sched1 placed it at pos 6 instead of pos 3. Storing through + * `u32 *out` (no `/s`) restores the anti-dep and the lw lands at idx 3. + * (§30's store-vs-load `/s` flag, used in the *opposite* direction.) + * + * L2 — BITFIELD addPrim, INLINE (keeps the ori off the "empty ready list" hole). + * With literal masks (`out[0] = (out[0] & 0xff000000) | (ot[2] & 0xffffff)`) + * sched1 hits a tick where the ori is the ONLY ready insn and freezes it at + * output pos 32 -> idx 14 (measured: 12 off — that is what every + * `.run/wave22/_a80140D68/*.c` 12-scorer was). The `((PTag*)…)->addr` + * bitfield form (cookbook §31 "BITFIELD STORE = THE MASK-ORDER DECOUPLER") + * keeps that tick occupied so the ori keeps drifting. + * + * L3 — KILL THE BIRTHING BOOST ON THE FIRST `addu` (the actual crack). + * sched1 queues each `lhu` 2 cycles behind its consumer (r3000 load latency, + * mips.md:157), which opens exactly ONE gap tick above them. While + * `q[0]+D_8011516A[idx].x` was a fresh single-set pseudo it was boosted, took + * the tick above the loads, and `dx-0xD` (also boosted) took the gap — so the + * forced emission order was sra, lhu, lhu, dx with no ALU-only tick left + * for the ori anywhere between sra and dx. Writing the sum into a variable + * that is assigned AGAIN (`a = q[0]+D…; a += dx;`) gives it REG_N_SETS==2, + * kills its boost, lets the boosted `dx-0xD` take the higher tick, and leaves + * the gap tick to the ori: sched1 now emits sra, lhu, lhu, ori, dx (ori at + * output pos 20) — exactly the LUID order sched2 needs. 11 -> 0. + * + * NEW IDIOM (for cookbook §S2/§49): an un-boosted 2-insn constant that floats to the + * top of its block cannot be moved by its OWN source position (126 header orderings + + * 150 body permutations measured: no effect at all). It is moved by DELETING A BOOST + * from whatever insn currently owns the load-latency gap above it — + * `x = A + B; x += C;` instead of `x = A + B + C;` is a zero-byte boost-kill that + * hands that gap tick to the floater. + * + * Measured dead ends kept for the record (every number from a real match_one run): + * register-pinned / plain mask locals over 12 positions -> 9 or 12, never 13 + * 126 header orderings / 150 body permutations -> 12 (ori stuck at 14) + * two `u32 *ot` locals / inline OTP / `u32 ot` int form -> 26 / 56 / 12 + * `__asm__("":"=r"(v):"0"(v))` re-tie on the lhu temps -> 32 / 64 (hoists the lhu) + * the same re-tie placed AFTER last use -> deleted as dead, 11 + * reusing one s32 temp for both lhu (S12 fence) -> 30 (loads float up) + * dropping the $8 pin on mhi -> 8 (REGALLOC-PERM $t0/$t1) + * dropping the (u16) casts on the two lhu -> 2 (WIDTH lh != lhu) + * + * Signature (read off the asm): a0 = SPRT out, a1 = s16 *src, a2 = s16 idx (in-callee + * sll16/sra14 => K&R narrow param, cookbook §99), a3 = s32 dx, 0x10($sp) = s16 ofs + * (K&R narrow; ANSI `s16 ofs` yields `lh`+`sll 1` = 2 ins instead of lw+sll16+sra15). + * + * Draft-local shims: these two typedefs already exist VERBATIM in src/shared/engine_types.h + * (Hw4 @833, Env_800D29F8). The guard makes the draft self-contained for match_one (which + * only prepends common.h) while collapsing to nothing once banked into a TU that includes + * the header. */ +#ifndef BFM_ENGINE_TYPES_H +typedef struct { s16 x; s16 y; } Hw4; +typedef struct { u16 f0; s16 f2; } Prim4; +#endif +/* §94 TYPE-CARRY FIX: Env_800D29F8 is NOT in engine_types.h (verified: 0 hits), so it must NOT sit + behind the BFM_ENGINE_TYPES_H guard — in the real TU that guard is DEFINED, the typedef vanished, + and the next line failed to parse. Hw4 stays guarded because it genuinely IS in the shared header + (redefining it would conflict). Kept draft-local per cookbook §100, not lifted. */ +typedef struct { + u32 *ot; /* 0x00 */ + u32 pad[4]; /* 0x04..0x13 */ +} Env_800D29F8; /* 0x14 stride */ + +extern Hw4 D_8011516A[]; +extern short D_800B9A02; +extern Env_800D29F8 D_800AE7BC[]; + +#define OTP_80140D68 (D_800AE7BC[*(volatile u16 *)&D_800B9A02].ot) + +s32 *func_80140D68(out, src, idx, dx, ofs) + s32 *out; + Prim4 *src; /* conformed to the fleet prototype; used only as (s32)src */ + s16 idx; + s32 dx; + s16 ofs; +{ + typedef struct { u32 addr : 24; u32 len : 8; } PTag_80140D68; + + register u32 mhi __asm__("$8"); + s16 *q; + s32 a; + + out[0] = 0x04000000; + *((u8 *)out + 0xC) = 0x70; + *((u8 *)out + 0xD) = 0x10; + mhi = 0x64808080; + out[1] = mhi; + *(u16 *)((u8 *)out + 0xE) = 0x4056; + + dx -= 0xD; + q = (s16 *)(ofs * 2 + (s32)src); + a = (u16)q[0] + (u16)D_8011516A[idx].x; + a += dx; + *(s16 *)((u8 *)out + 0x8) = a; + *(s16 *)((u8 *)out + 0xA) = q[1] - 4; + *(s16 *)((u8 *)out + 0x12) = 0x10; + *(s16 *)((u8 *)out + 0x10) = 0x10; + + ((PTag_80140D68 *)out)->addr = ((PTag_80140D68 *)(OTP_80140D68 + 2))->addr; + ((PTag_80140D68 *)(OTP_80140D68 + 2))->addr = (u32)out; + + return out + 5; +} diff --git a/.run/near6/d68_selfcontained.c b/.run/near6/d68_selfcontained.c new file mode 100644 index 0000000000..954f6f800f --- /dev/null +++ b/.run/near6/d68_selfcontained.c @@ -0,0 +1,132 @@ +/* func_80140D68 — SPRT (0x14) primitive builder + PsyQ addPrim() into OT_800D29F8[2]. + * + * @class: schedule + * @status: MATCH 0/65 (match_one, asm-subdir asm/ov_SC01_077/nonmatchings/ov_SC01_077) + * + * --------------------------------------------------------------------------- + * HOW THE "ori pinned at idx 3" WALL FELL — three sched1 levers, all byte-measured + * against cc1 RTL dumps (-dS/-dR = .i.sched/.i.sched2). + * + * The 0xFFFFFF mask is ONE movsi that sched1's own `try_split` (sched.c:4830 -> + * mips.md large_int split) turns into `lui`+`ori` BOTH SETTING THE SAME PSEUDO. + * reg_n_sets becomes 2 (sched.c:4617 update_flow_info), so `birthing_insn_p` + * (sched.c:2498, needs REG_N_SETS==1) never boosts either half. sched1 schedules + * BACKWARD, so an un-boosted priority-1 ALU insn only wins a tick when NOTHING + * boosted and NOTHING on the memory unit is ready — otherwise it drifts to the + * block head. That is the whole residual: the `ori` floated to sched1 output + * position 3, sched2 inherited that as its LUID, and it re-floated to idx 3. + * + * For the target, sched2 must sort LUID(sra a2) < LUID(ori) < LUID(addiu a3), + * i.e. sched1 must EMIT the ori between them. Three things had to be true: + * + * L1 — POINTER STORES, NOT A `Sprt *` STRUCT (fixes `lw $v1,0x10($sp)` @ idx 3). + * `p->tag = ...` through a struct pointer sets MEM_IN_STRUCT_P (`/s`) on the + * store, and `anti_dependence()` then says the `/s` store does NOT alias the + * plain `(mem (sp+16))` incoming-arg load — so the lw became ready 3 ticks + * early and sched1 placed it at pos 6 instead of pos 3. Storing through + * `u32 *out` (no `/s`) restores the anti-dep and the lw lands at idx 3. + * (§30's store-vs-load `/s` flag, used in the *opposite* direction.) + * + * L2 — BITFIELD addPrim, INLINE (keeps the ori off the "empty ready list" hole). + * With literal masks (`out[0] = (out[0] & 0xff000000) | (ot[2] & 0xffffff)`) + * sched1 hits a tick where the ori is the ONLY ready insn and freezes it at + * output pos 32 -> idx 14 (measured: 12 off — that is what every + * `.run/wave22/_a80140D68/*.c` 12-scorer was). The `((PTag*)…)->addr` + * bitfield form (cookbook §31 "BITFIELD STORE = THE MASK-ORDER DECOUPLER") + * keeps that tick occupied so the ori keeps drifting. + * + * L3 — KILL THE BIRTHING BOOST ON THE FIRST `addu` (the actual crack). + * sched1 queues each `lhu` 2 cycles behind its consumer (r3000 load latency, + * mips.md:157), which opens exactly ONE gap tick above them. While + * `q[0]+D_8011516A[idx].x` was a fresh single-set pseudo it was boosted, took + * the tick above the loads, and `dx-0xD` (also boosted) took the gap — so the + * forced emission order was sra, lhu, lhu, dx with no ALU-only tick left + * for the ori anywhere between sra and dx. Writing the sum into a variable + * that is assigned AGAIN (`a = q[0]+D…; a += dx;`) gives it REG_N_SETS==2, + * kills its boost, lets the boosted `dx-0xD` take the higher tick, and leaves + * the gap tick to the ori: sched1 now emits sra, lhu, lhu, ori, dx (ori at + * output pos 20) — exactly the LUID order sched2 needs. 11 -> 0. + * + * NEW IDIOM (for cookbook §S2/§49): an un-boosted 2-insn constant that floats to the + * top of its block cannot be moved by its OWN source position (126 header orderings + + * 150 body permutations measured: no effect at all). It is moved by DELETING A BOOST + * from whatever insn currently owns the load-latency gap above it — + * `x = A + B; x += C;` instead of `x = A + B + C;` is a zero-byte boost-kill that + * hands that gap tick to the floater. + * + * Measured dead ends kept for the record (every number from a real match_one run): + * register-pinned / plain mask locals over 12 positions -> 9 or 12, never 13 + * 126 header orderings / 150 body permutations -> 12 (ori stuck at 14) + * two `u32 *ot` locals / inline OTP / `u32 ot` int form -> 26 / 56 / 12 + * `__asm__("":"=r"(v):"0"(v))` re-tie on the lhu temps -> 32 / 64 (hoists the lhu) + * the same re-tie placed AFTER last use -> deleted as dead, 11 + * reusing one s32 temp for both lhu (S12 fence) -> 30 (loads float up) + * dropping the $8 pin on mhi -> 8 (REGALLOC-PERM $t0/$t1) + * dropping the (u16) casts on the two lhu -> 2 (WIDTH lh != lhu) + * + * Signature (read off the asm): a0 = SPRT out, a1 = s16 *src, a2 = s16 idx (in-callee + * sll16/sra14 => K&R narrow param, cookbook §99), a3 = s32 dx, 0x10($sp) = s16 ofs + * (K&R narrow; ANSI `s16 ofs` yields `lh`+`sll 1` = 2 ins instead of lw+sll16+sra15). + * + * Draft-local shims: these two typedefs already exist VERBATIM in src/shared/engine_types.h + * (Hw4 @833, Env_800D29F8). The guard makes the draft self-contained for match_one (which + * only prepends common.h) while collapsing to nothing once banked into a TU that includes + * the header. */ +#ifndef BFM_ENGINE_TYPES_H +typedef struct { s16 x; s16 y; } Hw4; +typedef struct { u16 f0; s16 f2; } Prim4; +#endif +/* §94 TYPE-CARRY FIX: Env_800D29F8 is NOT in engine_types.h (verified: 0 hits), so it must NOT sit + behind the BFM_ENGINE_TYPES_H guard — in the real TU that guard is DEFINED, the typedef vanished, + and the next line failed to parse. Hw4 stays guarded because it genuinely IS in the shared header + (redefining it would conflict). Kept draft-local per cookbook §100, not lifted. */ + +extern Hw4 D_8011516A[]; +extern short D_800B9A02; + + +s32 *func_80140D68(out, src, idx, dx, ofs) + s32 *out; + Prim4 *src; /* conformed to the fleet prototype; used only as (s32)src */ + s16 idx; + s32 dx; + s16 ofs; +{ + /* §94 TYPE-CARRY: this typedef and its extern MUST be BLOCK-scope. extract_unit carries + file-scope externs and #defines into each remapped sibling (Phase-27 _carry_macros) but + NOT file-scope typedefs — so a file-scope Env_800D29F8 is silently dropped from every + sibling and the family sweeps 0/137. Body-local typedefs DO survive (PTag_80140D68 below + is the proof), so it lives here. */ + typedef struct { + u32 *ot; /* 0x00 */ + u32 pad[4]; /* 0x04..0x13 */ + } Env_800D29F8; /* 0x14 stride */ + extern Env_800D29F8 D_800AE7BC[]; + + typedef struct { u32 addr : 24; u32 len : 8; } PTag_80140D68; + + register u32 mhi __asm__("$8"); + s16 *q; + s32 a; + + out[0] = 0x04000000; + *((u8 *)out + 0xC) = 0x70; + *((u8 *)out + 0xD) = 0x10; + mhi = 0x64808080; + out[1] = mhi; + *(u16 *)((u8 *)out + 0xE) = 0x4056; + + dx -= 0xD; + q = (s16 *)(ofs * 2 + (s32)src); + a = (u16)q[0] + (u16)D_8011516A[idx].x; + a += dx; + *(s16 *)((u8 *)out + 0x8) = a; + *(s16 *)((u8 *)out + 0xA) = q[1] - 4; + *(s16 *)((u8 *)out + 0x12) = 0x10; + *(s16 *)((u8 *)out + 0x10) = 0x10; + + ((PTag_80140D68 *)out)->addr = ((PTag_80140D68 *)((D_800AE7BC[*(volatile u16 *)&D_800B9A02].ot) + 2))->addr; + ((PTag_80140D68 *)((D_800AE7BC[*(volatile u16 *)&D_800B9A02].ot) + 2))->addr = (u32)out; + + return out + 5; +} diff --git a/.run/near6/d68_sigfix.c b/.run/near6/d68_sigfix.c new file mode 100644 index 0000000000..d0ac78219c --- /dev/null +++ b/.run/near6/d68_sigfix.c @@ -0,0 +1,126 @@ +/* func_80140D68 — SPRT (0x14) primitive builder + PsyQ addPrim() into OT_800D29F8[2]. + * + * @class: schedule + * @status: MATCH 0/65 (match_one, asm-subdir asm/ov_SC01_077/nonmatchings/ov_SC01_077) + * + * --------------------------------------------------------------------------- + * HOW THE "ori pinned at idx 3" WALL FELL — three sched1 levers, all byte-measured + * against cc1 RTL dumps (-dS/-dR = .i.sched/.i.sched2). + * + * The 0xFFFFFF mask is ONE movsi that sched1's own `try_split` (sched.c:4830 -> + * mips.md large_int split) turns into `lui`+`ori` BOTH SETTING THE SAME PSEUDO. + * reg_n_sets becomes 2 (sched.c:4617 update_flow_info), so `birthing_insn_p` + * (sched.c:2498, needs REG_N_SETS==1) never boosts either half. sched1 schedules + * BACKWARD, so an un-boosted priority-1 ALU insn only wins a tick when NOTHING + * boosted and NOTHING on the memory unit is ready — otherwise it drifts to the + * block head. That is the whole residual: the `ori` floated to sched1 output + * position 3, sched2 inherited that as its LUID, and it re-floated to idx 3. + * + * For the target, sched2 must sort LUID(sra a2) < LUID(ori) < LUID(addiu a3), + * i.e. sched1 must EMIT the ori between them. Three things had to be true: + * + * L1 — POINTER STORES, NOT A `Sprt *` STRUCT (fixes `lw $v1,0x10($sp)` @ idx 3). + * `p->tag = ...` through a struct pointer sets MEM_IN_STRUCT_P (`/s`) on the + * store, and `anti_dependence()` then says the `/s` store does NOT alias the + * plain `(mem (sp+16))` incoming-arg load — so the lw became ready 3 ticks + * early and sched1 placed it at pos 6 instead of pos 3. Storing through + * `u32 *out` (no `/s`) restores the anti-dep and the lw lands at idx 3. + * (§30's store-vs-load `/s` flag, used in the *opposite* direction.) + * + * L2 — BITFIELD addPrim, INLINE (keeps the ori off the "empty ready list" hole). + * With literal masks (`out[0] = (out[0] & 0xff000000) | (ot[2] & 0xffffff)`) + * sched1 hits a tick where the ori is the ONLY ready insn and freezes it at + * output pos 32 -> idx 14 (measured: 12 off — that is what every + * `.run/wave22/_a80140D68/*.c` 12-scorer was). The `((PTag*)…)->addr` + * bitfield form (cookbook §31 "BITFIELD STORE = THE MASK-ORDER DECOUPLER") + * keeps that tick occupied so the ori keeps drifting. + * + * L3 — KILL THE BIRTHING BOOST ON THE FIRST `addu` (the actual crack). + * sched1 queues each `lhu` 2 cycles behind its consumer (r3000 load latency, + * mips.md:157), which opens exactly ONE gap tick above them. While + * `q[0]+D_8011516A[idx].x` was a fresh single-set pseudo it was boosted, took + * the tick above the loads, and `dx-0xD` (also boosted) took the gap — so the + * forced emission order was sra, lhu, lhu, dx with no ALU-only tick left + * for the ori anywhere between sra and dx. Writing the sum into a variable + * that is assigned AGAIN (`a = q[0]+D…; a += dx;`) gives it REG_N_SETS==2, + * kills its boost, lets the boosted `dx-0xD` take the higher tick, and leaves + * the gap tick to the ori: sched1 now emits sra, lhu, lhu, ori, dx (ori at + * output pos 20) — exactly the LUID order sched2 needs. 11 -> 0. + * + * NEW IDIOM (for cookbook §S2/§49): an un-boosted 2-insn constant that floats to the + * top of its block cannot be moved by its OWN source position (126 header orderings + + * 150 body permutations measured: no effect at all). It is moved by DELETING A BOOST + * from whatever insn currently owns the load-latency gap above it — + * `x = A + B; x += C;` instead of `x = A + B + C;` is a zero-byte boost-kill that + * hands that gap tick to the floater. + * + * Measured dead ends kept for the record (every number from a real match_one run): + * register-pinned / plain mask locals over 12 positions -> 9 or 12, never 13 + * 126 header orderings / 150 body permutations -> 12 (ori stuck at 14) + * two `u32 *ot` locals / inline OTP / `u32 ot` int form -> 26 / 56 / 12 + * `__asm__("":"=r"(v):"0"(v))` re-tie on the lhu temps -> 32 / 64 (hoists the lhu) + * the same re-tie placed AFTER last use -> deleted as dead, 11 + * reusing one s32 temp for both lhu (S12 fence) -> 30 (loads float up) + * dropping the $8 pin on mhi -> 8 (REGALLOC-PERM $t0/$t1) + * dropping the (u16) casts on the two lhu -> 2 (WIDTH lh != lhu) + * + * Signature (read off the asm): a0 = SPRT out, a1 = s16 *src, a2 = s16 idx (in-callee + * sll16/sra14 => K&R narrow param, cookbook §99), a3 = s32 dx, 0x10($sp) = s16 ofs + * (K&R narrow; ANSI `s16 ofs` yields `lh`+`sll 1` = 2 ins instead of lw+sll16+sra15). + * + * Draft-local shims: these two typedefs already exist VERBATIM in src/shared/engine_types.h + * (Hw4 @833, Env_800D29F8). The guard makes the draft self-contained for match_one (which + * only prepends common.h) while collapsing to nothing once banked into a TU that includes + * the header. */ +#ifndef BFM_ENGINE_TYPES_H +typedef struct { s16 x; s16 y; } Hw4; +#endif +/* §94 TYPE-CARRY FIX: Env_800D29F8 is NOT in engine_types.h (verified: 0 hits), so it must NOT sit + behind the BFM_ENGINE_TYPES_H guard — in the real TU that guard is DEFINED, the typedef vanished, + and the next line failed to parse. Hw4 stays guarded because it genuinely IS in the shared header + (redefining it would conflict). Kept draft-local per cookbook §100, not lifted. */ +typedef struct { + u32 *ot; /* 0x00 */ + u32 pad[4]; /* 0x04..0x13 */ +} Env_800D29F8; /* 0x14 stride */ + +extern Hw4 D_8011516A[]; +extern short D_800B9A02; +extern Env_800D29F8 D_800AE7BC[]; + +#define OTP_80140D68 (D_800AE7BC[*(volatile u16 *)&D_800B9A02].ot) + +s32 *func_80140D68(out, src, idx, dx, ofs) + s32 *out; + s16 *src; + s16 idx; + s32 dx; + s16 ofs; +{ + typedef struct { u32 addr : 24; u32 len : 8; } PTag_80140D68; + + register u32 mhi __asm__("$8"); + s16 *q; + s32 a; + + out[0] = 0x04000000; + *((u8 *)out + 0xC) = 0x70; + *((u8 *)out + 0xD) = 0x10; + mhi = 0x64808080; + out[1] = mhi; + *(u16 *)((u8 *)out + 0xE) = 0x4056; + + dx -= 0xD; + q = (s16 *)(ofs * 2 + (s32)src); + a = (u16)q[0] + (u16)D_8011516A[idx].x; + a += dx; + *(s16 *)((u8 *)out + 0x8) = a; + *(s16 *)((u8 *)out + 0xA) = q[1] - 4; + *(s16 *)((u8 *)out + 0x12) = 0x10; + *(s16 *)((u8 *)out + 0x10) = 0x10; + + ((PTag_80140D68 *)out)->addr = ((PTag_80140D68 *)(OTP_80140D68 + 2))->addr; + ((PTag_80140D68 *)(OTP_80140D68 + 2))->addr = (u32)out; + + return out + 5; +} diff --git a/.run/near6/d68_typefix.c b/.run/near6/d68_typefix.c new file mode 100644 index 0000000000..5a223699aa --- /dev/null +++ b/.run/near6/d68_typefix.c @@ -0,0 +1,126 @@ +/* func_80140D68 — SPRT (0x14) primitive builder + PsyQ addPrim() into OT_800D29F8[2]. + * + * @class: schedule + * @status: MATCH 0/65 (match_one, asm-subdir asm/ov_SC01_077/nonmatchings/ov_SC01_077) + * + * --------------------------------------------------------------------------- + * HOW THE "ori pinned at idx 3" WALL FELL — three sched1 levers, all byte-measured + * against cc1 RTL dumps (-dS/-dR = .i.sched/.i.sched2). + * + * The 0xFFFFFF mask is ONE movsi that sched1's own `try_split` (sched.c:4830 -> + * mips.md large_int split) turns into `lui`+`ori` BOTH SETTING THE SAME PSEUDO. + * reg_n_sets becomes 2 (sched.c:4617 update_flow_info), so `birthing_insn_p` + * (sched.c:2498, needs REG_N_SETS==1) never boosts either half. sched1 schedules + * BACKWARD, so an un-boosted priority-1 ALU insn only wins a tick when NOTHING + * boosted and NOTHING on the memory unit is ready — otherwise it drifts to the + * block head. That is the whole residual: the `ori` floated to sched1 output + * position 3, sched2 inherited that as its LUID, and it re-floated to idx 3. + * + * For the target, sched2 must sort LUID(sra a2) < LUID(ori) < LUID(addiu a3), + * i.e. sched1 must EMIT the ori between them. Three things had to be true: + * + * L1 — POINTER STORES, NOT A `Sprt *` STRUCT (fixes `lw $v1,0x10($sp)` @ idx 3). + * `p->tag = ...` through a struct pointer sets MEM_IN_STRUCT_P (`/s`) on the + * store, and `anti_dependence()` then says the `/s` store does NOT alias the + * plain `(mem (sp+16))` incoming-arg load — so the lw became ready 3 ticks + * early and sched1 placed it at pos 6 instead of pos 3. Storing through + * `u32 *out` (no `/s`) restores the anti-dep and the lw lands at idx 3. + * (§30's store-vs-load `/s` flag, used in the *opposite* direction.) + * + * L2 — BITFIELD addPrim, INLINE (keeps the ori off the "empty ready list" hole). + * With literal masks (`out[0] = (out[0] & 0xff000000) | (ot[2] & 0xffffff)`) + * sched1 hits a tick where the ori is the ONLY ready insn and freezes it at + * output pos 32 -> idx 14 (measured: 12 off — that is what every + * `.run/wave22/_a80140D68/*.c` 12-scorer was). The `((PTag*)…)->addr` + * bitfield form (cookbook §31 "BITFIELD STORE = THE MASK-ORDER DECOUPLER") + * keeps that tick occupied so the ori keeps drifting. + * + * L3 — KILL THE BIRTHING BOOST ON THE FIRST `addu` (the actual crack). + * sched1 queues each `lhu` 2 cycles behind its consumer (r3000 load latency, + * mips.md:157), which opens exactly ONE gap tick above them. While + * `q[0]+D_8011516A[idx].x` was a fresh single-set pseudo it was boosted, took + * the tick above the loads, and `dx-0xD` (also boosted) took the gap — so the + * forced emission order was sra, lhu, lhu, dx with no ALU-only tick left + * for the ori anywhere between sra and dx. Writing the sum into a variable + * that is assigned AGAIN (`a = q[0]+D…; a += dx;`) gives it REG_N_SETS==2, + * kills its boost, lets the boosted `dx-0xD` take the higher tick, and leaves + * the gap tick to the ori: sched1 now emits sra, lhu, lhu, ori, dx (ori at + * output pos 20) — exactly the LUID order sched2 needs. 11 -> 0. + * + * NEW IDIOM (for cookbook §S2/§49): an un-boosted 2-insn constant that floats to the + * top of its block cannot be moved by its OWN source position (126 header orderings + + * 150 body permutations measured: no effect at all). It is moved by DELETING A BOOST + * from whatever insn currently owns the load-latency gap above it — + * `x = A + B; x += C;` instead of `x = A + B + C;` is a zero-byte boost-kill that + * hands that gap tick to the floater. + * + * Measured dead ends kept for the record (every number from a real match_one run): + * register-pinned / plain mask locals over 12 positions -> 9 or 12, never 13 + * 126 header orderings / 150 body permutations -> 12 (ori stuck at 14) + * two `u32 *ot` locals / inline OTP / `u32 ot` int form -> 26 / 56 / 12 + * `__asm__("":"=r"(v):"0"(v))` re-tie on the lhu temps -> 32 / 64 (hoists the lhu) + * the same re-tie placed AFTER last use -> deleted as dead, 11 + * reusing one s32 temp for both lhu (S12 fence) -> 30 (loads float up) + * dropping the $8 pin on mhi -> 8 (REGALLOC-PERM $t0/$t1) + * dropping the (u16) casts on the two lhu -> 2 (WIDTH lh != lhu) + * + * Signature (read off the asm): a0 = SPRT out, a1 = s16 *src, a2 = s16 idx (in-callee + * sll16/sra14 => K&R narrow param, cookbook §99), a3 = s32 dx, 0x10($sp) = s16 ofs + * (K&R narrow; ANSI `s16 ofs` yields `lh`+`sll 1` = 2 ins instead of lw+sll16+sra15). + * + * Draft-local shims: these two typedefs already exist VERBATIM in src/shared/engine_types.h + * (Hw4 @833, Env_800D29F8). The guard makes the draft self-contained for match_one (which + * only prepends common.h) while collapsing to nothing once banked into a TU that includes + * the header. */ +#ifndef BFM_ENGINE_TYPES_H +typedef struct { s16 x; s16 y; } Hw4; +#endif +/* §94 TYPE-CARRY FIX: Env_800D29F8 is NOT in engine_types.h (verified: 0 hits), so it must NOT sit + behind the BFM_ENGINE_TYPES_H guard — in the real TU that guard is DEFINED, the typedef vanished, + and the next line failed to parse. Hw4 stays guarded because it genuinely IS in the shared header + (redefining it would conflict). Kept draft-local per cookbook §100, not lifted. */ +typedef struct { + u32 *ot; /* 0x00 */ + u32 pad[4]; /* 0x04..0x13 */ +} Env_800D29F8; /* 0x14 stride */ + +extern Hw4 D_8011516A[]; +extern short D_800B9A02; +extern Env_800D29F8 D_800AE7BC[]; + +#define OTP_80140D68 (D_800AE7BC[*(volatile u16 *)&D_800B9A02].ot) + +u32 *func_80140D68(out, src, idx, dx, ofs) + u32 *out; + s16 *src; + s16 idx; + s32 dx; + s16 ofs; +{ + typedef struct { u32 addr : 24; u32 len : 8; } PTag_80140D68; + + register u32 mhi __asm__("$8"); + s16 *q; + s32 a; + + out[0] = 0x04000000; + *((u8 *)out + 0xC) = 0x70; + *((u8 *)out + 0xD) = 0x10; + mhi = 0x64808080; + out[1] = mhi; + *(u16 *)((u8 *)out + 0xE) = 0x4056; + + dx -= 0xD; + q = (s16 *)(ofs * 2 + (s32)src); + a = (u16)q[0] + (u16)D_8011516A[idx].x; + a += dx; + *(s16 *)((u8 *)out + 0x8) = a; + *(s16 *)((u8 *)out + 0xA) = q[1] - 4; + *(s16 *)((u8 *)out + 0x12) = 0x10; + *(s16 *)((u8 *)out + 0x10) = 0x10; + + ((PTag_80140D68 *)out)->addr = ((PTag_80140D68 *)(OTP_80140D68 + 2))->addr; + ((PTag_80140D68 *)(OTP_80140D68 + 2))->addr = (u32)out; + + return out + 5; +} diff --git a/.run/near6/g734_a.c b/.run/near6/g734_a.c new file mode 100644 index 0000000000..a775150e72 --- /dev/null +++ b/.run/near6/g734_a.c @@ -0,0 +1,366 @@ +/* func_80176734 (371 ins, ov_SC01_077_jr_801734BC, reach-138 core) — wave23 pass + * + * SCORE (match_one, asm-subdir asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_801734BC): + * wave22 seed .run/wave22/func_80176734.c : 366 ins, closeness 217 (LENGTH-DRIFT -5) + * Phase-27 fable seed .run/giants/*.fable.c : 370 ins, closeness 111 (LENGTH-DRIFT -1) + * THIS FILE : 371 ins, closeness 13, 13 real (masked, aligned) + * sibling .run/near6/wave23/x2.c : 371 ins, closeness 14, 11 real <- fewest REAL diffs + * Length now EXACT (371/371); all callee-saved assignments, the frame (0x40), region 1, + * regions 4/5/6, the switch, the whole D_80126CE0 / adjust / tail regions are byte-exact. + * + * FOUR LEVERS APPLIED THIS PASS (each byte-measured, all pin-free / generic-constraint): + * + * 1. NEW IDIOM — "the LOG_LINK next-use-only combine block" (the missing instruction). + * gcc-2.7.2 always expands `x = (a != b)` as `xor t; sne temp; move x,temp` + * (expmed.c:3978 `preserve_subexpressions_p()` is TRUE at -O2, so emit_store_flag always + * writes a fresh subtarget then copies). combine then merges the sne+copy whenever the + * temp dies at the copy — which is why every draft was 1 insn SHORT here. + * flow.c:2085-2091 builds a LOG_LINK only from a SET to the *NEXT FOLLOWING USE* of that + * reg, and only inside the same basic block. So one extra USE of the temp placed BETWEEN + * the sne and the copy moves the link off the copy: combine never sees the pair, both + * instructions survive, and NOTHING is emitted: + * tA = (st1->unk47 != b); + * __asm__("" :: "r"(tA)); <-- steals the LOG_LINK + * flag = tA; + * Crucially the anchor must go BEFORE the copy: placed after it (the Phase-27 fable v22 + * attempt) it also blocks reorg from stealing the copy into the `j` delay slot (+1 nop, + * net 0). Measured: closeness 111 -> 26. A basic-block split (`goto`/label between the + * two) also defeats combine — proven at 372 ins — but costs the arm's extra `j`. + * + * 2. cse qty-head steering (`make_regs_eqv`, cse.c:826) for the `unk47 == 0x80` re-test. + * After `bne tB,w` falls through, record_jump_equiv merges the two regs; the class head + * (and hence the register the NEXT test reads) becomes `w` iff w's regno_last_uid is later. + * Routing the load through the function-wide scratch tB AND giving tB a later last mention + * (`tB = w; st1->unk47 = tB;` in the last adjust arm) keeps tB the head: 26 -> 23. + * + * 3. Head-init placement. `st1` MUST stay a decl-initializer: moved into the body it loses + * the update_equiv_regs live-length doubling (local-alloc.c:1064), its allocno priority + * (global.c:594) explodes and the whole callee-saved bank rotates (frame 0x40 -> 0x48, + * +1 insn — measured on 12 permutations). Moving only `st2`/`ext` into the body, after the + * index chain and the `self = arg0` copy, reproduces the target's save/init interleave: + * 23 -> 15. + * + * 4. RC-15 density anchor on an explicit region-2 base pointer + * (`s32 base = (((s32)self<<16)>>14) + (s32)g; __asm__("" :: "r"(base));`) — the extra ref + * lifts the base allocno over the store temp, landing it in $a0 and making both + * `lw $a1,0x28($a0)` reloads exact: 15 -> 13 (real 13 -> 11 in the x2 spelling). + * + * RESIDUAL (13 positions here / 11 in x2.c), three clusters, all allocation/schedule ties: + * a) entry: `sw s1 / addiu s1,s3,0x48` scheduled one slot group off (sched2 LUID tie that + * is COUPLED to lever 3 — st1's LUID is pinned by the priority constraint). 4 positions. + * b) region 2: `d` (= st2->unk48, a cross-block => GLOBAL allocno) and the store temp swap + * $v1<->$a1. local-alloc gives the block-local store temp $v1 before global.c ever sees + * `d`; this is the named "local-vs-global allocation tie" (regalloc.md §H). 5 positions. + * c) region 3: `e` in $a0 vs $a1 (same tie). 2 positions. + * In p3 (this file) the region-1 `q` also takes $a2 instead of $a0 — that one is fixed in + * x2.c (put `self = arg0` before `q`), at the cost of the entry order; the two are coupled + * through when the `move s6,a0` frees $a0 at sched1 time. + * + * PIN-FREE / x138-safe: three generic-constraint identity/anchor asms (tB re-opaque for the + * cse.c:7511 fall-through delete, the tA LOG_LINK steal, the base density anchor) and the + * fable-era `fl` tail anchor. No `register __asm__("$N")` pins anywhere. + * + * What it does: per-track BGM/SFX state tick. g=&D_8011F7A8 (sound globals), st1/st2 = two + * SndSt state blocks inside it (+0x48/+0xE0), ext=&D_80078E78 (the engine-side SndSt). + * Region 1: master volume fade write to trk[arg0] (+4/+0x40). Region 2/3: unk48 state machine + * (fade-step table D_8018A2D8, program change via func_800183E0). Region 4: unk2E pan/tempo + * mirror + unk49 table. Region 5: unk1E 0x8000 flag toggle. Region 6: D_800B9A13 mode-change + * detect (changed). Switch: ext->unk48 in {3,4,5,6} -> fade checks -> flag = 0xFF/0xBA. Then + * volume ramp toward w (D_80126CE0 override or ext->unk47) with -3/-8 decay steps, and the + * func_801775E0(trk+0x64, ...) tail. + */ +#include "common.h" + +typedef struct Trk { + u8 pad00[4]; + u8 unk4; /* 0x04 */ + u8 pad05[8]; + u8 unkD; /* 0x0D */ + u8 pad0E[0x12]; + u16 unk20; /* 0x20 */ + u8 pad22[0x10]; + s16 unk32; /* 0x32 */ + u8 pad34[0xC]; + u8 unk40; /* 0x40 */ + u8 pad41[8]; + u8 unk49; /* 0x49 */ + u8 pad4A[0x1A]; + u8 unk64; /* 0x64 */ +} Trk; + +typedef struct SndSt { + u8 pad00[0x1E]; + s16 unk1E; /* 0x1E */ + u8 pad20[0xE]; + s16 unk2E; /* 0x2E */ + u8 pad30[0x17]; + u8 unk47; /* 0x47 */ + u8 unk48; /* 0x48 */ + u8 pad49[2]; + u8 unk4B; /* 0x4B */ + u8 pad4C[0x4C]; /* size 0x98 */ +} SndSt; + +typedef struct SndGlob { + u8 pad00[7]; + u8 unk7; /* 0x07 */ + u8 unk8; /* 0x08 */ + u8 pad09[9]; + u16 unk12; /* 0x12 */ + u8 pad14[0x14]; + Trk *trk[8]; /* 0x28 */ + SndSt st1; /* 0x48 */ + SndSt st2; /* 0xE0 */ +} SndGlob; + +extern SndGlob D_8011F7A8; +extern SndSt D_80078E78; + +extern s16 D_801152BA; +extern u8 D_8011F7B0; +extern u8 D_80115214; +extern u8 D_800B9A13; +extern u16 D_8018A238; +extern u8 D_8018A2D8[]; +extern u16 D_8018A22A[]; +extern void *D_8018A2E4[]; +extern u8 D_8018A2CC[]; +extern s16 D_80126D20; +extern s16 D_80126CE0; +extern s32 D_80126B58; +extern u8 D_800D45D4[]; +extern u8 D_800D43D4[]; +extern u8 D_800D4414[]; + +extern void func_800183E0(void *); +extern s32 func_801619D0(void *); +extern s32 func_80161A00(void *); +extern s32 func_80161A30(void *); +extern s32 func_80161A60(void *); +extern void func_801775E0(u8 *, s16); + +void func_80176734(arg0) +s16 arg0; +{ + SndGlob *g = &D_8011F7A8; + SndSt *st1 = &g->st1; + SndSt *st2; + SndSt *ext; + s32 flag; /* s0 */ + s32 changed; /* s5 */ + s32 tA; /* v0 scratch, reused */ + s32 tB; /* v1 scratch, reused */ + s32 w; /* a1 */ + u8 h; + s16 self; + s8 pad[4]; + + { + Trk *e; + Trk *q; + self = arg0; /* K8/RC-4 lifetime shaping: arg0's LAST use moved ABOVE q's + birth, so $a0 is dead when q is born and K3 first-fit gives + q $a0 instead of $a2 (target idx24 `addiu $a0,$a1,0x3C`). */ + e = g->trk[arg0]; + q = (Trk *)((u8 *)e + 0x3C); + st2 = &g->st2; + ext = &D_80078E78; + if (D_801152BA != 0) { + u8 b = D_8011F7B0; + u8 v; + if (b < 0x80U) { + v = b - 0x80; + } else { + v = ~b - 0x80; + } + q->unk4 = v; + e->unk4 = v; + g->unk8 = g->unk8 + D_80115214; + } else { + e->unk40 = 0x80; + e->unk4 = 0x80; + } + } + + if (st2->unk48 != 0) { + s32 base = (((s32)self << 16) >> 14) + (s32)g; + __asm__("" :: "r"(base)); + (*(Trk **)(base + 0x28))->unkD = D_8018A2D8[st2->unk48]; + if (st2->unk48 >= 4) { + u8 c = st1->unk48; + Trk *e = *(Trk **)(base + 0x28); + if (c & 0x80) { + e->unk20 = D_8018A238; + func_800183E0(D_800D45D4); + } else if (c != 0) { + e->unk20 = D_8018A22A[c]; + func_800183E0(D_8018A2E4[st1->unk48]); + } + { + u8 d = st2->unk48; + if (d == 5) { + if (st1->unk48 == 0) { + st2->unk48 = 0; + } else { + st2->unk48 = d + 1; + } + } else if (d == 0xA) { + st2->unk48 = 0; + } else { + st2->unk48 = d + 1; + } + } + } else { + st2->unk48++; + } + } else { + if (st1->unk48 != ext->unk48) { + if (st1->unk48 == 0 && ext->unk48 != 0) { + st2->unk48 = 5; + } else { + st2->unk48 = 0; + } + st1->unk48 = ext->unk48; + { + Trk *e = g->trk[self]; + u8 t = st2->unk48; + st2->unk48 = t + 1; + e->unkD = D_8018A2D8[t]; + } + } + } + + if (st1->unk2E == ext->unk2E) { + if (st2->unk2E != 0) { + st2->unk2E = 0; + goto upd49; + } + } else { + st1->unk2E = ext->unk2E; + st2->unk2E = 1; +upd49: + tB = (u16)st1->unk2E; + tB = tB << 16; + { + Trk *e = g->trk[self]; + if (tB != 0) { + e->unk49 = D_8018A2CC[tB >> 20]; + } else { + e->unk49 = 0xA0; + } + } + } + + { + s32 f1 = st1->unk1E & 0x8000; + if (f1 != (ext->unk1E & 0x8000)) { + if (f1 != 0) { + st1->unk1E = 0; + func_800183E0(D_800D43D4); + } else { + st1->unk1E = -0x8000; + func_800183E0(D_800D4414); + } + } + } + + { + u8 m = D_800B9A13; + if (m != 3) { + tA = (g->unk7 != m); + changed = tA; + if (tA != 0) { + g->unk7 = m; + } + } else { + changed = 0; + } + } + + flag = 0; + switch (ext->unk48) { + case 3: + if (func_801619D0(&D_80126B58) != 0) flag = 0xFF; + break; + case 4: + if (func_80161A00(&D_80126B58) != 0) flag = 0xFF; + break; + case 5: + if (func_80161A30(&D_80126B58) != 0) flag = 0xFF; + break; + case 6: + if (func_80161A60(&D_80126B58) != 0) flag = 0xBA; + break; + } + + tA = flag; + if (tA != 0) { + st2->unk47 = 1; + st1->unk4B = ext->unk48 | 0xF0; + st1->unk47 = (D_80126D20 << 7) / tA; + } else { + if (st1->unk4B >= 0xF0) { + st1->unk4B = 0; + } + w = *(u16 *)&D_80126CE0; + if (D_80126CE0 != 0) { + u8 b = w; + st1->unk4B = b; + tA = (st1->unk47 != b); + __asm__("" :: "r"(tA)); + flag = tA; + } else { + w = ext->unk47; + flag = 0; + tB = st1->unk47; + if (tB != w || tB == 0x80) { + flag = 1; + } + if (ext->unk47 != 0 && st1->unk4B != 0) { + st1->unk4B = 0; + st1->unk47 = ext->unk47; + } + } + + tB = changed; + if (flag != 0) goto adjust; + if (tB != 0) goto adjust; + if (st2->unk47 == 0) goto posttail; + __asm__("" : "=r"(tB) : "0"(tB)); + if (tB == 0) goto clear; +adjust: + tB = st1->unk47; + if (tB < (s16)w) { + st1->unk47 = w; + } else { + if ((s16)w != 0) { + tA = tB - 3; + st1->unk47 = tA; + } else { + tA = tB - 8; + st1->unk47 = tA; + } + if (st1->unk47 == 0 || st1->unk47 >= 0x81) { + st1->unk47 = 0; + st1->unk4B = 0; + } else if ((s32)st1->unk47 < (s16)w) { + tB = w; + st1->unk47 = tB; + } + } + st2->unk47 = 1; + goto posttail; +clear: + st2->unk47 = 0; +posttail:; + } + + { + s32 fl = (g->unk7 != 0) << 8; + s32 k = fl + 5; + __asm__("" :: "r"(fl)); + g->trk[self]->unk32 = g->unk12 + k; + fl += 9; + func_801775E0(&g->trk[self]->unk64, g->unk12 + fl); + } +} diff --git a/.run/near6/g734_b.c b/.run/near6/g734_b.c new file mode 100644 index 0000000000..84e5594089 --- /dev/null +++ b/.run/near6/g734_b.c @@ -0,0 +1,367 @@ +/* func_80176734 (371 ins, ov_SC01_077_jr_801734BC, reach-138 core) — wave23 pass + * + * SCORE (match_one, asm-subdir asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_801734BC): + * wave22 seed .run/wave22/func_80176734.c : 366 ins, closeness 217 (LENGTH-DRIFT -5) + * Phase-27 fable seed .run/giants/*.fable.c : 370 ins, closeness 111 (LENGTH-DRIFT -1) + * THIS FILE : 371 ins, closeness 13, 13 real (masked, aligned) + * sibling .run/near6/wave23/x2.c : 371 ins, closeness 14, 11 real <- fewest REAL diffs + * Length now EXACT (371/371); all callee-saved assignments, the frame (0x40), region 1, + * regions 4/5/6, the switch, the whole D_80126CE0 / adjust / tail regions are byte-exact. + * + * FOUR LEVERS APPLIED THIS PASS (each byte-measured, all pin-free / generic-constraint): + * + * 1. NEW IDIOM — "the LOG_LINK next-use-only combine block" (the missing instruction). + * gcc-2.7.2 always expands `x = (a != b)` as `xor t; sne temp; move x,temp` + * (expmed.c:3978 `preserve_subexpressions_p()` is TRUE at -O2, so emit_store_flag always + * writes a fresh subtarget then copies). combine then merges the sne+copy whenever the + * temp dies at the copy — which is why every draft was 1 insn SHORT here. + * flow.c:2085-2091 builds a LOG_LINK only from a SET to the *NEXT FOLLOWING USE* of that + * reg, and only inside the same basic block. So one extra USE of the temp placed BETWEEN + * the sne and the copy moves the link off the copy: combine never sees the pair, both + * instructions survive, and NOTHING is emitted: + * tA = (st1->unk47 != b); + * __asm__("" :: "r"(tA)); <-- steals the LOG_LINK + * flag = tA; + * Crucially the anchor must go BEFORE the copy: placed after it (the Phase-27 fable v22 + * attempt) it also blocks reorg from stealing the copy into the `j` delay slot (+1 nop, + * net 0). Measured: closeness 111 -> 26. A basic-block split (`goto`/label between the + * two) also defeats combine — proven at 372 ins — but costs the arm's extra `j`. + * + * 2. cse qty-head steering (`make_regs_eqv`, cse.c:826) for the `unk47 == 0x80` re-test. + * After `bne tB,w` falls through, record_jump_equiv merges the two regs; the class head + * (and hence the register the NEXT test reads) becomes `w` iff w's regno_last_uid is later. + * Routing the load through the function-wide scratch tB AND giving tB a later last mention + * (`tB = w; st1->unk47 = tB;` in the last adjust arm) keeps tB the head: 26 -> 23. + * + * 3. Head-init placement. `st1` MUST stay a decl-initializer: moved into the body it loses + * the update_equiv_regs live-length doubling (local-alloc.c:1064), its allocno priority + * (global.c:594) explodes and the whole callee-saved bank rotates (frame 0x40 -> 0x48, + * +1 insn — measured on 12 permutations). Moving only `st2`/`ext` into the body, after the + * index chain and the `self = arg0` copy, reproduces the target's save/init interleave: + * 23 -> 15. + * + * 4. RC-15 density anchor on an explicit region-2 base pointer + * (`s32 base = (((s32)self<<16)>>14) + (s32)g; __asm__("" :: "r"(base));`) — the extra ref + * lifts the base allocno over the store temp, landing it in $a0 and making both + * `lw $a1,0x28($a0)` reloads exact: 15 -> 13 (real 13 -> 11 in the x2 spelling). + * + * RESIDUAL (13 positions here / 11 in x2.c), three clusters, all allocation/schedule ties: + * a) entry: `sw s1 / addiu s1,s3,0x48` scheduled one slot group off (sched2 LUID tie that + * is COUPLED to lever 3 — st1's LUID is pinned by the priority constraint). 4 positions. + * b) region 2: `d` (= st2->unk48, a cross-block => GLOBAL allocno) and the store temp swap + * $v1<->$a1. local-alloc gives the block-local store temp $v1 before global.c ever sees + * `d`; this is the named "local-vs-global allocation tie" (regalloc.md §H). 5 positions. + * c) region 3: `e` in $a0 vs $a1 (same tie). 2 positions. + * In p3 (this file) the region-1 `q` also takes $a2 instead of $a0 — that one is fixed in + * x2.c (put `self = arg0` before `q`), at the cost of the entry order; the two are coupled + * through when the `move s6,a0` frees $a0 at sched1 time. + * + * PIN-FREE / x138-safe: three generic-constraint identity/anchor asms (tB re-opaque for the + * cse.c:7511 fall-through delete, the tA LOG_LINK steal, the base density anchor) and the + * fable-era `fl` tail anchor. No `register __asm__("$N")` pins anywhere. + * + * What it does: per-track BGM/SFX state tick. g=&D_8011F7A8 (sound globals), st1/st2 = two + * SndSt state blocks inside it (+0x48/+0xE0), ext=&D_80078E78 (the engine-side SndSt). + * Region 1: master volume fade write to trk[arg0] (+4/+0x40). Region 2/3: unk48 state machine + * (fade-step table D_8018A2D8, program change via func_800183E0). Region 4: unk2E pan/tempo + * mirror + unk49 table. Region 5: unk1E 0x8000 flag toggle. Region 6: D_800B9A13 mode-change + * detect (changed). Switch: ext->unk48 in {3,4,5,6} -> fade checks -> flag = 0xFF/0xBA. Then + * volume ramp toward w (D_80126CE0 override or ext->unk47) with -3/-8 decay steps, and the + * func_801775E0(trk+0x64, ...) tail. + */ +#include "common.h" + +typedef struct Trk { + u8 pad00[4]; + u8 unk4; /* 0x04 */ + u8 pad05[8]; + u8 unkD; /* 0x0D */ + u8 pad0E[0x12]; + u16 unk20; /* 0x20 */ + u8 pad22[0x10]; + s16 unk32; /* 0x32 */ + u8 pad34[0xC]; + u8 unk40; /* 0x40 */ + u8 pad41[8]; + u8 unk49; /* 0x49 */ + u8 pad4A[0x1A]; + u8 unk64; /* 0x64 */ +} Trk; + +typedef struct SndSt { + u8 pad00[0x1E]; + s16 unk1E; /* 0x1E */ + u8 pad20[0xE]; + s16 unk2E; /* 0x2E */ + u8 pad30[0x17]; + u8 unk47; /* 0x47 */ + u8 unk48; /* 0x48 */ + u8 pad49[2]; + u8 unk4B; /* 0x4B */ + u8 pad4C[0x4C]; /* size 0x98 */ +} SndSt; + +typedef struct SndGlob { + u8 pad00[7]; + u8 unk7; /* 0x07 */ + u8 unk8; /* 0x08 */ + u8 pad09[9]; + u16 unk12; /* 0x12 */ + u8 pad14[0x14]; + Trk *trk[8]; /* 0x28 */ + SndSt st1; /* 0x48 */ + SndSt st2; /* 0xE0 */ +} SndGlob; + +extern SndGlob D_8011F7A8; +extern SndSt D_80078E78; + +extern s16 D_801152BA; +extern u8 D_8011F7B0; +extern u8 D_80115214; +extern u8 D_800B9A13; +extern u16 D_8018A238; +extern u8 D_8018A2D8[]; +extern u16 D_8018A22A[]; +extern void *D_8018A2E4[]; +extern u8 D_8018A2CC[]; +extern s16 D_80126D20; +extern s16 D_80126CE0; +extern s32 D_80126B58; +extern u8 D_800D45D4[]; +extern u8 D_800D43D4[]; +extern u8 D_800D4414[]; + +extern void func_800183E0(void *); +extern s32 func_801619D0(void *); +extern s32 func_80161A00(void *); +extern s32 func_80161A30(void *); +extern s32 func_80161A60(void *); +extern void func_801775E0(u8 *, s16); + +void func_80176734(arg0) +s16 arg0; +{ + SndGlob *g = &D_8011F7A8; + SndSt *st1 = &g->st1; + SndSt *st2; + SndSt *ext; + s32 flag; /* s0 */ + s32 changed; /* s5 */ + s32 tA; /* v0 scratch, reused */ + s32 tB; /* v1 scratch, reused */ + s32 w; /* a1 */ + u8 h; + s16 self; + s8 pad[4]; + + { + Trk *e; + Trk *q; + e = g->trk[arg0]; + self = arg0; /* K8/RC-4: arg0 dies HERE, above q's birth, so $a0 is free + when q is born (K3 first-fit) -> q lands in $a0 not $a2. + Placed AFTER e's index computation to match the target's + emission order (idx4-6 sll/sra/addu, then s6=arg0). */ + q = (Trk *)((u8 *)e + 0x3C); + st2 = &g->st2; + ext = &D_80078E78; + if (D_801152BA != 0) { + u8 b = D_8011F7B0; + u8 v; + if (b < 0x80U) { + v = b - 0x80; + } else { + v = ~b - 0x80; + } + q->unk4 = v; + e->unk4 = v; + g->unk8 = g->unk8 + D_80115214; + } else { + e->unk40 = 0x80; + e->unk4 = 0x80; + } + } + + if (st2->unk48 != 0) { + s32 base = (((s32)self << 16) >> 14) + (s32)g; + __asm__("" :: "r"(base)); + (*(Trk **)(base + 0x28))->unkD = D_8018A2D8[st2->unk48]; + if (st2->unk48 >= 4) { + u8 c = st1->unk48; + Trk *e = *(Trk **)(base + 0x28); + if (c & 0x80) { + e->unk20 = D_8018A238; + func_800183E0(D_800D45D4); + } else if (c != 0) { + e->unk20 = D_8018A22A[c]; + func_800183E0(D_8018A2E4[st1->unk48]); + } + { + u8 d = st2->unk48; + if (d == 5) { + if (st1->unk48 == 0) { + st2->unk48 = 0; + } else { + st2->unk48 = d + 1; + } + } else if (d == 0xA) { + st2->unk48 = 0; + } else { + st2->unk48 = d + 1; + } + } + } else { + st2->unk48++; + } + } else { + if (st1->unk48 != ext->unk48) { + if (st1->unk48 == 0 && ext->unk48 != 0) { + st2->unk48 = 5; + } else { + st2->unk48 = 0; + } + st1->unk48 = ext->unk48; + { + Trk *e = g->trk[self]; + u8 t = st2->unk48; + st2->unk48 = t + 1; + e->unkD = D_8018A2D8[t]; + } + } + } + + if (st1->unk2E == ext->unk2E) { + if (st2->unk2E != 0) { + st2->unk2E = 0; + goto upd49; + } + } else { + st1->unk2E = ext->unk2E; + st2->unk2E = 1; +upd49: + tB = (u16)st1->unk2E; + tB = tB << 16; + { + Trk *e = g->trk[self]; + if (tB != 0) { + e->unk49 = D_8018A2CC[tB >> 20]; + } else { + e->unk49 = 0xA0; + } + } + } + + { + s32 f1 = st1->unk1E & 0x8000; + if (f1 != (ext->unk1E & 0x8000)) { + if (f1 != 0) { + st1->unk1E = 0; + func_800183E0(D_800D43D4); + } else { + st1->unk1E = -0x8000; + func_800183E0(D_800D4414); + } + } + } + + { + u8 m = D_800B9A13; + if (m != 3) { + tA = (g->unk7 != m); + changed = tA; + if (tA != 0) { + g->unk7 = m; + } + } else { + changed = 0; + } + } + + flag = 0; + switch (ext->unk48) { + case 3: + if (func_801619D0(&D_80126B58) != 0) flag = 0xFF; + break; + case 4: + if (func_80161A00(&D_80126B58) != 0) flag = 0xFF; + break; + case 5: + if (func_80161A30(&D_80126B58) != 0) flag = 0xFF; + break; + case 6: + if (func_80161A60(&D_80126B58) != 0) flag = 0xBA; + break; + } + + tA = flag; + if (tA != 0) { + st2->unk47 = 1; + st1->unk4B = ext->unk48 | 0xF0; + st1->unk47 = (D_80126D20 << 7) / tA; + } else { + if (st1->unk4B >= 0xF0) { + st1->unk4B = 0; + } + w = *(u16 *)&D_80126CE0; + if (D_80126CE0 != 0) { + u8 b = w; + st1->unk4B = b; + tA = (st1->unk47 != b); + __asm__("" :: "r"(tA)); + flag = tA; + } else { + w = ext->unk47; + flag = 0; + tB = st1->unk47; + if (tB != w || tB == 0x80) { + flag = 1; + } + if (ext->unk47 != 0 && st1->unk4B != 0) { + st1->unk4B = 0; + st1->unk47 = ext->unk47; + } + } + + tB = changed; + if (flag != 0) goto adjust; + if (tB != 0) goto adjust; + if (st2->unk47 == 0) goto posttail; + __asm__("" : "=r"(tB) : "0"(tB)); + if (tB == 0) goto clear; +adjust: + tB = st1->unk47; + if (tB < (s16)w) { + st1->unk47 = w; + } else { + if ((s16)w != 0) { + tA = tB - 3; + st1->unk47 = tA; + } else { + tA = tB - 8; + st1->unk47 = tA; + } + if (st1->unk47 == 0 || st1->unk47 >= 0x81) { + st1->unk47 = 0; + st1->unk4B = 0; + } else if ((s32)st1->unk47 < (s16)w) { + tB = w; + st1->unk47 = tB; + } + } + st2->unk47 = 1; + goto posttail; +clear: + st2->unk47 = 0; +posttail:; + } + + { + s32 fl = (g->unk7 != 0) << 8; + s32 k = fl + 5; + __asm__("" :: "r"(fl)); + g->trk[self]->unk32 = g->unk12 + k; + fl += 9; + func_801775E0(&g->trk[self]->unk64, g->unk12 + fl); + } +}