From e2aee8e274504edb9ec24df590af3fa3c16bdcf6 Mon Sep 17 00:00:00 2001 From: Drew T <50529377+Druthulu@users.noreply.github.com> Date: Sat, 5 Sep 2026 10:34:35 -0600 Subject: [PATCH] =?UTF-8?q?feat(phase-32):=20T3=20(12)+(13)=20=E2=80=94=20?= =?UTF-8?q?main:=20func=5F80015B6C=20(120=20ins)=20+=20func=5F8002FDE8=20(?= =?UTF-8?q?73=20ins)=20BANKED=20byte-identical=20143dbb89=20via=20gate=5Fm?= =?UTF-8?q?ain?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - gate_main .run/P32/t3s3/gate/slate_main2.json --apply: "slate 2 -> 2 compatible, 0 dropped"; clean EXE rebuild (extract + build) -> "BANKED 2 main functions -- 143dbb89f34491258bbc27810d0a12ec8b43a8dd BYTE-IDENTICAL", EXIT 0 (.run/P32/t3s3/gate/gate_main2.log) - func_80015B6C (src/800.c, Opus .run/P32/t3/opus/func_80015B6C.c): both journal walls fell — the "$v0/$v1 swap" was an artefact of the s32 wh[3] frame-slot hack; the corner-copy wall is cse.c canon_reg (make_regs_eqv keeps the FIRST register canonical) -> a barrier on the SOURCE variable `__asm__("" : "=r"(x) : "0"(x))`; four empty volatile fences partition block 2 into five phases at zero cost; the 0xE1000200 trailer lui hoist = reuse the $6-pinned tag variable - func_8002FDE8 (src/800_b_2.c, Opus .run/P32/t3/opus/func_8002FDE8.c): the "regalloc-priority wall" at 35 was the ARRAY spelling of D_800A46D2 — the block-scope SCALAR `extern s16 D_800A46D2;` (as same-TU func_8002FF0C/func_800301C8) gives 35 -> 3; third `return 1` inside the == -1 arm; the §47 live-length slider COMPUTED from -dl -dg (one fence where the constant is live and `data` dead); `one = 1;` and `i4 = idx * 4;` as separate statements - both rtu_match MATCH in the real TU (S82 re-verification); cookbook §500-B carries the closers - main: 12 -> 7 open (5 pinned walls + 2 NEAR) --- src/800.c | 144 +++++++++++++++++++++++++++++++++++++++++++++++++- src/800_b_2.c | 74 +++++++++++++++++++++++++- 2 files changed, 216 insertions(+), 2 deletions(-) diff --git a/src/800.c b/src/800.c index cbe4cb936..176bf4e90 100644 --- a/src/800.c +++ b/src/800.c @@ -3018,7 +3018,149 @@ subtract: return quotient; } -INCLUDE_ASM("asm/nonmatchings/800", func_80015B6C); +/* func_80015B6C — main/800. Allocate one 0x34-byte GPU packet from the frame + * heap (func_80010A08) holding three primitives and link all three, in order, + * into the current double-buffer's ordering table at depth 0xFFF + * (*(u32 *)(D_800A651C[D_800B9A02].a + 0x3FFC)) with the open-coded PSY-Q + * addPrim RMW pair (tag = {addr:24, len:8}), exactly as the matched sibling + * func_80015D4C in this TU does: + * + * +0x00 DR_MODE-style header : tag.len = 1, code word 0xE1000000 + * +0x08 POLY_G4 gradient quad: tag.len = 8, code 0x38, corners + * (x,y) (x+w,y) (x,y+h) (x+w,y+h), top colour c1/c2/c3 on v0+v1, + * bottom colour c4/c5/c6 on v2+v3 + * +0x2C DR_MODE-style trailer: tag.len = 1, code word 0xE1000200 + * + * =================================================================== + * WHY THE ODD-LOOKING SOURCE (all five levers were byte-verified; six earlier + * attempts plateaued at 20..47 without them, see the journal notes) + * =================================================================== + * + * 1. THE CORNER COPIES (`vx`/`vy`) + THE ZERO-BYTE INVALIDATORS. + * The target keeps the ORIGINAL x,y in scratch regs (addu $a0,$s1,$zero / + * addu $v1,$s2,$zero) and then DESTROYS $s1/$s2 with the sums + * (addu $s1,$s1,$t1). That shape needs a source-level copy + an in-place + * `x += w_`. But cse.c's canon_reg rewrites every use of `vx` back to `x` + * while `x` is still unmodified (make_regs_eqv keeps the FIRST reg of the + * quantity canonical, and a copy that dies inside the block can never + * become canonical), which silently turned `sh $a0,0x8` into `sh $s1,0x8` + * and deleted the `vy` copy outright (119 ins). + * Moving the `+=` above the stores does defeat cse, but it also shortens + * w_/h_'s live ranges, and local-alloc's priority + * (log2(n_refs)*n_refs/live_length) then hands w_/h_ the last two + * callee-saved registers and SPILLS TWO COLOURS (+5 ins). The target + * spills w_/h_ to 0x10/0x18(sp) instead. + * THE FIX that satisfies both: keep the `+=` in its natural late place and + * kill the cse equivalence with a zero-byte opaque copy on x and y — + * `__asm__("" : "=r"(x) : "0"(x));`. It emits nothing, re-defines x's + * quantity so canon_reg stops substituting, and leaves the live ranges + * (hence the spill decision) untouched. + * + * 2. THE `__asm__ __volatile__("")` SCHEDULING FENCES. gcc-2.7.2's sched1 + * gives the block-2 compute chain (the corner copies, the two `+=`, the + * packet-tag load, the 0xE1000200 constant) a longer path to the end of + * the block than the twelve leaf colour `sb`s, so it hoists ALL of it in + * front of the colours — and, without the first fence, hoists the trailer's + * `lui 0xE100` all the way to the top of the post-call region. Source + * order alone is INERT against this (measured: colours-first and + * colours-last both give the same schedule). The four empty volatile asms + * partition block 2 into exactly the target's five phases: + * colours | v0/v1 corners | tag load + x += w_ + v1/v2 | y += h_ + v3 + * | the addPrim pair. + * They cost nothing: the load-delay slots the fences would otherwise strand + * are filled from inside each phase (the tag load fills the first reload's + * slot, `and $a2,$a2,$a3` the second's), which is exactly what the target + * does at 0x80015C88 and 0x80015CA0. + * + * 3. `tag2` PINNED TO $6 ($a2) — the load-bearing regalloc lever, the same one + * func_80015D4C's header documents. Left alone, the block-2 tag lands in + * $a0, i.e. in the register the vx copy has just vacated, which drags + * `sh $a0,0x8` and `sh $a0,0x18` in front of the colour block and lets the + * trailer constant float free. With the tag in $a2 the vx copy stays live + * across the colours and the schedule falls into place. + * + * 4. `tag2` IS REUSED FOR THE 0xE1000200 TRAILER WORD, and the assignment sits + * AFTER the packet-tag store. gcc splits a large `li` into lui+ori only + * after reload, so sched2 hoists the `lui` as far as the register lets it: + * reusing $a2 pins it to the slot right after `sw $a2,0x0($v0)` — 0x80015CC8 + * in the target. Declaring a second constant, or assigning it earlier, + * moves the whole and/or chain out of $a2 (measured 5..36 mismatched). + * + * 5. `pmask` (the addPrim address mask) IS A NAMED LOCAL so that the second + * RMW's `and $a0,$v0,$a1` can still be scheduled INSIDE the first RMW + * (0x80015CB8) while fence 4 keeps the trailer's `lui` behind the store. + * Without it the fence would drag that `and` out with everything else. + * + * `mFF` stays a pinned local ($7/$a3): unpinning it re-shuffles the four + * post-call constants (measured 15 mismatched). `one`/`m24`/`otp`'s pins were + * measured INERT and dropped. + * + * @class: MATCH — 120/120 byte-exact under tools/match_one.py AND + * tools/rtu_match.py (whole-TU splice), and tools/reloc_identity.py AGREEs + * (relocs_checked=3): jal func_80010A08, %hi/%lo(D_800B9A02) via lhu (u16, + * per the TU's own declaration at src/800.c:2655), %hi/%lo(D_800A651C) via + * lw (the `.a` member at offset 0, 20-byte stride). + */ + +extern void *func_80010A08(s32 a0); +extern u16 D_800B9A02; + +void func_80015B6C(s32 x, s32 y, s32 w_, s32 h_, u8 c1, u8 c2, u8 c3, u8 c4, u8 c5, u8 c6) +{ + extern struct { s32 a; s32 b[4]; } D_800A651C[]; /* block scope, like func_80016450's sibling at src/800.c:3454 */ + s32 otp; + u8 *p; + register u32 mFF __asm__("$7"); + s32 vx, vy; + u32 pmask; + register u32 tag2 __asm__("$6"); + + otp = D_800A651C[D_800B9A02].a; + p = (u8 *)func_80010A08(0x34); + mFF = 0xFF000000; + + p[3] = 1; + *(u32 *)(p + 4) = 0xE1000000; + *(u32 *)p = (*(u32 *)p & mFF) | (*(u32 *)(otp + 0x3FFC) & 0xFFFFFF); + *(u32 *)(otp + 0x3FFC) = (*(u32 *)(otp + 0x3FFC) & mFF) | ((u32)p & 0xFFFFFF); + + p += 8; + p[3] = 8; + vx = x; + p[7] = 0x38; + __asm__ __volatile__(""); + vy = y; + __asm__("" : "=r"(x) : "0"(x)); + __asm__("" : "=r"(y) : "0"(y)); + p[4] = c1; p[5] = c2; p[6] = c3; + p[0xC] = c1; p[0xD] = c2; p[0xE] = c3; + p[0x14] = c4; p[0x15] = c5; p[0x16] = c6; + p[0x1C] = c4; p[0x1D] = c5; p[0x1E] = c6; + *(s16 *)(p + 8) = vx; + *(s16 *)(p + 0xA) = vy; + __asm__ __volatile__(""); + tag2 = *(u32 *)p; + x += w_; + *(s16 *)(p + 0x10) = x; + *(s16 *)(p + 0x12) = vy; + *(s16 *)(p + 0x18) = vx; + __asm__ __volatile__(""); + y += h_; + *(s16 *)(p + 0x1A) = y; + *(s16 *)(p + 0x20) = x; + *(s16 *)(p + 0x22) = y; + pmask = (u32)p & 0xFFFFFF; + *(u32 *)p = (tag2 & mFF) | (*(u32 *)(otp + 0x3FFC) & 0xFFFFFF); + __asm__ __volatile__(""); + tag2 = 0xE1000200; + *(u32 *)(otp + 0x3FFC) = (*(u32 *)(otp + 0x3FFC) & mFF) | pmask; + + p += 0x24; + p[3] = 1; + *(u32 *)(p + 4) = tag2; + *(u32 *)p = (*(u32 *)p & mFF) | (*(u32 *)(otp + 0x3FFC) & 0xFFFFFF); + *(u32 *)(otp + 0x3FFC) = (*(u32 *)(otp + 0x3FFC) & mFF) | ((u32)p & 0xFFFFFF); +} /* func_80015D4C — allocate an 0x18-byte flat-shaded quad GPU packet, fill its diff --git a/src/800_b_2.c b/src/800_b_2.c index cd03509f8..47a882dc9 100644 --- a/src/800_b_2.c +++ b/src/800_b_2.c @@ -2812,7 +2812,79 @@ void func_8002FDC8(void) { func_80037D74(); } -INCLUDE_ASM("asm/nonmatchings/800_b_2", func_8002FDE8); +extern s32 D_800A469C; +extern s16 D_800A46A0; +extern s16 D_800A46A2; +extern u8 D_800A46B0; +extern s32 D_800652F0[]; + +extern void func_8002ED90(void); +extern void func_800415A8(s32); +extern s16 func_80041A80(s32, s32, s32); +extern s16 func_800419B0(s32); + +/* func_8002FDE8 -- start resource `idx` playing out of the buffer `data`. + * + * Byte-shape notes (each measured against asm/nonmatchings/800_b_2/func_8002FDE8.s): + * - `extern s16 D_800A46D2;` at BLOCK SCOPE, exactly as func_8002FF0C and func_800301C8 do + * below: this TU declares the symbol `extern s16 D_800A46D2[]` further down, and the ARRAY + * spelling makes cse cache the address in a callee-saved register across the func_800419B0 + * call; the target keeps two independent %hi/%lo accesses (the `sh` at 0x2066C and the `lh` + * at 0x206C8), which only the scalar spelling emits. This was the residual that held the + * function at closeness 35 for four prior attempts. + * - the FIRST zero-byte fence (cookbook S194-A, AFTER placement) keeps `idx | 0x4000` at the + * head of the join block, which is what lets reorg steal it into the `bltz` delay slot and + * eager-duplicate it on the taken path (0x20618 / 0x20630). Without it the function is + * 2 instructions short. + * - `one` is a real local, and the fence sits BETWEEN its assignment and the first store: + * the `li` must be the block's first insn while the `sb` sinks below the index computation. + * Writing the literal 1 at all three sites instead costs 10 mismatched. + * - `i4 = idx * 4;` is a separate statement. The `sb D_800A46B0` and the `lw D_800652F0` + * carry a memory dependence, so sched1 orders them by LUID; splitting the index out gives + * the `sll` a lower LUID than the `sb` and reproduces `sll / addu $a0 / sb / lw`. + * - the SECOND fence is the S47 live-length slider. `data` (2 refs / 15 insns, pri 1333) and + * the `1` constant (4 refs / 58 insns, pri 1379) are adjacent in global.c's allocno_compare, + * so the constant allocated first and took $s1. One extra static insn where the constant is + * live and `data` is dead lengthens it to 59 (pri 1355 -> ties/loses) and hands $s1 back to + * `data`, $s2 to the constant, exactly as the target. + * - the third `return 1` is spelled inside the `== -1` arm, not as a trailing fallthrough: + * the other spelling inverts the final `beq` into a `bne` and swaps the two tail blocks. + */ +s32 func_8002FDE8(s32 idx, s32 data) { + /* block scope -- see the note above */ + extern s16 D_800A46D2; + s16 r; + s32 one; + s32 i4; + + func_8002ED90(); + if (D_800A46A2 >= 0) { + func_800415A8(D_800A46A2); + D_800A46A2 = -1; + } + D_800A46A0 = idx | 0x4000; + one = 1; + __asm__ __volatile__(""); + D_800A46B0 = one; + i4 = idx * 4; + D_800A469C = *(s32 *)((u8 *)D_800652F0 + i4); + r = func_80041A80(data, -1, D_800A469C); + D_800A46D2 = r; + D_800A46A2 = r; + __asm__ __volatile__(""); + if (r == -1) { + D_800A46B0 = one; + return 1; + } + if (func_800419B0(r) == -1) { + func_800415A8(D_800A46D2); + D_800A46A2 = -1; + D_800A46B0 = one; + return 1; + } + D_800A46B0 = 0; + return 1; +} /* func_8002FF0C -- start playback of resource `entry` from the buffer `arg`. *