diff --git a/.run/giants/c954_alloc.py b/.run/giants/c954_alloc.py new file mode 100644 index 000000000..7a2d370d8 --- /dev/null +++ b/.run/giants/c954_alloc.py @@ -0,0 +1,27 @@ +#!/usr/bin/env python3 +"""Allocno census: rank / refs / live-length / priority / granted hard reg, +for the long-lived (global) allocnos of one -da compile work dir. +usage: c954_alloc.py [minlen]""" +import re, math, sys + +WD = sys.argv[1] +MINLEN = int(sys.argv[2]) if len(sys.argv) > 2 else 150 +R = ['zero', 'at', 'v0', 'v1', 'a0', 'a1', 'a2', 'a3', 't0', 't1', 't2', 't3', 't4', 't5', 't6', 't7', + 's0', 's1', 's2', 's3', 's4', 's5', 's6', 's7', 't8', 't9', 'k0', 'k1', 'gp', 'sp', 'fp', 'ra'] +lreg = open(WD + '/t.i.lreg').read() +greg = open(WD + '/t.i.greg').read() +info = {} +for m in re.finditer(r'Register (\d+) used (\d+) times across (\d+) insns', lreg): + info[int(m.group(1))] = (int(m.group(2)), int(m.group(3))) +disp = {} +d = greg.split(';; Register dispositions:')[1].split('\n\n')[0] +for m in re.finditer(r'(\d+) in (\d+)', d): + disp[int(m.group(1))] = int(m.group(2)) +order = [int(x) for x in greg.split('regs to allocate:')[1].split('\n')[0].split()] +print('%-4s %6s %6s %6s %9s %s' % ('rank', 'pseudo', 'refs', 'len', 'pri', 'reg')) +for k, p in enumerate(order): + r, l = info.get(p, (0, 0)) + if l < MINLEN: + continue + pri = int(math.floor(math.log2(r)) * r / l * 10000) if r > 0 and l else -1 + print('%-4d %6d %6d %6d %9d %s' % (k, p, r, l, pri, R[disp[p]] if p in disp else 'SPILL')) diff --git a/.run/giants/c954_b0.c b/.run/giants/c954_b0.c new file mode 100644 index 000000000..bfc972d46 --- /dev/null +++ b/.run/giants/c954_b0.c @@ -0,0 +1,477 @@ +#include "common.h" +#include "/home/musashi/bfm-decomp/src/shared/engine_types.h" + +#define gte_ldv0(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_ldv3(r0, r1, r2) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 0( %1 );" \ + "lwc2 $3, 4( %1 );" \ + "lwc2 $4, 0( %2 );" \ + "lwc2 $5, 4( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) ) + +#define gte_ldv3c(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 8( %0 );" \ + "lwc2 $3, 12( %0 );" \ + "lwc2 $4, 16( %0 );" \ + "lwc2 $5, 20( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_rtps() __asm__ volatile ("nop;nop;rtps") +#define gte_rtpt() __asm__ volatile ("nop;nop;rtpt") +#define gte_nclip() __asm__ volatile ("nop;nop;nclip") + +#define gte_stsxy(r0) __asm__ volatile ( \ + "swc2 $14, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 0( %1 );" \ + "swc2 $14, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsxy3c(r0) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 4( %0 );" \ + "swc2 $14, 8( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_f3(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 12( %0 );" \ + "swc2 $14, 16( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_ft3(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 16( %0 );" \ + "swc2 $14, 24( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_f4(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 12( %0 );" \ + "swc2 $14, 16( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsz3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $17, 0( %0 );" \ + "swc2 $18, 0( %1 );" \ + "swc2 $19, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsz4(r0, r1, r2, r3) __asm__ volatile ( \ + "swc2 $16, 0( %0 );" \ + "swc2 $17, 0( %1 );" \ + "swc2 $18, 0( %2 );" \ + "swc2 $19, 0( %3 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ), "r"( r3 ) \ + : "memory" ) + +#define gte_stszotz(r0) __asm__ volatile ( \ + "mfc2 $12, $19;" \ + "nop;" \ + "sra $12, $12, 2;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stflg(r0) __asm__ volatile ( \ + "cfc2 $12, $31;" \ + "nop;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stopz(r0) __asm__ volatile ( \ + "swc2 $24, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +void func_8017C954(s32 arg0) +{ + typedef struct { u32 w0, w1, w2; } Prim; + + extern s32 func_800491EC(void); + extern void func_800547D8(s32, MATRIX2 *); + extern void func_80052E38(MATRIX2 *); + extern u8 *D_800A5E60; + extern u8 D_800A6610[]; + extern short D_800B9A02; /* TU-visible spelling (engine_core.h + ov_SC01_000.c col-0); unsigned access forced at use — §8d sub-class (b) */ + + DVECTOR2 tmpxy[4]; + SVECTOR2 box[8]; + SVECTOR2 sxy[8]; + MATRIX2 mtx; + struct { long otz, flag, opz, sz0, sz1, sz2, sz3; } g; + + s32 lim; + s32 nparts; + s32 j; + u32 nprim; + u32 i; + Part *part; + Prim *prim; + u8 *pkt; + u32 ot; + u8 *vtx; + u8 *va, *vb, *vc, *vd; + u32 w, code; + u32 wx, wy, wz; + s32 xa32, xb32, t32; + s32 xmn1, xmx1, xmn2, xmx2; + s32 mnc, mxc; + s16 my, mny, mx, mn; + + lim = func_800491EC() + *(s32 *)(arg0 + 0x64); + func_800547D8(arg0 + 0x10, &mtx); + func_80052E38(&mtx); + + pkt = D_800A5E60; + part = *(Part **)(arg0 + 0xC); + nparts = *(s32 *)(*(s32 *)(arg0 + 8) + 8); + vtx = *(u8 **)(*(s32 *)(arg0 + 8) + 0x10); + ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14]; + + for (j = 0; j < nparts; j++, part++) { + wx = part->xx; + mn = wx; + mx = wx >> 16; + wy = part->yy; + mny = wy; + my = wy >> 16; + wz = part->zz; + box[0].vx = mn; box[0].vy = mny; + box[1].vx = mx; box[1].vy = mny; + box[2].vx = mn; box[2].vy = mny; + box[3].vx = mx; box[3].vy = mny; + box[4].vx = mn; box[4].vy = my; + box[5].vx = mx; box[5].vy = my; + box[6].vx = mn; box[6].vy = my; + box[7].vx = mx; box[7].vy = my; + wy = wz >> 16; + box[0].vz = wz; + box[1].vz = wz; + box[4].vz = wz; + box[5].vz = wz; + box[2].vz = wy; + box[3].vz = wy; + box[6].vz = wy; + box[7].vz = wy; + + gte_ldv3c(&box[0]); + gte_rtpt(); + gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]); + gte_ldv0(&box[3]); + gte_rtps(); + gte_stsxy(&sxy[3]); + gte_ldv3c(&box[4]); + __asm__ volatile (""); /* §45-B live-length slider: +1 static insn splits the 228/230 allocno-priority tie (498/498 -> 497/498) */ + gte_rtpt(); + gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]); + gte_ldv0(&box[7]); + gte_rtps(); + gte_stsxy(&sxy[7]); + gte_stszotz(&g.otz); + + if (lim >= g.otz) { + xa32 = sxy[0].vx; + xb32 = sxy[1].vx; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vx; + xb32 = sxy[5].vx; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0xA0 && (s16)mnc < 0xA1) { + xa32 = sxy[0].vy; + xb32 = sxy[1].vy; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vy; + xb32 = sxy[5].vy; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0x78 && (s16)mnc < 0x79) { + prim = (Prim *)part->prim; + nprim = part->nprim; + for (i = 0; i < nprim; i++, prim++) { + w = prim->w1; + va = vtx + (w & 0xFFFF); + vb = vtx + (w >> 16); + w = prim->w2; + vc = vtx + (w & 0xFFFF); + w = w >> 16; + gte_ldv3(va, vb, vc); + gte_rtpt(); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_nclip(); + code = w & 7; + vd = vtx + (w & 0xFFF8); + gte_stopz(&g.opz); + if (g.opz > 0) { + switch (code) { + case 4: + case 5: + gte_stsxy3_f3(pkt); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (((PolyF3 *)pkt)->x0 > ((PolyF3 *)pkt)->x1) { + mx = ((PolyF3 *)pkt)->x0; + mn = ((PolyF3 *)pkt)->x1; + } else { + mn = ((PolyF3 *)pkt)->x0; + mx = ((PolyF3 *)pkt)->x1; + } + if (((PolyF3 *)pkt)->x2 > mx) mx = ((PolyF3 *)pkt)->x2; + else if (((PolyF3 *)pkt)->x2 < mn) mn = ((PolyF3 *)pkt)->x2; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyF3 *)pkt)->y0 > ((PolyF3 *)pkt)->y1) { + my = ((PolyF3 *)pkt)->y0; + mny = ((PolyF3 *)pkt)->y1; + } else { + mny = ((PolyF3 *)pkt)->y0; + my = ((PolyF3 *)pkt)->y1; + } + if (((PolyF3 *)pkt)->y2 > my) my = ((PolyF3 *)pkt)->y2; + else if (((PolyF3 *)pkt)->y2 < mny) mny = ((PolyF3 *)pkt)->y2; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code != 4) g.opz = za + 0x200; + ((PolyF3 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x4000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x14; + } + } + break; + case 6: + case 7: + gte_stsxy3_ft3(pkt); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (((PolyFT3 *)pkt)->x0 > ((PolyFT3 *)pkt)->x1) { + mx = ((PolyFT3 *)pkt)->x0; + mn = ((PolyFT3 *)pkt)->x1; + } else { + mn = ((PolyFT3 *)pkt)->x0; + mx = ((PolyFT3 *)pkt)->x1; + } + if (((PolyFT3 *)pkt)->x2 > mx) mx = ((PolyFT3 *)pkt)->x2; + else if (((PolyFT3 *)pkt)->x2 < mn) mn = ((PolyFT3 *)pkt)->x2; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT3 *)pkt)->y0 > ((PolyFT3 *)pkt)->y1) { + my = ((PolyFT3 *)pkt)->y0; + mny = ((PolyFT3 *)pkt)->y1; + } else { + mny = ((PolyFT3 *)pkt)->y0; + my = ((PolyFT3 *)pkt)->y1; + } + if (((PolyFT3 *)pkt)->y2 > my) my = ((PolyFT3 *)pkt)->y2; + else if (((PolyFT3 *)pkt)->y2 < mny) mny = ((PolyFT3 *)pkt)->y2; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code == 7) g.opz = za + 0x200; + tp = (u32 *)prim->w0; + ((PolyFT3 *)pkt)->rgbc = tp[0]; + ((PolyFT3 *)pkt)->uvc0 = tp[1]; + ((PolyFT3 *)pkt)->uvp1 = tp[2]; + ((PolyFT3 *)pkt)->uv2 = tp[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x20; + } + } + break; + case 0: + case 1: + gte_stsxy3_f4(pkt); + gte_ldv0(vd); + gte_rtps(); + if (((PolyF4 *)pkt)->x0 > ((PolyF4 *)pkt)->x1) { + mx = ((PolyF4 *)pkt)->x0; + mn = ((PolyF4 *)pkt)->x1; + } else { + mn = ((PolyF4 *)pkt)->x0; + mx = ((PolyF4 *)pkt)->x1; + } + if (((PolyF4 *)pkt)->x2 > mx) mx = ((PolyF4 *)pkt)->x2; + else if (((PolyF4 *)pkt)->x2 < mn) mn = ((PolyF4 *)pkt)->x2; + if (((PolyF4 *)pkt)->y0 > ((PolyF4 *)pkt)->y1) { + my = ((PolyF4 *)pkt)->y0; + mny = ((PolyF4 *)pkt)->y1; + } else { + mny = ((PolyF4 *)pkt)->y0; + my = ((PolyF4 *)pkt)->y1; + } + if (((PolyF4 *)pkt)->y2 > my) my = ((PolyF4 *)pkt)->y2; + else if (((PolyF4 *)pkt)->y2 < mny) mny = ((PolyF4 *)pkt)->y2; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyF4 *)pkt)->x3); + if (((PolyF4 *)pkt)->x3 < mn) mn = ((PolyF4 *)pkt)->x3; + else if (mx < ((PolyF4 *)pkt)->x3) mx = ((PolyF4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyF4 *)pkt)->y3 < mny) mny = ((PolyF4 *)pkt)->y3; + else if (my < ((PolyF4 *)pkt)->y3) my = ((PolyF4 *)pkt)->y3; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + ((PolyF4 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((za >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x5000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x18; + } + } + } + break; + case 2: + case 3: + gte_stsxy3c(&tmpxy[0]); + gte_ldv0(vd); + gte_rtps(); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyFT4 *)pkt)->x3); + if (((PolyFT4 *)pkt)->x3 < mn) mn = ((PolyFT4 *)pkt)->x3; + else if (mx < ((PolyFT4 *)pkt)->x3) mx = ((PolyFT4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT4 *)pkt)->y3 < mny) mny = ((PolyFT4 *)pkt)->y3; + else if (my < ((PolyFT4 *)pkt)->y3) my = ((PolyFT4 *)pkt)->y3; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + u32 uvw; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + if (code == 3) g.opz = za + 0x200; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + tp = (u32 *)prim->w0; + ((PolyFT4 *)pkt)->rgbc = tp[0]; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + } + } + } + break; + } + } + } + } + } + } + } + } + D_800A5E60 = pkt; +} + diff --git a/.run/giants/c954_b1.c b/.run/giants/c954_b1.c new file mode 100644 index 000000000..0c59a50e1 --- /dev/null +++ b/.run/giants/c954_b1.c @@ -0,0 +1,560 @@ +#include "common.h" +#include "/home/musashi/bfm-decomp/src/shared/engine_types.h" + +#define gte_ldv0(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_ldv3(r0, r1, r2) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 0( %1 );" \ + "lwc2 $3, 4( %1 );" \ + "lwc2 $4, 0( %2 );" \ + "lwc2 $5, 4( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) ) + +#define gte_ldv3c(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 8( %0 );" \ + "lwc2 $3, 12( %0 );" \ + "lwc2 $4, 16( %0 );" \ + "lwc2 $5, 20( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_rtps() __asm__ volatile ("nop;nop;rtps") +#define gte_rtpt() __asm__ volatile ("nop;nop;rtpt") +#define gte_nclip() __asm__ volatile ("nop;nop;nclip") + +#define gte_stsxy(r0) __asm__ volatile ( \ + "swc2 $14, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 0( %1 );" \ + "swc2 $14, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsxy3c(r0) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 4( %0 );" \ + "swc2 $14, 8( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_f3(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 12( %0 );" \ + "swc2 $14, 16( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_ft3(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 16( %0 );" \ + "swc2 $14, 24( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_f4(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 12( %0 );" \ + "swc2 $14, 16( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsz3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $17, 0( %0 );" \ + "swc2 $18, 0( %1 );" \ + "swc2 $19, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsz4(r0, r1, r2, r3) __asm__ volatile ( \ + "swc2 $16, 0( %0 );" \ + "swc2 $17, 0( %1 );" \ + "swc2 $18, 0( %2 );" \ + "swc2 $19, 0( %3 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ), "r"( r3 ) \ + : "memory" ) + +#define gte_stszotz(r0) __asm__ volatile ( \ + "mfc2 $12, $19;" \ + "nop;" \ + "sra $12, $12, 2;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stflg(r0) __asm__ volatile ( \ + "cfc2 $12, $31;" \ + "nop;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stopz(r0) __asm__ volatile ( \ + "swc2 $24, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +void func_8017C954(s32 arg0) +{ + typedef struct { u32 w0, w1, w2; } Prim; + + extern s32 func_800491EC(void); + extern void func_800547D8(s32, MATRIX2 *); + extern void func_80052E38(MATRIX2 *); + extern u8 *D_800A5E60; + extern u8 D_800A6610[]; + extern short D_800B9A02; + extern s32 D_801DCCA0; /* TU-visible spelling (engine_core.h + ov_SC01_000.c col-0); unsigned access forced at use — §8d sub-class (b) */ + + DVECTOR2 tmpxy[4]; + SVECTOR2 box[8]; + SVECTOR2 sxy[8]; + SVECTOR2 vv[4]; + MATRIX2 mtx; + struct { long otz, flag, opz, sz0, sz1, sz2, sz3; } g; + + s32 lim; + u32 colA; + u32 colB; + s32 nparts; + s32 j; + u32 nprim; + u32 i; + Part *part; + Prim *prim; + u8 *pkt; + u32 ot; + u8 *vtx; + u8 *va, *vb, *vc, *vd; + u32 w, code; + u32 wx, wy, wz; + s32 xa32, xb32, t32; + s32 xmn1, xmx1, xmn2, xmx2; + s32 mnc, mxc; + s16 my, mny, mx, mn; + s32 d; + u32 e; + + lim = func_800491EC() + *(s32 *)(arg0 + 0x64); + func_800547D8(arg0 + 0x10, &mtx); + func_80052E38(&mtx); + + pkt = D_800A5E60; + ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14]; + d = D_801DCCA0; + colA = (d << 16) | (d << 8) | d; + e = d * 2; + if (e > 0xFF) e = 0xFF; + colB = (e << 16) | (e << 8) | e; + part = *(Part **)(arg0 + 0xC); + nparts = *(s32 *)(*(s32 *)(arg0 + 8) + 8); + vtx = *(u8 **)(*(s32 *)(arg0 + 8) + 0x10); + + for (j = 0; j < nparts; j++, part++) { + wx = part->xx; + mn = wx; + mx = wx >> 16; + wy = part->yy; + mny = wy; + my = wy >> 16; + wz = part->zz; + box[0].vx = mn; box[0].vy = mny; + box[1].vx = mx; box[1].vy = mny; + box[2].vx = mn; box[2].vy = mny; + box[3].vx = mx; box[3].vy = mny; + box[4].vx = mn; box[4].vy = my; + box[5].vx = mx; box[5].vy = my; + box[6].vx = mn; box[6].vy = my; + box[7].vx = mx; box[7].vy = my; + wy = wz >> 16; + box[0].vz = wz; + box[1].vz = wz; + box[4].vz = wz; + box[5].vz = wz; + box[2].vz = wy; + box[3].vz = wy; + box[6].vz = wy; + box[7].vz = wy; + + gte_ldv3c(&box[0]); + gte_rtpt(); + gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]); + gte_ldv0(&box[3]); + gte_rtps(); + gte_stsxy(&sxy[3]); + gte_ldv3c(&box[4]); + __asm__ volatile (""); /* §45-B live-length slider: +1 static insn splits the 228/230 allocno-priority tie (498/498 -> 497/498) */ + gte_rtpt(); + gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]); + gte_ldv0(&box[7]); + gte_rtps(); + gte_stsxy(&sxy[7]); + gte_stszotz(&g.otz); + + if (lim >= g.otz) { + xa32 = sxy[0].vx; + xb32 = sxy[1].vx; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vx; + xb32 = sxy[5].vx; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0xA0 && (s16)mnc < 0xA1) { + xa32 = sxy[0].vy; + xb32 = sxy[1].vy; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vy; + xb32 = sxy[5].vy; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0x78 && (s16)mnc < 0x79) { + prim = (Prim *)part->prim; + nprim = part->nprim; + for (i = 0; i < nprim; i++, prim++) { + w = prim->w1; + va = vtx + (w & 0xFFFF); + vb = vtx + (w >> 16); + w = prim->w2; + vc = vtx + (w & 0xFFFF); + w = w >> 16; + gte_ldv3(va, vb, vc); + gte_rtpt(); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_nclip(); + code = w & 7; + vd = vtx + (w & 0xFFF8); + gte_stopz(&g.opz); + if (g.opz > 0) { + switch (code) { + case 4: + case 5: + gte_stsxy3_f3(pkt); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (((PolyF3 *)pkt)->x0 > ((PolyF3 *)pkt)->x1) { + mx = ((PolyF3 *)pkt)->x0; + mn = ((PolyF3 *)pkt)->x1; + } else { + mn = ((PolyF3 *)pkt)->x0; + mx = ((PolyF3 *)pkt)->x1; + } + if (((PolyF3 *)pkt)->x2 > mx) mx = ((PolyF3 *)pkt)->x2; + else if (((PolyF3 *)pkt)->x2 < mn) mn = ((PolyF3 *)pkt)->x2; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyF3 *)pkt)->y0 > ((PolyF3 *)pkt)->y1) { + my = ((PolyF3 *)pkt)->y0; + mny = ((PolyF3 *)pkt)->y1; + } else { + mny = ((PolyF3 *)pkt)->y0; + my = ((PolyF3 *)pkt)->y1; + } + if (((PolyF3 *)pkt)->y2 > my) my = ((PolyF3 *)pkt)->y2; + else if (((PolyF3 *)pkt)->y2 < mny) mny = ((PolyF3 *)pkt)->y2; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code != 4) g.opz = za + 0x200; + ((PolyF3 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x4000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x14; + } + } + break; + case 6: + case 7: + gte_stsxy3_ft3(pkt); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (((PolyFT3 *)pkt)->x0 > ((PolyFT3 *)pkt)->x1) { + mx = ((PolyFT3 *)pkt)->x0; + mn = ((PolyFT3 *)pkt)->x1; + } else { + mn = ((PolyFT3 *)pkt)->x0; + mx = ((PolyFT3 *)pkt)->x1; + } + if (((PolyFT3 *)pkt)->x2 > mx) mx = ((PolyFT3 *)pkt)->x2; + else if (((PolyFT3 *)pkt)->x2 < mn) mn = ((PolyFT3 *)pkt)->x2; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT3 *)pkt)->y0 > ((PolyFT3 *)pkt)->y1) { + my = ((PolyFT3 *)pkt)->y0; + mny = ((PolyFT3 *)pkt)->y1; + } else { + mny = ((PolyFT3 *)pkt)->y0; + my = ((PolyFT3 *)pkt)->y1; + } + if (((PolyFT3 *)pkt)->y2 > my) my = ((PolyFT3 *)pkt)->y2; + else if (((PolyFT3 *)pkt)->y2 < mny) mny = ((PolyFT3 *)pkt)->y2; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code == 7) g.opz = za + 0x200; + tp = (u32 *)prim->w0; + ((PolyFT3 *)pkt)->rgbc = tp[0]; + ((PolyFT3 *)pkt)->uvc0 = tp[1]; + ((PolyFT3 *)pkt)->uvp1 = tp[2]; + ((PolyFT3 *)pkt)->uv2 = tp[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x20; + } + } + break; + case 0: + case 1: + gte_stsxy3_f4(pkt); + gte_ldv0(vd); + gte_rtps(); + if (((PolyF4 *)pkt)->x0 > ((PolyF4 *)pkt)->x1) { + mx = ((PolyF4 *)pkt)->x0; + mn = ((PolyF4 *)pkt)->x1; + } else { + mn = ((PolyF4 *)pkt)->x0; + mx = ((PolyF4 *)pkt)->x1; + } + if (((PolyF4 *)pkt)->x2 > mx) mx = ((PolyF4 *)pkt)->x2; + else if (((PolyF4 *)pkt)->x2 < mn) mn = ((PolyF4 *)pkt)->x2; + if (((PolyF4 *)pkt)->y0 > ((PolyF4 *)pkt)->y1) { + my = ((PolyF4 *)pkt)->y0; + mny = ((PolyF4 *)pkt)->y1; + } else { + mny = ((PolyF4 *)pkt)->y0; + my = ((PolyF4 *)pkt)->y1; + } + if (((PolyF4 *)pkt)->y2 > my) my = ((PolyF4 *)pkt)->y2; + else if (((PolyF4 *)pkt)->y2 < mny) mny = ((PolyF4 *)pkt)->y2; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyF4 *)pkt)->x3); + if (((PolyF4 *)pkt)->x3 < mn) mn = ((PolyF4 *)pkt)->x3; + else if (mx < ((PolyF4 *)pkt)->x3) mx = ((PolyF4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyF4 *)pkt)->y3 < mny) mny = ((PolyF4 *)pkt)->y3; + else if (my < ((PolyF4 *)pkt)->y3) my = ((PolyF4 *)pkt)->y3; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + ((PolyF4 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((za >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x5000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x18; + } + } + } + break; + case 2: + gte_stsxy3c(&tmpxy[0]); + gte_ldv0(vd); + gte_rtps(); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyFT4 *)pkt)->x3); + if (((PolyFT4 *)pkt)->x3 < mn) mn = ((PolyFT4 *)pkt)->x3; + else if (mx < ((PolyFT4 *)pkt)->x3) mx = ((PolyFT4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT4 *)pkt)->y3 < mny) mny = ((PolyFT4 *)pkt)->y3; + else if (my < ((PolyFT4 *)pkt)->y3) my = ((PolyFT4 *)pkt)->y3; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + u32 uvw; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + if (code == 3) g.opz = za + 0x200; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + tp = (u32 *)prim->w0; + ((PolyFT4 *)pkt)->rgbc = tp[0]; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + } + } + } + break; + case 3: + gte_stsxy3c(&tmpxy[0]); + vv[3] = *(SVECTOR2 *)vd; + gte_ldv0(&vv[3]); + gte_rtps(); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&tmpxy[3]); + if (tmpxy[3].vx < mn) mn = tmpxy[3].vx; + else if (mx < tmpxy[3].vx) mx = tmpxy[3].vx; + if (mx >= -0xA0 && mn < 0xA1) { + if (tmpxy[3].vy < mny) mny = tmpxy[3].vy; + else if (my < tmpxy[3].vy) my = tmpxy[3].vy; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + u32 uvw; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + *(u32 *)&((PolyFT4 *)pkt)->x3 = *(u32 *)&tmpxy[3]; + ((PolyFT4 *)pkt)->rgbc = colA | 0x2E000000; + tp = (u32 *)prim->w0; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + pkt[3] = 6; + *(u32 *)(pkt + 4) = 0xE1000040; + *(u32 *)(pkt + 8) = colB | 0x2A000000; + *(u32 *)(pkt + 0xC) = *(u32 *)&tmpxy[0]; + *(u32 *)(pkt + 0x10) = *(u32 *)&tmpxy[1]; + *(u32 *)(pkt + 0x14) = *(u32 *)&tmpxy[2]; + *(u32 *)(pkt + 0x18) = *(u32 *)&tmpxy[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x6000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x1C; + } + } + } + break; + } + } + } + } + } + } + } + } + D_800A5E60 = pkt; +} + diff --git a/.run/giants/c954_b2.c b/.run/giants/c954_b2.c new file mode 100644 index 000000000..bbc4e83cd --- /dev/null +++ b/.run/giants/c954_b2.c @@ -0,0 +1,560 @@ +#include "common.h" +#include "/home/musashi/bfm-decomp/src/shared/engine_types.h" + +#define gte_ldv0(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_ldv3(r0, r1, r2) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 0( %1 );" \ + "lwc2 $3, 4( %1 );" \ + "lwc2 $4, 0( %2 );" \ + "lwc2 $5, 4( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) ) + +#define gte_ldv3c(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 8( %0 );" \ + "lwc2 $3, 12( %0 );" \ + "lwc2 $4, 16( %0 );" \ + "lwc2 $5, 20( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_rtps() __asm__ volatile ("nop;nop;rtps") +#define gte_rtpt() __asm__ volatile ("nop;nop;rtpt") +#define gte_nclip() __asm__ volatile ("nop;nop;nclip") + +#define gte_stsxy(r0) __asm__ volatile ( \ + "swc2 $14, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 0( %1 );" \ + "swc2 $14, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsxy3c(r0) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 4( %0 );" \ + "swc2 $14, 8( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_f3(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 12( %0 );" \ + "swc2 $14, 16( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_ft3(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 16( %0 );" \ + "swc2 $14, 24( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_f4(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 12( %0 );" \ + "swc2 $14, 16( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsz3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $17, 0( %0 );" \ + "swc2 $18, 0( %1 );" \ + "swc2 $19, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsz4(r0, r1, r2, r3) __asm__ volatile ( \ + "swc2 $16, 0( %0 );" \ + "swc2 $17, 0( %1 );" \ + "swc2 $18, 0( %2 );" \ + "swc2 $19, 0( %3 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ), "r"( r3 ) \ + : "memory" ) + +#define gte_stszotz(r0) __asm__ volatile ( \ + "mfc2 $12, $19;" \ + "nop;" \ + "sra $12, $12, 2;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stflg(r0) __asm__ volatile ( \ + "cfc2 $12, $31;" \ + "nop;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stopz(r0) __asm__ volatile ( \ + "swc2 $24, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +void func_8017C954(s32 arg0) +{ + typedef struct { u32 w0, w1, w2; } Prim; + + extern s32 func_800491EC(void); + extern void func_800547D8(s32, MATRIX2 *); + extern void func_80052E38(MATRIX2 *); + extern u8 *D_800A5E60; + extern u8 D_800A6610[]; + extern short D_800B9A02; + extern s32 D_801DCCA0; /* TU-visible spelling (engine_core.h + ov_SC01_000.c col-0); unsigned access forced at use — §8d sub-class (b) */ + + DVECTOR2 tmpxy[4]; + SVECTOR2 box[8]; + SVECTOR2 sxy[8]; + SVECTOR2 vv[4]; + MATRIX2 mtx; + struct { long otz, flag, opz, sz0, sz1, sz2, sz3; } g; + + s32 lim; + u32 colA; + u32 colB; + s32 nparts; + s32 j; + u32 nprim; + u32 i; + Part *part; + Prim *prim; + u8 *pkt; + u32 ot; + u8 *vtx; + u8 *va, *vb, *vc, *vd; + u32 w, code; + u32 wx, wy, wz; + s32 xa32, xb32, t32; + s32 xmn1, xmx1, xmn2, xmx2; + s32 mnc, mxc; + s16 my, mny, mx, mn; + s32 d; + u32 e; + + lim = func_800491EC() + *(s32 *)(arg0 + 0x64); + func_800547D8(arg0 + 0x10, &mtx); + func_80052E38(&mtx); + + pkt = D_800A5E60; + ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14]; + d = D_801DCCA0; + colA = (d << 16) | (d << 8) | d; + e = d * 2; + if (e > 0xFF) e = 0xFF; + colB = (e << 16) | (e << 8) | e; + part = *(Part **)(arg0 + 0xC); + nparts = *(s32 *)(*(s32 *)(arg0 + 8) + 8); + vtx = *(u8 **)(*(s32 *)(arg0 + 8) + 0x10); + + for (j = 0; j < nparts; j++, part++) { + wx = part->xx; + mn = wx; + mx = wx >> 16; + wy = part->yy; + mny = wy; + my = wy >> 16; + wz = part->zz; + box[0].vx = mn; box[0].vy = mny; + box[1].vx = mx; box[1].vy = mny; + box[2].vx = mn; box[2].vy = mny; + box[3].vx = mx; box[3].vy = mny; + box[4].vx = mn; box[4].vy = my; + box[5].vx = mx; box[5].vy = my; + box[6].vx = mn; box[6].vy = my; + box[7].vx = mx; box[7].vy = my; + wy = wz >> 16; + box[0].vz = wz; + box[1].vz = wz; + box[4].vz = wz; + box[5].vz = wz; + box[2].vz = wy; + box[3].vz = wy; + box[6].vz = wy; + box[7].vz = wy; + + gte_ldv3c(&box[0]); + gte_rtpt(); + gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]); + gte_ldv0(&box[3]); + gte_rtps(); + gte_stsxy(&sxy[3]); + gte_ldv3c(&box[4]); + __asm__ volatile (""); /* §45-B live-length slider: +1 static insn splits the 228/230 allocno-priority tie (498/498 -> 497/498) */ + gte_rtpt(); + gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]); + gte_ldv0(&box[7]); + gte_rtps(); + gte_stsxy(&sxy[7]); + gte_stszotz(&g.otz); + + if (lim >= g.otz) { + xa32 = sxy[0].vx; + xb32 = sxy[1].vx; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vx; + xb32 = sxy[5].vx; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0xA0 && (s16)mnc < 0xA1) { + xa32 = sxy[0].vy; + xb32 = sxy[1].vy; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vy; + xb32 = sxy[5].vy; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0x78 && (s16)mnc < 0x79) { + prim = (Prim *)part->prim; + nprim = part->nprim; + for (i = 0; i < nprim; i++, prim++) { + w = prim->w1; + va = vtx + (w & 0xFFFF); + vb = vtx + (w >> 16); + w = prim->w2; + vc = vtx + (w & 0xFFFF); + w = w >> 16; + gte_ldv3(va, vb, vc); + gte_rtpt(); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_nclip(); + code = w & 7; + vd = vtx + (w & 0xFFF8); + gte_stopz(&g.opz); + if (g.opz > 0) { + switch (code) { + case 4: + case 5: + gte_stsxy3_f3(pkt); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (((PolyF3 *)pkt)->x0 > ((PolyF3 *)pkt)->x1) { + mx = ((PolyF3 *)pkt)->x0; + mn = ((PolyF3 *)pkt)->x1; + } else { + mn = ((PolyF3 *)pkt)->x0; + mx = ((PolyF3 *)pkt)->x1; + } + if (((PolyF3 *)pkt)->x2 > mx) mx = ((PolyF3 *)pkt)->x2; + else if (((PolyF3 *)pkt)->x2 < mn) mn = ((PolyF3 *)pkt)->x2; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyF3 *)pkt)->y0 > ((PolyF3 *)pkt)->y1) { + my = ((PolyF3 *)pkt)->y0; + mny = ((PolyF3 *)pkt)->y1; + } else { + mny = ((PolyF3 *)pkt)->y0; + my = ((PolyF3 *)pkt)->y1; + } + if (((PolyF3 *)pkt)->y2 > my) my = ((PolyF3 *)pkt)->y2; + else if (((PolyF3 *)pkt)->y2 < mny) mny = ((PolyF3 *)pkt)->y2; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code != 4) g.opz = za + 0x200; + ((PolyF3 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x4000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x14; + } + } + break; + case 6: + case 7: + gte_stsxy3_ft3(pkt); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (((PolyFT3 *)pkt)->x0 > ((PolyFT3 *)pkt)->x1) { + mx = ((PolyFT3 *)pkt)->x0; + mn = ((PolyFT3 *)pkt)->x1; + } else { + mn = ((PolyFT3 *)pkt)->x0; + mx = ((PolyFT3 *)pkt)->x1; + } + if (((PolyFT3 *)pkt)->x2 > mx) mx = ((PolyFT3 *)pkt)->x2; + else if (((PolyFT3 *)pkt)->x2 < mn) mn = ((PolyFT3 *)pkt)->x2; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT3 *)pkt)->y0 > ((PolyFT3 *)pkt)->y1) { + my = ((PolyFT3 *)pkt)->y0; + mny = ((PolyFT3 *)pkt)->y1; + } else { + mny = ((PolyFT3 *)pkt)->y0; + my = ((PolyFT3 *)pkt)->y1; + } + if (((PolyFT3 *)pkt)->y2 > my) my = ((PolyFT3 *)pkt)->y2; + else if (((PolyFT3 *)pkt)->y2 < mny) mny = ((PolyFT3 *)pkt)->y2; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code == 7) g.opz = za + 0x200; + tp = (u32 *)prim->w0; + ((PolyFT3 *)pkt)->rgbc = tp[0]; + ((PolyFT3 *)pkt)->uvc0 = tp[1]; + ((PolyFT3 *)pkt)->uvp1 = tp[2]; + ((PolyFT3 *)pkt)->uv2 = tp[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x20; + } + } + break; + case 0: + case 1: + gte_stsxy3_f4(pkt); + gte_ldv0(vd); + gte_rtps(); + if (((PolyF4 *)pkt)->x0 > ((PolyF4 *)pkt)->x1) { + mx = ((PolyF4 *)pkt)->x0; + mn = ((PolyF4 *)pkt)->x1; + } else { + mn = ((PolyF4 *)pkt)->x0; + mx = ((PolyF4 *)pkt)->x1; + } + if (((PolyF4 *)pkt)->x2 > mx) mx = ((PolyF4 *)pkt)->x2; + else if (((PolyF4 *)pkt)->x2 < mn) mn = ((PolyF4 *)pkt)->x2; + if (((PolyF4 *)pkt)->y0 > ((PolyF4 *)pkt)->y1) { + my = ((PolyF4 *)pkt)->y0; + mny = ((PolyF4 *)pkt)->y1; + } else { + mny = ((PolyF4 *)pkt)->y0; + my = ((PolyF4 *)pkt)->y1; + } + if (((PolyF4 *)pkt)->y2 > my) my = ((PolyF4 *)pkt)->y2; + else if (((PolyF4 *)pkt)->y2 < mny) mny = ((PolyF4 *)pkt)->y2; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyF4 *)pkt)->x3); + if (((PolyF4 *)pkt)->x3 < mn) mn = ((PolyF4 *)pkt)->x3; + else if (mx < ((PolyF4 *)pkt)->x3) mx = ((PolyF4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyF4 *)pkt)->y3 < mny) mny = ((PolyF4 *)pkt)->y3; + else if (my < ((PolyF4 *)pkt)->y3) my = ((PolyF4 *)pkt)->y3; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + ((PolyF4 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((za >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x5000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x18; + } + } + } + break; + case 2: + gte_stsxy3c(&tmpxy[0]); + gte_ldv0(vd); + gte_rtps(); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyFT4 *)pkt)->x3); + if (((PolyFT4 *)pkt)->x3 < mn) mn = ((PolyFT4 *)pkt)->x3; + else if (mx < ((PolyFT4 *)pkt)->x3) mx = ((PolyFT4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT4 *)pkt)->y3 < mny) mny = ((PolyFT4 *)pkt)->y3; + else if (my < ((PolyFT4 *)pkt)->y3) my = ((PolyFT4 *)pkt)->y3; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + u32 uvw; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + if (code == 3) g.opz = za + 0x200; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + tp = (u32 *)prim->w0; + ((PolyFT4 *)pkt)->rgbc = tp[0]; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + } + } + } + break; + case 3: + gte_stsxy3c(&tmpxy[0]); + vv[3] = *(SVECTOR2 *)vd; + gte_ldv0(&vv[3]); + gte_rtps(); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&tmpxy[3]); + if (tmpxy[3].vx < mn) mn = tmpxy[3].vx; + else if (mx < tmpxy[3].vx) mx = tmpxy[3].vx; + if (mx >= -0xA0 && mn < 0xA1) { + if (tmpxy[3].vy < mny) mny = tmpxy[3].vy; + else if (my < tmpxy[3].vy) my = tmpxy[3].vy; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + u32 uvw; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + *(u32 *)&((PolyFT4 *)pkt)->x3 = *(u32 *)&tmpxy[3]; + ((PolyFT4 *)pkt)->rgbc = colA | 0x2E000000; + tp = (u32 *)prim->w0; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((za >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + pkt[3] = 6; + *(u32 *)(pkt + 4) = 0xE1000040; + *(u32 *)(pkt + 8) = colB | 0x2A000000; + *(u32 *)(pkt + 0xC) = *(u32 *)&tmpxy[0]; + *(u32 *)(pkt + 0x10) = *(u32 *)&tmpxy[1]; + *(u32 *)(pkt + 0x14) = *(u32 *)&tmpxy[2]; + *(u32 *)(pkt + 0x18) = *(u32 *)&tmpxy[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x6000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x1C; + } + } + } + break; + } + } + } + } + } + } + } + } + D_800A5E60 = pkt; +} + diff --git a/.run/giants/c954_b3.c b/.run/giants/c954_b3.c new file mode 100644 index 000000000..b1088b030 --- /dev/null +++ b/.run/giants/c954_b3.c @@ -0,0 +1,561 @@ +#include "common.h" +#include "/home/musashi/bfm-decomp/src/shared/engine_types.h" + +#define gte_ldv0(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_ldv3(r0, r1, r2) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 0( %1 );" \ + "lwc2 $3, 4( %1 );" \ + "lwc2 $4, 0( %2 );" \ + "lwc2 $5, 4( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) ) + +#define gte_ldv3c(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 8( %0 );" \ + "lwc2 $3, 12( %0 );" \ + "lwc2 $4, 16( %0 );" \ + "lwc2 $5, 20( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_rtps() __asm__ volatile ("nop;nop;rtps") +#define gte_rtpt() __asm__ volatile ("nop;nop;rtpt") +#define gte_nclip() __asm__ volatile ("nop;nop;nclip") + +#define gte_stsxy(r0) __asm__ volatile ( \ + "swc2 $14, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 0( %1 );" \ + "swc2 $14, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsxy3c(r0) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 4( %0 );" \ + "swc2 $14, 8( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_f3(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 12( %0 );" \ + "swc2 $14, 16( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_ft3(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 16( %0 );" \ + "swc2 $14, 24( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_f4(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 12( %0 );" \ + "swc2 $14, 16( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsz3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $17, 0( %0 );" \ + "swc2 $18, 0( %1 );" \ + "swc2 $19, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsz4(r0, r1, r2, r3) __asm__ volatile ( \ + "swc2 $16, 0( %0 );" \ + "swc2 $17, 0( %1 );" \ + "swc2 $18, 0( %2 );" \ + "swc2 $19, 0( %3 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ), "r"( r3 ) \ + : "memory" ) + +#define gte_stszotz(r0) __asm__ volatile ( \ + "mfc2 $12, $19;" \ + "nop;" \ + "sra $12, $12, 2;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stflg(r0) __asm__ volatile ( \ + "cfc2 $12, $31;" \ + "nop;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stopz(r0) __asm__ volatile ( \ + "swc2 $24, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +void func_8017C954(s32 arg0) +{ + typedef struct { u32 w0, w1, w2; } Prim; + + extern s32 func_800491EC(void); + extern void func_800547D8(s32, MATRIX2 *); + extern void func_80052E38(MATRIX2 *); + extern u8 *D_800A5E60; + extern u8 D_800A6610[]; + extern short D_800B9A02; + extern s32 D_801DCCA0; /* TU-visible spelling (engine_core.h + ov_SC01_000.c col-0); unsigned access forced at use — §8d sub-class (b) */ + + DVECTOR2 tmpxy[4]; + SVECTOR2 box[8]; + SVECTOR2 sxy[8]; + SVECTOR2 vv[4]; + MATRIX2 mtx; + struct { long otz, flag, opz, sz0, sz1, sz2, sz3; } g; + + s32 lim; + u32 colA; + u32 colB; + u32 nprim; + s32 nparts; + s32 j; + u32 i; + Part *part; + Prim *prim; + u8 *pkt; + u32 ot; + u8 *vtx; + u8 *va, *vb, *vc, *vd; + u32 w, code; + u32 wx, wy, wz; + s32 xa32, xb32, t32; + s32 xmn1, xmx1, xmn2, xmx2; + s32 mnc, mxc; + s16 my, mny, mx, mn; + s32 d; + u32 e; + + lim = func_800491EC() + *(s32 *)(arg0 + 0x64); + func_800547D8(arg0 + 0x10, &mtx); + func_80052E38(&mtx); + + pkt = D_800A5E60; + ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14]; + d = D_801DCCA0; + colA = (d << 16) | (d << 8) | d; + e = d * 2; + if (e > 0xFF) e = 0xFF; + colB = (e << 16) | (e << 8) | e; + part = *(Part **)(arg0 + 0xC); + nparts = *(s32 *)(*(s32 *)(arg0 + 8) + 8); + vtx = *(u8 **)(*(s32 *)(arg0 + 8) + 0x10); + + for (j = 0; j < nparts; j++, part++) { + wx = part->xx; + mn = wx; + mx = wx >> 16; + wy = part->yy; + mny = wy; + my = wy >> 16; + wz = part->zz; + box[0].vx = mn; box[0].vy = mny; + box[1].vx = mx; box[1].vy = mny; + box[2].vx = mn; box[2].vy = mny; + box[3].vx = mx; box[3].vy = mny; + box[4].vx = mn; box[4].vy = my; + box[5].vx = mx; box[5].vy = my; + box[6].vx = mn; box[6].vy = my; + box[7].vx = mx; box[7].vy = my; + wy = wz >> 16; + box[0].vz = wz; + box[1].vz = wz; + box[4].vz = wz; + box[5].vz = wz; + box[2].vz = wy; + box[3].vz = wy; + box[6].vz = wy; + box[7].vz = wy; + + gte_ldv3c(&box[0]); + gte_rtpt(); + gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]); + gte_ldv0(&box[3]); + gte_rtps(); + gte_stsxy(&sxy[3]); + gte_ldv3c(&box[4]); + __asm__ volatile (""); /* §45-B live-length slider: +1 static insn splits the 228/230 allocno-priority tie (498/498 -> 497/498) */ + gte_rtpt(); + gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]); + gte_ldv0(&box[7]); + gte_rtps(); + gte_stsxy(&sxy[7]); + gte_stszotz(&g.otz); + + if (lim >= g.otz) { + xa32 = sxy[0].vx; + xb32 = sxy[1].vx; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vx; + xb32 = sxy[5].vx; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0xA0 && (s16)mnc < 0xA1) { + xa32 = sxy[0].vy; + xb32 = sxy[1].vy; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vy; + xb32 = sxy[5].vy; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0x78 && (s16)mnc < 0x79) { + prim = (Prim *)part->prim; + nprim = part->nprim; + for (i = 0; i < nprim; i++, prim++) { + w = prim->w1; + va = vtx + (w & 0xFFFF); + vb = vtx + (w >> 16); + w = prim->w2; + vc = vtx + (w & 0xFFFF); + w = w >> 16; + gte_ldv3(va, vb, vc); + gte_rtpt(); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_nclip(); + code = w & 7; + vd = vtx + (w & 0xFFF8); + gte_stopz(&g.opz); + if (g.opz > 0) { + switch (code) { + case 4: + case 5: + gte_stsxy3_f3(pkt); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (((PolyF3 *)pkt)->x0 > ((PolyF3 *)pkt)->x1) { + mx = ((PolyF3 *)pkt)->x0; + mn = ((PolyF3 *)pkt)->x1; + } else { + mn = ((PolyF3 *)pkt)->x0; + mx = ((PolyF3 *)pkt)->x1; + } + if (((PolyF3 *)pkt)->x2 > mx) mx = ((PolyF3 *)pkt)->x2; + else if (((PolyF3 *)pkt)->x2 < mn) mn = ((PolyF3 *)pkt)->x2; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyF3 *)pkt)->y0 > ((PolyF3 *)pkt)->y1) { + my = ((PolyF3 *)pkt)->y0; + mny = ((PolyF3 *)pkt)->y1; + } else { + mny = ((PolyF3 *)pkt)->y0; + my = ((PolyF3 *)pkt)->y1; + } + if (((PolyF3 *)pkt)->y2 > my) my = ((PolyF3 *)pkt)->y2; + else if (((PolyF3 *)pkt)->y2 < mny) mny = ((PolyF3 *)pkt)->y2; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code != 4) g.opz = za + 0x200; + ((PolyF3 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x4000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x14; + } + } + break; + case 6: + case 7: + gte_stsxy3_ft3(pkt); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (((PolyFT3 *)pkt)->x0 > ((PolyFT3 *)pkt)->x1) { + mx = ((PolyFT3 *)pkt)->x0; + mn = ((PolyFT3 *)pkt)->x1; + } else { + mn = ((PolyFT3 *)pkt)->x0; + mx = ((PolyFT3 *)pkt)->x1; + } + if (((PolyFT3 *)pkt)->x2 > mx) mx = ((PolyFT3 *)pkt)->x2; + else if (((PolyFT3 *)pkt)->x2 < mn) mn = ((PolyFT3 *)pkt)->x2; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT3 *)pkt)->y0 > ((PolyFT3 *)pkt)->y1) { + my = ((PolyFT3 *)pkt)->y0; + mny = ((PolyFT3 *)pkt)->y1; + } else { + mny = ((PolyFT3 *)pkt)->y0; + my = ((PolyFT3 *)pkt)->y1; + } + if (((PolyFT3 *)pkt)->y2 > my) my = ((PolyFT3 *)pkt)->y2; + else if (((PolyFT3 *)pkt)->y2 < mny) mny = ((PolyFT3 *)pkt)->y2; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code == 7) g.opz = za + 0x200; + tp = (u32 *)prim->w0; + ((PolyFT3 *)pkt)->rgbc = tp[0]; + ((PolyFT3 *)pkt)->uvc0 = tp[1]; + ((PolyFT3 *)pkt)->uvp1 = tp[2]; + ((PolyFT3 *)pkt)->uv2 = tp[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x20; + } + } + break; + case 0: + case 1: + gte_stsxy3_f4(pkt); + gte_ldv0(vd); + gte_rtps(); + if (((PolyF4 *)pkt)->x0 > ((PolyF4 *)pkt)->x1) { + mx = ((PolyF4 *)pkt)->x0; + mn = ((PolyF4 *)pkt)->x1; + } else { + mn = ((PolyF4 *)pkt)->x0; + mx = ((PolyF4 *)pkt)->x1; + } + if (((PolyF4 *)pkt)->x2 > mx) mx = ((PolyF4 *)pkt)->x2; + else if (((PolyF4 *)pkt)->x2 < mn) mn = ((PolyF4 *)pkt)->x2; + if (((PolyF4 *)pkt)->y0 > ((PolyF4 *)pkt)->y1) { + my = ((PolyF4 *)pkt)->y0; + mny = ((PolyF4 *)pkt)->y1; + } else { + mny = ((PolyF4 *)pkt)->y0; + my = ((PolyF4 *)pkt)->y1; + } + if (((PolyF4 *)pkt)->y2 > my) my = ((PolyF4 *)pkt)->y2; + else if (((PolyF4 *)pkt)->y2 < mny) mny = ((PolyF4 *)pkt)->y2; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyF4 *)pkt)->x3); + if (((PolyF4 *)pkt)->x3 < mn) mn = ((PolyF4 *)pkt)->x3; + else if (mx < ((PolyF4 *)pkt)->x3) mx = ((PolyF4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyF4 *)pkt)->y3 < mny) mny = ((PolyF4 *)pkt)->y3; + else if (my < ((PolyF4 *)pkt)->y3) my = ((PolyF4 *)pkt)->y3; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + ((PolyF4 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((za >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x5000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x18; + } + } + } + break; + case 2: + gte_stsxy3c(&tmpxy[0]); + gte_ldv0(vd); + gte_rtps(); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyFT4 *)pkt)->x3); + if (((PolyFT4 *)pkt)->x3 < mn) mn = ((PolyFT4 *)pkt)->x3; + else if (mx < ((PolyFT4 *)pkt)->x3) mx = ((PolyFT4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT4 *)pkt)->y3 < mny) mny = ((PolyFT4 *)pkt)->y3; + else if (my < ((PolyFT4 *)pkt)->y3) my = ((PolyFT4 *)pkt)->y3; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + u32 uvw; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + if (code == 3) g.opz = za + 0x200; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + tp = (u32 *)prim->w0; + ((PolyFT4 *)pkt)->rgbc = tp[0]; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + } + } + } + break; + case 3: + gte_stsxy3c(&tmpxy[0]); + vv[3] = *(SVECTOR2 *)vd; + gte_ldv0(&vv[3]); + gte_rtps(); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + __asm__ __volatile__ ("" ::: "$3"); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&tmpxy[3]); + if (tmpxy[3].vx < mn) mn = tmpxy[3].vx; + else if (mx < tmpxy[3].vx) mx = tmpxy[3].vx; + if (mx >= -0xA0 && mn < 0xA1) { + if (tmpxy[3].vy < mny) mny = tmpxy[3].vy; + else if (my < tmpxy[3].vy) my = tmpxy[3].vy; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + u32 uvw; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + *(u32 *)&((PolyFT4 *)pkt)->x3 = *(u32 *)&tmpxy[3]; + ((PolyFT4 *)pkt)->rgbc = colA | 0x2E000000; + tp = (u32 *)prim->w0; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + pkt[3] = 6; + *(u32 *)(pkt + 4) = 0xE1000040; + *(u32 *)(pkt + 8) = colB | 0x2A000000; + *(u32 *)(pkt + 0xC) = *(u32 *)&tmpxy[0]; + *(u32 *)(pkt + 0x10) = *(u32 *)&tmpxy[1]; + *(u32 *)(pkt + 0x14) = *(u32 *)&tmpxy[2]; + *(u32 *)(pkt + 0x18) = *(u32 *)&tmpxy[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x6000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x1C; + } + } + } + break; + } + } + } + } + } + } + } + } + D_800A5E60 = pkt; +} + diff --git a/.run/giants/c954_b4.c b/.run/giants/c954_b4.c new file mode 100644 index 000000000..91925f257 --- /dev/null +++ b/.run/giants/c954_b4.c @@ -0,0 +1,566 @@ +#include "common.h" +#include "/home/musashi/bfm-decomp/src/shared/engine_types.h" + +#define gte_ldv0(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_ldv3(r0, r1, r2) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 0( %1 );" \ + "lwc2 $3, 4( %1 );" \ + "lwc2 $4, 0( %2 );" \ + "lwc2 $5, 4( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) ) + +#define gte_ldv3c(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 8( %0 );" \ + "lwc2 $3, 12( %0 );" \ + "lwc2 $4, 16( %0 );" \ + "lwc2 $5, 20( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_rtps() __asm__ volatile ("nop;nop;rtps") +#define gte_rtpt() __asm__ volatile ("nop;nop;rtpt") +#define gte_nclip() __asm__ volatile ("nop;nop;nclip") + +#define gte_stsxy(r0) __asm__ volatile ( \ + "swc2 $14, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 0( %1 );" \ + "swc2 $14, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsxy3c(r0) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 4( %0 );" \ + "swc2 $14, 8( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_f3(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 12( %0 );" \ + "swc2 $14, 16( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_ft3(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 16( %0 );" \ + "swc2 $14, 24( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_f4(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 12( %0 );" \ + "swc2 $14, 16( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsz3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $17, 0( %0 );" \ + "swc2 $18, 0( %1 );" \ + "swc2 $19, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsz4(r0, r1, r2, r3) __asm__ volatile ( \ + "swc2 $16, 0( %0 );" \ + "swc2 $17, 0( %1 );" \ + "swc2 $18, 0( %2 );" \ + "swc2 $19, 0( %3 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ), "r"( r3 ) \ + : "memory" ) + +#define gte_stszotz(r0) __asm__ volatile ( \ + "mfc2 $12, $19;" \ + "nop;" \ + "sra $12, $12, 2;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stflg(r0) __asm__ volatile ( \ + "cfc2 $12, $31;" \ + "nop;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stopz(r0) __asm__ volatile ( \ + "swc2 $24, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +void func_8017C954(s32 arg0) +{ + typedef struct { u32 w0, w1, w2; } Prim; + + extern s32 func_800491EC(void); + extern void func_800547D8(s32, MATRIX2 *); + extern void func_80052E38(MATRIX2 *); + extern u8 *D_800A5E60; + extern u8 D_800A6610[]; + extern short D_800B9A02; + extern s32 D_801DCCA0; /* TU-visible spelling (engine_core.h + ov_SC01_000.c col-0); unsigned access forced at use — §8d sub-class (b) */ + + DVECTOR2 tmpxy[4]; + SVECTOR2 box[8]; + SVECTOR2 sxy[8]; + SVECTOR2 vv[4]; + MATRIX2 mtx; + struct { long otz, flag, opz, sz0, sz1, sz2, sz3; } g; + + s32 lim; + u32 colA; + u32 colB; + u32 nprim; + s32 nparts; + s32 j; + u32 i; + Part *part; + Prim *prim; + u8 *pkt; + u32 ot; + u8 *vtx; + u8 *va, *vb, *vc, *vd; + u32 w, code; + u32 wx, wy, wz; + s32 xa32, xb32, t32; + s32 xmn1, xmx1, xmn2, xmx2; + s32 mnc, mxc; + s16 my, mny, mx, mn; + s32 d; + u32 e; + + lim = func_800491EC() + *(s32 *)(arg0 + 0x64); + func_800547D8(arg0 + 0x10, &mtx); + func_80052E38(&mtx); + + pkt = D_800A5E60; + ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14]; + d = D_801DCCA0; + colA = (d << 16) | (d << 8) | d; + e = d * 2; + if (e > 0xFF) e = 0xFF; + colB = (e << 16) | (e << 8) | e; + part = *(Part **)(arg0 + 0xC); + nparts = *(s32 *)(*(s32 *)(arg0 + 8) + 8); + vtx = *(u8 **)(*(s32 *)(arg0 + 8) + 0x10); + + for (j = 0; j < nparts; j++, part++) { + wx = part->xx; + mn = wx; + mx = wx >> 16; + wy = part->yy; + mny = wy; + my = wy >> 16; + wz = part->zz; + box[0].vx = mn; box[0].vy = mny; + box[1].vx = mx; box[1].vy = mny; + box[2].vx = mn; box[2].vy = mny; + box[3].vx = mx; box[3].vy = mny; + box[4].vx = mn; box[4].vy = my; + box[5].vx = mx; box[5].vy = my; + box[6].vx = mn; box[6].vy = my; + box[7].vx = mx; box[7].vy = my; + wy = wz >> 16; + box[0].vz = wz; + box[1].vz = wz; + box[4].vz = wz; + box[5].vz = wz; + box[2].vz = wy; + box[3].vz = wy; + box[6].vz = wy; + box[7].vz = wy; + + gte_ldv3c(&box[0]); + gte_rtpt(); + gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]); + gte_ldv0(&box[3]); + gte_rtps(); + gte_stsxy(&sxy[3]); + gte_ldv3c(&box[4]); + __asm__ volatile (""); /* §45-B live-length slider: +1 static insn splits the 228/230 allocno-priority tie (498/498 -> 497/498) */ + __asm__ volatile (""); /* §45-B live-length slider: +1 static insn splits the 228/230 allocno-priority tie (498/498 -> 497/498) */ + gte_rtpt(); + gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]); + gte_ldv0(&box[7]); + gte_rtps(); + gte_stsxy(&sxy[7]); + gte_stszotz(&g.otz); + + if (lim >= g.otz) { + xa32 = sxy[0].vx; + xb32 = sxy[1].vx; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vx; + xb32 = sxy[5].vx; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0xA0 && (s16)mnc < 0xA1) { + xa32 = sxy[0].vy; + xb32 = sxy[1].vy; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vy; + xb32 = sxy[5].vy; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0x78 && (s16)mnc < 0x79) { + prim = (Prim *)part->prim; + nprim = part->nprim; + for (i = 0; i < nprim; i++, prim++) { + w = prim->w1; + va = vtx + (w & 0xFFFF); + vb = vtx + (w >> 16); + w = prim->w2; + vc = vtx + (w & 0xFFFF); + w = w >> 16; + gte_ldv3(va, vb, vc); + gte_rtpt(); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_nclip(); + code = w & 7; + vd = vtx + (w & 0xFFF8); + gte_stopz(&g.opz); + if (g.opz > 0) { + switch (code) { + case 4: + case 5: + gte_stsxy3_f3(pkt); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (((PolyF3 *)pkt)->x0 > ((PolyF3 *)pkt)->x1) { + mx = ((PolyF3 *)pkt)->x0; + mn = ((PolyF3 *)pkt)->x1; + } else { + mn = ((PolyF3 *)pkt)->x0; + mx = ((PolyF3 *)pkt)->x1; + } + if (((PolyF3 *)pkt)->x2 > mx) mx = ((PolyF3 *)pkt)->x2; + else if (((PolyF3 *)pkt)->x2 < mn) mn = ((PolyF3 *)pkt)->x2; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyF3 *)pkt)->y0 > ((PolyF3 *)pkt)->y1) { + my = ((PolyF3 *)pkt)->y0; + mny = ((PolyF3 *)pkt)->y1; + } else { + mny = ((PolyF3 *)pkt)->y0; + my = ((PolyF3 *)pkt)->y1; + } + if (((PolyF3 *)pkt)->y2 > my) my = ((PolyF3 *)pkt)->y2; + else if (((PolyF3 *)pkt)->y2 < mny) mny = ((PolyF3 *)pkt)->y2; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code != 4) g.opz = za + 0x200; + ((PolyF3 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x4000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x14; + } + } + break; + case 6: + case 7: + gte_stsxy3_ft3(pkt); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (((PolyFT3 *)pkt)->x0 > ((PolyFT3 *)pkt)->x1) { + mx = ((PolyFT3 *)pkt)->x0; + mn = ((PolyFT3 *)pkt)->x1; + } else { + mn = ((PolyFT3 *)pkt)->x0; + mx = ((PolyFT3 *)pkt)->x1; + } + if (((PolyFT3 *)pkt)->x2 > mx) mx = ((PolyFT3 *)pkt)->x2; + else if (((PolyFT3 *)pkt)->x2 < mn) mn = ((PolyFT3 *)pkt)->x2; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT3 *)pkt)->y0 > ((PolyFT3 *)pkt)->y1) { + my = ((PolyFT3 *)pkt)->y0; + mny = ((PolyFT3 *)pkt)->y1; + } else { + mny = ((PolyFT3 *)pkt)->y0; + my = ((PolyFT3 *)pkt)->y1; + } + if (((PolyFT3 *)pkt)->y2 > my) my = ((PolyFT3 *)pkt)->y2; + else if (((PolyFT3 *)pkt)->y2 < mny) mny = ((PolyFT3 *)pkt)->y2; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code == 7) g.opz = za + 0x200; + tp = (u32 *)prim->w0; + ((PolyFT3 *)pkt)->rgbc = tp[0]; + ((PolyFT3 *)pkt)->uvc0 = tp[1]; + ((PolyFT3 *)pkt)->uvp1 = tp[2]; + ((PolyFT3 *)pkt)->uv2 = tp[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x20; + } + } + break; + case 0: + case 1: + gte_stsxy3_f4(pkt); + gte_ldv0(vd); + gte_rtps(); + if (((PolyF4 *)pkt)->x0 > ((PolyF4 *)pkt)->x1) { + mx = ((PolyF4 *)pkt)->x0; + mn = ((PolyF4 *)pkt)->x1; + } else { + mn = ((PolyF4 *)pkt)->x0; + mx = ((PolyF4 *)pkt)->x1; + } + if (((PolyF4 *)pkt)->x2 > mx) mx = ((PolyF4 *)pkt)->x2; + else if (((PolyF4 *)pkt)->x2 < mn) mn = ((PolyF4 *)pkt)->x2; + if (((PolyF4 *)pkt)->y0 > ((PolyF4 *)pkt)->y1) { + my = ((PolyF4 *)pkt)->y0; + mny = ((PolyF4 *)pkt)->y1; + } else { + mny = ((PolyF4 *)pkt)->y0; + my = ((PolyF4 *)pkt)->y1; + } + if (((PolyF4 *)pkt)->y2 > my) my = ((PolyF4 *)pkt)->y2; + else if (((PolyF4 *)pkt)->y2 < mny) mny = ((PolyF4 *)pkt)->y2; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyF4 *)pkt)->x3); + if (((PolyF4 *)pkt)->x3 < mn) mn = ((PolyF4 *)pkt)->x3; + else if (mx < ((PolyF4 *)pkt)->x3) mx = ((PolyF4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyF4 *)pkt)->y3 < mny) mny = ((PolyF4 *)pkt)->y3; + else if (my < ((PolyF4 *)pkt)->y3) my = ((PolyF4 *)pkt)->y3; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + ((PolyF4 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((za >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x5000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x18; + } + } + } + break; + case 2: + gte_stsxy3c(&tmpxy[0]); + gte_ldv0(vd); + gte_rtps(); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyFT4 *)pkt)->x3); + if (((PolyFT4 *)pkt)->x3 < mn) mn = ((PolyFT4 *)pkt)->x3; + else if (mx < ((PolyFT4 *)pkt)->x3) mx = ((PolyFT4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT4 *)pkt)->y3 < mny) mny = ((PolyFT4 *)pkt)->y3; + else if (my < ((PolyFT4 *)pkt)->y3) my = ((PolyFT4 *)pkt)->y3; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + u32 uvw; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + if (code == 3) g.opz = za + 0x200; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + tp = (u32 *)prim->w0; + ((PolyFT4 *)pkt)->rgbc = tp[0]; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + } + } + } + break; + case 3: + gte_stsxy3c(&tmpxy[0]); + vv[3] = *(SVECTOR2 *)vd; + gte_ldv0(&vv[3]); + gte_rtps(); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + __asm__ __volatile__ ("" ::: "$3"); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&tmpxy[3]); + if (tmpxy[3].vx < mn) mn = tmpxy[3].vx; + else if (mx < tmpxy[3].vx) mx = tmpxy[3].vx; + if (mx >= -0xA0 && mn < 0xA1) { + if (tmpxy[3].vy < mny) mny = tmpxy[3].vy; + else if (my < tmpxy[3].vy) my = tmpxy[3].vy; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + u32 uvw; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + *(u32 *)&((PolyFT4 *)pkt)->x3 = *(u32 *)&tmpxy[3]; + tp = (u32 *)prim->w0; + ((PolyFT4 *)pkt)->rgbc = colA | 0x2E000000; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + { + u32 *op1 = (u32 *)(((za >> 2) << 2) + ot); + *(u32 *)pkt = (*op1 & 0xFFFFFF) | 0x9000000; + *op1 = (*op1 & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + } + pkt += 0x28; + pkt[3] = 6; + *(u32 *)(pkt + 4) = 0xE1000040; + *(u32 *)(pkt + 8) = colB | 0x2A000000; + *(u32 *)(pkt + 0xC) = *(u32 *)&tmpxy[0]; + *(u32 *)(pkt + 0x10) = *(u32 *)&tmpxy[1]; + *(u32 *)(pkt + 0x14) = *(u32 *)&tmpxy[2]; + *(u32 *)(pkt + 0x18) = *(u32 *)&tmpxy[3]; + { + u32 *op2 = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*op2 & 0xFFFFFF) | 0x6000000; + *op2 = (*op2 & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + } + pkt += 0x1C; + } + } + } + break; + } + } + } + } + } + } + } + } + D_800A5E60 = pkt; +} + diff --git a/.run/giants/c954_cc.sh b/.run/giants/c954_cc.sh new file mode 100644 index 000000000..5cbe6a77b --- /dev/null +++ b/.run/giants/c954_cc.sh @@ -0,0 +1,22 @@ +#!/bin/bash +# compile ONE draft of func_8017C954 the pinned-triple way, into $2 (a work dir). +# usage: c954_cc.sh [extra cc1 flags e.g. -da] +set -e +cd /home/musashi/bfm-decomp +SRC="$1"; WD="$2"; shift 2 +mkdir -p "$WD" +CPPFLAGS="-lang-c -Iinclude -undef -Wall -fno-builtin -Dmips -D__GNUC__=2 -D__OPTIMIZE__ -Dpsx -D_PSYQ -D_MIPSEL -D_LANGUAGE_C" +CC1FLAGS="-quiet -O2 -G0 -mips1 -mcpu=3000 -mgas -msoft-float -fgnu-linker" +python3 - "$SRC" "$WD/t.c" <<'PY' +import sys, os +sys.path.insert(0, '/home/musashi/bfm-decomp/tools') +import masked_diff +src = masked_diff.strip_scalar_typedefs(open(sys.argv[1]).read()) +if '#include "common.h"' not in src: src = '#include "common.h"\n' + src +open(sys.argv[2], 'w').write(src) +PY +mipsel-linux-gnu-cpp $CPPFLAGS "$WD/t.c" > "$WD/t.i" +( cd "$WD" && /home/musashi/bfm-decomp/tools/bin/gcc-2.7.2-psx/cc1 $CC1FLAGS "$@" t.i -o t.s.raw ) +.venv/bin/python tools/maspsx/maspsx.py --aspsx-version=2.56 --expand-div < "$WD/t.s.raw" > "$WD/t.s" +mipsel-linux-gnu-as -Iinclude -march=r3000 -mtune=r3000 -no-pad-sections -O1 -G0 -o "$WD/t.o" "$WD/t.s" +echo "OK -> $WD/t.o" diff --git a/.run/giants/c954_full.py b/.run/giants/c954_full.py new file mode 100644 index 000000000..0efcee962 --- /dev/null +++ b/.run/giants/c954_full.py @@ -0,0 +1,16 @@ +#!/usr/bin/env python3 +"""Full (uncapped) masked index-wise diff of a compiled object vs the target .s. +usage: c954_full.py [fn]""" +import sys, os +sys.path.insert(0, '/home/musashi/bfm-decomp/tools') +import masked_diff + +OBJ = sys.argv[1] +FN = sys.argv[2] if len(sys.argv) > 2 else 'func_8017C954' +TGT = '/home/musashi/bfm-decomp/asm/ov_SC06_029/nonmatchings/ov_SC06_029_jr_8017AE2C/%s.s' % FN +mine = masked_diff.insns_from_object(OBJ, FN) +tgt = masked_diff.insns_from_s(TGT) +diffs = masked_diff.structured_diff(mine, tgt) +print('mine=%d target=%d %d mismatched' % (len(mine), len(tgt), len(diffs))) +for i, me, tg in diffs: + print(' %4d | %-32s | %s' % (i, me, tg)) diff --git a/.run/giants/c954_mk.py b/.run/giants/c954_mk.py new file mode 100644 index 000000000..72231e7eb --- /dev/null +++ b/.run/giants/c954_mk.py @@ -0,0 +1,106 @@ +#!/usr/bin/env python3 +"""Variant generator for func_8017C954. + +usage: c954_mk.py [lever ...] + +Every lever is a list of (old, new) string replacements applied to the base text. +Each replacement is ASSERTED to apply (count must match) so a "neutral" reading +can never be a silent no-op. Unknown lever name -> hard error. +""" +import sys + +# ---------------------------------------------------------------- levers +L = {} + +# --- declaration-order levers ------------------------------------------- +L['nprim_first'] = [(""" s32 nparts; + s32 j; + u32 nprim; + u32 i;""", """ u32 nprim; + s32 nparts; + s32 j; + u32 i;""")] + +L['nprim_first_part'] = [(""" s32 nparts; + s32 j; + u32 nprim; + u32 i; + Part *part;""", """ u32 nprim; + s32 nparts; + Part *part; + s32 j; + u32 i;""")] + +L['col_after_lim'] = [] # identity (b1 already has it) + +L['col_first'] = [(""" s32 lim; + u32 colA; + u32 colB;""", """ u32 colA; + u32 colB; + s32 lim;""")] + +L['col_last'] = [(""" s32 lim; + u32 colA; + u32 colB; + s32 nparts;""", """ s32 lim; + s32 nparts;"""), + (""" s16 my, mny, mx, mn;""", """ s16 my, mny, mx, mn; + u32 colA; + u32 colB;""")] + +L['de_last'] = [(""" s16 my, mny, mx, mn; + s32 d; + u32 e;""", """ s16 my, mny, mx, mn;"""), + (""" s32 lim; + u32 colA;""", """ s32 lim; + s32 d; + u32 e; + u32 colA;""")] + +# --- prologue statement-order levers ------------------------------------ +L['ot_last'] = [(""" pkt = D_800A5E60; + ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14]; + d = D_801DCCA0;""", """ pkt = D_800A5E60; + d = D_801DCCA0;"""), + (""" colB = (e << 16) | (e << 8) | e; + part =""", """ colB = (e << 16) | (e << 8) | e; + ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14]; + part =""")] + +L['ot_first'] = [(""" pkt = D_800A5E60; + ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14];""", + """ ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14]; + pkt = D_800A5E60;""")] + +L['ot_hoist'] = [(""" pkt = D_800A5E60; + ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14];""", """ pkt = D_800A5E60;"""), + (""" lim = func_800491EC()""", """ ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14]; + lim = func_800491EC()""")] + +# --- prim/nprim read order ---------------------------------------------- +L['prim_first'] = [(""" prim = (Prim *)part->prim; + nprim = part->nprim;""", """ nprim = part->nprim; + prim = (Prim *)part->prim;""")] + +# --- misc --------------------------------------------------------------- +L['e_shift'] = [(" e = d * 2;", " e = d << 1;")] +L['e_signed'] = [(""" u32 e;""", """ s32 e;"""), + (" if (e > 0xFF) e = 0xFF;", " if ((u32)e > 0xFF) e = 0xFF;")] + + +def main(): + base, out = sys.argv[1], sys.argv[2] + src = open(base).read() + for name in sys.argv[3:]: + if name not in L: + raise SystemExit("unknown lever: %s" % name) + for old, new in L[name]: + n = src.count(old) + if n != 1: + raise SystemExit("lever %s: pattern count %d (want 1):\n%s" % (name, n, old[:200])) + src = src.replace(old, new, 1) + open(out, 'w').write(src) + print("wrote %s (levers: %s)" % (out, ' '.join(sys.argv[3:]) or 'none')) + + +main() diff --git a/.run/giants/c954_probe.sh b/.run/giants/c954_probe.sh new file mode 100644 index 000000000..2189660c0 --- /dev/null +++ b/.run/giants/c954_probe.sh @@ -0,0 +1,8 @@ +#!/bin/bash +# c954_probe.sh [tag] -> prints "TAG :: " / "TAG :: " +cd /home/musashi/bfm-decomp +SRC="$1"; TAG="${2:-$(basename $1 .c)}" +OUT=$(timeout 900 python3 tools/match_one.py func_8017C954 --c "$SRC" \ + --asm-subdir asm/ov_SC06_029/nonmatchings/ov_SC06_029_jr_8017AE2C 2>&1 | head -2) +echo "$TAG :: $(echo "$OUT" | head -1)" +echo "$TAG :: $(echo "$OUT" | sed -n 2p)" diff --git a/.run/giants/c954_probe2.sh b/.run/giants/c954_probe2.sh new file mode 100644 index 000000000..735752a1b --- /dev/null +++ b/.run/giants/c954_probe2.sh @@ -0,0 +1,32 @@ +#!/bin/bash +# c954_probe2.sh [tag] -- compiles with -da, prints score + the my/mny/mx/mn grants +cd /home/musashi/bfm-decomp +SRC="$1"; TAG="${2:-$(basename "$1" .c)}"; WD=".run/c954/p2/$TAG" +mkdir -p "$WD" +if ! bash .run/giants/c954_cc.sh "$SRC" "$WD" -da >"$WD/cc.log" 2>&1; then + echo "$TAG :: CCFAIL $(grep -m2 -E 'error|undeclared|parse' "$WD/cc.log" | tr '\n' ' ')"; exit 0 +fi +S=$(bash .run/giants/c954_score.sh "$SRC" "$TAG" "$WD" 2>/dev/null) +python3 - "$WD" "$S" <<'PY' +import re, sys +wd = sys.argv[1] +txt = open(wd + '/t.i.greg').read() +lreg = open(wd + '/t.i.lreg').read() +info = {} +for m in re.finditer(r'Register (\d+) used (\d+) times across (\d+) insns', lreg): + info[int(m.group(1))] = (int(m.group(2)), int(m.group(3))) +disp = {} +d = txt.split(';; Register dispositions:')[1].split('\n\n')[0] +for m in re.finditer(r'(\d+) in (\d+)', d): + disp[int(m.group(1))] = int(m.group(2)) +R = ['zero','at','v0','v1','a0','a1','a2','a3','t0','t1','t2','t3','t4','t5','t6','t7', + 's0','s1','s2','s3','s4','s5','s6','s7','t8','t9','k0','k1','gp','sp','fp','ra'] +# the my/mny/mx/mn quad = the four consecutive pseudos with the largest equal ref counts >60 +cand = sorted([p for p, (r, l) in info.items() if r > 60 and l > 150]) +out = [] +for p in cand[:8]: + mm = re.search(r'^;; %d conflicts:(.*)$' % p, txt, re.M) + hs = [int(x) for x in mm.group(1).split()] if mm else [] + out.append('%d=%s%s' % (p, R[disp[p]] if p in disp else 'SPILL', '' if 3 in [n for n in hs if n < 32] else '!')) +print('%s quad: %s' % (sys.argv[2].strip(), ' '.join(out))) +PY diff --git a/.run/giants/c954_reg.py b/.run/giants/c954_reg.py new file mode 100644 index 000000000..43d2080bd --- /dev/null +++ b/.run/giants/c954_reg.py @@ -0,0 +1,83 @@ +#!/usr/bin/env python3 +"""Per-region aligned-identical report (registers KEPT and register-MASKED). +usage: c954_reg.py [tag] +Regions are the target's five switch arms + prologue/dispatch/epilogue.""" +import re, subprocess, difflib, sys + +OBJ = sys.argv[1] +TAG = sys.argv[2] if len(sys.argv) > 2 else OBJ +TGT = '/home/musashi/bfm-decomp/asm/ov_SC06_029/nonmatchings/ov_SC06_029_jr_8017AE2C/func_8017C954.s' +REGIONS = [("head", 0, 326), ("dispatch", 326, 373), ("A:F3", 373, 485), ("B:FT3", 485, 608), + ("C:F4", 608, 761), ("D:FT4", 761, 942), ("E:new", 942, 1162), ("tail", 1162, 1194)] + +out = subprocess.run(["mipsel-linux-gnu-objdump", "-drz", OBJ], capture_output=True, text=True).stdout + + +def words_mine(): + w = [] + for line in out.splitlines(): + m = re.match(r'\s*[0-9a-f]+:\s+([0-9a-f]{8})\s', line) + if m: + w.append(int(m.group(1), 16)) + elif 'R_MIPS_' in line and w and not isinstance(w[-1], tuple) and (w[-1] >> 26) not in (2, 3): + w[-1] = ('R', w[-1]) + return w + + +def words_tgt(): + w = [] + for line in open(TGT): + m = re.match(r'\s*/\* \w+ [0-9A-F]{8} ([0-9A-F]{8}) \*/', line) + if m: + v = int.from_bytes(bytes.fromhex(m.group(1)), 'little') + w.append(('R', v) if ('%hi(' in line or '%lo(' in line) else v) + return w + + +def mask(x): + rel = isinstance(x, tuple); v = x[1] if rel else x + op = (v >> 26) & 0x3F + if op == 0 or op == 0x1C: + k = (op << 6) | (v & 0x3F) | (((v >> 6) & 0x1F) << 12) + elif op in (2, 3): + k = (op << 26) + elif op in (0x12,): + k = v & 0xFC1F07FF + elif op in (1, 4, 5, 6, 7): + k = (op << 20) | (((v >> 16) & 0x1F) << 8) + else: + k = (op << 20) | (v & 0xFFFF if not rel else 0) + return (k, 'R' if rel else '') + + +def full(x): + rel = isinstance(x, tuple); v = x[1] if rel else x + op = (v >> 26) & 0x3F + if rel: + return (v & 0xFFFF0000, 'R') + if op in (2, 3): + return (op << 26, '') + if op in (1, 4, 5, 6, 7): + return (v & 0xFFFF0000, '') + return (v, '') + + +wm, wt = words_mine(), words_tgt() +res = {} +for name, fn in (("masked", mask), ("kept", full)): + a = [fn(x) for x in wm]; b = [fn(x) for x in wt] + ok = [False] * len(b) + for t, i1, i2, j1, j2 in difflib.SequenceMatcher(None, a, b, autojunk=False).get_opcodes(): + if t == 'equal': + for j in range(j1, j2): + ok[j] = True + res[name] = ok +print("%-22s mine=%d target=%d" % (TAG, len(wm), len(wt))) +for nm, lo, hi in REGIONS: + n = hi - lo + mk = sum(res['masked'][lo:hi]); kp = sum(res['kept'][lo:hi]) + print(" %-9s %4d ins masked %3d/%3d %5.1f%% kept %3d/%3d %5.1f%%" + % (nm, n, mk, n, 100.0 * mk / n, kp, n, 100.0 * kp / n)) +print(" %-9s %4d ins masked %3d %5.1f%% kept %3d %5.1f%%" + % ("TOTAL", len(wt), sum(res['masked']), 100.0 * sum(res['masked']) / len(wt), + sum(res['kept']), 100.0 * sum(res['kept']) / len(wt))) diff --git a/.run/giants/c954_regmap.py b/.run/giants/c954_regmap.py new file mode 100644 index 000000000..74525fb13 --- /dev/null +++ b/.run/giants/c954_regmap.py @@ -0,0 +1,43 @@ +"""Register-correspondence census over index-aligned instructions (same length only).""" +import sys,os,collections +sys.path.insert(0,'/home/musashi/bfm-decomp/tools') +import masked_diff, re, subprocess, shutil +OBJ=sys.argv[1] +TGT='/home/musashi/bfm-decomp/asm/ov_SC06_029/nonmatchings/ov_SC06_029_jr_8017AE2C/func_8017C954.s' +mine=masked_diff.insns_from_object(OBJ,'func_8017C954') +tgt=masked_diff.insns_from_s(TGT) +R=['zero','at','v0','v1','a0','a1','a2','a3','t0','t1','t2','t3','t4','t5','t6','t7', + 's0','s1','s2','s3','s4','s5','s6','s7','t8','t9','k0','k1','gp','sp','s8','ra'] +pairs=collections.Counter() +for i,(a,b) in enumerate(zip(mine,tgt)): + aw = a if isinstance(a,int) else None + # use raw words +for i,(a,b) in enumerate(zip(mine,tgt)): + pass +# simpler: raw words via objdump/target parse +def words_t(): + w=[] + for l in open(TGT): + m=re.match(r'\s*/\* \w+ [0-9A-F]{8} ([0-9A-F]{8}) \*/',l) + if m: w.append(int.from_bytes(bytes.fromhex(m.group(1)),'little')) + return w +OD=[c for c in ["mips-linux-gnu-objdump","mipsel-linux-gnu-objdump"] if shutil.which(c)][0] +out=subprocess.run([OD,"-drz",OBJ],capture_output=True,text=True).stdout +def words_m(): + w=[];inside=False + for l in out.splitlines(): + if re.match(r'^[0-9a-f]+ :',l): inside=True;continue + if inside: + if re.match(r'^[0-9a-f]+ <',l) and w: break + m=re.match(r'\s*[0-9a-f]+:\s+([0-9a-f]{8})\s',l) + if m: w.append(int(m.group(1),16)) + return w +T,M=words_t(),words_m() +print('len',len(M),len(T)) +for a,b in zip(M,T): + if a==b: continue + if (a>>26)!=(b>>26): continue + for sh in (21,16,11): + ra=(a>>sh)&0x1F; rb=(b>>sh)&0x1F + if ra!=rb: pairs[(R[ra],R[rb])]+=1 +for (a,b),n in pairs.most_common(20): print(' mine $%-4s -> target $%-4s x%d'%(a,b,n)) diff --git a/.run/giants/c954_score.sh b/.run/giants/c954_score.sh new file mode 100644 index 000000000..1e181fab6 --- /dev/null +++ b/.run/giants/c954_score.sh @@ -0,0 +1,21 @@ +#!/bin/bash +# c954_score.sh [tag] [workdir] +# compiles the draft and prints: TAG nins maskedPct keptPct keptIdent +# (b3_align numbers -- shape-agnostic, valid even when the length drifts) +cd /home/musashi/bfm-decomp +SRC="$1"; TAG="${2:-$(basename "$1" .c)}"; WD="${3:-.run/c954/score/$TAG}" +mkdir -p "$WD" +if ! bash .run/giants/c954_cc.sh "$SRC" "$WD" >"$WD/cc.log" 2>&1; then + echo "$TAG :: CCFAIL $(tail -2 "$WD/cc.log" | tr '\n' ' ')"; exit 0 +fi +python3 .run/giants/b3_align.py "$WD/t.o" \ + asm/ov_SC06_029/nonmatchings/ov_SC06_029_jr_8017AE2C/func_8017C954.s 0 2>/dev/null \ + | python3 -c " +import sys,re +t='$TAG' +lines=sys.stdin.read().splitlines() +nins=re.search(r'mine (\d+)',lines[0]).group(1) +pct=[l for l in lines if 'aligned-identical' in l] +m=[re.search(r'(\d+) / (\d+) target ins = ([\d.]+)%',p) for p in pct] +print('%-28s nins=%s masked=%s%% (%s) kept=%s%% (%s)' % (t,nins,m[0].group(3),m[0].group(1),m[1].group(3),m[1].group(1))) +" diff --git a/.run/giants/c954_side.py b/.run/giants/c954_side.py new file mode 100644 index 000000000..e2a5d8ac1 --- /dev/null +++ b/.run/giants/c954_side.py @@ -0,0 +1,32 @@ +#!/usr/bin/env python3 +"""Side-by-side index-wise listing of a compiled object vs target .s over a range. +usage: c954_side.py [fn]""" +import sys, os, re, subprocess, shutil +sys.path.insert(0, '/home/musashi/bfm-decomp/tools') +import masked_diff + +OBJ = sys.argv[1]; LO = int(sys.argv[2]); HI = int(sys.argv[3]) +FN = sys.argv[4] if len(sys.argv) > 4 else 'func_8017C954' +TGT = '/home/musashi/bfm-decomp/asm/ov_SC06_029/nonmatchings/ov_SC06_029_jr_8017AE2C/%s.s' % FN + +OD = [c for c in ["mips-linux-gnu-objdump", "mipsel-linux-gnu-objdump"] if shutil.which(c)][0] +out = subprocess.run([OD, "-drz", OBJ], capture_output=True, text=True).stdout +mt = [] +inside = False +for line in out.splitlines(): + if re.match(r'^[0-9a-f]+ <%s>:' % FN, line): inside = True; continue + if inside: + if re.match(r'^[0-9a-f]+ <', line): break + m = re.match(r'\s*[0-9a-f]+:\s+[0-9a-f]{8}\s+(.*)', line) + if m: mt.append(re.sub(r'\s+', ' ', m.group(1).strip())) +tt = [] +for line in open(TGT): + m = re.match(r'\s*/\* \w+ [0-9A-F]{8} [0-9A-F]{8} \*/\s+(.*)', line) + if m: tt.append(re.sub(r'\s+', ' ', m.group(1).strip().rstrip())) + +mine = masked_diff.insns_from_object(OBJ, FN) +tgt = masked_diff.insns_from_s(TGT) +diffs = {i for i, a, b in masked_diff.structured_diff(mine, tgt)} +for k in range(LO, min(HI, max(len(mt), len(tt)))): + flag = '*' if k in diffs else ' ' + print('%s %4d | %-38s | %s' % (flag, k, mt[k] if k < len(mt) else '', tt[k] if k < len(tt) else '')) diff --git a/.run/giants/c954_slots.py b/.run/giants/c954_slots.py new file mode 100644 index 000000000..628961bd9 --- /dev/null +++ b/.run/giants/c954_slots.py @@ -0,0 +1,26 @@ +"""Stack-slot census: sp-offset -> access counts, for target and one object, side by side.""" +import re,subprocess,sys,collections +OBJ=sys.argv[1] +TGT='/home/musashi/bfm-decomp/asm/ov_SC06_029/nonmatchings/ov_SC06_029_jr_8017AE2C/func_8017C954.s' +def cens_t(): + c=collections.defaultdict(collections.Counter) + for l in open(TGT): + m=re.match(r'\s*/\* \w+ \w+ \w+ \*/\s+(\S+)\s+\$\w+, (-?0x[0-9A-Fa-f]+)\(\$sp\)',l) + if m: c[int(m.group(2),16)][m.group(1)]+=1 + return c +def cens_m(): + out=subprocess.run(['mipsel-linux-gnu-objdump','-drz',OBJ],capture_output=True,text=True).stdout + c=collections.defaultdict(collections.Counter); inside=False + for l in out.splitlines(): + if re.match(r'^[0-9a-f]+ :',l): inside=True; continue + if inside: + if re.match(r'^[0-9a-f]+ <',l): break + m=re.match(r'\s*[0-9a-f]+:\s+[0-9a-f]{8}\s+(\S+)\s+\S+,(-?\d+)\(sp\)',l) + if m: c[int(m.group(2))][m.group(1)]+=1 + return c +T,M=cens_t(),cens_m() +def fmt(cc): return ' '.join('%s:%d'%(k,v) for k,v in sorted(cc.items())) +print('target slots=%d mine slots=%d'%(len(T),len(M))) +ks=sorted(set(T)|set(M)) +for k in ks: + print('0x%03X T[%-28s] M[%s]'%(k,fmt(T[k]),fmt(M[k]))) diff --git a/.run/giants/c954_sw.sh b/.run/giants/c954_sw.sh new file mode 100644 index 000000000..e5856a7dd --- /dev/null +++ b/.run/giants/c954_sw.sh @@ -0,0 +1,20 @@ +#!/bin/bash +# c954_sw.sh [lever-spec ...] +# each lever-spec = comma-separated lever names (or "-" for the bare base) +# 8-way parallel; prints the b3_align score for each. +cd /home/musashi/bfm-decomp +BASE="$1"; shift +D=.run/c954/sweep +mkdir -p "$D" +run_one() { + local spec="$1" base="$2" d="$3" + local tag=$(echo "$spec" | tr ',' '+') + local f="$d/v_$tag.c" + if [ "$spec" = "-" ]; then cp "$base" "$f"; else + python3 .run/giants/c954_mk.py "$base" "$f" $(echo "$spec" | tr ',' ' ') >/dev/null 2>"$d/e_$tag.txt" \ + || { echo "$tag :: MKFAIL $(head -3 "$d/e_$tag.txt"|tr '\n' ' ')"; return; } + fi + bash .run/giants/c954_score.sh "$f" "$tag" "$d/w_$tag" +} +export -f run_one +printf '%s\n' "$@" | xargs -P 8 -I{} bash -c 'run_one "$@"' _ {} "$BASE" "$D" diff --git a/.run/giants/c954_sweep.sh b/.run/giants/c954_sweep.sh new file mode 100644 index 000000000..28f6286bb --- /dev/null +++ b/.run/giants/c954_sweep.sh @@ -0,0 +1,19 @@ +#!/bin/bash +# c954_sweep.sh [lever-spec ...] +# each lever-spec = comma-separated lever names, e.g. "nprim_first,ot_last" +# runs all specs 8-way parallel, prints "spec :: line" +cd /home/musashi/bfm-decomp +BASE="$1"; shift +D=.run/c954/sweep +mkdir -p "$D" +run_one() { + local spec="$1" base="$2" d="$3" + local tag=$(echo "$spec" | tr ',' '+') + local f="$d/v_$tag.c" + python3 .run/giants/c954_mk.py "$base" "$f" $(echo "$spec" | tr ',' ' ') >/dev/null 2>"$d/e_$tag.txt" || { echo "$tag :: MKFAIL $(head -3 $d/e_$tag.txt|tr '\n' ' ')"; return; } + local out=$(timeout 900 python3 tools/match_one.py func_8017C954 --c "$f" \ + --asm-subdir asm/ov_SC06_029/nonmatchings/ov_SC06_029_jr_8017AE2C 2>&1 | head -2 | tr '\n' ' ') + echo "$tag :: $out" +} +export -f run_one +printf '%s\n' "$@" | xargs -P 8 -I{} bash -c 'run_one "$@"' _ {} "$BASE" "$D" diff --git a/.run/giants/s19_c954_report.md b/.run/giants/s19_c954_report.md new file mode 100644 index 000000000..e76f9dce9 --- /dev/null +++ b/.run/giants/s19_c954_report.md @@ -0,0 +1,300 @@ +# func_8017C954 (behemoth #5, 1,194 ins, `ov_SC06_029`) — session 20 report + +## Verdict + +**CRACKED. `match_one` = MATCH (1194 ins).** + +``` +python3 tools/match_one.py func_8017C954 \ + --c .run/giants/s19_func_8017C954_b1.c \ + --asm-subdir asm/ov_SC06_029/nonmatchings/ov_SC06_029_jr_8017AE2C +-> MATCH (1194 ins) func_8017C954 +``` + +Independent re-verification with a private pipeline (`.run/giants/c954_cc.sh` + +`c954_full.py`, same pinned triple, separate work dir): + +``` +mine=1194 target=1194 0 mismatched +b3_align: register-MASKED 1194/1194 = 100.0% register-KEPT 1194/1194 = 100.0% +``` + +`symcheck` reports `SYMS-DIFF … MISSING jtbl_801DB70C` — that is the **name** of the +compiler-emitted `.rodata` jump table, not a code difference. The table my compile +emits is byte-identical in content: + +| index | my `.rodata` offset | target `jtbl_801DB70C` | arm | +|---|---|---|---| +| 0,1 | 0x980 | 0x8017D2D4 − 0x8017C954 = 0x980 | C (POLY_F4) | +| 2 | 0xBE4 | 0x8017D538 − base = 0xBE4 | D (POLY_FT4) | +| 3 | 0xEB8 | 0x8017D80C − base = 0xEB8 | **E (new)** | +| 4,5 | 0x5D4 | 0x8017CF28 − base = 0x5D4 | A (POLY_F3) | +| 6,7 | 0x794 | 0x8017D0E8 − base = 0x794 | B (POLY_FT3) | + +`match_one` is the **candidate** gate. The whole-binary SHA1 rebuild (G3/P9) is the +arbiter and has **not** been run — the task forbade touching `src/`, `config/` or the +build tree. Two integration items for the banker: the `jtbl_801DB70C` rodata carve, +and `extern s32 D_801DCCA0;` (already declared in the TU at +`src/ov_SC06_029/ov_SC06_029_jr_8017AE2C.c:3302`). + +Deliverable draft: `/home/musashi/bfm-decomp/.run/giants/s19_func_8017C954_b1.c` +(~160-line dossier header, b4/b5/b2 style). + +Starting point: base = the matched 952-ins `func_8017CA80` renamed → 1129 mismatched, +`SIZE-MISMATCH/short`. Progression: 1129 → 1069 (structure decoded, −3 ins) → 37 → 28 → **MATCH**. + +--- + +## The delta vs the matched base `func_8017CA80` (952 ins) + +Identical skeleton — same 3-call prologue, same Part[] outer loop / 8-corner AABB / +otz + screen-bbox reject, same Prim[] inner loop, same OT insert, same +`D_800A5E60 = pkt`. **+242 instructions in exactly two places:** + +### 1. Prologue, +14 ins — two replicated grey colour words + +```c +pkt = D_800A5E60; +ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14]; +d = D_801DCCA0; /* fade / flash level */ +colA = (d << 16) | (d << 8) | d; /* -> spill slot 0x108 */ +e = d * 2; +if (e > 0xFF) e = 0xFF; /* u32 => `sltiu $v0,$a1,0x100` */ +colB = (e << 16) | (e << 8) | e; /* -> spill slot 0x110 */ +part = …; nparts = …; vtx = …; /* AFTER the clamp branch */ +``` + +`e` must be **unsigned** (the target compares with `sltiu`, not `slti`). The split of +this block around the clamp `if` is directly readable from the target: sched.c cannot +move insns across a basic-block boundary, so everything emitted before +`bnez $v0,.L8017CA00` is in source before the `if`, and `part`/`nparts`/`vtx` — emitted +after — are in source after it. First guess was right, first try. + +### 2. The switch gains a **fifth arm**, +228 ins + +`case 2` and `case 3` split (proved against the real `jtbl_801DB70C` in +`asm/ov_SC06_029/data/tail21.data.s`, table above). `case 2` keeps the base's FT4 body +verbatim — including the now-dead `if (code == 3) g.opz = za + 0x200;`, i.e. the +original author copy-pasted. `case 3` = **arm E**, which emits **two** packets per prim: + +* **P1** — POLY_FT4 (0x28, OT tag `0x9000000`): same geometry and UVs as arm D, but + `rgbc = colA | 0x2E000000` (GPU 0x2E = textured quad, semi-transparent) instead of + `tp[0]` — so `tp[0]` is never read at all. +* **P2** — a 7-word (0x1C, OT tag `0x6000000`) overlay: + `+0x00` tag, `+0x04` `0xE1000040` (GPU **E1 draw-mode**, semi-transparency mode 2 = + `B − F`, i.e. subtractive), `+0x08` `colB | 0x2A000000` (GPU 0x2A = flat quad, + semi-transparent), `+0x0C…+0x18` the four screen-xy words. + `pkt[3] = 6;` writes the `P_TAG` len byte by hand before the tag word overwrites it + (a PsyQ `setlen()` habit). + +Arm E also differs from arm D in how it loads the 4th vertex: + +```c +vv[3] = *(SVECTOR2 *)vd; /* SVECTOR2 align = 2 => mips.c output_block_move */ +gte_ldv0(&vv[3]); /* (align < 4) emits the target's lwl/lwr + swl/swr */ +``` + +and it `gte_stsxy`'s the 4th projected vertex into `tmpxy[3]` (not straight into the +packet) because **both** packets need all four xy words. + +### Frame (0x298) — matches exactly + +``` +0x00 args | 0x10 tmpxy[4] | 0x20 box[8] | 0x60 sxy[8] | 0xA0 vv[4] | +0xC0 mtx | 0xE0 g{otz,flag,opz,sz0..sz3} | 0x100..0x26F 46×8-byte spills | +0x270..0x294 saved regs (s0-s7, fp, ra) +``` + +The §79 frame-slot oracle did the decoding: the *only* sp offsets used in 0xA0..0xBF +are 0xB8/0xBB/0xBC/0xBF, and `mtx` sits at 0xC0 rather than the base's 0xA0 — so a new +**0x20-byte** object is declared between `sxy` and `mtx` and the 8-byte copy lands at +its +0x18. `SVECTOR2 vv[4]` with only `vv[3]` used reproduces it (gcc allocates a whole +array regardless of which elements are referenced). + +Spill order (§79 = pseudo number = declaration order): +`lim 0x100, colA 0x108, colB 0x110, nprim 0x118, nparts 0x120, part 0x128, i 0x260, +(part+0xC giv) 0x268` — note **nprim before nparts**, which is why the draft declares +`u32 nprim;` ahead of `s32 nparts;` unlike the base. + +--- + +## Every lever tried, with its MEASURED result + +The quoted metric for the first three rows is `c954_reg.py` **register-KEPT +aligned-identical %** (length-drift-proof; `match_one`'s index-wise count is useless +while the length is off). Once the length hit 1194 the metric is `match_one`. + +| # | Lever | Measured | +|---|---|---| +| — | base (952-ins `func_8017CA80` renamed) | 1129 mismatched, SIZE-MISMATCH/short | +| — | + prologue colour block + `vv[4]` + case-2/3 split + arm E (**b1**) | 1191 ins, masked 95.9 %, **kept 62.6 %** | +| **L1** | `__asm__ __volatile__ ("" ::: "$3")` after arm E's y min/max | **kept 94.9 %** (masked 98.8 %) | +| **L2** | `u32 nprim;` declared before `s32 nparts;` | +3 structural ins → kept **95.2 %**, masked **99.2 %** | +| **L3** | arm E P1 store order: 4×xy, `tp`, rgbc, 3×uv | kept **95.5 %**, masked **99.4 %** | +| **L4** | **two separately-scoped `otp`s in arm E** (one per OT insert) + first insert reads `za` | 1202 → **1194 ins**, **37 mismatched** | +| **L5** | arm E P2 store order: len, +0x08 colour, +0x0C xy0, +0x04 `0xE1000040`, +0x10/+0x14/+0x18 | **28 mismatched** | +| **L6** | **zero** `__asm__ volatile ("")` live-length sliders | **MATCH** | + +### L1 — the dominant lever, and the reusable finding + +`my`, `mny`, `mx`, `mn` are four `s16` **global** allocnos (pseudos 103..106; refs 103, +live length 245..282). The target grants them `my=$a2 mny=$a3 mx=$t0 mn=$t1` — exactly +what the matched 952-ins base compiles to, and the base's grant is byte-identical to +this function's in all four *unchanged* arms. + +Adding the fifth arm makes `my` and `mny` **lose their hard-register conflict with +`$v1`** and instead acquire a *copy preference* for it. Straight from the `.greg` dump: + +``` +;; 103 conflicts: … 2 12 29 <- $v1 (3) absent +;; 103 preferences: 3 <- and now preferred +``` +vs. the four-arm control, where all four read `… 2 3 12 29` and have no preference. +`my` therefore takes `$v1` and the **whole quad slides one slot down +`reg_alloc_order`** — `v1,a2,a3,t0` instead of `a2,a3,t0,t1`. One slide renames ~40 % +of every switch arm (arms A–D fell from 97–98 % kept to 57–63 %). + +Bisected to the statement: with arm E truncated right after its y min/max the conflict +survives; adding **any** block after it — even +`if (!(g.flag & 0x7F85E000)) { pkt += 4; }` — destroys it. A *small* 5th arm, or a +verbatim duplicate of arm D, does not break it, so it is the size/shape of arm E, not +the arm count. + +Only `"$3"` works. `__asm__ volatile("")`, `"$2"`, `"$4"` and `"memory"` clobbers are +all exact no-ops here (each measured). The dial emits `#APP / / #NO_APP`, i.e. +**zero instructions** — the compile is 1194 EXACT with it and 1182 without it. + +**Generalisable:** when a giant's residual is "a whole block of registers renamed by one +slot", read the `;; N conflicts:` **and** `;; N preferences:` lines of the `.greg` dump +for the highest-priority allocno in the block. A *missing* hard-register conflict plus a +*new* copy preference is the signature of a one-slot slide, and it is one dial away — +not forty bugs. (Extends §76: the allocno **class** lever has a sibling, the allocno +**conflict** lever, and C reaches the latter only through what happens to be live.) + +### L4 — the family's `otp` lever, applied inside a single arm + +One `otp` used by both OT inserts has **2 deaths** → `local-alloc.c:472` +(`REG_BASIC_BLOCK >= 0 && REG_N_DEATHS == 1`) rejects it → global allocno → +`combine_regs` (`local-alloc.c:1825`) cannot tie the `(za>>2)<<2` shift chain into it, +so the chain needs a separate `$v0` scratch. `$v0` is exactly the register the four +`lw/sw` xy pairs use, so the scheduler can no longer put `sra/sll/addu` into their +load-delay slots and **maspsx emits three `nop`s** — those three nops plus the extra +`lw`+`lui` were the entire remaining length drift. Scoping each insert's `otp` in its +own block makes both 1-death/1-block pseudos, the chain ties in place in `$a0`, and the +nops become the target's `sra $a0 / sll $a0 / addu $a0,$a0,$s0`. + +The first insert must read **`za`**, not `g.opz`: `g`'s address is taken by the gte +macros, so `g.opz` would have to be re-loaded after the aliasing `pkt` stores and could +never be hoisted into the delay slots. (Arm D keeps `g.opz` — it has the conditional +`+0x200` — and that is exactly why the target's arm D holds `za` in `$v1` while arms C +and E hold it in `$a0`.) + +### L6 — the §47 slider, with the opposite sign + +The matched 952-ins base ships **one** `__asm__ volatile ("")` to split the +`&g.sz1` / `&g.sz2` allocno tie. Here the correct count is **none**: + +| sliders | result | +|---|---| +| **0** | **MATCH (1194)** | +| 1 | 1193 ins, 1090 mismatched | +| 2 | 1194 ins, 28 mismatched | +| 3 | 1194 ins, 10 mismatched | +| 4 | 1194 ins, 28 mismatched | +| 5 | 1193 ins, 1088 mismatched | + +Same mechanism: `allocno_compare` (`global.c:594`) gives the three `&g.szN` pointers +refs 16 and live lengths within 2 of each other, so `int(4*16*10000/L)` puts them one +apart and *every static instruction added anywhere in the outer loop* re-ranks them. +The final 28 mismatched instructions were exactly this — a 3-cycle on +{`vtx`, `&g.flag`, `0x7F85E000`} (`$s7/$s5/$s6` → `$s6/$s7/$s5`) and a swap on +{`&g.sz1`, `&g.sz2`} (`$s1` ↔ `$t8`) — and they all fell in one edit. + +--- + +## Do-not-re-buy list (byte-measured; each entry is scoped to its base — §80(i)) + +### On the b1 base (1191 ins, kept 62.6 %) + +| Lever | Result | +|---|---| +| **all 24 permutations** of `s16 my, mny, mx, mn` | **all exactly 62.6 % / 95.9 %** — the four are not tied | +| 9 positions for that declaration in the decl list | all exactly neutral | +| `colA/colB` declared first / last / adjacent to `d,e` | 62.4 / 62.5 / 62.6 (neutral-to-worse) | +| `ot` hoisted above the three calls | 61.7 % (worse) | +| `ot` moved after the clamp `if` | −1 ins, much worse | +| `e = d << 1` instead of `d * 2` | exactly neutral | +| `nprim` read before `prim` in the part loop | 62.6 % (neutral) | +| dedicated arm-E min/max vars — function-scope, block-scope, x-only, y-only | 59.6–62.6 % (all neutral or worse) | +| reversed y comparison in arm E | 62.5 % | +| `t32` temp for `tmpxy[2].vx/.vy` in arm E | 55.4 % | +| `gte_ldv0(vd)` instead of the `vv[3]` block copy | 53–57 % and loses the target's lwl/lwr | +| arm E y-block **before** x-block | 75.0 % kept, +7 ins — right REGISTERS, x/y swapped. **The diagnostic** that proved the residual was one allocno slide | +| two separately-scoped `otp` (before L1/L2/L3) | 64.7 % | +| first `otp` from `za` (before L1) | worse | +| `x_e1swap` (P2 colour before E1) | exactly neutral **on this base** — it only pays at 37 | + +### On the clobber base (kept 94.9 %) + +| Lever | Result | +|---|---| +| first `otp` from `za` **alone** | 92.2 % — flips a `$fp` tie in the head (pseudo 78 `j` vs 80 `part`, both refs 8, lengths 985/984 → 984/983) | +| same + one extra zero-byte slider | 94.8 % (tie restored, but still +7 ins) | +| `otp` assignment moved right after `g.opz = za` | identical to leaving it in place (sched normalises) | +| six P1 store orderings (`tp` first / rgbc first / `tp` mid / no `uvw` / rgbc after x2 / x3 last) | 95.0–95.5 % | + +### On the MATCH base + +| Lever | Result | +|---|---| +| remove the `"$3"` dial | 1182 ins, 1081 mismatched | +| replace it with `""`, `"$2"` or `"memory"` | 1182 ins, 1085 mismatched | +| separate arm-E min/max vars (x, y, or both) | 1182 ins, 1081–1084 mismatched | +| arm E y-block before x-block | 1194 ins, 172 mismatched | +| P2 orderings: E1 last / len last / E1+len / colour+2×xy | 92 / 176 / 39 / 37 mismatched | + +**§80(i) confirmed again, twice.** `x_e1swap` measured *exactly neutral* at kept 62.6 % +and was worth −2 mismatched at 37. The `za` lever measured *worse* on the clobber base +and became *necessary* two levers later. Both were on the discard pile. + +--- + +## The one honest caveat + +Lever L1 is a hand-placed **register-clobber dial**, not a construct a 1998 programmer +would have typed. It is byte-free (empty template; `#APP`/`#NO_APP` only) and the +compile is 1194 EXACT, so the delivered object is the target — but it stands in for +whatever the original source did to keep `$v1` busy across arm E's y min/max, and that +source shape has not been found. Fourteen "natural" spellings were tried and measured +(table above). **The single most promising next move** is to find it: bisect further +*inside* the `if (!(g.flag & 0x7F85E000))` block that first destroys the conflict, and +compare the `t.i.lreg` (post-local-alloc) RTL of the four-arm control against the +five-arm draft over arm A's y2 block — the value the target keeps in `$v1` there is a +two-block copy pseudo, and finding which local-alloc decision moves it off `$v1` should +name the missing statement. Until then the dial is a documented, measured stand-in and +the byte result is unaffected. + +--- + +## Artifacts preserved (all under `.run/giants/`) + +| File | What | +|---|---| +| `s19_func_8017C954_b1.c` | **the MATCHing draft** (dossier header) | +| `s19_c954_report.md` | this report | +| `c954_cc.sh` | one-draft compile through the pinned triple into a private work dir, with `-da` RTL dumps | +| `c954_probe.sh` | one-line `match_one` score for a variant (parallel-safe) | +| `c954_score.sh` | length-drift-proof score (b3_align masked % + kept %) | +| `c954_probe2.sh` | score **plus** the my/mny/mx/mn grants and their `$v1` conflict flag | +| `c954_full.py` | uncapped index-wise masked diff (`match_one` caps at 40) | +| `c954_side.py` | side-by-side mine/target listing over an index range, mismatches starred | +| `c954_reg.py` | **per-region** aligned-identical report (head / dispatch / 5 arms / tail) — the tool that made this crack tractable | +| `c954_alloc.py` | allocno census: rank / refs / live length / `allocno_compare` priority / granted hard reg | +| `c954_slots.py` | §79 frame-slot census, target vs draft | +| `c954_regmap.py` | register-correspondence census | +| `c954_mk.py` / `c954_sweep.sh` / `c954_sw.sh` | asserting variant generator + 8-way parallel sweeps | + +Working variants and RTL dumps live in `.run/c954/` (untracked). The two dumps worth +keeping are `.run/c954/w_dump/t.i.{lreg,greg}` (the b1 baseline, where `my`/`mny` lack +the `$v1` conflict) and `.run/c954/w_abl_noEd/t.i.greg` (the four-arm control, where +they have it) — that pair *is* the L1 finding. diff --git a/.run/giants/s19_func_8017C954_b1.c b/.run/giants/s19_func_8017C954_b1.c new file mode 100644 index 000000000..ca34da79e --- /dev/null +++ b/.run/giants/s19_func_8017C954_b1.c @@ -0,0 +1,733 @@ +#include "common.h" +#include "/home/musashi/bfm-decomp/src/shared/engine_types.h" + +/* =========================================================================== + * func_8017C954 -- 1,194 ins, ov_SC06_029 (behemoth #5). *** MATCH *** + * + * STATUS (2026-07-25, session 20, gcc-2.7.2 pinned triple): + * python3 tools/match_one.py func_8017C954 --c \ + * --asm-subdir asm/ov_SC06_029/nonmatchings/ov_SC06_029_jr_8017AE2C + * -> MATCH (1194 ins) + * Independent re-check (private pipeline .run/giants/c954_cc.sh + c954_full.py): + * mine=1194 target=1194 0 mismatched + * b3_align: register-MASKED 1194/1194 = 100.0%, register-KEPT 1194/1194 = 100.0% + * The compiler-generated .rodata jump table is byte-correct too: + * offsets 0x980 0x980 0xBE4 0xEB8 0x5D4 0x5D4 0x794 0x794 + * == target jtbl_801DB70C (symcheck's only diff is the jtbl NAME, which is a + * splat rodata-carve integration item, not a code difference). + * match_one is the CANDIDATE gate; the whole-binary SHA1 rebuild (G3/P9) is the + * arbiter and has NOT been run here (the task forbade touching the build tree). + * ONE register pin-equivalent survives: a single zero-byte + * `__asm__ __volatile__ ("" ::: "$3")` -- see L1 below. It emits + * `#APP / / #NO_APP`, i.e. NO instruction; the compile is 1194 EXACT. + * + * WHAT IT IS: the "constant-colour + subtractive overlay quad" variant of the + * mesh-renderer family. Skeleton is byte-for-byte the MATCHED unlit base + * func_8017CA80 (952 ins, src/ov_SC03_090/ov_SC03_090_jr_8017CA80.c): + * same 3-call prologue (func_800491EC / func_800547D8 / func_80052E38), same + * Part[] outer loop (stride 0x14) with the 8-corner AABB rtpt/rtps + otz + + * screen-bbox reject, same Prim[] inner loop (stride 0xC) with + * rtpt/stflg/nclip/stopz, same OT insert, same `D_800A5E60 = pkt`. + * + * THE TWO DELTAS vs the base (952 -> 1194 = +242 ins): + * + * (1) PROLOGUE, +14 ins: two replicated grey colour words derived from the + * global s32 D_801DCCA0 (a fade/flash level, 0..0x80): + * d = D_801DCCA0; + * colA = (d << 16) | (d << 8) | d; -> stack slot 0x108 + * e = d * 2; if (e > 0xFF) e = 0xFF; (UNSIGNED: `sltiu ,0x100`) + * colB = (e << 16) | (e << 8) | e; -> stack slot 0x110 + * Placement is load-bearing: `pkt`, `ot`, `d`, `colA`, `e` and the clamp + * `if` are all in basic block 1; `part`, `nparts`, `vtx` are read AFTER + * the clamp branch (gcc's sched cannot cross the branch, so the emission + * order reads the source order back directly). + * + * (2) THE SWITCH GAINS A FIFTH ARM, +228 ins. The base maps + * {4,5}->POLY_F3 {6,7}->POLY_FT3 {0,1}->POLY_F4 {2,3}->POLY_FT4. + * Here `case 2` and `case 3` SPLIT (verified against the real + * jtbl_801DB70C in asm/ov_SC06_029/data/tail21.data.s): + * [0][1] -> arm C POLY_F4 (unchanged) + * [2] -> arm D POLY_FT4 (unchanged -- it even KEEPS the now-dead + * `if (code == 3) g.opz = za + 0x200;`) + * [3] -> arm E NEW + * [4][5] -> arm A POLY_F3 (unchanged) + * [6][7] -> arm B POLY_FT3 (unchanged) + * Arm E emits TWO packets per primitive: + * P1 = POLY_FT4 (0x28, OT tag 0x9000000) -- same geometry/UVs as arm D + * but `rgbc = colA | 0x2E000000` (GPU code 0x2E = textured quad, + * semi-transparent) INSTEAD of tp[0], so tp[0] is never read; + * P2 = a 7-word (0x1C, OT tag 0x6000000) overlay: + * +0x00 tag, +0x04 0xE1000040 (GPU E1 draw-mode: semi-transparency + * mode 2 = B-F, subtractive), +0x08 `colB | 0x2A000000` + * (GPU 0x2A = flat quad, semi-transparent), +0x0C..+0x18 the four + * screen xy words. `pkt[3] = 6;` writes the P_TAG len byte by hand + * before the tag word overwrites it (a PsyQ `setlen()` habit). + * Arm E also differs from arm D in HOW it gets the 4th vertex: + * `vv[3] = *(SVECTOR2 *)vd; gte_ldv0(&vv[3]);` + * (SVECTOR2 has 2-byte alignment, so mips.c's `output_block_move` + * (align < 4) emits the lwl/lwr + swl/swr pair the target has), and it + * stsxy's the 4th projected vertex into `tmpxy[3]` rather than straight + * into the packet, because both packets need all four xy words. + * + * FRAME (0x298, measured, matches exactly): + * 0x00 args | 0x10 tmpxy[4] | 0x20 box[8] | 0x60 sxy[8] | 0xA0 vv[4] | + * 0xC0 mtx | 0xE0 g{otz,flag,opz,sz0..sz3} | 0x100..0x26F = 46 EIGHT-BYTE + * spill slots | 0x270..0x294 saved regs (s0-s7, fp, ra). + * The 0xA0 block is a NEW 0x20-byte local declared between `sxy` and `mtx`; + * only vv[3] (0xB8) is ever touched -- an array element, so gcc allocates the + * whole 0x20 (Sec.79 frame-slot oracle: this is what puts mtx at 0xC0 and g at + * 0xE0 instead of the base's 0xA0/0xC0). + * Spill order (Sec.79 = declaration order): lim 0x100, colA 0x108, colB 0x110, + * nprim 0x118, nparts 0x120, part 0x128, i 0x260, (part+0xC giv) 0x268. + * NOTE nprim BEFORE nparts -- that is why `u32 nprim;` is declared ahead of + * `s32 nparts;` here, unlike the base. (Measured: +3 structural ins.) + * + * --------------------------------------------------------------------------- + * THE FIVE LEVERS THAT TOOK 1129 mismatched -> MATCH. Each byte-measured. + * Metric quoted is `c954_reg.py` register-KEPT aligned-identical %, which is + * length-drift-proof (match_one's index-wise count is useless while len != 1194). + * + * L1 ONE ZERO-BYTE $v1 CONFLICT DIAL, first statement after arm E's y min/max: + * __asm__ __volatile__ ("" ::: "$3"); + * 62.6% -> 94.9% kept. THE dominant lever; everything else is small. + * WHY (this is the reusable finding): + * my/mny/mx/mn are four s16 GLOBAL allocnos (pseudos 103..106, refs 103, + * live length 245..282). In the target they are granted + * my=$a2 mny=$a3 mx=$t0 mn=$t1 -- exactly what the MATCHED 952-ins base + * compiles to. Adding the 5th arm makes `my` and `mny` LOSE their hard-reg + * conflict with $v1 and instead acquire a COPY PREFERENCE for it + * (`;; 103 conflicts: ... 2 12 29` / `;; 103 preferences: 3`, greg dump), + * so `my` takes $v1 and the whole quad slides one slot down the + * reg_alloc_order (v1,a2,a3,t0 instead of a2,a3,t0,t1). That single slide + * renames ~40% of every switch arm. + * Bisected to the instruction: with arm E truncated after its y min/max + * the conflict is present; adding ANY block after it (even + * `if (!(g.flag & 0x7F85E000)) pkt += 4;`) removes it. A 5th arm that is + * *small* (or a verbatim duplicate of arm D) does not break it -- so it is + * the SIZE/shape of arm E, not the arm count. + * Only "$3" works: a bare `__asm__ volatile("")`, a "$2"/"$4" clobber or a + * "memory" clobber are all no-ops here (measured). Six "natural" spellings + * were tried and all failed (see the report's do-not-re-buy list), so this + * dial stands in for whatever the original source did to keep $v1 busy. + * + * L2 `u32 nprim;` DECLARED BEFORE `s32 nparts;` -> +3 structural ins, + * 94.9% -> 95.2% kept. Sec.79: spill slots are handed out in pseudo-number + * (= declaration) order, and the target has nprim at 0x118, nparts at 0x120. + * + * L3 ARM E's PACKET-1 STORE ORDER: the four xy words, then `tp`, then rgbc, + * then the three uv words. 95.2% -> 95.5% kept. + * + * L4 TWO SEPARATELY-SCOPED `otp`s IN ARM E (one per OT insert), not one + * function-scope-style `otp` assigned twice. 95.5% -> 97.3% kept AND + * 1202 -> 1194 ins (the whole remaining LENGTH DRIFT). + * This is the family's L1 lever (Sec.76) applied inside one arm: + * `local-alloc.c:472` accepts an allocno only if REG_BASIC_BLOCK >= 0 && + * REG_N_DEATHS == 1. One `otp` with two OT inserts has 2 deaths -> GLOBAL + * allocno -> `combine_regs` (local-alloc.c:1825) cannot tie the + * `(za>>2)<<2` shift chain into it, so the chain needs a separate $v0 scratch + * -- and $v0 is exactly the register the four `lw/sw` xy pairs are using, so + * the scheduler can no longer slot `sra/sll/addu` into their load-delay slots + * and maspsx emits three `nop`s instead. Per-insert scoping makes each a + * 1-death, 1-block pseudo, the chain ties in place in $a0, and the three nops + * turn back into the target's `sra $a0 / sll $a0 / addu $a0,$a0,$s0`. + * (The first insert also has to read `za`, not `g.opz`: `g` has its address + * taken by the gte macros, so `g.opz` would have to be RE-LOADED after the + * aliasing `pkt` stores and could not be hoisted at all.) + * + * L5 ARM E's PACKET-2 STORE ORDER: len byte, +0x08 colour word, +0x0C xy0, + * +0x04 0xE1000040, then +0x10/+0x14/+0x18. 37 -> 28 mismatched. + * The odd interleave is what keeps the two-insn 0xE1000040 constant from + * being materialised early: with the store any earlier, `lui $a1` gets + * scheduled into the packet-1 load-delay slot at tgt[1093] where the target + * has a real `nop` (Sec.78 -- a nop the target has and you lack is a + * register-liveness fact). + * + * L6 ZERO `__asm__ volatile ("")` LIVE-LENGTH SLIDERS. The matched 952-ins base + * ships ONE (Sec.47, to split the &g.sz1 / &g.sz2 allocno tie). Here the + * correct count is NONE: 0 -> MATCH, 2 -> 28, 3 -> 10, 1 and 5 -> length -1. + * Same mechanism, opposite sign: `allocno_compare` (global.c:594) gives the + * three &g.szN pointers refs 16 and lengths within 2 of each other, so + * int(4*16*10000/L) puts them 1 apart and every static instruction added + * anywhere in the outer loop re-ranks them. The last 28 mismatched + * instructions were exactly this: a 3-cycle on {vtx, &g.flag, 0x7F85E000} + * ($s7/$s5/$s6 -> $s6/$s7/$s5) plus a swap on {&g.sz1, &g.sz2} ($s1/$t8). + * + * ===== TRIED AND REJECTED (byte-measured; scoped to the base named) ===== + * On the 1129 base: all 24 permutations of `s16 my, mny, mx, mn` (ALL exactly + * neutral -- the four are not tied); 9 positions for that declaration in the + * decl list (all neutral); colA/colB declared first / last / next to d,e; + * `ot` hoisted before the 3 calls (worse) or moved after the clamp (-1 ins, + * much worse); `e = d << 1`; reading nprim before prim; dedicated arm-E + * min/max variables (function-scope, block-scope, x-only, y-only -- all + * neutral or worse); a reversed y comparison; a `t32` temp for tmpxy[2]; + * `gte_ldv0(vd)` instead of the vv[3] block copy (much worse -- and it loses + * the target's lwl/lwr). Arm E's y-block BEFORE its x-block reaches the right + * REGISTERS (a2,a3,t0,t1) but with x and y swapped, and costs +7 ins: it was + * the diagnostic that proved the residual was one allocno-slide, not 40 bugs. + * On the MATCH base: removing the "$3" dial, or replacing it with "" / "$2" / + * "memory", drops to 1182 ins / ~1085 mismatched. + * =========================================================================== */ + +#define gte_ldv0(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_ldv3(r0, r1, r2) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 0( %1 );" \ + "lwc2 $3, 4( %1 );" \ + "lwc2 $4, 0( %2 );" \ + "lwc2 $5, 4( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) ) + +#define gte_ldv3c(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 8( %0 );" \ + "lwc2 $3, 12( %0 );" \ + "lwc2 $4, 16( %0 );" \ + "lwc2 $5, 20( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_rtps() __asm__ volatile ("nop;nop;rtps") +#define gte_rtpt() __asm__ volatile ("nop;nop;rtpt") +#define gte_nclip() __asm__ volatile ("nop;nop;nclip") + +#define gte_stsxy(r0) __asm__ volatile ( \ + "swc2 $14, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 0( %1 );" \ + "swc2 $14, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsxy3c(r0) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 4( %0 );" \ + "swc2 $14, 8( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_f3(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 12( %0 );" \ + "swc2 $14, 16( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_ft3(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 16( %0 );" \ + "swc2 $14, 24( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_f4(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 12( %0 );" \ + "swc2 $14, 16( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsz3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $17, 0( %0 );" \ + "swc2 $18, 0( %1 );" \ + "swc2 $19, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsz4(r0, r1, r2, r3) __asm__ volatile ( \ + "swc2 $16, 0( %0 );" \ + "swc2 $17, 0( %1 );" \ + "swc2 $18, 0( %2 );" \ + "swc2 $19, 0( %3 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ), "r"( r3 ) \ + : "memory" ) + +#define gte_stszotz(r0) __asm__ volatile ( \ + "mfc2 $12, $19;" \ + "nop;" \ + "sra $12, $12, 2;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stflg(r0) __asm__ volatile ( \ + "cfc2 $12, $31;" \ + "nop;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stopz(r0) __asm__ volatile ( \ + "swc2 $24, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +void func_8017C954(s32 arg0) +{ + typedef struct { u32 w0, w1, w2; } Prim; + + extern s32 func_800491EC(void); + extern void func_800547D8(s32, MATRIX2 *); + extern void func_80052E38(MATRIX2 *); + extern u8 *D_800A5E60; + extern u8 D_800A6610[]; + extern short D_800B9A02; /* TU-visible spelling (engine_core.h + ov_SC01_000.c col-0); unsigned access forced at use — §8d sub-class (b) */ + extern s32 D_801DCCA0; /* fade/flash level driving both overlay colours */ + + DVECTOR2 tmpxy[4]; + SVECTOR2 box[8]; + SVECTOR2 sxy[8]; + SVECTOR2 vv[4]; /* 0xA0..0xBF; only vv[3] is used (arm E vertex copy) */ + MATRIX2 mtx; + struct { long otz, flag, opz, sz0, sz1, sz2, sz3; } g; + + s32 lim; + u32 colA; + u32 colB; + u32 nprim; + s32 nparts; + s32 j; + u32 i; + Part *part; + Prim *prim; + u8 *pkt; + u32 ot; + u8 *vtx; + u8 *va, *vb, *vc, *vd; + u32 w, code; + u32 wx, wy, wz; + s32 xa32, xb32, t32; + s32 xmn1, xmx1, xmn2, xmx2; + s32 mnc, mxc; + s16 my, mny, mx, mn; + s32 d; + u32 e; + + lim = func_800491EC() + *(s32 *)(arg0 + 0x64); + func_800547D8(arg0 + 0x10, &mtx); + func_80052E38(&mtx); + + pkt = D_800A5E60; + ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14]; + d = D_801DCCA0; + colA = (d << 16) | (d << 8) | d; + e = d * 2; + if (e > 0xFF) e = 0xFF; + colB = (e << 16) | (e << 8) | e; + part = *(Part **)(arg0 + 0xC); + nparts = *(s32 *)(*(s32 *)(arg0 + 8) + 8); + vtx = *(u8 **)(*(s32 *)(arg0 + 8) + 0x10); + + for (j = 0; j < nparts; j++, part++) { + wx = part->xx; + mn = wx; + mx = wx >> 16; + wy = part->yy; + mny = wy; + my = wy >> 16; + wz = part->zz; + box[0].vx = mn; box[0].vy = mny; + box[1].vx = mx; box[1].vy = mny; + box[2].vx = mn; box[2].vy = mny; + box[3].vx = mx; box[3].vy = mny; + box[4].vx = mn; box[4].vy = my; + box[5].vx = mx; box[5].vy = my; + box[6].vx = mn; box[6].vy = my; + box[7].vx = mx; box[7].vy = my; + wy = wz >> 16; + box[0].vz = wz; + box[1].vz = wz; + box[4].vz = wz; + box[5].vz = wz; + box[2].vz = wy; + box[3].vz = wy; + box[6].vz = wy; + box[7].vz = wy; + + gte_ldv3c(&box[0]); + gte_rtpt(); + gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]); + gte_ldv0(&box[3]); + gte_rtps(); + gte_stsxy(&sxy[3]); + gte_ldv3c(&box[4]); + gte_rtpt(); + gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]); + gte_ldv0(&box[7]); + gte_rtps(); + gte_stsxy(&sxy[7]); + gte_stszotz(&g.otz); + + if (lim >= g.otz) { + xa32 = sxy[0].vx; + xb32 = sxy[1].vx; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vx; + xb32 = sxy[5].vx; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0xA0 && (s16)mnc < 0xA1) { + xa32 = sxy[0].vy; + xb32 = sxy[1].vy; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vy; + xb32 = sxy[5].vy; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0x78 && (s16)mnc < 0x79) { + prim = (Prim *)part->prim; + nprim = part->nprim; + for (i = 0; i < nprim; i++, prim++) { + w = prim->w1; + va = vtx + (w & 0xFFFF); + vb = vtx + (w >> 16); + w = prim->w2; + vc = vtx + (w & 0xFFFF); + w = w >> 16; + gte_ldv3(va, vb, vc); + gte_rtpt(); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_nclip(); + code = w & 7; + vd = vtx + (w & 0xFFF8); + gte_stopz(&g.opz); + if (g.opz > 0) { + switch (code) { + case 4: + case 5: + gte_stsxy3_f3(pkt); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (((PolyF3 *)pkt)->x0 > ((PolyF3 *)pkt)->x1) { + mx = ((PolyF3 *)pkt)->x0; + mn = ((PolyF3 *)pkt)->x1; + } else { + mn = ((PolyF3 *)pkt)->x0; + mx = ((PolyF3 *)pkt)->x1; + } + if (((PolyF3 *)pkt)->x2 > mx) mx = ((PolyF3 *)pkt)->x2; + else if (((PolyF3 *)pkt)->x2 < mn) mn = ((PolyF3 *)pkt)->x2; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyF3 *)pkt)->y0 > ((PolyF3 *)pkt)->y1) { + my = ((PolyF3 *)pkt)->y0; + mny = ((PolyF3 *)pkt)->y1; + } else { + mny = ((PolyF3 *)pkt)->y0; + my = ((PolyF3 *)pkt)->y1; + } + if (((PolyF3 *)pkt)->y2 > my) my = ((PolyF3 *)pkt)->y2; + else if (((PolyF3 *)pkt)->y2 < mny) mny = ((PolyF3 *)pkt)->y2; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code != 4) g.opz = za + 0x200; + ((PolyF3 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x4000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x14; + } + } + break; + case 6: + case 7: + gte_stsxy3_ft3(pkt); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (((PolyFT3 *)pkt)->x0 > ((PolyFT3 *)pkt)->x1) { + mx = ((PolyFT3 *)pkt)->x0; + mn = ((PolyFT3 *)pkt)->x1; + } else { + mn = ((PolyFT3 *)pkt)->x0; + mx = ((PolyFT3 *)pkt)->x1; + } + if (((PolyFT3 *)pkt)->x2 > mx) mx = ((PolyFT3 *)pkt)->x2; + else if (((PolyFT3 *)pkt)->x2 < mn) mn = ((PolyFT3 *)pkt)->x2; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT3 *)pkt)->y0 > ((PolyFT3 *)pkt)->y1) { + my = ((PolyFT3 *)pkt)->y0; + mny = ((PolyFT3 *)pkt)->y1; + } else { + mny = ((PolyFT3 *)pkt)->y0; + my = ((PolyFT3 *)pkt)->y1; + } + if (((PolyFT3 *)pkt)->y2 > my) my = ((PolyFT3 *)pkt)->y2; + else if (((PolyFT3 *)pkt)->y2 < mny) mny = ((PolyFT3 *)pkt)->y2; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code == 7) g.opz = za + 0x200; + tp = (u32 *)prim->w0; + ((PolyFT3 *)pkt)->rgbc = tp[0]; + ((PolyFT3 *)pkt)->uvc0 = tp[1]; + ((PolyFT3 *)pkt)->uvp1 = tp[2]; + ((PolyFT3 *)pkt)->uv2 = tp[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x20; + } + } + break; + case 0: + case 1: + gte_stsxy3_f4(pkt); + gte_ldv0(vd); + gte_rtps(); + if (((PolyF4 *)pkt)->x0 > ((PolyF4 *)pkt)->x1) { + mx = ((PolyF4 *)pkt)->x0; + mn = ((PolyF4 *)pkt)->x1; + } else { + mn = ((PolyF4 *)pkt)->x0; + mx = ((PolyF4 *)pkt)->x1; + } + if (((PolyF4 *)pkt)->x2 > mx) mx = ((PolyF4 *)pkt)->x2; + else if (((PolyF4 *)pkt)->x2 < mn) mn = ((PolyF4 *)pkt)->x2; + if (((PolyF4 *)pkt)->y0 > ((PolyF4 *)pkt)->y1) { + my = ((PolyF4 *)pkt)->y0; + mny = ((PolyF4 *)pkt)->y1; + } else { + mny = ((PolyF4 *)pkt)->y0; + my = ((PolyF4 *)pkt)->y1; + } + if (((PolyF4 *)pkt)->y2 > my) my = ((PolyF4 *)pkt)->y2; + else if (((PolyF4 *)pkt)->y2 < mny) mny = ((PolyF4 *)pkt)->y2; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyF4 *)pkt)->x3); + if (((PolyF4 *)pkt)->x3 < mn) mn = ((PolyF4 *)pkt)->x3; + else if (mx < ((PolyF4 *)pkt)->x3) mx = ((PolyF4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyF4 *)pkt)->y3 < mny) mny = ((PolyF4 *)pkt)->y3; + else if (my < ((PolyF4 *)pkt)->y3) my = ((PolyF4 *)pkt)->y3; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + ((PolyF4 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((za >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x5000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x18; + } + } + } + break; + case 2: + gte_stsxy3c(&tmpxy[0]); + gte_ldv0(vd); + gte_rtps(); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyFT4 *)pkt)->x3); + if (((PolyFT4 *)pkt)->x3 < mn) mn = ((PolyFT4 *)pkt)->x3; + else if (mx < ((PolyFT4 *)pkt)->x3) mx = ((PolyFT4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT4 *)pkt)->y3 < mny) mny = ((PolyFT4 *)pkt)->y3; + else if (my < ((PolyFT4 *)pkt)->y3) my = ((PolyFT4 *)pkt)->y3; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + u32 uvw; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + if (code == 3) g.opz = za + 0x200; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + tp = (u32 *)prim->w0; + ((PolyFT4 *)pkt)->rgbc = tp[0]; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + } + } + } + break; + case 3: + gte_stsxy3c(&tmpxy[0]); + vv[3] = *(SVECTOR2 *)vd; + gte_ldv0(&vv[3]); + gte_rtps(); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + /* L1: zero-byte $v1 conflict dial -- keeps my/mny off $v1 so the + * my/mny/mx/mn quad lands on $a2/$a3/$t0/$t1 (see header). */ + __asm__ __volatile__ ("" ::: "$3"); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&tmpxy[3]); + if (tmpxy[3].vx < mn) mn = tmpxy[3].vx; + else if (mx < tmpxy[3].vx) mx = tmpxy[3].vx; + if (mx >= -0xA0 && mn < 0xA1) { + if (tmpxy[3].vy < mny) mny = tmpxy[3].vy; + else if (my < tmpxy[3].vy) my = tmpxy[3].vy; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + u32 uvw; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + *(u32 *)&((PolyFT4 *)pkt)->x3 = *(u32 *)&tmpxy[3]; + tp = (u32 *)prim->w0; + ((PolyFT4 *)pkt)->rgbc = colA | 0x2E000000; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + { + u32 *op1 = (u32 *)(((za >> 2) << 2) + ot); + *(u32 *)pkt = (*op1 & 0xFFFFFF) | 0x9000000; + *op1 = (*op1 & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + } + pkt += 0x28; + pkt[3] = 6; + *(u32 *)(pkt + 8) = colB | 0x2A000000; + *(u32 *)(pkt + 0xC) = *(u32 *)&tmpxy[0]; + *(u32 *)(pkt + 4) = 0xE1000040; + *(u32 *)(pkt + 0x10) = *(u32 *)&tmpxy[1]; + *(u32 *)(pkt + 0x14) = *(u32 *)&tmpxy[2]; + *(u32 *)(pkt + 0x18) = *(u32 *)&tmpxy[3]; + { + u32 *op2 = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*op2 & 0xFFFFFF) | 0x6000000; + *op2 = (*op2 & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + } + pkt += 0x1C; + } + } + } + break; + } + } + } + } + } + } + } + } + D_800A5E60 = pkt; +} + diff --git a/phase-ends/CURRENT_PHASE.md b/phase-ends/CURRENT_PHASE.md index c8774f094..461507b52 100644 --- a/phase-ends/CURRENT_PHASE.md +++ b/phase-ends/CURRENT_PHASE.md @@ -3913,6 +3913,44 @@ conditional) · main-EXE/B9 + GLM/B6 + resident's 14 walls (P30) · behemoths B7 99% round-1 result as a stall.** Also found a **5th original-source copy-paste artefact** (QUAD vertex-1 box-3 y-axis accumulates into `a2v` while its kill branch still says `a3v`). +- **✅/⛔ 2026-07-25 (SESSION-19) — `func_8017C954` (1,194 ins) **MATCHED** but **NOT BANKED**: the + blocker is a 3-deep build-infra chain, named exactly, NOT a matching problem.** + Opus 5 @ xHigh cracked it: `match_one` → **MATCH (1194 ins)**, 100% on every region, progression + 1129 → 1069 → 37 → 28 → MATCH. **I verified it independently.** It is the matched base + `func_8017CA80` (952) **+ two deltas**: a 14-ins prologue computing two replicated grey colour words + from `D_801DCCA0`, and a **fifth switch arm** (`case 2`/`case 3` split, proved against the real jump + table) emitting a POLY_FT4 plus a 7-word subtractive overlay. + **⛔ THE WHOLE-BINARY GATE SAID `DIFF` — and it is RIGHT.** This is a **jr (jump-table) function** + (`jr $v0` at .s:409; its table is `jtbl_801DB70C` in `asm/ov_SC06_029/data/tail21.data.s`). Matching + the C makes gcc emit that jtbl into `.rodata` while the raw copy stays in the data tail ⇒ duplicate + + wrong address. **`match_one` masks jal/HI16/LO16, so it CANNOT see this** — exactly the §53 carve law. + **THE CHAIN, each step failing LOUD with its own remedy (good tooling, R32/R35):** + 1. `harvest_verify` → **DIFF** (not PLUMBING — a real byte difference). + 2. `jtbl_carve ov_SC06_029 --func func_8017C954` → refuses: subseg `ov_SC06_029_jr_8017AE2C` would + host **NON-CONTIGUOUS** `.rodata` carves (0xb3468 and 0xb35b4) — one object cannot leave a gap for + the unmatched jtbl between them. Remedy it names: isolate into its own code subseg first. + 3. `jr_isolate_all ov_SC06_029 --dry-run` (47 jr in 21 objects) → **REFUSES**: 2 file-scope decls + (`extern struct PW8017E6D8 D_801E1EC4;` / `…EC8;`) could not be placed, and it will not emit a + region that silently omits them — *"a dropped prototype is a SILENT BYTE-CHANGER"* (in C89 an + undeclared function is implicitly `int f()`, and return type drives delay-slot fill here). + Remedy it names: carry the naming type (`file_scope_types`) or add it to `engine_types.h`. + **NB `struct PW8017E6D8` IS already in `engine_types.h:658`** — so this looks like a placement-logic + gap, not a missing type. That is the precise next thing to check. + **⇒ BANKING IS A BOUNDED BUILD-INFRA TASK (T2 — config resegment ⇒ full R22), NOT more matching.** + Deliberately not started this deep into the session. **The match is preserved and tracked:** + `.run/giants/s19_func_8017C954_b1.c` (~160-line dossier) + `s19_c954_report.md` + the harness + `c954_{cc,probe,probe2,score,sweep,sw}.sh` / `c954_{full,side,reg,alloc,slots,regmap,mk}.py`. + **AGENT FINDINGS WORTH KEEPING:** **§80(i) confirmed twice more** — `x_e1swap` measured *exactly + neutral* then later paid −2; the `za` lever measured *worse* and became necessary two levers later. + **New diagnostic proposed:** when a residual is "a whole block of registers renamed by ONE SLOT", + read the `.greg` `;; N conflicts:` **and** `;; N preferences:` lines for the block's top allocno — a + *missing* hard-reg conflict plus a *new* copy preference is the signature of a one-slot slide, one + dial away rather than forty bugs. The per-region scorer `c954_reg.py` is the reusable tool. + **HONEST CAVEAT FROM THE AGENT:** its lever 1 is a hand-placed byte-free `__asm__` register-clobber + dial, not a construct the original author would have typed; 14 natural spellings were tried and + measured. The bytes are unaffected but the true source shape is unfound — the report names the exact + next probe. + > **🛑 SESSION-19 CLOSING CHECKPOINT (2026-07-25, Opus 5 @ High) — REFRESHED mid-session; supersedes > both the SESSION-18 block and the earlier SESSION-19 block (which was written before the > ENGINE_SHB / class-B / dedup_extend-bug work and went stale). Fresh session safe here.**