mirror of
https://github.com/Druthulu/BFM-decomp
synced 2026-09-26 13:33:34 -04:00
phase-36: S104 s104_e22 — func_8017C954 banked at 0 through the whole-object gate + propagated — barrier → 0: a reused value split into s16 wq (local-alloc.c:472; sched.c:2469-2490) + g.opz in both arms (an integer tie, global.c:594-607)
This commit is contained in:
@@ -0,0 +1,445 @@
|
||||
void func_8017C954(s32 arg0)
|
||||
{
|
||||
typedef struct { u32 w0, w1, w2; } Prim;
|
||||
|
||||
extern s32 func_800491EC(void);
|
||||
extern void func_800547D8(s32, MATRIX2 *);
|
||||
extern void func_80052E38(MATRIX2 *);
|
||||
extern u8 *D_800A5E60;
|
||||
extern u8 D_800A6610[];
|
||||
extern short D_800B9A02; /* TU-visible spelling (engine_core.h + ov_SC01_000.c col-0); unsigned access forced at use — §8d sub-class (b) */
|
||||
extern s32 D_801DCCA0; /* fade/flash level driving both overlay colours */
|
||||
|
||||
DVECTOR2 tmpxy[4];
|
||||
SVECTOR2 box[8];
|
||||
SVECTOR2 sxy[8];
|
||||
SVECTOR2 vv[4]; /* 0xA0..0xBF; only vv[3] is used (arm E vertex copy) */
|
||||
MATRIX2 mtx;
|
||||
struct { long otz, flag, opz, sz0, sz1, sz2, sz3; } g;
|
||||
|
||||
s32 lim;
|
||||
u32 colA;
|
||||
u32 colB;
|
||||
u32 nprim;
|
||||
s32 nparts;
|
||||
s32 j;
|
||||
u32 i;
|
||||
Part *part;
|
||||
Prim *prim;
|
||||
u8 *pkt;
|
||||
u32 ot;
|
||||
u8 *vtx;
|
||||
u8 *va, *vb, *vc, *vd;
|
||||
u32 w, code;
|
||||
u32 wx, wy, wz;
|
||||
s16 wq; /* P36 S104 e22: the box's high z half, its own s16 (see mechanism.md) */
|
||||
s32 xa32, xb32, t32;
|
||||
s32 xmn1, xmx1, xmn2, xmx2;
|
||||
s32 mnc, mxc;
|
||||
s16 my, mny, mx, mn;
|
||||
s32 d;
|
||||
u32 e;
|
||||
|
||||
lim = func_800491EC() + *(s32 *)(arg0 + 0x64);
|
||||
func_800547D8(arg0 + 0x10, &mtx);
|
||||
func_80052E38(&mtx);
|
||||
|
||||
pkt = D_800A5E60;
|
||||
ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14];
|
||||
d = D_801DCCA0;
|
||||
colA = (d << 16) | (d << 8) | d;
|
||||
e = d * 2;
|
||||
if (e > 0xFF) e = 0xFF;
|
||||
colB = (e << 16) | (e << 8) | e;
|
||||
part = *(Part **)(arg0 + 0xC);
|
||||
nparts = *(s32 *)(*(s32 *)(arg0 + 8) + 8);
|
||||
vtx = *(u8 **)(*(s32 *)(arg0 + 8) + 0x10);
|
||||
|
||||
for (j = 0; j < nparts; j++, part++) {
|
||||
wx = part->xx;
|
||||
mn = wx;
|
||||
mx = wx >> 16;
|
||||
wy = part->yy;
|
||||
mny = wy;
|
||||
my = wy >> 16;
|
||||
wz = part->zz;
|
||||
box[0].vx = mn; box[0].vy = mny;
|
||||
box[1].vx = mx; box[1].vy = mny;
|
||||
box[2].vx = mn; box[2].vy = mny;
|
||||
box[3].vx = mx; box[3].vy = mny;
|
||||
box[4].vx = mn; box[4].vy = my;
|
||||
box[5].vx = mx; box[5].vy = my;
|
||||
box[6].vx = mn; box[6].vy = my;
|
||||
box[7].vx = mx; box[7].vy = my;
|
||||
wq = wz >> 16;
|
||||
box[0].vz = wz;
|
||||
box[1].vz = wz;
|
||||
box[4].vz = wz;
|
||||
box[5].vz = wz;
|
||||
box[2].vz = wq;
|
||||
box[3].vz = wq;
|
||||
box[6].vz = wq;
|
||||
box[7].vz = wq;
|
||||
|
||||
gte_ldv3c(&box[0]);
|
||||
gte_rtpt();
|
||||
gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]);
|
||||
gte_ldv0(&box[3]);
|
||||
gte_rtps();
|
||||
gte_stsxy(&sxy[3]);
|
||||
gte_ldv3c(&box[4]);
|
||||
gte_rtpt();
|
||||
gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]);
|
||||
gte_ldv0(&box[7]);
|
||||
gte_rtps();
|
||||
gte_stsxy(&sxy[7]);
|
||||
gte_stszotz(&g.otz);
|
||||
|
||||
if (lim >= g.otz) {
|
||||
xa32 = sxy[0].vx;
|
||||
xb32 = sxy[1].vx;
|
||||
if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; }
|
||||
t32 = sxy[2].vx;
|
||||
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
|
||||
t32 = sxy[3].vx;
|
||||
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
|
||||
xa32 = sxy[4].vx;
|
||||
xb32 = sxy[5].vx;
|
||||
if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; }
|
||||
t32 = sxy[6].vx;
|
||||
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
|
||||
t32 = sxy[7].vx;
|
||||
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
|
||||
mnc = xmn1;
|
||||
if (xmn2 < xmn1) mnc = xmn2;
|
||||
mxc = xmx1;
|
||||
if (mxc < xmx2) mxc = xmx2;
|
||||
if ((s16)mxc >= -0xA0 && (s16)mnc < 0xA1) {
|
||||
xa32 = sxy[0].vy;
|
||||
xb32 = sxy[1].vy;
|
||||
if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; }
|
||||
t32 = sxy[2].vy;
|
||||
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
|
||||
t32 = sxy[3].vy;
|
||||
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
|
||||
xa32 = sxy[4].vy;
|
||||
xb32 = sxy[5].vy;
|
||||
if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; }
|
||||
t32 = sxy[6].vy;
|
||||
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
|
||||
t32 = sxy[7].vy;
|
||||
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
|
||||
mnc = xmn1;
|
||||
if (xmn2 < xmn1) mnc = xmn2;
|
||||
mxc = xmx1;
|
||||
if (mxc < xmx2) mxc = xmx2;
|
||||
if ((s16)mxc >= -0x78 && (s16)mnc < 0x79) {
|
||||
prim = (Prim *)part->prim;
|
||||
nprim = part->nprim;
|
||||
for (i = 0; i < nprim; i++, prim++) {
|
||||
w = prim->w1;
|
||||
va = vtx + (w & 0xFFFF);
|
||||
vb = vtx + (w >> 16);
|
||||
w = prim->w2;
|
||||
vc = vtx + (w & 0xFFFF);
|
||||
w = w >> 16;
|
||||
gte_ldv3(va, vb, vc);
|
||||
gte_rtpt();
|
||||
gte_stflg(&g.flag);
|
||||
if (!(g.flag & 0x7F85E000)) {
|
||||
gte_nclip();
|
||||
code = w & 7;
|
||||
vd = vtx + (w & 0xFFF8);
|
||||
gte_stopz(&g.opz);
|
||||
if (g.opz > 0) {
|
||||
switch (code) {
|
||||
case 4:
|
||||
case 5:
|
||||
gte_stsxy3_f3(pkt);
|
||||
gte_stsz3(&g.sz0, &g.sz1, &g.sz2);
|
||||
if (((PolyF3 *)pkt)->x0 > ((PolyF3 *)pkt)->x1) {
|
||||
mx = ((PolyF3 *)pkt)->x0;
|
||||
mn = ((PolyF3 *)pkt)->x1;
|
||||
} else {
|
||||
mn = ((PolyF3 *)pkt)->x0;
|
||||
mx = ((PolyF3 *)pkt)->x1;
|
||||
}
|
||||
if (((PolyF3 *)pkt)->x2 > mx) mx = ((PolyF3 *)pkt)->x2;
|
||||
else if (((PolyF3 *)pkt)->x2 < mn) mn = ((PolyF3 *)pkt)->x2;
|
||||
if (mx >= -0xA0 && mn < 0xA1) {
|
||||
if (((PolyF3 *)pkt)->y0 > ((PolyF3 *)pkt)->y1) {
|
||||
my = ((PolyF3 *)pkt)->y0;
|
||||
mny = ((PolyF3 *)pkt)->y1;
|
||||
} else {
|
||||
mny = ((PolyF3 *)pkt)->y0;
|
||||
my = ((PolyF3 *)pkt)->y1;
|
||||
}
|
||||
if (((PolyF3 *)pkt)->y2 > my) my = ((PolyF3 *)pkt)->y2;
|
||||
else if (((PolyF3 *)pkt)->y2 < mny) mny = ((PolyF3 *)pkt)->y2;
|
||||
if (my >= -0x78 && mny < 0x79) {
|
||||
s32 za, zb;
|
||||
u32 *otp;
|
||||
if (g.sz0 > g.sz1) {
|
||||
za = g.sz0;
|
||||
if (za < g.sz2) za = g.sz2;
|
||||
} else {
|
||||
za = g.sz1;
|
||||
if (za < g.sz2) za = g.sz2;
|
||||
}
|
||||
g.opz = za;
|
||||
if (code != 4) g.opz = za + 0x200;
|
||||
((PolyF3 *)pkt)->rgbc = prim->w0;
|
||||
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
|
||||
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x4000000;
|
||||
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
|
||||
pkt += 0x14;
|
||||
}
|
||||
}
|
||||
break;
|
||||
case 6:
|
||||
case 7:
|
||||
gte_stsxy3_ft3(pkt);
|
||||
gte_stsz3(&g.sz0, &g.sz1, &g.sz2);
|
||||
if (((PolyFT3 *)pkt)->x0 > ((PolyFT3 *)pkt)->x1) {
|
||||
mx = ((PolyFT3 *)pkt)->x0;
|
||||
mn = ((PolyFT3 *)pkt)->x1;
|
||||
} else {
|
||||
mn = ((PolyFT3 *)pkt)->x0;
|
||||
mx = ((PolyFT3 *)pkt)->x1;
|
||||
}
|
||||
if (((PolyFT3 *)pkt)->x2 > mx) mx = ((PolyFT3 *)pkt)->x2;
|
||||
else if (((PolyFT3 *)pkt)->x2 < mn) mn = ((PolyFT3 *)pkt)->x2;
|
||||
if (mx >= -0xA0 && mn < 0xA1) {
|
||||
if (((PolyFT3 *)pkt)->y0 > ((PolyFT3 *)pkt)->y1) {
|
||||
my = ((PolyFT3 *)pkt)->y0;
|
||||
mny = ((PolyFT3 *)pkt)->y1;
|
||||
} else {
|
||||
mny = ((PolyFT3 *)pkt)->y0;
|
||||
my = ((PolyFT3 *)pkt)->y1;
|
||||
}
|
||||
if (((PolyFT3 *)pkt)->y2 > my) my = ((PolyFT3 *)pkt)->y2;
|
||||
else if (((PolyFT3 *)pkt)->y2 < mny) mny = ((PolyFT3 *)pkt)->y2;
|
||||
if (my >= -0x78 && mny < 0x79) {
|
||||
s32 za, zb;
|
||||
u32 *otp;
|
||||
u32 *tp;
|
||||
if (g.sz0 > g.sz1) {
|
||||
za = g.sz0;
|
||||
if (za < g.sz2) za = g.sz2;
|
||||
g.opz = za;
|
||||
} else {
|
||||
za = g.sz1;
|
||||
if (za < g.sz2) za = g.sz2;
|
||||
g.opz = za;
|
||||
}
|
||||
if (code == 7) g.opz = za + 0x200;
|
||||
tp = (u32 *)prim->w0;
|
||||
((PolyFT3 *)pkt)->rgbc = tp[0];
|
||||
((PolyFT3 *)pkt)->uvc0 = tp[1];
|
||||
((PolyFT3 *)pkt)->uvp1 = tp[2];
|
||||
((PolyFT3 *)pkt)->uv2 = tp[3];
|
||||
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
|
||||
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000;
|
||||
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
|
||||
pkt += 0x20;
|
||||
}
|
||||
}
|
||||
break;
|
||||
case 0:
|
||||
case 1:
|
||||
gte_stsxy3_f3(pkt);
|
||||
gte_ldv0(vd);
|
||||
gte_rtps();
|
||||
if (((PolyF4 *)pkt)->x0 > ((PolyF4 *)pkt)->x1) {
|
||||
mx = ((PolyF4 *)pkt)->x0;
|
||||
mn = ((PolyF4 *)pkt)->x1;
|
||||
} else {
|
||||
mn = ((PolyF4 *)pkt)->x0;
|
||||
mx = ((PolyF4 *)pkt)->x1;
|
||||
}
|
||||
if (((PolyF4 *)pkt)->x2 > mx) mx = ((PolyF4 *)pkt)->x2;
|
||||
else if (((PolyF4 *)pkt)->x2 < mn) mn = ((PolyF4 *)pkt)->x2;
|
||||
if (((PolyF4 *)pkt)->y0 > ((PolyF4 *)pkt)->y1) {
|
||||
my = ((PolyF4 *)pkt)->y0;
|
||||
mny = ((PolyF4 *)pkt)->y1;
|
||||
} else {
|
||||
mny = ((PolyF4 *)pkt)->y0;
|
||||
my = ((PolyF4 *)pkt)->y1;
|
||||
}
|
||||
if (((PolyF4 *)pkt)->y2 > my) my = ((PolyF4 *)pkt)->y2;
|
||||
else if (((PolyF4 *)pkt)->y2 < mny) mny = ((PolyF4 *)pkt)->y2;
|
||||
gte_stflg(&g.flag);
|
||||
if (!(g.flag & 0x7F85E000)) {
|
||||
gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3);
|
||||
gte_stsxy((long *)&((PolyF4 *)pkt)->x3);
|
||||
if (((PolyF4 *)pkt)->x3 < mn) mn = ((PolyF4 *)pkt)->x3;
|
||||
else if (mx < ((PolyF4 *)pkt)->x3) mx = ((PolyF4 *)pkt)->x3;
|
||||
if (mx >= -0xA0 && mn < 0xA1) {
|
||||
if (((PolyF4 *)pkt)->y3 < mny) mny = ((PolyF4 *)pkt)->y3;
|
||||
else if (my < ((PolyF4 *)pkt)->y3) my = ((PolyF4 *)pkt)->y3;
|
||||
if (my >= -0x78 && mny < 0x79) {
|
||||
s32 za, zb;
|
||||
u32 *otp;
|
||||
zb = g.sz2;
|
||||
if (zb < g.sz3) zb = g.sz3;
|
||||
za = g.sz0;
|
||||
if (za < g.sz1) za = g.sz1;
|
||||
if (za < zb) za = zb;
|
||||
g.opz = za;
|
||||
((PolyF4 *)pkt)->rgbc = prim->w0;
|
||||
otp = (u32 *)(((za >> 2) << 2) + ot);
|
||||
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x5000000;
|
||||
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
|
||||
pkt += 0x18;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
case 2:
|
||||
gte_stsxy3c(&tmpxy[0]);
|
||||
gte_ldv0(vd);
|
||||
gte_rtps();
|
||||
if (tmpxy[0].vx > tmpxy[1].vx) {
|
||||
mx = tmpxy[0].vx;
|
||||
mn = tmpxy[1].vx;
|
||||
} else {
|
||||
mn = tmpxy[0].vx;
|
||||
mx = tmpxy[1].vx;
|
||||
}
|
||||
if (tmpxy[2].vx > mx) mx = tmpxy[2].vx;
|
||||
else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx;
|
||||
if (tmpxy[0].vy > tmpxy[1].vy) {
|
||||
my = tmpxy[0].vy;
|
||||
mny = tmpxy[1].vy;
|
||||
} else {
|
||||
mny = tmpxy[0].vy;
|
||||
my = tmpxy[1].vy;
|
||||
}
|
||||
if (tmpxy[2].vy > my) my = tmpxy[2].vy;
|
||||
else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy;
|
||||
gte_stflg(&g.flag);
|
||||
if (!(g.flag & 0x7F85E000)) {
|
||||
gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3);
|
||||
gte_stsxy((long *)&((PolyFT4 *)pkt)->x3);
|
||||
if (((PolyFT4 *)pkt)->x3 < mn) mn = ((PolyFT4 *)pkt)->x3;
|
||||
else if (mx < ((PolyFT4 *)pkt)->x3) mx = ((PolyFT4 *)pkt)->x3;
|
||||
if (mx >= -0xA0 && mn < 0xA1) {
|
||||
if (((PolyFT4 *)pkt)->y3 < mny) mny = ((PolyFT4 *)pkt)->y3;
|
||||
else if (my < ((PolyFT4 *)pkt)->y3) my = ((PolyFT4 *)pkt)->y3;
|
||||
if (my >= -0x78 && mny < 0x79) {
|
||||
s32 za, zb;
|
||||
u32 *otp;
|
||||
u32 *tp;
|
||||
u32 uvw;
|
||||
zb = g.sz2;
|
||||
if (zb < g.sz3) zb = g.sz3;
|
||||
za = g.sz0;
|
||||
if (za < g.sz1) za = g.sz1;
|
||||
if (za < zb) za = zb;
|
||||
g.opz = za;
|
||||
if (code == 3) g.opz = za + 0x200;
|
||||
*(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0];
|
||||
*(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1];
|
||||
*(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2];
|
||||
tp = (u32 *)prim->w0;
|
||||
((PolyFT4 *)pkt)->rgbc = tp[0];
|
||||
((PolyFT4 *)pkt)->uvc0 = tp[1];
|
||||
((PolyFT4 *)pkt)->uvp1 = tp[2];
|
||||
uvw = tp[3];
|
||||
((PolyFT4 *)pkt)->uv2 = uvw;
|
||||
((PolyFT4 *)pkt)->uv3 = uvw >> 16;
|
||||
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
|
||||
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000;
|
||||
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
|
||||
pkt += 0x28;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
case 3:
|
||||
gte_stsxy3c(&tmpxy[0]);
|
||||
vv[3] = *(SVECTOR2 *)vd;
|
||||
gte_ldv0(&vv[3]);
|
||||
gte_rtps();
|
||||
if (tmpxy[0].vx > tmpxy[1].vx) {
|
||||
mx = tmpxy[0].vx;
|
||||
mn = tmpxy[1].vx;
|
||||
} else {
|
||||
mn = tmpxy[0].vx;
|
||||
mx = tmpxy[1].vx;
|
||||
}
|
||||
if (tmpxy[2].vx > mx) mx = tmpxy[2].vx;
|
||||
else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx;
|
||||
if (tmpxy[0].vy > tmpxy[1].vy) {
|
||||
my = tmpxy[0].vy;
|
||||
mny = tmpxy[1].vy;
|
||||
} else {
|
||||
mny = tmpxy[0].vy;
|
||||
my = tmpxy[1].vy;
|
||||
}
|
||||
if (tmpxy[2].vy > my) my = tmpxy[2].vy;
|
||||
else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy;
|
||||
gte_stflg(&g.flag);
|
||||
if (!(g.flag & 0x7F85E000)) {
|
||||
gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3);
|
||||
gte_stsxy((long *)&tmpxy[3]);
|
||||
if (tmpxy[3].vx < mn) mn = tmpxy[3].vx;
|
||||
else if (mx < tmpxy[3].vx) mx = tmpxy[3].vx;
|
||||
if (mx >= -0xA0 && mn < 0xA1) {
|
||||
if (tmpxy[3].vy < mny) mny = tmpxy[3].vy;
|
||||
else if (my < tmpxy[3].vy) my = tmpxy[3].vy;
|
||||
if (my >= -0x78 && mny < 0x79) {
|
||||
s32 za, zb;
|
||||
u32 *otp;
|
||||
u32 *tp;
|
||||
u32 uvw;
|
||||
zb = g.sz2;
|
||||
if (zb < g.sz3) zb = g.sz3;
|
||||
za = g.sz0;
|
||||
if (za < g.sz1) za = g.sz1;
|
||||
if (za < zb) za = zb;
|
||||
g.opz = za;
|
||||
*(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0];
|
||||
*(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1];
|
||||
*(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2];
|
||||
*(u32 *)&((PolyFT4 *)pkt)->x3 = *(u32 *)&tmpxy[3];
|
||||
tp = (u32 *)prim->w0;
|
||||
((PolyFT4 *)pkt)->rgbc = colA | 0x2E000000;
|
||||
((PolyFT4 *)pkt)->uvc0 = tp[1];
|
||||
((PolyFT4 *)pkt)->uvp1 = tp[2];
|
||||
uvw = tp[3];
|
||||
((PolyFT4 *)pkt)->uv2 = uvw;
|
||||
((PolyFT4 *)pkt)->uv3 = uvw >> 16;
|
||||
{
|
||||
u32 *op1 = (u32 *)(((za >> 2) << 2) + ot);
|
||||
*(u32 *)pkt = (*op1 & 0xFFFFFF) | 0x9000000;
|
||||
*op1 = (*op1 & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
|
||||
}
|
||||
pkt += 0x28;
|
||||
pkt[3] = 6;
|
||||
*(u32 *)(pkt + 8) = colB | 0x2A000000;
|
||||
*(u32 *)(pkt + 0xC) = *(u32 *)&tmpxy[0];
|
||||
*(u32 *)(pkt + 4) = 0xE1000040;
|
||||
*(u32 *)(pkt + 0x10) = *(u32 *)&tmpxy[1];
|
||||
*(u32 *)(pkt + 0x14) = *(u32 *)&tmpxy[2];
|
||||
*(u32 *)(pkt + 0x18) = *(u32 *)&tmpxy[3];
|
||||
{
|
||||
u32 *op2 = (u32 *)(((g.opz >> 2) << 2) + ot);
|
||||
*(u32 *)pkt = (*op2 & 0xFFFFFF) | 0x6000000;
|
||||
*op2 = (*op2 & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
|
||||
}
|
||||
pkt += 0x1C;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
D_800A5E60 = pkt;
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
# func_8017C954 (ov_SC06_029) — agent e22, P36 S104
|
||||
|
||||
**Result: score 0, ZERO levers (the `__asm__ __volatile__ ("" ::: "$3")` "L1 dial" is gone). Levers 1 -> 0.**
|
||||
Two plain-C edits to body_free.c (1194 = 1194 ins):
|
||||
1. the box's second use of `wy` (`wy = wz >> 16;`, the high z half) becomes its own local `s16 wq;`;
|
||||
2. in arm B (case 6/7, PolyFT3) `g.opz = za;` is written inside BOTH branches of the za max-of-3 instead of once after
|
||||
the if/else (the spelling the lever-free family members ov_SC03_014 func_8017DF84 :4812 and func_8017FA5C :3410
|
||||
already use in the same FT3 arm).
|
||||
The function's file-level header (L1 dial, L6 "zero sliders") describes the old levers and should be rewritten when
|
||||
this banks (I did not edit it — outside the body).
|
||||
|
||||
## (a) The residual
|
||||
Lever-free: 632 (COUNT, 1182 vs 1194). The header's diagnosis was right about the symptom: the s16 quad
|
||||
my/mny/mx/mn slides one slot down reg_alloc_order (v1,a2,a3,t0 instead of a2,a3,t0,t1), which renames ~40% of every
|
||||
switch arm and changes the spill count (frame 656 vs 664, -12 ins).
|
||||
|
||||
## (b) The passes and decisions (all proven on bytes/dumps)
|
||||
Step 1 — why my/mny take $v1:
|
||||
- `wy` holds TWO values (`part->yy`, later `wz >> 16`) -> dies in 2 places -> refused by local-alloc
|
||||
(`reg_n_deaths == 1`, local-alloc.c:472) -> a GLOBAL allocno. So no LOCAL pseudo in $v1 is live while my/mny are:
|
||||
my/mny have no hard conflict with $v1 (`;; 103 conflicts: ... 2 12 29`) and keep a $v1 preference, which
|
||||
`prune_preferences` (global.c:834-874) only deletes for a conflicting register; my takes $v1 (find_reg's preference
|
||||
pass, global.c:1037-1071) and, via `regs_someone_prefers`, every earlier per-arm HI temp (r277, r360, ...) is pushed
|
||||
from $v1 to $a0 — the cascade (r277 identical conflicts in both compiles, different register: a0 vs v1).
|
||||
- With `wq` its own variable, `wy` is single-value -> local ($v1), and `wq` is local too. Its WIDTH is what matters:
|
||||
as `s16` the shift is `(set (subreg:SI (reg:HI wq) 0) (lshiftrt ...))` — a SUBREG destination, so
|
||||
`birthing_insn_p` (sched.c:2469-2490) fails and sched1 keeps the shift early (right after `my = wy >> 16`), while
|
||||
the `box[].vz = wz` stores still follow: `wz` stays live, `wq` cannot tie to it, local-alloc gives `wq` $v1 for a
|
||||
stretch where all four of my/mny/mx/mn are live -> each gets a HARD conflict with $v1 (`;; 104..107 conflicts:
|
||||
2 3 12 29`) -> the quad lands on a2/a3/t0/t1, exactly the tree's asm clobber, but for a real reason.
|
||||
As `u32 wq` the set is a REG, gets the birthing boost, is scheduled after the wz stores, ties to wz's $v0 and the
|
||||
conflict never appears (gen_box.py: 459; s16/u16: 28).
|
||||
Step 2 — the last 28 (header L6: the 3-cycle {vtx, &g.flag, 0x7F85E000} and the {&g.sz1, &g.sz2} swap):
|
||||
- Every loop-invariant allocno's live length (recomputed by sched1, sched.c:4911-4947) was exactly ONE insn shorter
|
||||
than with the tree's zero-byte asm: &g.sz{1,2,3} 967/968/969 -> `allocno_compare` (global.c:594-607)
|
||||
floor(640000/L) = 661/661/660 — a TIE, broken by allocno number the wrong way (tree: 968/969/970 -> 661/660/659);
|
||||
same for &g.flag vs the constant (390000/L).
|
||||
- A diagnostic `__asm__("")` probe (not delivered) at each of 267 statement positions gave 0 at 138 of them: ANY
|
||||
one zero-byte insn inside the outer loop is the whole residual.
|
||||
- The plain-C source of that insn: `g.opz = za;` duplicated into both branches is TWO stores through allocation
|
||||
(9 `sw ...,232(sp)` in `.greg`) and ONE in the bytes (8 in `.jump2`/`.dbr`/objdump): the post-reload jump pass's
|
||||
cross-jump (`find_cross_jump`, jump.c:2371, run from toplev.c after reload) merges the identical tails, so the bytes
|
||||
are unchanged and every live range through arm B is one insn longer. Duplicating it in arms A AND B is +2 -> 115;
|
||||
either arm alone -> 0; arm B chosen because the family's lever-free members spell arm B that way.
|
||||
(Found by running the R2-R41 recipe families (`tools/delever.py recipe_candidates`, read-only import) over the
|
||||
28-body: 1,655 candidates, exactly two at 0 — R39 dup-join in arm A and in arm B.)
|
||||
|
||||
## (c) The moves
|
||||
1. `s16 wq; ... wq = wz >> 16; box[2/3/6/7].vz = wq;` (632 -> 28)
|
||||
2. arm B: `if (g.sz0 > g.sz1) { za = g.sz0; if (za < g.sz2) za = g.sz2; g.opz = za; } else { ...; g.opz = za; }`
|
||||
(28 -> 0)
|
||||
Both single-step scores are better than the start; neither alone closes.
|
||||
|
||||
## (d) Generator proposals
|
||||
- R-WIDTH-SPLIT: when a local is assigned twice (two unrelated values, "dies in 2 places" in `.lreg`) and one value
|
||||
is a narrowing `x >> 16` / truncation stored to 16-bit fields, split that value into its own local AT THE FIELD
|
||||
WIDTH (s16/u16): the SUBREG destination removes sched1's birthing boost, keeping the definition where the reused
|
||||
variable had it (a u32 split is a different schedule).
|
||||
- R39 dup-join as a LIVE-LENGTH dial: when a register residual is a permutation among long-lived loop invariants whose
|
||||
`alloc_table` priorities TIE (or sit 1 apart), and a zero-byte `asm("")` probe anywhere in the loop closes it, try
|
||||
R39 (duplicate a join-point store into both arms) — cross-jump removes the copy post-reload, so it is +1 live length
|
||||
at zero bytes. Tooling: an `asm("")` position probe is a cheap oracle for "is the residual a live-length off-by-one".
|
||||
|
||||
## (e) Tried and failed (bytes)
|
||||
- per-arm / per-box scoping of my/mny/mx/mn (64 variants, gen_scope.py): best 503.
|
||||
- wq as u32 in 36 placements/store orders/assignment orders (gen_box.py): best 459 (my first, `my = wy >> 16;`).
|
||||
- wq s16/u16: 28 in every placement tried; comparison operand flips on all 84 ifs (gen_flip.py): all 28 (the front
|
||||
end canonicalises them).
|
||||
- dup-join in arms A and B together: 115.
|
||||
- The earlier `$3` dial and the header's "six natural spellings" are superseded; the header's L2-L5 moves are already
|
||||
in body_free and stay.
|
||||
|
||||
## (f) Where the method fell short
|
||||
- The regen sweep (s104_all) was run on the 632 body; its best (581) never met R39 because the dial's residual hid
|
||||
behind the big slide. Re-running the generator families AFTER a structural fix found the finisher in one pass —
|
||||
the sweep should be re-run on every agent's improved body, not only the free one.
|
||||
- alloc_table.py prints preferences only after pruning; the $v1 preference that drove the slide was invisible in the
|
||||
tree's dump (pruned by the clobber's conflict). A "preferences before prune" column would have shown it.
|
||||
- Sibling ov_SC06_000 func_8017EF68 (same family, one launder on `wq`): `s16 wq` there scores 14 = its lever-free
|
||||
score (scratch/sib/) — a different residual (a sched order), not closed by this text.
|
||||
|
||||
Scratch kept (scratch/c/keep/): dupjoin_armB.c (= body.c before the comment clean-up), dupjoin_armA.c (also 0),
|
||||
dupjoin_AB_115.c, wq_s16_28.c, wq_u32_459.c; generators gen_box.py / gen_box2.py / gen_scope.py / gen_flip.py /
|
||||
regen.py; dump helpers quick.sh, annot.py (RTL with each pseudo's local/global hard reg), regmap.py, live.py.
|
||||
|
||||
## (g) Structs
|
||||
Not the channel here: the decisions are local-alloc eligibility (a reused variable), sched1's birthing rule (the
|
||||
declared width of a local), and a post-reload cross-jump. The Part/Prim/POLY_* accesses already go through struct
|
||||
types; turning `g` into a named struct or `tmpxy` into DVECTOR changes nothing these passes look at.
|
||||
@@ -0,0 +1,19 @@
|
||||
void func_80181708(int param_1) {
|
||||
typedef struct { s16 vx, vy, vz, pad; } SV;
|
||||
typedef struct { u8 pad[0x3C]; s32 x3C; s32 x40; s32 x44; s32 x48; s32 x4C; s32 x50; } OB;
|
||||
extern s16 D_801904C6_a[] __asm__("D_801904C6");
|
||||
extern s16 D_801904CE_a[] __asm__("D_801904CE");
|
||||
SV *p1 = (SV *)D_801DFD8C;
|
||||
SV *p2 = (SV *)D_801DFD90;
|
||||
|
||||
((OB *)param_1)->x48 = p1->vx;
|
||||
((OB *)param_1)->x50 = p1->vz;
|
||||
((OB *)param_1)->x3C = p2->vx;
|
||||
((OB *)param_1)->x44 = p2->vz;
|
||||
((OB *)param_1)->x4C += (s16)func_80012CB8(D_801904C6_a[0], p1->vy, 0xC0);
|
||||
((OB *)param_1)->x40 += (s16)func_80012CB8(D_801904CE_a[0], ((SV *)D_801DFD90)->vy, 0xC0);
|
||||
if (((OB *)param_1)->x4C < ((SV *)D_801DFD8C)->vy)
|
||||
((OB *)param_1)->x4C = ((SV *)D_801DFD8C)->vy;
|
||||
if (((OB *)param_1)->x40 < ((SV *)D_801DFD90)->vy)
|
||||
((OB *)param_1)->x40 = ((SV *)D_801DFD90)->vy;
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
# func_80181708 (ov_SC06_029) — agent e22, P36 S104
|
||||
|
||||
**Result: score 0, ZERO levers (the $3 pin and the `__asm__("")` barrier both gone). Levers 2 -> 0.**
|
||||
The body uses two body-local STRUCT typedefs (`SV` = {vx,vy,vz,pad} for the two SVECTOR-pointer globals, `OB` for the
|
||||
object at `param_1`) and two body-local array-typed DECLARATION aliases of the s16 globals
|
||||
(`extern s16 D_801904C6_a[] __asm__("D_801904C6");`, same for `D_801904CE`). No pin, no asm statement, no volatile,
|
||||
no signature change. The file-scope equivalent of the aliases is `extern s16 D_801904C6[];` / `D_801904CE[]` in the TU
|
||||
(a declaration-type change outside the body — the STRUCTS phase can make it and drop the aliases).
|
||||
|
||||
## (a) The residual
|
||||
Lever-free: score 26 (COUNT, 66 vs 65). Two independent defects:
|
||||
1. block 0: `psVar1`/`psVar2` land in v1/a0 where the target has a1/v1; with psVar2 in a0 sched2 cannot hoist its load
|
||||
above `move s0,a0`, so the prologue interleave differs (+1 ins).
|
||||
2. after call 1: sched1 order `sll, sra, lw 76, lw t, addu, sw` vs the target's `sll, lw 76, sra, addu, lw t, sw`, so
|
||||
the reloaded `D_801DFD90` (t) is born while the `sra` result still holds v0 and takes a1 instead of v0.
|
||||
|
||||
## (b) The pass and the decision (read, then proven on bytes)
|
||||
- sched.c:817-839 `true_dependence` / :845-861 `anti_dependence`: a MEM_IN_STRUCT access at a VARYING address never
|
||||
conflicts with a NON-struct access at a FIXED address. expr.c:4568-4577 sets MEM_IN_STRUCT for an INDIRECT_REF of a
|
||||
PLUS_EXPR or an aggregate; `*(s32 *)(param_1 + K)` is a NOP_EXPR of an int sum -> NOT struct; `((OB *)param_1)->xK`
|
||||
-> struct; `D_801DFD8C` (s32 scalar) -> non-struct fixed; `D_801904C6` as `s16` -> non-struct fixed, as `s16[]`
|
||||
element -> struct fixed.
|
||||
- With the stores struct-varying, the loads of the two pointer globals (non-struct fixed) lose every memory edge:
|
||||
sched1 (backward, birthing boost sched.c:2469-2544 since each is set once) places `p2 = D_801DFD90` right before its
|
||||
first use, so p2's quantity is born late: local-alloc `qty_compare_1` (local-alloc.c:1598) ranks p2 (3 refs, short
|
||||
life) ABOVE p1 (4 refs over 22 numbers, 3636) -> p2 takes v1 first; p1 then finds v0 (temps), v1 (p2), a0 (the
|
||||
`D_801904C6` arg load precedes p1's death) busy and takes a1 — the target's assignment. sched2 then hoists both loads
|
||||
above the prologue (no memory or register edge left), exactly the target's first four instructions.
|
||||
- The same independence lets the `D_801DFD90` reload after call 1 and the `D_801DFD8C` reload after call 2 schedule
|
||||
above the `x4C`/`x40` stores (the target's order; the free body only got it by writing the loads first).
|
||||
- The `D_801904C6`/`D_801904CE` argument loads must STAY behind the object stores (target: after `sw 68(s0)` / after
|
||||
`sw 76(s0)`): with struct-varying stores that needs the loads to be struct too -> the array-typed alias. Without the
|
||||
aliases (n5) the arg load floats to just after `move s0,a0` (sched2) and the score is 21; one alias alone 10-11.
|
||||
|
||||
## (c) The moves (all needed; single steps score worse)
|
||||
1. object stores/loads through a struct type: `((OB *)param_1)->x48 = ...` (MEM_IN_STRUCT varying).
|
||||
2. the two pointer globals read directly (`((SV *)D_801DFD90)->vy`) instead of via the `t`/`psVar5` temps, and `SV`
|
||||
field access for the pointed-to vectors (readability; `p1[0]`/`p1[2]` score the same — n4 = 0 too).
|
||||
3. `D_801904C6`/`D_801904CE` read as elements of an array-typed declaration (struct-FIXED), keeping them behind the stores.
|
||||
4. `+=` with the `(s16)` cast inline instead of `sVar4`/`acc` temps (the barrier's job — acc before the t load — falls
|
||||
out of the dependence graph once the loads are free).
|
||||
|
||||
## (d) Generator proposal
|
||||
When a register/ORDER residual involves loads of GLOBALS that the target hoists above stores through a parameter
|
||||
pointer (or leaves behind them), rewrite each access's MEM_IN_STRUCT class to match: object fields through a struct
|
||||
cast (`((T *)p)->f`), global scalars as plain names (non-struct fixed), and globals the target keeps ordered after the
|
||||
stores as array elements (`extern T G_a[] __asm__("G")`, struct fixed) — enumerate the 2^k struct/non-struct choices
|
||||
per access group, it is a small search that no current generator (R2-R41) spans.
|
||||
|
||||
## (e) Tried and failed (bytes)
|
||||
- n1 (inline temps, all `*(s32 *)(p+K)` casts) 28; n2 (struct `OB *o = (OB *)param_1;` local + SV) 28 — the arg loads
|
||||
float to the top; n3 (struct only on x4C) 48; n10 (n11 with `OB *o` as a local) 9 — the separate local costs a
|
||||
copy; n12 (`(*(s16 (*)[1])&D_801904C6)[0]`, no alias) 21 — the ARRAY_REF of a cast ADDR_EXPR is not marked;
|
||||
n6 (`(&D_801904C6)[0]`) 21 — fold removes `+ 0`.
|
||||
- The s104_all regen best was 7 (a marked do-while around one store) and history's best 2 (do-while + inline t).
|
||||
|
||||
## (f) Where the method fell short
|
||||
Step 12-16 name the aggregate channel only for ONE lever class (a table load hoisted over a store). Here it closed a
|
||||
pin AND a barrier: the register residual was a sched1 ORDER consequence (births/deaths feeding qty_compare), and
|
||||
the order was set by memory dependence edges. The allocation table (and localalloc_sim) explained the v1/a0 choice
|
||||
but could not suggest the fix; reading the sched1 ready-list trace for WHY `lw D_801DFD90` only became ready after
|
||||
the first store (an anti-dependence edge) was the step that found it.
|
||||
|
||||
## (g) Structs — YES, this is the structs channel, proven on bytes
|
||||
`param_1` is an object with s32 fields at 0x3C/0x40/0x44/0x48/0x4C/0x50; `D_801DFD8C`/`D_801DFD90` are `SVECTOR *`
|
||||
(set in func_80180F08 to `&D_801904D4[0]` / `[2]` of an s32 array = 8-byte SVECTORs); `D_801904C6`/`D_801904CE` are
|
||||
s16 members of an aggregate (8 bytes apart — plausibly the `pad` of two SVECTORs at D_801904C0/C8, or two fields of one
|
||||
struct). Giving these their struct types is exactly what moves the sched.c dependence decision; with file-scope types
|
||||
the two aliases disappear.
|
||||
@@ -0,0 +1,34 @@
|
||||
void func_80184EFC(s32 a0) {
|
||||
extern s32 D_80190A08;
|
||||
extern u16 D_80190810[];
|
||||
s32 ret;
|
||||
s32 p20;
|
||||
|
||||
ret = func_8012C1B8();
|
||||
*(s32 *)(a0 + 0x20) = ret;
|
||||
if (ret == 0) {
|
||||
func_8012CAE4((void *)a0);
|
||||
return;
|
||||
}
|
||||
|
||||
func_8001C214(ret, 0);
|
||||
|
||||
*(u16 *)(*(s32 *)(a0 + 0x20) + 0x2C) |= 0x10;
|
||||
/* P36 S104 e22: the tag is ONE expression (each `|` a fresh single-set pseudo, so sched1's birthing boost,
|
||||
* sched.c:2469-2544, keeps the chain after the flags store) and the flags update is a plain `|=` — the old
|
||||
* shared `tmp` (a two-death pseudo, refused by local-alloc.c:472) needed a $2 pin; the `s0 = a0` copy needed $16. */
|
||||
*(s32 *)(a0 + 0x58) = (s32)&D_80190A08 | 0x10000000 | 0x20000000 | 0x40000000;
|
||||
if (*(s16 *)(a0 + 0x70) & 0x8000) {
|
||||
*(s16 *)(a0 + 0x34) = 1;
|
||||
}
|
||||
|
||||
p20 = *(s32 *)(*(s32 *)(a0 + 0x64) + 0x20);
|
||||
*(u16 *)(*(s32 *)(a0 + 0x20) + 0x12) =
|
||||
*(u16 *)(p20 + 0x12) + D_80190810[(*(u16 *)(a0 + 0x70) & 0xF) * 6] + 0xE00;
|
||||
func_8012B2CC(a0);
|
||||
|
||||
func_8012B178(a0, (s32)0xFFFC0000);
|
||||
|
||||
*(s32 *)(a0 + 0x1C) = 0x20;
|
||||
*(u16 *)(a0 + 0x2) += 1;
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
# func_80184EFC (ov_SC06_029) — agent e22, P36 S104
|
||||
|
||||
**Result: score 0, ZERO levers (both pins, $16 and $2, gone). Levers 2 -> 0.** Plain C: no asm, no volatile, no alias,
|
||||
no signature change.
|
||||
|
||||
## (a) The residual
|
||||
Lever-free: score 35 (COUNT, 66 vs 63): an extra callee-saved register (`move s1,s0`, 32-byte frame) and a
|
||||
v0/v1/a0 rotation through the flags/tag block (`sub` v0 vs v1, flags temp a0 vs v0, tag accumulator a0 vs v0,
|
||||
0x40000000 v1 vs a0, the `lh 112` test v0 vs v1).
|
||||
|
||||
## (b) The passes and decisions
|
||||
1. `s32 s0; s0 = a0;` — the parameter pseudo and its copy survive as TWO callee-saved pseudos (the copy is live past
|
||||
calls on two paths); deleting the copy and using `a0` (step 15 d24) removes the `$16` pin's job: 35 -> 16 alone.
|
||||
2. `tmp` reused for the flags update AND the tag accumulator dies twice (at the `sh 44` and at the `sw 88`), so
|
||||
local-alloc refuses it (`reg_n_deaths == 1`, local-alloc.c:472) and global-alloc sees v0 already taken by the
|
||||
locals `sub` and the `lh 112` test value -> tmp lands in a0 (localalloc_sim: `sub` q0 -> v0, r83 -> v1, 0 mismatches).
|
||||
The target's assignment (tag in v0 first, `sub` v1, 0x40000000 a0, test value v1) is what local-alloc gives when
|
||||
the tag is a LOCAL quantity of high priority.
|
||||
3. Splitting the temp naively (one `tag` variable, `tag |= C` three times) makes tag single-death but lets sched1 hoist
|
||||
the whole tag chain above the flags load/store (21): the chain's insns have priority 1 (`priority()` sched.c:1425,
|
||||
latency-1 chain), the flags store priority 3, and a 4-set `tag` gets no birthing boost (`birthing_insn_p`
|
||||
sched.c:2469 needs `reg_n_sets == 1`), so the backward scheduler picks the store first = places it last.
|
||||
4. Written as ONE expression, `(s32)&D_80190A08 | 0x10000000 | 0x20000000 | 0x40000000` (fold cannot associate the
|
||||
constants onto a SYMBOL address, so the three `or`s survive), every step is a fresh single-set pseudo: each gets
|
||||
the birthing boost (adjust_priority, sched.c:2507-2544) the moment its consumer is scheduled, so the chain is placed
|
||||
right before the `sw 88` — after the flags store — and the tag quantity is local, ranks first, takes v0.
|
||||
Proven on bytes: a2 (one 4-set tag var) 21 vs a6 (four single-set vars g1..g4) 0 vs a8 (one expression) 0.
|
||||
|
||||
## (c) The moves
|
||||
1. delete `s0 = a0`, use the parameter (`$16` pin gone).
|
||||
2. flags update as `*(u16 *)(*(s32 *)(a0 + 0x20) + 0x2C) |= 0x10;` (no shared temp).
|
||||
3. tag as one expression `(s32)&D_80190A08 | 0x10000000 | 0x20000000 | 0x40000000` (the `$2` pin gone).
|
||||
(cosmetic, all 0: the cast on func_8012C1B8 dropped — the TU's prototype already returns s32; the `val`/`sub2`/`nib`
|
||||
temps folded into one statement with `D_80190810[nib * 6]`; `+= 1` on the u16 counter.)
|
||||
Six other lever-free TUs spell the same tag the same way (e.g. ov_SC06_022 `(s32)&D_80190F8C | 0x10000000 |
|
||||
0x40000000`, ov_SC02_011:686), which corroborates the idiom.
|
||||
|
||||
## (d) Generator proposal
|
||||
When a register residual sits on a temp that is assigned a chain `t = X; t = t | C1; t = t | C2; ...; *M = t;`
|
||||
(or a temp reused for two unrelated chains), rewrite the chain as ONE expression stored directly (`*M = X | C1 | C2`)
|
||||
and any read-modify-write through the temp as `*P |= K;` — the single-set pseudos restore sched1's birthing boost and
|
||||
local-alloc eligibility at once.
|
||||
|
||||
## (e) Tried and failed (bytes)
|
||||
- a1 (only the `s0` copy removed) 16; a2/a3/a4 (shared temp split into flags var + ONE 4-set tag var) 21 — the
|
||||
chain hoists above the flags store (the tree header's warning, confirmed, and why the old author kept one shared
|
||||
pseudo plus a $2 pin); a5 (flags temp as u16 `nib`) 12.
|
||||
- s104_all regen best 11 (R10 param-alias + R6 inline tmp): the generators inline one `tmp` step at a time and never
|
||||
produce the whole chain as one expression.
|
||||
|
||||
## (f) Where the method fell short
|
||||
Nothing in steps 12-16 names "a multi-set temp kills the birthing boost" as a SCHEDULING fact (step 12's d8 covers
|
||||
width, the 80185D44 header covers a 2-set s16); the `.sched` ready list (`(3)` vs `(7f000001)`) made it visible.
|
||||
|
||||
## (g) Structs
|
||||
Not needed here and would not change the closing decision (register-only chain; the flags/tag stores are to
|
||||
different objects). An object struct for `a0` (0x1C/0x20/0x2C/0x34/0x58/0x64/0x70) would read better but is
|
||||
cosmetic; untested on bytes.
|
||||
@@ -0,0 +1,33 @@
|
||||
void func_80185D44(void *a0) {
|
||||
s32 v0;
|
||||
s32 v1;
|
||||
s32 m;
|
||||
s16 t;
|
||||
u8 *p;
|
||||
|
||||
v0 = func_8012C1B8();
|
||||
*(s32 *)((s32)a0 + 0x20) = v0;
|
||||
if (v0 == 0) {
|
||||
func_8012CAE4(a0);
|
||||
return;
|
||||
}
|
||||
func_8001C214(v0, (s32)D_801A0734);
|
||||
m = 0x7FFF0000;
|
||||
__asm__("" : "=r"(m) : "0"(m)); // !FAKE: launder m — cse.c fold_rtx folds `m |= 0xFFFF` to one large_int movsi whose sched1 split halves get ADJACENT luids (sched.c:4830, rank_for_schedule :2428); the target needs luid(lui) < luid(t=0x80) and luid(t=-2) < luid(ori) (P36 S104 e22 minimum-lever)
|
||||
t = 0x80;
|
||||
*(s16 *)((s32)a0 + 0xE0) = t;
|
||||
*(s16 *)((s32)a0 + 0xDE) = t;
|
||||
*(s16 *)((s32)a0 + 0xDC) = t;
|
||||
t = -2;
|
||||
*(s16 *)((s32)a0 + 0xE2) = t;
|
||||
p = (u8 *)D_801DDBF4;
|
||||
v1 = *(s32 *)((s32)a0 + 0x20);
|
||||
*(u16 *)(v1 + 0x2C) |= 0x80;
|
||||
v1 = *(s32 *)((s32)a0 + 0x20);
|
||||
*(s32 *)(v1 + 0x80) = (s32)a0 + 0xDC;
|
||||
*(s16 *)(p + 0x1A) = 0x800;
|
||||
*(s16 *)(p + 0x18) = 0x800;
|
||||
m |= 0xFFFF;
|
||||
*(s32 *)(p + 4) &= m;
|
||||
*(u16 *)((s32)a0 + 2) = *(u16 *)((s32)a0 + 2) + 1;
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
# func_80185D44 (ov_SC06_029) — agent e22, P36 S104
|
||||
|
||||
**Result: score 0 with ONE marked lever (the tree's launder, re-marked in the step-9 format). Levers 1 -> 1.
|
||||
No plain-C spelling found; the lever is argued irreducible below (proven on bytes for every alternative tried).**
|
||||
|
||||
## (a) The residual
|
||||
Lever-free: score 2 (ORDER, 47 = 47 ins). The mask `0x7FFFFFFF` is emitted `lui a1,0x7fff` ... `ori a1,a1,0xffff`;
|
||||
the target puts the `lui` FIRST in the block after `jal func_8001C214` (before `li v0,128`) and the `ori` after
|
||||
`lw v1,32(s0)`; mine puts the `lui` after `li v0,-2`. Same instructions, one moved.
|
||||
|
||||
## (b) The pass and the decision
|
||||
- cse.c `fold_rtx`: `m = 0x7FFF0000; ... m |= 0xFFFF;` folds to ONE `(set m (const_int 0x7fffffff))` at the `|=`'s
|
||||
position (mips.h:2669 `CONST_COSTS` returns 0 for every CONST_INT, so the constant always wins); the first set is dead.
|
||||
Seen in the `.combine` dump: insn 92 `(set (reg/v:SI 75) (const_int 2147483647))` after the `+0x18` store.
|
||||
- sched.c:4830 `try_split` splits that insn by mips.md:3208 (`large_int`) into `lui`+`ori` (uids 112/113) with
|
||||
ADJACENT luids (`sched_analyze` numbers insns in order, sched.c:2170-2175).
|
||||
- sched.c:2428 `rank_for_schedule` ties on `INSN_LUID` (higher luid picked first by the backward scheduler). The target's
|
||||
backward pick order is ... 60, **113 (ori)**, 66, 57, 54, 51, 48, 45, **112 (lui)**: the ori must beat `t = -2` (57) and
|
||||
the `lw v1` (66) on luid, the lui must lose to 45..57 on luid — i.e. luid(lui) < luid(45) < luid(57) < luid(ori), which
|
||||
two adjacent luids cannot satisfy. None of these insns gets the birthing boost (`t` is set twice, the split halves
|
||||
set reg 75 twice; `birthing_insn_p` sched.c:2469 needs `reg_n_sets == 1`), so priority does not break the tie either.
|
||||
=> the original compiled the two halves as TWO RTL insns, which needs cse to NOT know m's value at the `|=`.
|
||||
|
||||
## (c) The move
|
||||
None closes it in plain C. Delivered: the tree body with its one launder, marked
|
||||
`// !FAKE: launder m — ... (P36 S104 e22 minimum-lever)`; `--try` = 0.
|
||||
|
||||
## (d) Generator proposal
|
||||
When the residual is a `lui X,hi` / `ori X,X,lo` pair whose lui sits EARLIER than the lever-free placement and the ori
|
||||
stays put, the source must emit the halves as two insns: try `m = HI<<16; <stmts>; m |= LO;` with the first set placed
|
||||
at the lui's block position (a generator can compute that position from the target objdump); if cse folds it (it
|
||||
always has so far), the case is a launder class for the STRUCTS phase, not a search target.
|
||||
|
||||
## (e) Tried and failed (bytes)
|
||||
- `m = 0x7FFFFFFF;` placed before each of the 7 statement positions (scratch/gen1.py, g1/v0..v6): 4, 6, 2, 2, 2, 2, 2.
|
||||
v0 (at the top) puts the lui first as the target does but drags the ori with it (adjacent luids) — the prediction.
|
||||
- `do { m = 0x7FFF0000; } while (0);` (cse1 stops at NOTE_INSN_LOOP_END, cse.c:8055) -> 2: cse2 (`after_loop`)
|
||||
ignores the note and folds anyway. Same with `t = 0x80` inside the do-while -> 2.
|
||||
- `*(s32 *)(p + 4) &= m | 0xFFFF;` -> 2. The s104_all regen (R2-R41) best is also 2 (177 compiles), and an
|
||||
EXHAUSTIVE pass of every R-family candidate on body_free (244 candidates via `tools/delever.py recipe_candidates`,
|
||||
read-only import, scratch/regen.py) has best 2 too.
|
||||
- m = 0x7FFFFFFF at 7 positions x the a0/p stores as struct vs non-struct accesses (scratch/gen3.py, 28 bodies): best 2
|
||||
— memory-dependence edges do not reach two halves of one split constant.
|
||||
- Every cse-visible spelling of the constant (shift of a narrow const, union field insert, bitfield clear at bit 31)
|
||||
reaches the same `(const_int 0x7fffffff)` by reasoning (store_fixed_bit_field's `mask_rtx` is a CONST_INT); not all
|
||||
compiled.
|
||||
|
||||
## (f) Where the method fell short
|
||||
Nothing in steps 12-16 targets a sched1 SPLIT insn; the luid arithmetic (adjacent split halves) is the whole answer and
|
||||
it needs the `.sched` ready-list trace plus the pre-sched `.combine` order, not the allocation table.
|
||||
Survey: across the 4,284 baseline objects, lui/ori gaps >= 4 are common, but the ones I sampled (func_8017E590,
|
||||
func_80183A70, func_801810B0, func_801439FC) are all reorg filling a branch delay slot with the lui — a different, natural
|
||||
mechanism that does not apply here (no branch after the call).
|
||||
|
||||
## (g) Structs
|
||||
No. The lever keeps a CONSTANT opaque to cse's constant folding; giving `p`/`a0` struct types changes the addressing
|
||||
of the stores (aggregate MEM flags, expr.c:4568-4577) and so possibly their dependence edges in sched, but it cannot
|
||||
stop fold_rtx from folding `0x7FFF0000 | 0xFFFF`, which is what forces the single insn. The split needs the halves to
|
||||
be distinct RTL, and no struct shape produces that.
|
||||
+187
-187
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"head": "aa0564d72",
|
||||
"head": "8733d36da",
|
||||
"stamp": "15956e4a96c4",
|
||||
"generated": "2026-09-11 03:27",
|
||||
"generated": "2026-09-11 03:53",
|
||||
"aliases": [
|
||||
"main",
|
||||
"ov_SC03_014",
|
||||
@@ -21,207 +21,207 @@
|
||||
"main": {
|
||||
"objects": 85,
|
||||
"identical": 85,
|
||||
"seconds": 10.808999999999996,
|
||||
"mean_s": 0.127
|
||||
"seconds": 7.850000000000001,
|
||||
"mean_s": 0.092
|
||||
},
|
||||
"ov_SC03_014": {
|
||||
"objects": 32,
|
||||
"identical": 32,
|
||||
"seconds": 6.97,
|
||||
"mean_s": 0.218
|
||||
"seconds": 5.756,
|
||||
"mean_s": 0.18
|
||||
},
|
||||
"ov_SC03_015": {
|
||||
"objects": 32,
|
||||
"identical": 32,
|
||||
"seconds": 7.423,
|
||||
"mean_s": 0.232
|
||||
"seconds": 5.114,
|
||||
"mean_s": 0.16
|
||||
},
|
||||
"ov_SC04_011": {
|
||||
"objects": 28,
|
||||
"identical": 28,
|
||||
"seconds": 6.887,
|
||||
"mean_s": 0.246
|
||||
"seconds": 4.3870000000000005,
|
||||
"mean_s": 0.157
|
||||
}
|
||||
},
|
||||
"per_object_seconds": {
|
||||
"build/src/800.o": 0.954,
|
||||
"build/src/800_b.o": 0.169,
|
||||
"build/src/800_b_2.o": 0.411,
|
||||
"build/src/800_b_o0a.o": 0.093,
|
||||
"build/src/800_c.o": 0.256,
|
||||
"build/src/800b2.o": 0.146,
|
||||
"build/src/apicard1.o": 0.097,
|
||||
"build/src/apicard2.o": 0.12,
|
||||
"build/src/apicard3.o": 0.147,
|
||||
"build/src/apicard4.o": 0.101,
|
||||
"build/src/apicard5.o": 0.105,
|
||||
"build/src/apicard6.o": 0.143,
|
||||
"build/src/apicard7.o": 0.12,
|
||||
"build/src/boot.o": 0.15,
|
||||
"build/src/800.o": 0.566,
|
||||
"build/src/800_b.o": 0.146,
|
||||
"build/src/800_b_2.o": 0.277,
|
||||
"build/src/800_b_o0a.o": 0.113,
|
||||
"build/src/800_c.o": 0.238,
|
||||
"build/src/800b2.o": 0.135,
|
||||
"build/src/apicard1.o": 0.089,
|
||||
"build/src/apicard2.o": 0.107,
|
||||
"build/src/apicard3.o": 0.081,
|
||||
"build/src/apicard4.o": 0.12,
|
||||
"build/src/apicard5.o": 0.107,
|
||||
"build/src/apicard6.o": 0.102,
|
||||
"build/src/apicard7.o": 0.1,
|
||||
"build/src/boot.o": 0.158,
|
||||
"build/src/gap.o": 0.102,
|
||||
"build/src/libapi1.o": 0.127,
|
||||
"build/src/libapi2.o": 0.073,
|
||||
"build/src/libc2_1.o": 0.087,
|
||||
"build/src/libc2_2.o": 0.084,
|
||||
"build/src/libcd1.o": 0.143,
|
||||
"build/src/libcd2.o": 0.125,
|
||||
"build/src/libetc.o": 0.116,
|
||||
"build/src/libgpu.o": 0.134,
|
||||
"build/src/libgpu2.o": 0.104,
|
||||
"build/src/libgs1.o": 0.13,
|
||||
"build/src/libgs2.o": 0.08,
|
||||
"build/src/libgs3.o": 0.087,
|
||||
"build/src/libgs4.o": 0.099,
|
||||
"build/src/libgs5.o": 0.119,
|
||||
"build/src/libgs6.o": 0.101,
|
||||
"build/src/libgs7.o": 0.087,
|
||||
"build/src/libgs8.o": 0.087,
|
||||
"build/src/libgte1.o": 0.072,
|
||||
"build/src/libgte10.o": 0.121,
|
||||
"build/src/libgte11.o": 0.084,
|
||||
"build/src/libgte12.o": 0.101,
|
||||
"build/src/libgte13.o": 0.134,
|
||||
"build/src/libgte14.o": 0.077,
|
||||
"build/src/libgte15.o": 0.114,
|
||||
"build/src/libgte16.o": 0.118,
|
||||
"build/src/libgte17.o": 0.137,
|
||||
"build/src/libgte18.o": 0.135,
|
||||
"build/src/libgte19.o": 0.118,
|
||||
"build/src/libgte2.o": 0.096,
|
||||
"build/src/libgte20.o": 0.09,
|
||||
"build/src/libgte21.o": 0.112,
|
||||
"build/src/libgte22.o": 0.108,
|
||||
"build/src/libgte23.o": 0.095,
|
||||
"build/src/libgte24.o": 0.103,
|
||||
"build/src/libgte25.o": 0.122,
|
||||
"build/src/libgte26.o": 0.096,
|
||||
"build/src/libgte27.o": 0.124,
|
||||
"build/src/libgte28.o": 0.11,
|
||||
"build/src/libgte29.o": 0.092,
|
||||
"build/src/libgte3.o": 0.121,
|
||||
"build/src/libgte30.o": 0.134,
|
||||
"build/src/libgte4.o": 0.17,
|
||||
"build/src/libgte5.o": 0.086,
|
||||
"build/src/libgte6.o": 0.123,
|
||||
"build/src/libgte7.o": 0.104,
|
||||
"build/src/libgte8.o": 0.112,
|
||||
"build/src/libgte9.o": 0.131,
|
||||
"build/src/libmcrd1.o": 0.109,
|
||||
"build/src/libmcrd2.o": 0.114,
|
||||
"build/src/libpad1.o": 0.131,
|
||||
"build/src/libpad2.o": 0.131,
|
||||
"build/src/sgap.o": 0.093,
|
||||
"build/src/sgap_2.o": 0.089,
|
||||
"build/src/sgap_3.o": 0.123,
|
||||
"build/src/sgap_4.o": 0.132,
|
||||
"build/src/sgap_5.o": 0.115,
|
||||
"build/src/sgap_6.o": 0.078,
|
||||
"build/src/sgap_8.o": 0.107,
|
||||
"build/src/snd1.o": 0.104,
|
||||
"build/src/snd10.o": 0.095,
|
||||
"build/src/snd11.o": 0.121,
|
||||
"build/src/snd12.o": 0.097,
|
||||
"build/src/snd2.o": 0.128,
|
||||
"build/src/snd3.o": 0.094,
|
||||
"build/src/snd4.o": 0.167,
|
||||
"build/src/snd5.o": 0.104,
|
||||
"build/src/snd6.o": 0.171,
|
||||
"build/src/snd7.o": 0.093,
|
||||
"build/src/snd8.o": 0.072,
|
||||
"build/src/snd9.o": 0.104,
|
||||
"build/src/ov_SC03_014/ov_SC03_014.o": 0.209,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_after.o": 0.722,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8012ACE0.o": 0.574,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80135888.o": 0.099,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80135A4C.o": 0.136,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80135D20.o": 0.15,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_801380E0.o": 0.283,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8013C98C.o": 0.19,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8013F350.o": 0.109,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8013FFD8.o": 0.138,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80140608.o": 0.253,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8015444C.o": 0.098,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80154C24.o": 0.222,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_801588CC.o": 0.132,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80159C84.o": 0.101,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8015A3C8.o": 0.11,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8015AE2C.o": 0.117,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8015C32C.o": 0.678,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8016AB6C.o": 0.387,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80171B4C.o": 0.135,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_801734BC.o": 0.268,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_801789AC.o": 0.124,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80178D40.o": 0.142,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8017A4AC.o": 0.118,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8017AE2C.o": 0.256,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8017EB7C.o": 0.276,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80184440.o": 0.087,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_801848E4.o": 0.408,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_o0b.o": 0.074,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_o0c.o": 0.125,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_o0d.o": 0.114,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_o0e.o": 0.135,
|
||||
"build/src/ov_SC03_015/ov_SC03_015.o": 0.244,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_after.o": 0.731,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8012ACE0.o": 0.593,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80135888.o": 0.081,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80135A4C.o": 0.078,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80135D20.o": 0.182,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_801380E0.o": 0.232,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8013C98C.o": 0.173,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8013F350.o": 0.113,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8013FFD8.o": 0.121,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80140608.o": 0.294,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8015444C.o": 0.149,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80154C24.o": 0.291,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_801588CC.o": 0.141,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80159C84.o": 0.121,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8015A3C8.o": 0.111,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8015AE2C.o": 0.169,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8015C32C.o": 0.746,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8016AB6C.o": 0.429,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80171B4C.o": 0.177,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_801734BC.o": 0.315,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_801789AC.o": 0.121,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80178D40.o": 0.163,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8017A4AC.o": 0.154,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8017AE2C.o": 0.291,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8017EB7C.o": 0.278,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80184440.o": 0.118,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_801848E4.o": 0.404,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_o0b.o": 0.092,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_o0c.o": 0.095,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_o0d.o": 0.084,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_o0e.o": 0.132,
|
||||
"build/src/ov_SC04_011/ov_SC04_011.o": 0.196,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_after.o": 0.695,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8012ACE0.o": 0.597,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80135888.o": 0.118,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80135A4C.o": 0.108,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80135D20.o": 0.206,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_801380E0.o": 0.274,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8013C98C.o": 0.193,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8013F350.o": 0.163,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8013FFD8.o": 0.115,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80140608.o": 0.279,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8015444C.o": 0.15,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80154C24.o": 0.288,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_801588CC.o": 0.192,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80159C84.o": 0.118,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8015A3C8.o": 0.122,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8015AE2C.o": 0.156,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8015C32C.o": 0.563,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8016AB6C.o": 0.408,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80171B4C.o": 0.176,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_801734BC.o": 0.275,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_801789AC.o": 0.111,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80178D40.o": 0.179,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8017A4AC.o": 0.124,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8017AE2C.o": 0.176,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8017D494.o": 0.689,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_o0b.o": 0.1,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_o0c.o": 0.116
|
||||
"build/src/libapi1.o": 0.1,
|
||||
"build/src/libapi2.o": 0.066,
|
||||
"build/src/libc2_1.o": 0.083,
|
||||
"build/src/libc2_2.o": 0.081,
|
||||
"build/src/libcd1.o": 0.069,
|
||||
"build/src/libcd2.o": 0.066,
|
||||
"build/src/libetc.o": 0.065,
|
||||
"build/src/libgpu.o": 0.073,
|
||||
"build/src/libgpu2.o": 0.081,
|
||||
"build/src/libgs1.o": 0.066,
|
||||
"build/src/libgs2.o": 0.057,
|
||||
"build/src/libgs3.o": 0.038,
|
||||
"build/src/libgs4.o": 0.04,
|
||||
"build/src/libgs5.o": 0.037,
|
||||
"build/src/libgs6.o": 0.043,
|
||||
"build/src/libgs7.o": 0.049,
|
||||
"build/src/libgs8.o": 0.044,
|
||||
"build/src/libgte1.o": 0.044,
|
||||
"build/src/libgte10.o": 0.047,
|
||||
"build/src/libgte11.o": 0.05,
|
||||
"build/src/libgte12.o": 0.049,
|
||||
"build/src/libgte13.o": 0.047,
|
||||
"build/src/libgte14.o": 0.053,
|
||||
"build/src/libgte15.o": 0.127,
|
||||
"build/src/libgte16.o": 0.123,
|
||||
"build/src/libgte17.o": 0.139,
|
||||
"build/src/libgte18.o": 0.113,
|
||||
"build/src/libgte19.o": 0.094,
|
||||
"build/src/libgte2.o": 0.085,
|
||||
"build/src/libgte20.o": 0.125,
|
||||
"build/src/libgte21.o": 0.135,
|
||||
"build/src/libgte22.o": 0.138,
|
||||
"build/src/libgte23.o": 0.081,
|
||||
"build/src/libgte24.o": 0.1,
|
||||
"build/src/libgte25.o": 0.112,
|
||||
"build/src/libgte26.o": 0.095,
|
||||
"build/src/libgte27.o": 0.12,
|
||||
"build/src/libgte28.o": 0.118,
|
||||
"build/src/libgte29.o": 0.152,
|
||||
"build/src/libgte3.o": 0.076,
|
||||
"build/src/libgte30.o": 0.082,
|
||||
"build/src/libgte4.o": 0.08,
|
||||
"build/src/libgte5.o": 0.065,
|
||||
"build/src/libgte6.o": 0.064,
|
||||
"build/src/libgte7.o": 0.065,
|
||||
"build/src/libgte8.o": 0.072,
|
||||
"build/src/libgte9.o": 0.057,
|
||||
"build/src/libmcrd1.o": 0.067,
|
||||
"build/src/libmcrd2.o": 0.059,
|
||||
"build/src/libpad1.o": 0.053,
|
||||
"build/src/libpad2.o": 0.055,
|
||||
"build/src/sgap.o": 0.045,
|
||||
"build/src/sgap_2.o": 0.048,
|
||||
"build/src/sgap_3.o": 0.048,
|
||||
"build/src/sgap_4.o": 0.047,
|
||||
"build/src/sgap_5.o": 0.049,
|
||||
"build/src/sgap_6.o": 0.041,
|
||||
"build/src/sgap_8.o": 0.042,
|
||||
"build/src/snd1.o": 0.042,
|
||||
"build/src/snd10.o": 0.042,
|
||||
"build/src/snd11.o": 0.039,
|
||||
"build/src/snd12.o": 0.107,
|
||||
"build/src/snd2.o": 0.104,
|
||||
"build/src/snd3.o": 0.092,
|
||||
"build/src/snd4.o": 0.127,
|
||||
"build/src/snd5.o": 0.118,
|
||||
"build/src/snd6.o": 0.085,
|
||||
"build/src/snd7.o": 0.13,
|
||||
"build/src/snd8.o": 0.125,
|
||||
"build/src/snd9.o": 0.078,
|
||||
"build/src/ov_SC03_014/ov_SC03_014.o": 0.196,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_after.o": 0.426,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8012ACE0.o": 0.365,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80135888.o": 0.111,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80135A4C.o": 0.114,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80135D20.o": 0.182,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_801380E0.o": 0.226,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8013C98C.o": 0.134,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8013F350.o": 0.091,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8013FFD8.o": 0.08,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80140608.o": 0.161,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8015444C.o": 0.091,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80154C24.o": 0.149,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_801588CC.o": 0.091,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80159C84.o": 0.075,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8015A3C8.o": 0.063,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8015AE2C.o": 0.061,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8015C32C.o": 0.756,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8016AB6C.o": 0.166,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80171B4C.o": 0.064,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_801734BC.o": 0.296,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_801789AC.o": 0.115,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80178D40.o": 0.136,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8017A4AC.o": 0.121,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8017AE2C.o": 0.217,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_8017EB7C.o": 0.286,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_80184440.o": 0.122,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_jr_801848E4.o": 0.316,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_o0b.o": 0.156,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_o0c.o": 0.104,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_o0d.o": 0.152,
|
||||
"build/src/ov_SC03_014/ov_SC03_014_o0e.o": 0.133,
|
||||
"build/src/ov_SC03_015/ov_SC03_015.o": 0.2,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_after.o": 0.311,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8012ACE0.o": 0.398,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80135888.o": 0.106,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80135A4C.o": 0.091,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80135D20.o": 0.123,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_801380E0.o": 0.152,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8013C98C.o": 0.117,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8013F350.o": 0.098,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8013FFD8.o": 0.079,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80140608.o": 0.173,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8015444C.o": 0.073,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80154C24.o": 0.16,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_801588CC.o": 0.061,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80159C84.o": 0.052,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8015A3C8.o": 0.044,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8015AE2C.o": 0.061,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8015C32C.o": 0.288,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8016AB6C.o": 0.283,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80171B4C.o": 0.167,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_801734BC.o": 0.242,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_801789AC.o": 0.119,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80178D40.o": 0.178,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8017A4AC.o": 0.15,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8017AE2C.o": 0.216,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_8017EB7C.o": 0.255,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_80184440.o": 0.123,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_jr_801848E4.o": 0.271,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_o0b.o": 0.14,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_o0c.o": 0.13,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_o0d.o": 0.117,
|
||||
"build/src/ov_SC03_015/ov_SC03_015_o0e.o": 0.136,
|
||||
"build/src/ov_SC04_011/ov_SC04_011.o": 0.173,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_after.o": 0.413,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8012ACE0.o": 0.3,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80135888.o": 0.064,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80135A4C.o": 0.053,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80135D20.o": 0.104,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_801380E0.o": 0.097,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8013C98C.o": 0.072,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8013F350.o": 0.048,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8013FFD8.o": 0.038,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80140608.o": 0.109,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8015444C.o": 0.044,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80154C24.o": 0.103,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_801588CC.o": 0.055,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80159C84.o": 0.101,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8015A3C8.o": 0.125,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8015AE2C.o": 0.134,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8015C32C.o": 0.419,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8016AB6C.o": 0.266,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80171B4C.o": 0.133,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_801734BC.o": 0.211,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_801789AC.o": 0.129,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_80178D40.o": 0.151,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8017A4AC.o": 0.12,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8017AE2C.o": 0.158,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_jr_8017D494.o": 0.492,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_o0b.o": 0.131,
|
||||
"build/src/ov_SC04_011/ov_SC04_011_o0c.o": 0.144
|
||||
},
|
||||
"ok": true,
|
||||
"seconds": 3.4
|
||||
"seconds": 18.4
|
||||
}
|
||||
|
||||
File diff suppressed because one or more lines are too long
@@ -2895,6 +2895,7 @@ void func_8017C954(s32 arg0)
|
||||
u8 *va, *vb, *vc, *vd;
|
||||
u32 w, code;
|
||||
u32 wx, wy, wz;
|
||||
s16 wq; /* P36 S104 e22: the box's high z half, its own s16 (see mechanism.md) */
|
||||
s32 xa32, xb32, t32;
|
||||
s32 xmn1, xmx1, xmn2, xmx2;
|
||||
s32 mnc, mxc;
|
||||
@@ -2933,15 +2934,15 @@ void func_8017C954(s32 arg0)
|
||||
box[5].vx = mx; box[5].vy = my;
|
||||
box[6].vx = mn; box[6].vy = my;
|
||||
box[7].vx = mx; box[7].vy = my;
|
||||
wy = wz >> 16;
|
||||
wq = wz >> 16;
|
||||
box[0].vz = wz;
|
||||
box[1].vz = wz;
|
||||
box[4].vz = wz;
|
||||
box[5].vz = wz;
|
||||
box[2].vz = wy;
|
||||
box[3].vz = wy;
|
||||
box[6].vz = wy;
|
||||
box[7].vz = wy;
|
||||
box[2].vz = wq;
|
||||
box[3].vz = wq;
|
||||
box[6].vz = wq;
|
||||
box[7].vz = wq;
|
||||
|
||||
gte_ldv3c(&box[0]);
|
||||
gte_rtpt();
|
||||
@@ -3088,11 +3089,12 @@ void func_8017C954(s32 arg0)
|
||||
if (g.sz0 > g.sz1) {
|
||||
za = g.sz0;
|
||||
if (za < g.sz2) za = g.sz2;
|
||||
g.opz = za;
|
||||
} else {
|
||||
za = g.sz1;
|
||||
if (za < g.sz2) za = g.sz2;
|
||||
g.opz = za;
|
||||
}
|
||||
g.opz = za;
|
||||
if (code == 7) g.opz = za + 0x200;
|
||||
tp = (u32 *)prim->w0;
|
||||
((PolyFT3 *)pkt)->rgbc = tp[0];
|
||||
@@ -3240,9 +3242,6 @@ void func_8017C954(s32 arg0)
|
||||
}
|
||||
if (tmpxy[2].vy > my) my = tmpxy[2].vy;
|
||||
else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy;
|
||||
/* L1: zero-byte $v1 conflict dial -- keeps my/mny off $v1 so the
|
||||
* my/mny/mx/mn quad lands on $a2/$a3/$t0/$t1 (see header). */
|
||||
__asm__ __volatile__ ("" ::: "$3"); // !FAKE: barrier — NEEDED DIFFERS (P36 rung B tus10)
|
||||
gte_stflg(&g.flag);
|
||||
if (!(g.flag & 0x7F85E000)) {
|
||||
gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3);
|
||||
|
||||
Reference in New Issue
Block a user