feat(phase-29): func_801299C8 sweep chunk 2 — 45/45 banked

This commit is contained in:
Drew T
2026-07-22 14:36:09 -06:00
parent 048edcc6c2
commit 8887cf87bc
91 changed files with 4503 additions and 948 deletions
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_093/nonmatchings/ov_SC03_093", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801CAFA8/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801CC570 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801CC66A;
extern u8 D_801CC592;
extern u8 D_801CC570;
extern u8 D_801CAFA8[];
extern u8 D_801CAFA9[];
extern u8 D_801CAFAA[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801CC66A;
g = D_801CC592;
b = D_801CC570;
D_801CAFA8[0] = r * 5 >> 3;
D_801CAFA9[0] = g << 3;
D_801CAFAA[0] = (s32)(b * 255) >> 4;
D_801CAFA8[4] = (s32)(r * 143) >> 4;
D_801CAFA9[4] = g * 25 >> 1;
D_801CAFAA[4] = (s32)(b * 255) >> 4;
D_801CAFA8[8] = (s32)(r * 255) >> 4;
D_801CAFA9[8] = (s32)(g * 255) >> 4;
D_801CAFAA[8] = (s32)(b * 255) >> 4;
D_801CAFA8[12] = (s32)(r * 143) >> 4;
D_801CAFA9[12] = (s32)(g * 255) >> 4;
D_801CAFAA[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801CAFA8[i] = arg2[0x44] * D_801CC66A >> 4;
D_801CAFA9[i] = arg2[0x45] * D_801CC592 >> 4;
D_801CAFAA[i] = arg2[0x46] * D_801CC570 >> 4;
i = (arg1 + 1) * 4;
D_801CAFA8[i] = arg2[0x47] * D_801CC66A >> 4;
D_801CAFA9[i] = arg2[0x48] * D_801CC592 >> 4;
D_801CAFAA[i] = arg2[0x49] * D_801CC570 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801CC66A << 3;
arg2[0x21] = D_801CC592 << 3;
arg2[0x22] = D_801CC570 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_094/nonmatchings/ov_SC03_094", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801D1DE8/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801D3128 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801D3222;
extern u8 D_801D314A;
extern u8 D_801D3128;
extern u8 D_801D1DE8[];
extern u8 D_801D1DE9[];
extern u8 D_801D1DEA[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801D3222;
g = D_801D314A;
b = D_801D3128;
D_801D1DE8[0] = r * 5 >> 3;
D_801D1DE9[0] = g << 3;
D_801D1DEA[0] = (s32)(b * 255) >> 4;
D_801D1DE8[4] = (s32)(r * 143) >> 4;
D_801D1DE9[4] = g * 25 >> 1;
D_801D1DEA[4] = (s32)(b * 255) >> 4;
D_801D1DE8[8] = (s32)(r * 255) >> 4;
D_801D1DE9[8] = (s32)(g * 255) >> 4;
D_801D1DEA[8] = (s32)(b * 255) >> 4;
D_801D1DE8[12] = (s32)(r * 143) >> 4;
D_801D1DE9[12] = (s32)(g * 255) >> 4;
D_801D1DEA[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801D1DE8[i] = arg2[0x44] * D_801D3222 >> 4;
D_801D1DE9[i] = arg2[0x45] * D_801D314A >> 4;
D_801D1DEA[i] = arg2[0x46] * D_801D3128 >> 4;
i = (arg1 + 1) * 4;
D_801D1DE8[i] = arg2[0x47] * D_801D3222 >> 4;
D_801D1DE9[i] = arg2[0x48] * D_801D314A >> 4;
D_801D1DEA[i] = arg2[0x49] * D_801D3128 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801D3222 << 3;
arg2[0x21] = D_801D314A << 3;
arg2[0x22] = D_801D3128 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_095/nonmatchings/ov_SC03_095", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_8019A9E0/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_8019BD28 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_8019BE22;
extern u8 D_8019BD4A;
extern u8 D_8019BD28;
extern u8 D_8019A9E0[];
extern u8 D_8019A9E1[];
extern u8 D_8019A9E2[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_8019BE22;
g = D_8019BD4A;
b = D_8019BD28;
D_8019A9E0[0] = r * 5 >> 3;
D_8019A9E1[0] = g << 3;
D_8019A9E2[0] = (s32)(b * 255) >> 4;
D_8019A9E0[4] = (s32)(r * 143) >> 4;
D_8019A9E1[4] = g * 25 >> 1;
D_8019A9E2[4] = (s32)(b * 255) >> 4;
D_8019A9E0[8] = (s32)(r * 255) >> 4;
D_8019A9E1[8] = (s32)(g * 255) >> 4;
D_8019A9E2[8] = (s32)(b * 255) >> 4;
D_8019A9E0[12] = (s32)(r * 143) >> 4;
D_8019A9E1[12] = (s32)(g * 255) >> 4;
D_8019A9E2[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_8019A9E0[i] = arg2[0x44] * D_8019BE22 >> 4;
D_8019A9E1[i] = arg2[0x45] * D_8019BD4A >> 4;
D_8019A9E2[i] = arg2[0x46] * D_8019BD28 >> 4;
i = (arg1 + 1) * 4;
D_8019A9E0[i] = arg2[0x47] * D_8019BE22 >> 4;
D_8019A9E1[i] = arg2[0x48] * D_8019BD4A >> 4;
D_8019A9E2[i] = arg2[0x49] * D_8019BD28 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_8019BE22 << 3;
arg2[0x21] = D_8019BD4A << 3;
arg2[0x22] = D_8019BD28 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_096/nonmatchings/ov_SC03_096", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_80199498/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_8019A7E0 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_8019A8DA;
extern u8 D_8019A802;
extern u8 D_8019A7E0;
extern u8 D_80199498[];
extern u8 D_80199499[];
extern u8 D_8019949A[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_8019A8DA;
g = D_8019A802;
b = D_8019A7E0;
D_80199498[0] = r * 5 >> 3;
D_80199499[0] = g << 3;
D_8019949A[0] = (s32)(b * 255) >> 4;
D_80199498[4] = (s32)(r * 143) >> 4;
D_80199499[4] = g * 25 >> 1;
D_8019949A[4] = (s32)(b * 255) >> 4;
D_80199498[8] = (s32)(r * 255) >> 4;
D_80199499[8] = (s32)(g * 255) >> 4;
D_8019949A[8] = (s32)(b * 255) >> 4;
D_80199498[12] = (s32)(r * 143) >> 4;
D_80199499[12] = (s32)(g * 255) >> 4;
D_8019949A[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_80199498[i] = arg2[0x44] * D_8019A8DA >> 4;
D_80199499[i] = arg2[0x45] * D_8019A802 >> 4;
D_8019949A[i] = arg2[0x46] * D_8019A7E0 >> 4;
i = (arg1 + 1) * 4;
D_80199498[i] = arg2[0x47] * D_8019A8DA >> 4;
D_80199499[i] = arg2[0x48] * D_8019A802 >> 4;
D_8019949A[i] = arg2[0x49] * D_8019A7E0 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_8019A8DA << 3;
arg2[0x21] = D_8019A802 << 3;
arg2[0x22] = D_8019A7E0 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_097/nonmatchings/ov_SC03_097", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801A9520/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801AAD60 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801AAE5A;
extern u8 D_801AAD82;
extern u8 D_801AAD60;
extern u8 D_801A9520[];
extern u8 D_801A9521[];
extern u8 D_801A9522[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801AAE5A;
g = D_801AAD82;
b = D_801AAD60;
D_801A9520[0] = r * 5 >> 3;
D_801A9521[0] = g << 3;
D_801A9522[0] = (s32)(b * 255) >> 4;
D_801A9520[4] = (s32)(r * 143) >> 4;
D_801A9521[4] = g * 25 >> 1;
D_801A9522[4] = (s32)(b * 255) >> 4;
D_801A9520[8] = (s32)(r * 255) >> 4;
D_801A9521[8] = (s32)(g * 255) >> 4;
D_801A9522[8] = (s32)(b * 255) >> 4;
D_801A9520[12] = (s32)(r * 143) >> 4;
D_801A9521[12] = (s32)(g * 255) >> 4;
D_801A9522[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801A9520[i] = arg2[0x44] * D_801AAE5A >> 4;
D_801A9521[i] = arg2[0x45] * D_801AAD82 >> 4;
D_801A9522[i] = arg2[0x46] * D_801AAD60 >> 4;
i = (arg1 + 1) * 4;
D_801A9520[i] = arg2[0x47] * D_801AAE5A >> 4;
D_801A9521[i] = arg2[0x48] * D_801AAD82 >> 4;
D_801A9522[i] = arg2[0x49] * D_801AAD60 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801AAE5A << 3;
arg2[0x21] = D_801AAD82 << 3;
arg2[0x22] = D_801AAD60 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_098/nonmatchings/ov_SC03_098", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801C4190/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801C5748 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801C5842;
extern u8 D_801C576A;
extern u8 D_801C5748;
extern u8 D_801C4190[];
extern u8 D_801C4191[];
extern u8 D_801C4192[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801C5842;
g = D_801C576A;
b = D_801C5748;
D_801C4190[0] = r * 5 >> 3;
D_801C4191[0] = g << 3;
D_801C4192[0] = (s32)(b * 255) >> 4;
D_801C4190[4] = (s32)(r * 143) >> 4;
D_801C4191[4] = g * 25 >> 1;
D_801C4192[4] = (s32)(b * 255) >> 4;
D_801C4190[8] = (s32)(r * 255) >> 4;
D_801C4191[8] = (s32)(g * 255) >> 4;
D_801C4192[8] = (s32)(b * 255) >> 4;
D_801C4190[12] = (s32)(r * 143) >> 4;
D_801C4191[12] = (s32)(g * 255) >> 4;
D_801C4192[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801C4190[i] = arg2[0x44] * D_801C5842 >> 4;
D_801C4191[i] = arg2[0x45] * D_801C576A >> 4;
D_801C4192[i] = arg2[0x46] * D_801C5748 >> 4;
i = (arg1 + 1) * 4;
D_801C4190[i] = arg2[0x47] * D_801C5842 >> 4;
D_801C4191[i] = arg2[0x48] * D_801C576A >> 4;
D_801C4192[i] = arg2[0x49] * D_801C5748 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801C5842 << 3;
arg2[0x21] = D_801C576A << 3;
arg2[0x22] = D_801C5748 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_099/nonmatchings/ov_SC03_099", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801BDE60/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801BF2F0 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801BF3EA;
extern u8 D_801BF312;
extern u8 D_801BF2F0;
extern u8 D_801BDE60[];
extern u8 D_801BDE61[];
extern u8 D_801BDE62[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801BF3EA;
g = D_801BF312;
b = D_801BF2F0;
D_801BDE60[0] = r * 5 >> 3;
D_801BDE61[0] = g << 3;
D_801BDE62[0] = (s32)(b * 255) >> 4;
D_801BDE60[4] = (s32)(r * 143) >> 4;
D_801BDE61[4] = g * 25 >> 1;
D_801BDE62[4] = (s32)(b * 255) >> 4;
D_801BDE60[8] = (s32)(r * 255) >> 4;
D_801BDE61[8] = (s32)(g * 255) >> 4;
D_801BDE62[8] = (s32)(b * 255) >> 4;
D_801BDE60[12] = (s32)(r * 143) >> 4;
D_801BDE61[12] = (s32)(g * 255) >> 4;
D_801BDE62[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801BDE60[i] = arg2[0x44] * D_801BF3EA >> 4;
D_801BDE61[i] = arg2[0x45] * D_801BF312 >> 4;
D_801BDE62[i] = arg2[0x46] * D_801BF2F0 >> 4;
i = (arg1 + 1) * 4;
D_801BDE60[i] = arg2[0x47] * D_801BF3EA >> 4;
D_801BDE61[i] = arg2[0x48] * D_801BF312 >> 4;
D_801BDE62[i] = arg2[0x49] * D_801BF2F0 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801BF3EA << 3;
arg2[0x21] = D_801BF312 << 3;
arg2[0x22] = D_801BF2F0 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_100/nonmatchings/ov_SC03_100", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801C70E8/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801C87F8 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801C88F2;
extern u8 D_801C881A;
extern u8 D_801C87F8;
extern u8 D_801C70E8[];
extern u8 D_801C70E9[];
extern u8 D_801C70EA[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801C88F2;
g = D_801C881A;
b = D_801C87F8;
D_801C70E8[0] = r * 5 >> 3;
D_801C70E9[0] = g << 3;
D_801C70EA[0] = (s32)(b * 255) >> 4;
D_801C70E8[4] = (s32)(r * 143) >> 4;
D_801C70E9[4] = g * 25 >> 1;
D_801C70EA[4] = (s32)(b * 255) >> 4;
D_801C70E8[8] = (s32)(r * 255) >> 4;
D_801C70E9[8] = (s32)(g * 255) >> 4;
D_801C70EA[8] = (s32)(b * 255) >> 4;
D_801C70E8[12] = (s32)(r * 143) >> 4;
D_801C70E9[12] = (s32)(g * 255) >> 4;
D_801C70EA[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801C70E8[i] = arg2[0x44] * D_801C88F2 >> 4;
D_801C70E9[i] = arg2[0x45] * D_801C881A >> 4;
D_801C70EA[i] = arg2[0x46] * D_801C87F8 >> 4;
i = (arg1 + 1) * 4;
D_801C70E8[i] = arg2[0x47] * D_801C88F2 >> 4;
D_801C70E9[i] = arg2[0x48] * D_801C881A >> 4;
D_801C70EA[i] = arg2[0x49] * D_801C87F8 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801C88F2 << 3;
arg2[0x21] = D_801C881A << 3;
arg2[0x22] = D_801C87F8 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_101/nonmatchings/ov_SC03_101", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_8019C9B0/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_8019E0B8 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_8019E1B2;
extern u8 D_8019E0DA;
extern u8 D_8019E0B8;
extern u8 D_8019C9B0[];
extern u8 D_8019C9B1[];
extern u8 D_8019C9B2[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_8019E1B2;
g = D_8019E0DA;
b = D_8019E0B8;
D_8019C9B0[0] = r * 5 >> 3;
D_8019C9B1[0] = g << 3;
D_8019C9B2[0] = (s32)(b * 255) >> 4;
D_8019C9B0[4] = (s32)(r * 143) >> 4;
D_8019C9B1[4] = g * 25 >> 1;
D_8019C9B2[4] = (s32)(b * 255) >> 4;
D_8019C9B0[8] = (s32)(r * 255) >> 4;
D_8019C9B1[8] = (s32)(g * 255) >> 4;
D_8019C9B2[8] = (s32)(b * 255) >> 4;
D_8019C9B0[12] = (s32)(r * 143) >> 4;
D_8019C9B1[12] = (s32)(g * 255) >> 4;
D_8019C9B2[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_8019C9B0[i] = arg2[0x44] * D_8019E1B2 >> 4;
D_8019C9B1[i] = arg2[0x45] * D_8019E0DA >> 4;
D_8019C9B2[i] = arg2[0x46] * D_8019E0B8 >> 4;
i = (arg1 + 1) * 4;
D_8019C9B0[i] = arg2[0x47] * D_8019E1B2 >> 4;
D_8019C9B1[i] = arg2[0x48] * D_8019E0DA >> 4;
D_8019C9B2[i] = arg2[0x49] * D_8019E0B8 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_8019E1B2 << 3;
arg2[0x21] = D_8019E0DA << 3;
arg2[0x22] = D_8019E0B8 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_102/nonmatchings/ov_SC03_102", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801B8628/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801B9AB8 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801B9BB2;
extern u8 D_801B9ADA;
extern u8 D_801B9AB8;
extern u8 D_801B8628[];
extern u8 D_801B8629[];
extern u8 D_801B862A[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801B9BB2;
g = D_801B9ADA;
b = D_801B9AB8;
D_801B8628[0] = r * 5 >> 3;
D_801B8629[0] = g << 3;
D_801B862A[0] = (s32)(b * 255) >> 4;
D_801B8628[4] = (s32)(r * 143) >> 4;
D_801B8629[4] = g * 25 >> 1;
D_801B862A[4] = (s32)(b * 255) >> 4;
D_801B8628[8] = (s32)(r * 255) >> 4;
D_801B8629[8] = (s32)(g * 255) >> 4;
D_801B862A[8] = (s32)(b * 255) >> 4;
D_801B8628[12] = (s32)(r * 143) >> 4;
D_801B8629[12] = (s32)(g * 255) >> 4;
D_801B862A[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801B8628[i] = arg2[0x44] * D_801B9BB2 >> 4;
D_801B8629[i] = arg2[0x45] * D_801B9ADA >> 4;
D_801B862A[i] = arg2[0x46] * D_801B9AB8 >> 4;
i = (arg1 + 1) * 4;
D_801B8628[i] = arg2[0x47] * D_801B9BB2 >> 4;
D_801B8629[i] = arg2[0x48] * D_801B9ADA >> 4;
D_801B862A[i] = arg2[0x49] * D_801B9AB8 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801B9BB2 << 3;
arg2[0x21] = D_801B9ADA << 3;
arg2[0x22] = D_801B9AB8 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_103/nonmatchings/ov_SC03_103", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801C37E0/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801C4C68 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801C4D62;
extern u8 D_801C4C8A;
extern u8 D_801C4C68;
extern u8 D_801C37E0[];
extern u8 D_801C37E1[];
extern u8 D_801C37E2[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801C4D62;
g = D_801C4C8A;
b = D_801C4C68;
D_801C37E0[0] = r * 5 >> 3;
D_801C37E1[0] = g << 3;
D_801C37E2[0] = (s32)(b * 255) >> 4;
D_801C37E0[4] = (s32)(r * 143) >> 4;
D_801C37E1[4] = g * 25 >> 1;
D_801C37E2[4] = (s32)(b * 255) >> 4;
D_801C37E0[8] = (s32)(r * 255) >> 4;
D_801C37E1[8] = (s32)(g * 255) >> 4;
D_801C37E2[8] = (s32)(b * 255) >> 4;
D_801C37E0[12] = (s32)(r * 143) >> 4;
D_801C37E1[12] = (s32)(g * 255) >> 4;
D_801C37E2[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801C37E0[i] = arg2[0x44] * D_801C4D62 >> 4;
D_801C37E1[i] = arg2[0x45] * D_801C4C8A >> 4;
D_801C37E2[i] = arg2[0x46] * D_801C4C68 >> 4;
i = (arg1 + 1) * 4;
D_801C37E0[i] = arg2[0x47] * D_801C4D62 >> 4;
D_801C37E1[i] = arg2[0x48] * D_801C4C8A >> 4;
D_801C37E2[i] = arg2[0x49] * D_801C4C68 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801C4D62 << 3;
arg2[0x21] = D_801C4C8A << 3;
arg2[0x22] = D_801C4C68 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_104/nonmatchings/ov_SC03_104", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801BED78/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801C0328 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801C0422;
extern u8 D_801C034A;
extern u8 D_801C0328;
extern u8 D_801BED78[];
extern u8 D_801BED79[];
extern u8 D_801BED7A[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801C0422;
g = D_801C034A;
b = D_801C0328;
D_801BED78[0] = r * 5 >> 3;
D_801BED79[0] = g << 3;
D_801BED7A[0] = (s32)(b * 255) >> 4;
D_801BED78[4] = (s32)(r * 143) >> 4;
D_801BED79[4] = g * 25 >> 1;
D_801BED7A[4] = (s32)(b * 255) >> 4;
D_801BED78[8] = (s32)(r * 255) >> 4;
D_801BED79[8] = (s32)(g * 255) >> 4;
D_801BED7A[8] = (s32)(b * 255) >> 4;
D_801BED78[12] = (s32)(r * 143) >> 4;
D_801BED79[12] = (s32)(g * 255) >> 4;
D_801BED7A[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801BED78[i] = arg2[0x44] * D_801C0422 >> 4;
D_801BED79[i] = arg2[0x45] * D_801C034A >> 4;
D_801BED7A[i] = arg2[0x46] * D_801C0328 >> 4;
i = (arg1 + 1) * 4;
D_801BED78[i] = arg2[0x47] * D_801C0422 >> 4;
D_801BED79[i] = arg2[0x48] * D_801C034A >> 4;
D_801BED7A[i] = arg2[0x49] * D_801C0328 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801C0422 << 3;
arg2[0x21] = D_801C034A << 3;
arg2[0x22] = D_801C0328 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_105/nonmatchings/ov_SC03_105", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801B8308/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801BCBA8 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801BCCAA;
extern u8 D_801BCBD2;
extern u8 D_801BCBA8;
extern u8 D_801B8308[];
extern u8 D_801B8309[];
extern u8 D_801B830A[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801BCCAA;
g = D_801BCBD2;
b = D_801BCBA8;
D_801B8308[0] = r * 5 >> 3;
D_801B8309[0] = g << 3;
D_801B830A[0] = (s32)(b * 255) >> 4;
D_801B8308[4] = (s32)(r * 143) >> 4;
D_801B8309[4] = g * 25 >> 1;
D_801B830A[4] = (s32)(b * 255) >> 4;
D_801B8308[8] = (s32)(r * 255) >> 4;
D_801B8309[8] = (s32)(g * 255) >> 4;
D_801B830A[8] = (s32)(b * 255) >> 4;
D_801B8308[12] = (s32)(r * 143) >> 4;
D_801B8309[12] = (s32)(g * 255) >> 4;
D_801B830A[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801B8308[i] = arg2[0x44] * D_801BCCAA >> 4;
D_801B8309[i] = arg2[0x45] * D_801BCBD2 >> 4;
D_801B830A[i] = arg2[0x46] * D_801BCBA8 >> 4;
i = (arg1 + 1) * 4;
D_801B8308[i] = arg2[0x47] * D_801BCCAA >> 4;
D_801B8309[i] = arg2[0x48] * D_801BCBD2 >> 4;
D_801B830A[i] = arg2[0x49] * D_801BCBA8 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801BCCAA << 3;
arg2[0x21] = D_801BCBD2 << 3;
arg2[0x22] = D_801BCBA8 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_108/nonmatchings/ov_SC03_108", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801A05F8/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801A1938 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801A1A32;
extern u8 D_801A195A;
extern u8 D_801A1938;
extern u8 D_801A05F8[];
extern u8 D_801A05F9[];
extern u8 D_801A05FA[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801A1A32;
g = D_801A195A;
b = D_801A1938;
D_801A05F8[0] = r * 5 >> 3;
D_801A05F9[0] = g << 3;
D_801A05FA[0] = (s32)(b * 255) >> 4;
D_801A05F8[4] = (s32)(r * 143) >> 4;
D_801A05F9[4] = g * 25 >> 1;
D_801A05FA[4] = (s32)(b * 255) >> 4;
D_801A05F8[8] = (s32)(r * 255) >> 4;
D_801A05F9[8] = (s32)(g * 255) >> 4;
D_801A05FA[8] = (s32)(b * 255) >> 4;
D_801A05F8[12] = (s32)(r * 143) >> 4;
D_801A05F9[12] = (s32)(g * 255) >> 4;
D_801A05FA[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801A05F8[i] = arg2[0x44] * D_801A1A32 >> 4;
D_801A05F9[i] = arg2[0x45] * D_801A195A >> 4;
D_801A05FA[i] = arg2[0x46] * D_801A1938 >> 4;
i = (arg1 + 1) * 4;
D_801A05F8[i] = arg2[0x47] * D_801A1A32 >> 4;
D_801A05F9[i] = arg2[0x48] * D_801A195A >> 4;
D_801A05FA[i] = arg2[0x49] * D_801A1938 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801A1A32 << 3;
arg2[0x21] = D_801A195A << 3;
arg2[0x22] = D_801A1938 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_109/nonmatchings/ov_SC03_109", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_80199008/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_8019A340 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_8019A43A;
extern u8 D_8019A362;
extern u8 D_8019A340;
extern u8 D_80199008[];
extern u8 D_80199009[];
extern u8 D_8019900A[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_8019A43A;
g = D_8019A362;
b = D_8019A340;
D_80199008[0] = r * 5 >> 3;
D_80199009[0] = g << 3;
D_8019900A[0] = (s32)(b * 255) >> 4;
D_80199008[4] = (s32)(r * 143) >> 4;
D_80199009[4] = g * 25 >> 1;
D_8019900A[4] = (s32)(b * 255) >> 4;
D_80199008[8] = (s32)(r * 255) >> 4;
D_80199009[8] = (s32)(g * 255) >> 4;
D_8019900A[8] = (s32)(b * 255) >> 4;
D_80199008[12] = (s32)(r * 143) >> 4;
D_80199009[12] = (s32)(g * 255) >> 4;
D_8019900A[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_80199008[i] = arg2[0x44] * D_8019A43A >> 4;
D_80199009[i] = arg2[0x45] * D_8019A362 >> 4;
D_8019900A[i] = arg2[0x46] * D_8019A340 >> 4;
i = (arg1 + 1) * 4;
D_80199008[i] = arg2[0x47] * D_8019A43A >> 4;
D_80199009[i] = arg2[0x48] * D_8019A362 >> 4;
D_8019900A[i] = arg2[0x49] * D_8019A340 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_8019A43A << 3;
arg2[0x21] = D_8019A362 << 3;
arg2[0x22] = D_8019A340 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_110/nonmatchings/ov_SC03_110", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801A0DB0/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801A2370 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801A246A;
extern u8 D_801A2392;
extern u8 D_801A2370;
extern u8 D_801A0DB0[];
extern u8 D_801A0DB1[];
extern u8 D_801A0DB2[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801A246A;
g = D_801A2392;
b = D_801A2370;
D_801A0DB0[0] = r * 5 >> 3;
D_801A0DB1[0] = g << 3;
D_801A0DB2[0] = (s32)(b * 255) >> 4;
D_801A0DB0[4] = (s32)(r * 143) >> 4;
D_801A0DB1[4] = g * 25 >> 1;
D_801A0DB2[4] = (s32)(b * 255) >> 4;
D_801A0DB0[8] = (s32)(r * 255) >> 4;
D_801A0DB1[8] = (s32)(g * 255) >> 4;
D_801A0DB2[8] = (s32)(b * 255) >> 4;
D_801A0DB0[12] = (s32)(r * 143) >> 4;
D_801A0DB1[12] = (s32)(g * 255) >> 4;
D_801A0DB2[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801A0DB0[i] = arg2[0x44] * D_801A246A >> 4;
D_801A0DB1[i] = arg2[0x45] * D_801A2392 >> 4;
D_801A0DB2[i] = arg2[0x46] * D_801A2370 >> 4;
i = (arg1 + 1) * 4;
D_801A0DB0[i] = arg2[0x47] * D_801A246A >> 4;
D_801A0DB1[i] = arg2[0x48] * D_801A2392 >> 4;
D_801A0DB2[i] = arg2[0x49] * D_801A2370 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801A246A << 3;
arg2[0x21] = D_801A2392 << 3;
arg2[0x22] = D_801A2370 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_111/nonmatchings/ov_SC03_111", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801B82B8/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801B9A18 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801B9B12;
extern u8 D_801B9A3A;
extern u8 D_801B9A18;
extern u8 D_801B82B8[];
extern u8 D_801B82B9[];
extern u8 D_801B82BA[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801B9B12;
g = D_801B9A3A;
b = D_801B9A18;
D_801B82B8[0] = r * 5 >> 3;
D_801B82B9[0] = g << 3;
D_801B82BA[0] = (s32)(b * 255) >> 4;
D_801B82B8[4] = (s32)(r * 143) >> 4;
D_801B82B9[4] = g * 25 >> 1;
D_801B82BA[4] = (s32)(b * 255) >> 4;
D_801B82B8[8] = (s32)(r * 255) >> 4;
D_801B82B9[8] = (s32)(g * 255) >> 4;
D_801B82BA[8] = (s32)(b * 255) >> 4;
D_801B82B8[12] = (s32)(r * 143) >> 4;
D_801B82B9[12] = (s32)(g * 255) >> 4;
D_801B82BA[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801B82B8[i] = arg2[0x44] * D_801B9B12 >> 4;
D_801B82B9[i] = arg2[0x45] * D_801B9A3A >> 4;
D_801B82BA[i] = arg2[0x46] * D_801B9A18 >> 4;
i = (arg1 + 1) * 4;
D_801B82B8[i] = arg2[0x47] * D_801B9B12 >> 4;
D_801B82B9[i] = arg2[0x48] * D_801B9A3A >> 4;
D_801B82BA[i] = arg2[0x49] * D_801B9A18 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801B9B12 << 3;
arg2[0x21] = D_801B9A3A << 3;
arg2[0x22] = D_801B9A18 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_112/nonmatchings/ov_SC03_112", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801ABB18/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801ACFE0 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801AD0DA;
extern u8 D_801AD002;
extern u8 D_801ACFE0;
extern u8 D_801ABB18[];
extern u8 D_801ABB19[];
extern u8 D_801ABB1A[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801AD0DA;
g = D_801AD002;
b = D_801ACFE0;
D_801ABB18[0] = r * 5 >> 3;
D_801ABB19[0] = g << 3;
D_801ABB1A[0] = (s32)(b * 255) >> 4;
D_801ABB18[4] = (s32)(r * 143) >> 4;
D_801ABB19[4] = g * 25 >> 1;
D_801ABB1A[4] = (s32)(b * 255) >> 4;
D_801ABB18[8] = (s32)(r * 255) >> 4;
D_801ABB19[8] = (s32)(g * 255) >> 4;
D_801ABB1A[8] = (s32)(b * 255) >> 4;
D_801ABB18[12] = (s32)(r * 143) >> 4;
D_801ABB19[12] = (s32)(g * 255) >> 4;
D_801ABB1A[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801ABB18[i] = arg2[0x44] * D_801AD0DA >> 4;
D_801ABB19[i] = arg2[0x45] * D_801AD002 >> 4;
D_801ABB1A[i] = arg2[0x46] * D_801ACFE0 >> 4;
i = (arg1 + 1) * 4;
D_801ABB18[i] = arg2[0x47] * D_801AD0DA >> 4;
D_801ABB19[i] = arg2[0x48] * D_801AD002 >> 4;
D_801ABB1A[i] = arg2[0x49] * D_801ACFE0 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801AD0DA << 3;
arg2[0x21] = D_801AD002 << 3;
arg2[0x22] = D_801ACFE0 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_113/nonmatchings/ov_SC03_113", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801A46A8/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801A5B80 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801A5C7A;
extern u8 D_801A5BA2;
extern u8 D_801A5B80;
extern u8 D_801A46A8[];
extern u8 D_801A46A9[];
extern u8 D_801A46AA[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801A5C7A;
g = D_801A5BA2;
b = D_801A5B80;
D_801A46A8[0] = r * 5 >> 3;
D_801A46A9[0] = g << 3;
D_801A46AA[0] = (s32)(b * 255) >> 4;
D_801A46A8[4] = (s32)(r * 143) >> 4;
D_801A46A9[4] = g * 25 >> 1;
D_801A46AA[4] = (s32)(b * 255) >> 4;
D_801A46A8[8] = (s32)(r * 255) >> 4;
D_801A46A9[8] = (s32)(g * 255) >> 4;
D_801A46AA[8] = (s32)(b * 255) >> 4;
D_801A46A8[12] = (s32)(r * 143) >> 4;
D_801A46A9[12] = (s32)(g * 255) >> 4;
D_801A46AA[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801A46A8[i] = arg2[0x44] * D_801A5C7A >> 4;
D_801A46A9[i] = arg2[0x45] * D_801A5BA2 >> 4;
D_801A46AA[i] = arg2[0x46] * D_801A5B80 >> 4;
i = (arg1 + 1) * 4;
D_801A46A8[i] = arg2[0x47] * D_801A5C7A >> 4;
D_801A46A9[i] = arg2[0x48] * D_801A5BA2 >> 4;
D_801A46AA[i] = arg2[0x49] * D_801A5B80 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801A5C7A << 3;
arg2[0x21] = D_801A5BA2 << 3;
arg2[0x22] = D_801A5B80 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_114/nonmatchings/ov_SC03_114", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801A8340/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801A96F8 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801A97F2;
extern u8 D_801A971A;
extern u8 D_801A96F8;
extern u8 D_801A8340[];
extern u8 D_801A8341[];
extern u8 D_801A8342[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801A97F2;
g = D_801A971A;
b = D_801A96F8;
D_801A8340[0] = r * 5 >> 3;
D_801A8341[0] = g << 3;
D_801A8342[0] = (s32)(b * 255) >> 4;
D_801A8340[4] = (s32)(r * 143) >> 4;
D_801A8341[4] = g * 25 >> 1;
D_801A8342[4] = (s32)(b * 255) >> 4;
D_801A8340[8] = (s32)(r * 255) >> 4;
D_801A8341[8] = (s32)(g * 255) >> 4;
D_801A8342[8] = (s32)(b * 255) >> 4;
D_801A8340[12] = (s32)(r * 143) >> 4;
D_801A8341[12] = (s32)(g * 255) >> 4;
D_801A8342[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801A8340[i] = arg2[0x44] * D_801A97F2 >> 4;
D_801A8341[i] = arg2[0x45] * D_801A971A >> 4;
D_801A8342[i] = arg2[0x46] * D_801A96F8 >> 4;
i = (arg1 + 1) * 4;
D_801A8340[i] = arg2[0x47] * D_801A97F2 >> 4;
D_801A8341[i] = arg2[0x48] * D_801A971A >> 4;
D_801A8342[i] = arg2[0x49] * D_801A96F8 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801A97F2 << 3;
arg2[0x21] = D_801A971A << 3;
arg2[0x22] = D_801A96F8 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_115/nonmatchings/ov_SC03_115", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_80193C98/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_80194FD0 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801950CA;
extern u8 D_80194FF2;
extern u8 D_80194FD0;
extern u8 D_80193C98[];
extern u8 D_80193C99[];
extern u8 D_80193C9A[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801950CA;
g = D_80194FF2;
b = D_80194FD0;
D_80193C98[0] = r * 5 >> 3;
D_80193C99[0] = g << 3;
D_80193C9A[0] = (s32)(b * 255) >> 4;
D_80193C98[4] = (s32)(r * 143) >> 4;
D_80193C99[4] = g * 25 >> 1;
D_80193C9A[4] = (s32)(b * 255) >> 4;
D_80193C98[8] = (s32)(r * 255) >> 4;
D_80193C99[8] = (s32)(g * 255) >> 4;
D_80193C9A[8] = (s32)(b * 255) >> 4;
D_80193C98[12] = (s32)(r * 143) >> 4;
D_80193C99[12] = (s32)(g * 255) >> 4;
D_80193C9A[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_80193C98[i] = arg2[0x44] * D_801950CA >> 4;
D_80193C99[i] = arg2[0x45] * D_80194FF2 >> 4;
D_80193C9A[i] = arg2[0x46] * D_80194FD0 >> 4;
i = (arg1 + 1) * 4;
D_80193C98[i] = arg2[0x47] * D_801950CA >> 4;
D_80193C99[i] = arg2[0x48] * D_80194FF2 >> 4;
D_80193C9A[i] = arg2[0x49] * D_80194FD0 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801950CA << 3;
arg2[0x21] = D_80194FF2 << 3;
arg2[0x22] = D_80194FD0 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_116/nonmatchings/ov_SC03_116", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_80196828/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_80197BE0 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_80197CDA;
extern u8 D_80197C02;
extern u8 D_80197BE0;
extern u8 D_80196828[];
extern u8 D_80196829[];
extern u8 D_8019682A[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_80197CDA;
g = D_80197C02;
b = D_80197BE0;
D_80196828[0] = r * 5 >> 3;
D_80196829[0] = g << 3;
D_8019682A[0] = (s32)(b * 255) >> 4;
D_80196828[4] = (s32)(r * 143) >> 4;
D_80196829[4] = g * 25 >> 1;
D_8019682A[4] = (s32)(b * 255) >> 4;
D_80196828[8] = (s32)(r * 255) >> 4;
D_80196829[8] = (s32)(g * 255) >> 4;
D_8019682A[8] = (s32)(b * 255) >> 4;
D_80196828[12] = (s32)(r * 143) >> 4;
D_80196829[12] = (s32)(g * 255) >> 4;
D_8019682A[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_80196828[i] = arg2[0x44] * D_80197CDA >> 4;
D_80196829[i] = arg2[0x45] * D_80197C02 >> 4;
D_8019682A[i] = arg2[0x46] * D_80197BE0 >> 4;
i = (arg1 + 1) * 4;
D_80196828[i] = arg2[0x47] * D_80197CDA >> 4;
D_80196829[i] = arg2[0x48] * D_80197C02 >> 4;
D_8019682A[i] = arg2[0x49] * D_80197BE0 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_80197CDA << 3;
arg2[0x21] = D_80197C02 << 3;
arg2[0x22] = D_80197BE0 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_117/nonmatchings/ov_SC03_117", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801CCFA0/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801CE7D8 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801CE8D2;
extern u8 D_801CE7FA;
extern u8 D_801CE7D8;
extern u8 D_801CCFA0[];
extern u8 D_801CCFA1[];
extern u8 D_801CCFA2[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801CE8D2;
g = D_801CE7FA;
b = D_801CE7D8;
D_801CCFA0[0] = r * 5 >> 3;
D_801CCFA1[0] = g << 3;
D_801CCFA2[0] = (s32)(b * 255) >> 4;
D_801CCFA0[4] = (s32)(r * 143) >> 4;
D_801CCFA1[4] = g * 25 >> 1;
D_801CCFA2[4] = (s32)(b * 255) >> 4;
D_801CCFA0[8] = (s32)(r * 255) >> 4;
D_801CCFA1[8] = (s32)(g * 255) >> 4;
D_801CCFA2[8] = (s32)(b * 255) >> 4;
D_801CCFA0[12] = (s32)(r * 143) >> 4;
D_801CCFA1[12] = (s32)(g * 255) >> 4;
D_801CCFA2[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801CCFA0[i] = arg2[0x44] * D_801CE8D2 >> 4;
D_801CCFA1[i] = arg2[0x45] * D_801CE7FA >> 4;
D_801CCFA2[i] = arg2[0x46] * D_801CE7D8 >> 4;
i = (arg1 + 1) * 4;
D_801CCFA0[i] = arg2[0x47] * D_801CE8D2 >> 4;
D_801CCFA1[i] = arg2[0x48] * D_801CE7FA >> 4;
D_801CCFA2[i] = arg2[0x49] * D_801CE7D8 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801CE8D2 << 3;
arg2[0x21] = D_801CE7FA << 3;
arg2[0x22] = D_801CE7D8 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_118/nonmatchings/ov_SC03_118", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801D2EF0/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801D46E0 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801D47DA;
extern u8 D_801D4702;
extern u8 D_801D46E0;
extern u8 D_801D2EF0[];
extern u8 D_801D2EF1[];
extern u8 D_801D2EF2[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801D47DA;
g = D_801D4702;
b = D_801D46E0;
D_801D2EF0[0] = r * 5 >> 3;
D_801D2EF1[0] = g << 3;
D_801D2EF2[0] = (s32)(b * 255) >> 4;
D_801D2EF0[4] = (s32)(r * 143) >> 4;
D_801D2EF1[4] = g * 25 >> 1;
D_801D2EF2[4] = (s32)(b * 255) >> 4;
D_801D2EF0[8] = (s32)(r * 255) >> 4;
D_801D2EF1[8] = (s32)(g * 255) >> 4;
D_801D2EF2[8] = (s32)(b * 255) >> 4;
D_801D2EF0[12] = (s32)(r * 143) >> 4;
D_801D2EF1[12] = (s32)(g * 255) >> 4;
D_801D2EF2[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801D2EF0[i] = arg2[0x44] * D_801D47DA >> 4;
D_801D2EF1[i] = arg2[0x45] * D_801D4702 >> 4;
D_801D2EF2[i] = arg2[0x46] * D_801D46E0 >> 4;
i = (arg1 + 1) * 4;
D_801D2EF0[i] = arg2[0x47] * D_801D47DA >> 4;
D_801D2EF1[i] = arg2[0x48] * D_801D4702 >> 4;
D_801D2EF2[i] = arg2[0x49] * D_801D46E0 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801D47DA << 3;
arg2[0x21] = D_801D4702 << 3;
arg2[0x22] = D_801D46E0 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_119/nonmatchings/ov_SC03_119", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801D2EF0/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801D46E0 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801D47DA;
extern u8 D_801D4702;
extern u8 D_801D46E0;
extern u8 D_801D2EF0[];
extern u8 D_801D2EF1[];
extern u8 D_801D2EF2[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801D47DA;
g = D_801D4702;
b = D_801D46E0;
D_801D2EF0[0] = r * 5 >> 3;
D_801D2EF1[0] = g << 3;
D_801D2EF2[0] = (s32)(b * 255) >> 4;
D_801D2EF0[4] = (s32)(r * 143) >> 4;
D_801D2EF1[4] = g * 25 >> 1;
D_801D2EF2[4] = (s32)(b * 255) >> 4;
D_801D2EF0[8] = (s32)(r * 255) >> 4;
D_801D2EF1[8] = (s32)(g * 255) >> 4;
D_801D2EF2[8] = (s32)(b * 255) >> 4;
D_801D2EF0[12] = (s32)(r * 143) >> 4;
D_801D2EF1[12] = (s32)(g * 255) >> 4;
D_801D2EF2[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801D2EF0[i] = arg2[0x44] * D_801D47DA >> 4;
D_801D2EF1[i] = arg2[0x45] * D_801D4702 >> 4;
D_801D2EF2[i] = arg2[0x46] * D_801D46E0 >> 4;
i = (arg1 + 1) * 4;
D_801D2EF0[i] = arg2[0x47] * D_801D47DA >> 4;
D_801D2EF1[i] = arg2[0x48] * D_801D4702 >> 4;
D_801D2EF2[i] = arg2[0x49] * D_801D46E0 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801D47DA << 3;
arg2[0x21] = D_801D4702 << 3;
arg2[0x22] = D_801D46E0 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_121/nonmatchings/ov_SC03_121", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801AD998/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801AEE40 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801AEF3A;
extern u8 D_801AEE62;
extern u8 D_801AEE40;
extern u8 D_801AD998[];
extern u8 D_801AD999[];
extern u8 D_801AD99A[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801AEF3A;
g = D_801AEE62;
b = D_801AEE40;
D_801AD998[0] = r * 5 >> 3;
D_801AD999[0] = g << 3;
D_801AD99A[0] = (s32)(b * 255) >> 4;
D_801AD998[4] = (s32)(r * 143) >> 4;
D_801AD999[4] = g * 25 >> 1;
D_801AD99A[4] = (s32)(b * 255) >> 4;
D_801AD998[8] = (s32)(r * 255) >> 4;
D_801AD999[8] = (s32)(g * 255) >> 4;
D_801AD99A[8] = (s32)(b * 255) >> 4;
D_801AD998[12] = (s32)(r * 143) >> 4;
D_801AD999[12] = (s32)(g * 255) >> 4;
D_801AD99A[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801AD998[i] = arg2[0x44] * D_801AEF3A >> 4;
D_801AD999[i] = arg2[0x45] * D_801AEE62 >> 4;
D_801AD99A[i] = arg2[0x46] * D_801AEE40 >> 4;
i = (arg1 + 1) * 4;
D_801AD998[i] = arg2[0x47] * D_801AEF3A >> 4;
D_801AD999[i] = arg2[0x48] * D_801AEE62 >> 4;
D_801AD99A[i] = arg2[0x49] * D_801AEE40 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801AEF3A << 3;
arg2[0x21] = D_801AEE62 << 3;
arg2[0x22] = D_801AEE40 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_124/nonmatchings/ov_SC03_124", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801E04C0/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801E2448 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801E2572;
extern u8 D_801E2472;
extern u8 D_801E2448;
extern u8 D_801E04C0[];
extern u8 D_801E04C1[];
extern u8 D_801E04C2[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801E2572;
g = D_801E2472;
b = D_801E2448;
D_801E04C0[0] = r * 5 >> 3;
D_801E04C1[0] = g << 3;
D_801E04C2[0] = (s32)(b * 255) >> 4;
D_801E04C0[4] = (s32)(r * 143) >> 4;
D_801E04C1[4] = g * 25 >> 1;
D_801E04C2[4] = (s32)(b * 255) >> 4;
D_801E04C0[8] = (s32)(r * 255) >> 4;
D_801E04C1[8] = (s32)(g * 255) >> 4;
D_801E04C2[8] = (s32)(b * 255) >> 4;
D_801E04C0[12] = (s32)(r * 143) >> 4;
D_801E04C1[12] = (s32)(g * 255) >> 4;
D_801E04C2[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801E04C0[i] = arg2[0x44] * D_801E2572 >> 4;
D_801E04C1[i] = arg2[0x45] * D_801E2472 >> 4;
D_801E04C2[i] = arg2[0x46] * D_801E2448 >> 4;
i = (arg1 + 1) * 4;
D_801E04C0[i] = arg2[0x47] * D_801E2572 >> 4;
D_801E04C1[i] = arg2[0x48] * D_801E2472 >> 4;
D_801E04C2[i] = arg2[0x49] * D_801E2448 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801E2572 << 3;
arg2[0x21] = D_801E2472 << 3;
arg2[0x22] = D_801E2448 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_125/nonmatchings/ov_SC03_125", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801C2DB8/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801C43A0 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801C44CA;
extern u8 D_801C43CA;
extern u8 D_801C43A0;
extern u8 D_801C2DB8[];
extern u8 D_801C2DB9[];
extern u8 D_801C2DBA[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801C44CA;
g = D_801C43CA;
b = D_801C43A0;
D_801C2DB8[0] = r * 5 >> 3;
D_801C2DB9[0] = g << 3;
D_801C2DBA[0] = (s32)(b * 255) >> 4;
D_801C2DB8[4] = (s32)(r * 143) >> 4;
D_801C2DB9[4] = g * 25 >> 1;
D_801C2DBA[4] = (s32)(b * 255) >> 4;
D_801C2DB8[8] = (s32)(r * 255) >> 4;
D_801C2DB9[8] = (s32)(g * 255) >> 4;
D_801C2DBA[8] = (s32)(b * 255) >> 4;
D_801C2DB8[12] = (s32)(r * 143) >> 4;
D_801C2DB9[12] = (s32)(g * 255) >> 4;
D_801C2DBA[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801C2DB8[i] = arg2[0x44] * D_801C44CA >> 4;
D_801C2DB9[i] = arg2[0x45] * D_801C43CA >> 4;
D_801C2DBA[i] = arg2[0x46] * D_801C43A0 >> 4;
i = (arg1 + 1) * 4;
D_801C2DB8[i] = arg2[0x47] * D_801C44CA >> 4;
D_801C2DB9[i] = arg2[0x48] * D_801C43CA >> 4;
D_801C2DBA[i] = arg2[0x49] * D_801C43A0 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801C44CA << 3;
arg2[0x21] = D_801C43CA << 3;
arg2[0x22] = D_801C43A0 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_126/nonmatchings/ov_SC03_126", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_8018E680/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_8018F9B8 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_8018FAB2;
extern u8 D_8018F9DA;
extern u8 D_8018F9B8;
extern u8 D_8018E680[];
extern u8 D_8018E681[];
extern u8 D_8018E682[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_8018FAB2;
g = D_8018F9DA;
b = D_8018F9B8;
D_8018E680[0] = r * 5 >> 3;
D_8018E681[0] = g << 3;
D_8018E682[0] = (s32)(b * 255) >> 4;
D_8018E680[4] = (s32)(r * 143) >> 4;
D_8018E681[4] = g * 25 >> 1;
D_8018E682[4] = (s32)(b * 255) >> 4;
D_8018E680[8] = (s32)(r * 255) >> 4;
D_8018E681[8] = (s32)(g * 255) >> 4;
D_8018E682[8] = (s32)(b * 255) >> 4;
D_8018E680[12] = (s32)(r * 143) >> 4;
D_8018E681[12] = (s32)(g * 255) >> 4;
D_8018E682[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_8018E680[i] = arg2[0x44] * D_8018FAB2 >> 4;
D_8018E681[i] = arg2[0x45] * D_8018F9DA >> 4;
D_8018E682[i] = arg2[0x46] * D_8018F9B8 >> 4;
i = (arg1 + 1) * 4;
D_8018E680[i] = arg2[0x47] * D_8018FAB2 >> 4;
D_8018E681[i] = arg2[0x48] * D_8018F9DA >> 4;
D_8018E682[i] = arg2[0x49] * D_8018F9B8 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_8018FAB2 << 3;
arg2[0x21] = D_8018F9DA << 3;
arg2[0x22] = D_8018F9B8 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_000/nonmatchings/ov_SC04_000", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801A8730/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801AAEE8 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801AAFE2;
extern u8 D_801AAF0A;
extern u8 D_801AAEE8;
extern u8 D_801A8730[];
extern u8 D_801A8731[];
extern u8 D_801A8732[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801AAFE2;
g = D_801AAF0A;
b = D_801AAEE8;
D_801A8730[0] = r * 5 >> 3;
D_801A8731[0] = g << 3;
D_801A8732[0] = (s32)(b * 255) >> 4;
D_801A8730[4] = (s32)(r * 143) >> 4;
D_801A8731[4] = g * 25 >> 1;
D_801A8732[4] = (s32)(b * 255) >> 4;
D_801A8730[8] = (s32)(r * 255) >> 4;
D_801A8731[8] = (s32)(g * 255) >> 4;
D_801A8732[8] = (s32)(b * 255) >> 4;
D_801A8730[12] = (s32)(r * 143) >> 4;
D_801A8731[12] = (s32)(g * 255) >> 4;
D_801A8732[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801A8730[i] = arg2[0x44] * D_801AAFE2 >> 4;
D_801A8731[i] = arg2[0x45] * D_801AAF0A >> 4;
D_801A8732[i] = arg2[0x46] * D_801AAEE8 >> 4;
i = (arg1 + 1) * 4;
D_801A8730[i] = arg2[0x47] * D_801AAFE2 >> 4;
D_801A8731[i] = arg2[0x48] * D_801AAF0A >> 4;
D_801A8732[i] = arg2[0x49] * D_801AAEE8 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801AAFE2 << 3;
arg2[0x21] = D_801AAF0A << 3;
arg2[0x22] = D_801AAEE8 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_002/nonmatchings/ov_SC04_002", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801BFA40/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801C15C0 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801C16C0;
extern u8 D_801C15E2;
extern u8 D_801C15C0;
extern u8 D_801BFA40[];
extern u8 D_801BFA41[];
extern u8 D_801BFA42[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801C16C0;
g = D_801C15E2;
b = D_801C15C0;
D_801BFA40[0] = r * 5 >> 3;
D_801BFA41[0] = g << 3;
D_801BFA42[0] = (s32)(b * 255) >> 4;
D_801BFA40[4] = (s32)(r * 143) >> 4;
D_801BFA41[4] = g * 25 >> 1;
D_801BFA42[4] = (s32)(b * 255) >> 4;
D_801BFA40[8] = (s32)(r * 255) >> 4;
D_801BFA41[8] = (s32)(g * 255) >> 4;
D_801BFA42[8] = (s32)(b * 255) >> 4;
D_801BFA40[12] = (s32)(r * 143) >> 4;
D_801BFA41[12] = (s32)(g * 255) >> 4;
D_801BFA42[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801BFA40[i] = arg2[0x44] * D_801C16C0 >> 4;
D_801BFA41[i] = arg2[0x45] * D_801C15E2 >> 4;
D_801BFA42[i] = arg2[0x46] * D_801C15C0 >> 4;
i = (arg1 + 1) * 4;
D_801BFA40[i] = arg2[0x47] * D_801C16C0 >> 4;
D_801BFA41[i] = arg2[0x48] * D_801C15E2 >> 4;
D_801BFA42[i] = arg2[0x49] * D_801C15C0 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801C16C0 << 3;
arg2[0x21] = D_801C15E2 << 3;
arg2[0x22] = D_801C15C0 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_003/nonmatchings/ov_SC04_003", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_8019A6A8/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_8019BA80 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_8019BB7A;
extern u8 D_8019BAA2;
extern u8 D_8019BA80;
extern u8 D_8019A6A8[];
extern u8 D_8019A6A9[];
extern u8 D_8019A6AA[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_8019BB7A;
g = D_8019BAA2;
b = D_8019BA80;
D_8019A6A8[0] = r * 5 >> 3;
D_8019A6A9[0] = g << 3;
D_8019A6AA[0] = (s32)(b * 255) >> 4;
D_8019A6A8[4] = (s32)(r * 143) >> 4;
D_8019A6A9[4] = g * 25 >> 1;
D_8019A6AA[4] = (s32)(b * 255) >> 4;
D_8019A6A8[8] = (s32)(r * 255) >> 4;
D_8019A6A9[8] = (s32)(g * 255) >> 4;
D_8019A6AA[8] = (s32)(b * 255) >> 4;
D_8019A6A8[12] = (s32)(r * 143) >> 4;
D_8019A6A9[12] = (s32)(g * 255) >> 4;
D_8019A6AA[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_8019A6A8[i] = arg2[0x44] * D_8019BB7A >> 4;
D_8019A6A9[i] = arg2[0x45] * D_8019BAA2 >> 4;
D_8019A6AA[i] = arg2[0x46] * D_8019BA80 >> 4;
i = (arg1 + 1) * 4;
D_8019A6A8[i] = arg2[0x47] * D_8019BB7A >> 4;
D_8019A6A9[i] = arg2[0x48] * D_8019BAA2 >> 4;
D_8019A6AA[i] = arg2[0x49] * D_8019BA80 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_8019BB7A << 3;
arg2[0x21] = D_8019BAA2 << 3;
arg2[0x22] = D_8019BA80 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_004/nonmatchings/ov_SC04_004", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801B24A8/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801B3AE0 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801B3BE0;
extern u8 D_801B3B02;
extern u8 D_801B3AE0;
extern u8 D_801B24A8[];
extern u8 D_801B24A9[];
extern u8 D_801B24AA[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801B3BE0;
g = D_801B3B02;
b = D_801B3AE0;
D_801B24A8[0] = r * 5 >> 3;
D_801B24A9[0] = g << 3;
D_801B24AA[0] = (s32)(b * 255) >> 4;
D_801B24A8[4] = (s32)(r * 143) >> 4;
D_801B24A9[4] = g * 25 >> 1;
D_801B24AA[4] = (s32)(b * 255) >> 4;
D_801B24A8[8] = (s32)(r * 255) >> 4;
D_801B24A9[8] = (s32)(g * 255) >> 4;
D_801B24AA[8] = (s32)(b * 255) >> 4;
D_801B24A8[12] = (s32)(r * 143) >> 4;
D_801B24A9[12] = (s32)(g * 255) >> 4;
D_801B24AA[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801B24A8[i] = arg2[0x44] * D_801B3BE0 >> 4;
D_801B24A9[i] = arg2[0x45] * D_801B3B02 >> 4;
D_801B24AA[i] = arg2[0x46] * D_801B3AE0 >> 4;
i = (arg1 + 1) * 4;
D_801B24A8[i] = arg2[0x47] * D_801B3BE0 >> 4;
D_801B24A9[i] = arg2[0x48] * D_801B3B02 >> 4;
D_801B24AA[i] = arg2[0x49] * D_801B3AE0 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801B3BE0 << 3;
arg2[0x21] = D_801B3B02 << 3;
arg2[0x22] = D_801B3AE0 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_005/nonmatchings/ov_SC04_005", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801C0680/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801C21C0 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801C22BA;
extern u8 D_801C21E2;
extern u8 D_801C21C0;
extern u8 D_801C0680[];
extern u8 D_801C0681[];
extern u8 D_801C0682[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801C22BA;
g = D_801C21E2;
b = D_801C21C0;
D_801C0680[0] = r * 5 >> 3;
D_801C0681[0] = g << 3;
D_801C0682[0] = (s32)(b * 255) >> 4;
D_801C0680[4] = (s32)(r * 143) >> 4;
D_801C0681[4] = g * 25 >> 1;
D_801C0682[4] = (s32)(b * 255) >> 4;
D_801C0680[8] = (s32)(r * 255) >> 4;
D_801C0681[8] = (s32)(g * 255) >> 4;
D_801C0682[8] = (s32)(b * 255) >> 4;
D_801C0680[12] = (s32)(r * 143) >> 4;
D_801C0681[12] = (s32)(g * 255) >> 4;
D_801C0682[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801C0680[i] = arg2[0x44] * D_801C22BA >> 4;
D_801C0681[i] = arg2[0x45] * D_801C21E2 >> 4;
D_801C0682[i] = arg2[0x46] * D_801C21C0 >> 4;
i = (arg1 + 1) * 4;
D_801C0680[i] = arg2[0x47] * D_801C22BA >> 4;
D_801C0681[i] = arg2[0x48] * D_801C21E2 >> 4;
D_801C0682[i] = arg2[0x49] * D_801C21C0 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801C22BA << 3;
arg2[0x21] = D_801C21E2 << 3;
arg2[0x22] = D_801C21C0 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_006/nonmatchings/ov_SC04_006", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_8019C928/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_8019DC60 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_8019DD5A;
extern u8 D_8019DC82;
extern u8 D_8019DC60;
extern u8 D_8019C928[];
extern u8 D_8019C929[];
extern u8 D_8019C92A[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_8019DD5A;
g = D_8019DC82;
b = D_8019DC60;
D_8019C928[0] = r * 5 >> 3;
D_8019C929[0] = g << 3;
D_8019C92A[0] = (s32)(b * 255) >> 4;
D_8019C928[4] = (s32)(r * 143) >> 4;
D_8019C929[4] = g * 25 >> 1;
D_8019C92A[4] = (s32)(b * 255) >> 4;
D_8019C928[8] = (s32)(r * 255) >> 4;
D_8019C929[8] = (s32)(g * 255) >> 4;
D_8019C92A[8] = (s32)(b * 255) >> 4;
D_8019C928[12] = (s32)(r * 143) >> 4;
D_8019C929[12] = (s32)(g * 255) >> 4;
D_8019C92A[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_8019C928[i] = arg2[0x44] * D_8019DD5A >> 4;
D_8019C929[i] = arg2[0x45] * D_8019DC82 >> 4;
D_8019C92A[i] = arg2[0x46] * D_8019DC60 >> 4;
i = (arg1 + 1) * 4;
D_8019C928[i] = arg2[0x47] * D_8019DD5A >> 4;
D_8019C929[i] = arg2[0x48] * D_8019DC82 >> 4;
D_8019C92A[i] = arg2[0x49] * D_8019DC60 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_8019DD5A << 3;
arg2[0x21] = D_8019DC82 << 3;
arg2[0x22] = D_8019DC60 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_007/nonmatchings/ov_SC04_007", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801B3EE0/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801B5A20 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801B5B1A;
extern u8 D_801B5A42;
extern u8 D_801B5A20;
extern u8 D_801B3EE0[];
extern u8 D_801B3EE1[];
extern u8 D_801B3EE2[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801B5B1A;
g = D_801B5A42;
b = D_801B5A20;
D_801B3EE0[0] = r * 5 >> 3;
D_801B3EE1[0] = g << 3;
D_801B3EE2[0] = (s32)(b * 255) >> 4;
D_801B3EE0[4] = (s32)(r * 143) >> 4;
D_801B3EE1[4] = g * 25 >> 1;
D_801B3EE2[4] = (s32)(b * 255) >> 4;
D_801B3EE0[8] = (s32)(r * 255) >> 4;
D_801B3EE1[8] = (s32)(g * 255) >> 4;
D_801B3EE2[8] = (s32)(b * 255) >> 4;
D_801B3EE0[12] = (s32)(r * 143) >> 4;
D_801B3EE1[12] = (s32)(g * 255) >> 4;
D_801B3EE2[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801B3EE0[i] = arg2[0x44] * D_801B5B1A >> 4;
D_801B3EE1[i] = arg2[0x45] * D_801B5A42 >> 4;
D_801B3EE2[i] = arg2[0x46] * D_801B5A20 >> 4;
i = (arg1 + 1) * 4;
D_801B3EE0[i] = arg2[0x47] * D_801B5B1A >> 4;
D_801B3EE1[i] = arg2[0x48] * D_801B5A42 >> 4;
D_801B3EE2[i] = arg2[0x49] * D_801B5A20 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801B5B1A << 3;
arg2[0x21] = D_801B5A42 << 3;
arg2[0x22] = D_801B5A20 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_008/nonmatchings/ov_SC04_008", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_80190BB8/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_80193350 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_8019344A;
extern u8 D_80193372;
extern u8 D_80193350;
extern u8 D_80190BB8[];
extern u8 D_80190BB9[];
extern u8 D_80190BBA[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_8019344A;
g = D_80193372;
b = D_80193350;
D_80190BB8[0] = r * 5 >> 3;
D_80190BB9[0] = g << 3;
D_80190BBA[0] = (s32)(b * 255) >> 4;
D_80190BB8[4] = (s32)(r * 143) >> 4;
D_80190BB9[4] = g * 25 >> 1;
D_80190BBA[4] = (s32)(b * 255) >> 4;
D_80190BB8[8] = (s32)(r * 255) >> 4;
D_80190BB9[8] = (s32)(g * 255) >> 4;
D_80190BBA[8] = (s32)(b * 255) >> 4;
D_80190BB8[12] = (s32)(r * 143) >> 4;
D_80190BB9[12] = (s32)(g * 255) >> 4;
D_80190BBA[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_80190BB8[i] = arg2[0x44] * D_8019344A >> 4;
D_80190BB9[i] = arg2[0x45] * D_80193372 >> 4;
D_80190BBA[i] = arg2[0x46] * D_80193350 >> 4;
i = (arg1 + 1) * 4;
D_80190BB8[i] = arg2[0x47] * D_8019344A >> 4;
D_80190BB9[i] = arg2[0x48] * D_80193372 >> 4;
D_80190BBA[i] = arg2[0x49] * D_80193350 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_8019344A << 3;
arg2[0x21] = D_80193372 << 3;
arg2[0x22] = D_80193350 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_009/nonmatchings/ov_SC04_009", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_80196DC8/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_80198100 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801981FA;
extern u8 D_80198122;
extern u8 D_80198100;
extern u8 D_80196DC8[];
extern u8 D_80196DC9[];
extern u8 D_80196DCA[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801981FA;
g = D_80198122;
b = D_80198100;
D_80196DC8[0] = r * 5 >> 3;
D_80196DC9[0] = g << 3;
D_80196DCA[0] = (s32)(b * 255) >> 4;
D_80196DC8[4] = (s32)(r * 143) >> 4;
D_80196DC9[4] = g * 25 >> 1;
D_80196DCA[4] = (s32)(b * 255) >> 4;
D_80196DC8[8] = (s32)(r * 255) >> 4;
D_80196DC9[8] = (s32)(g * 255) >> 4;
D_80196DCA[8] = (s32)(b * 255) >> 4;
D_80196DC8[12] = (s32)(r * 143) >> 4;
D_80196DC9[12] = (s32)(g * 255) >> 4;
D_80196DCA[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_80196DC8[i] = arg2[0x44] * D_801981FA >> 4;
D_80196DC9[i] = arg2[0x45] * D_80198122 >> 4;
D_80196DCA[i] = arg2[0x46] * D_80198100 >> 4;
i = (arg1 + 1) * 4;
D_80196DC8[i] = arg2[0x47] * D_801981FA >> 4;
D_80196DC9[i] = arg2[0x48] * D_80198122 >> 4;
D_80196DCA[i] = arg2[0x49] * D_80198100 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801981FA << 3;
arg2[0x21] = D_80198122 << 3;
arg2[0x22] = D_80198100 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_010/nonmatchings/ov_SC04_010", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_8018E260/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_8018F5D8 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_8018F6D2;
extern u8 D_8018F5FA;
extern u8 D_8018F5D8;
extern u8 D_8018E260[];
extern u8 D_8018E261[];
extern u8 D_8018E262[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_8018F6D2;
g = D_8018F5FA;
b = D_8018F5D8;
D_8018E260[0] = r * 5 >> 3;
D_8018E261[0] = g << 3;
D_8018E262[0] = (s32)(b * 255) >> 4;
D_8018E260[4] = (s32)(r * 143) >> 4;
D_8018E261[4] = g * 25 >> 1;
D_8018E262[4] = (s32)(b * 255) >> 4;
D_8018E260[8] = (s32)(r * 255) >> 4;
D_8018E261[8] = (s32)(g * 255) >> 4;
D_8018E262[8] = (s32)(b * 255) >> 4;
D_8018E260[12] = (s32)(r * 143) >> 4;
D_8018E261[12] = (s32)(g * 255) >> 4;
D_8018E262[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_8018E260[i] = arg2[0x44] * D_8018F6D2 >> 4;
D_8018E261[i] = arg2[0x45] * D_8018F5FA >> 4;
D_8018E262[i] = arg2[0x46] * D_8018F5D8 >> 4;
i = (arg1 + 1) * 4;
D_8018E260[i] = arg2[0x47] * D_8018F6D2 >> 4;
D_8018E261[i] = arg2[0x48] * D_8018F5FA >> 4;
D_8018E262[i] = arg2[0x49] * D_8018F5D8 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_8018F6D2 << 3;
arg2[0x21] = D_8018F5FA << 3;
arg2[0x22] = D_8018F5D8 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_011/nonmatchings/ov_SC04_011", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801ED9C8/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801F1540 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801F163A;
extern u8 D_801F1562;
extern u8 D_801F1540;
extern u8 D_801ED9C8[];
extern u8 D_801ED9C9[];
extern u8 D_801ED9CA[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801F163A;
g = D_801F1562;
b = D_801F1540;
D_801ED9C8[0] = r * 5 >> 3;
D_801ED9C9[0] = g << 3;
D_801ED9CA[0] = (s32)(b * 255) >> 4;
D_801ED9C8[4] = (s32)(r * 143) >> 4;
D_801ED9C9[4] = g * 25 >> 1;
D_801ED9CA[4] = (s32)(b * 255) >> 4;
D_801ED9C8[8] = (s32)(r * 255) >> 4;
D_801ED9C9[8] = (s32)(g * 255) >> 4;
D_801ED9CA[8] = (s32)(b * 255) >> 4;
D_801ED9C8[12] = (s32)(r * 143) >> 4;
D_801ED9C9[12] = (s32)(g * 255) >> 4;
D_801ED9CA[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801ED9C8[i] = arg2[0x44] * D_801F163A >> 4;
D_801ED9C9[i] = arg2[0x45] * D_801F1562 >> 4;
D_801ED9CA[i] = arg2[0x46] * D_801F1540 >> 4;
i = (arg1 + 1) * 4;
D_801ED9C8[i] = arg2[0x47] * D_801F163A >> 4;
D_801ED9C9[i] = arg2[0x48] * D_801F1562 >> 4;
D_801ED9CA[i] = arg2[0x49] * D_801F1540 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801F163A << 3;
arg2[0x21] = D_801F1562 << 3;
arg2[0x22] = D_801F1540 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_012/nonmatchings/ov_SC04_012", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_80190938/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801930D0 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801931CA;
extern u8 D_801930F2;
extern u8 D_801930D0;
extern u8 D_80190938[];
extern u8 D_80190939[];
extern u8 D_8019093A[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801931CA;
g = D_801930F2;
b = D_801930D0;
D_80190938[0] = r * 5 >> 3;
D_80190939[0] = g << 3;
D_8019093A[0] = (s32)(b * 255) >> 4;
D_80190938[4] = (s32)(r * 143) >> 4;
D_80190939[4] = g * 25 >> 1;
D_8019093A[4] = (s32)(b * 255) >> 4;
D_80190938[8] = (s32)(r * 255) >> 4;
D_80190939[8] = (s32)(g * 255) >> 4;
D_8019093A[8] = (s32)(b * 255) >> 4;
D_80190938[12] = (s32)(r * 143) >> 4;
D_80190939[12] = (s32)(g * 255) >> 4;
D_8019093A[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_80190938[i] = arg2[0x44] * D_801931CA >> 4;
D_80190939[i] = arg2[0x45] * D_801930F2 >> 4;
D_8019093A[i] = arg2[0x46] * D_801930D0 >> 4;
i = (arg1 + 1) * 4;
D_80190938[i] = arg2[0x47] * D_801931CA >> 4;
D_80190939[i] = arg2[0x48] * D_801930F2 >> 4;
D_8019093A[i] = arg2[0x49] * D_801930D0 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801931CA << 3;
arg2[0x21] = D_801930F2 << 3;
arg2[0x22] = D_801930D0 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_015/nonmatchings/ov_SC04_015", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801C7620/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801C9240 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801C936A;
extern u8 D_801C926A;
extern u8 D_801C9240;
extern u8 D_801C7620[];
extern u8 D_801C7621[];
extern u8 D_801C7622[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801C936A;
g = D_801C926A;
b = D_801C9240;
D_801C7620[0] = r * 5 >> 3;
D_801C7621[0] = g << 3;
D_801C7622[0] = (s32)(b * 255) >> 4;
D_801C7620[4] = (s32)(r * 143) >> 4;
D_801C7621[4] = g * 25 >> 1;
D_801C7622[4] = (s32)(b * 255) >> 4;
D_801C7620[8] = (s32)(r * 255) >> 4;
D_801C7621[8] = (s32)(g * 255) >> 4;
D_801C7622[8] = (s32)(b * 255) >> 4;
D_801C7620[12] = (s32)(r * 143) >> 4;
D_801C7621[12] = (s32)(g * 255) >> 4;
D_801C7622[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801C7620[i] = arg2[0x44] * D_801C936A >> 4;
D_801C7621[i] = arg2[0x45] * D_801C926A >> 4;
D_801C7622[i] = arg2[0x46] * D_801C9240 >> 4;
i = (arg1 + 1) * 4;
D_801C7620[i] = arg2[0x47] * D_801C936A >> 4;
D_801C7621[i] = arg2[0x48] * D_801C926A >> 4;
D_801C7622[i] = arg2[0x49] * D_801C9240 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801C936A << 3;
arg2[0x21] = D_801C926A << 3;
arg2[0x22] = D_801C9240 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_016/nonmatchings/ov_SC04_016", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801A57C8/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801AC170 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801AC26A;
extern u8 D_801AC192;
extern u8 D_801AC170;
extern u8 D_801A57C8[];
extern u8 D_801A57C9[];
extern u8 D_801A57CA[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801AC26A;
g = D_801AC192;
b = D_801AC170;
D_801A57C8[0] = r * 5 >> 3;
D_801A57C9[0] = g << 3;
D_801A57CA[0] = (s32)(b * 255) >> 4;
D_801A57C8[4] = (s32)(r * 143) >> 4;
D_801A57C9[4] = g * 25 >> 1;
D_801A57CA[4] = (s32)(b * 255) >> 4;
D_801A57C8[8] = (s32)(r * 255) >> 4;
D_801A57C9[8] = (s32)(g * 255) >> 4;
D_801A57CA[8] = (s32)(b * 255) >> 4;
D_801A57C8[12] = (s32)(r * 143) >> 4;
D_801A57C9[12] = (s32)(g * 255) >> 4;
D_801A57CA[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801A57C8[i] = arg2[0x44] * D_801AC26A >> 4;
D_801A57C9[i] = arg2[0x45] * D_801AC192 >> 4;
D_801A57CA[i] = arg2[0x46] * D_801AC170 >> 4;
i = (arg1 + 1) * 4;
D_801A57C8[i] = arg2[0x47] * D_801AC26A >> 4;
D_801A57C9[i] = arg2[0x48] * D_801AC192 >> 4;
D_801A57CA[i] = arg2[0x49] * D_801AC170 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801AC26A << 3;
arg2[0x21] = D_801AC192 << 3;
arg2[0x22] = D_801AC170 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_018/nonmatchings/ov_SC04_018", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801E5A18/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801E7988 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801E7AB2;
extern u8 D_801E79B2;
extern u8 D_801E7988;
extern u8 D_801E5A18[];
extern u8 D_801E5A19[];
extern u8 D_801E5A1A[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801E7AB2;
g = D_801E79B2;
b = D_801E7988;
D_801E5A18[0] = r * 5 >> 3;
D_801E5A19[0] = g << 3;
D_801E5A1A[0] = (s32)(b * 255) >> 4;
D_801E5A18[4] = (s32)(r * 143) >> 4;
D_801E5A19[4] = g * 25 >> 1;
D_801E5A1A[4] = (s32)(b * 255) >> 4;
D_801E5A18[8] = (s32)(r * 255) >> 4;
D_801E5A19[8] = (s32)(g * 255) >> 4;
D_801E5A1A[8] = (s32)(b * 255) >> 4;
D_801E5A18[12] = (s32)(r * 143) >> 4;
D_801E5A19[12] = (s32)(g * 255) >> 4;
D_801E5A1A[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801E5A18[i] = arg2[0x44] * D_801E7AB2 >> 4;
D_801E5A19[i] = arg2[0x45] * D_801E79B2 >> 4;
D_801E5A1A[i] = arg2[0x46] * D_801E7988 >> 4;
i = (arg1 + 1) * 4;
D_801E5A18[i] = arg2[0x47] * D_801E7AB2 >> 4;
D_801E5A19[i] = arg2[0x48] * D_801E79B2 >> 4;
D_801E5A1A[i] = arg2[0x49] * D_801E7988 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801E7AB2 << 3;
arg2[0x21] = D_801E79B2 << 3;
arg2[0x22] = D_801E7988 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */
+78 -1
View File
@@ -560,7 +560,84 @@ void func_8012956C(void) {
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298F4 (src/shared) */
INCLUDE_ASM("asm/ov_SC04_019/nonmatchings/ov_SC04_019", func_801299C8);
// @class: schedule
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
//
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801E5A18/1/2[] (a 4-row x 3-comp
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801E7988 load.
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
extern u8 D_801E7AB2;
extern u8 D_801E79B2;
extern u8 D_801E7988;
extern u8 D_801E5A18[];
extern u8 D_801E5A19[];
extern u8 D_801E5A1A[];
void func_801299C8(arg0, arg1, arg2)
s16 arg0;
s16 arg1;
u8 *arg2;
{
u32 r;
u32 g;
u32 b;
s32 i;
if (arg0 != 0) {
switch (arg1) {
case 1:
r = D_801E7AB2;
g = D_801E79B2;
b = D_801E7988;
D_801E5A18[0] = r * 5 >> 3;
D_801E5A19[0] = g << 3;
D_801E5A1A[0] = (s32)(b * 255) >> 4;
D_801E5A18[4] = (s32)(r * 143) >> 4;
D_801E5A19[4] = g * 25 >> 1;
D_801E5A1A[4] = (s32)(b * 255) >> 4;
D_801E5A18[8] = (s32)(r * 255) >> 4;
D_801E5A19[8] = (s32)(g * 255) >> 4;
D_801E5A1A[8] = (s32)(b * 255) >> 4;
D_801E5A18[12] = (s32)(r * 143) >> 4;
D_801E5A19[12] = (s32)(g * 255) >> 4;
D_801E5A1A[12] = (s32)(b * 45) >> 2;
break;
case 0:
case 2:
i = arg1 * 4;
D_801E5A18[i] = arg2[0x44] * D_801E7AB2 >> 4;
D_801E5A19[i] = arg2[0x45] * D_801E79B2 >> 4;
D_801E5A1A[i] = arg2[0x46] * D_801E7988 >> 4;
i = (arg1 + 1) * 4;
D_801E5A18[i] = arg2[0x47] * D_801E7AB2 >> 4;
D_801E5A19[i] = arg2[0x48] * D_801E79B2 >> 4;
D_801E5A1A[i] = arg2[0x49] * D_801E7988 >> 4;
break;
case 3:
case 4:
arg2[0x20] = D_801E7AB2 << 3;
arg2[0x21] = D_801E79B2 << 3;
arg2[0x22] = D_801E7988 << 3;
break;
}
}
}
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129C40 (src/shared) */