mirror of
https://github.com/Druthulu/BFM-decomp
synced 2026-09-28 06:49:47 -04:00
feat(phase-30 S48-0b): jtbl family func_801299C8 — 3/3 blocked-binary siblings banked
This commit is contained in:
+3
-3
@@ -4915,7 +4915,7 @@ ov_MAIN_012_ELF := $(ov_MAIN_012_OUT).elf
|
||||
ov_MAIN_012_MAPFILE := $(ov_MAIN_012_OUT).map
|
||||
ov_MAIN_012_LD_SCRIPT := $(ov_MAIN_012_OUT).ld
|
||||
ov_MAIN_012_SPLAT_YAML := config/splat.ov_MAIN_012.yaml
|
||||
ov_MAIN_012_JTBL_INTERLEAVE := --order tail.data.o,ov_MAIN_012_jr_80131340.o,tail2.data.o,ov_MAIN_012_jr_80135A4C.o,tail3.data.o,ov_MAIN_012_jr_80135EB0.o,tail4.data.o,ov_MAIN_012_jr_801380E0.o,tail5.data.o,ov_MAIN_012_jr_8013F350.o,tail6.data.o,ov_MAIN_012_jr_80159C84.o,tail7.data.o,ov_MAIN_012_jr_8015A3C8.o,tail8.data.o,ov_MAIN_012_jr_801789AC.o,tail9.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve
|
||||
ov_MAIN_012_JTBL_INTERLEAVE := --order tail.data.o,ov_MAIN_012.o,tail2.data.o,ov_MAIN_012_jr_80131340.o,tail3.data.o,ov_MAIN_012_jr_80135A4C.o,tail4.data.o,ov_MAIN_012_jr_80135EB0.o,tail5.data.o,ov_MAIN_012_jr_801380E0.o,tail6.data.o,ov_MAIN_012_jr_8013F350.o,tail7.data.o,ov_MAIN_012_jr_80159C84.o,tail8.data.o,ov_MAIN_012_jr_8015A3C8.o,tail9.data.o,ov_MAIN_012_jr_801789AC.o,tail10.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve
|
||||
build/src/ov_MAIN_012/ov_MAIN_012_jr_8013F350.o: JTBL_PADS := 0,0,4,0,4,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20,+0x38,+0x58,+0x70,+0x90
|
||||
build/src/ov_MAIN_012/ov_MAIN_012_jr_80159C84.o: JTBL_PADS := 0,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20
|
||||
build/src/ov_MAIN_012/ov_MAIN_012_jr_801789AC.o: JTBL_PADS := 0,0,0,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x18,+0x190,+0x220
|
||||
@@ -4940,7 +4940,7 @@ ov_SC02_037_ELF := $(ov_SC02_037_OUT).elf
|
||||
ov_SC02_037_MAPFILE := $(ov_SC02_037_OUT).map
|
||||
ov_SC02_037_LD_SCRIPT := $(ov_SC02_037_OUT).ld
|
||||
ov_SC02_037_SPLAT_YAML := config/splat.ov_SC02_037.yaml
|
||||
ov_SC02_037_JTBL_INTERLEAVE := --order tail.data.o,ov_SC02_037_jr_80131340.o,tail2.data.o,ov_SC02_037_jr_80135A4C.o,tail3.data.o,ov_SC02_037_jr_80135EB0.o,tail4.data.o,ov_SC02_037_jr_801380E0.o,tail5.data.o,ov_SC02_037_jr_8013F350.o,tail6.data.o,ov_SC02_037_jr_80159C84.o,tail7.data.o,ov_SC02_037_jr_8015A3C8.o,tail8.data.o,ov_SC02_037_jr_801789AC.o,tail9.data.o,ov_SC02_037_jr_8017AE2C.o,tail10.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve
|
||||
ov_SC02_037_JTBL_INTERLEAVE := --order tail.data.o,ov_SC02_037.o,tail2.data.o,ov_SC02_037_jr_80131340.o,tail3.data.o,ov_SC02_037_jr_80135A4C.o,tail4.data.o,ov_SC02_037_jr_80135EB0.o,tail5.data.o,ov_SC02_037_jr_801380E0.o,tail6.data.o,ov_SC02_037_jr_8013F350.o,tail7.data.o,ov_SC02_037_jr_80159C84.o,tail8.data.o,ov_SC02_037_jr_8015A3C8.o,tail9.data.o,ov_SC02_037_jr_801789AC.o,tail10.data.o,ov_SC02_037_jr_8017AE2C.o,tail11.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve
|
||||
build/src/ov_SC02_037/ov_SC02_037_jr_8013F350.o: JTBL_PADS := 0,0,4,0,4,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20,+0x38,+0x58,+0x70,+0x90
|
||||
build/src/ov_SC02_037/ov_SC02_037_jr_80159C84.o: JTBL_PADS := 0,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20
|
||||
build/src/ov_SC02_037/ov_SC02_037_jr_8015A3C8.o: JTBL_PADS := 0,4,4,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20,+0x40,+0x60
|
||||
@@ -4966,7 +4966,7 @@ ov_SC03_107_ELF := $(ov_SC03_107_OUT).elf
|
||||
ov_SC03_107_MAPFILE := $(ov_SC03_107_OUT).map
|
||||
ov_SC03_107_LD_SCRIPT := $(ov_SC03_107_OUT).ld
|
||||
ov_SC03_107_SPLAT_YAML := config/splat.ov_SC03_107.yaml
|
||||
ov_SC03_107_JTBL_INTERLEAVE := --order tail.data.o,ov_SC03_107_jr_80131340.o,tail2.data.o,ov_SC03_107_jr_80135A4C.o,tail3.data.o,ov_SC03_107_jr_80135EB0.o,tail4.data.o,ov_SC03_107_jr_801380E0.o,tail5.data.o,ov_SC03_107_jr_8013F350.o,tail6.data.o,ov_SC03_107_jr_80159C84.o,tail7.data.o,ov_SC03_107_jr_8015A3C8.o,tail8.data.o,ov_SC03_107_jr_801789AC.o,tail9.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve
|
||||
ov_SC03_107_JTBL_INTERLEAVE := --order tail.data.o,ov_SC03_107.o,tail2.data.o,ov_SC03_107_jr_80131340.o,tail3.data.o,ov_SC03_107_jr_80135A4C.o,tail4.data.o,ov_SC03_107_jr_80135EB0.o,tail5.data.o,ov_SC03_107_jr_801380E0.o,tail6.data.o,ov_SC03_107_jr_8013F350.o,tail7.data.o,ov_SC03_107_jr_80159C84.o,tail8.data.o,ov_SC03_107_jr_8015A3C8.o,tail9.data.o,ov_SC03_107_jr_801789AC.o,tail10.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve
|
||||
build/src/ov_SC03_107/ov_SC03_107_jr_8013F350.o: JTBL_PADS := 0,0,4,0,4,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20,+0x38,+0x58,+0x70,+0x90
|
||||
build/src/ov_SC03_107/ov_SC03_107_jr_80159C84.o: JTBL_PADS := 0,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20
|
||||
build/src/ov_SC03_107/ov_SC03_107_jr_801789AC.o: JTBL_PADS := 0,0,0,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x18,+0x190,+0x220
|
||||
|
||||
@@ -103,22 +103,24 @@ segments:
|
||||
- [0x32270, c, ov_MAIN_012_jr_8015A3C8]
|
||||
- [0x50854, c, ov_MAIN_012_jr_801789AC]
|
||||
- [0x561e0, data, tail]
|
||||
- [0x5abd0, .rodata, ov_MAIN_012] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x5abe4, data, tail2]
|
||||
- [0x5ad78, .rodata, ov_MAIN_012_jr_80131340] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x5ad8c, data, tail2]
|
||||
- [0x5ad8c, data, tail3]
|
||||
- [0x5adbc, .rodata, ov_MAIN_012_jr_80135A4C] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x5add0, data, tail3]
|
||||
- [0x5add0, data, tail4]
|
||||
- [0x5adec, .rodata, ov_MAIN_012_jr_80135EB0] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x5ae18, data, tail4]
|
||||
- [0x5ae18, data, tail5]
|
||||
- [0x5ae24, .rodata, ov_MAIN_012_jr_801380E0] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x5ae88, data, tail5]
|
||||
- [0x5ae88, data, tail6]
|
||||
- [0x5b494, .rodata, ov_MAIN_012_jr_8013F350] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x5b538, data, tail6]
|
||||
- [0x5b538, data, tail7]
|
||||
- [0x5b730, .rodata, ov_MAIN_012_jr_80159C84] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x5b764, data, tail7]
|
||||
- [0x5b764, data, tail8]
|
||||
- [0x5b768, .rodata, ov_MAIN_012_jr_8015A3C8] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x5b784, data, tail8]
|
||||
- [0x5b784, data, tail9]
|
||||
- [0x5ba40, .rodata, ov_MAIN_012_jr_801789AC] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x5bcec, data, tail9]
|
||||
- [0x5bcec, data, tail10]
|
||||
- [0x5DB24, bin, trailing] # final 3 bytes (EOF not 4-aligned; spimdisasm drops a partial word)
|
||||
- [0x5DB27] # EOF marker = the 0.4.dec byte length
|
||||
# @TRAILING@ (above) is replaced by tools/new_overlay.sh: for a non-4-aligned overlay it becomes
|
||||
|
||||
@@ -104,24 +104,26 @@ segments:
|
||||
- [0x50854, c, ov_SC02_037_jr_801789AC]
|
||||
- [0x52cd4, c, ov_SC02_037_jr_8017AE2C]
|
||||
- [0x5ba68, data, tail]
|
||||
- [0x9f0b4, .rodata, ov_SC02_037] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x9f0c8, data, tail2]
|
||||
- [0x9f25c, .rodata, ov_SC02_037_jr_80131340] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x9f270, data, tail2]
|
||||
- [0x9f270, data, tail3]
|
||||
- [0x9f2a0, .rodata, ov_SC02_037_jr_80135A4C] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x9f2b4, data, tail3]
|
||||
- [0x9f2b4, data, tail4]
|
||||
- [0x9f2d0, .rodata, ov_SC02_037_jr_80135EB0] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x9f2fc, data, tail4]
|
||||
- [0x9f2fc, data, tail5]
|
||||
- [0x9f308, .rodata, ov_SC02_037_jr_801380E0] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x9f36c, data, tail5]
|
||||
- [0x9f36c, data, tail6]
|
||||
- [0x9f978, .rodata, ov_SC02_037_jr_8013F350] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x9fa1c, data, tail6]
|
||||
- [0x9fa1c, data, tail7]
|
||||
- [0x9fc14, .rodata, ov_SC02_037_jr_80159C84] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x9fc48, data, tail7]
|
||||
- [0x9fc48, data, tail8]
|
||||
- [0x9fc4c, .rodata, ov_SC02_037_jr_8015A3C8] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x9fcc8, data, tail8]
|
||||
- [0x9fcc8, data, tail9]
|
||||
- [0x9ff24, .rodata, ov_SC02_037_jr_801789AC] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x9ff3c, data, tail9]
|
||||
- [0x9ff3c, data, tail10]
|
||||
- [0xa01d4, .rodata, ov_SC02_037_jr_8017AE2C] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0xa0238, data, tail10]
|
||||
- [0xa0238, data, tail11]
|
||||
- [0xA198C, bin, trailing] # final 3 bytes (EOF not 4-aligned; spimdisasm drops a partial word)
|
||||
- [0xA198F] # EOF marker = the 0.4.dec byte length
|
||||
# @TRAILING@ (above) is replaced by tools/new_overlay.sh: for a non-4-aligned overlay it becomes
|
||||
|
||||
@@ -103,22 +103,24 @@ segments:
|
||||
- [0x32270, c, ov_SC03_107_jr_8015A3C8]
|
||||
- [0x50854, c, ov_SC03_107_jr_801789AC]
|
||||
- [0x5a304, data, tail]
|
||||
- [0x713e8, .rodata, ov_SC03_107] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x713fc, data, tail2]
|
||||
- [0x71590, .rodata, ov_SC03_107_jr_80131340] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x715a4, data, tail2]
|
||||
- [0x715a4, data, tail3]
|
||||
- [0x715d4, .rodata, ov_SC03_107_jr_80135A4C] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x715e8, data, tail3]
|
||||
- [0x715e8, data, tail4]
|
||||
- [0x71604, .rodata, ov_SC03_107_jr_80135EB0] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x71630, data, tail4]
|
||||
- [0x71630, data, tail5]
|
||||
- [0x7163c, .rodata, ov_SC03_107_jr_801380E0] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x716a0, data, tail5]
|
||||
- [0x716a0, data, tail6]
|
||||
- [0x71cac, .rodata, ov_SC03_107_jr_8013F350] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x71d50, data, tail6]
|
||||
- [0x71d50, data, tail7]
|
||||
- [0x71f48, .rodata, ov_SC03_107_jr_80159C84] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x71f7c, data, tail7]
|
||||
- [0x71f7c, data, tail8]
|
||||
- [0x71f80, .rodata, ov_SC03_107_jr_8015A3C8] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x71f9c, data, tail8]
|
||||
- [0x71f9c, data, tail9]
|
||||
- [0x72258, .rodata, ov_SC03_107_jr_801789AC] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py)
|
||||
- [0x72504, data, tail9]
|
||||
- [0x72504, data, tail10]
|
||||
- [0x73BE4, bin, trailing] # final 3 bytes (EOF not 4-aligned; spimdisasm drops a partial word)
|
||||
- [0x73BE7] # EOF marker = the 0.4.dec byte length
|
||||
# @TRAILING@ (above) is replaced by tools/new_overlay.sh: for a non-4-aligned overlay it becomes
|
||||
|
||||
@@ -640,7 +640,86 @@ void func_8012956C(void) {
|
||||
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298f4 (src/shared) */
|
||||
|
||||
|
||||
INCLUDE_ASM("asm/ov_MAIN_012/nonmatchings/ov_MAIN_012", func_801299C8);
|
||||
|
||||
|
||||
|
||||
// @class: schedule
|
||||
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
|
||||
//
|
||||
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
|
||||
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
|
||||
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
|
||||
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
|
||||
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801840B8/1/2[] (a 4-row x 3-comp
|
||||
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
|
||||
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
|
||||
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
|
||||
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
|
||||
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
|
||||
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
|
||||
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
|
||||
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
|
||||
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
|
||||
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_80185B10 load.
|
||||
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
|
||||
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
|
||||
|
||||
extern u8 D_80185C0A;
|
||||
extern u8 D_80185B32;
|
||||
extern u8 D_80185B10;
|
||||
extern u8 D_801840B8[];
|
||||
extern u8 D_801840B9[];
|
||||
extern u8 D_801840BA[];
|
||||
|
||||
void func_801299C8(arg0, arg1, arg2)
|
||||
s16 arg0;
|
||||
s16 arg1;
|
||||
u8 *arg2;
|
||||
{
|
||||
u32 r;
|
||||
u32 g;
|
||||
u32 b;
|
||||
s32 i;
|
||||
|
||||
if (arg0 != 0) {
|
||||
switch (arg1) {
|
||||
case 1:
|
||||
r = D_80185C0A;
|
||||
g = D_80185B32;
|
||||
b = D_80185B10;
|
||||
D_801840B8[0] = r * 5 >> 3;
|
||||
D_801840B9[0] = g << 3;
|
||||
D_801840BA[0] = (s32)(b * 255) >> 4;
|
||||
D_801840B8[4] = (s32)(r * 143) >> 4;
|
||||
D_801840B9[4] = g * 25 >> 1;
|
||||
D_801840BA[4] = (s32)(b * 255) >> 4;
|
||||
D_801840B8[8] = (s32)(r * 255) >> 4;
|
||||
D_801840B9[8] = (s32)(g * 255) >> 4;
|
||||
D_801840BA[8] = (s32)(b * 255) >> 4;
|
||||
D_801840B8[12] = (s32)(r * 143) >> 4;
|
||||
D_801840B9[12] = (s32)(g * 255) >> 4;
|
||||
D_801840BA[12] = (s32)(b * 45) >> 2;
|
||||
break;
|
||||
case 0:
|
||||
case 2:
|
||||
i = arg1 * 4;
|
||||
D_801840B8[i] = arg2[0x44] * D_80185C0A >> 4;
|
||||
D_801840B9[i] = arg2[0x45] * D_80185B32 >> 4;
|
||||
D_801840BA[i] = arg2[0x46] * D_80185B10 >> 4;
|
||||
i = (arg1 + 1) * 4;
|
||||
D_801840B8[i] = arg2[0x47] * D_80185C0A >> 4;
|
||||
D_801840B9[i] = arg2[0x48] * D_80185B32 >> 4;
|
||||
D_801840BA[i] = arg2[0x49] * D_80185B10 >> 4;
|
||||
break;
|
||||
case 3:
|
||||
case 4:
|
||||
arg2[0x20] = D_80185C0A << 3;
|
||||
arg2[0x21] = D_80185B32 << 3;
|
||||
arg2[0x22] = D_80185B10 << 3;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129c40 (src/shared) */
|
||||
|
||||
|
||||
@@ -623,7 +623,86 @@ void func_8012956C(void) {
|
||||
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298f4 (src/shared) */
|
||||
|
||||
|
||||
INCLUDE_ASM("asm/ov_SC02_037/nonmatchings/ov_SC02_037", func_801299C8);
|
||||
|
||||
|
||||
|
||||
// @class: schedule
|
||||
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
|
||||
//
|
||||
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
|
||||
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
|
||||
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
|
||||
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
|
||||
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801C83B0/1/2[] (a 4-row x 3-comp
|
||||
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
|
||||
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
|
||||
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
|
||||
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
|
||||
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
|
||||
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
|
||||
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
|
||||
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
|
||||
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
|
||||
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801C9978 load.
|
||||
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
|
||||
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
|
||||
|
||||
extern u8 D_801C9A72;
|
||||
extern u8 D_801C999A;
|
||||
extern u8 D_801C9978;
|
||||
extern u8 D_801C83B0[];
|
||||
extern u8 D_801C83B1[];
|
||||
extern u8 D_801C83B2[];
|
||||
|
||||
void func_801299C8(arg0, arg1, arg2)
|
||||
s16 arg0;
|
||||
s16 arg1;
|
||||
u8 *arg2;
|
||||
{
|
||||
u32 r;
|
||||
u32 g;
|
||||
u32 b;
|
||||
s32 i;
|
||||
|
||||
if (arg0 != 0) {
|
||||
switch (arg1) {
|
||||
case 1:
|
||||
r = D_801C9A72;
|
||||
g = D_801C999A;
|
||||
b = D_801C9978;
|
||||
D_801C83B0[0] = r * 5 >> 3;
|
||||
D_801C83B1[0] = g << 3;
|
||||
D_801C83B2[0] = (s32)(b * 255) >> 4;
|
||||
D_801C83B0[4] = (s32)(r * 143) >> 4;
|
||||
D_801C83B1[4] = g * 25 >> 1;
|
||||
D_801C83B2[4] = (s32)(b * 255) >> 4;
|
||||
D_801C83B0[8] = (s32)(r * 255) >> 4;
|
||||
D_801C83B1[8] = (s32)(g * 255) >> 4;
|
||||
D_801C83B2[8] = (s32)(b * 255) >> 4;
|
||||
D_801C83B0[12] = (s32)(r * 143) >> 4;
|
||||
D_801C83B1[12] = (s32)(g * 255) >> 4;
|
||||
D_801C83B2[12] = (s32)(b * 45) >> 2;
|
||||
break;
|
||||
case 0:
|
||||
case 2:
|
||||
i = arg1 * 4;
|
||||
D_801C83B0[i] = arg2[0x44] * D_801C9A72 >> 4;
|
||||
D_801C83B1[i] = arg2[0x45] * D_801C999A >> 4;
|
||||
D_801C83B2[i] = arg2[0x46] * D_801C9978 >> 4;
|
||||
i = (arg1 + 1) * 4;
|
||||
D_801C83B0[i] = arg2[0x47] * D_801C9A72 >> 4;
|
||||
D_801C83B1[i] = arg2[0x48] * D_801C999A >> 4;
|
||||
D_801C83B2[i] = arg2[0x49] * D_801C9978 >> 4;
|
||||
break;
|
||||
case 3:
|
||||
case 4:
|
||||
arg2[0x20] = D_801C9A72 << 3;
|
||||
arg2[0x21] = D_801C999A << 3;
|
||||
arg2[0x22] = D_801C9978 << 3;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129c40 (src/shared) */
|
||||
|
||||
|
||||
@@ -640,7 +640,86 @@ void func_8012956C(void) {
|
||||
DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298f4 (src/shared) */
|
||||
|
||||
|
||||
INCLUDE_ASM("asm/ov_SC03_107/nonmatchings/ov_SC03_107", func_801299C8);
|
||||
|
||||
|
||||
|
||||
// @class: schedule
|
||||
// @stuck: none — MATCH (158 ins, match_one relocation-masked)
|
||||
//
|
||||
// Levers that landed it (2 iterations, 56 mismatched -> MATCH):
|
||||
// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the
|
||||
// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`)
|
||||
// stashed in the jtbl branch delay slot and RE-extended per use in the case body.
|
||||
// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_8019A710/1/2[] (a 4-row x 3-comp
|
||||
// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)`
|
||||
// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case.
|
||||
// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block,
|
||||
// so `case 1:` must be written before `case 0/2:` and `case 3/4:`.
|
||||
// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0],
|
||||
// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS
|
||||
// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays
|
||||
// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE
|
||||
// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting
|
||||
// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_8019BBD0 load.
|
||||
// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit
|
||||
// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block).
|
||||
|
||||
extern u8 D_8019BCCA;
|
||||
extern u8 D_8019BBF2;
|
||||
extern u8 D_8019BBD0;
|
||||
extern u8 D_8019A710[];
|
||||
extern u8 D_8019A711[];
|
||||
extern u8 D_8019A712[];
|
||||
|
||||
void func_801299C8(arg0, arg1, arg2)
|
||||
s16 arg0;
|
||||
s16 arg1;
|
||||
u8 *arg2;
|
||||
{
|
||||
u32 r;
|
||||
u32 g;
|
||||
u32 b;
|
||||
s32 i;
|
||||
|
||||
if (arg0 != 0) {
|
||||
switch (arg1) {
|
||||
case 1:
|
||||
r = D_8019BCCA;
|
||||
g = D_8019BBF2;
|
||||
b = D_8019BBD0;
|
||||
D_8019A710[0] = r * 5 >> 3;
|
||||
D_8019A711[0] = g << 3;
|
||||
D_8019A712[0] = (s32)(b * 255) >> 4;
|
||||
D_8019A710[4] = (s32)(r * 143) >> 4;
|
||||
D_8019A711[4] = g * 25 >> 1;
|
||||
D_8019A712[4] = (s32)(b * 255) >> 4;
|
||||
D_8019A710[8] = (s32)(r * 255) >> 4;
|
||||
D_8019A711[8] = (s32)(g * 255) >> 4;
|
||||
D_8019A712[8] = (s32)(b * 255) >> 4;
|
||||
D_8019A710[12] = (s32)(r * 143) >> 4;
|
||||
D_8019A711[12] = (s32)(g * 255) >> 4;
|
||||
D_8019A712[12] = (s32)(b * 45) >> 2;
|
||||
break;
|
||||
case 0:
|
||||
case 2:
|
||||
i = arg1 * 4;
|
||||
D_8019A710[i] = arg2[0x44] * D_8019BCCA >> 4;
|
||||
D_8019A711[i] = arg2[0x45] * D_8019BBF2 >> 4;
|
||||
D_8019A712[i] = arg2[0x46] * D_8019BBD0 >> 4;
|
||||
i = (arg1 + 1) * 4;
|
||||
D_8019A710[i] = arg2[0x47] * D_8019BCCA >> 4;
|
||||
D_8019A711[i] = arg2[0x48] * D_8019BBF2 >> 4;
|
||||
D_8019A712[i] = arg2[0x49] * D_8019BBD0 >> 4;
|
||||
break;
|
||||
case 3:
|
||||
case 4:
|
||||
arg2[0x20] = D_8019BCCA << 3;
|
||||
arg2[0x21] = D_8019BBF2 << 3;
|
||||
arg2[0x22] = D_8019BBD0 << 3;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129c40 (src/shared) */
|
||||
|
||||
|
||||
Reference in New Issue
Block a user