From 2edb23a535934f96cb41584e60e322ce9c5afff7 Mon Sep 17 00:00:00 2001 From: Drew T <50529377+Druthulu@users.noreply.github.com> Date: Tue, 11 Aug 2026 15:45:47 -0600 Subject: [PATCH] =?UTF-8?q?feat(phase-30=20S48-0b):=20jtbl=20family=20func?= =?UTF-8?q?=5F801299C8=20=E2=80=94=203/3=20blocked-binary=20siblings=20ban?= =?UTF-8?q?ked?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- config/overlays.mk | 6 +-- config/splat.ov_MAIN_012.yaml | 18 ++++---- config/splat.ov_SC02_037.yaml | 20 +++++---- config/splat.ov_SC03_107.yaml | 18 ++++---- src/ov_MAIN_012/ov_MAIN_012.c | 81 ++++++++++++++++++++++++++++++++++- src/ov_SC02_037/ov_SC02_037.c | 81 ++++++++++++++++++++++++++++++++++- src/ov_SC03_107/ov_SC03_107.c | 81 ++++++++++++++++++++++++++++++++++- 7 files changed, 274 insertions(+), 31 deletions(-) diff --git a/config/overlays.mk b/config/overlays.mk index 042e6adc9..b7e54bfcd 100644 --- a/config/overlays.mk +++ b/config/overlays.mk @@ -4915,7 +4915,7 @@ ov_MAIN_012_ELF := $(ov_MAIN_012_OUT).elf ov_MAIN_012_MAPFILE := $(ov_MAIN_012_OUT).map ov_MAIN_012_LD_SCRIPT := $(ov_MAIN_012_OUT).ld ov_MAIN_012_SPLAT_YAML := config/splat.ov_MAIN_012.yaml -ov_MAIN_012_JTBL_INTERLEAVE := --order tail.data.o,ov_MAIN_012_jr_80131340.o,tail2.data.o,ov_MAIN_012_jr_80135A4C.o,tail3.data.o,ov_MAIN_012_jr_80135EB0.o,tail4.data.o,ov_MAIN_012_jr_801380E0.o,tail5.data.o,ov_MAIN_012_jr_8013F350.o,tail6.data.o,ov_MAIN_012_jr_80159C84.o,tail7.data.o,ov_MAIN_012_jr_8015A3C8.o,tail8.data.o,ov_MAIN_012_jr_801789AC.o,tail9.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve +ov_MAIN_012_JTBL_INTERLEAVE := --order tail.data.o,ov_MAIN_012.o,tail2.data.o,ov_MAIN_012_jr_80131340.o,tail3.data.o,ov_MAIN_012_jr_80135A4C.o,tail4.data.o,ov_MAIN_012_jr_80135EB0.o,tail5.data.o,ov_MAIN_012_jr_801380E0.o,tail6.data.o,ov_MAIN_012_jr_8013F350.o,tail7.data.o,ov_MAIN_012_jr_80159C84.o,tail8.data.o,ov_MAIN_012_jr_8015A3C8.o,tail9.data.o,ov_MAIN_012_jr_801789AC.o,tail10.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve build/src/ov_MAIN_012/ov_MAIN_012_jr_8013F350.o: JTBL_PADS := 0,0,4,0,4,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20,+0x38,+0x58,+0x70,+0x90 build/src/ov_MAIN_012/ov_MAIN_012_jr_80159C84.o: JTBL_PADS := 0,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20 build/src/ov_MAIN_012/ov_MAIN_012_jr_801789AC.o: JTBL_PADS := 0,0,0,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x18,+0x190,+0x220 @@ -4940,7 +4940,7 @@ ov_SC02_037_ELF := $(ov_SC02_037_OUT).elf ov_SC02_037_MAPFILE := $(ov_SC02_037_OUT).map ov_SC02_037_LD_SCRIPT := $(ov_SC02_037_OUT).ld ov_SC02_037_SPLAT_YAML := config/splat.ov_SC02_037.yaml -ov_SC02_037_JTBL_INTERLEAVE := --order tail.data.o,ov_SC02_037_jr_80131340.o,tail2.data.o,ov_SC02_037_jr_80135A4C.o,tail3.data.o,ov_SC02_037_jr_80135EB0.o,tail4.data.o,ov_SC02_037_jr_801380E0.o,tail5.data.o,ov_SC02_037_jr_8013F350.o,tail6.data.o,ov_SC02_037_jr_80159C84.o,tail7.data.o,ov_SC02_037_jr_8015A3C8.o,tail8.data.o,ov_SC02_037_jr_801789AC.o,tail9.data.o,ov_SC02_037_jr_8017AE2C.o,tail10.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve +ov_SC02_037_JTBL_INTERLEAVE := --order tail.data.o,ov_SC02_037.o,tail2.data.o,ov_SC02_037_jr_80131340.o,tail3.data.o,ov_SC02_037_jr_80135A4C.o,tail4.data.o,ov_SC02_037_jr_80135EB0.o,tail5.data.o,ov_SC02_037_jr_801380E0.o,tail6.data.o,ov_SC02_037_jr_8013F350.o,tail7.data.o,ov_SC02_037_jr_80159C84.o,tail8.data.o,ov_SC02_037_jr_8015A3C8.o,tail9.data.o,ov_SC02_037_jr_801789AC.o,tail10.data.o,ov_SC02_037_jr_8017AE2C.o,tail11.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve build/src/ov_SC02_037/ov_SC02_037_jr_8013F350.o: JTBL_PADS := 0,0,4,0,4,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20,+0x38,+0x58,+0x70,+0x90 build/src/ov_SC02_037/ov_SC02_037_jr_80159C84.o: JTBL_PADS := 0,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20 build/src/ov_SC02_037/ov_SC02_037_jr_8015A3C8.o: JTBL_PADS := 0,4,4,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20,+0x40,+0x60 @@ -4966,7 +4966,7 @@ ov_SC03_107_ELF := $(ov_SC03_107_OUT).elf ov_SC03_107_MAPFILE := $(ov_SC03_107_OUT).map ov_SC03_107_LD_SCRIPT := $(ov_SC03_107_OUT).ld ov_SC03_107_SPLAT_YAML := config/splat.ov_SC03_107.yaml -ov_SC03_107_JTBL_INTERLEAVE := --order tail.data.o,ov_SC03_107_jr_80131340.o,tail2.data.o,ov_SC03_107_jr_80135A4C.o,tail3.data.o,ov_SC03_107_jr_80135EB0.o,tail4.data.o,ov_SC03_107_jr_801380E0.o,tail5.data.o,ov_SC03_107_jr_8013F350.o,tail6.data.o,ov_SC03_107_jr_80159C84.o,tail7.data.o,ov_SC03_107_jr_8015A3C8.o,tail8.data.o,ov_SC03_107_jr_801789AC.o,tail9.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve +ov_SC03_107_JTBL_INTERLEAVE := --order tail.data.o,ov_SC03_107.o,tail2.data.o,ov_SC03_107_jr_80131340.o,tail3.data.o,ov_SC03_107_jr_80135A4C.o,tail4.data.o,ov_SC03_107_jr_80135EB0.o,tail5.data.o,ov_SC03_107_jr_801380E0.o,tail6.data.o,ov_SC03_107_jr_8013F350.o,tail7.data.o,ov_SC03_107_jr_80159C84.o,tail8.data.o,ov_SC03_107_jr_8015A3C8.o,tail9.data.o,ov_SC03_107_jr_801789AC.o,tail10.data.o,trailing.o # Phase-26 §8 jtbl-rodata carve build/src/ov_SC03_107/ov_SC03_107_jr_8013F350.o: JTBL_PADS := 0,0,4,0,4,0 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20,+0x38,+0x58,+0x70,+0x90 build/src/ov_SC03_107/ov_SC03_107_jr_80159C84.o: JTBL_PADS := 0,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x20 build/src/ov_SC03_107/ov_SC03_107_jr_801789AC.o: JTBL_PADS := 0,0,0,4 # §8e pads (jtbl_carve.py) tables=+0x0,+0x18,+0x190,+0x220 diff --git a/config/splat.ov_MAIN_012.yaml b/config/splat.ov_MAIN_012.yaml index 27811a1e8..56abe9b5d 100644 --- a/config/splat.ov_MAIN_012.yaml +++ b/config/splat.ov_MAIN_012.yaml @@ -103,22 +103,24 @@ segments: - [0x32270, c, ov_MAIN_012_jr_8015A3C8] - [0x50854, c, ov_MAIN_012_jr_801789AC] - [0x561e0, data, tail] + - [0x5abd0, .rodata, ov_MAIN_012] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) + - [0x5abe4, data, tail2] - [0x5ad78, .rodata, ov_MAIN_012_jr_80131340] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x5ad8c, data, tail2] + - [0x5ad8c, data, tail3] - [0x5adbc, .rodata, ov_MAIN_012_jr_80135A4C] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x5add0, data, tail3] + - [0x5add0, data, tail4] - [0x5adec, .rodata, ov_MAIN_012_jr_80135EB0] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x5ae18, data, tail4] + - [0x5ae18, data, tail5] - [0x5ae24, .rodata, ov_MAIN_012_jr_801380E0] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x5ae88, data, tail5] + - [0x5ae88, data, tail6] - [0x5b494, .rodata, ov_MAIN_012_jr_8013F350] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x5b538, data, tail6] + - [0x5b538, data, tail7] - [0x5b730, .rodata, ov_MAIN_012_jr_80159C84] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x5b764, data, tail7] + - [0x5b764, data, tail8] - [0x5b768, .rodata, ov_MAIN_012_jr_8015A3C8] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x5b784, data, tail8] + - [0x5b784, data, tail9] - [0x5ba40, .rodata, ov_MAIN_012_jr_801789AC] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x5bcec, data, tail9] + - [0x5bcec, data, tail10] - [0x5DB24, bin, trailing] # final 3 bytes (EOF not 4-aligned; spimdisasm drops a partial word) - [0x5DB27] # EOF marker = the 0.4.dec byte length # @TRAILING@ (above) is replaced by tools/new_overlay.sh: for a non-4-aligned overlay it becomes diff --git a/config/splat.ov_SC02_037.yaml b/config/splat.ov_SC02_037.yaml index 1c5a76f2f..094d87be0 100644 --- a/config/splat.ov_SC02_037.yaml +++ b/config/splat.ov_SC02_037.yaml @@ -104,24 +104,26 @@ segments: - [0x50854, c, ov_SC02_037_jr_801789AC] - [0x52cd4, c, ov_SC02_037_jr_8017AE2C] - [0x5ba68, data, tail] + - [0x9f0b4, .rodata, ov_SC02_037] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) + - [0x9f0c8, data, tail2] - [0x9f25c, .rodata, ov_SC02_037_jr_80131340] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x9f270, data, tail2] + - [0x9f270, data, tail3] - [0x9f2a0, .rodata, ov_SC02_037_jr_80135A4C] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x9f2b4, data, tail3] + - [0x9f2b4, data, tail4] - [0x9f2d0, .rodata, ov_SC02_037_jr_80135EB0] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x9f2fc, data, tail4] + - [0x9f2fc, data, tail5] - [0x9f308, .rodata, ov_SC02_037_jr_801380E0] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x9f36c, data, tail5] + - [0x9f36c, data, tail6] - [0x9f978, .rodata, ov_SC02_037_jr_8013F350] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x9fa1c, data, tail6] + - [0x9fa1c, data, tail7] - [0x9fc14, .rodata, ov_SC02_037_jr_80159C84] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x9fc48, data, tail7] + - [0x9fc48, data, tail8] - [0x9fc4c, .rodata, ov_SC02_037_jr_8015A3C8] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x9fcc8, data, tail8] + - [0x9fcc8, data, tail9] - [0x9ff24, .rodata, ov_SC02_037_jr_801789AC] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x9ff3c, data, tail9] + - [0x9ff3c, data, tail10] - [0xa01d4, .rodata, ov_SC02_037_jr_8017AE2C] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0xa0238, data, tail10] + - [0xa0238, data, tail11] - [0xA198C, bin, trailing] # final 3 bytes (EOF not 4-aligned; spimdisasm drops a partial word) - [0xA198F] # EOF marker = the 0.4.dec byte length # @TRAILING@ (above) is replaced by tools/new_overlay.sh: for a non-4-aligned overlay it becomes diff --git a/config/splat.ov_SC03_107.yaml b/config/splat.ov_SC03_107.yaml index a741118c1..d3a3a3be8 100644 --- a/config/splat.ov_SC03_107.yaml +++ b/config/splat.ov_SC03_107.yaml @@ -103,22 +103,24 @@ segments: - [0x32270, c, ov_SC03_107_jr_8015A3C8] - [0x50854, c, ov_SC03_107_jr_801789AC] - [0x5a304, data, tail] + - [0x713e8, .rodata, ov_SC03_107] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) + - [0x713fc, data, tail2] - [0x71590, .rodata, ov_SC03_107_jr_80131340] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x715a4, data, tail2] + - [0x715a4, data, tail3] - [0x715d4, .rodata, ov_SC03_107_jr_80135A4C] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x715e8, data, tail3] + - [0x715e8, data, tail4] - [0x71604, .rodata, ov_SC03_107_jr_80135EB0] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x71630, data, tail4] + - [0x71630, data, tail5] - [0x7163c, .rodata, ov_SC03_107_jr_801380E0] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x716a0, data, tail5] + - [0x716a0, data, tail6] - [0x71cac, .rodata, ov_SC03_107_jr_8013F350] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x71d50, data, tail6] + - [0x71d50, data, tail7] - [0x71f48, .rodata, ov_SC03_107_jr_80159C84] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x71f7c, data, tail7] + - [0x71f7c, data, tail8] - [0x71f80, .rodata, ov_SC03_107_jr_8015A3C8] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x71f9c, data, tail8] + - [0x71f9c, data, tail9] - [0x72258, .rodata, ov_SC03_107_jr_801789AC] # Phase-26 §8 jtbl-rodata carve (jtbl_carve.py) - - [0x72504, data, tail9] + - [0x72504, data, tail10] - [0x73BE4, bin, trailing] # final 3 bytes (EOF not 4-aligned; spimdisasm drops a partial word) - [0x73BE7] # EOF marker = the 0.4.dec byte length # @TRAILING@ (above) is replaced by tools/new_overlay.sh: for a non-4-aligned overlay it becomes diff --git a/src/ov_MAIN_012/ov_MAIN_012.c b/src/ov_MAIN_012/ov_MAIN_012.c index 8809ecfde..67c0c5b55 100644 --- a/src/ov_MAIN_012/ov_MAIN_012.c +++ b/src/ov_MAIN_012/ov_MAIN_012.c @@ -640,7 +640,86 @@ void func_8012956C(void) { DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298f4 (src/shared) */ -INCLUDE_ASM("asm/ov_MAIN_012/nonmatchings/ov_MAIN_012", func_801299C8); + + + +// @class: schedule +// @stuck: none — MATCH (158 ins, match_one relocation-masked) +// +// Levers that landed it (2 iterations, 56 mismatched -> MATCH): +// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the +// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`) +// stashed in the jtbl branch delay slot and RE-extended per use in the case body. +// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801840B8/1/2[] (a 4-row x 3-comp +// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)` +// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case. +// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block, +// so `case 1:` must be written before `case 0/2:` and `case 3/4:`. +// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0], +// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS +// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays +// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE +// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting +// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_80185B10 load. +// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit +// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block). + +extern u8 D_80185C0A; +extern u8 D_80185B32; +extern u8 D_80185B10; +extern u8 D_801840B8[]; +extern u8 D_801840B9[]; +extern u8 D_801840BA[]; + +void func_801299C8(arg0, arg1, arg2) +s16 arg0; +s16 arg1; +u8 *arg2; +{ + u32 r; + u32 g; + u32 b; + s32 i; + + if (arg0 != 0) { + switch (arg1) { + case 1: + r = D_80185C0A; + g = D_80185B32; + b = D_80185B10; + D_801840B8[0] = r * 5 >> 3; + D_801840B9[0] = g << 3; + D_801840BA[0] = (s32)(b * 255) >> 4; + D_801840B8[4] = (s32)(r * 143) >> 4; + D_801840B9[4] = g * 25 >> 1; + D_801840BA[4] = (s32)(b * 255) >> 4; + D_801840B8[8] = (s32)(r * 255) >> 4; + D_801840B9[8] = (s32)(g * 255) >> 4; + D_801840BA[8] = (s32)(b * 255) >> 4; + D_801840B8[12] = (s32)(r * 143) >> 4; + D_801840B9[12] = (s32)(g * 255) >> 4; + D_801840BA[12] = (s32)(b * 45) >> 2; + break; + case 0: + case 2: + i = arg1 * 4; + D_801840B8[i] = arg2[0x44] * D_80185C0A >> 4; + D_801840B9[i] = arg2[0x45] * D_80185B32 >> 4; + D_801840BA[i] = arg2[0x46] * D_80185B10 >> 4; + i = (arg1 + 1) * 4; + D_801840B8[i] = arg2[0x47] * D_80185C0A >> 4; + D_801840B9[i] = arg2[0x48] * D_80185B32 >> 4; + D_801840BA[i] = arg2[0x49] * D_80185B10 >> 4; + break; + case 3: + case 4: + arg2[0x20] = D_80185C0A << 3; + arg2[0x21] = D_80185B32 << 3; + arg2[0x22] = D_80185B10 << 3; + break; + } + } +} DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129c40 (src/shared) */ diff --git a/src/ov_SC02_037/ov_SC02_037.c b/src/ov_SC02_037/ov_SC02_037.c index 898d7be91..0f214de2a 100644 --- a/src/ov_SC02_037/ov_SC02_037.c +++ b/src/ov_SC02_037/ov_SC02_037.c @@ -623,7 +623,86 @@ void func_8012956C(void) { DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298f4 (src/shared) */ -INCLUDE_ASM("asm/ov_SC02_037/nonmatchings/ov_SC02_037", func_801299C8); + + + +// @class: schedule +// @stuck: none — MATCH (158 ins, match_one relocation-masked) +// +// Levers that landed it (2 iterations, 56 mismatched -> MATCH): +// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the +// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`) +// stashed in the jtbl branch delay slot and RE-extended per use in the case body. +// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_801C83B0/1/2[] (a 4-row x 3-comp +// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)` +// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case. +// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block, +// so `case 1:` must be written before `case 0/2:` and `case 3/4:`. +// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0], +// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS +// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays +// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE +// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting +// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_801C9978 load. +// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit +// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block). + +extern u8 D_801C9A72; +extern u8 D_801C999A; +extern u8 D_801C9978; +extern u8 D_801C83B0[]; +extern u8 D_801C83B1[]; +extern u8 D_801C83B2[]; + +void func_801299C8(arg0, arg1, arg2) +s16 arg0; +s16 arg1; +u8 *arg2; +{ + u32 r; + u32 g; + u32 b; + s32 i; + + if (arg0 != 0) { + switch (arg1) { + case 1: + r = D_801C9A72; + g = D_801C999A; + b = D_801C9978; + D_801C83B0[0] = r * 5 >> 3; + D_801C83B1[0] = g << 3; + D_801C83B2[0] = (s32)(b * 255) >> 4; + D_801C83B0[4] = (s32)(r * 143) >> 4; + D_801C83B1[4] = g * 25 >> 1; + D_801C83B2[4] = (s32)(b * 255) >> 4; + D_801C83B0[8] = (s32)(r * 255) >> 4; + D_801C83B1[8] = (s32)(g * 255) >> 4; + D_801C83B2[8] = (s32)(b * 255) >> 4; + D_801C83B0[12] = (s32)(r * 143) >> 4; + D_801C83B1[12] = (s32)(g * 255) >> 4; + D_801C83B2[12] = (s32)(b * 45) >> 2; + break; + case 0: + case 2: + i = arg1 * 4; + D_801C83B0[i] = arg2[0x44] * D_801C9A72 >> 4; + D_801C83B1[i] = arg2[0x45] * D_801C999A >> 4; + D_801C83B2[i] = arg2[0x46] * D_801C9978 >> 4; + i = (arg1 + 1) * 4; + D_801C83B0[i] = arg2[0x47] * D_801C9A72 >> 4; + D_801C83B1[i] = arg2[0x48] * D_801C999A >> 4; + D_801C83B2[i] = arg2[0x49] * D_801C9978 >> 4; + break; + case 3: + case 4: + arg2[0x20] = D_801C9A72 << 3; + arg2[0x21] = D_801C999A << 3; + arg2[0x22] = D_801C9978 << 3; + break; + } + } +} DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129c40 (src/shared) */ diff --git a/src/ov_SC03_107/ov_SC03_107.c b/src/ov_SC03_107/ov_SC03_107.c index 5e581e61f..fee379310 100644 --- a/src/ov_SC03_107/ov_SC03_107.c +++ b/src/ov_SC03_107/ov_SC03_107.c @@ -640,7 +640,86 @@ void func_8012956C(void) { DEFINE_func_801298F4() /* dedup: shared engine-core @0x801298f4 (src/shared) */ -INCLUDE_ASM("asm/ov_SC03_107/nonmatchings/ov_SC03_107", func_801299C8); + + + +// @class: schedule +// @stuck: none — MATCH (158 ins, match_one relocation-masked) +// +// Levers that landed it (2 iterations, 56 mismatched -> MATCH): +// 1. §43 K&R s16-param definition: `void f(a0,a1,a2) s16 a0; s16 a1; u8 *a2;` reproduces the +// in-place `sll $a0,$a0,16` zero-test on the arg reg + the raw-$a1 copy (`addu $a3,$a1,$zero`) +// stashed in the jtbl branch delay slot and RE-extended per use in the case body. +// 2. §18 array-of-struct %lo-fold: three sibling extern arrays D_8019A710/1/2[] (a 4-row x 3-comp +// RGB gradient table, stride 4) give `lui $at,%hi(sym); addu $at,$at,idx4; sb $v0,%lo(sym)($at)` +// for the indexed case and plain `lui/sb %lo(sym+k)` for the constant-index case. +// 3. Switch CASE-ORDER = source order: the jump table dispatches case 1 to the FIRST emitted block, +// so `case 1:` must be written before `case 0/2:` and `case 3/4:`. +// 4. THE residual (56 -> 0): the case-1 body must be written ROW-MAJOR (BE0[0],BE1[0],BE2[0], +// BE0[4],BE1[4],BE2[4],...), i.e. the natural table fill. gcc-2.7.2's sched pass then REORDERS +// the stores itself (BE0,BE1,BE5,BE8,BE4,BEC,BE9,BED,BE2,BE6,BEA,BEE) because the three arrays +// are distinct declarations => provably non-aliasing. Writing the source in the target's STORE +// order is the trap: it pins the b*255 / r*143 CSEs at their late store sites instead of letting +// them hoist into $a0/$v1 at rows 0/1, and mis-schedules the D_8019BBD0 load. +// 5. Shift signedness: `u32` component locals give `srl` for r*5>>3 and g*25>>1; an explicit +// `(s32)(x * 255) >> 4` gives `sra` for the *255 / *143 / *45 terms (mixed within one block). + +extern u8 D_8019BCCA; +extern u8 D_8019BBF2; +extern u8 D_8019BBD0; +extern u8 D_8019A710[]; +extern u8 D_8019A711[]; +extern u8 D_8019A712[]; + +void func_801299C8(arg0, arg1, arg2) +s16 arg0; +s16 arg1; +u8 *arg2; +{ + u32 r; + u32 g; + u32 b; + s32 i; + + if (arg0 != 0) { + switch (arg1) { + case 1: + r = D_8019BCCA; + g = D_8019BBF2; + b = D_8019BBD0; + D_8019A710[0] = r * 5 >> 3; + D_8019A711[0] = g << 3; + D_8019A712[0] = (s32)(b * 255) >> 4; + D_8019A710[4] = (s32)(r * 143) >> 4; + D_8019A711[4] = g * 25 >> 1; + D_8019A712[4] = (s32)(b * 255) >> 4; + D_8019A710[8] = (s32)(r * 255) >> 4; + D_8019A711[8] = (s32)(g * 255) >> 4; + D_8019A712[8] = (s32)(b * 255) >> 4; + D_8019A710[12] = (s32)(r * 143) >> 4; + D_8019A711[12] = (s32)(g * 255) >> 4; + D_8019A712[12] = (s32)(b * 45) >> 2; + break; + case 0: + case 2: + i = arg1 * 4; + D_8019A710[i] = arg2[0x44] * D_8019BCCA >> 4; + D_8019A711[i] = arg2[0x45] * D_8019BBF2 >> 4; + D_8019A712[i] = arg2[0x46] * D_8019BBD0 >> 4; + i = (arg1 + 1) * 4; + D_8019A710[i] = arg2[0x47] * D_8019BCCA >> 4; + D_8019A711[i] = arg2[0x48] * D_8019BBF2 >> 4; + D_8019A712[i] = arg2[0x49] * D_8019BBD0 >> 4; + break; + case 3: + case 4: + arg2[0x20] = D_8019BCCA << 3; + arg2[0x21] = D_8019BBF2 << 3; + arg2[0x22] = D_8019BBD0 << 3; + break; + } + } +} DEFINE_func_80129C40() /* dedup: shared engine-core @0x80129c40 (src/shared) */