mirror of
https://github.com/Druthulu/BFM-decomp
synced 2026-10-01 23:52:03 -04:00
docs(phase-32): T4b (7) ledger — func_800CF3E8 banked (md_MAIN_003 100% C), backlog re-rendered, the Fable draft + report + verdict kept; queue updated
This commit is contained in:
@@ -0,0 +1,8 @@
|
||||
CC build/src/md_MAIN_003/md_MAIN_003.o
|
||||
src/md_MAIN_003/md_MAIN_003.c: In function `func_800CFC1C':
|
||||
src/md_MAIN_003/md_MAIN_003.c:582: warning: return makes integer from pointer without a cast
|
||||
src/md_MAIN_003/md_MAIN_003.c:666: warning: return makes integer from pointer without a cast
|
||||
LD build/md_MAIN_003/md_MAIN_003.elf
|
||||
OBJCOPY build/md_MAIN_003/md_MAIN_003
|
||||
[ OK ] build/md_MAIN_003/md_MAIN_003
|
||||
sha1 dd1b32ecf1103c6f7cf1943d25546a3046e17b14 == config/check.md_MAIN_003.sha (BYTE-IDENTICAL)
|
||||
@@ -0,0 +1,242 @@
|
||||
|
||||
typedef struct {
|
||||
u8 pad0, pad1, pad2, len;
|
||||
u32 tpage;
|
||||
u8 r0, g0, b0, code;
|
||||
s16 x0, y0;
|
||||
u8 u0, v0;
|
||||
u16 clut;
|
||||
s16 w, h;
|
||||
} Sprt24;
|
||||
|
||||
extern s32 D_800EC690;
|
||||
extern s32 D_800EC694;
|
||||
extern s32 D_800EC68C;
|
||||
extern s16 D_800EC678;
|
||||
extern u16 D_800B9A02;
|
||||
extern u8 D_800AA60C[];
|
||||
typedef struct { u16 v; } S16_800CF3E8;
|
||||
typedef struct { u32 addr : 24; u32 len : 8; } PTag_800CF3E8;
|
||||
#define setaddr_(p, a) (((PTag_800CF3E8 *)(p))->addr = (u32)(a))
|
||||
#define getaddr_(p) ((u32)(((PTag_800CF3E8 *)(p))->addr))
|
||||
#define ADDPRIMB(ot, p) setaddr_((p), getaddr_(ot)); setaddr_((ot), (p));
|
||||
|
||||
extern S16_800CF3E8 sD800B9A02 __asm__("D_800B9A02");
|
||||
#define OTIDX ((fr = sD800B9A02), fr.v)
|
||||
extern Sprt24 D_800EC7C8[];
|
||||
extern Sprt24 D_800EC7F8[];
|
||||
extern Sprt24 D_800EC828[];
|
||||
extern Sprt24 D_800EC858[];
|
||||
extern Sprt24 D_800EC6A8[];
|
||||
extern Sprt24 D_800EC6D8[];
|
||||
extern Sprt24 D_800EC708[];
|
||||
extern Sprt24 D_800EC738[];
|
||||
extern Sprt24 D_800EC768[];
|
||||
extern Sprt24 D_800EC798[];
|
||||
extern u16 D_800EC7DA[];
|
||||
extern u16 D_800EC80A[];
|
||||
extern u16 D_800EC83A[];
|
||||
extern u16 D_800EC86A[];
|
||||
|
||||
#define OTG (*(u32 *)(D_800AA60C + OTIDX * 0x4000))
|
||||
#define ADDPRIM(ot, p) \
|
||||
*(u32 *)(p) = (*(u32 *)(p) & 0xFF000000) | ((ot) & 0xFFFFFF); \
|
||||
(ot) = ((ot) & 0xFF000000) | ((u32)(p) & 0xFFFFFF);
|
||||
#define ADDPRIM2(ot, p, LO, HI) \
|
||||
*(u32 *)(p) = (*(u32 *)(p) & (HI)) | ((ot) & (LO)); \
|
||||
(ot) = ((ot) & (HI)) | ((u32)(p) & (LO));
|
||||
|
||||
void func_800CF3E8(void) {
|
||||
Sprt24 *pA, *pF, *pD, *p1, *p2, *p4;
|
||||
s32 c24;
|
||||
s32 w60;
|
||||
u32 tv0;
|
||||
u32 tag6;
|
||||
u32 tv1;
|
||||
register u32 tv2 __asm__("$15");
|
||||
u32 tv3;
|
||||
u32 tv4;
|
||||
u32 b1a;
|
||||
u32 b1c;
|
||||
register s32 c5 __asm__("$6");
|
||||
register s32 q5 __asm__("$13");
|
||||
register u32 mq2 __asm__("$8");
|
||||
register Sprt24 *pC __asm__("$3");
|
||||
register Sprt24 *p3 __asm__("$9");
|
||||
register Sprt24 *p5 __asm__("$4");
|
||||
register Sprt24 *p6 __asm__("$3");
|
||||
register u32 *ot __asm__("$5");
|
||||
s32 idx = D_800B9A02;
|
||||
S16_800CF3E8 fr;
|
||||
|
||||
if (D_800EC690 == 1) goto blk1;
|
||||
if (D_800EC690 < 2) goto final;
|
||||
if (D_800EC690 == 2) goto blk2;
|
||||
goto final;
|
||||
|
||||
blk1:
|
||||
if (D_800EC694 != 0) {
|
||||
b1a = 0xE100008A;
|
||||
mq2 = 0xFFFFFF;
|
||||
b1c = 0xE100008C;
|
||||
{ Sprt24 *b_ = D_800EC7C8; pA = &b_[idx]; }
|
||||
q5 = 5;
|
||||
pA->len = q5;
|
||||
((u32 *)&D_800EC7C8[idx])[1] = b1a;
|
||||
pA->code = 0x64;
|
||||
pA->r0 = pA->g0 = pA->b0 = D_800EC68C;
|
||||
pA->x0 = -0xB0;
|
||||
pA->y0 = 0x54;
|
||||
pA->u0 = 0;
|
||||
pA->v0 = 0;
|
||||
pA->clut = 0x7800;
|
||||
pA->w = 0x100;
|
||||
pA->h = 0x20;
|
||||
D_800EC7DA[idx * 12] = 0x7880;
|
||||
ADDPRIM2(OTG, pA, mq2, 0xFF000000);
|
||||
|
||||
__asm__("");
|
||||
{ Sprt24 *b_ = D_800EC7F8; pC = &b_[idx]; }
|
||||
pC->len = q5;
|
||||
((u32 *)&D_800EC7F8[idx])[1] = b1c;
|
||||
pC->code = 0x64;
|
||||
pC->r0 = pC->g0 = pC->b0 = D_800EC68C;
|
||||
pC->x0 = 0x50;
|
||||
w60 = 0x60;
|
||||
pC->y0 = 0x54;
|
||||
pC->u0 = 0;
|
||||
pC->v0 = 0;
|
||||
__asm__("");
|
||||
pC->clut = 0x7800;
|
||||
pC->w = w60;
|
||||
pC->h = 0x20;
|
||||
D_800EC80A[idx * 12] = 0x7880;
|
||||
ADDPRIM2(OTG, pC, mq2, 0xFF000000);
|
||||
|
||||
}
|
||||
goto final;
|
||||
blk2:
|
||||
{ Sprt24 *b_ = D_800EC828; pF = &b_[idx]; }
|
||||
pF->len = 5;
|
||||
((u32 *)&D_800EC828[idx])[1] = 0xE100008F;
|
||||
mq2 = 0xFFFFFF;
|
||||
pF->code = 0x64;
|
||||
pF->r0 = pF->g0 = pF->b0 = D_800EC68C;
|
||||
pF->x0 = -0x40;
|
||||
pF->y0 = 0x44;
|
||||
pF->u0 = 0;
|
||||
pF->v0 = 0;
|
||||
pF->clut = 0x7800;
|
||||
pF->w = 0x80;
|
||||
pF->h = 0x40;
|
||||
D_800EC83A[idx * 12] = 0x7A80;
|
||||
ADDPRIM2(OTG, pF, mq2, 0xFF000000);
|
||||
|
||||
{ Sprt24 *b_ = D_800EC858; pD = &b_[idx]; }
|
||||
pD->len = 5;
|
||||
((u32 *)&D_800EC858[idx])[1] = 0xE100008D;
|
||||
pD->code = 0x64;
|
||||
pD->r0 = pD->g0 = pD->b0 = D_800EC68C;
|
||||
pD->x0 = -0x70;
|
||||
pD->y0 = D_800EC678 * 32 + 0x44;
|
||||
pD->u0 = 0;
|
||||
pD->v0 = 0;
|
||||
pD->clut = 0x7800;
|
||||
pD->w = 0xE0;
|
||||
pD->h = 0x20;
|
||||
D_800EC86A[idx * 12] = 0x7800;
|
||||
ADDPRIM2(OTG, pD, mq2, 0xFF000000);
|
||||
|
||||
final:
|
||||
tv0 = 0xE100008A;
|
||||
tv1 = 0xE100008C;
|
||||
tv2 = 0xE100008E;
|
||||
tv3 = 0xE100009A;
|
||||
tv4 = 0xE100009C;
|
||||
{ Sprt24 *b_ = D_800EC6A8; p1 = &b_[idx]; }
|
||||
c5 = 5;
|
||||
p1->len = c5;
|
||||
((u32 *)&D_800EC6A8[idx])[1] = tv0;
|
||||
p1->code = 0x64;
|
||||
p1->r0 = p1->g0 = p1->b0 = D_800EC68C;
|
||||
p1->x0 = -0x140;
|
||||
p1->y0 = -0xDC;
|
||||
p1->u0 = 0;
|
||||
c24 = 0x24;
|
||||
p1->v0 = c24;
|
||||
p1->clut = 0x78C0;
|
||||
p1->w = 0x100;
|
||||
p1->h = 0xDC;
|
||||
{ Sprt24 *b_ = D_800EC6D8; p2 = &b_[idx]; }
|
||||
p2->len = c5;
|
||||
((u32 *)&D_800EC6D8[idx])[1] = tv1;
|
||||
p2->code = 0x64;
|
||||
p2->r0 = p2->g0 = p2->b0 = D_800EC68C;
|
||||
p2->x0 = -0x40;
|
||||
p2->y0 = -0xDC;
|
||||
p2->u0 = 0;
|
||||
p2->v0 = c24;
|
||||
p2->clut = 0x78C0;
|
||||
p2->w = 0x100;
|
||||
p2->h = 0xDC;
|
||||
{ Sprt24 *b_ = D_800EC708; p3 = &b_[idx]; }
|
||||
p3->len = c5;
|
||||
((u32 *)&D_800EC708[idx])[1] = tv2;
|
||||
p3->code = 0x64;
|
||||
p3->r0 = p3->g0 = p3->b0 = D_800EC68C;
|
||||
p3->x0 = 0xC0;
|
||||
p3->y0 = -0xDC;
|
||||
p3->u0 = 0;
|
||||
p3->v0 = c24;
|
||||
p3->clut = 0x78C0;
|
||||
p3->w = 0x80;
|
||||
p3->h = 0xDC;
|
||||
tv2 = 0xE100009E;
|
||||
{ Sprt24 *b_ = D_800EC738; p4 = &b_[idx]; }
|
||||
p4->len = c5;
|
||||
((u32 *)&D_800EC738[idx])[1] = tv3;
|
||||
p4->code = 0x64;
|
||||
p4->r0 = p4->g0 = p4->b0 = D_800EC68C;
|
||||
p4->x0 = -0x140;
|
||||
p4->y0 = 0;
|
||||
p4->u0 = 0;
|
||||
p4->v0 = 0;
|
||||
p4->clut = 0x78C0;
|
||||
p4->w = 0x100;
|
||||
p4->h = 0xDC;
|
||||
{ Sprt24 *b_ = D_800EC768; p5 = &b_[idx]; }
|
||||
p5->len = c5;
|
||||
((u32 *)&D_800EC768[idx])[1] = tv4;
|
||||
p5->code = 0x64;
|
||||
p5->r0 = p5->g0 = p5->b0 = D_800EC68C;
|
||||
p5->x0 = -0x40;
|
||||
p5->y0 = 0;
|
||||
p5->u0 = 0;
|
||||
p5->v0 = 0;
|
||||
__asm__("");
|
||||
{ Sprt24 *b_ = D_800EC798; p6 = &b_[idx]; }
|
||||
p5->clut = 0x78C0;
|
||||
p5->w = 0x100;
|
||||
p5->h = 0xDC;
|
||||
p6->len = c5;
|
||||
((u32 *)&D_800EC798[idx])[1] = tv2;
|
||||
tag6 = *(u32 *)p6;
|
||||
p6->code = 0x64;
|
||||
p6->r0 = p6->g0 = p6->b0 = D_800EC68C;
|
||||
p6->x0 = 0xC0;
|
||||
p6->y0 = 0;
|
||||
p6->u0 = 0;
|
||||
p6->v0 = 0;
|
||||
p6->clut = 0x78C0;
|
||||
p6->w = 0x80;
|
||||
p6->h = 0xDC;
|
||||
ot = (u32 *)(D_800AA60C + idx * 0x4000);
|
||||
tag6 &= 0xFF000000;
|
||||
*(u32 *)p6 = tag6 | (*ot & 0xFFFFFF);
|
||||
*ot = (*ot & 0xFF000000) | ((u32)p6 & 0xFFFFFF);
|
||||
ADDPRIMB(ot, p5)
|
||||
ADDPRIMB(ot, p4)
|
||||
ADDPRIMB(ot, p3)
|
||||
ADDPRIMB(ot, p2)
|
||||
ADDPRIMB(ot, p1)
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
# md_MAIN_003:func_800CF3E8 (469 ins) — Fable T5x: **MATCH** (leaf `match_one` 469/469, real-TU `rtu_match` MATCH)
|
||||
|
||||
Draft: `.run/P32/t5x/fable/func_800CF3E8.c` (= work `v_X5.c`). Alternate byte-identical drafts: `v_C1.c` (adds the S83 pointer
|
||||
launder), `v_X4.c` (tag read spelled at the chain). Baseline reproduced first: prior best 27 @ 469; S83 launder variant 79 @ 470.
|
||||
Relocation audit: all 20 data symbols present with the target's exact counts (`relocs_mine.txt` vs `relocs_target.txt` in the work dir).
|
||||
|
||||
## The mechanism, as READ from the dumps (all in `.run/P32/t5x/work/func_800CF3E8/rtl_*/t.i.{sched,sched2,lreg}`)
|
||||
Both prior reports located the residual in the right window (the p6 tag load at 380 vs 399/384) but attributed the placement to
|
||||
cse/sched1 alone. The dumps show THREE passes each contributing one defect, and the row closes only when all three are fixed:
|
||||
|
||||
1. **sched1: the tag load was BIRTHING-BOOSTED.** In the launder variant's `-dS` the load (insn 1182) is `7f000001` in the
|
||||
ready list at T-73 and is picked the instant its consumer is placed — `adjust_priority` (sched.c:2507) → `birthing_insn_p`
|
||||
(:2469, `reg_n_sets == 1`, dest live). A boosted load sinks to just before its use; NO source spelling moves it while the boost
|
||||
is alive (this is why S83's 14-position `tag6` birth sweep and the 32-position hoist sweep were inert).
|
||||
**Lever: a second LIVE set of the loaded pseudo — `tag6 = *(u32*)p6; … tag6 &= 0xFF000000;`** (compound assignment reuses the
|
||||
variable's pseudo as the `and`'s dest: `(set (reg 81) (and (reg 81) (reg mhi)))`). reg_n_sets = 2 → priority 6 in `-dS`
|
||||
(`v_A` trace: "1121 (6)"), and sched1's output becomes `sb len, [tpage addr], sw tpage, lw tag, lui m24, sb code, lbu, ori m24,
|
||||
x0 y0 u0 v0 clut w h, colour×3, sll, la, addu ot, lui mhi, lw *ot, and tag, and otv, or, sw` — the tag load is now ahead of
|
||||
`sb code`/`lbu` in LUID order. (Same gate as §501-C/§501-D; the compound form keeps the variable single-death/local, unlike the
|
||||
`__asm__ volatile("" : "=r"(x))` dial, which measured 39 here — it makes the tag a 2-death global allocno.)
|
||||
|
||||
2. **sched2 decides the FINAL slot, not sched1.** Post-reload both the load `(mem:SI (reg 3))` and the field stores
|
||||
`(mem/s (plus (reg 3) N))` share the hard base, `memrefs_conflict_p` disambiguates, and the unit-hazard rule "a load is blocked
|
||||
for 1 cycle right after a store" (`-dR`: `blocking insn 1182 for 1 cycles` at every store pick) walks the load upward past every
|
||||
consecutive store; it stops where it loses a LUID tie to `lbu`(1130)/`sb code`(1125) — which is exactly what (1) fixes.
|
||||
Two more sched2 facts, both byte-verified:
|
||||
* the S83 +1 `nop` was the ot-USE phantom `__asm__("" :: "r"(ot))` (insn 1179): in sched2 it ties `and tag` at priority 7,
|
||||
wins on LUID (sched1 parked it at the block end), and is picked between `lw *ot` and `and otv` — a zero-byte insn filling the
|
||||
load-delay slot in the MODEL only, so gas emits a real nop. **Remove it** (re-adding it: 70 @ 470, ablation X2).
|
||||
* `lui m24` lands at 378 (between `sb $a2,3($v1)` and `sw tpage`) purely from its sched2 anti-dependence on `$a2` (c5's last
|
||||
use) + the ALU-vs-memory hazard order — which requires m24 to BE in `$a2`, i.e. the allocation below. The fence
|
||||
`__asm__("")` after the tpage store must go (everything after a traditional asm depends on it, so `lui m24` could never reach
|
||||
378; re-adding it: 8 @ 469, ablation X3). The fence BEFORE p6's birth stays (removing it: 461 @ 471, ablation X6).
|
||||
|
||||
3. **local-alloc: m24 must out-rank mhi.** `qty_compare` (local-alloc.c:1579) = `floor_log2(n_refs)·n_refs·size/(death−birth)`
|
||||
with birth/death from the POST-sched1 positions and `n_refs` from flow (flow.c:2067/2315/2501/2711 — written BEFORE combine;
|
||||
combine.c:56 documents that it does not adjust them). With the hand-written `(x & 0xFF000000) | (y & 0xFFFFFF)` macro both
|
||||
masks have 13 refs; mhi's boosted `lui` sinks to the chain (born at 344) while m24's is born at 327, so mhi's range is 16
|
||||
shorter → mhi is allocated first and takes the lowest free register `$a2` (`v_A` lreg: `Register 452 in 6`, m24 `455 in 10`,
|
||||
tag `81 in 8`) — the target needs m24→`$a2`, mhi→`$t0`, tag→`$t2`.
|
||||
**Lever: spell the OT link the way libgpu does — the `P_TAG` bitfield `setaddr(p, getaddr(ot)); setaddr(ot, p)`.**
|
||||
`store_fixed_bit_field` (expmed.c) re-masks the already-masked value with `0xFFFFFF` (`must_and`), an `and` that cse cannot
|
||||
fold (both operands are registers) and combine removes later — but flow has already counted it: m24 gets 3 refs per prim
|
||||
instead of 2 (18–19 total, floor_log2 = 4) and its priority roughly doubles → allocated before mhi → `$a2`; mhi then takes
|
||||
`$t0`; the tag (range [380,401], overlapping 0xDC/mhi in `$t0`) takes `$t2`; every block-head constant keeps its register.
|
||||
Measured: bitfield on prims 5..1 (p6 kept manual): 74 → **2**; the same draft with the manual macro on all six: 74 (B5).
|
||||
§364's -O2 warning (a `/s` tag store lets cse drop the OT re-read) does not bite here because BOTH sides of the link are
|
||||
`P_TAG` accesses, so the `*ot` re-read is `/s` too and is invalidated correctly — the 469-instruction count is preserved.
|
||||
|
||||
4. The last 2 rows (p5 `x0`/`y0` order, 362/363) were the baseline's deliberate `y0; x0` source swap for the old basin; natural
|
||||
order (`x0; y0`) closes it: 2 → **MATCH**. (S83 measured the swap at 31 — in the old basin, where it was paid for elsewhere.)
|
||||
|
||||
## Ablations on the MATCH (leaf match_one; every one re-measured, not inferred)
|
||||
| change | result | meaning |
|
||||
|---|---|---|
|
||||
| X1: `(tag6 & 0xFF000000)` instead of `tag6 &= …` | 89 @ 469 | the second SET (boost kill) is load-bearing |
|
||||
| X2: re-add `__asm__("" :: "r"(ot))` | 70 @ 470 | the phantom steals the load-delay slot in sched2 |
|
||||
| X3: re-add `__asm__("")` after the tpage store | 8 @ 469 | blocks `lui m24` from reaching 378 |
|
||||
| X4: tag read spelled at the chain (after `ot = …`) | MATCH | source position of the read is NOT load-bearing once un-boosted |
|
||||
| X5: drop the S83 pointer launder | MATCH (delivered) | with the read spelled before the field stores, the pseudo-base deps pin it anyway |
|
||||
| X6: drop the fence before p6's birth | 461 @ 471 | §194-A fence still load-bearing |
|
||||
| B5: manual mask macro on all six prims | 74 @ 469 | the bitfield's redundant `and` (m24 refs) is load-bearing |
|
||||
| V1/V2: `__asm__ volatile("" : "=r"(tag6))` instead of `&=` | 39 @ 469 | 2-death → global allocno; different basin |
|
||||
|
||||
## For the cookbook (proposed §500-H amendment / new §)
|
||||
* "A load through a pinned base is scheduled late" has THREE owners: sched1's birthing boost (kill with a second live set of the
|
||||
loaded variable — compound assignment, zero bytes), sched2's load-after-store hazard walk (the final slot; LUID ties with
|
||||
neighbouring memory ops decide where it stops), and local-alloc's `qty_compare` (register roles follow the ranges the new
|
||||
order creates). Read `-dS` for `7f000001` on the load, `-dR` for `blocking insn N`, `-dl` for `Register N in R`.
|
||||
* A zero-byte `__asm__("" :: "r"(x))` USE is not free in the scheduler: it occupies a cycle and can fill a load-delay slot in
|
||||
the model only → a real nop from gas. Check the `-dR` ready list around every `lw`/`and` pair before keeping one.
|
||||
* `reg_n_refs` is flow's count and survives combine: any RTL redundancy that only combine removes (a bitfield's `must_and`
|
||||
re-mask, expmed.c `store_fixed_bit_field`) still weights local-alloc's priority. The libgpu `P_TAG` bitfield `addPrim` is
|
||||
therefore not just an operand-order lever (§364) but a REGISTER-PRIORITY lever at -O2 for the 0xFFFFFF mask.
|
||||
@@ -1,8 +1,7 @@
|
||||
# T4b Fable resume queue (Drew 2026-09-05: "resume agents, but not all at once, just 3 at a time"). Delete a line when resumed.
|
||||
# fn agentId closeness
|
||||
func_80185810 a547e70e9a7735c8a 35
|
||||
func_8017DC80 a7c2c1e4865e5590d 46
|
||||
func_800CF408 aac6ce9fdfdcc093c 49
|
||||
func_800CF6D0 a7f7e477e9878acdd 137
|
||||
func_80011380 a60ffea4022b891fc 6 (-O0, §474 PROVED — last)
|
||||
# RUNNING: func_800CD92C a5587b9d5c7006214 · func_80039308 a55fbfb4fb896bd71 · func_800CF3E8 ad27049c8fdf1e816 (DONE: func_800391D4 func_80039DEC func_800CD674 func_8017DF28 func_80020DA4 func_801834A4 MATCH; func_80032A74 NEAR 1)
|
||||
# RUNNING: func_800CD92C a5587b9d5c7006214 · func_80039308 a55fbfb4fb896bd71 · func_80185810 a547e70e9a7735c8a (DONE MATCH: 391D4 39DEC CD674 DF28 20DA4 1834A4 CF3E8; NEAR: 32A74 1)
|
||||
|
||||
@@ -5,3 +5,4 @@
|
||||
{"fn": "func_80020DA4", "binary": "main", "arm": "fable", "status": "MATCH", "closeness": 0, "compiles": true, "draft_path": ".run/P32/t5x/fable/func_80020DA4.c", "note": "unpin e0 ($6 pin made $a2 regs_ever_live -> bad_spill_regs -> forbidden at reload's retry_global_alloc, reload1.c:486/3497/3651-3660/709; the LO-parked mult results are retried first-fit) + two zero-byte launders steering local-alloc qty_compare: e1 launder before `dst[6] = -e1` (qty 6666->8750, allocated before e0, holds $v1) and p1 launder between its andi and sll (undoes the global tie). Ladder 2->37->15->6->MATCH. Coordinator rtu MATCH 100/100; gate_main BANKED 143dbb89", "session": "491895ad"}
|
||||
{"fn": "func_801834A4", "binary": "ov_SC03_105", "arm": "fable", "status": "MATCH", "closeness": 0, "compiles": true, "draft_path": ".run/P32/t5x/fable/func_801834A4.c", "note": "CSE-quantity split, not loop.c: expand_divmod (expmed.c:3034-3057) emits const/mult/sra/subu adjacently from ONE op0 so loop.c treats mult and sra alike (force_movables loop.c:1226 doubles K's savings -> hoists all, or a hard-reg dividend kills all). cse.c canon_reg never rewrites a hard reg but hashes by reg_qty: `sign = half >> 31` at the loop TOP (movable, maybe_never==0) + `hh = half` (register $2) + `pos[0] -= hh / 3` makes the mult read $2 (not movable; K unlinked, inline) while the division's sra is CSE'd into sign (hoisted); combine folds the copy into the mult. Traps: a callee-saved pin enters regs_ever_live before combine -> global pass 0 gives it to j (7); an asm feed for the unread sign store adds +2 weighted refs and swaps half/sign in allocno_compare (4) -> a dead `sign = 0` initializer instead (flow deletes it). Coordinator rtu MATCH 106/106; bank.sh byte-identical", "session": "491895ad"}
|
||||
{"fn": "func_80032A74", "binary": "main", "arm": "fable", "status": "NEAR", "closeness": 1, "compiles": true, "draft_path": ".run/P32/t5x/fable/func_80032A74.c", "note": "rtu DIFF 1 in src/800_b_2.c (idx 244 lh vs lhu; frame 0x78 exact); the lhu respelling is code-exact 422/422 with vars=56. Residual mechanism READ + byte-reproduced: the never-referenced 8-byte slot at 0x48 is a GHOST pseudo — a stranded middle temp of a 3-insn combine whose refs-zeroing is skipped when newi2pat!=0 (combine.c:2306-2313); zero occurrences -> regclass 'ST_REGS or none' -> reload1.c:658 alter_reg(i,-1) 8-byte slot in regno order = right after the three param slots (ghost1 reproducer vars=8, no code). The caller-save-area hypothesis (S83 HYPOTHESIS.md) is REFUTED: order_regs_for_reload (reload1.c:3606-3700) picks zero-use regs first so a pseudo homed in $t0 makes $t1 the spill reg (+30 rows); an area needs caller_save_needed (global.c:1085-1091) and a pseudo kept in a call-used reg with save/restore at every live call (caller-save.c:264, 349-470). The only ghost species from a memory value is the SIGN_EXTEND narrow-load split (combine.c:1887-1930) whose signature IS lh; a jump-target second promotion is folded by cse follow-jumps (n0a/b/c: vars 56, +86 rows); a fall-through one needs a register sign_extend MIPS lacks; the generic two-SETs split (combine.c:1963-2020) has no candidate (cse pre-folds constant offsets). Inert: expA/expB, n16 (2 ghosts), n0a/n0b/n0c. 402k tokens, 23 min", "session": "491895ad"}
|
||||
{"fn": "func_800CF3E8", "binary": "md_MAIN_003", "arm": "fable", "status": "MATCH", "closeness": 0, "compiles": true, "draft_path": ".run/P32/t5x/fable/func_800CF3E8.c", "note": "27 -> MATCH 469/469; three passes each owned one defect: (1) sched1 birthing boost on the tag load (sched.c:2507/2469) killed with a SECOND LIVE SET of the loaded variable `tag6 = *(u32*)p6; ... tag6 &= 0xFF000000;` (reg_n_sets=2, zero bytes); (2) the S83 +1 nop was the phantom `__asm__(\"\" :: \"r\"(ot))` USE filling the load-delay slot in sched2's model only — removed, plus the fence after the tpage store removed so `lui m24` reaches slot 378 via its $a2 anti-dep; (3) local-alloc qty_compare had mhi out-ranking m24 for $a2 — spelling prims 5..1's OT link as libgpu's P_TAG bitfield adds store_fixed_bit_field's redundant 0xFFFFFF re-mask (folded by combine, counted by flow: m24 13 -> 18 refs) so m24 wins $a2, mhi -> $t0, tag -> $t2; last 2 rows = p5 x0/y0 in natural source order. Ablations: non-compound tag 89; phantom re-added 70@470; fence re-added 8; manual masks on all six prims 74. Coordinator rtu MATCH 469/469; bank.sh byte-identical", "session": "491895ad"}
|
||||
|
||||
+6
-7
@@ -2,7 +2,7 @@
|
||||
|
||||
> Generated by `tools/backlog.py render` from `.run/backlog.jsonl`. These are functions the Phase-21 automation got **close** on but did NOT byte-match. The whole-binary byte-gate is the sole arbiter (G3/P9): **byte-matches bank and are NOT listed here** — only genuine near-misses/blockers are. Ranked by hand-session priority: **reach** (×N propagation leverage) → **closeness** (match_one mismatch count, lower = closer) → **size**. Each row's `best_draft` is the closest C the machine reached — resume from there.
|
||||
|
||||
**Open near-misses:** 10 · by status {'near': 9, 'failed': 1} · by class {'WALL-CANDIDATE': 1, 'WALL-PROVED': 1, 'SCHED': 4, 'REGALLOC': 1, 'ALIAS': 1, 'FRAME': 1, None: 1}
|
||||
**Open near-misses:** 9 · by status {'near': 8, 'failed': 1} · by class {'WALL-CANDIDATE': 1, 'WALL-PROVED': 1, 'SCHED': 4, 'REGALLOC': 1, 'FRAME': 1, None: 1}
|
||||
|
||||
| # | addr | reach | class | nins | status | closeness | where it stuck | best draft |
|
||||
|--:|------|------:|-------|-----:|--------|----------:|----------------|------------|
|
||||
@@ -10,9 +10,8 @@
|
||||
| 2 | func_80011380 | None | WALL-PROVED | 192 | near | 6 | §474 PROVED C-level floor (boot -O0): fold-const.c:882 split_tree merges MULT(MULT(i,2),2); the two escapes each cost one instruction (stupid.c:497 adjacency / expand_decl use-brackets); §388 -O0 colouring oracle. Pinned S79 #8; re-probed S83 in the real TU: DIFF 6 (unchanged) | `.run/m3/opus/func_80011380.c` |
|
||||
| 3 | func_800CD92C | None | SCHED | 247 | near | 15 | map §S7 prologue WEAVE: the {sw,lui,ori} groups for 0xE100008D/8F land after the 9-insn li block instead of before — the §17 pins reproduce the ALLOCATION but the hoist happens in sched2. Same SPRT family as func_800CD674 (§364 mirror levers applied) | `.run/P32/t3/opus/func_800CD92C.c` |
|
||||
| 4 | func_80039308 | None | REGALLOC | 518 | near | 17 | sched2 + cross-block regalloc: preheader 49/50 swap, un-spellable addu $a2,$a0,$zero (every p=r form cse-propagated), a temp on $t0 vs $s7, and 11 insns of one alias fact (2nd D_80073140[j] load cannot schedule above the D_800C7D20 store from C; /s unlock costs the address allocation, net 20-24). 34->17 via s16 b4 widening copy + dead-local identity sweep (.run/P32/t3/restored/sweep_func_80039308.py) + $2 pin. permuter_ils --klass REGALLOC 2x150s: no gain | `.run/P32/t3/opus/func_80039308.c` |
|
||||
| 5 | func_800CF3E8 | None | ALIAS | 469 | near | 27 | ONE cause: the pinned-base alias basin (§500-D1) in the p5/p6 tail; blocks 1-2 byte-exact (idx 0-361). 54->27 via blk2 constant birth order + birthing-boost local w60 + §194-A fence relocation. Inert: all 9 pins load-bearing (+5..+1409), asm position x7, h6 hoist 32x2, tag reshape, P_TAG ADDPRIM, array p6 stores, ~92k annealed variants. Untested: an unpinned alias of p6 for the tag load alone (ONE Opus second look allowed) | `.run/P32/t3/opus/func_800CF3E8.c` |
|
||||
| 6 | func_80185810 | None | SCHED | 489 | near | 35 | [permuter] 4 emission windows (see report .run/P32/t3/reports/func_80185810__opus__*.md); exact length, rtu-clean | `.run/P32/t3/opus/func_80185810.c` |
|
||||
| 7 | func_8017DC80 | None | FRAME | 346 | near | 46 | the historic -33 LENGTH wall CLOSED (GTE macros must be REAL macros — the TU house block; the splat Handwritten tag is wrong): 346/346, exact 0x70 frame + 9 callee-saved. Residual: reload-slot frame + the la $a0 slot; cse1 unifies OT index and n<4 across func_80010A08(8) (§500-D2 zero-byte asm retire) | `.run/P32/t3/opus/func_8017DC80.c` |
|
||||
| 8 | func_800CF408 | None | SCHED | 178 | near | 49 | [permuter] 3 hunks: two prologue sched2 slots, an mlo/mhi allocno tie, a 3-insn block-2 head hoist. Two LENGTH-bearing pins found (tp $17 shared by 0xE1000087/97 = the 6th callee-saved; ob $10 fixes the $t1/$t2/$t3 rotation, 56->49). §351 family (func_8001212C -O0 / func_8017DD04 -O2 exemplars) | `.run/P32/t3/opus/func_800CF408.c` |
|
||||
| 9 | func_800CF6D0 | None | SCHED | 249 | near | 137 | sched1 rank_for_schedule last-insn-CLASS tie (every store priority 2, equal refs; QImode stores grouped, loads floated, HImode after — 5 of 6 blocks) + $t1<->$t3 local-alloc swap of the two masks. 249/249 exact length only with tpage-before-len field order (19 swept). Inert at 137: pins on tpage constants/masks, asm re-ties, volatile/memory fences, /s-denial on any store subset, *0x4000 vs <<14, p++ vs p+0x18, / swap. decomp-permuter 122 was semantically wrong (R63) | `.run/P32/t3/opus/func_800CF6D0.c` |
|
||||
| 10 | func_80062144 | None | | None | failed | | won't compile standalone (loose-typing / missing decl) | |
|
||||
| 5 | func_80185810 | None | SCHED | 489 | near | 35 | [permuter] 4 emission windows (see report .run/P32/t3/reports/func_80185810__opus__*.md); exact length, rtu-clean | `.run/P32/t3/opus/func_80185810.c` |
|
||||
| 6 | func_8017DC80 | None | FRAME | 346 | near | 46 | the historic -33 LENGTH wall CLOSED (GTE macros must be REAL macros — the TU house block; the splat Handwritten tag is wrong): 346/346, exact 0x70 frame + 9 callee-saved. Residual: reload-slot frame + the la $a0 slot; cse1 unifies OT index and n<4 across func_80010A08(8) (§500-D2 zero-byte asm retire) | `.run/P32/t3/opus/func_8017DC80.c` |
|
||||
| 7 | func_800CF408 | None | SCHED | 178 | near | 49 | [permuter] 3 hunks: two prologue sched2 slots, an mlo/mhi allocno tie, a 3-insn block-2 head hoist. Two LENGTH-bearing pins found (tp $17 shared by 0xE1000087/97 = the 6th callee-saved; ob $10 fixes the $t1/$t2/$t3 rotation, 56->49). §351 family (func_8001212C -O0 / func_8017DD04 -O2 exemplars) | `.run/P32/t3/opus/func_800CF408.c` |
|
||||
| 8 | func_800CF6D0 | None | SCHED | 249 | near | 137 | sched1 rank_for_schedule last-insn-CLASS tie (every store priority 2, equal refs; QImode stores grouped, loads floated, HImode after — 5 of 6 blocks) + $t1<->$t3 local-alloc swap of the two masks. 249/249 exact length only with tpage-before-len field order (19 swept). Inert at 137: pins on tpage constants/masks, asm re-ties, volatile/memory fences, /s-denial on any store subset, *0x4000 vs <<14, p++ vs p+0x18, / swap. decomp-permuter 122 was semantically wrong (R63) | `.run/P32/t3/opus/func_800CF6D0.c` |
|
||||
| 9 | func_80062144 | None | | None | failed | | won't compile standalone (loose-typing / missing decl) | |
|
||||
|
||||
Reference in New Issue
Block a user