diff --git a/.run/P32/t3s3/bank_md_MAIN_003_184302.log b/.run/P32/t3s3/bank_md_MAIN_003_184302.log new file mode 100644 index 0000000000..1a1811b104 --- /dev/null +++ b/.run/P32/t3s3/bank_md_MAIN_003_184302.log @@ -0,0 +1,8 @@ + CC build/src/md_MAIN_003/md_MAIN_003.o +src/md_MAIN_003/md_MAIN_003.c: In function `func_800CFC1C': +src/md_MAIN_003/md_MAIN_003.c:582: warning: return makes integer from pointer without a cast +src/md_MAIN_003/md_MAIN_003.c:666: warning: return makes integer from pointer without a cast + LD build/md_MAIN_003/md_MAIN_003.elf + OBJCOPY build/md_MAIN_003/md_MAIN_003 +[ OK ] build/md_MAIN_003/md_MAIN_003 + sha1 dd1b32ecf1103c6f7cf1943d25546a3046e17b14 == config/check.md_MAIN_003.sha (BYTE-IDENTICAL) diff --git a/.run/P32/t5x/fable/func_800CF3E8.c b/.run/P32/t5x/fable/func_800CF3E8.c new file mode 100644 index 0000000000..048193b6fc --- /dev/null +++ b/.run/P32/t5x/fable/func_800CF3E8.c @@ -0,0 +1,242 @@ + +typedef struct { + u8 pad0, pad1, pad2, len; + u32 tpage; + u8 r0, g0, b0, code; + s16 x0, y0; + u8 u0, v0; + u16 clut; + s16 w, h; +} Sprt24; + +extern s32 D_800EC690; +extern s32 D_800EC694; +extern s32 D_800EC68C; +extern s16 D_800EC678; +extern u16 D_800B9A02; +extern u8 D_800AA60C[]; +typedef struct { u16 v; } S16_800CF3E8; +typedef struct { u32 addr : 24; u32 len : 8; } PTag_800CF3E8; +#define setaddr_(p, a) (((PTag_800CF3E8 *)(p))->addr = (u32)(a)) +#define getaddr_(p) ((u32)(((PTag_800CF3E8 *)(p))->addr)) +#define ADDPRIMB(ot, p) setaddr_((p), getaddr_(ot)); setaddr_((ot), (p)); + +extern S16_800CF3E8 sD800B9A02 __asm__("D_800B9A02"); +#define OTIDX ((fr = sD800B9A02), fr.v) +extern Sprt24 D_800EC7C8[]; +extern Sprt24 D_800EC7F8[]; +extern Sprt24 D_800EC828[]; +extern Sprt24 D_800EC858[]; +extern Sprt24 D_800EC6A8[]; +extern Sprt24 D_800EC6D8[]; +extern Sprt24 D_800EC708[]; +extern Sprt24 D_800EC738[]; +extern Sprt24 D_800EC768[]; +extern Sprt24 D_800EC798[]; +extern u16 D_800EC7DA[]; +extern u16 D_800EC80A[]; +extern u16 D_800EC83A[]; +extern u16 D_800EC86A[]; + +#define OTG (*(u32 *)(D_800AA60C + OTIDX * 0x4000)) +#define ADDPRIM(ot, p) \ + *(u32 *)(p) = (*(u32 *)(p) & 0xFF000000) | ((ot) & 0xFFFFFF); \ + (ot) = ((ot) & 0xFF000000) | ((u32)(p) & 0xFFFFFF); +#define ADDPRIM2(ot, p, LO, HI) \ + *(u32 *)(p) = (*(u32 *)(p) & (HI)) | ((ot) & (LO)); \ + (ot) = ((ot) & (HI)) | ((u32)(p) & (LO)); + +void func_800CF3E8(void) { + Sprt24 *pA, *pF, *pD, *p1, *p2, *p4; + s32 c24; + s32 w60; + u32 tv0; + u32 tag6; + u32 tv1; + register u32 tv2 __asm__("$15"); + u32 tv3; + u32 tv4; + u32 b1a; + u32 b1c; + register s32 c5 __asm__("$6"); + register s32 q5 __asm__("$13"); + register u32 mq2 __asm__("$8"); + register Sprt24 *pC __asm__("$3"); + register Sprt24 *p3 __asm__("$9"); + register Sprt24 *p5 __asm__("$4"); + register Sprt24 *p6 __asm__("$3"); + register u32 *ot __asm__("$5"); + s32 idx = D_800B9A02; + S16_800CF3E8 fr; + + if (D_800EC690 == 1) goto blk1; + if (D_800EC690 < 2) goto final; + if (D_800EC690 == 2) goto blk2; + goto final; + +blk1: + if (D_800EC694 != 0) { + b1a = 0xE100008A; + mq2 = 0xFFFFFF; + b1c = 0xE100008C; + { Sprt24 *b_ = D_800EC7C8; pA = &b_[idx]; } + q5 = 5; + pA->len = q5; + ((u32 *)&D_800EC7C8[idx])[1] = b1a; + pA->code = 0x64; + pA->r0 = pA->g0 = pA->b0 = D_800EC68C; + pA->x0 = -0xB0; + pA->y0 = 0x54; + pA->u0 = 0; + pA->v0 = 0; + pA->clut = 0x7800; + pA->w = 0x100; + pA->h = 0x20; + D_800EC7DA[idx * 12] = 0x7880; + ADDPRIM2(OTG, pA, mq2, 0xFF000000); + + __asm__(""); + { Sprt24 *b_ = D_800EC7F8; pC = &b_[idx]; } + pC->len = q5; + ((u32 *)&D_800EC7F8[idx])[1] = b1c; + pC->code = 0x64; + pC->r0 = pC->g0 = pC->b0 = D_800EC68C; + pC->x0 = 0x50; + w60 = 0x60; + pC->y0 = 0x54; + pC->u0 = 0; + pC->v0 = 0; + __asm__(""); + pC->clut = 0x7800; + pC->w = w60; + pC->h = 0x20; + D_800EC80A[idx * 12] = 0x7880; + ADDPRIM2(OTG, pC, mq2, 0xFF000000); + + } + goto final; +blk2: + { Sprt24 *b_ = D_800EC828; pF = &b_[idx]; } + pF->len = 5; + ((u32 *)&D_800EC828[idx])[1] = 0xE100008F; + mq2 = 0xFFFFFF; + pF->code = 0x64; + pF->r0 = pF->g0 = pF->b0 = D_800EC68C; + pF->x0 = -0x40; + pF->y0 = 0x44; + pF->u0 = 0; + pF->v0 = 0; + pF->clut = 0x7800; + pF->w = 0x80; + pF->h = 0x40; + D_800EC83A[idx * 12] = 0x7A80; + ADDPRIM2(OTG, pF, mq2, 0xFF000000); + + { Sprt24 *b_ = D_800EC858; pD = &b_[idx]; } + pD->len = 5; + ((u32 *)&D_800EC858[idx])[1] = 0xE100008D; + pD->code = 0x64; + pD->r0 = pD->g0 = pD->b0 = D_800EC68C; + pD->x0 = -0x70; + pD->y0 = D_800EC678 * 32 + 0x44; + pD->u0 = 0; + pD->v0 = 0; + pD->clut = 0x7800; + pD->w = 0xE0; + pD->h = 0x20; + D_800EC86A[idx * 12] = 0x7800; + ADDPRIM2(OTG, pD, mq2, 0xFF000000); + +final: + tv0 = 0xE100008A; + tv1 = 0xE100008C; + tv2 = 0xE100008E; + tv3 = 0xE100009A; + tv4 = 0xE100009C; + { Sprt24 *b_ = D_800EC6A8; p1 = &b_[idx]; } + c5 = 5; + p1->len = c5; + ((u32 *)&D_800EC6A8[idx])[1] = tv0; + p1->code = 0x64; + p1->r0 = p1->g0 = p1->b0 = D_800EC68C; + p1->x0 = -0x140; + p1->y0 = -0xDC; + p1->u0 = 0; + c24 = 0x24; + p1->v0 = c24; + p1->clut = 0x78C0; + p1->w = 0x100; + p1->h = 0xDC; + { Sprt24 *b_ = D_800EC6D8; p2 = &b_[idx]; } + p2->len = c5; + ((u32 *)&D_800EC6D8[idx])[1] = tv1; + p2->code = 0x64; + p2->r0 = p2->g0 = p2->b0 = D_800EC68C; + p2->x0 = -0x40; + p2->y0 = -0xDC; + p2->u0 = 0; + p2->v0 = c24; + p2->clut = 0x78C0; + p2->w = 0x100; + p2->h = 0xDC; + { Sprt24 *b_ = D_800EC708; p3 = &b_[idx]; } + p3->len = c5; + ((u32 *)&D_800EC708[idx])[1] = tv2; + p3->code = 0x64; + p3->r0 = p3->g0 = p3->b0 = D_800EC68C; + p3->x0 = 0xC0; + p3->y0 = -0xDC; + p3->u0 = 0; + p3->v0 = c24; + p3->clut = 0x78C0; + p3->w = 0x80; + p3->h = 0xDC; + tv2 = 0xE100009E; + { Sprt24 *b_ = D_800EC738; p4 = &b_[idx]; } + p4->len = c5; + ((u32 *)&D_800EC738[idx])[1] = tv3; + p4->code = 0x64; + p4->r0 = p4->g0 = p4->b0 = D_800EC68C; + p4->x0 = -0x140; + p4->y0 = 0; + p4->u0 = 0; + p4->v0 = 0; + p4->clut = 0x78C0; + p4->w = 0x100; + p4->h = 0xDC; + { Sprt24 *b_ = D_800EC768; p5 = &b_[idx]; } + p5->len = c5; + ((u32 *)&D_800EC768[idx])[1] = tv4; + p5->code = 0x64; + p5->r0 = p5->g0 = p5->b0 = D_800EC68C; + p5->x0 = -0x40; + p5->y0 = 0; + p5->u0 = 0; + p5->v0 = 0; + __asm__(""); + { Sprt24 *b_ = D_800EC798; p6 = &b_[idx]; } + p5->clut = 0x78C0; + p5->w = 0x100; + p5->h = 0xDC; + p6->len = c5; + ((u32 *)&D_800EC798[idx])[1] = tv2; + tag6 = *(u32 *)p6; + p6->code = 0x64; + p6->r0 = p6->g0 = p6->b0 = D_800EC68C; + p6->x0 = 0xC0; + p6->y0 = 0; + p6->u0 = 0; + p6->v0 = 0; + p6->clut = 0x78C0; + p6->w = 0x80; + p6->h = 0xDC; + ot = (u32 *)(D_800AA60C + idx * 0x4000); + tag6 &= 0xFF000000; + *(u32 *)p6 = tag6 | (*ot & 0xFFFFFF); + *ot = (*ot & 0xFF000000) | ((u32)p6 & 0xFFFFFF); + ADDPRIMB(ot, p5) + ADDPRIMB(ot, p4) + ADDPRIMB(ot, p3) + ADDPRIMB(ot, p2) + ADDPRIMB(ot, p1) +} diff --git a/.run/P32/t5x/reports/func_800CF3E8.md b/.run/P32/t5x/reports/func_800CF3E8.md new file mode 100644 index 0000000000..52351c30bb --- /dev/null +++ b/.run/P32/t5x/reports/func_800CF3E8.md @@ -0,0 +1,74 @@ +# md_MAIN_003:func_800CF3E8 (469 ins) — Fable T5x: **MATCH** (leaf `match_one` 469/469, real-TU `rtu_match` MATCH) + +Draft: `.run/P32/t5x/fable/func_800CF3E8.c` (= work `v_X5.c`). Alternate byte-identical drafts: `v_C1.c` (adds the S83 pointer +launder), `v_X4.c` (tag read spelled at the chain). Baseline reproduced first: prior best 27 @ 469; S83 launder variant 79 @ 470. +Relocation audit: all 20 data symbols present with the target's exact counts (`relocs_mine.txt` vs `relocs_target.txt` in the work dir). + +## The mechanism, as READ from the dumps (all in `.run/P32/t5x/work/func_800CF3E8/rtl_*/t.i.{sched,sched2,lreg}`) +Both prior reports located the residual in the right window (the p6 tag load at 380 vs 399/384) but attributed the placement to +cse/sched1 alone. The dumps show THREE passes each contributing one defect, and the row closes only when all three are fixed: + +1. **sched1: the tag load was BIRTHING-BOOSTED.** In the launder variant's `-dS` the load (insn 1182) is `7f000001` in the + ready list at T-73 and is picked the instant its consumer is placed — `adjust_priority` (sched.c:2507) → `birthing_insn_p` + (:2469, `reg_n_sets == 1`, dest live). A boosted load sinks to just before its use; NO source spelling moves it while the boost + is alive (this is why S83's 14-position `tag6` birth sweep and the 32-position hoist sweep were inert). + **Lever: a second LIVE set of the loaded pseudo — `tag6 = *(u32*)p6; … tag6 &= 0xFF000000;`** (compound assignment reuses the + variable's pseudo as the `and`'s dest: `(set (reg 81) (and (reg 81) (reg mhi)))`). reg_n_sets = 2 → priority 6 in `-dS` + (`v_A` trace: "1121 (6)"), and sched1's output becomes `sb len, [tpage addr], sw tpage, lw tag, lui m24, sb code, lbu, ori m24, + x0 y0 u0 v0 clut w h, colour×3, sll, la, addu ot, lui mhi, lw *ot, and tag, and otv, or, sw` — the tag load is now ahead of + `sb code`/`lbu` in LUID order. (Same gate as §501-C/§501-D; the compound form keeps the variable single-death/local, unlike the + `__asm__ volatile("" : "=r"(x))` dial, which measured 39 here — it makes the tag a 2-death global allocno.) + +2. **sched2 decides the FINAL slot, not sched1.** Post-reload both the load `(mem:SI (reg 3))` and the field stores + `(mem/s (plus (reg 3) N))` share the hard base, `memrefs_conflict_p` disambiguates, and the unit-hazard rule "a load is blocked + for 1 cycle right after a store" (`-dR`: `blocking insn 1182 for 1 cycles` at every store pick) walks the load upward past every + consecutive store; it stops where it loses a LUID tie to `lbu`(1130)/`sb code`(1125) — which is exactly what (1) fixes. + Two more sched2 facts, both byte-verified: + * the S83 +1 `nop` was the ot-USE phantom `__asm__("" :: "r"(ot))` (insn 1179): in sched2 it ties `and tag` at priority 7, + wins on LUID (sched1 parked it at the block end), and is picked between `lw *ot` and `and otv` — a zero-byte insn filling the + load-delay slot in the MODEL only, so gas emits a real nop. **Remove it** (re-adding it: 70 @ 470, ablation X2). + * `lui m24` lands at 378 (between `sb $a2,3($v1)` and `sw tpage`) purely from its sched2 anti-dependence on `$a2` (c5's last + use) + the ALU-vs-memory hazard order — which requires m24 to BE in `$a2`, i.e. the allocation below. The fence + `__asm__("")` after the tpage store must go (everything after a traditional asm depends on it, so `lui m24` could never reach + 378; re-adding it: 8 @ 469, ablation X3). The fence BEFORE p6's birth stays (removing it: 461 @ 471, ablation X6). + +3. **local-alloc: m24 must out-rank mhi.** `qty_compare` (local-alloc.c:1579) = `floor_log2(n_refs)·n_refs·size/(death−birth)` + with birth/death from the POST-sched1 positions and `n_refs` from flow (flow.c:2067/2315/2501/2711 — written BEFORE combine; + combine.c:56 documents that it does not adjust them). With the hand-written `(x & 0xFF000000) | (y & 0xFFFFFF)` macro both + masks have 13 refs; mhi's boosted `lui` sinks to the chain (born at 344) while m24's is born at 327, so mhi's range is 16 + shorter → mhi is allocated first and takes the lowest free register `$a2` (`v_A` lreg: `Register 452 in 6`, m24 `455 in 10`, + tag `81 in 8`) — the target needs m24→`$a2`, mhi→`$t0`, tag→`$t2`. + **Lever: spell the OT link the way libgpu does — the `P_TAG` bitfield `setaddr(p, getaddr(ot)); setaddr(ot, p)`.** + `store_fixed_bit_field` (expmed.c) re-masks the already-masked value with `0xFFFFFF` (`must_and`), an `and` that cse cannot + fold (both operands are registers) and combine removes later — but flow has already counted it: m24 gets 3 refs per prim + instead of 2 (18–19 total, floor_log2 = 4) and its priority roughly doubles → allocated before mhi → `$a2`; mhi then takes + `$t0`; the tag (range [380,401], overlapping 0xDC/mhi in `$t0`) takes `$t2`; every block-head constant keeps its register. + Measured: bitfield on prims 5..1 (p6 kept manual): 74 → **2**; the same draft with the manual macro on all six: 74 (B5). + §364's -O2 warning (a `/s` tag store lets cse drop the OT re-read) does not bite here because BOTH sides of the link are + `P_TAG` accesses, so the `*ot` re-read is `/s` too and is invalidated correctly — the 469-instruction count is preserved. + +4. The last 2 rows (p5 `x0`/`y0` order, 362/363) were the baseline's deliberate `y0; x0` source swap for the old basin; natural + order (`x0; y0`) closes it: 2 → **MATCH**. (S83 measured the swap at 31 — in the old basin, where it was paid for elsewhere.) + +## Ablations on the MATCH (leaf match_one; every one re-measured, not inferred) +| change | result | meaning | +|---|---|---| +| X1: `(tag6 & 0xFF000000)` instead of `tag6 &= …` | 89 @ 469 | the second SET (boost kill) is load-bearing | +| X2: re-add `__asm__("" :: "r"(ot))` | 70 @ 470 | the phantom steals the load-delay slot in sched2 | +| X3: re-add `__asm__("")` after the tpage store | 8 @ 469 | blocks `lui m24` from reaching 378 | +| X4: tag read spelled at the chain (after `ot = …`) | MATCH | source position of the read is NOT load-bearing once un-boosted | +| X5: drop the S83 pointer launder | MATCH (delivered) | with the read spelled before the field stores, the pseudo-base deps pin it anyway | +| X6: drop the fence before p6's birth | 461 @ 471 | §194-A fence still load-bearing | +| B5: manual mask macro on all six prims | 74 @ 469 | the bitfield's redundant `and` (m24 refs) is load-bearing | +| V1/V2: `__asm__ volatile("" : "=r"(tag6))` instead of `&=` | 39 @ 469 | 2-death → global allocno; different basin | + +## For the cookbook (proposed §500-H amendment / new §) +* "A load through a pinned base is scheduled late" has THREE owners: sched1's birthing boost (kill with a second live set of the + loaded variable — compound assignment, zero bytes), sched2's load-after-store hazard walk (the final slot; LUID ties with + neighbouring memory ops decide where it stops), and local-alloc's `qty_compare` (register roles follow the ranges the new + order creates). Read `-dS` for `7f000001` on the load, `-dR` for `blocking insn N`, `-dl` for `Register N in R`. +* A zero-byte `__asm__("" :: "r"(x))` USE is not free in the scheduler: it occupies a cycle and can fill a load-delay slot in + the model only → a real nop from gas. Check the `-dR` ready list around every `lw`/`and` pair before keeping one. +* `reg_n_refs` is flow's count and survives combine: any RTL redundancy that only combine removes (a bitfield's `must_and` + re-mask, expmed.c `store_fixed_bit_field`) still weights local-alloc's priority. The libgpu `P_TAG` bitfield `addPrim` is + therefore not just an operand-order lever (§364) but a REGISTER-PRIORITY lever at -O2 for the 0xFFFFFF mask. diff --git a/.run/P32/t5x/resume_queue.txt b/.run/P32/t5x/resume_queue.txt index 50303ca074..8b7eb5e378 100644 --- a/.run/P32/t5x/resume_queue.txt +++ b/.run/P32/t5x/resume_queue.txt @@ -1,8 +1,7 @@ # T4b Fable resume queue (Drew 2026-09-05: "resume agents, but not all at once, just 3 at a time"). Delete a line when resumed. # fn agentId closeness -func_80185810 a547e70e9a7735c8a 35 func_8017DC80 a7c2c1e4865e5590d 46 func_800CF408 aac6ce9fdfdcc093c 49 func_800CF6D0 a7f7e477e9878acdd 137 func_80011380 a60ffea4022b891fc 6 (-O0, §474 PROVED — last) -# RUNNING: func_800CD92C a5587b9d5c7006214 · func_80039308 a55fbfb4fb896bd71 · func_800CF3E8 ad27049c8fdf1e816 (DONE: func_800391D4 func_80039DEC func_800CD674 func_8017DF28 func_80020DA4 func_801834A4 MATCH; func_80032A74 NEAR 1) +# RUNNING: func_800CD92C a5587b9d5c7006214 · func_80039308 a55fbfb4fb896bd71 · func_80185810 a547e70e9a7735c8a (DONE MATCH: 391D4 39DEC CD674 DF28 20DA4 1834A4 CF3E8; NEAR: 32A74 1) diff --git a/.run/P32/t5x/verdicts.jsonl b/.run/P32/t5x/verdicts.jsonl index d34af9e943..9bf71ec132 100644 --- a/.run/P32/t5x/verdicts.jsonl +++ b/.run/P32/t5x/verdicts.jsonl @@ -5,3 +5,4 @@ {"fn": "func_80020DA4", "binary": "main", "arm": "fable", "status": "MATCH", "closeness": 0, "compiles": true, "draft_path": ".run/P32/t5x/fable/func_80020DA4.c", "note": "unpin e0 ($6 pin made $a2 regs_ever_live -> bad_spill_regs -> forbidden at reload's retry_global_alloc, reload1.c:486/3497/3651-3660/709; the LO-parked mult results are retried first-fit) + two zero-byte launders steering local-alloc qty_compare: e1 launder before `dst[6] = -e1` (qty 6666->8750, allocated before e0, holds $v1) and p1 launder between its andi and sll (undoes the global tie). Ladder 2->37->15->6->MATCH. Coordinator rtu MATCH 100/100; gate_main BANKED 143dbb89", "session": "491895ad"} {"fn": "func_801834A4", "binary": "ov_SC03_105", "arm": "fable", "status": "MATCH", "closeness": 0, "compiles": true, "draft_path": ".run/P32/t5x/fable/func_801834A4.c", "note": "CSE-quantity split, not loop.c: expand_divmod (expmed.c:3034-3057) emits const/mult/sra/subu adjacently from ONE op0 so loop.c treats mult and sra alike (force_movables loop.c:1226 doubles K's savings -> hoists all, or a hard-reg dividend kills all). cse.c canon_reg never rewrites a hard reg but hashes by reg_qty: `sign = half >> 31` at the loop TOP (movable, maybe_never==0) + `hh = half` (register $2) + `pos[0] -= hh / 3` makes the mult read $2 (not movable; K unlinked, inline) while the division's sra is CSE'd into sign (hoisted); combine folds the copy into the mult. Traps: a callee-saved pin enters regs_ever_live before combine -> global pass 0 gives it to j (7); an asm feed for the unread sign store adds +2 weighted refs and swaps half/sign in allocno_compare (4) -> a dead `sign = 0` initializer instead (flow deletes it). Coordinator rtu MATCH 106/106; bank.sh byte-identical", "session": "491895ad"} {"fn": "func_80032A74", "binary": "main", "arm": "fable", "status": "NEAR", "closeness": 1, "compiles": true, "draft_path": ".run/P32/t5x/fable/func_80032A74.c", "note": "rtu DIFF 1 in src/800_b_2.c (idx 244 lh vs lhu; frame 0x78 exact); the lhu respelling is code-exact 422/422 with vars=56. Residual mechanism READ + byte-reproduced: the never-referenced 8-byte slot at 0x48 is a GHOST pseudo — a stranded middle temp of a 3-insn combine whose refs-zeroing is skipped when newi2pat!=0 (combine.c:2306-2313); zero occurrences -> regclass 'ST_REGS or none' -> reload1.c:658 alter_reg(i,-1) 8-byte slot in regno order = right after the three param slots (ghost1 reproducer vars=8, no code). The caller-save-area hypothesis (S83 HYPOTHESIS.md) is REFUTED: order_regs_for_reload (reload1.c:3606-3700) picks zero-use regs first so a pseudo homed in $t0 makes $t1 the spill reg (+30 rows); an area needs caller_save_needed (global.c:1085-1091) and a pseudo kept in a call-used reg with save/restore at every live call (caller-save.c:264, 349-470). The only ghost species from a memory value is the SIGN_EXTEND narrow-load split (combine.c:1887-1930) whose signature IS lh; a jump-target second promotion is folded by cse follow-jumps (n0a/b/c: vars 56, +86 rows); a fall-through one needs a register sign_extend MIPS lacks; the generic two-SETs split (combine.c:1963-2020) has no candidate (cse pre-folds constant offsets). Inert: expA/expB, n16 (2 ghosts), n0a/n0b/n0c. 402k tokens, 23 min", "session": "491895ad"} +{"fn": "func_800CF3E8", "binary": "md_MAIN_003", "arm": "fable", "status": "MATCH", "closeness": 0, "compiles": true, "draft_path": ".run/P32/t5x/fable/func_800CF3E8.c", "note": "27 -> MATCH 469/469; three passes each owned one defect: (1) sched1 birthing boost on the tag load (sched.c:2507/2469) killed with a SECOND LIVE SET of the loaded variable `tag6 = *(u32*)p6; ... tag6 &= 0xFF000000;` (reg_n_sets=2, zero bytes); (2) the S83 +1 nop was the phantom `__asm__(\"\" :: \"r\"(ot))` USE filling the load-delay slot in sched2's model only — removed, plus the fence after the tpage store removed so `lui m24` reaches slot 378 via its $a2 anti-dep; (3) local-alloc qty_compare had mhi out-ranking m24 for $a2 — spelling prims 5..1's OT link as libgpu's P_TAG bitfield adds store_fixed_bit_field's redundant 0xFFFFFF re-mask (folded by combine, counted by flow: m24 13 -> 18 refs) so m24 wins $a2, mhi -> $t0, tag -> $t2; last 2 rows = p5 x0/y0 in natural source order. Ablations: non-compound tag 89; phantom re-added 70@470; fence re-added 8; manual masks on all six prims 74. Coordinator rtu MATCH 469/469; bank.sh byte-identical", "session": "491895ad"} diff --git a/docs/backlog.md b/docs/backlog.md index aeb18a3119..0d835f1de8 100644 --- a/docs/backlog.md +++ b/docs/backlog.md @@ -2,7 +2,7 @@ > Generated by `tools/backlog.py render` from `.run/backlog.jsonl`. These are functions the Phase-21 automation got **close** on but did NOT byte-match. The whole-binary byte-gate is the sole arbiter (G3/P9): **byte-matches bank and are NOT listed here** — only genuine near-misses/blockers are. Ranked by hand-session priority: **reach** (×N propagation leverage) → **closeness** (match_one mismatch count, lower = closer) → **size**. Each row's `best_draft` is the closest C the machine reached — resume from there. -**Open near-misses:** 10 · by status {'near': 9, 'failed': 1} · by class {'WALL-CANDIDATE': 1, 'WALL-PROVED': 1, 'SCHED': 4, 'REGALLOC': 1, 'ALIAS': 1, 'FRAME': 1, None: 1} +**Open near-misses:** 9 · by status {'near': 8, 'failed': 1} · by class {'WALL-CANDIDATE': 1, 'WALL-PROVED': 1, 'SCHED': 4, 'REGALLOC': 1, 'FRAME': 1, None: 1} | # | addr | reach | class | nins | status | closeness | where it stuck | best draft | |--:|------|------:|-------|-----:|--------|----------:|----------------|------------| @@ -10,9 +10,8 @@ | 2 | func_80011380 | None | WALL-PROVED | 192 | near | 6 | §474 PROVED C-level floor (boot -O0): fold-const.c:882 split_tree merges MULT(MULT(i,2),2); the two escapes each cost one instruction (stupid.c:497 adjacency / expand_decl use-brackets); §388 -O0 colouring oracle. Pinned S79 #8; re-probed S83 in the real TU: DIFF 6 (unchanged) | `.run/m3/opus/func_80011380.c` | | 3 | func_800CD92C | None | SCHED | 247 | near | 15 | map §S7 prologue WEAVE: the {sw,lui,ori} groups for 0xE100008D/8F land after the 9-insn li block instead of before — the §17 pins reproduce the ALLOCATION but the hoist happens in sched2. Same SPRT family as func_800CD674 (§364 mirror levers applied) | `.run/P32/t3/opus/func_800CD92C.c` | | 4 | func_80039308 | None | REGALLOC | 518 | near | 17 | sched2 + cross-block regalloc: preheader 49/50 swap, un-spellable addu $a2,$a0,$zero (every p=r form cse-propagated), a temp on $t0 vs $s7, and 11 insns of one alias fact (2nd D_80073140[j] load cannot schedule above the D_800C7D20 store from C; /s unlock costs the address allocation, net 20-24). 34->17 via s16 b4 widening copy + dead-local identity sweep (.run/P32/t3/restored/sweep_func_80039308.py) + $2 pin. permuter_ils --klass REGALLOC 2x150s: no gain | `.run/P32/t3/opus/func_80039308.c` | -| 5 | func_800CF3E8 | None | ALIAS | 469 | near | 27 | ONE cause: the pinned-base alias basin (§500-D1) in the p5/p6 tail; blocks 1-2 byte-exact (idx 0-361). 54->27 via blk2 constant birth order + birthing-boost local w60 + §194-A fence relocation. Inert: all 9 pins load-bearing (+5..+1409), asm position x7, h6 hoist 32x2, tag reshape, P_TAG ADDPRIM, array p6 stores, ~92k annealed variants. Untested: an unpinned alias of p6 for the tag load alone (ONE Opus second look allowed) | `.run/P32/t3/opus/func_800CF3E8.c` | -| 6 | func_80185810 | None | SCHED | 489 | near | 35 | [permuter] 4 emission windows (see report .run/P32/t3/reports/func_80185810__opus__*.md); exact length, rtu-clean | `.run/P32/t3/opus/func_80185810.c` | -| 7 | func_8017DC80 | None | FRAME | 346 | near | 46 | the historic -33 LENGTH wall CLOSED (GTE macros must be REAL macros — the TU house block; the splat Handwritten tag is wrong): 346/346, exact 0x70 frame + 9 callee-saved. Residual: reload-slot frame + the la $a0 slot; cse1 unifies OT index and n<4 across func_80010A08(8) (§500-D2 zero-byte asm retire) | `.run/P32/t3/opus/func_8017DC80.c` | -| 8 | func_800CF408 | None | SCHED | 178 | near | 49 | [permuter] 3 hunks: two prologue sched2 slots, an mlo/mhi allocno tie, a 3-insn block-2 head hoist. Two LENGTH-bearing pins found (tp $17 shared by 0xE1000087/97 = the 6th callee-saved; ob $10 fixes the $t1/$t2/$t3 rotation, 56->49). §351 family (func_8001212C -O0 / func_8017DD04 -O2 exemplars) | `.run/P32/t3/opus/func_800CF408.c` | -| 9 | func_800CF6D0 | None | SCHED | 249 | near | 137 | sched1 rank_for_schedule last-insn-CLASS tie (every store priority 2, equal refs; QImode stores grouped, loads floated, HImode after — 5 of 6 blocks) + $t1<->$t3 local-alloc swap of the two masks. 249/249 exact length only with tpage-before-len field order (19 swept). Inert at 137: pins on tpage constants/masks, asm re-ties, volatile/memory fences, /s-denial on any store subset, *0x4000 vs <<14, p++ vs p+0x18, / swap. decomp-permuter 122 was semantically wrong (R63) | `.run/P32/t3/opus/func_800CF6D0.c` | -| 10 | func_80062144 | None | | None | failed | | won't compile standalone (loose-typing / missing decl) | | +| 5 | func_80185810 | None | SCHED | 489 | near | 35 | [permuter] 4 emission windows (see report .run/P32/t3/reports/func_80185810__opus__*.md); exact length, rtu-clean | `.run/P32/t3/opus/func_80185810.c` | +| 6 | func_8017DC80 | None | FRAME | 346 | near | 46 | the historic -33 LENGTH wall CLOSED (GTE macros must be REAL macros — the TU house block; the splat Handwritten tag is wrong): 346/346, exact 0x70 frame + 9 callee-saved. Residual: reload-slot frame + the la $a0 slot; cse1 unifies OT index and n<4 across func_80010A08(8) (§500-D2 zero-byte asm retire) | `.run/P32/t3/opus/func_8017DC80.c` | +| 7 | func_800CF408 | None | SCHED | 178 | near | 49 | [permuter] 3 hunks: two prologue sched2 slots, an mlo/mhi allocno tie, a 3-insn block-2 head hoist. Two LENGTH-bearing pins found (tp $17 shared by 0xE1000087/97 = the 6th callee-saved; ob $10 fixes the $t1/$t2/$t3 rotation, 56->49). §351 family (func_8001212C -O0 / func_8017DD04 -O2 exemplars) | `.run/P32/t3/opus/func_800CF408.c` | +| 8 | func_800CF6D0 | None | SCHED | 249 | near | 137 | sched1 rank_for_schedule last-insn-CLASS tie (every store priority 2, equal refs; QImode stores grouped, loads floated, HImode after — 5 of 6 blocks) + $t1<->$t3 local-alloc swap of the two masks. 249/249 exact length only with tpage-before-len field order (19 swept). Inert at 137: pins on tpage constants/masks, asm re-ties, volatile/memory fences, /s-denial on any store subset, *0x4000 vs <<14, p++ vs p+0x18, / swap. decomp-permuter 122 was semantically wrong (R63) | `.run/P32/t3/opus/func_800CF6D0.c` | +| 9 | func_80062144 | None | | None | failed | | won't compile standalone (loose-typing / missing decl) | |