From 99b65ef94a0e95017304e35eb20a04268fb44bb7 Mon Sep 17 00:00:00 2001 From: Drew T <50529377+Druthulu@users.noreply.github.com> Date: Sat, 5 Sep 2026 19:09:36 -0600 Subject: [PATCH] =?UTF-8?q?docs(phase-32):=20T4b=20=E2=80=94=20func=5F8018?= =?UTF-8?q?5810=2035=20->=2013=20NEAR=20(Fable,=20471k=20tokens):=203=20of?= =?UTF-8?q?=204=20windows=20closed=20(P=5FTAG=20bitfield=20qty=20prioritie?= =?UTF-8?q?s,=20sched1=20flush=5Fpending=5Flists=20load=20order,=20hard-re?= =?UTF-8?q?g=20dests=20not=20boosted);=20residual=20=3D=20the=20cl=20secon?= =?UTF-8?q?d-set=20fence;=20backlog=20row=20+=20draft=20+=20report=20kept?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .run/P32/t5x/fable/func_80185810.c | 368 ++++++++++++++++++++++++++ .run/P32/t5x/reports/func_80185810.md | 89 +++++++ .run/P32/t5x/resume_queue.txt | 3 +- .run/P32/t5x/verdicts.jsonl | 1 + .run/backlog.jsonl | 1 + docs/backlog.md | 6 +- 6 files changed, 463 insertions(+), 5 deletions(-) create mode 100644 .run/P32/t5x/fable/func_80185810.c create mode 100644 .run/P32/t5x/reports/func_80185810.md diff --git a/.run/P32/t5x/fable/func_80185810.c b/.run/P32/t5x/fable/func_80185810.c new file mode 100644 index 000000000..03c78b684 --- /dev/null +++ b/.run/P32/t5x/fable/func_80185810.c @@ -0,0 +1,368 @@ +/* func_80185810 (ov_SC03_105, sub ov_SC03_105_jr_80181C84) -- 489 ins. + * + * COMPILER-EMITTED gcc-2.7.2 -O2 C using PsyQ GTE inline-asm macros (splat's + * "Handwritten function" banner is wrong). Billboard-sprite drawer: builds the + * rot/trans matrix (3 paths), RTPS the actor position, rejects it outside + * +-200/+-160, RTPT the two extent vectors through D_801BC9B4, fills a 0x28-byte + * POLY_FT4 packet off the D_800A5E60 bump allocator and addPrim()s it into the OT + * at &D_800A6610[D_800B9A02 << 14] (read base-relative: D_800AF630 + 0xA3D2). + * + * STATUS (fable arm, P32 T5x): exact length 489/489, closeness 13 (from 35). + * Every register allocation matches; the 13 rows are one sched1/sched2 window + * (idx 363-380): the zero-byte fence after `p[7] |= ...` forbids the target's + * interleave of `sll $a2,14` / `andi $a3,0xFFFF` / `li 2; subu 2-mode` into the + * tpage/code store window, and `la D_800A6610` vs the two `lhu 0x16($s0)` reloads + * take $v1/$v0 instead of $v0/$v0. Mechanism as read from the -dS/-dl dumps: the + * fence's REGISTER effect is that it forces the unboosted 2nd set `cl &= 0xFFFF` + * (pri 3) to be picked before the fence in sched1's backward pass; without it that + * insn is starved by boosted insns until T-34 and the w-chain's two `cl` reads + * (which carry anti-dependences on it) float to the block head, take $a0/$a1 in + * local-alloc, and rotate the quartet. See .run/P32/t5x/reports/func_80185810.md. + * + * LEVERS (each measured with match_one; the mechanism for each is in the report): + * 1. libgpu P_TAG bitfield for the OT link (S364/S500-C, exemplar src/800.c + * func_80023BF0): flips the otz/(*p) local-alloc qty priorities (35 -> 26); + * integer add `(otz << 2) + (u32)ob` puts the shifted index first (-> 25). + * 2. HI temps `t20/t22` so the D_801BA6B0 load sits between the two coordinate + * loads and the two stores in RTL order: sched1's 33rd-memory-op pending-list + * FLUSH hangs the last five memory ops off `sh 0x18(s0)`; the lbu needs the + * highest LUID of the three loads yet must precede the stores (-> 19). + * 3. `register` pins on the block-local quartet uu $4 / mode $5 / ot16 $6: a + * hard-reg destination is NOT birthing-boosted (trace: all three loads at + * priority 2), so the LUID tie-break gives the target's load order (-> 14). + * 4. `register u32 shf __asm__("$3")` for `2 - mode` (-> 13). + * Kept from the opus draft (still load-bearing, each re-measured): base as a local, + * the packet word as ONE expression, the two-use `tb` temp, the fence. + */ + +#ifndef BFM_ENGINE_TYPES_H +typedef struct { short m[3][3]; long t[3]; } MATRIX_80188114; +#endif + +extern u8 D_800AF630[]; +extern MATRIX_80188114 D_801BC9B4; +extern u8 *D_800A5E60; +extern u8 D_800A6610[]; +extern u8 D_801BA6B0; +extern void func_80185FB4(s32 a0, s32 a1, s32 a2); +extern void func_8001F730(s32 a0, void *a1, void *a2); + +#define gte_SetRotMatrix_85810(r0) __asm__ volatile ( \ + "lw $12, 0( %0 );" \ + "lw $13, 4( %0 );" \ + "ctc2 $12, $0;" \ + "ctc2 $13, $1;" \ + "lw $12, 8( %0 );" \ + "lw $13, 12( %0 );" \ + "lw $14, 16( %0 );" \ + "ctc2 $12, $2;" \ + "ctc2 $13, $3;" \ + "ctc2 $14, $4" \ + : \ + : "r"( r0 ) \ + : "$12", "$13", "$14" ) + +#define gte_SetTransMatrix_85810(r0) __asm__ volatile ( \ + "lw $12, 20( %0 );" \ + "lw $13, 24( %0 );" \ + "ctc2 $12, $5;" \ + "lw $14, 28( %0 );" \ + "ctc2 $13, $6;" \ + "ctc2 $14, $7" \ + : \ + : "r"( r0 ) \ + : "$12", "$13", "$14" ) + +#define gte_ldclmv_85810(r0) __asm__ volatile ( \ + "lhu $12, 0( %0 );" \ + "lhu $13, 6( %0 );" \ + "lhu $14, 12( %0 );" \ + "mtc2 $12, $9;" \ + "mtc2 $13, $10;" \ + "mtc2 $14, $11" \ + : \ + : "r"( r0 ) \ + : "$12", "$13", "$14" ) + +#define gte_rtir_85810() __asm__ volatile ("nop;nop;mvmva 1, 0, 3, 3, 0") + +#define gte_stclmv_85810(r0) __asm__ volatile ( \ + "mfc2 $12, $9;" \ + "mfc2 $13, $10;" \ + "mfc2 $14, $11;" \ + "sh $12, 0( %0 );" \ + "sh $13, 6( %0 );" \ + "sh $14, 12( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "$13", "$14", "memory" ) + +#define gte_ldlvl_85810(r0) __asm__ volatile ( \ + "lhu $13, 4( %0 );" \ + "lhu $12, 0( %0 );" \ + "sll $13, $13, 16;" \ + "or $12, $12, $13;" \ + "mtc2 $12, $0;" \ + "lwc2 $1, 8( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "$13" ) + +#define gte_rt_85810() __asm__ volatile ("nop;nop;mvmva 1, 0, 0, 0, 0") + +#define gte_stlvnl_85810(r0) __asm__ volatile ( \ + "swc2 $25, 0( %0 );" \ + "swc2 $26, 4( %0 );" \ + "swc2 $27, 8( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_ldv0_85810(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_rtps_85810() __asm__ volatile ("nop;nop;rtps") + +#define gte_stsxy_85810(r0) __asm__ volatile ( \ + "swc2 $14, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsz_85810(r0) __asm__ volatile ( \ + "swc2 $19, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_ldv3_85810(r0, r1, r2) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 0( %1 );" \ + "lwc2 $3, 4( %1 );" \ + "lwc2 $4, 0( %2 );" \ + "lwc2 $5, 4( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) ) + +#define gte_rtpt_85810() __asm__ volatile ("nop;nop;rtpt") + +#define gte_stsxy0_85810(r0) __asm__ volatile ( \ + "swc2 $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy1_85810(r0) __asm__ volatile ( \ + "swc2 $13, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stflg_85810(r0) __asm__ volatile ( \ + "cfc2 $12, $31;" \ + "nop;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stszotz_85810(r0) __asm__ volatile ( \ + "mfc2 $12, $19;" \ + "nop;" \ + "sra $12, $12, 2;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +void func_80185810(s32 a0) +{ + typedef struct { u16 vx, vy, vz, pad; } UV_85810; + typedef struct { short m[3][3]; long t[3]; } MTX_85810; + typedef struct { u32 addr : 24; u32 len : 8; } PTag_85810; + + UV_85810 v[5]; /* sp+0x10 .. sp+0x37 */ + MTX_85810 m; /* sp+0x38 */ + long sz; /* sp+0x58 */ + long flag; /* sp+0x5C */ + long otz; /* sp+0x60 */ + u8 *base; + u8 *p; + u8 *ob; + register u32 ot16 __asm__("$6"); + PTag_85810 *q; + u32 flags; + s32 spr; + s32 t0; + s32 sx; + s32 sy; + register u32 mode __asm__("$5"); + u32 w; + register u32 uu __asm__("$4"); + u32 cl; + u32 tp; + register u32 shf __asm__("$3"); + u16 t20; + u16 t22; + + base = D_800AF630; + flags = *(u32 *)a0; + spr = *(s32 *)(a0 + 0x24); + + if (*(s32 *)(a0 + 0x20) != 0) { + if (flags & 0x800000) { + gte_SetRotMatrix_85810(base + 0x18); + gte_ldclmv_85810((u8 *)(*(s32 *)(*(s32 *)(a0 + 0x20) + 0x20) + 0x34)); + gte_rtir_85810(); + gte_stclmv_85810(&m.m[0][0]); + gte_ldclmv_85810((u8 *)(*(s32 *)(*(s32 *)(a0 + 0x20) + 0x20) + 0x36)); + gte_rtir_85810(); + gte_stclmv_85810(&m.m[0][1]); + gte_ldclmv_85810((u8 *)(*(s32 *)(*(s32 *)(a0 + 0x20) + 0x20) + 0x38)); + gte_rtir_85810(); + gte_stclmv_85810(&m.m[0][2]); + gte_SetTransMatrix_85810(base + 0x18); + gte_ldlvl_85810((u8 *)(*(s32 *)(*(s32 *)(a0 + 0x20) + 0x20) + 0x48)); + gte_rt_85810(); + gte_stlvnl_85810(&m.t[0]); + } else { + func_80185FB4(*(s32 *)(*(s32 *)(a0 + 0x20) + 0x20), flags, (s32)&m); + gte_SetRotMatrix_85810(base + 0x18); + gte_ldclmv_85810(&m.m[0][0]); + gte_rtir_85810(); + gte_stclmv_85810(&m.m[0][0]); + gte_ldclmv_85810(&m.m[0][1]); + gte_rtir_85810(); + gte_stclmv_85810(&m.m[0][1]); + gte_ldclmv_85810(&m.m[0][2]); + gte_rtir_85810(); + gte_stclmv_85810(&m.m[0][2]); + gte_SetTransMatrix_85810(base + 0x18); + gte_ldlvl_85810(&m.t[0]); + gte_rt_85810(); + gte_stlvnl_85810(&m.t[0]); + } + gte_SetRotMatrix_85810(&m); + gte_SetTransMatrix_85810(&m); + t0 = *(s32 *)(*(s32 *)(a0 + 0x20) + 0x20) + 0x10; + } else { + gte_SetRotMatrix_85810(base + 0x18); + gte_SetTransMatrix_85810(base + 0x18); + t0 = a0 + 4; + } + + gte_ldv0_85810((u8 *)(a0 + 0x14)); + gte_rtps_85810(); + gte_stsxy_85810(&v[0]); + gte_stsz_85810(&sz); + + sx = *(s16 *)&v[0].vx; + if (sx < 0) sx = -sx; + if (sx >= 0xC9) return; + sy = *(s16 *)&v[0].vy; + if (sy < 0) sy = -sy; + if (sy >= 0xA1) return; + + *(u16 *)&D_801BC9B4 = *(u16 *)(a0 + 0xC); + *(u16 *)((u8 *)&D_801BC9B4 + 8) = *(u16 *)(a0 + 0xE); + gte_SetRotMatrix_85810(&D_801BC9B4); + gte_SetTransMatrix_85810(&D_801BC9B4); + + v[3].vx = *(u16 *)(spr + 8) + *(u16 *)(a0 + 0x1C); + v[3].vy = *(u16 *)(spr + 0xA) + *(u16 *)(a0 + 0x1E); + v[3].vz = sz; + v[4].vx = *(u8 *)(spr + 2); + v[4].vy = *(u8 *)(spr + 3); + v[4].vz = sz; + + gte_ldv3_85810(&v[3], &v[4], &v[4]); + gte_rtpt_85810(); + gte_stsxy0_85810(&v[1]); + gte_stsxy1_85810(&v[2]); + gte_stflg_85810(&flag); + gte_stszotz_85810(&otz); + if (flag & ~0x1000) return; + + mode = (flags >> 24) & 3; + p = D_800A5E60; + D_800A5E60 = p + 0x28; + w = mode << 7; + p[3] = 9; + p[7] = 0x2C; + ot16 = *(u16 *)(base + 0xA3D2); + uu = *(u16 *)(spr + 4); + cl = *(u16 *)(spr + 6); + p[7] = 0x2E; + *(u16 *)(p + 0x16) = w | ((flags >> 23) & 0x60) | ((cl & 0x100) >> 4) + | ((uu & 0x3C0) >> 6) | ((cl & 0x200) << 2); + p[7] |= (flags & 0x40) >> 6; + __asm__ volatile(""); + ob = &D_800A6610[ot16 << 14]; + uu -= (*(u16 *)(p + 0x16) & 0xF) << 6; + shf = 2 - mode; + uu <<= shf; + cl &= 0xFFFF; + p[0xC] = uu; + if (*(u16 *)(p + 0x16) & 0x10) cl -= 0x100; + p[0xD] = cl; + + p[0x14] = p[0xC] + *(u8 *)(spr + 2) - 1; + p[0x15] = p[0xD]; + p[0x1C] = p[0xC]; + p[0x1D] = p[0xD] + *(u8 *)(spr + 3) - 1; + p[6] = 0x80; + p[5] = 0x80; + p[4] = 0x80; + p[0x24] = p[0x14]; + p[0x25] = p[0x1D]; + + *(u16 *)(p + 8) = v[0].vx + v[1].vx; + *(u16 *)(p + 0xA) = v[0].vy + v[1].vy; + *(u16 *)(p + 0x10) = *(u16 *)(p + 8) + v[2].vx; + *(u16 *)(p + 0x1A) = *(u16 *)(p + 0xA) + v[2].vy; + *(u16 *)(p + 0x12) = *(u16 *)(p + 0xA); + *(u16 *)(p + 0x18) = *(u16 *)(p + 8); + t20 = *(u16 *)(p + 0x10); + t22 = *(u16 *)(p + 0x1A); + tp = D_801BA6B0; + *(u16 *)(p + 0x20) = t20; + *(u16 *)(p + 0x22) = t22; + + if (tp == 0) { + u32 t2 = *(u8 *)(spr + 1); + u32 tb = (t2 + 0x100) << 6; + if (t2 < 0xE0) *(u16 *)(p + 0xE) = tb | 0x16; + else *(u16 *)(p + 0xE) = tb | 0x10; + } else { + u32 tb = (tp + 0x100) << 6; + { u32 vv; if (tp < 0xE0) vv = tb | 0x16; else vv = tb | 0x10; *(u16 *)(p + 0xE) = vv; } + } + + if ((flags & 0x300000) == 0x200000) { + func_8001F730(t0, &v[0], p); + } + + q = (PTag_85810 *)((otz << 2) + (u32)ob); + ((PTag_85810 *)p)->addr = q[1].addr; + q[1].addr = (u32)p; +} + +/* RESIDUAL (35 masked mismatches, [permuter] class -- pure emission order, all + * registers and all 489 instructions otherwise identical): + * 343-347 the three loads: mine emits lhu(spr+4), lhu(spr+6), lui/addu/lhu(A3D2); + * the target emits the A3D2 group first. Same multiset, rotated. + * 363-380 the fence boundary: the target interleaves `sll $a2,14` (ob) and + * `andi $a3,0xFFFF` (cl) INTO the `sh 0x16 / lbu 7 / or / sb 7` window + * and computes `2 - mode` there too; the fence that fixes the register + * allocation also forbids exactly that interleave. Both halves of this + * trade were measured: no fence = right order, wrong registers (67). + * 432-438 the D_801BA6B0 lui/lbu sits between the two 0x20/0x22 loads instead of + * after both. + * 466-479 the OT link: otz/(*(s32*)p) occupy $a0/$v1 instead of $v1/$a0. + * All four are rank_for_schedule / INSN_LUID ties (cookbook S49), not reachable + * from any source form tried here. + */ diff --git a/.run/P32/t5x/reports/func_80185810.md b/.run/P32/t5x/reports/func_80185810.md new file mode 100644 index 000000000..546f53f8e --- /dev/null +++ b/.run/P32/t5x/reports/func_80185810.md @@ -0,0 +1,89 @@ +# func_80185810 (ov_SC03_105, 489 ins) — fable arm, session 491895ad (2026-09-05) + +**Result: NEAR, closeness 13** (from 35; exact length 489/489 throughout; every register allocation matches). +Draft: `.run/P32/t5x/fable/func_80185810.c` (= work `v7.c`). Leaf `match_one` 13; real-TU `rtu_match` (split +ov_SC03_105_jr_80181C84) 13, the same rows — no TU plumbing problem. Symbol audit unchanged from the opus report (every +D_/func_ symbol appears in the target's own relocation lines; `D_801BC9BC` reached as `(u8*)&D_801BC9B4 + 8`). +Scratch: `.run/P32/t5x/work/func_80185810/` (dumps_*/ = -dr/-ds/-dc/-dS/-dl/-dg/-dR dumps per variant via `dump.sh`; +`exp/run.py` = named-edit A/B harness; `exp/sweep.py` = the 3,360-variant region-2 sweep; `exp/rtlsum.py` = one-line-per-insn RTL +summariser for a dump). Three of the four windows closed; each mechanism below was READ from the dumps, not inferred. + +## 1. Window 4 (OT link, idx 466–479, 10 rows) → CLOSED. 35 → 25 +Lever: the libgpu `P_TAG` bitfield (§364/§500-C; exemplar `src/800.c func_80023BF0`): +`((PTag*)p)->addr = q[1].addr; q[1].addr = (u32)p;` instead of the hand-written `& 0xFF000000 | & 0xFFFFFF` pair. +Mechanism (`dumps_base/base.i.lreg`, `.i.sched`): local-alloc ties `lw otz → sll → addu q` into ONE qty (block_alloc ties each dest +to its dying operand; the `plus (reg ob) (reg shift)` operand order does not matter because block_alloc tries every operand, +local-alloc.c:1123ff) with 8 refs, and the `*(u32*)p` chain into another with 6. With the hand-written form sched1 hoists +`lw otz` to insn #2 of the block (range 30 → `qty_compare` 3·8·4/30 = 3.2) while the *p chain spans 10 (2·6·4/10 = 4.8) → *p is +allocated first and takes `$v1`, otz gets `$a0` — the observed swap. The bitfield store expands VALUE-first +(`store_fixed_bit_field`), which reorders sched1's output so the two priorities flip. The last row of the window +(`addu v1,s1,v1` vs `addu v1,v1,s1`): c-typeck's `pointer_int_sum` always puts the POINTER first in `ptr + int`; an +INTEGER add `(otz << 2) + (u32)ob` puts the shifted index first and `expand_binop` does not swap two REG operands (optabs.c:399ff). + +## 2. Window 3 (idx 432–438, 6 rows) → CLOSED. 25 → 19 +Lever (§49 LUID dial): `t20 = *(u16*)(p+0x10); t22 = *(u16*)(p+0x1A); tp = D_801BA6B0; *(u16*)(p+0x20) = t20; *(u16*)(p+0x22) = t22;` +Mechanism: sched1's pending-memory-list FLUSH — `sched_analyze_1/2`: `if (pending_lists_length > 32) flush_pending_lists (insn)` — +fires at the 33rd memory op of the coordinate block (`sh 0x18(s0)`, insn 588 in the base dump; `.i.sched` shows every later +memory op with `REG_DEP_ANTI 588`). Among the five memory ops after the flush the target picks the `lbu D_801BA6B0` FIRST in the +backward pass (T-5 in the trace), i.e. the lbu must have the HIGHEST LUID of the three loads while still preceding both stores in +RTL order (a fixed-address load after a store through a pseudo base is a true dependence and sinks below the stores — measured: +`tp` after the two stores = 80 rows, 492 ins). Only HI temps can express "after both loads, before both stores". +The opus report's "9 placements inert" was correct for STATEMENT placements; the temp form is the 10th. + +## 3. Window 1 (idx 343–347, 5 rows) → CLOSED. 19 → 14 +Lever: `register u32 uu __asm__("$4"), mode ("$5"), ot16 ("$6")` on the block-local quartet. +Mechanism (`dumps_PIN3/PIN3.i.sched`, block 13, T-37): with pseudos the `ot16` load is `birthing_insn_p` (single set → 7f000001) +and is picked before the two unboosted `spr` loads, i.e. placed AFTER them; **a hard-register destination is NOT boosted in this +compiler** — all three loads sit at priority 2 in the pinned trace (and the pinned `mode` `and` at priority 1), so the LUID +tie-break gives the source order ot16, uu, cl. (`ot16 <<= 14` as a second set also un-boosts the load and fixes the window, but +the now-unboosted `sll` is starved to the block head: 24.) NB this contradicts sched.md's "hard-reg dests ARE boosted" note and +agrees with §501-D's `regclass.c:1791` reading only if `reg_n_sets` for hard regs is not what `birthing_insn_p` sees — +byte-measured here: **a `register … __asm__` pin on a single-set variable is a zero-byte boost-kill dial.** + +## 4. `register u32 shf __asm__("$3")` for `2 - mode`. 14 → 13 +Local-alloc: the `li 2 → subu` qty has 4–5 refs over 2 insns (pri 2.0–2.5) vs `la D_800A6610` 2 refs over 2 insns (0.5), so `shf` +takes `$v0` and `la` `$v1`; the target has `la` in `$v0`. The pin swaps them; the other 12 rows are unchanged. + +## 5. Residual — 13 rows, idx 363–380, ONE cause: the §194-A fence that the allocation still needs +The zero-byte `__asm__ volatile("")` after `p[7] |= …` forbids the target's interleave (`sll $a2,14` at 364, `andi $a3,0xFFFF` at +367 inside the tpage/code window, `li/subu 2-mode` at 370/372) and keeps `la`/`addu` after the `sllv`. READ FROM THE DUMPS +(`dumps_NOFENCE`, `dumps_P3NF`): the fence's REGISTER effect is this — `cl &= 0xFFFF` is a 2nd set of `cl` (unboosted, priority 3); +the w-chain's two `cl` reads (`andi cl,0x100` / `andi cl,0x200`) carry ANTI-dependences on it, so they cannot be picked in the +backward pass until it is; without the fence that insn is starved by boosted insns and equal-priority memory ops until T-34 +(the trace shows it losing every tie: `sb 7` at T-14 and `sh 0x16` at T-20 win on `potential_hazard`), the two reads are then +placed right after the loads, their qtys span the whole w chain (`Register 192 … across 8 insns`, `203 … 12 insns`) and take +`$a0/$a1` (or `$a3/$t0` with the pins) in local-alloc → the quartet rotates. With the fence the insn is the last region-2 pick +(T-14) and is forced before the fence, so the reads are released early and stay adjacent to their consumers. +Every way found to release the reads without the fence loses the `andi` or the allocation: +* a fresh single-set `clx = cl & 0xFFFF` is boosted, but `combine` then folds the `and` into any same-block consumer using + `nonzero_bits(cl) = 0xFFFF` (F2: `move`, 59) — the `andi` survives only when its consumers are in OTHER blocks, i.e. `clx` must be + set again in the arm (2 sets → unboosted, back to starvation: E 49 / 39 / 42) or the store duplicated into both arms; the + arm-duplicated form (G) does not cross-jump because the arm's `clr` cannot be tied to the GLOBAL `clx` (490 ins, 164); +* the andi needs priority 4 to win the T-14 gap on its own (all region-1 insns are one load deep = 3; the only pri-3 load is the + tpage reload, and a dependence on it puts the andi after `lhu 0x16` — the wrong side of the target's 367); +* a launder `__asm__("" : "=r"(clx) : "0"(cl))` is not a copy for `set_preference`/`block_alloc` (asm insns skip the tying code), + so cl/clx land in different registers and reload inserts a real `move`. +The target's sched2 (`-dR`) explains the rows once the allocation is right: `sll $a2` / `andi $a3` / `li 2` are stall FILLERS placed +by `rank_for_schedule`'s LUID tie among equal-priority ready insns (the `lbu 7 → or` gap takes `andi $a3`; `sh 0x16` is not a +candidate there because `lbu v1` redefines `$v1`), and `la $v0` sits at 376 only because of its `$v0` anti-dependence on +`subu a0,a0,v0` (375) — which is why the la/shf register swap above was worth a row. + +## 6. Measured inert / worse (from v3b=19 or v6=14; all exact length unless noted) +`ot16 <<= 14` at 4 positions (24, sll floats to the head); `w` as a two-set variable (46); `(flags&0x40)>>6` via a multi-set +variable (66) or `x = flags & 0x40; x >>= 6` (61); fresh `clx` 2-set (49; with pins 39/42); `shf = 2 - mode` before the fence +(24); no fence (51; with pins 43); `cl` pinned `$7` (151, 488 ins); ternary `p[0xD] = t ? clx-0x100 : clx` (483 ins); +`obase` pinned `$2` for the `la` (20; with shf pin 16); **the 3,360-variant sweep of region 2** (ob fused / in-place / temp × +shf inline / temp × cl in-place / clx × test temp × every statement order × fence on/off, with the pins): best 14, every +fence-free variant worse — statement order inside the fenced region 2 is fully inert (sched2 decides it). + +## 7. For the cookbook (generalisable, each byte-measured here) +* **The pending-memory-list FLUSH is a scheduling barrier you can compute:** count memory ops from the block head; the 33rd + becomes a fence for everything after it (`sched.c flush_pending_lists`). A load that must sit "after the last two loads but + before their stores" needs HI temps (window 3). +* **A `register __asm__` pin is a zero-byte birthing-boost kill** (hard-reg dest → priority stays 2); use it when a single-set + load must NOT sink to its consumer and a second set would move the consumer (window 1). +* **A redefinition of a variable the block has already read holds every earlier read hostage** (anti-deps + starvation of the + unboosted 2nd set): the reads float to the head and take the low caller-saved registers. A volatile fence between the reads and + the redefinition is the cheap fix; the honest fix (a boosted fresh pseudo) is blocked whenever `combine` can see the loaded + value's `nonzero_bits` from a same-block consumer. +* **P_TAG bitfield at -O2 is a local-alloc PRIORITY lever** (third witness after §500-C/§501-G), not only an operand-order one. diff --git a/.run/P32/t5x/resume_queue.txt b/.run/P32/t5x/resume_queue.txt index be5a5eaa1..f18d927ee 100644 --- a/.run/P32/t5x/resume_queue.txt +++ b/.run/P32/t5x/resume_queue.txt @@ -1,6 +1,5 @@ # T4b Fable resume queue (Drew 2026-09-05: "resume agents, but not all at once, just 3 at a time"). Delete a line when resumed. # fn agentId closeness -func_800CF408 aac6ce9fdfdcc093c 49 func_800CF6D0 a7f7e477e9878acdd 137 func_80011380 a60ffea4022b891fc 6 (-O0, §474 PROVED — last) -# RUNNING: func_80039308 a55fbfb4fb896bd71 · func_80185810 a547e70e9a7735c8a · func_8017DC80 a7c2c1e4865e5590d (DONE MATCH: 391D4 39DEC CD674 DF28 20DA4 1834A4 CF3E8 CD92C; NEAR: 32A74 1) +# RUNNING: func_80039308 a55fbfb4fb896bd71 · func_8017DC80 a7c2c1e4865e5590d · func_800CF408 aac6ce9fdfdcc093c (DONE MATCH: 391D4 39DEC CD674 DF28 20DA4 1834A4 CF3E8 CD92C; NEAR: 32A74 1, 185810 13) diff --git a/.run/P32/t5x/verdicts.jsonl b/.run/P32/t5x/verdicts.jsonl index b881fe71c..ae3a27839 100644 --- a/.run/P32/t5x/verdicts.jsonl +++ b/.run/P32/t5x/verdicts.jsonl @@ -7,3 +7,4 @@ {"fn": "func_80032A74", "binary": "main", "arm": "fable", "status": "NEAR", "closeness": 1, "compiles": true, "draft_path": ".run/P32/t5x/fable/func_80032A74.c", "note": "rtu DIFF 1 in src/800_b_2.c (idx 244 lh vs lhu; frame 0x78 exact); the lhu respelling is code-exact 422/422 with vars=56. Residual mechanism READ + byte-reproduced: the never-referenced 8-byte slot at 0x48 is a GHOST pseudo — a stranded middle temp of a 3-insn combine whose refs-zeroing is skipped when newi2pat!=0 (combine.c:2306-2313); zero occurrences -> regclass 'ST_REGS or none' -> reload1.c:658 alter_reg(i,-1) 8-byte slot in regno order = right after the three param slots (ghost1 reproducer vars=8, no code). The caller-save-area hypothesis (S83 HYPOTHESIS.md) is REFUTED: order_regs_for_reload (reload1.c:3606-3700) picks zero-use regs first so a pseudo homed in $t0 makes $t1 the spill reg (+30 rows); an area needs caller_save_needed (global.c:1085-1091) and a pseudo kept in a call-used reg with save/restore at every live call (caller-save.c:264, 349-470). The only ghost species from a memory value is the SIGN_EXTEND narrow-load split (combine.c:1887-1930) whose signature IS lh; a jump-target second promotion is folded by cse follow-jumps (n0a/b/c: vars 56, +86 rows); a fall-through one needs a register sign_extend MIPS lacks; the generic two-SETs split (combine.c:1963-2020) has no candidate (cse pre-folds constant offsets). Inert: expA/expB, n16 (2 ghosts), n0a/n0b/n0c. 402k tokens, 23 min", "session": "491895ad"} {"fn": "func_800CF3E8", "binary": "md_MAIN_003", "arm": "fable", "status": "MATCH", "closeness": 0, "compiles": true, "draft_path": ".run/P32/t5x/fable/func_800CF3E8.c", "note": "27 -> MATCH 469/469; three passes each owned one defect: (1) sched1 birthing boost on the tag load (sched.c:2507/2469) killed with a SECOND LIVE SET of the loaded variable `tag6 = *(u32*)p6; ... tag6 &= 0xFF000000;` (reg_n_sets=2, zero bytes); (2) the S83 +1 nop was the phantom `__asm__(\"\" :: \"r\"(ot))` USE filling the load-delay slot in sched2's model only — removed, plus the fence after the tpage store removed so `lui m24` reaches slot 378 via its $a2 anti-dep; (3) local-alloc qty_compare had mhi out-ranking m24 for $a2 — spelling prims 5..1's OT link as libgpu's P_TAG bitfield adds store_fixed_bit_field's redundant 0xFFFFFF re-mask (folded by combine, counted by flow: m24 13 -> 18 refs) so m24 wins $a2, mhi -> $t0, tag -> $t2; last 2 rows = p5 x0/y0 in natural source order. Ablations: non-compound tag 89; phantom re-added 70@470; fence re-added 8; manual masks on all six prims 74. Coordinator rtu MATCH 469/469; bank.sh byte-identical", "session": "491895ad"} {"fn": "func_800CD92C", "binary": "md_MAIN_009", "arm": "fable", "status": "MATCH", "closeness": 0, "compiles": true, "draft_path": ".run/P32/t5x/fable/func_800CD92C.c", "note": "MATCH 247/247, ZERO register pins. (1) m24 takes $t1 by REF COUNT: local-alloc qty_compare uses flow's reg_n_refs (pre-combine, combine.c:56); the libgpu P_TAG bitfield store's must_and re-mask (expmed.c:608) leaves a combine-foldable (and (and ot m24) m24) (combine.c:3140-3170) so m24 keeps 19 refs vs la's 13 -> spelled (ot & m24) & m24, unpinned. (2) tp8D/tp8F float to the top when the TAG LOAD is an in-place multi-set pseudo (bitfield RMW shape t=*p; t&=FF00; t|=v; *p=t): sched1 splits large consts (sched.c:4826) and update_n_sets kills their birthing boost, so they are pri-1 floaters eaten only by the lhu->sll latency stall of each RMW chain; a single-set tag load is boosted and glued to its and, a multi-set one fills those stalls -> local-alloc gives $16..$19 in the target order. Sub-levers: the mask temp v must be a FRESH expression (in-place v&=m24 steals $2); v (OT read) precedes t&=FF00; plain scalar D_800BAE22 (struct/array/cast spellings force one shared la); tag store non-/s *(u32*)p; TU spelling extern u8 *D_800A71D0. Coordinator rtu MATCH 247/247; bank.sh d270f695. 509k tokens, 39 min", "session": "491895ad"} +{"fn": "func_80185810", "binary": "ov_SC03_105", "arm": "fable", "status": "NEAR", "closeness": 13, "compiles": true, "draft_path": ".run/P32/t5x/fable/func_80185810.c", "note": "35 -> 13 at exact length 489/489 (rtu agrees). Closed 3 of 4 windows: (1) OT link = libgpu P_TAG bitfield flips the otz/(*p) local-alloc qty priorities + integer add (otz<<2)+(u32)ob for the addu operand order (35->25); (2) sched1 flush_pending_lists at the 33rd memory op (sh 0x18) makes the D_801BA6B0 lbu need the highest LUID of the three loads while preceding both stores -> HI temps t20/t22 with tp loaded between loads and stores (25->19); (3) pins uu $4/mode $5/ot16 $6: a hard-reg destination is NOT birthing-boosted, so the LUID tie-break gives the target load order (19->14); (4) shf pinned $3 swaps the la/shf local qtys (14->13). Residual ONE cause (idx 363-380): the zero-byte fence after p[7]|= is needed because `cl &= 0xFFFF` is an unboosted 2nd set whose anti-deps hold the w-chain's two cl reads hostage (without the fence they float to the block head and take $a0/$a1: 43/51), yet the fence forbids sched2's fillers (sll $a2, andi $a3, li/subu) from crossing into the tpage/code window. Fresh single-set clx is boosted but combine folds the andi via nonzero_bits(cl) into any same-block consumer (59); 2-set clx unboosted (49/39/42); arm-duplicated store does not cross-jump (490 ins); a launder asm is a real move for block_alloc. Inert: 3,360-variant region-2 sweep best 14; ot16<<=14 (24); w two-set (46); multi-set x (66/61); obase pin $2 (20); cl pin $7 (151). 471k tokens, 26 min", "session": "491895ad"} diff --git a/.run/backlog.jsonl b/.run/backlog.jsonl index 764b9e03a..d1ce97c75 100644 --- a/.run/backlog.jsonl +++ b/.run/backlog.jsonl @@ -22,3 +22,4 @@ {"ts": "2026-09-05 12:09:39", "addr": "0x80039dec", "name": "func_80039DEC", "reach": null, "klass": "WALL-CANDIDATE", "nins": 74, "status": "near", "closeness": 2, "where_stuck": "WALL candidate CONFIRMED (S83 sandbox-TU rtu DIFF 2: idx 0 'move t1,a2' vs 'addu a3,a2,zero'; idx 56 'sb t1' vs 'sb a3'): 74/74 exact length; the $a3<->$t0/$t1 swap of the two K&R raw-preserve parameter copies is fixed by ARGUMENT POSITION in gcc-2.7.2 narrow-parameter promotion (2nd param -> $a3, 3rd -> $t0) before the global allocator runs - every pin regresses to 60-75; 3 attempts + permuter_ils 9 -> 2 (S80). Banking would need the TU decl -> no-proto (byte-neutral commit) - only worth it at closeness 0", "best_draft": ".run/S79w/permuter/func_80039DEC.c", "binary": "main", "source": "P32-T4 S83 re-probe (R40): the drafts were CC1 FAIL in src/800_c.c only because the TU's prototype 'extern void func_80039DEC(void *, s16, u8)' (line 3496) rejects the K&R definition; in a sandbox TU copy with the no-proto spelling the permuter draft re-runs DIFF 2 - the recorded residual, confirmed in TU context (leaf 2; the Sonnet draft leaf 9)", "residual": null, "passes_tried": null} {"ts": "2026-09-05 12:09:40", "addr": "0x800391d4", "name": "func_800391D4", "reach": null, "klass": "WALL-CANDIDATE", "nins": 75, "status": "near", "closeness": 3, "where_stuck": "WALL candidate CONFIRMED (S83 sandbox-TU rtu DIFF 3: idx 9-11 'move t0,zero' before vs after arg1 sll/sra): 75/75, a 3-insn SCHEDULE-REORDER - move_movables splices hoisted invariants after any pre-existing preheader flow code (loop.c 2.7.2:1529, map loop.md L4), so off init cannot follow arg1 hoisted sign-extend from C; 4 prior attempts + S79 Sonnet 64->3 + permuter_ils null (S80). Banking would need the TU externs -> [][1] (whole-EXE sha decides) - only worth it at closeness 0", "best_draft": ".run/S79w/sonnet/func_800391D4.c", "binary": "main", "source": "P32-T4 S83 re-probe (R40): CC1 FAIL in src/800_c.c only because the draft's lever spelling 'extern s32 D_80073140[][1]' conflicts with the TU's three 'extern s32 D_80073140[]'; in a sandbox TU copy with the externs spelled [][1] (+ (s32*) casts at the 3 uses) the Sonnet draft re-runs DIFF 3 - the recorded residual; the [][1] spelling is LOAD-BEARING: the TU's [] spelling, a (s32 (*)[1]) cast and a byte-offset form all regress to 65 @ 76 ins", "residual": null, "passes_tried": null} {"ts": "2026-09-05 18:35:15", "addr": "0x80032a74", "name": "func_80032A74", "reach": null, "klass": "WALL-CANDIDATE", "nins": 422, "status": "near", "closeness": 1, "where_stuck": "WALL candidate STRENGTHENED (S83 Fable): the never-referenced 8-byte slot at 0x48 is a GHOST pseudo (a stranded middle temp of a 3-insn combine, refs-zeroing skipped when newi2pat!=0, combine.c:2306-2313; regclass ST_REGS-or-none; reload1.c:658 alter_reg 8-byte slot in regno order). Caller-save-area route REFUTED (order_regs_for_reload picks zero-use regs first; an area needs caller_save_needed + saves at every live call). The only ghost species from a memory value is the SIGN_EXTEND narrow-load split whose signature IS lh \u2014 and the target load is lhu; jump-target second promotions are folded by cse follow-jumps; a fall-through one needs a register sign_extend MIPS lacks; the generic two-SETs split has no candidate. Frame 0x78 needs one ghost the lhu code cannot mint. Inert: expA/expB, n16, n0a/b/c (+ the S79 ~200 probes)", "best_draft": ".run/P32/t5x/fable/func_80032A74.c", "binary": "main", "source": "P32-T4b S83 Fable agent (402k tokens, 23 min, ~40 probes) after the S83 hand pass; report .run/P32/t5x/reports/func_80032A74.md", "residual": null, "passes_tried": null} +{"ts": "2026-09-05 19:09:22", "addr": "0x80185810", "name": "func_80185810", "reach": null, "klass": "SCHED", "nins": 489, "status": "near", "closeness": 13, "where_stuck": "S83 Fable: 35 -> 13 at exact length; 3 of 4 windows closed (P_TAG bitfield OT link + integer add for the addu operand order; sched1 flush_pending_lists at the 33rd memory op explains the load order -> HI temps; hard-reg destinations are not birthing-boosted -> pins uu $4 / mode $5 / ot16 $6 give the LUID order; shf pin $3). Residual ONE cause idx 363-380: `cl &= 0xFFFF` is an unboosted 2nd set \u2014 the fence after p[7]|= is needed (else its two reads float to the block head, 43/51) yet it blocks sched2 fillers crossing into the tpage/code window. NEXT: a spelling in which cl is single-set (its high half cleared at birth: cl = *(u16*)... or the shift form) so no fence is needed, or the two cl reads consume a fresh single-set copy that combine cannot fold (nonzero_bits defeats a plain andi copy; try a subreg/HI-mode temp)", "best_draft": ".run/P32/t5x/fable/func_80185810.c", "binary": "ov_SC03_105", "source": "P32-T4b S83 Fable agent (471k tokens, 26 min, ~3,400 compiles); report .run/P32/t5x/reports/func_80185810.md", "residual": null, "passes_tried": null} diff --git a/docs/backlog.md b/docs/backlog.md index 91a089435..5a77390fe 100644 --- a/docs/backlog.md +++ b/docs/backlog.md @@ -2,14 +2,14 @@ > Generated by `tools/backlog.py render` from `.run/backlog.jsonl`. These are functions the Phase-21 automation got **close** on but did NOT byte-match. The whole-binary byte-gate is the sole arbiter (G3/P9): **byte-matches bank and are NOT listed here** — only genuine near-misses/blockers are. Ranked by hand-session priority: **reach** (×N propagation leverage) → **closeness** (match_one mismatch count, lower = closer) → **size**. Each row's `best_draft` is the closest C the machine reached — resume from there. -**Open near-misses:** 8 · by status {'near': 7, 'failed': 1} · by class {'WALL-CANDIDATE': 1, 'WALL-PROVED': 1, 'REGALLOC': 1, 'SCHED': 3, 'FRAME': 1, None: 1} +**Open near-misses:** 8 · by status {'near': 7, 'failed': 1} · by class {'WALL-CANDIDATE': 1, 'WALL-PROVED': 1, 'SCHED': 3, 'REGALLOC': 1, 'FRAME': 1, None: 1} | # | addr | reach | class | nins | status | closeness | where it stuck | best draft | |--:|------|------:|-------|-----:|--------|----------:|----------------|------------| | 1 | func_80032A74 | None | WALL-CANDIDATE | 422 | near | 1 | WALL candidate CONFIRMED in the real TU (S83): 422/422, sole residual idx 244 `lh v0,0x18(s1)` vs target `lhu` — extendhisi2 is a force_not_mem EXPAND (the orphan frame slot is minted only at an lh; §172 producer 3 caller-save area, reload1.c:1445), so lhu loses the 8 frame bytes; ~200 byte-probes + 100-variant retyping sweep (S79) + permuter_ils 8x150s null (S80). Citation current (§172, reload1.c:1445). Draft synced to the TU (typedefs stripped via cdecl.strip_provided_typedefs; D_80064D44/D_8006A970/func_8003F144/func_800316F8 spelled as the TU) | `.run/P32/t4/drafts/func_80032A74_tuclean.c` | | 2 | func_80011380 | None | WALL-PROVED | 192 | near | 6 | §474 PROVED C-level floor (boot -O0): fold-const.c:882 split_tree merges MULT(MULT(i,2),2); the two escapes each cost one instruction (stupid.c:497 adjacency / expand_decl use-brackets); §388 -O0 colouring oracle. Pinned S79 #8; re-probed S83 in the real TU: DIFF 6 (unchanged) | `.run/m3/opus/func_80011380.c` | -| 3 | func_80039308 | None | REGALLOC | 518 | near | 17 | sched2 + cross-block regalloc: preheader 49/50 swap, un-spellable addu $a2,$a0,$zero (every p=r form cse-propagated), a temp on $t0 vs $s7, and 11 insns of one alias fact (2nd D_80073140[j] load cannot schedule above the D_800C7D20 store from C; /s unlock costs the address allocation, net 20-24). 34->17 via s16 b4 widening copy + dead-local identity sweep (.run/P32/t3/restored/sweep_func_80039308.py) + $2 pin. permuter_ils --klass REGALLOC 2x150s: no gain | `.run/P32/t3/opus/func_80039308.c` | -| 4 | func_80185810 | None | SCHED | 489 | near | 35 | [permuter] 4 emission windows (see report .run/P32/t3/reports/func_80185810__opus__*.md); exact length, rtu-clean | `.run/P32/t3/opus/func_80185810.c` | +| 3 | func_80185810 | None | SCHED | 489 | near | 13 | S83 Fable: 35 -> 13 at exact length; 3 of 4 windows closed (P_TAG bitfield OT link + integer add for the addu operand order; sched1 flush_pending_lists at the 33rd memory op explains the load order -> HI temps; hard-reg destinations are not birthing-boosted -> pins uu $4 / mode $5 / ot16 $6 give the LUID order; shf pin $3). Residual ONE cause idx 363-380: `cl &= 0xFFFF` is an unboosted 2nd set — the fence after p[7]/= is needed (else its two reads float to the block head, 43/51) yet it blocks sched2 fillers crossing into the tpage/code window. NEXT: a spelling in which cl is single-set (its high half cleared at birth: cl = *(u16*)... or the shift form) so no fence is needed, or the two cl reads consume a fresh single-set copy that combine cannot fold (nonzero_bits defeats a plain andi copy; try a subreg/HI-mode temp) | `.run/P32/t5x/fable/func_80185810.c` | +| 4 | func_80039308 | None | REGALLOC | 518 | near | 17 | sched2 + cross-block regalloc: preheader 49/50 swap, un-spellable addu $a2,$a0,$zero (every p=r form cse-propagated), a temp on $t0 vs $s7, and 11 insns of one alias fact (2nd D_80073140[j] load cannot schedule above the D_800C7D20 store from C; /s unlock costs the address allocation, net 20-24). 34->17 via s16 b4 widening copy + dead-local identity sweep (.run/P32/t3/restored/sweep_func_80039308.py) + $2 pin. permuter_ils --klass REGALLOC 2x150s: no gain | `.run/P32/t3/opus/func_80039308.c` | | 5 | func_8017DC80 | None | FRAME | 346 | near | 46 | the historic -33 LENGTH wall CLOSED (GTE macros must be REAL macros — the TU house block; the splat Handwritten tag is wrong): 346/346, exact 0x70 frame + 9 callee-saved. Residual: reload-slot frame + the la $a0 slot; cse1 unifies OT index and n<4 across func_80010A08(8) (§500-D2 zero-byte asm retire) | `.run/P32/t3/opus/func_8017DC80.c` | | 6 | func_800CF408 | None | SCHED | 178 | near | 49 | [permuter] 3 hunks: two prologue sched2 slots, an mlo/mhi allocno tie, a 3-insn block-2 head hoist. Two LENGTH-bearing pins found (tp $17 shared by 0xE1000087/97 = the 6th callee-saved; ob $10 fixes the $t1/$t2/$t3 rotation, 56->49). §351 family (func_8001212C -O0 / func_8017DD04 -O2 exemplars) | `.run/P32/t3/opus/func_800CF408.c` | | 7 | func_800CF6D0 | None | SCHED | 249 | near | 137 | sched1 rank_for_schedule last-insn-CLASS tie (every store priority 2, equal refs; QImode stores grouped, loads floated, HImode after — 5 of 6 blocks) + $t1<->$t3 local-alloc swap of the two masks. 249/249 exact length only with tpage-before-len field order (19 swept). Inert at 137: pins on tpage constants/masks, asm re-ties, volatile/memory fences, /s-denial on any store subset, *0x4000 vs <<14, p++ vs p+0x18, / swap. decomp-permuter 122 was semantically wrong (R63) | `.run/P32/t3/opus/func_800CF6D0.c` |