From 05430eb26e575bb8f9f3edad259132df276abb63 Mon Sep 17 00:00:00 2001 From: Drew T <50529377+Druthulu@users.noreply.github.com> Date: Tue, 28 Jul 2026 11:28:59 -0600 Subject: [PATCH] =?UTF-8?q?feat(phase-29):=20T32=20=E2=80=94=20NEAR-6=20cr?= =?UTF-8?q?ack=20wave=20banks=204=20reach-138=20cores=20=C3=971;=20R22=201?= =?UTF-8?q?40/140?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ultracode fan-out (12 agents, 1.70M subagent tokens): 6 crack agents, one per NEAR target, each carrying its byte-measured residual + T31's disproved routes, then a distill agent per target that adversarially re-checks the claim. - BANKED ×1 (whole-binary gate, gate_stage --no-propagate per §55b law 1): func_80140958 (260 ins), func_80177B5C (147), func_80132F40 (72), func_8012E364 (67). Every agent MATCH claim was RE-MEASURED BY ME with match_one before it was believed (G3/P9: match_one is a candidate, a bank is the whole-binary gate), and verified against the SOURCE not gate_stage's accumulating verified-list (§55b trap 4). R22 clean-fleet: 140 passed, 0 failed of 140. - engine_core.h moved by exactly one byte-neutral arity fix (void -> no-proto) = fleet-shared, so R22 was mandatory (§61/§63), not the per-binary gate. - func_80140D68 MATCHes standalone but NOT whole-binary — the §30a integration class; its distill agent named the likely cause in advance (DEFINE_func_* extern must return u32*, not void). - func_80176734 217 -> 13 with the instruction count now EXACT (371/371). cse_expr.md §H's "no bank, 5 permuter-shaped clusters" is BYTE-REFUTED: 4 of 5 were steerable from C; the -1 length delta was a combine/LOG_LINK effect (flow.c links a SET only to the next use in the SAME bb), not frame pressure. Two coupled allocator/sched ties survive. - FOUND: docs/gcc-2.7.2-map/sched.md cites gcc-2.8.1 line numbers (birthing_insn_p 2498->2469, adjust_priority 2534->2507, potential_hazard 1345->1318, schedule_select 2646->2616) — surviving papermario numbers Phase 23's source-version correction never swept. One is LOAD-BEARING: §1.7 and §S12 claim the S2 boost needs SET(REG_pseudo,...) so pins must be removed; sched.c:2477 tests only GET_CODE(SET_DEST)==REG with NO pseudo check, discriminator is reg_n_sets==1 (2490). Verified by me against tools/reference/gcc-2.7.2, not taken from the agents. Map edits owed (next task). - MY DEFECT: all 6 agents shared one scratch dir (1,452 files); deliverables are uniquely named and verified intact, but short-named scratch could collide. Per-agent subdirs next wave. --- phase-ends/CURRENT_PHASE.md | 83 ++++++++ src/ov_SC01_077/ov_SC01_077.c | 181 +++++++++++++++- src/ov_SC01_077/ov_SC01_077_jr_8012ACE0.c | 246 +++++++++++++++++++++- src/ov_SC01_077/ov_SC01_077_jr_80135888.c | 2 +- src/ov_SC01_077/ov_SC01_077_jr_80135A4C.c | 2 +- src/ov_SC01_077/ov_SC01_077_jr_80135D20.c | 2 +- src/ov_SC01_077/ov_SC01_077_jr_801380E0.c | 2 +- src/ov_SC01_077/ov_SC01_077_jr_801734BC.c | 180 +++++++++++++++- src/ov_SC01_077/ov_SC01_077_jr_80182268.c | 2 +- src/shared/engine_core.h | 2 +- 10 files changed, 690 insertions(+), 12 deletions(-) diff --git a/phase-ends/CURRENT_PHASE.md b/phase-ends/CURRENT_PHASE.md index 54d69d98ed..d24c7dff38 100644 --- a/phase-ends/CURRENT_PHASE.md +++ b/phase-ends/CURRENT_PHASE.md @@ -6635,3 +6635,86 @@ C-level structural fix, not a scheduling tie-break. Untried. Six independent hard functions with precise per-function diagnoses is **breadth-shaped**, and I worked them serially in the main loop. That is the shape `breadth-isolated-agents-not-serial` names as the expensive one. Prompted Drew at the end of this task rather than continuing to grind. + +## ✅ T32 — the NEAR-6 Ultracode crack wave: 4 cores BANKED ×1, a 5th at 13-from-217, R22 140/140 + +Drew enabled `/effort ultracode` at the T31 hand-off (R26/R27 — prompted, waited for the toggle, +a system-reminder confirmed it). Fanned out **6 isolated agents, one per NEAR target**, each given +its byte-measured residual AND T31's disproved routes so nobody re-derived them; then a **distill +agent per target** that adversarially re-checks the crack claim and extracts the idiom. +12 agents, 1.70M subagent tokens, 545 tool calls, 51 min wall. + +### The result — 5 agent MATCH claims, all 5 independently re-measured BY ME, then gated +`match_one` MATCH is a candidate, never a bank (G3/P9). I re-ran the oracle myself on every +deliverable before believing any of it: + +| fn | before | agent claim | MY re-measure | whole-binary gate | templated ins | +|---|---|---|---|---|---| +| `func_80140958` | 10 | MATCH | **MATCH 260/260** | ✅ BANKED | 35,880 | +| `func_80177B5C` | 11 (7 post-ILS) | MATCH | **MATCH 147/147** | ✅ BANKED | 20,286 | +| `func_80132F40` | 6 | MATCH | **MATCH 72/72** | ✅ BANKED | 9,936 | +| `func_8012E364` | 7 | MATCH | **MATCH 67/67** | ✅ BANKED | 9,246 | +| `func_80140D68` | 9 | MATCH | **MATCH 65/65** | ⚠️ NEAR whole-binary | 8,970 | +| `func_80176734` | 217 | IMPROVED | **13** (371/371, count exact) | not banked | 51,198 | + +**`gate_stage --no-propagate` (§55b law 1): drafts 5 → banked 4, near 1, failed 0.** +**R22 clean-fleet `make clean && extract-all && check-all` → 140 passed, 0 failed of 140.** +Verified by the SOURCE, not the report (§55b trap 4 — the verified-list accumulates and has no +`--verified-out`; I unlinked it first and then grepped `INCLUDE_ASM` out of `src/`): all four +stubs are gone and replaced by real definitions; `func_80140D68` is correctly still stubbed. +`src/shared/engine_core.h` moved by exactly ONE byte-neutral arity fix +(`extern void func_8012E364(void)` → `()`, the no-proto class) — fleet-shared, hence R22 mandatory +per §61/§63, and green. + +### ⚠️ `func_80140D68` — MATCHes standalone, does NOT bank whole-binary +The §30a integration class. Its distill agent predicted the exact cause in advance: the draft's +`#ifndef BFM_ENGINE_TYPES_H` shim collapses once the TU includes `engine_types.h`, but the +`DEFINE_func_*` caller macro's extern return type must be `u32 *` and not `void` — a discarding +`void` extern DCEs the return and shrinks the frame. Not chased this task; it is the cheapest +open item in the set (8,970 ins for a declaration fix). + +### `func_80176734` — 217 → 13, and §H's verdict is byte-REFUTED +The 371-ins top prize (51,198 ins). Instruction count is now EXACT (371/371) and 4 of the 5 +clusters Phase-27's Fable5 pass called permuter-shaped were **steerable from C**. The −1 length +delta was never a frame-pressure lock: gcc emits `xor/sne/move` for `x = (a != b)` and combine +merges the last two only when it has a LOG_LINK, which `flow.c` builds solely from a SET to the +NEXT use in the SAME basic block — so an extra use between the sne and the copy (`__asm__("" :: +"r"(t))`) removes the link and both insns survive. 111 → 26. +**So `cse_expr.md` §H's "no bank (5 permuter-shaped clusters)" is superseded: the residual was map +INCOMPLETENESS, not a compiler wall.** What survives is two coupled ties (a local-vs-global +allocation tie in regions 2/3, and an entry-block sched2 LUID tie that shares one screw with it). + +### 🔧 THE MAP IS CITING THE WRONG COMPILER — verified by me against the pinned source +The distill agents flagged stale line numbers; I checked them myself rather than swapping one +unverified citation for another (`tools/reference/gcc-2.7.2/sched.c`): + +| symbol | `sched.md` says | ACTUAL 2.7.2 | +|---|---|---| +| `birthing_insn_p` | 2498 | **2469** | +| `adjust_priority` | 2534 | **2507** | +| `potential_hazard` | 1345 | **1318** | +| `schedule_select` | 2646 | **2616** | + +A consistent ~30-line offset = surviving **gcc-2.8.1 (papermario)** numbers. Phase 23 established +the reference was 2.8.1 and staged vanilla 2.7.2, but `sched.md`'s citations were never re-derived. +**And one is load-bearing, not cosmetic:** §1 item 7 and §S12 both claim `birthing_insn_p` requires +`SET(REG_pseudo, …)`, so a `register __asm__` pin on the dest kills the S2 boost ("Unpin first"). +`sched.c:2477` tests only `GET_CODE (SET_DEST (pat)) == REG` — **there is no +`>= FIRST_PSEUDO_REGISTER` check anywhere in the function**; the real discriminator is +`reg_n_sets[i] == 1` (2490). Hard-reg dests ARE boosted. We have been telling agents to drop pins +for no reason. + +### ⚠️ A wave-harness defect I introduced +I pointed all 6 agents at ONE shared scratch dir (`.run/near6/wave23/`, now 1,452 files). The +per-function deliverables are uniquely named and verified intact, but short-named scratch +(`c5a.c`, `n1.c`, …) and helper scripts could collide between siblings in either direction. +**Next wave: one subdirectory per agent.** Caught by an agent, not by me. + +## ▶ NEXT +1. Regen the family map (`make sig-overlays` + `family_hseq.py`) — the 4 freshly-cracked exemplars + are still `draft-ov077` in the stale map — then `family_sweep --hseq --only ` per core. + **NOT `--reconcile-raw`** (mishandles per-overlay data externs: 0/137 last wave). ~75,348 ins. +2. `func_80140D68`'s `DEFINE_func_*` return-type fix (8,970 ins, a declaration). +3. Land the sched.md corrections + the §H supersession. +4. `func_80176734` at 13: permuter on the delivered draft (it is in the permuter bucket), then the + `reg_renumber`-swap gdb oracle (regalloc.md §H) before ANY further C-tier spend. diff --git a/src/ov_SC01_077/ov_SC01_077.c b/src/ov_SC01_077/ov_SC01_077.c index d1db8de917..a8527c5b98 100644 --- a/src/ov_SC01_077/ov_SC01_077.c +++ b/src/ov_SC01_077/ov_SC01_077.c @@ -1781,7 +1781,7 @@ s32 func_8013F350(void) { extern void func_80140E6C(void); extern void func_80140F00(void); -extern s32 *func_80140958(s32 *, s32, s32); +extern s32 *func_80140958(); extern int func_80141100(int); extern s16 func_8014168C(s16); extern s32 func_8013FFD8(s16, s32, s32 *); @@ -2299,7 +2299,184 @@ void func_801407F4(void) *(s16 *)(puVar1 + 0x1c) = *puVar3++; } -INCLUDE_ASM("asm/ov_SC01_077/nonmatchings/ov_SC01_077", func_80140958); + + + + + + + + + + + + + + + + + + +extern s16 func_8014168C(s16); +extern short D_800B9A02; +extern u16 D_80115110; +extern u8 D_80115140[]; +extern s16 D_8011514E; +extern u8 D_80115158[]; +extern Hw4 D_8011516A[]; +extern Prim4 D_8018798C[]; +extern Prim4 *D_80187A80[]; +extern u8 D_80115143; +extern Prim4 D_8018793C[]; +extern s16 D_80187E9C[]; +extern s16 D_80187EAC[]; +extern u16 D_801879BC; +extern u16 D_801879BE; +extern s32 *func_80140D68(s32 *, Prim4 *, s32, s32, s32); +/* func_80140958 -- MATCH (260 ins), verified by + * python3 tools/match_one.py func_80140958 --c .run/near6/wave23/func_80140958.c \ + * --asm-subdir asm/ov_SC01_077/nonmatchings/ov_SC01_077 + * + * Three levers took the seed from 6 -> 0 (see the numbered notes in the inner block): + * [L1] `m == i` (not `m == 3`) -- keeps b[0] a RUNTIME value; `m == 3` let gcc + * const-fold it and materialise `li a3,3` / `li t3,12` instead of + * `addu a3,v1,zero` / `sll t3,v1,2`. + * [L2] one dead `__asm__ volatile("" :: "r"(j))` -- +1 weighted REG_N_REFS on j so + * global.c:594 allocno_compare ranks j above k (j -> $a2, k -> $a3). + * [L3] `t3v = m * 4` as an EXPLICIT preheader statement (instead of letting loop.c + * hoist D_8011516A[m]'s index) -- gives it a LUID *below* the three constant + * assignments, which is what puts `move a3,v1 / sll t3,v1,2` ahead of them in + * sched1's backward LUID tie-break. + */ +s32 *func_80140958(ot, i, n) +s32 *ot; +s16 i; +s16 n; +{ + typedef struct + { + u32 *ot; + u32 pad[4]; + } Env_80140958; + extern Env_80140958 D_800AE7BC[]; + Prim4 *p; + s16 t; + s32 c3; +register u16 *a __asm__("$20"); +register u16 *b __asm__("$21"); +register u16 *c __asm__("$23"); +register u16 *e __asm__("$22"); +register u16 *pb __asm__("$10"); +register u32 m24 __asm__("$8"); +register u32 mhi __asm__("$9"); +register s32 eight __asm__("$2"); + if (i < n) + { + c3 = 3; + a = &D_80115110; + b = a + 5; + c = a + 3; + e = &D_801879BE; + do + { + if (((a[0] == 0) && (a[1] != c3)) && (a[1] < 6)) + { + if (i == a[5]) + { + ot = func_80140D68(ot, &D_8018793C[i], i, D_80187E9C[a[3] & 7], 0); + } + } + else + { + p = D_80187A80[i]; + if (((p != 0) && (i == b[0])) && (i != 6)) + { + if ((i == 2) && ((*((s16 *) (b + 7))) != 0)) + { + p = D_8018798C; + } + if (i != c3) + { + t = ((s32 (*)(s16)) func_8014168C)(i) * 2; + } + else + { + t = ((*((u8 *) (&D_8011514E))) - D_80115143) * 2; + } + ot = func_80140D68(ot, p, i, D_80187EAC[c[0] & 7], t); + if (((i == 2) && ((*((s16 *) (c + 9))) == 1)) && ((*((s16 *) (c + 12))) != 0)) + { + ot = func_80140D68(ot, p, 2, 8, ((*((s16 *) (c + 12))) & 0xF) * 2); + } + } + } + if (i == c3) + { + s32 m = b[0]; + /* [L1] compare against `i`, NOT against the literal 3. */ + if ((m == i) && ((b[-2] & 8) != 0)) + { + s16 j; + s32 k; + s32 t3v; + u8 *q = ((u8 *) ot) + 0x14; + s16 y; + j = 0; + k = m; + t3v = m * 4; /* [L3] explicit, must sit before the 3 constants */ + pb = (u16 *) (&D_800B9A02); + m24 = 0xFFFFFF; + mhi = 0xFF000000; + for (; j < 2; j++) + { + if (j == 0) + { + if (D_80115140[k] == 0) + { + continue; + } + q[-7] = 0x30; + y = (*e) - 4; + } + else + { + s32 k2 = k * 2; + if ((((s8 *) D_80115158)[k2] - ((s8 *) D_80115140)[k]) < 2) + { + continue; + } + q[-7] = 0x38; + y = (*e) + 3; + } + *((s16 *) (q - 10)) = y; +__asm__("" ::: "memory"); + *((u32 *) ot) = 0x4000000; + q[-8] = 0x78; + *((u32 *) (q - 0x10)) = 0x64808080; + *((s16 *) (q - 6)) = 0x4056; + *((s16 *) (q - 0xC)) = (D_801879BC + ((u16) *((u16 *) (((u8 *) D_8011516A) + t3v)))) + 0x4A; + eight = 8; + *((s16 *) (q - 2)) = eight; + *((s16 *) (q - 4)) = eight; + *((u32 *) ot) = ((*((u32 *) ot)) & mhi) | (D_800AE7BC[*pb].ot[2] & m24); + { +register u32 *op __asm__("$4"); + op = D_800AE7BC[*pb].ot; + op[2] = (op[2] & mhi) | (((u32) ot) & m24); + } + q += 0x14; + ot += 5; + __asm__ volatile("" :: "r"(j)); /* [L2] zero code, +1 ref on j */ + } + + } + } + i = i + 1; + } + while (i < n); + } + return ot; +} INCLUDE_ASM("asm/ov_SC01_077/nonmatchings/ov_SC01_077", func_80140D68); diff --git a/src/ov_SC01_077/ov_SC01_077_jr_8012ACE0.c b/src/ov_SC01_077/ov_SC01_077_jr_8012ACE0.c index 198893351d..5ee9b1f916 100644 --- a/src/ov_SC01_077/ov_SC01_077_jr_8012ACE0.c +++ b/src/ov_SC01_077/ov_SC01_077_jr_8012ACE0.c @@ -895,7 +895,123 @@ DEFINE_func_8012E28C() /* dedup: shared engine-core @0x8012E28C (src/shared) */ DEFINE_func_8012E32C() /* dedup: shared engine-core @0x8012E32C (src/shared) */ -INCLUDE_ASM("asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_8012ACE0", func_8012E364); +/* func_8012E364 (ov_SC01_077_jr_8012ACE0) — MATCH, 67/67 ins, 0 mismatched (reloc-masked). + * + * Verified: + * python3 tools/match_one.py func_8012E364 --c .run/near6/wave23/func_8012E364.c \ + * --asm-subdir asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_8012ACE0 + * -> MATCH (67 ins) func_8012E364 + * + * Symbols: sig_hints listed no callees and no data decls for this fn, so the three externs below + * are derived from the asm's %hi/%lo pairs: + * D_80126CE0 -> `lh` (0x8012E370) => s16 + * D_801D9498 -> `lw`/`sw` => s32 + * D_801D949C -> `lw`/`sw` => s32 + * D_80126CE0 is already declared `extern s16 D_80126CE0;` inside ov_SC03_099_jr_8016AB6C.c etc., + * so the s16 typing is consistent with the rest of the tree. + * + * --------------------------------------------------------------------------------------------- + * HOW THE LAST 7 SLOTS CAME OFF (wave22 plateaued here at closeness 7; that header's + * "genuine regalloc hard tail / not C-expressible" verdict was WRONG on both clusters). + * + * CLUSTER B - idx 59-62 (`nop`/`negu $v1,$v1` vs `addu $v0,$v1,$zero`/`negu $v0,$v0`): 7 -> 3. + * NOT a delay-slot (dbr) residual at all. `mips.md:1526 abssi2` is a 3-instruction `multi` + * template that emits its OWN branch AND fills its OWN slot: + * REGNO(op0) == REGNO(op1): "bgez %1,1f%#\n\tsubu %0,%z2,%0\n1:" (slot -> maspsx nop) + * REGNO(op0) != REGNO(op1): "%(bgez %1,1f\n\tmove %0,%1\n\tsubu %0,%z2,%0\n1:%)" + * The target IS the second form (`.set noreorder` + `move` in the slot). So the whole residual + * was one question: does the abs DEST get a different hard reg from its SOURCE? + * Two edits, both required: + * (a) spell the abs as `__builtin_abs(v)` so a real `(abs:SI ...)` insn exists. A hand-rolled + * `d = v; if (d<0) d = -d;` is equivalent (cse folds it to the same abs insn - the wave22 + * draft was ALREADY going through abssi2, it just hit the dest==src arm), but a + * `(v<0) ? -v : v` ternary does NOT fold here: 68 ins, 19 off. + * (b) PIN the abs result to $2. Unpinned, local-alloc's combine_regs (K8) ties the abs dest + * into its dying input `v` -> REGNO(op0)==REGNO(op1) -> the nop form. Measured: the same + * file with `s32 d;` instead of the pin is 5 off; with the pin, 3 off. + * (Rejected on measurement: `d = v; asm("" :: "r"(v))` dead-read to break the tie = 68 ins/13; + * RC-12 `$0`-add opaque copy = 9; pinning d to $3 = 5.) + * + * CLUSTER A - idx 43-45 (the D_801D949C load vs the 0x20($a2) load, swapped): 3 -> MATCH. + * Pure sched1 rank, and it IS steerable (sched.md S2, the birthing boost). The `.i.sched` dump + * of the wave22 draft says it outright: + * ;; insn[ 104]: priority = 1 (prev = D_801D949C) + * ;; insn[ 107]: priority = 1 (a = *(arg0+0x20)) + * ;; ready list at T-17: 107 (1) 104 (7f000001), now 104 107 + * Equal base priority, but 104 carries the `adjust_priority`/`birthing_insn_p` boost + * (sched.c:2506/2469 - `reg_n_sets[dest] == 1`) and 107 does NOT, because wave22 reused ONE + * variable `a` for four roles (hint value, division result, both tail entity loads) -> 4 sets. + * Backward scheduling means picked-first = placed-LAST, so the boosted 104 got pushed BELOW 107. + * Fix: give each tail entity load its own single-set local (`e1`, `e2`). Now both loads are + * boosted, the rank falls through to `rank_for_schedule`'s LUID tie-break (sched.c:2427, + * "highest LUID first" = ascending source order forward) and the two loads come out in + * statement order = the target order. + * NB the wave22 header's claim "splitting `a` regresses 2 slots" only applies to splitting the + * FIRST two roles (hint value / division result) - those must stay one multi-set cross-block + * variable. Splitting only the TAIL roles is what pays. + * + * LOAD-BEARING constructs (do not "simplify"): + * 1. `spd` - a local holding 0x1000 SET BEFORE the if/else chain, so the constant lives in a + * pseudo across the branch: `addiu $a3,$zero,0x1000` in the `blez` slot at 0x8012E3DC and the + * `sw $a3` / `addu $v1,$v1,$a3` forms at L8012E410. + * 2. `a` reused for the D_80126CE0 value AND the division result (multi-set, cross-block) - that + * is what puts the quotient in $a0 (`subu $a0,$v0,$v1`) instead of coalescing into $v0. + * 3. `flags` pinned to $2 (dropping it = 13 off) and `prev` pinned to $5 (dropping it = 6 off). + * 4. `d` pinned to $2 - see CLUSTER B(b). + * 5. `e1`/`e2` must be SEPARATE single-set locals - see CLUSTER A. + * The `arg0` $6 pin is NOT load-bearing any more (verified: still MATCH without it); it is kept + * because it costs nothing and documents the target's `addu $a2,$a0,$zero`. + */ +extern s16 D_80126CE0; +extern s32 D_801D9498; +extern s32 D_801D949C; + +void func_8012E364(s32 arg0_) +{ + register s32 arg0 __asm__("$6"); + register s32 prev __asm__("$5"); + register u16 flags __asm__("$2"); + s32 a; + s32 diff; + s32 v; + register s32 d __asm__("$2"); + s32 spd; + s32 e1; + s32 e2; + + arg0 = arg0_; + *(s16 *)(arg0 + 0x5C) = 0; + a = D_80126CE0; + if (a == 0) { + D_801D9498 = 0x1000; + D_801D949C = 0x1000; + } + a = ((0x90 - a) << 12) / 0x90; + *(s32 *)(arg0 + 0x1C) += 1; + spd = 0x1000; + + diff = D_801D9498 - a; + if (diff > 0) { + D_801D9498 -= diff >> 2; + } else if (diff < 0) { + D_801D9498 += (-diff) / 4; + } + + prev = D_801D949C; + e1 = *(s32 *)(arg0 + 0x20); + v = D_801D9498 - prev + spd; + D_801D949C = spd; + flags = *(u16 *)(e1 + 0x2C); + D_801D9498 = v; + *(u16 *)(e1 + 0x2C) = flags | 0x10; + + e2 = *(s32 *)(arg0 + 0x20); + d = __builtin_abs(v); + *(s16 *)(e2 + 0x1C) = d; + *(s16 *)(e2 + 0x18) = d; + *(s16 *)(*(s32 *)(arg0 + 0x20) + 0x1A) = 0x1000; +} + DEFINE_func_8012E470() /* dedup: shared engine-core @0x8012E470 (src/shared) */ @@ -2174,7 +2290,133 @@ DEFINE_func_80132EC4() /* dedup: shared engine-core @0x80132EC4 (src/shared) */ DEFINE_func_80132EF4() /* dedup: shared engine-core @0x80132EF4 (src/shared) */ -INCLUDE_ASM("asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_8012ACE0", func_80132F40); +/* func_80132F40 — ov_SC01_077 (jr_8012ACE0 region), 72 ins, -O2. *** MATCH *** + * + * Verified: + * python3 tools/match_one.py func_80132F40 --c .run/near6/wave23/func_80132F40.c \ + * --asm-subdir asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_8012ACE0 + * -> MATCH (72 ins) + * + * --------------------------------------------------------------------------- + * WHAT THE WAVE-22 SEED GOT WRONG (the 6-mismatch plateau, and why s32 "cost 3") + * --------------------------------------------------------------------------- + * The seed's note blamed the `addiu $s3,$sp,0x10` hoist on a *whole-function CSE + * fork keyed on s16-vs-s32 w/h*, and concluded the s32 world was unreachable. + * Both halves of that are wrong, and the real mechanism is a reusable idiom. + * + * The hoisted register is the block-move source-address pseudo (reg 83 = + * `(plus fp 16)`, created by expand for `v[1] = v[0]`). Whether it survives is + * decided by ONE thing: does CSE's *extended basic block* still contain it when + * CSE reaches the two `&v[0]` call arguments? + * + * cse.c:cse_end_of_basic_block scans `while (p && GET_CODE (p) != CODE_LABEL)`. + * It can walk PAST a conditional jump only via + * - follow_jumps : target label preceded by a BARRIER, LABEL_NUSES == 1, or + * - skip_blocks : "branch around a block", no labels inside (-O2 sets both) + * and it BREAKS EARLY on + * `if (! after_loop && NOTE_LINE_NUMBER (p) == NOTE_INSN_LOOP_END) break;` + * + * With s16 w the min block stayed a *diamond* (jump.c's `if(..) x=a; else x=b;` + * -> `x=b; if(..) x=a;` at jump.c:699 is blocked when the moved insn carries a + * REG_EQUAL note — the sign_extend note from the sll/sra pair). The diamond's + * BARRIER + join label ended CSE's block before the calls, so `&v[0]` was + * recomputed at each site. With s32 w the diamond was flattened in jump1, and + * skip_blocks then walked CSE straight through to both call sites -> reg 83 was + * substituted, went live across a call, and took a 5th callee-saved register. + * So the fork was never about the *type* — it was about the CFG shape at CSE. + * + * --------------------------------------------------------------------------- + * THE NEW LEVER (cookbook candidate: "zero-instruction CSE path cut") + * --------------------------------------------------------------------------- + * `do { } while (0);` emits NOTE_INSN_LOOP_BEG/CONT/END and ZERO instructions. + * A NOTE_INSN_LOOP_END is exactly what cse1 (after_loop == 0) breaks its + * extended basic block on. Dropping one between the struct copy and the call + * sites cuts the path, kills the hoist, and costs nothing: 73 ins -> 70 ins. + * (Measured: without it this same file is 75 ins / 58 mismatched.) + * It is *not* an ordering/pressure hack — 1..8 empty `__asm__ __volatile__("")` + * barriers, which lengthen live ranges but emit no LOOP notes, changed the hoist + * by exactly nothing, which refutes the global.c allocno-priority explanation. + * + * --------------------------------------------------------------------------- + * THE MIN BLOCK (idx 37..44) + * --------------------------------------------------------------------------- + * lh $v1,8($v0) w (both loads are SImode sign_extend -> `lh`) + * lh $v0,0xA($v0) h (reuses the dying pointer's $v0) + * nop (load-delay; nothing schedulable) + * addu $a0,$v0,$zero hh = h (a REAL source-level carrier) + * slt $v0,$v0,$v1 c = h 75 ins / 58 drop `p` pin -> 71 ins / 40 + * drop `h` pin -> 71 ins / 33 drop `c` pin -> 70 ins / 36 + * drop `hh` pin -> 71 ins / 32 drop `q` pin -> 72 ins / 12 + * adding a `w`->$3 pin or an `m`->$18 pin -> still MATCH (so both are omitted) + * + * Canonical decls (wave22_targets.json sig_hints) verbatim; D_80126BE0 is + * declared exactly as the 20+ sibling TUs already declare it. + */ +extern void func_8012F038(); +extern void func_8012F14C(); +extern s32 func_80135888(s32, s32, s32, s32); +extern u16 D_80126B5E; +extern u16 D_80126B62; +extern u16 D_80126B66; +extern u8 D_80126BE0[]; + +void func_80132F40(s32 arg0) +{ + typedef struct { u16 vx, vy, vz, pad; } Svec_80132F40; + + Svec_80132F40 v[4]; + register Svec_80132F40 *q __asm__("$17"); + register s16 *p __asm__("$2"); + register s32 h __asm__("$2"); + register s32 c __asm__("$2"); + register s32 hh __asm__("$4"); + s32 w; + s32 m; + + q = &v[1]; + v[0].vx = D_80126B5E; + v[0].vy = D_80126B62; + v[0].vz = D_80126B66; + v[1] = v[0]; + + if (func_80135888(*(s32 *)(arg0 + 0x20), *(s32 *)(arg0 + 0x58), + (s32)D_80126BE0, (s32)q) != 0) { + /* zero-instruction NOTE_INSN_LOOP_END: cuts cse1's extended basic block + * so the block-move address pseudo cannot reach the two &v[0] args. */ + do { } while (0); + + p = (s16 *)((*(s32 *)(arg0 + 0x58) & 0x0FFFFFFF) | 0x80000000); + w = p[4]; + h = p[5]; + hh = h; + c = (h < w); + if (c) { m = hh; } else { m = w; } + + func_8012F038(*(s32 *)(arg0 + 0x20) + 0x34, &v[0], q); + v[1].vy = m; + func_8012F14C(*(s32 *)(arg0 + 0x20) + 0x34, q, &v[0]); + D_80126B5E = v[0].vx; + D_80126B62 = v[0].vy; + D_80126B66 = v[0].vz; + } +} + DEFINE_func_80133060() /* dedup: shared engine-core @0x80133060 (src/shared) */ diff --git a/src/ov_SC01_077/ov_SC01_077_jr_80135888.c b/src/ov_SC01_077/ov_SC01_077_jr_80135888.c index de6c2a263b..8b911f71d6 100644 --- a/src/ov_SC01_077/ov_SC01_077_jr_80135888.c +++ b/src/ov_SC01_077/ov_SC01_077_jr_80135888.c @@ -545,7 +545,7 @@ extern void func_801301E8(u8 *a0); extern void func_80130278(s32 arg0); extern void func_80130314(s32 a0); extern void func_80130360(s32 a0); -extern void func_8012E364(void); +extern void func_8012E364(); extern void func_801303A0(s32 a0); extern void func_801303EC(void *a0); extern void func_80143CD4(s32 a0); diff --git a/src/ov_SC01_077/ov_SC01_077_jr_80135A4C.c b/src/ov_SC01_077/ov_SC01_077_jr_80135A4C.c index 2ae554fc63..06377fc810 100644 --- a/src/ov_SC01_077/ov_SC01_077_jr_80135A4C.c +++ b/src/ov_SC01_077/ov_SC01_077_jr_80135A4C.c @@ -545,7 +545,7 @@ extern void func_801301E8(u8 *a0); extern void func_80130278(s32 arg0); extern void func_80130314(s32 a0); extern void func_80130360(s32 a0); -extern void func_8012E364(void); +extern void func_8012E364(); extern void func_801303A0(s32 a0); extern void func_801303EC(void *a0); extern void func_80143CD4(s32 a0); diff --git a/src/ov_SC01_077/ov_SC01_077_jr_80135D20.c b/src/ov_SC01_077/ov_SC01_077_jr_80135D20.c index 79432eac1e..ad054eabeb 100644 --- a/src/ov_SC01_077/ov_SC01_077_jr_80135D20.c +++ b/src/ov_SC01_077/ov_SC01_077_jr_80135D20.c @@ -545,7 +545,7 @@ extern void func_801301E8(u8 *a0); extern void func_80130278(s32 arg0); extern void func_80130314(s32 a0); extern void func_80130360(s32 a0); -extern void func_8012E364(void); +extern void func_8012E364(); extern void func_801303A0(s32 a0); extern void func_801303EC(void *a0); extern void func_80143CD4(s32 a0); diff --git a/src/ov_SC01_077/ov_SC01_077_jr_801380E0.c b/src/ov_SC01_077/ov_SC01_077_jr_801380E0.c index f635dedaaf..acf213d891 100644 --- a/src/ov_SC01_077/ov_SC01_077_jr_801380E0.c +++ b/src/ov_SC01_077/ov_SC01_077_jr_801380E0.c @@ -545,7 +545,7 @@ extern void func_801301E8(u8 *a0); extern void func_80130278(s32 arg0); extern void func_80130314(s32 a0); extern void func_80130360(s32 a0); -extern void func_8012E364(void); +extern void func_8012E364(); extern void func_801303A0(s32 a0); extern void func_801303EC(void *a0); extern void func_80143CD4(s32 a0); diff --git a/src/ov_SC01_077/ov_SC01_077_jr_801734BC.c b/src/ov_SC01_077/ov_SC01_077_jr_801734BC.c index a2c210c294..03209a4e9a 100644 --- a/src/ov_SC01_077/ov_SC01_077_jr_801734BC.c +++ b/src/ov_SC01_077/ov_SC01_077_jr_801734BC.c @@ -3065,7 +3065,7 @@ void func_80175AB8(param_1) extern u32 *func_801770E0(void *param_1, u32 param_2, s16 param_3_); extern u32 func_801783D0(s32 a0, s32 a1); extern u32 *func_80177EA4(u32 *param_1, s32 param_2, u32 param_3, s32 param_4); - extern u32 *func_80177B5C(u32 *a0, s32 a1, s32 a2, s32 a3, s32 a4); + extern u32 *func_80177B5C(); extern void func_80177940(u32 *p, u32 a_, u32 b_, u32 c_); extern u32 *func_80178298(u32 *param_1, u8 *param_2, short param_3, short param_4); extern s32 func_80024054(u8*, u8*); @@ -3644,7 +3644,183 @@ DEFINE_func_80177940() /* dedup: shared engine-core @0x80177940 (src/shared) */ DEFINE_func_80177AD4() /* dedup: shared engine-core @0x80177AD4 (src/shared) */ -INCLUDE_ASM("asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_801734BC", func_80177B5C); +/* func_80177B5C - MATCH (147/147 ins), wave23. + * + * Seed: .run/near6/func_80177B5C_ils.c (permuter-improved, closeness 7). + * Two residual clusters remained; both cracked, each by a sourced gcc-2.7.2 mechanism. + * + * CLUSTER C (idx 86-89) - the "cl | ((n*8+8) | 0x4000)" re-association. + * NOT a cse fold. It happens in the FRONT END: fold-const.c fold(), the `associate:` + * arm at :3685. For `A | (X | C)`, split_tree(arg1) (:3759, decomposer at :882) splits + * arg1 into var=X / con=C because TREE_CONSTANT(op1) holds, then rebuilds it as + * `(A | C) | X` at :3785. Verified in the FIRST RTL dump (t.i.rtl insn 215 already + * reads `(ior v1 16384)`), so no RTL-level lever (the cse if/else diamond, tie or + * volatile barriers, operand swap, shift-vs-multiply) can ever reach it. + * ANTIDOTE: hoist the inner IOR into its own statement. The outer arg1 is then a + * VAR_DECL, split_tree returns 0, and the associate arm is skipped. (This function + * already proved the shape at `tt = uv | 0x1000; p[3] = cl | tt;`, which matched.) + * + * That exposed a REGALLOC residual: the chain landed in $a3 (`sll a3,a3,3`) instead + * of $v0. Cause: `n` was pinned to a HARD reg, and local-alloc.c combine_regs():1798 + * unconditionally records a dying hard-reg SOURCE in qty_phys_sugg[] for the dest + * pseudo, so the shift dest inherited $a3. A *pseudo* source cannot do that - :1763 + * bails when reg_qty[ureg] < 0 (i.e. not block-local), and the target's `n` crosses + * the join, so it is exactly such a pseudo. + * ANTIDOTE: move the pin off `n` (which dies into the shift) onto `nn` (the tested + * value). `nn` pinned to $2 keeps the two distinct so the `addu $a3,$v0,$zero` copy + * survives (combine_regs :1841 refuses to tie when the DEST is non-block-local), + * while `n` stays a pseudo and the shift chain gets an ordinary local quantity -> + * $v0 - which also restores the target schedule, because the $v0 anti-dependence on + * `sw $v0,-0x4($a1)` is what stops sched2 hoisting the chain above the store. + * [7 -> 8 -> 3] + * + * CLUSTER B (idx 24-27) - `lui $t0,0x300` two slots early. + * Pure sched1 LUID tie-break, read straight off the -da trace (t.i.sched, T-36): + * ready = { 58 (7f000001), 72 (7f000001), ... } - insn 58 (lui $t0) IS birthing- + * boosted (sched.c birthing_insn_p:2469 works on hard regs too; reg_n_sets[$t0]==1), + * so 58/71/72 all tie at max_priority and rank_for_schedule:2427 falls through to + * DESCENDING LUID. Source order put `ca = 0x3000000;` before the mask, so + * LUID(58) < LUID(72) and the lui was placed first. + * (The companion mask 0xFFFFFF correctly stays at idx 6-7 because lui+ori is TWO + * sets of $t1 -> reg_n_sets==2 -> no boost -> it sinks to the block head. Same + * mechanism, opposite sign - the model predicts both.) + * ANTIDOTE: split the mask into its own statement (`gg`) and materialise the + * constant BETWEEN it and the OR, so expand emits addiu, and, lui, or in that order + * and LUID(58) > LUID(72). Sweeping the plain statement position of `ca = ...` was + * inert (all 11 slots scored 3) - only interposing the temp moves the LUID past the + * AND. [3 -> 0] + * + * Dead ends measured, not guessed: pin nv to $2 = 27; reuse tt = 24; reuse uv = 29; + * two-step |= = 15; unpin n = 84 (146 ins, the copy coalesces away); unpin ca = 139; + * ca as a bare literal = 139 (145 ins - cse merges it with the loop copy, so the hard + * pin is what keeps the pre-loop and in-loop constants separate); both literal = 143; + * `(ca = 0x3000000)` as an assignment-EXPRESSION = 114 (148 ins, extra move); reusing + * the existing `g` for the mask temp instead of a fresh one = 8 (g has a 2nd set later). + */ +extern u8 D_8018A300[]; +u32 *func_80177B5C(p, bits, tbli, x, y) +u32 *p; +u32 bits; +s32 tbli; +s32 x; +s32 y; +{ +register u32 bb __asm__("$14"); + u32 *q; +register u32 v __asm__("$25"); +register u32 cl __asm__("$3"); +register u32 cs __asm__("$5"); +register u32 ca __asm__("$8"); + s16 i; + u32 mk1; + u32 cc1; + u32 flag; +register u32 nn __asm__("$2"); + u32 n; +register u32 t __asm__("$13"); + u32 col; + u32 uv; + u32 tt; + u32 nv; + u32 x1; + u32 x2; + u32 w; + u32 g; + u32 gg; + u32 w3; +register u32 yr __asm__("$16"); +register u32 yt __asm__("$4"); +register u32 tr __asm__("$21"); +register u32 xr __asm__("$17"); +register u32 c3 __asm__("$18"); +register s32 ff __asm__("$19"); +register s32 two __asm__("$20"); + yt = y; + tr = tbli; +__asm__("" : "=r"(tr) : "0"(tr)); + xr = x; +__asm__("" : "=r"(xr) : "0"(xr)); + mk1 = 0xFFFFFF; + cc1 = 0x74808080; + bb = bits; + t = x + 0xE; + flag = 0x1000000; + i = 0; + two = 2; + ff = 255; + ; + v = D_8018A300[(s16) tbli]; + gg = ((u32) (p - 5)) & mk1; + ca = 0x3000000; + p[0] = gg | ca; + x1 = (x - 3) & 0xFFFF; + x2 = (x + 5) & 0xFFFF; + p[1] = cc1; + yr = yt; +__asm__("" : "=r"(yr) : "0"(yr)); + yt = (s16) yt; + cs = (yt + 1) << 16; + w = cs | x1; +__asm__("" : "=r"(w) : "0"(w)); + cl = ((v << 6) | 0x4016) << 16; + p[2] = w; + p[3] = cl | 0x1800; + p += 5; + p[0] = (((u32) (p - 5)) & mk1) | ca; + p[1] = cc1; + p[2] = cs | x2; + p[3] = cl | 0x1808; + p += 5; + q = p; + yt = yt << 16; + { + for (; i < 3; i++) + { + nn = ((bb << 16) >> 18) >> 10; + n = nn; + if (((nn != 0) || (i == two)) || (i == ff)) + { + flag = 0; + } + q[0] = (((u32) (q - 5)) & 0xFFFFFF) | 0x3000000; + q[2] = (yt | (t & 0xFFFF)) | flag; + col = 0x74808080; + q[1] = col; + nv = ((n * 8) + 8) | 0x4000; + q[3] = cl | nv; + q += 5; + t += 8; + bb <<= 4; + } + + } + p = q; +__asm__("" : "=r"(v) : "0"(v)); + g = (((u32) (p - 5)) & 0xFFFFFF) | 0x3000000; +__asm__ __volatile__(""); + cs = yr << 16; + p[0] = g; + w3 = cs | ((xr + 0x2A) & 0xFFFF); +__asm__ __volatile__(""); + cl = ((v << 6) | 0x4016) << 16; + uv = ((s16) tr) << 4; + p[2] = w3; + tt = uv | 0x1000; + p[1] = col; + p[3] = cl | tt; + p += 5; + p[0] = (((u32) (p - 5)) & 0xFFFFFF) | 0x3000000; +__asm__ __volatile__(""); + cs = cs | ((xr + 0x32) & 0xFFFF); + uv = uv | 0x1008; + cl = cl | uv; + p[1] = col; + p[2] = cs; + p[3] = cl; + p += 5; +__asm__("" :: "r"(tr), "r"(xr)); + return p; +} INCLUDE_ASM("asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_801734BC", func_80177DA8); diff --git a/src/ov_SC01_077/ov_SC01_077_jr_80182268.c b/src/ov_SC01_077/ov_SC01_077_jr_80182268.c index 5395dee3db..3f529af338 100644 --- a/src/ov_SC01_077/ov_SC01_077_jr_80182268.c +++ b/src/ov_SC01_077/ov_SC01_077_jr_80182268.c @@ -3495,7 +3495,7 @@ INCLUDE_ASM("asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_80182268", func_8018332 INCLUDE_ASM("asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_80182268", func_80183834); -extern void func_8012E364(void); +extern void func_8012E364(); void func_80183A50(void) { func_8012E364(); diff --git a/src/shared/engine_core.h b/src/shared/engine_core.h index 50d2b28909..a6c124b01c 100644 --- a/src/shared/engine_core.h +++ b/src/shared/engine_core.h @@ -16924,7 +16924,7 @@ } #define DEFINE_func_801303A0() \ - extern void func_8012E364(void); \ + extern void func_8012E364(); \ extern void func_80131CA8(int a0, int a1); \ void func_801303A0(s32 a0) { \ if (*(s32 *)(a0 + 0xB4) & 0x2) { \