diff --git a/.run/P36/agents/ov_SC04_011__func_80133784/body.c b/.run/P36/agents/ov_SC04_011__func_80133784/body.c index 6f72a4367..36165df32 100644 --- a/.run/P36/agents/ov_SC04_011__func_80133784/body.c +++ b/.run/P36/agents/ov_SC04_011__func_80133784/body.c @@ -12,13 +12,12 @@ s32 func_80133784(s32 arg0, void *arg1, s32 arg2) { s32 s4; s16 a0v; s16 arg0s; - s32 dx, dy, dz; s32 r; a0v = ((s16)arg0); s1 = 0; - s4 = 0; s3 = 0; + s4 = 0; s2 = 0; arg0s = a0v; D_801909BC->f6 = -0x7FFF; @@ -35,8 +34,7 @@ s32 func_80133784(s32 arg0, void *arg1, s32 arg2) { s16 sy = ((Box_80133784 *)arg2)->f2 - ((Box_80133784 *)arg1)->f2; s16 sz = ((Box_80133784 *)arg2)->f4 - ((Box_80133784 *)arg1)->f4; if (sx == 0 && sy == 0) { - s32 zt = (sz == 0); - s2 = zt; + s2 = (sz == 0); } D_801909BC->f2 = ((Box_80133784 *)arg1)->f2 - 4; r = func_80047D3C(sx * sx + sz * sz); @@ -78,17 +76,21 @@ loop: goto store_out; after: + { + s32 lim = -0xBCB; + s32 mask = 0xFFFF; + if ((s16)s4 != 0 || D_801EDA3C != 0) { s16 t; t = D_801909BC->f6; - if (t >= -0xBCB) { + if (t >= lim) { if (t < -0x578) { s1 |= 0x4000; } else { s1 |= 0x8000; } } - if ((s16)D_801909C0->f6 < -0xBCB) { + if ((s16)D_801909C0->f6 < lim) { s1 |= 0x2000; } store_out: @@ -96,7 +98,8 @@ after: ((Box_80133784 *)arg2)->f2 = D_801909C0->f2; ((Box_80133784 *)arg2)->f4 = D_801909C0->f4; ((Box_80133784 *)arg2)->f6 = D_801EDA40; - return s1 & 0xFFFF; + return s1 & mask; + } } ((Box_80133784 *)arg2)->f6 = D_801EDA40; return 0; diff --git a/.run/P36/agents/ov_SC04_011__func_80133784/mechanism.md b/.run/P36/agents/ov_SC04_011__func_80133784/mechanism.md index b00ec1b56..e2aa45844 100644 --- a/.run/P36/agents/ov_SC04_011__func_80133784/mechanism.md +++ b/.run/P36/agents/ov_SC04_011__func_80133784/mechanism.md @@ -1,44 +1,165 @@ -# func_80133784 (ov_SC04_011) — mechanism (WORK IN PROGRESS) +# func_80133784 (ov_SC04_011) — mechanism -Best so far: **score 19** (from 21; mechanical best was 12 — but that was a different family, see below). -Best text: `PACK/scratch/v/J.c` (= `PACK/body.c`). +**Final score 2** (from 21; the mechanical search's best was 12). `mine 203 ins, target 203`, one +two-instruction residual left. Text: `PACK/body.c`. Harness calibrated first (R39): the tree's own +`body_tree.c` scores **0** through the same `--try` path, so the instrument was measuring this function. ## (a) The residual in one sentence -The lever-free body is 2 instructions SHORT of the target (201 vs 203): two register-to-register -copies that the target keeps (`sltiu v0,v0,1` + `move s2,v0`, and `move v1,v0` after the call) are -coalesced away in mine — and, before the two moves are even reachable, mine hoists the call -argument's sign-extension `sll` OUT of the loop and swaps two callee-saved registers. -## (b) The pass and the decision, read from the compiler's own source -**loop.c — `scan_loop` / `invariant_p` (the `sll` hoist).** `loop_optimize` only sees a loop when the -RTL carries `NOTE_INSN_LOOP_BEG`/`NOTE_INSN_LOOP_END`, and those notes are emitted only by the -`while`/`for`/`do` statement expanders (`stmt.c: expand_start_loop`). Written as `while (1) { … }`, -the body's first call argument `(ashift (reg arg0s) 16)` is loop-invariant -(`tools/reference/gcc-2.7.2/loop.c:2745-2751`, `invariant_p` case REG returns -`n_times_set[REGNO] == 0`), so `scan_loop` records it as a movable -(`loop.c:645-710`) and `move_movables` hoists it into the preheader -(`loop.c:1631` — `threshold * savings * m->lifetime >= insn_count`, and with a call in the loop -`threshold = 1 * (1 + n_non_fixed_regs)` is ~60, so the test always passes). -Written as a **label + `goto`**, no loop notes exist, `loop_optimize` never sees a loop, and the -`sll` stays where the target has it (at the branch target, re-executed every iteration). -PROVEN on bytes: the `sll a0,s5,0x10` / `sra a0,a0,0x10` pair and the `arg0s`/flag register swap -all disappear from the diff the moment the `while (1)` becomes `loop: … goto loop;`. +The lever-free body is two instructions SHORT of the target (201 vs 203) because two register-to-register +copies the target keeps are coalesced away (`sltiu v0,v0,1` + `move s2,v0`, and `move v1,v0` after the +loop's call), and — before those are even reachable — mine hoists the call argument's sign-extension `sll` +out of the loop and swaps two pairs of callee-saved registers (`arg0s`/flag, counter/accumulator). -**global.c — `allocno_compare` (the `$s3`/`$s4` swap).** With the loop fixed, the counter and the -accumulator are two 4-reference global allocnos whose priorities -`floor(log2 refs) * refs / live_length * 1e4` differ by ~2%: measured with -`tools/alloc_table.py`, source order `s3 = 0; s4 = 0;` gives counter 8/89 = 898.9 and accumulator -8/87 = 919.5, so the accumulator is allocated first and takes `$s3`. Exchanging the two -initialisers moves the counter to 8/86 = 930.2 ahead of the accumulator's 8/90 = 888.9 and the -two registers come out as the target has them. PROVEN on bytes (score 22 -> 19). +Four independent defects, four different passes. Each was read from the compiler source, then proven on bytes. -## (c) Source moves so far -1. `while (1) { … }` -> `loop: { … if (oldc < 5) goto loop; }` — kills the loop-invariant hoist. -2. `s3 = 0; s4 = 0;` -> `s4 = 0; s3 = 0;` — flips `allocno_compare` so the counter gets `$s3`. +## (b) The passes and the decisions, with `file:line` -## (d) Generator proposal -(placeholder — see the final version) +### 1. The hoisted `sll` and the `$s2`/`$s5` swap — `loop.c` -## (e) Not yet closed -- The two surviving copies (`move s2,v0`, `move v1,v0`). -- The prologue emission order of the two zero-initialisers (a side effect of move 2). +`loop_optimize` only sees a loop where the RTL carries `NOTE_INSN_LOOP_BEG`/`NOTE_INSN_LOOP_END`, and those +notes are emitted **only** by the statement expanders (`stmt.c: expand_start_loop` emits the note then the +start label; `expand_end_loop` emits `NOTE_INSN_LOOP_END`). Inside such a loop the first call argument's +`(ashift (reg arg0s) 16)` is invariant — `invariant_p`'s REG case returns `n_times_set[REGNO] == 0` +(`tools/reference/gcc-2.7.2/loop.c:2745-2751`, the same place that makes a **hard** register never invariant, +because `loop.c:595-596` forces `n_times_set[i] = 1` for every hard reg: that is exactly what the tree's +`register s16 arg0s __asm__("$21")` pin was buying). `scan_loop` therefore records the `sll` as a movable +(`loop.c:645-710`; the dest is a compiler temp, so the `! REG_USERVAR_P && ! REG_LOOP_TEST_P` arm of the +three-way gate at `loop.c:694-700` passes), and `move_movables` hoists it into the preheader — +`loop.c:1631` `threshold * savings * m->lifetime >= insn_count`, and with a call in the loop +`threshold = 1 * (1 + n_non_fixed_regs)` (`loop.c:532`) is ~60 against an `insn_count` of ~15, so the test +never fails. Allocation then ties `arg0s` to the hoisted temp (`sll s2,s2,0x10` destroys the variable, +which is legal because every later use goes through the shifted value), and the flag/`arg0s` priorities swap. + +**Move:** write the loop as `loop: { … if (oldc < 5) goto loop; }`. A label+`goto` loop carries no loop +notes, `loop_optimize` never registers a loop object, and the `sll` stays at the branch target exactly where +the target has it. PROVEN: the six diffs at indices 22/25/56/103/109/117 all vanish. +Controls: `while (1) {…break;}`, `do {…} while (s3++ < 5)` and `for (;;)` were each tried — **all three hoist** +(score 21); only the goto form does not. This is the *reverse* of cookbook §176 (`func_80186A04`, where a +goto-loop was rewritten as a `for` to GET the hoist). + +### 2. The missing `move v1,v0` after the loop's call — `global.c: find_reg` + +A call result reaches its pseudo through a real copy insn `(set p (reg:SI 2))`. Combine cannot delete it +(the pseudo has two uses), so it survives or dies purely by allocation: local-alloc's `combine_regs` +records `qty_phys_sugg` for a hard-reg source (`local-alloc.c`, the `if (ureg < FIRST_PSEUDO_REGISTER)` +arm), and for a *global* allocno `global.c: find_reg` follows `hard_reg_copy_preferences`. With a +dedicated `ret` variable that preference is free and the copy is coalesced into `$v0`. The function's +other call (`func_80047D3C`) already emits `move v1,v0` in **both** mine and the target, because that +result's live range does conflict with `$v0`. + +**Move:** use **one variable `r` for both call results**. One allocno now spans both call sites, `$v0` is +no longer free across it, `find_reg` gives it `$v1`, and the copy out of the return register survives at +*both* sites — which is what the target has. PROVEN: 19 → 15, the whole loop region matches. + +### 3. The missing `move s2,v0` on the flag — `combine.c: try_combine` + +The dumps settle this one. In my body `.combine` holds a single +`(insn 128 (set (reg/v:SI 76) (eq:SI (reg:SI 118) (const_int 0))) 236 {seq_si_zero})` — combine merged +`(set r119 (eq r118 0))` with the copy `(set r76 r119)`; in `body_tree`'s dump the same two insns survive +with the `asm` wedged between them. Any *same-mode* copy whose source dies at the copy is merged +(`can_combine_p` `combine.c:803-…` lets it through: same block, LOG_LINK present, source dies, and +`INSN_CUID (i2) < last_call_cuid` does not apply because no call separates them). + +**Move (partial):** declare the flag `u16`. The store is then a HImode subreg store that combine cannot +fold into the SImode `seq`, so `sltiu v0,v0,1 ; move s2,v0` comes out byte-exact. **Cost:** the loop test +`if (s2 != 0)` needs a zero-extension `andi v0,s2,0xffff`, and it lands in a delay slot that the target +fills with `nop` — this is the entire residual 2. + +### 4. The prologue order of the two zero-inits / the `$s3`/`$s4` swap — `global.c: allocno_compare` + `flow.c` + +`allocno_compare` ranks by `floor(log2 refs) * refs / live_length * 1e4` and breaks ties by allocno number +(declaration order — the counter is declared before the accumulator). The emitted order of +`move s3,zero` / `move s4,zero` (and of their prologue `sw`s, which sched2 pairs with them) follows the +**source** order of the two initialisers, so the source must read `s3 = 0; s4 = 0;`. + +In a real `while` loop `flow.c:2071` (`reg_n_refs[regno] += loop_depth`) weights in-loop references ×2: +counter 1+3×2 = **7 refs**, accumulator 1+2×2+1 = **6** — measured with `tools/alloc_table.py` +(`free`: r77 7/89 = 1573, r78 6/87 = 1379) and the counter wins. In the goto loop there is no loop depth, +both have **4 refs**, and with the counter born one insn earlier its live length is 91 vs 89 +(`u2`: r77 4/91 = 879.1, r78 4/89 = 898.9) — so the accumulator wins and takes `$s3`. + +**Move:** hoist two repeated integer literals into named locals at the head of the `after:` block +(`lim = -0xBCB`, used twice; `mask = 0xFFFF`). `flow_analysis` runs **once, before combine** +(`toplev.c:2983` vs `:3004`), so those two sets are counted in `reg_live_length` even though cse/combine +fold them away and no instruction is emitted for them. The accumulator's live length goes 89 → **91** = +the counter's; the tie then breaks by allocno number in the counter's favour and it takes `$s3`. +PROVEN: 6 → 2, the prologue diff disappears, instruction count unchanged at 203. + +## (c) The source moves (one line each) + +1. `while (1) { … }` → `loop: { … if (oldc < 5) goto loop; }` — kills `loop.c`'s invariant hoist of the + call argument's `sll` (21 → 22 alone, but it fixes 6 later diffs). +2. One variable `r` for **both** call results instead of `r` + `ret`/`ret0`/`retc` — makes the loop call's + result land in `$v1`, so `move v1,v0` survives (19 → 15). +3. `s32 s2` → `u16 s2` — the HImode subreg store blocks `try_combine`, so `move s2,v0` survives (15 → 6). +4. `lim = -0xBCB` / `mask = 0xFFFF` hoisted into locals at the head of `after:` — pads the accumulator's + flow-time live length by 2 at zero instruction cost, flipping `allocno_compare` (6 → **2**). + +(Also: init order stays the natural `s1, s3, s4, s2`; `dx, dy, dz` dropped; `zt` inlined — all byte-neutral.) + +## (d) Generator proposals + +**Primary — `delever.deloop` (the reverse of the `for`-loop rewrite):** when the residual shows a +loop-invariant sub-expression of a call argument (typically an `sll`/`sra` of a callee-saved register) +sitting in the loop **preheader** in mine but at the **branch target** in the target, together with a +register swap involving that same register, rewrite the `while`/`for`/`do` loop as a label plus `goto` so +no `NOTE_INSN_LOOP_BEG` is emitted and `loop.c` cannot register a loop at all. + +Two more, each byte-proven here: + +* **`delever.share_call_result`:** when the diff is a missing `move ,v0` immediately after a call, + assign that call's result to a variable **already used for another call's result** in the same function — + one allocno spanning both sites loses `$v0` and the copy survives at both. +* **`delever.liveness_pad`:** when two allocnos with equal `n_refs` are within a few live-length units in + `allocno_compare` and the wrong one is allocated first, hoist a **repeated integer literal** into a named + local declared at the head of the block where the allocno that must win is still live; `flow` runs once + before combine, so the set counts in `reg_live_length` while cse/combine delete it (zero bytes). One unit + per surviving literal; a literal whose only use is in a *nearby* block is folded early and buys nothing — + `-0xBCB` (two uses) and `0xFFFF` (used in the far `store_out` block) each bought 1, `-0x578`, `0x4000`, + `0x8000` and a hoisted `0` bought nothing. + +## (e) What did NOT work, with byte evidence + +* Ten spellings of the flag assignment with an `s32` flag — `s2 = (sz==0)`, `!sz`, `(sz?0:1)`, `(s16)/(u8)` + casts, a named `zt` at function scope, `zt` with two sets, a dead second consumer, splitting the + `&&` so the store lands in another block, a `goto`-split, a duplicated guard — **all score 19** (or worse): + combine merges every same-mode copy whose source dies. Only a mode change blocks it. +* `do {…} while (s3++<5)`, `for (;;)`, `while(1)` — all hoist the `sll` (21). `arg0s = arg0s;` and + `arg0s = a0v;` inside the loop do not defeat `invariant_p`: jump.c deletes the no-op move and cse + propagates the copy's source, so the `sll` is invariant again (21). +* `do { … } while (0)` wedged inside the goto loop to buy back `flow`'s loop-depth ref weighting: the refs + do inflate (r77 7/89, r78 6/87, verified in `alloc_table`) but the allocation rotates three ways + (counter→`$s4`, accum→`$s2`, flag→`$s3`) — **26**, worse than doing nothing. +* All 24 permutations of the four zero-initialisers, on four different bases: the emitted prologue order + always follows the source order, so no permutation gives both the right order and the right registers + (best 6 either way). +* Swapping the `||` in the `after:` test (`D_801EDA3C` first) does buy the tie (both live lengths 91) and + fixes the prologue — but reorders five instructions: net **10**, no gain. That measurement is what + identified "+2 live units on the accumulator" as the actual requirement. +* Widths of `s1`, `s3`, `s4`, `a0v`, `arg0s`, `ret`, `retc`, `zt` (s8/u8/s16/u16/s32/u32): none helped; + `s16 ret` gives `sll/sra` (2 insns) where the target wants one `move`. + +## (f) Where the method fell short + +* **`sites.txt` is a decoy at the shape level.** It listed 7 NEEDED sites; none of them named what actually + had to change. One source-shape move (the goto loop) retired three of them at once, and the two + `instruction addu` sites turned out to be two *different* mechanisms (allocation for one, combine for the + other) that look identical in the table. +* **The residual text cannot separate "instruction missing" from "registers wrong".** Counting first + (201 vs 203) and mapping each diff hunk to an index in the target's own `objdump` was what made the four + defects separable. Recommend the pack print the target's disassembly of the function alongside `residual.txt`. +* **`alloc_table.py` was decisive but only after I stopped reading it as a ranking and started doing + arithmetic on it.** The winning move came from "I need +2 live-length units on r78", which the table + gives directly. A `--diff` mode (two tags side by side, priorities and live lengths only) would have + saved most of the search. +* **The knowledge base has the `for`-rewrite direction but not the `goto`-rewrite direction.** Cookbook §176 documents + turning a goto-loop into a `for` to *obtain* a hoist; the inverse — de-looping to *defeat* one — is the move + that unlocked this body and should be banked next to it. + +## Paths + +* `PACK/body.c` — the best text (score 2). +* `PACK/mechanism.md` — this file. +* `PACK/scratch/` — the probe harness (`t.sh`, `mkdump.sh`, `dis.sh`, `sweep.py`, `probe.py`, `cut.py`), + every candidate under `PACK/scratch/v/`, and the cc1 pass dumps under `PACK/scratch/dumps/`. diff --git a/.run/P36/agents/ov_SC04_011__func_8013F350/body.c b/.run/P36/agents/ov_SC04_011__func_8013F350/body.c index 23abfcc3c..d3ed3ccc9 100644 --- a/.run/P36/agents/ov_SC04_011__func_8013F350/body.c +++ b/.run/P36/agents/ov_SC04_011__func_8013F350/body.c @@ -3,7 +3,7 @@ s32 func_8013F350(void) { u16 *ps; u16 *pf; u16 *pg; - u16 pad; + s32 pad; u16 st; u8 *pcur; u8 *pmax; @@ -20,7 +20,7 @@ s32 func_8013F350(void) { s32 chg; pad = *pd; - chg = 0; + do { chg = 0; } while (0); if (pad != 0) { if (pad == pd[2]) { D_80115122 = D_80115122 - 1; @@ -167,9 +167,9 @@ s32 func_8013F350(void) { ps = &D_8011511A; st = *ps; + off = st << 1; p2e = (u8 *)ps + 0x2E; p3e = (u8 *)ps + 0x3E; - off = st << 1; pcur = p2e + off; pmax = p3e + off; if (st == 2 || 1 < pmax[0]) { diff --git a/.run/P36/agents/ov_SC04_011__func_8013F350/mechanism.md b/.run/P36/agents/ov_SC04_011__func_8013F350/mechanism.md new file mode 100644 index 000000000..5a762f248 --- /dev/null +++ b/.run/P36/agents/ov_SC04_011__func_8013F350/mechanism.md @@ -0,0 +1,164 @@ +# func_8013F350 (ov_SC04_011) — T7 agent reading + +**Start 70 · FINAL 16 · NOT closed.** All 16 points are in instructions 0-32; instructions 32-489 (458 of +490) are byte-identical in plain C, and the `register s32 off __asm__("$4")` pin is fully replaced by a +one-line statement move. The remainder is the `pd` launder + its `$5` pin, and that half is a wall (below). +Levers in the tree: `register u16 *pd __asm__("$5")`, `register s32 off __asm__("$4")`, and a +`__asm__ __volatile__("" : "=r"(pd) : "0"(pd))` launder on `pd` (sites.txt: all three NEEDED). + +## (a) The residual, in one sentence + +Two independent defects, and they are **not** the same class: a **TAIL** one — the two derived pointers +`pcur`/`pmax` are computed in the wrong order and take each other's colours (`a0<->a2` ×20 across the whole +byte-poke section) — and a **HEAD** one — the target holds `&D_8011511C` in a register (`lui a1 / addiu a1`, +then `lhu 0(a1)`, `lhu 4(a1)`, `sh 4(a1)`) while every lever-free spelling gets that pointer +**constant-folded away** into three separate `lui/%lo` accesses. + +## (b) The passes, read from the compiler's own source and from its dumps + +### TAIL — solved. It is a COLOURING move, not a scheduling one. `local-alloc.c` + `reorg.c`. + +`off = st << 1;` was written *after* `p2e`/`p3e`. Hoisting it above them is worth **59 -> 23** on the bytes and +takes the entire tail (all 20 `a0<->a2` register pairs and the missing `nop`) to zero. Read off the dumps +(`scratch/dumps/dumps_d_m1` vs `dumps_d_t2`, produced from the two candidates), the chain is: + +1. **The RTL order does not change at all.** In BOTH candidates `.sched` and `.sched2` emit the same stream: + `789` base, ..., `804` = `pcur = off + p2e`, `807` = `pmax = off + p3e`, `811` = `li v0,2`, `812` = `beq`. + sched1/sched2 are innocent. The tree's header comment says "gcc otherwise emits pmax before pcur" — true of + the ASSEMBLY, but the RTL order never changes; what moves is the colouring, and reorg then reorders the + printed stream. Worth correcting in the tree's note, because the wrong attribution sends the next reader + to `sched.c`. +2. **The COLOURS change.** With `off` written last: `p2e`->`$a0`, `off`->`$v1`, `pcur`->`$a0` (reusing the dead + `p2e`), `pmax`->`$a2`. With `off` written first: `off`->`$a0`, `p2e`->`$v1`, `pcur`->`$a2`, + `pmax`->`$a0` (reusing the dead `off`) — the target's assignment. This is `block_alloc`'s ordering of the + block's quantities (`local-alloc.c:1483-1512`, `qty_compare` at `:1579`) feeding `find_free_reg`'s + lowest-free-regno live-range scan (`local-alloc.c:2109-2158`): moving the statement moves `off`'s birth + ahead of `p2e`/`p3e`, so `off` takes `$a0` and dies into `pmax`. +3. **`reorg` then decides the `nop`.** `fill_simple_delay_slots` scans BACKWARD from the branch + (`reorg.c:2906-2952`), skipping insns that conflict and **accumulating their `set`/`needed` as it goes**. + Branch `beq $a1,$v0`: `needed = {a1,v0}`. + * mine: `li v0,2` conflicts (sets `$v0`); `pmax = addu $a2,$v1,$v0` conflicts (reads `$v0`, now in `set`); + `pcur = addu $a0,$v1,$a0` touches nothing in `set`/`needed` -> **eligible, stolen into the delay slot**. + * target: `li v0,2` conflicts; `pmax = addu $a0,$a0,$v0` conflicts; `pcur = addu $a2,$a0,$v1` now READS + `$a0`, which the skipped `pmax` put in `set` -> conflict; every earlier insn conflicts too -> **`nop`**. + + So the `nop` is a *consequence* of `pmax` reusing `off`'s register, not an independent scheduling fact. + +### HEAD — a wall for plain C. `cse.c:2663-2665`, in `find_best_addr`. + +```c + if (GET_CODE (addr) != REG + && validate_change (insn, loc, fold_rtx (addr, insn), 0)) + addr = *loc; +``` + +`fold_rtx` resolves `(reg pd)` through `equiv_constant` (`qty_const`, set because `pd`'s only set is +`(set (reg 72) (symbol_ref "D_8011511C"))`), so `(mem (plus (reg pd) 4))` — i.e. `pd[2]` — is rewritten to +`(mem (const (plus (symbol_ref "D_8011511C") 4)))`, an address MIPS accepts, so `validate_change` succeeds. +**Verified in the dumps**: `dumps_cur/cur.i.cse` already carries the folded `(const (plus (symbol_ref +"D_8011511C") (const_int 4)))` for both `pd[2]` sites while `.jump` (pre-cse) does not. That leaves `pd` with +one remaining use, `(mem (reg pd))` for `*pd`, which **combine** then substitutes and deletes +(`dumps_cur/cur.i.combine` line 17 is `(mem:HI (symbol_ref "D_8011511C"))`; the pseudo is gone by `.lreg`). + +The zero-byte launder works because the asm gives `pd` a second SET and an unknowable value, so `qty_const` +is never set and neither fold can fire. **There is no plain-C equivalent**: any spelling of a link-time +constant address is a constant to cse, so the moment one use carries a non-zero offset that use is folded, +and the remaining zero-offset use is then single-use and folded by combine. + +The known-true control is `func_8013F138` (`src/ov_SC06_008/ov_SC06_008_jr_8013C98C.c:2209`), which keeps its +base in plain C — `u16 *p = &D_80115118; *p += 0x10;` — and does so **only** because it has TWO uses and both +are at offset 0: `find_best_addr` returns early for a plain `(reg)` address, and combine cannot substitute a +reg with two uses. Our function needs offsets 0 **and** 4, so it falls outside that escape. + +## (c) The source moves + +1. `off = st << 1;` moved **above** `p2e = ...; p3e = ...;` (a 1-line statement move). 59 -> 23. Closes the + whole tail; this is the plain-C replacement for the `register s32 off __asm__("$4")` pin. +2. `s32 pad;` instead of `u16 pad;` (R12). 70 -> 59. **This one is a compensating error, not a fix**: it + deletes the target's real `andi v1,a0,0xffff`, which cancels the ONE extra instruction the folded head + costs (3 x `lui/%lo` = 6 insns where the target's held base needs 5). It keeps the instruction count at + 490 = 490 and is worth 11 points, but it is not part of any byte-identical spelling. +3. `do { chg = 0; } while (0);` (R7). 23 -> 16. Also a compensating move, and not readable: it only shuffles + the head's remaining wrong instructions closer to the target's. **`body.c` is the best-scoring text, not a + proposal for the tree** — a readable stop would be `u16 pad; chg = 0;` plus move 1 alone, which scores 42 + but contains only the one honest fix. Nothing here is bankable: the score is not 0. + +## (d) GENERATOR PROPOSAL + +**When two derived pointers built from one shared operand (`x = A + t; y = B + t;`) swap registers in the +residual, and the target has a `nop` in a delay slot where mine has one of them, hoist the SHARED operand's +defining statement to the front of its straight-line run.** + +Mechanically: for every local `t` that is read by two or more later statements of the same brace block, offer +the candidate with `t`'s defining statement moved to the first position of the run that contains all of its +readers. That is one candidate per such local — a handful, not a search. It is `R18 bystander_moves` with the +independence test INVERTED: R18 only moves a statement that shares NO identifier with what it crosses, which is +precisely why seven mechanical runs and ~4,000 compiles never generated this one; the move that matters here +shares `st`/`off` with everything around it. The payoff is that the shared operand is born first, so it wins the +low register and DIES into the second consumer, which is what the `register T x __asm__("$4")` pin was faking. +**Byte-proven here: 59 -> 23 in one compile, closing 458 of 490 instructions.** + +## (e) What did NOT work, with byte evidence (every number is a `--try` score) + +* `u16 *pp = pd + 2` / `u16 *pe = &D_80115120` with `*pp` at both `pd[2]` sites — **70 / 39**. The dumps say + why: with a plain-`(reg)` address cse1 keeps it (`dumps_s5/s5.i.loop` still has `(mem (plus (reg 72) 4))`), + but cse1's own `ADDRESS_COST` tie-break at `cse.c:2710-2726` — *equal address cost, higher rtx cost wins* — + rewrites `(reg pe)` back to `(plus (reg pd) 4)`, and cse2 then folds it (`.flow` has the constant). +* A second, redundant `pd = &D_8011511C;` in either arm of the `if` (the §349 "make the base set more than + once" lever) — **23, i.e. no change**: jump/flow delete the redundant set before `reg_n_sets` is read. +* `register u16 *pd` (the plain-C keyword, no asm) — **70**. DECL_REGISTER does not reach cse. +* Struct-pointer spelling (`struct { u16 f0, f2, f4; } *pd`) — **23** (identical RTL). +* Splitting the declaration from the initializer, reordering `chg = 0;` / `pad = *pd;`, `{ }` and + `do { } while (0)` around `pad = *pd;`, inverting the outer `if`, swapping the inner arms, a `goto` tail — + **23, 23, 23, 33, 46, 26, 46**. None changes the fold. +* Reusing `pd` for the tail's `ps` (two sets, natural) — **67**: it merges the two bases and breaks the tail. +* Widths on `off` (u16/s16/s8/u32) — **65/65/65/59**. +* **The positive control**: `body.c` + the launder alone scores **9** and its class is `REG-caller` — the only + thing left after the launder is the 3-cycle `pd v1->a1 / pad a1->a0 / masked-pad a0->v1`, which is what the + `$5` pin buys. So the head is worth 16 of the residual and is the whole remaining distance. + +* **`volatile` does not rescue the head either** (run only as a control — it is banned in the deliverable): + `volatile u16 *pd` scores **29 at 493 instructions** (three EXTRA insns; volatile blocks the fold but also + blocks every reuse), and a `(u16 *)(volatile u16 *)` cast scores **42**, identical to no change. So the head + is not a "banned-construct" wall that a rules change would open — only a value cse cannot know opens it. +* **One-move exhaustion**: after reaching 16 I ran every delever generator (R2,R3,R4,R5,R6,R7,R8,R9,R10,R12, + R13,R14,R15,R16,R17,R18,R19,R20,R21 — 313 candidates) over the 16-body through `--try`. **No single move + improves it.** 16 is a one-move local optimum, and it is the same 16 the mechanical runs reached by a + different four-move path. +* **Final residual scope**: every difference is inside instructions 0-32. Instructions 32-489 (458 of 490) are + byte-identical, in plain C, with the `$4` pin removed. + +## (f) Where the method fell short + +* Step 2's "COUNT FIRST" was misleading here: the counts are **equal** (490 = 490) and the class prints + `COUNT`, yet nothing is missing — two opposite errors cancel (the folded head is +1 insn, the `s32 pad` + widening is -1). A count check that only compares totals hides that. What actually located both defects was + the whole-function `objdump` of the tree's own object versus the candidate's `.s` (step 1), read side by + side around the two `lui` sites. +* `alloc_table` was of no use on this body: the defect never reaches allocation, because the pseudo is + **deleted by combine** before `.lreg`. The table's 158 rows are all downstream of that. A cheap pre-check — + "is the pseudo you are reasoning about still present in `.lreg`?" — would have saved the detour. +* The pack's `history.txt` reports the mechanical best as 16 with a four-move path; hand-reconstructing that + path from the descriptions landed at 51 (the `@NNNN` line numbers are relative to the *evolving* text, not + to `body_free.c`). Re-running delever's own generators through `--try` reproduced it in one round. A pack + that shipped the best candidate's TEXT, not just its move names, would have saved ~30 minutes. +* **`neighbours.txt` dropped the one document that mattered.** It reproduced this function's own + `// @class: regalloc-order` / `// @stuck: none — MATCH (490 ins, ...)` pair and stopped there — but the + header comment those two lines belong to (`src/ov_SC04_011/ov_SC04_011_jr_8013F350.c:910-948`) is a + **numbered, eight-point English explanation of every lever in this body**, including "gcc otherwise emits + pmax before pcur, letting pcur sink into the beq delay slot (target has a nop there)" — which is the tail + crack, stated outright, and which I re-derived from the dumps before I found it. The pack should carry the + function's own header comment IN FULL, not just the two tag lines grepped out of it. + +## Score ladder (every number from `--try`, in order) + +| body | score | +|---|---| +| `body_free.c` (start) | 70 | +| `+ s32 pad` (R12) | 59 | +| `+ off = st << 1;` hoisted above `p2e`/`p3e` | **23** | +| `+ do { chg = 0; } while (0);` (R7) | **16** | +| `off`-hoist alone, `u16 pad` kept | 42 | +| control: `body_free.c` + the launder only | 45 | +| control: the 23-body + the launder, `s32 pad` | 27 | +| control: the 23-body + the launder, `u16 pad` | **9 (REG-caller — only the `$5` pin's 3-cycle left)** | diff --git a/decomp-architect/corpus/tools/P10/delever_pack.py b/decomp-architect/corpus/tools/P10/delever_pack.py index 70402df3e..7150c0061 100644 --- a/decomp-architect/corpus/tools/P10/delever_pack.py +++ b/decomp-architect/corpus/tools/P10/delever_pack.py @@ -112,6 +112,25 @@ def build(a): if head: out_n.append(f"--- {defs_at[k][1]} (line {ln + 1}) ---") out_n += list(reversed(head)) + # THE TARGET'S OWN HEADER, IN FULL. neighbours.txt used to carry only the @class/@stuck LINES, and agent b9 + # (S102) lost hours to exactly that: its function's header comment is an eight-point English explanation of + # every lever it carries, including the tail crack stated outright, and the pack had reduced it to two + # tagged lines. A grep for tags is not a substitute for the paragraph they sit in. + if here is not None: + ln0 = ntxt[:defs_at[here][0]].count("\n") + own, j = [], ln0 - 1 + while j >= 0 and len(own) < 80 and (nlines[j].lstrip().startswith("//") + or nlines[j].lstrip().startswith("*") + or nlines[j].lstrip().startswith("/*") + or nlines[j].strip() == ""): + if nlines[j].strip() == "" and own: + break + if nlines[j].strip(): + own.append(nlines[j]) + j -= 1 + if own: + out_n = [f"=== THIS FUNCTION'S OWN HEADER ({fn}, line {ln0 + 1}) — read it in full ==="] \ + + list(reversed(own)) + [""] + out_n tagged = [l for l in nlines if "@class:" in l or "@stuck:" in l or "@crack:" in l] if tagged: out_n.append("--- every @class/@stuck/@crack note in this translation unit ---") @@ -119,6 +138,15 @@ def build(a): (d / "neighbours.txt").write_text("\n".join(out_n[:400]) + "\n") except OSError: pass + # the BEST CANDIDATE'S TEXT, not just its move path: history.txt's `@NNNN` line numbers are relative to the + # EVOLVING text, so hand-reconstructing a path lands somewhere else (agent b9 reconstructed 51 where the engine's + # own generators reproduce 16 in one round). + try: + bp = ds.RUN / "bodies" / f"{e['alias']}__{fn}.c" + if bp.exists(): + (d / "best_body.c").write_text(bp.read_text(errors="surrogateescape"), errors="surrogateescape") + except Exception: + pass (d / "history.txt").write_text("\n".join(hist) + "\n") order.append(f"{k}\t{fn}\t{e['alias']}\t{e['copies']}\t{e['best']}\t{e['needed']}\t{','.join(e['kinds'])}\t{','.join(e['regs'])}\t{tu}") print(f" {k:3d} {fn} {e['copies']:4d} copies best {e['best']} -> {d.relative_to(REPO)}", flush=True) diff --git a/phase-ends/CURRENT_PHASE.md b/phase-ends/CURRENT_PHASE.md index 54030f273..bd59d5cb7 100644 --- a/phase-ends/CURRENT_PHASE.md +++ b/phase-ends/CURRENT_PHASE.md @@ -1306,6 +1306,27 @@ accumulate here as the phase produces them.** instrument: none of the seven NEEDED sites named what actually had to change, and one shape move retired three of them** — a site list says which levers the byte oracle could not remove ALONE, not which source facts are load-bearing. +- **S102 — T7 agent b9: `func_8013F350` NOT closed (70 → 16) and it REFUSED TO OFFER ITS OWN IMPROVEMENTS AS A BANK.** + It solved the whole tail — **instructions 32 to 489 of 490 are byte-identical in plain C with the `$4` pin gone** — by + ONE statement move: hoisting `off = st << 1;` above the two derived pointers, which is a COLOURING move and not a + scheduling one (the `.sched`/`.sched2` RTL order is identical in both candidates, so the tree's own header note + blaming the scheduler describes the assembly, not the RTL). With `off` born first it takes `$a0` and dies into the + later pointer, and `fill_simple_delay_slots`' backward scan (`reorg.c:2906-2952`), which ACCUMULATES set/needed over + skipped insns, then finds the conflict that produces the target's `nop`. **The head is a proven wall**: `find_best_addr` + (`cse.c:2663-2665`) folds the base to an absolute address because its only set is a `symbol_ref`, so combine deletes + the pseudo outright — and its known-true control `func_8013F138` keeps its base in plain C ONLY because both its uses + are at offset 0, where `find_best_addr` returns early. Ours needs offsets 0 and 4. **The honesty that matters: it + labelled its remaining two moves COMPENSATING ERRORS** — a width change that deletes the target's real `andi` to cancel + an extra instruction the folded head costs — and wrote "nothing here is bankable" rather than hand back a 16 dressed as + progress. **Its generator proposal is R18 with the independence test INVERTED** (hoist the shared operand's defining + statement to the front of its run — R18 refuses it precisely because it shares identifiers with what it crosses, which + is why ~4,000 mechanical compiles never tried it). **Two pack defects it found, both fixed:** `neighbours.txt` carried + the `@class`/`@stuck` LINES but not the header comment they sit in — and that comment is an eight-point English + explanation of every lever in the body, including the tail crack stated outright; the pack now ships the target's own + header IN FULL. And `history.txt`'s `@NNNN` line numbers are relative to the EVOLVING text, so reconstructing a path by + hand lands elsewhere (it reached 51 where the engine's own generators reproduce 16 in one round) — the pack now ships + the best candidate's TEXT as `best_body.c`. + ## 🛑 SESSION CHECKPOINT — S101 (2026-09-09) / LIVE, refreshed S102 (2026-09-10): T0–T6 ☑, **T7 RUNNING — agent a1 BANKED + HARVESTED: `func_80156044` (130 bodies) and its move toolified as generator **R15, the sink**, which then closed 6 more exemplars (267 bodies) in 4 compiles each with no tokens; agent a2 banked 125 more; 30,358 → 29,572 sites, R22 218/218**; **lane B DELIVERED + 4 claims verified on bytes; lane A = rung G, `tools/delever_search.py`, BUILT, CONTROLLED, MEASURED over six runs (g1–g6b: 33,427 → 30,358 sites, 12,048 → 9,747 bodies, every bank R22 218/218, no drafting tokens); the head is where the number is (57 classes ≥100 copies = 7,318 of 9,796 residue bodies) and the wide search is spent on it; T7 APPROVED by Drew as ONE AGENT AT A TIME — the packs, the brief and the agent's scorer are built; NEXT = §2: start the serial agent loop IN THIS FRESH SESSION** | the number at this commit: **27,984 sites** in 8,249 bodies · marked 27,984 · UNMARKED 0 · orphans 0 — `lever_census --check` OK · `lever_progress --check` OK (20 milestones). **S102's loop state: the burst of 20 is landing; a4/a7/a8/a12/a18/a19 banked (756 bodies); the per-file scratch-object collision is FIXED (per-function tag); the biggest class found is a truncated `(void)` DECLARATION, three cases of which need the types phase.** ### 0. How to use this block diff --git a/tools/delever_pack.py b/tools/delever_pack.py index 70402df3e..7150c0061 100644 --- a/tools/delever_pack.py +++ b/tools/delever_pack.py @@ -112,6 +112,25 @@ def build(a): if head: out_n.append(f"--- {defs_at[k][1]} (line {ln + 1}) ---") out_n += list(reversed(head)) + # THE TARGET'S OWN HEADER, IN FULL. neighbours.txt used to carry only the @class/@stuck LINES, and agent b9 + # (S102) lost hours to exactly that: its function's header comment is an eight-point English explanation of + # every lever it carries, including the tail crack stated outright, and the pack had reduced it to two + # tagged lines. A grep for tags is not a substitute for the paragraph they sit in. + if here is not None: + ln0 = ntxt[:defs_at[here][0]].count("\n") + own, j = [], ln0 - 1 + while j >= 0 and len(own) < 80 and (nlines[j].lstrip().startswith("//") + or nlines[j].lstrip().startswith("*") + or nlines[j].lstrip().startswith("/*") + or nlines[j].strip() == ""): + if nlines[j].strip() == "" and own: + break + if nlines[j].strip(): + own.append(nlines[j]) + j -= 1 + if own: + out_n = [f"=== THIS FUNCTION'S OWN HEADER ({fn}, line {ln0 + 1}) — read it in full ==="] \ + + list(reversed(own)) + [""] + out_n tagged = [l for l in nlines if "@class:" in l or "@stuck:" in l or "@crack:" in l] if tagged: out_n.append("--- every @class/@stuck/@crack note in this translation unit ---") @@ -119,6 +138,15 @@ def build(a): (d / "neighbours.txt").write_text("\n".join(out_n[:400]) + "\n") except OSError: pass + # the BEST CANDIDATE'S TEXT, not just its move path: history.txt's `@NNNN` line numbers are relative to the + # EVOLVING text, so hand-reconstructing a path lands somewhere else (agent b9 reconstructed 51 where the engine's + # own generators reproduce 16 in one round). + try: + bp = ds.RUN / "bodies" / f"{e['alias']}__{fn}.c" + if bp.exists(): + (d / "best_body.c").write_text(bp.read_text(errors="surrogateescape"), errors="surrogateescape") + except Exception: + pass (d / "history.txt").write_text("\n".join(hist) + "\n") order.append(f"{k}\t{fn}\t{e['alias']}\t{e['copies']}\t{e['best']}\t{e['needed']}\t{','.join(e['kinds'])}\t{','.join(e['regs'])}\t{tu}") print(f" {k:3d} {fn} {e['copies']:4d} copies best {e['best']} -> {d.relative_to(REPO)}", flush=True)