diff --git a/.run/P36/agents/ov_SC04_011__func_801833D4/body.c b/.run/P36/agents/ov_SC04_011__func_801833D4/body.c new file mode 100644 index 0000000000..3969820d39 --- /dev/null +++ b/.run/P36/agents/ov_SC04_011__func_801833D4/body.c @@ -0,0 +1,65 @@ +s32 impl_801833D4(s32 a0, s32 a1) +{ + /* $a1 is call-saved across func_801852BC/func_801853D0/func_80185648 until its one use + far below; gcc puts its param->hardreg move in the branch's delay slot rather than up + front. The copy is a STATEMENT after the first load, not the declaration's initialiser: + an initialiser sits right after the parameter copy, cse retargets that copy + (cse.c:7454-7480) and sched1 never moves it (sched.c:3188-3205); after a real insn, + combine folds the parameter copy into it and it is scheduled like any other insn + (P36 S104 e11, the $19 pin removed). */ + s32 r1; + u16 s0; + s32 obj; + s32 s4; + /* v[2..4] is a 3-word (vel-like x/y/z) struct passed by address to func_800484EC; the + real local apparently has 2 leading words of other data ahead of it in the frame (its + address-taken struct forces gcc to reserve stack starting 2 words earlier than our x + field) -- v[0]/v[1] are unused padding needed only to reproduce the frame layout. */ + s32 v[5]; + + s0 = D_801EFD20; + s4 = 0; + r1 = a1; + obj = aFC4C[s0]; + + if (D_801EFD40 & 0x4) { + func_801852BC(s0); + func_801853D0(s0); + func_80185648(s0); + + v[3] = 0; + v[2] = 0; + if (D_801EFD40 & 0x20) { + v[4] = 0xFFD80000; + } else { + v[4] = 0xFFEC0000; + } + + func_800484EC(*(s32 *)(obj + 0x20) + 0x34, (s32)&v[2], (s32)&v[2]); + + *(s32 *)(obj + 0x4) += v[2]; + *(s32 *)(obj + 0x8) += v[3]; + *(s32 *)(obj + 0xC) += v[4]; + + *(u16 *)(*(s32 *)(obj + 0x20) + 0x14) = + *(u16 *)(*(s32 *)(obj + 0x20) + 0x14) + r1; + func_80183BF0(a0); + + *(u16 *)(*(s32 *)(obj + 0x20) + 0x10) = + *(u16 *)(a0 + 0xFE) + *(u16 *)(a0 + 0x102); + + if (func_801836D4((void *)a0, (void *)obj) != 0) { + s4 = 1; + } else { + if (D_801EFD40 & 0x20) { + s0 = 0x40; + } else { + s0 = 0x20; + } + func_80185960(*(s16 *)(a0 + 0x106), (u16 *)(*(s32 *)(obj + 0x20) + 0x12), s0); + func_80185960(*(s16 *)(a0 + 0x100), (u16 *)(a0 + 0xFE), s0); + } + } + + return s4; +} diff --git a/.run/P36/agents/ov_SC04_011__func_801833D4/mechanism.md b/.run/P36/agents/ov_SC04_011__func_801833D4/mechanism.md new file mode 100644 index 0000000000..67962a20b3 --- /dev/null +++ b/.run/P36/agents/ov_SC04_011__func_801833D4/mechanism.md @@ -0,0 +1,62 @@ +# func_801833D4 (impl_801833D4, ov_SC04_011_jr_8017D494.c): mechanism (P36 T7 S104, agent e11) + +**Result: score 0 in plain C.** No pin, no asm, no added volatile. Levers go from 1 to 0 (the `$19` pin on `r1`). +Signature unchanged. No copy of this class in another TU (the 0xFFD80000 / `obj + 0x20) + 0x12), s0)` greps find +unrelated bodies). + +## (a) The residual in one sentence +Same count, same registers. Two moves of the first block are swapped. The target does `sw s4; move s4,zero` early +and fills the `beqz` delay slot with `move s3,a1`. Mine does `sw s3; move s3,a1` early and puts `move s4,zero` in +the slot. + +## (b) The pass and the decision (proven on dumps and bytes) +- **The initialiser `s32 r1 = a1;` sits directly after the parameter copy** `(set 74 $5)`. cse's + `(set REG0 REG1)` swap (`cse.c:7454-7480`: `NEXT_INSN (PREV_INSN (insn)) == insn`, the previous insn sets REG1, + REG0 is the canonical reg) retargets the parameter copy to set `r1` directly. The `.cse` dump of + `scratch/v_first.c` shows `(insn 6 (set (reg/v 74) (reg 5 a1)))`. +- **sched1 never moves a parameter copy.** Before reload, block 0's leading SETs before `NOTE_INSN_FUNCTION_BEG` + are removed from scheduling (`sched.c:3188-3205`: "don't delay getting parameters from hard registers into + pseudo registers"), so the copy keeps the lowest LUID. In sched2 all first-block insns have priority 1, and + `rank_for_schedule`'s last tie-break is `INSN_LUID` (`sched.c:2425-2428`). The `s4 = 0` insn (higher LUID) is + therefore scheduled closer to the branch, and reorg's backward scan puts it in the delay slot. +- **Written as a statement after a real insn** (`s0 = D_801EFD20;` first), the copy `r1 = ` stays a + separate insn at its own position. Combine then folds the dying parameter pseudo into it: in the `.lreg` of + `scratch/v_before_s4.c`, reg 73 is gone and `(insn 16 (set (reg/v 74) (reg 5 a1)) REG_DEAD a1)` sits after the + `D_801EFD20` load. sched1 now schedules it with `LAUNCH_PRIORITY` (`sched.c:187`, the 0x7f000001 in the trace), + ahead of `s4 = 0` (priority 1). That places it after `s4 = 0`, sched2's LUID tie-break keeps the order, and reorg + takes `move s3,a1` into the delay slot, which is the target. + +## (c) The move that closed it +`s32 r1 = a1;` becomes `s32 r1;` plus the statement `r1 = a1;` after `s4 = 0;`. Every position after the first +real insn scores 0: after `s0 = …` (`v_before_s4`), after `s4 = 0` (`body.c`), after `obj = …` (`v_after_obj`), +and as the declaration list `u16 s0 = D_801EFD20; s32 r1 = a1;` (`v_declinit`). The body comment that explained +the pin now explains this. + +## (d) Generator proposal +When a register residual is two first-block moves swapped, and one of them is a parameter copy (`move sN,aK`, which +the target has later or in a delay slot), turn `T x = aK;` at the top of the body into a plain `x = aK;` statement +placed after the first real statement (and try each later position). This takes the copy out of cse's +adjacent-producer swap and sched1's parameter-copy exclusion. + +## (e) What did NOT work (byte evidence) +- `r1 = a1;` as the FIRST statement (`v_first.c`): score 5, the same residual. It is still adjacent to the + parameter copy, so cse still swaps it. +- Deleting `r1` and using `a1` directly (`v_noR1.c`): score 5. The parameter copy is the only copy and stays + unscheduled. +- `s4 = 0;` moved first (`v_s4_first.c`): score 5. +- `r1 = a1;` inside the `if` (`v_in_if.c`): score 8. The copy lands in block 1, which is too late. +- The sweep's R6 (inline r1) and R9 (swap statements) stayed at 5. No generator splits an initialiser off its + declaration. + +## (f) Where the method fell short +The residual's "register pairs zero->a1, s4->s3" line reads like an allocation defect, but it was purely order +(ORDER class, same registers). The sched1 trace (the 0x7f000001 LAUNCH_PRIORITY on the copy) and sched.c:3188 gave +the reason. The allocation table would have been a detour. + +## (g) Structs question +No. The lever was about where a register copy sits (cse adjacency, sched1's parameter-copy exclusion), not about +memory. A struct for `obj`/`a0` would not change it. The `v[5]` padding array could plausibly become a real struct +local (2 leading words + an x/y/z vector) in the structs phase, but it is not part of this lever. + +Files: `body.c` (score 0), `scratch/v_*.c` (variants above), `scratch/dumps_{free,first,before_s4}/`, +`scratch/splice.py`. diff --git a/.run/P36/agents/ov_SC04_011__func_80183E20/body.c b/.run/P36/agents/ov_SC04_011__func_80183E20/body.c new file mode 100644 index 0000000000..957e823453 --- /dev/null +++ b/.run/P36/agents/ov_SC04_011__func_80183E20/body.c @@ -0,0 +1,27 @@ +void aF80183E20(void *a0) { + extern u16 D_801EFD40; + extern u16 aEFD24 __asm__("D_801EFD24"); + u16 flags; + u16 cnt; + s32 v0; + s32 v1; + flags = D_801EFD40; + if (flags & 0x80) { + v0 = *(u16 *)((u8 *)a0 + 0x76); + v1 = *(u16 *)((u8 *)a0 + 0x60); + *(u16 *)((u8 *)a0 + 0x60) = 0; + *(u16 *)((u8 *)a0 + 0x76) = v0 - v1; + D_801EFD40 = flags & 0xFF7F; + if (((s16 *)a0)[0x3B] < 0) { + *(u16 *)((u8 *)a0 + 0x76) = 0; + } + } + cnt = aEFD24; + if (cnt != 0) { + cnt = cnt - 1; + aEFD24 = cnt; + if (cnt == 0) { + func_80186AB8(); + } + } +} diff --git a/.run/P36/agents/ov_SC04_011__func_80183E20/mechanism.md b/.run/P36/agents/ov_SC04_011__func_80183E20/mechanism.md new file mode 100644 index 0000000000..96095b8107 --- /dev/null +++ b/.run/P36/agents/ov_SC04_011__func_80183E20/mechanism.md @@ -0,0 +1,59 @@ +# func_80183E20 (ov_SC04_011_jr_8017D494.c): mechanism (P36 T7 S104, agent e11) + +**Result: score 0 in plain C, on the first `--try`.** No pin, no asm, no added volatile. Levers go from 1 to 0 (the +`__asm__("":::"memory")` barrier at tree line 7408). Signature unchanged. No other copy of the class in `src/` (grep of +`a0 + 0x60) = 0;` with a 0x76/0xFF7F neighbour finds only this one). + +## (a) The residual in one sentence +Same count (34/34). The target RELOADS the field it has just stored (`sh v0,118(a0); lh v1,118(a0)`). Mine has +`sll v0,v0,16` instead: cse forwarded the stored value into the `(s16)` test, so no load is left. + +## (b) The pass and the decision (proven on bytes) +The original stored the global flag BEFORE it read the field back. Two passes are involved: +- **cse forgets the field at the scalar store.** A store to the fixed address `D_801EFD40` goes through + `note_mem_written` (`cse.c:7539-7578`). Its address does not vary, so it only sets `writes.var`. + `invalidate_memory` (`cse.c:1701-1720`) then removes every in-memory table entry whose address DOES vary + (`cse_rtx_addr_varies_p`, `cse.c:2474`). `a0+0x76` varies, so its entry is removed. The load after the flag + store now finds nothing to forward, and the `lh` survives. +- **sched1 hoists the load back above the flag store.** `true_dependence` (`sched.c:817-839`) drops the + store→load dependence when the load is `MEM_IN_STRUCT_P` with a varying address and the store is a scalar with a + fixed address (the second `! (...)` clause). expand_expr sets `MEM_IN_STRUCT_P` on an INDIRECT_REF whose operand + is a PLUS_EXPR (`expr.c:4569-4577`). `((s16 *)a0)[0x3B]` gives exactly that. `*(s16 *)((u8 *)a0 + 0x76)` is a + NOP_EXPR around the PLUS, so it is not marked, and the load stays below the store. + +## (c) The move that closed it +```c + *(u16 *)((u8 *)a0 + 0x76) = v0 - v1; + D_801EFD40 = flags & 0xFF7F; /* flag store moved BEFORE the reload */ + if (((s16 *)a0)[0x3B] < 0) { /* array-indexed reload: in-struct MEM */ + *(u16 *)((u8 *)a0 + 0x76) = 0; + } +``` +The `v1b` temp is gone, and so is the `v0 = v0 - v1;` self-update (it is now stored directly). Both moves are needed: +- cast-style reload after the flag store (`scratch/v_cast.c`): score 5, 35 ins. The `lh` is kept but sched cannot + lift it above `sh D_801EFD40`. +- indexed reload read into a temp BEFORE the flag store (`scratch/v_before.c`): score 5. This is the free body's + residual exactly: cse forwards. + +## (d) Generator proposal +When the target has `sw/sh X,K(rA); l[hw] rB,K(rA)` (a store immediately reloaded) and mine forwards the value +(a missing load, often an extra `sll`/`sra` or `move`), look for a later store to a GLOBAL scalar in the same block. +Move that store textually between the field store and the field read, and respell the read as an array index +`((T *)p)[K/sizeof(T)]` (or a struct field). This replaces the `"memory"` barrier lever. + +## (e) What did NOT work +Recorded under (c). The sweep's best was 2 (R7 do-while + R9 swap). Neither move reaches this, because no generator +changes a cast deref into an index expression. + +## (f) Where the method fell short +Nothing blocked. What found it was the S103 c11 note in METHOD step 3 ("`p[i]` is an aggregate access that +`true_dependence` treats as independent of a scalar store; a cast-wrapped byte-offset read is not marked") plus +reading the target order (`lh` BEFORE the flag's `andi`/`sh`). No dumps were needed. + +## (g) Structs question +Yes, and here it is the whole answer. A struct type for `a0` (`struct { … u16 f60 @0x60; … s16 f76 @0x76; }`, with +`a0->f76` as a COMPONENT_REF) makes the reload an in-struct MEM: the COMPONENT_REF case sets `MEM_IN_STRUCT_P` +(`expr.c:4873/4888`). The array index sets the same flag through `expr.c:4569`, so it is the same channel. Not tested with a struct declaration. The array index already proves the channel +on bytes, and a struct field would take the same `true_dependence` clause. + +Files: `body.c` (score 0), `scratch/v_cast.c`, `scratch/v_before.c` (both score 5, the two halves of the proof). diff --git a/.run/P36/agents/ov_SC04_011__func_80185214/body.c b/.run/P36/agents/ov_SC04_011__func_80185214/body.c new file mode 100644 index 0000000000..ff6a5f683a --- /dev/null +++ b/.run/P36/agents/ov_SC04_011__func_80185214/body.c @@ -0,0 +1,20 @@ +void func_80185214(s32 a0) +{ + extern s32 D_801EFC48; + extern u8 D_80194724[]; + extern void func_80185A18(s32 idx, s32 val); + s32 p; + s32 i; + s32 c; + + p = *(s32 *)(D_801EFC48 + 0xCC); + for (i = 0; i < 5; i++) { + c = *(u8 *)(p + i * 8 + 6); + if (c < 0x15 && D_80194724[c] == 0) { + func_80185A18(c, (s16)(*(u16 *)(p + i * 8 + 2) + a0)); + } + if (*(s16 *)(p + i * 8 + 6) & 0x8000) { + break; + } + } +} diff --git a/.run/P36/agents/ov_SC04_011__func_80185214/mechanism.md b/.run/P36/agents/ov_SC04_011__func_80185214/mechanism.md new file mode 100644 index 0000000000..36758e1b4c --- /dev/null +++ b/.run/P36/agents/ov_SC04_011__func_80185214/mechanism.md @@ -0,0 +1,69 @@ +# func_80185214 (ov_SC04_011_jr_8017D494.c): mechanism (P36 T7 S104, agent e11) + +**Result: score 0 in plain C.** No pin, no asm, no added volatile, no goto. Levers go from 1 to 0 (the `$16` pin on +`s0`). Signature unchanged. No copy of this class elsewhere: `c = *(u8 *)(s0 + 6);` occurs only here. The +same-name functions in ov_SC02_005 and ov_SC06_029 are different bodies, and the similar walk in `func_80184CCC` +(same TU) is already lever-free. + +## (a) The residual in one sentence +One extra instruction. Mine walks a pointer to the entry's `+6` field (`addiu s0,v0,6` and `addiu s1,v0,46`, then +fields at `0(s0)`/`-4(s0)`). The target walks the entry base (`lw s0,204(v0)` straight into s0, `addiu s1,s0,40`, +fields at `6(s0)`/`2(s0)`). + +## (b) The pass and the decision (proven on the `-dL` dumps and bytes) +loop.c strength reduction. In the free body `s0` is a pointer biv, and the three field ADDRESSES are its givs: +`Insn 26/42/59: dest address src reg 73 … mult 1 add 6 / add 2 / add 6` (`record_giv`, `loop.c:4341`, dump text +`:4508`). `combine_givs` (`loop.c:5494`, `:5527`) merges them onto the add-6 giv ("giv at 42 combined with giv at +59", "giv at 26 combined with giv at 59"). That giv is reduced to a new register holding `s0+6`, and "biv 73 was +eliminated", so the exit test is rewritten against `s1+6`. The pin hid `s0` from loop.c (a hard register is never +a biv), which is why the lever "worked". + +## (c) The move that closed it +Index the entries by a counter instead of walking a pointer (the S103 c2 / S104 d16/d18 family): +```c + p = *(s32 *)(D_801EFC48 + 0xCC); + for (i = 0; i < 5; i++) { + c = *(u8 *)(p + i * 8 + 6); + if (c < 0x15 && D_80194724[c] == 0) { + func_80185A18(c, (s16)(*(u16 *)(p + i * 8 + 2) + a0)); + } + if (*(s16 *)(p + i * 8 + 6) & 0x8000) { + break; + } + } +``` +Now the givs are the SUMS `p + i*8` (`Insn 31/78: giv reg … mult 8 add (reg 73)`, a DEST_REG giv with +`add_val = p`). The reduced register holds the entry base itself (initial value `p`), and the field offsets stay in +the addressing mode (`6(s0)`, `2(s0)`). Biv `i` is eliminated against `p + 40`, which gives `addiu s1,s0,40` and +the signed `slt`. `s2 = a0` is gone (`a0` is used directly). Also proven: +- `scratch/v_do_i.c` (the same loop as a do-while with `i++`): 0. +- `scratch/v_goto.c` (the ORIGINAL pointer walk as a goto loop, no LOOP notes, so loop.c never runs): 0. +- `scratch/v_struct.c` (a body-local `struct Ent { u16 f0, f2, f4; union { u8 id; s16 flags; } f6; } *e;` indexed + as `e[i]`): 0. + +## (d) Generator proposal +When a loop's residual shows a walked pointer whose reduced register sits at a field offset (fields read at +`0(sN)`/negative offsets, the end pointer `base+K+off`), and the target reads the same fields at `off(sN)` with +`sN` = the loaded base: rewrite `for/do (p…; p < end; p += S)` as `for (i = 0; i < (end-base)/S; i++)` with every +`*(T *)(p + K)` becoming `*(T *)(base + i * S + K)`. The DEST_REG giv `base + i*S` then becomes the reduced register. + +## (e) What did NOT work (byte evidence) +- `for (s1 = s0 + 0x28; s0 < s1; s0 += 8)` (`scratch/v_for_ptr.c`): 11. It keeps the entry test and the + offset giv. +- The sweep (R4 declaration moves, R6/R10 parameter moves, R7/R8/R9) stayed at 6. No generator turns a pointer walk + into an indexed loop. + +## (f) Where the method fell short +METHOD step 3 (S103 c2) and step 14 (d16/d18) already name this family. The `-dL` dump's "giv … combined … reduced +to" lines settle it in one compile. The residual's "register pairs v0->s0" line pointed at allocation, but the +defect was the +6 bias of the reduced register. + +## (g) Structs question +Yes, plausibly for readability, but it is not needed for the bytes. `D_801EFC48 + 0xCC` points to an array of +8-byte entries (`+2` u16 value, `+6` u8 id whose halfword also carries a 0x8000 end flag). With a struct array +indexed `e[i]` (`scratch/v_struct.c`) the loop also scores 0, because the channel is the loop's giv shape (an +indexed sum vs a walked pointer), not MEM_IN_STRUCT_P. A struct POINTER walked with `e++` would presumably +reproduce the free body's defect. Not tested. + +Files: `body.c` (score 0), `scratch/v_{for_i,do_i,goto,struct}.c` (0), `scratch/v_for_ptr.c` (11), +`scratch/dumps_{free,for_i}/` (`.loop` dumps). diff --git a/.run/P36/agents/ov_SC04_011__func_80186020/body.c b/.run/P36/agents/ov_SC04_011__func_80186020/body.c new file mode 100644 index 0000000000..f476e9806d --- /dev/null +++ b/.run/P36/agents/ov_SC04_011__func_80186020/body.c @@ -0,0 +1,45 @@ +void func_80186020(void *a0) { + s32 p; + + if ((D_801EFD40 & 2) != 0) { + s32 a; + s32 b; + + a = (s16)D_801EFD48; + if (a >= 0x120) { + *(s16 *)&D_801EFD4C = -0x60; + } else if (a < -0x5F) { + *(s16 *)&D_801EFD4C = 0x60; + } + a = D_801EFD48; + b = D_801EFD4C; + p = *(s32 *)((s32)a0 + 0x20); + a = a + b; + b = D_801EFD44; + D_801EFD48 = a; + b = b + a; + *(u16 *)(p + 0x10) = b; + } else { + u16 x; + u16 y; + + if ((D_801EFD40 & 1) != 0) { + x = D_801EFD48; + if ((s16)x >= 0x120) { + return; + } + y = x + 0x60; + } else { + x = D_801EFD48; + if ((s16)x < -0x5F) { + return; + } + y = x - 0x60; + } + x = D_801EFD44; + p = *(s32 *)((s32)a0 + 0x20); + D_801EFD48 = y; + x = x + y; + *(u16 *)(p + 0x10) = x; + } +} diff --git a/.run/P36/agents/ov_SC04_011__func_80186020/mechanism.md b/.run/P36/agents/ov_SC04_011__func_80186020/mechanism.md new file mode 100644 index 0000000000..b67bd71a7b --- /dev/null +++ b/.run/P36/agents/ov_SC04_011__func_80186020/mechanism.md @@ -0,0 +1,106 @@ +# func_80186020 (ov_SC04_011_jr_8017D494.c): mechanism (P36 T7 S104, agent e11) + +**Result: score 0 in plain C.** No pin, no asm, no added volatile. Levers go from 1 to 0 (the `$2` pin on `y`). +Signature unchanged. No copy of this class elsewhere (greps of the `-0x60`/`0x120` pair and of the pin find only +this body). + +## (a) The residual in one sentence +Same count (60/60). Every value the target keeps in `v0` is in `v1` in mine and the other way round, in BOTH arms. +The free body's shared `u16 x, y` get x→v0 and y→v1; the target has x→v1 and y→v0. + +## (b) The pass and the decision (proven on the `.lreg`/`.greg` dumps and bytes) +global.c allocates in priority order (`allocno_compare`, `global.c:594-610`). `find_reg` gives each allocno the +first register that is free of conflicts and preferences (`global.c:945-990`). In the free body: +- In arm 1, `x = x + y` is HImode arithmetic. It expands into an SI temp (r91) and a HI copy into `x`. cse then + routes the two later readers (the `D_801EFD48` store and `y + x`) through r91, so the copy into `x` dies and + r91 is a LOCAL. local-alloc allocates it before global runs and gives it `v0`, its first choice. +- Global then sees `y` (reloaded with `D_801EFD44` while r91 is live) as conflicting with `v0`. `x`, the + highest-priority allocno (26666 vs y's 21428), has no `v0` conflict and a copy preference for r91's `v0`, so it + takes `v0`. `y` falls to `v1`. The pin forced `y` into `v0`. + +What the target's allocation needs (read off the working body's `.greg`: order `x a b y`, dispositions +x→3, a→3, b→2, y→2): +- The two `v1` roles (arm 1's D48 accumulator, arm 2's D48/D44 value) must CONFLICT with `v0`. That happens when + the variable is live while a `v0` compare temp (`slti v0,…`) is live, so `find_reg` skips `v0` for them. +- The two `v0` roles must be global and must not overlap any local that sits in `v0`. +- Arm 1's sum must not be a separate local temp. + +## (c) The moves that closed it (joint; each alone scores worse) +```c + if ((D_801EFD40 & 2) != 0) { + s32 a; /* arm-1 locals, SImode */ + s32 b; + a = (s16)D_801EFD48; /* the test operand hoisted into `a` */ + if (a >= 0x120) { … } else if (a < -0x5F) { … } + a = D_801EFD48; + b = D_801EFD4C; p = …; a = a + b; b = D_801EFD44; D_801EFD48 = a; b = b + a; … = b; + } else { + u16 x; /* arm-2 locals, HImode */ + u16 y; + if (D_801EFD40 & 1) { x = D_801EFD48; if ((s16)x >= 0x120) return; y = x + 0x60; } + else { x = D_801EFD48; if ((s16)x < -0x5F) return; y = x - 0x60; } + x = D_801EFD44; p = …; D_801EFD48 = y; x = x + y; … = x; + } +``` +1. **Split the shared variables per arm** (S104 d7/d22: one variable per case), and give each arm the width its + arithmetic proves. Arm 1 as `s32` makes `a = a + b` a plain SImode add into `a`, so no HI→SI temp and no local + r91. Arm 2 stays `u16`: its `move v1,v0` (the HI copy of the `lh`) and the 16-byte frame (the dead + sign-extension intermediate r96/r100 with class `ST_REGS or none` that reload spills; see `.lreg`) are + HImode-only artefacts. With `s32` in arm 2 both vanish (`scratch/c3/w_s32_s32.c`, 56 ins). +2. **Hoist the arm-1 test operand into `a`** (`a = (s16)D_801EFD48; if (a >= 0x120) … else if (a < -0x5F)`). + `a` is now set in one block and read in the else-if block, so it is block-global + (`local-alloc.c:469-476` requires a single block and `reg_n_deaths == 1`). It is also live across the + `slti v0` temps, so it conflicts with `v0` and takes `v1`. The target shows the same thing: the `lh v1,D48` of + the test and the later `lhu v1,D48` share `v1`. +3. **Arm 2 reads `D_801EFD48` into `x` in each sub-block** (`x = D_801EFD48; if ((s16)x …) …; y = x ± 0x60;`). + `x` now spans three blocks, and its HI copy is born while the `lh` result (local `v0`) is still read by + `slti`, so `x` conflicts with `v0` and takes `v1`. `y` takes `v0`. This is what the target's + `lh v0; move v1,v0; slti v0,v0,…` is. + +Proven on bytes: +- `body.c` (block-scoped declarations) and `scratch/c10/g_abb_s32s32_v.c` (function-scope declarations): 0. +- Without move 2 (`scratch/c11/n3_nohoist.c`): 15, because cross-jump merges the arm tails once the registers + line up. +- Without move 3 (`n1_arm2free.c`): 14. +- `a` as `s16` (`n4_a_s16.c`): 22. +- Without the `*(s16 *)&D_801EFD4C` store casts (`n5_nocast.c`): 1. + +## (d) Generator proposal +When a register residual swaps `v0`/`v1` (or any pair) between variables SHARED by two if/else arms, first split +them per arm (one variable per role per arm, at the widths the instructions prove: `s32` where the target adds in +a register with no HI copy, `u16` where it keeps a `move` of an `lh`). Then, for each variable the target puts in +the register that also holds an earlier TEST operand (`lh vN,G; slti …,vN,K` followed by a later reload of `G` +into `vN`), hoist the test operand into that variable (`v = (s16)G; if (v >= K)`). This makes it block-global and +makes it conflict with the compare temp's register. + +## (e) What did NOT work (byte evidence, scratch/c*/ with their scores) +- The free body: 22. The regen sweep's 122 candidates: best 10 (`scratch/regen_scores.txt`). The history's "7" path + could not be reproduced from its label. +- Renaming or splitting per arm with `u16` everywhere (`c/`, `c2/`): 10-24. Arm 1's cse temp r91 always takes `v0`. +- Widths alone (`c3/`, 16 combinations): best 15. `s32` in arm 2 deletes the HI `move` and the frame. +- Arm 1 `s32` + arm 2 `u16` `y2` with a shared `s32 x` (`c4/t_s32s32_x_s32u16.c`): 4. The `x + y2` add + zero-extends (`andi 0xffff`). Folding the add into the store removes the `andi` but lets cross-jump merge the + tails (`c9/`: 7). +- A single-set D44 temp to get the scheduler's birthing boost (`c8/`): 10. The D44 load is then hoisted above the + sum (`sched.c:2477-2490` boosts it, `sched.c:2425-2428` breaks the tie), so it overlaps `y`. + +## (f) Where the method fell short +- The allocation table (`tools/alloc_table.py`) was the deciding read. It showed r91 as a LOCAL in `v0` and + `y` "conflicts v0". What it cannot show is WHICH local causes a conflict. Grepping `.lreg` for the local's + block and life settled it. +- The winning move 2 (hoisting a test operand into an existing variable, so that it conflicts with the compare + temp) is not in METHOD. It is the inverse of S103 c18's "same register, same role → one variable": here the + target's `lh v1,G … lhu v1,G` said "same register, same variable". A block-by-block read of the target for + "which values share a register" found it. The hunk view hid it. +- Enumeration was effective: about 300 bodies in 8 batches through `--try`. Each batch was shaped by the previous + batch's residual. + +## (g) Structs question +No. The lever came from register-allocation conflicts between user variables (local-alloc/global), with no memory +ordering involved. The globals `D_801EFD44/48/4C` are three separate scalars (x-offset, position, velocity-like). +Making them fields of one struct would put `MEM_IN_STRUCT_P` on the accesses. Nothing here depends on alias +decisions, although cse's store-forwarding into `x = x + y` (`cse.c` invalidation of varying-address entries, +`cse.c:1701-1720`) would be worth re-checking in the structs phase. Not tested. + +Files: `body.c` (score 0), `scratch/c{,2..11}/` (candidate batches), `scratch/tryall.sh`, `scratch/splice.py`, +`scratch/dumps_{free,i1,m_xyc_u,k1_u16,match}/`. diff --git a/.run/P36/delever/calibration.json b/.run/P36/delever/calibration.json index 24c750db3c..d09f92fc58 100644 --- a/.run/P36/delever/calibration.json +++ b/.run/P36/delever/calibration.json @@ -1,7 +1,7 @@ { - "head": "580d7bfd8", + "head": "860ed9153", "stamp": "15956e4a96c4", - "generated": "2026-09-11 02:05", + "generated": "2026-09-11 02:14", "aliases": [ "main", "ov_SC03_014", @@ -21,207 +21,207 @@ "main": { "objects": 85, "identical": 85, - "seconds": 6.937999999999999, - "mean_s": 0.082 + "seconds": 6.653, + "mean_s": 0.078 }, "ov_SC03_014": { "objects": 32, "identical": 32, - "seconds": 4.733, - "mean_s": 0.148 + "seconds": 4.586, + "mean_s": 0.143 }, "ov_SC03_015": { "objects": 32, "identical": 32, - "seconds": 4.500000000000001, - "mean_s": 0.141 + "seconds": 4.437, + "mean_s": 0.139 }, "ov_SC04_011": { "objects": 28, "identical": 28, - "seconds": 3.8809999999999993, - "mean_s": 0.139 + "seconds": 3.7390000000000008, + "mean_s": 0.134 } }, "per_object_seconds": { - "build/src/800.o": 0.742, - "build/src/800_b.o": 0.077, - "build/src/800_b_2.o": 0.303, - "build/src/800_b_o0a.o": 0.077, - "build/src/800_c.o": 0.203, - "build/src/800b2.o": 0.083, - "build/src/apicard1.o": 0.08, - "build/src/apicard2.o": 0.072, - "build/src/apicard3.o": 0.079, - "build/src/apicard4.o": 0.067, - "build/src/apicard5.o": 0.087, - "build/src/apicard6.o": 0.076, - "build/src/apicard7.o": 0.082, - "build/src/boot.o": 0.117, - "build/src/gap.o": 0.07, - "build/src/libapi1.o": 0.071, - "build/src/libapi2.o": 0.061, - "build/src/libc2_1.o": 0.064, - "build/src/libc2_2.o": 0.059, - "build/src/libcd1.o": 0.066, - "build/src/libcd2.o": 0.064, - "build/src/libetc.o": 0.058, - "build/src/libgpu.o": 0.07, - "build/src/libgpu2.o": 0.07, - "build/src/libgs1.o": 0.081, - "build/src/libgs2.o": 0.048, - "build/src/libgs3.o": 0.066, - "build/src/libgs4.o": 0.066, - "build/src/libgs5.o": 0.06, - "build/src/libgs6.o": 0.077, - "build/src/libgs7.o": 0.059, - "build/src/libgs8.o": 0.068, - "build/src/libgte1.o": 0.06, - "build/src/libgte10.o": 0.058, - "build/src/libgte11.o": 0.061, - "build/src/libgte12.o": 0.064, + "build/src/800.o": 0.712, + "build/src/800_b.o": 0.087, + "build/src/800_b_2.o": 0.276, + "build/src/800_b_o0a.o": 0.062, + "build/src/800_c.o": 0.197, + "build/src/800b2.o": 0.072, + "build/src/apicard1.o": 0.072, + "build/src/apicard2.o": 0.071, + "build/src/apicard3.o": 0.068, + "build/src/apicard4.o": 0.062, + "build/src/apicard5.o": 0.078, + "build/src/apicard6.o": 0.066, + "build/src/apicard7.o": 0.079, + "build/src/boot.o": 0.086, + "build/src/gap.o": 0.06, + "build/src/libapi1.o": 0.08, + "build/src/libapi2.o": 0.057, + "build/src/libc2_1.o": 0.071, + "build/src/libc2_2.o": 0.067, + "build/src/libcd1.o": 0.078, + "build/src/libcd2.o": 0.061, + "build/src/libetc.o": 0.056, + "build/src/libgpu.o": 0.063, + "build/src/libgpu2.o": 0.067, + "build/src/libgs1.o": 0.055, + "build/src/libgs2.o": 0.079, + "build/src/libgs3.o": 0.063, + "build/src/libgs4.o": 0.05, + "build/src/libgs5.o": 0.061, + "build/src/libgs6.o": 0.072, + "build/src/libgs7.o": 0.075, + "build/src/libgs8.o": 0.062, + "build/src/libgte1.o": 0.059, + "build/src/libgte10.o": 0.061, + "build/src/libgte11.o": 0.057, + "build/src/libgte12.o": 0.067, "build/src/libgte13.o": 0.064, - "build/src/libgte14.o": 0.064, - "build/src/libgte15.o": 0.066, - "build/src/libgte16.o": 0.06, - "build/src/libgte17.o": 0.062, - "build/src/libgte18.o": 0.061, - "build/src/libgte19.o": 0.069, - "build/src/libgte2.o": 0.061, - "build/src/libgte20.o": 0.071, - "build/src/libgte21.o": 0.064, - "build/src/libgte22.o": 0.065, - "build/src/libgte23.o": 0.061, - "build/src/libgte24.o": 0.069, - "build/src/libgte25.o": 0.067, - "build/src/libgte26.o": 0.068, - "build/src/libgte27.o": 0.065, - "build/src/libgte28.o": 0.073, - "build/src/libgte29.o": 0.065, - "build/src/libgte3.o": 0.069, - "build/src/libgte30.o": 0.071, - "build/src/libgte4.o": 0.063, + "build/src/libgte14.o": 0.049, + "build/src/libgte15.o": 0.063, + "build/src/libgte16.o": 0.07, + "build/src/libgte17.o": 0.069, + "build/src/libgte18.o": 0.063, + "build/src/libgte19.o": 0.06, + "build/src/libgte2.o": 0.077, + "build/src/libgte20.o": 0.065, + "build/src/libgte21.o": 0.063, + "build/src/libgte22.o": 0.059, + "build/src/libgte23.o": 0.063, + "build/src/libgte24.o": 0.063, + "build/src/libgte25.o": 0.064, + "build/src/libgte26.o": 0.067, + "build/src/libgte27.o": 0.077, + "build/src/libgte28.o": 0.053, + "build/src/libgte29.o": 0.067, + "build/src/libgte3.o": 0.072, + "build/src/libgte30.o": 0.075, + "build/src/libgte4.o": 0.066, "build/src/libgte5.o": 0.066, - "build/src/libgte6.o": 0.072, - "build/src/libgte7.o": 0.076, - "build/src/libgte8.o": 0.068, - "build/src/libgte9.o": 0.076, - "build/src/libmcrd1.o": 0.079, - "build/src/libmcrd2.o": 0.075, - "build/src/libpad1.o": 0.073, + "build/src/libgte6.o": 0.059, + "build/src/libgte7.o": 0.069, + "build/src/libgte8.o": 0.063, + "build/src/libgte9.o": 0.062, + "build/src/libmcrd1.o": 0.077, + "build/src/libmcrd2.o": 0.063, + "build/src/libpad1.o": 0.068, "build/src/libpad2.o": 0.073, - "build/src/sgap.o": 0.084, - "build/src/sgap_2.o": 0.072, - "build/src/sgap_3.o": 0.067, - "build/src/sgap_4.o": 0.072, - "build/src/sgap_5.o": 0.067, - "build/src/sgap_6.o": 0.063, - "build/src/sgap_8.o": 0.067, - "build/src/snd1.o": 0.077, - "build/src/snd10.o": 0.065, + "build/src/sgap.o": 0.071, + "build/src/sgap_2.o": 0.055, + "build/src/sgap_3.o": 0.071, + "build/src/sgap_4.o": 0.075, + "build/src/sgap_5.o": 0.064, + "build/src/sgap_6.o": 0.059, + "build/src/sgap_8.o": 0.069, + "build/src/snd1.o": 0.068, + "build/src/snd10.o": 0.072, "build/src/snd11.o": 0.061, - "build/src/snd12.o": 0.066, - "build/src/snd2.o": 0.071, - "build/src/snd3.o": 0.07, - "build/src/snd4.o": 0.068, - "build/src/snd5.o": 0.072, - "build/src/snd6.o": 0.081, - "build/src/snd7.o": 0.068, - "build/src/snd8.o": 0.069, - "build/src/snd9.o": 0.076, - "build/src/ov_SC03_014/ov_SC03_014.o": 0.132, - "build/src/ov_SC03_014/ov_SC03_014_after.o": 0.519, - "build/src/ov_SC03_014/ov_SC03_014_jr_8012ACE0.o": 0.406, - "build/src/ov_SC03_014/ov_SC03_014_jr_80135888.o": 0.059, - "build/src/ov_SC03_014/ov_SC03_014_jr_80135A4C.o": 0.063, - "build/src/ov_SC03_014/ov_SC03_014_jr_80135D20.o": 0.121, - "build/src/ov_SC03_014/ov_SC03_014_jr_801380E0.o": 0.162, - "build/src/ov_SC03_014/ov_SC03_014_jr_8013C98C.o": 0.122, - "build/src/ov_SC03_014/ov_SC03_014_jr_8013F350.o": 0.109, - "build/src/ov_SC03_014/ov_SC03_014_jr_8013FFD8.o": 0.078, - "build/src/ov_SC03_014/ov_SC03_014_jr_80140608.o": 0.183, - "build/src/ov_SC03_014/ov_SC03_014_jr_8015444C.o": 0.076, - "build/src/ov_SC03_014/ov_SC03_014_jr_80154C24.o": 0.166, - "build/src/ov_SC03_014/ov_SC03_014_jr_801588CC.o": 0.096, - "build/src/ov_SC03_014/ov_SC03_014_jr_80159C84.o": 0.074, - "build/src/ov_SC03_014/ov_SC03_014_jr_8015A3C8.o": 0.069, - "build/src/ov_SC03_014/ov_SC03_014_jr_8015AE2C.o": 0.102, - "build/src/ov_SC03_014/ov_SC03_014_jr_8015C32C.o": 0.473, - "build/src/ov_SC03_014/ov_SC03_014_jr_8016AB6C.o": 0.255, - "build/src/ov_SC03_014/ov_SC03_014_jr_80171B4C.o": 0.103, - "build/src/ov_SC03_014/ov_SC03_014_jr_801734BC.o": 0.2, - "build/src/ov_SC03_014/ov_SC03_014_jr_801789AC.o": 0.061, - "build/src/ov_SC03_014/ov_SC03_014_jr_80178D40.o": 0.103, - "build/src/ov_SC03_014/ov_SC03_014_jr_8017A4AC.o": 0.073, - "build/src/ov_SC03_014/ov_SC03_014_jr_8017AE2C.o": 0.169, - "build/src/ov_SC03_014/ov_SC03_014_jr_8017EB7C.o": 0.205, - "build/src/ov_SC03_014/ov_SC03_014_jr_80184440.o": 0.052, - "build/src/ov_SC03_014/ov_SC03_014_jr_801848E4.o": 0.268, - "build/src/ov_SC03_014/ov_SC03_014_o0b.o": 0.057, - "build/src/ov_SC03_014/ov_SC03_014_o0c.o": 0.059, - "build/src/ov_SC03_014/ov_SC03_014_o0d.o": 0.051, + "build/src/snd12.o": 0.059, + "build/src/snd2.o": 0.076, + "build/src/snd3.o": 0.074, + "build/src/snd4.o": 0.066, + "build/src/snd5.o": 0.068, + "build/src/snd6.o": 0.073, + "build/src/snd7.o": 0.066, + "build/src/snd8.o": 0.065, + "build/src/snd9.o": 0.063, + "build/src/ov_SC03_014/ov_SC03_014.o": 0.122, + "build/src/ov_SC03_014/ov_SC03_014_after.o": 0.501, + "build/src/ov_SC03_014/ov_SC03_014_jr_8012ACE0.o": 0.393, + "build/src/ov_SC03_014/ov_SC03_014_jr_80135888.o": 0.065, + "build/src/ov_SC03_014/ov_SC03_014_jr_80135A4C.o": 0.061, + "build/src/ov_SC03_014/ov_SC03_014_jr_80135D20.o": 0.124, + "build/src/ov_SC03_014/ov_SC03_014_jr_801380E0.o": 0.156, + "build/src/ov_SC03_014/ov_SC03_014_jr_8013C98C.o": 0.101, + "build/src/ov_SC03_014/ov_SC03_014_jr_8013F350.o": 0.091, + "build/src/ov_SC03_014/ov_SC03_014_jr_8013FFD8.o": 0.099, + "build/src/ov_SC03_014/ov_SC03_014_jr_80140608.o": 0.18, + "build/src/ov_SC03_014/ov_SC03_014_jr_8015444C.o": 0.084, + "build/src/ov_SC03_014/ov_SC03_014_jr_80154C24.o": 0.176, + "build/src/ov_SC03_014/ov_SC03_014_jr_801588CC.o": 0.092, + "build/src/ov_SC03_014/ov_SC03_014_jr_80159C84.o": 0.068, + "build/src/ov_SC03_014/ov_SC03_014_jr_8015A3C8.o": 0.063, + "build/src/ov_SC03_014/ov_SC03_014_jr_8015AE2C.o": 0.082, + "build/src/ov_SC03_014/ov_SC03_014_jr_8015C32C.o": 0.463, + "build/src/ov_SC03_014/ov_SC03_014_jr_8016AB6C.o": 0.258, + "build/src/ov_SC03_014/ov_SC03_014_jr_80171B4C.o": 0.099, + "build/src/ov_SC03_014/ov_SC03_014_jr_801734BC.o": 0.186, + "build/src/ov_SC03_014/ov_SC03_014_jr_801789AC.o": 0.058, + "build/src/ov_SC03_014/ov_SC03_014_jr_80178D40.o": 0.094, + "build/src/ov_SC03_014/ov_SC03_014_jr_8017A4AC.o": 0.064, + "build/src/ov_SC03_014/ov_SC03_014_jr_8017AE2C.o": 0.172, + "build/src/ov_SC03_014/ov_SC03_014_jr_8017EB7C.o": 0.188, + "build/src/ov_SC03_014/ov_SC03_014_jr_80184440.o": 0.05, + "build/src/ov_SC03_014/ov_SC03_014_jr_801848E4.o": 0.253, + "build/src/ov_SC03_014/ov_SC03_014_o0b.o": 0.063, + "build/src/ov_SC03_014/ov_SC03_014_o0c.o": 0.056, + "build/src/ov_SC03_014/ov_SC03_014_o0d.o": 0.057, "build/src/ov_SC03_014/ov_SC03_014_o0e.o": 0.067, - "build/src/ov_SC03_015/ov_SC03_015.o": 0.112, - "build/src/ov_SC03_015/ov_SC03_015_after.o": 0.509, - "build/src/ov_SC03_015/ov_SC03_015_jr_8012ACE0.o": 0.391, - "build/src/ov_SC03_015/ov_SC03_015_jr_80135888.o": 0.054, - "build/src/ov_SC03_015/ov_SC03_015_jr_80135A4C.o": 0.052, - "build/src/ov_SC03_015/ov_SC03_015_jr_80135D20.o": 0.1, - "build/src/ov_SC03_015/ov_SC03_015_jr_801380E0.o": 0.143, - "build/src/ov_SC03_015/ov_SC03_015_jr_8013C98C.o": 0.108, - "build/src/ov_SC03_015/ov_SC03_015_jr_8013F350.o": 0.076, - "build/src/ov_SC03_015/ov_SC03_015_jr_8013FFD8.o": 0.063, - "build/src/ov_SC03_015/ov_SC03_015_jr_80140608.o": 0.182, - "build/src/ov_SC03_015/ov_SC03_015_jr_8015444C.o": 0.076, - "build/src/ov_SC03_015/ov_SC03_015_jr_80154C24.o": 0.158, - "build/src/ov_SC03_015/ov_SC03_015_jr_801588CC.o": 0.091, - "build/src/ov_SC03_015/ov_SC03_015_jr_80159C84.o": 0.07, - "build/src/ov_SC03_015/ov_SC03_015_jr_8015A3C8.o": 0.068, - "build/src/ov_SC03_015/ov_SC03_015_jr_8015AE2C.o": 0.081, - "build/src/ov_SC03_015/ov_SC03_015_jr_8015C32C.o": 0.443, - "build/src/ov_SC03_015/ov_SC03_015_jr_8016AB6C.o": 0.262, - "build/src/ov_SC03_015/ov_SC03_015_jr_80171B4C.o": 0.095, - "build/src/ov_SC03_015/ov_SC03_015_jr_801734BC.o": 0.21, - "build/src/ov_SC03_015/ov_SC03_015_jr_801789AC.o": 0.057, - "build/src/ov_SC03_015/ov_SC03_015_jr_80178D40.o": 0.094, - "build/src/ov_SC03_015/ov_SC03_015_jr_8017A4AC.o": 0.068, - "build/src/ov_SC03_015/ov_SC03_015_jr_8017AE2C.o": 0.168, - "build/src/ov_SC03_015/ov_SC03_015_jr_8017EB7C.o": 0.209, - "build/src/ov_SC03_015/ov_SC03_015_jr_80184440.o": 0.048, + "build/src/ov_SC03_015/ov_SC03_015.o": 0.131, + "build/src/ov_SC03_015/ov_SC03_015_after.o": 0.494, + "build/src/ov_SC03_015/ov_SC03_015_jr_8012ACE0.o": 0.407, + "build/src/ov_SC03_015/ov_SC03_015_jr_80135888.o": 0.046, + "build/src/ov_SC03_015/ov_SC03_015_jr_80135A4C.o": 0.05, + "build/src/ov_SC03_015/ov_SC03_015_jr_80135D20.o": 0.108, + "build/src/ov_SC03_015/ov_SC03_015_jr_801380E0.o": 0.149, + "build/src/ov_SC03_015/ov_SC03_015_jr_8013C98C.o": 0.109, + "build/src/ov_SC03_015/ov_SC03_015_jr_8013F350.o": 0.073, + "build/src/ov_SC03_015/ov_SC03_015_jr_8013FFD8.o": 0.059, + "build/src/ov_SC03_015/ov_SC03_015_jr_80140608.o": 0.174, + "build/src/ov_SC03_015/ov_SC03_015_jr_8015444C.o": 0.072, + "build/src/ov_SC03_015/ov_SC03_015_jr_80154C24.o": 0.159, + "build/src/ov_SC03_015/ov_SC03_015_jr_801588CC.o": 0.077, + "build/src/ov_SC03_015/ov_SC03_015_jr_80159C84.o": 0.059, + "build/src/ov_SC03_015/ov_SC03_015_jr_8015A3C8.o": 0.066, + "build/src/ov_SC03_015/ov_SC03_015_jr_8015AE2C.o": 0.095, + "build/src/ov_SC03_015/ov_SC03_015_jr_8015C32C.o": 0.44, + "build/src/ov_SC03_015/ov_SC03_015_jr_8016AB6C.o": 0.272, + "build/src/ov_SC03_015/ov_SC03_015_jr_80171B4C.o": 0.094, + "build/src/ov_SC03_015/ov_SC03_015_jr_801734BC.o": 0.199, + "build/src/ov_SC03_015/ov_SC03_015_jr_801789AC.o": 0.062, + "build/src/ov_SC03_015/ov_SC03_015_jr_80178D40.o": 0.1, + "build/src/ov_SC03_015/ov_SC03_015_jr_8017A4AC.o": 0.056, + "build/src/ov_SC03_015/ov_SC03_015_jr_8017AE2C.o": 0.16, + "build/src/ov_SC03_015/ov_SC03_015_jr_8017EB7C.o": 0.161, + "build/src/ov_SC03_015/ov_SC03_015_jr_80184440.o": 0.053, "build/src/ov_SC03_015/ov_SC03_015_jr_801848E4.o": 0.27, - "build/src/ov_SC03_015/ov_SC03_015_o0b.o": 0.062, - "build/src/ov_SC03_015/ov_SC03_015_o0c.o": 0.058, - "build/src/ov_SC03_015/ov_SC03_015_o0d.o": 0.058, - "build/src/ov_SC03_015/ov_SC03_015_o0e.o": 0.064, - "build/src/ov_SC04_011/ov_SC04_011.o": 0.115, - "build/src/ov_SC04_011/ov_SC04_011_after.o": 0.419, - "build/src/ov_SC04_011/ov_SC04_011_jr_8012ACE0.o": 0.346, - "build/src/ov_SC04_011/ov_SC04_011_jr_80135888.o": 0.047, - "build/src/ov_SC04_011/ov_SC04_011_jr_80135A4C.o": 0.056, - "build/src/ov_SC04_011/ov_SC04_011_jr_80135D20.o": 0.114, - "build/src/ov_SC04_011/ov_SC04_011_jr_801380E0.o": 0.147, - "build/src/ov_SC04_011/ov_SC04_011_jr_8013C98C.o": 0.107, - "build/src/ov_SC04_011/ov_SC04_011_jr_8013F350.o": 0.076, - "build/src/ov_SC04_011/ov_SC04_011_jr_8013FFD8.o": 0.063, - "build/src/ov_SC04_011/ov_SC04_011_jr_80140608.o": 0.172, - "build/src/ov_SC04_011/ov_SC04_011_jr_8015444C.o": 0.071, - "build/src/ov_SC04_011/ov_SC04_011_jr_80154C24.o": 0.17, - "build/src/ov_SC04_011/ov_SC04_011_jr_801588CC.o": 0.092, - "build/src/ov_SC04_011/ov_SC04_011_jr_80159C84.o": 0.066, - "build/src/ov_SC04_011/ov_SC04_011_jr_8015A3C8.o": 0.07, - "build/src/ov_SC04_011/ov_SC04_011_jr_8015AE2C.o": 0.092, - "build/src/ov_SC04_011/ov_SC04_011_jr_8015C32C.o": 0.34, - "build/src/ov_SC04_011/ov_SC04_011_jr_8016AB6C.o": 0.214, - "build/src/ov_SC04_011/ov_SC04_011_jr_80171B4C.o": 0.095, - "build/src/ov_SC04_011/ov_SC04_011_jr_801734BC.o": 0.175, - "build/src/ov_SC04_011/ov_SC04_011_jr_801789AC.o": 0.057, - "build/src/ov_SC04_011/ov_SC04_011_jr_80178D40.o": 0.09, - "build/src/ov_SC04_011/ov_SC04_011_jr_8017A4AC.o": 0.061, - "build/src/ov_SC04_011/ov_SC04_011_jr_8017AE2C.o": 0.085, - "build/src/ov_SC04_011/ov_SC04_011_jr_8017D494.o": 0.426, - "build/src/ov_SC04_011/ov_SC04_011_o0b.o": 0.05, - "build/src/ov_SC04_011/ov_SC04_011_o0c.o": 0.065 + "build/src/ov_SC03_015/ov_SC03_015_o0b.o": 0.056, + "build/src/ov_SC03_015/ov_SC03_015_o0c.o": 0.064, + "build/src/ov_SC03_015/ov_SC03_015_o0d.o": 0.054, + "build/src/ov_SC03_015/ov_SC03_015_o0e.o": 0.068, + "build/src/ov_SC04_011/ov_SC04_011.o": 0.128, + "build/src/ov_SC04_011/ov_SC04_011_after.o": 0.411, + "build/src/ov_SC04_011/ov_SC04_011_jr_8012ACE0.o": 0.329, + "build/src/ov_SC04_011/ov_SC04_011_jr_80135888.o": 0.049, + "build/src/ov_SC04_011/ov_SC04_011_jr_80135A4C.o": 0.052, + "build/src/ov_SC04_011/ov_SC04_011_jr_80135D20.o": 0.113, + "build/src/ov_SC04_011/ov_SC04_011_jr_801380E0.o": 0.138, + "build/src/ov_SC04_011/ov_SC04_011_jr_8013C98C.o": 0.101, + "build/src/ov_SC04_011/ov_SC04_011_jr_8013F350.o": 0.069, + "build/src/ov_SC04_011/ov_SC04_011_jr_8013FFD8.o": 0.061, + "build/src/ov_SC04_011/ov_SC04_011_jr_80140608.o": 0.16, + "build/src/ov_SC04_011/ov_SC04_011_jr_8015444C.o": 0.062, + "build/src/ov_SC04_011/ov_SC04_011_jr_80154C24.o": 0.159, + "build/src/ov_SC04_011/ov_SC04_011_jr_801588CC.o": 0.078, + "build/src/ov_SC04_011/ov_SC04_011_jr_80159C84.o": 0.061, + "build/src/ov_SC04_011/ov_SC04_011_jr_8015A3C8.o": 0.059, + "build/src/ov_SC04_011/ov_SC04_011_jr_8015AE2C.o": 0.084, + "build/src/ov_SC04_011/ov_SC04_011_jr_8015C32C.o": 0.335, + "build/src/ov_SC04_011/ov_SC04_011_jr_8016AB6C.o": 0.211, + "build/src/ov_SC04_011/ov_SC04_011_jr_80171B4C.o": 0.098, + "build/src/ov_SC04_011/ov_SC04_011_jr_801734BC.o": 0.177, + "build/src/ov_SC04_011/ov_SC04_011_jr_801789AC.o": 0.052, + "build/src/ov_SC04_011/ov_SC04_011_jr_80178D40.o": 0.087, + "build/src/ov_SC04_011/ov_SC04_011_jr_8017A4AC.o": 0.064, + "build/src/ov_SC04_011/ov_SC04_011_jr_8017AE2C.o": 0.092, + "build/src/ov_SC04_011/ov_SC04_011_jr_8017D494.o": 0.414, + "build/src/ov_SC04_011/ov_SC04_011_o0b.o": 0.044, + "build/src/ov_SC04_011/ov_SC04_011_o0c.o": 0.051 }, "ok": true, - "seconds": 2.3 + "seconds": 2.2 } diff --git a/.run/P36/delever/ledger.jsonl b/.run/P36/delever/ledger.jsonl index e973cdb690..98db9f57ae 100644 --- a/.run/P36/delever/ledger.jsonl +++ b/.run/P36/delever/ledger.jsonl @@ -29418,3 +29418,4 @@ {"ts": "2026-09-11 02:04:50", "label": "s104_e10", "rung": "E", "calib": {"head": "7e7d9c523", "stamp": "15956e4a96c4"}, "tu": "src/800.c", "fn": "func_80021174", "addr": 2147619188, "aliases": null, "header": false, "includers": 0, "nhash_before": "ff01f6d8bf282c431329206e903f5ab8fb9a952c", "nhash_after": "27b3d5dffa268ac2b379b13c87142ab329756edf", "source": ".run/P36/agents/main__func_80021174/body.c", "verdict": "LEVER-FREE", "sites": [], "compiles": 1, "seconds": 0.438, "objects": ["build/src/800.o"], "before_text": "s32 func_80021174(s32 a0, s32 a1)\n{\n s32 sp[4];\n s32 lim;\ns32 *p_a3;\nregister s32 ret __asm__(\"$2\"); // !FAKE: pin $2 \u2014 NEEDED DIFFERS (P36 rung B tus9)\n if (a0 == 0x7FFF7FFF)\n {\n return 1;\n }\n p_a3 = D_800AE688;\n__asm__ volatile ( \"lw $12, 0( %0 );\" \"lw $13, 4( %0 );\" \"ctc2 $12, $0;\" \"ctc2 $13, $1;\" \"lw $12, 8( %0 );\" \"lw $13, 12( %0 );\" \"lw $14, 16( %0 );\" \"ctc2 $12, $2;\" \"ctc2 $13, $3;\" \"ctc2 $14, $4;\" \"lw $12, 20( %0 );\" \"lw $13, 24( %0 );\" \"ctc2 $12, $5;\" \"lw $14, 28( %0 );\" \"ctc2 $13, $6;\" \"ctc2 $14, $7\" : : \"r\"( p_a3 ) : \"$12\", \"$13\", \"$14\", \"memory\" ); // !FAKE: gte direct \u2014 clobbers ['memory'] (gte_SetRotTransMatrix_m) beyond Sony's (P36 T5 gte1)\n__asm__ volatile ( \"lhu $13, 4( %0 );\" \"lhu $12, 0( %0 );\" \"sll $13, $13, 16;\" \"or $12, $12, $13;\" \"mtc2 $12, $0;\" \"lwc2 $1, 8( %0 )\" : : \"r\"( (SV_80021174 *)a1 ) : \"$12\", \"$13\", \"memory\" ); // !FAKE: gte direct \u2014 clobbers ['memory'] (gte_ldlv0_m) beyond Sony's (P36 T5 gte1)\n__asm__ volatile ( \"nop;\" \"nop;\" \"rtps\" : : : \"memory\" ); // !FAKE: gte direct \u2014 clobbers ['memory'] (gte_rtps_m) beyond Sony's (P36 T5 gte1)\ngte_stsxy(sp);\ngte_stflg(&sp[1]);\ngte_stszotz(&sp[2]);\n if (sp[1] < 0)\n {\n return 0;\n }\n ret = 0;\n lim = (s16) a0;\n ;\n if ((*((s16 *) sp)) <= (-lim))\n {\n return ret;\n }\n if (lim < (*((s16 *) sp)))\n {\n return ret;\n }\n a0 = a0 >> 16;\n a1 = *((s16 *) (((char *) sp) + 2));\n if (a1 <= (-a0))\n {\n return ret;\n }\n ret = !(a0 < a1);\n return ret;\n}\n", "after_text": "s32 func_80021174(s32 a0, s32 a1)\n{\n struct {\n s16 sx, sy;\n s32 flag;\n s32 otz;\n } r;\n s32 lim;\n\n if (a0 == 0x7FFF7FFF) {\n return 1;\n }\n gte_SetRotTransMatrix(D_800AE688);\n gte_ldlv0((SV_80021174 *)a1);\n gte_rtps();\n gte_stsxy(&r.sx);\n gte_stflg(&r.flag);\n gte_stszotz(&r.otz);\n if (r.flag < 0) {\n return 0;\n }\n lim = (s16)a0;\n if (-lim < r.sx && r.sx <= lim && -(a0 >> 16) < r.sy && r.sy <= (a0 >> 16)) {\n return 1;\n }\n return 0;\n}\n"} {"ts": "2026-09-11 02:05:20", "label": "s104_e13", "rung": "E", "calib": {"head": "c37ef9c15", "stamp": "15956e4a96c4"}, "tu": "src/ov_SC02_017/ov_SC02_017_jr_8017DF34.c", "fn": "func_80185574", "addr": 2149078388, "aliases": null, "header": false, "includers": 0, "nhash_before": "9d423d4275920c5121fd435c05fc402c1fba561c", "nhash_after": "0bee5bbfb3d4a6371c187915f491596b1b83bab8", "source": ".run/P36/agents/ov_SC02_017__func_80185574/body.c", "verdict": "LEVER-FREE", "sites": [], "compiles": 1, "seconds": 0.202, "objects": ["build/src/ov_SC02_017/ov_SC02_017_jr_8017DF34.o"], "before_text": "void func_80185574(s32 arg0) {\n s32 self;\n register s32 zr __asm__(\"$0\"); // !FAKE: pin $0 \u2014 NEEDED DIFFERS (P36 rung B tus8)\n s32 mask;\n s32 flag;\n\n self = arg0 + zr;\n flag = *(s16 *)(self + 0xAA);\n *(s16 *)(self + 0x5C) = 0;\n *(s16 *)(self + 0x98) = 0;\n *(s32 *)(self + 0x1C) = 0;\n if (flag == 0) {\n s32 p;\n register s32 f __asm__(\"$2\"); // !FAKE: pin $2 \u2014 NEEDED DIFFERS (P36 rung B tus8)\n mask = ~0x80;\n p = *(s32 *)(self + 0x20);\n f = *(s32 *)(self + 0xDC);\n p = *(u16 *)(p + 0x18);\n *(s32 *)(self + 0xDC) = f & mask;\n *(s16 *)(self + 0x100) = p;\n } else {\n s32 q;\n register s32 g __asm__(\"$3\"); // !FAKE: pin $3 \u2014 NEEDED DIFFERS (P36 rung B tus8)\n q = *(s32 *)(self + 0x20);\n g = *(s32 *)(self + 0xDC);\n q = *(u16 *)(q + 0x18);\n g |= 0x80;\n *(s32 *)(self + 0xDC) = g;\n *(s16 *)(self + 0x104) = q;\n }\n}\n", "after_text": "void func_80185574(s32 arg0) {\n *(s16 *)(arg0 + 0x5C) = 0;\n *(s16 *)(arg0 + 0x98) = 0;\n *(s32 *)(arg0 + 0x1C) = 0;\n if (*(s16 *)(arg0 + 0xAA) == 0) {\n *(s16 *)(arg0 + 0x100) = *(u16 *)(*(s32 *)(arg0 + 0x20) + 0x18);\n *(s32 *)(arg0 + 0xDC) &= ~0x80;\n } else {\n *(s16 *)(arg0 + 0x104) = *(u16 *)(*(s32 *)(arg0 + 0x20) + 0x18);\n *(s32 *)(arg0 + 0xDC) |= 0x80;\n }\n}\n"} {"ts": "2026-09-11 02:05:38", "label": "s104_e13", "rung": "E", "calib": {"head": "580d7bfd8", "stamp": "15956e4a96c4"}, "tu": "src/ov_SC02_017/ov_SC02_017_jr_8017DF34.c", "fn": "func_801842E0", "addr": 2149073632, "aliases": null, "header": false, "includers": 0, "nhash_before": "5a24bef2d7747b0700b66f9dd4d611717951cb5d", "nhash_after": "11c0fffaf83ae81cbc2d75d40ee56dacc3c3679b", "source": ".run/P36/agents/ov_SC02_017__func_801842E0/body.c", "verdict": "LEVER-FREE", "sites": [], "compiles": 1, "seconds": 0.195, "objects": ["build/src/ov_SC02_017/ov_SC02_017_jr_8017DF34.o"], "before_text": "void func_801842E0(void) {\n register s32 t __asm__(\"$3\"); // !FAKE: pin $3 \u2014 NEEDED DIFFERS (P36 rung B tus8)\n s32 a0;\n s32 v0;\n s32 a1;\n register s32 a1p __asm__(\"$5\"); // !FAKE: pin $5 \u2014 NEEDED DIFFERS (P36 rung B tus8)\n\n t = D_800B99DA;\n if (t != D_8018EF88) {\n a0 = (s16)D_80126B62;\n D_8018EF88 = t;\n\n if (a0 >= -0x7FF) {\n t = D_8018EF84;\n } else if (a0 < -0xC00) {\n t = D_8018EF84;\n } else {\n SV3_8012CC88 sp10;\n SV3_8012CC88 sp18;\n\n sp10.vx = D_80126B5E;\n sp10.vy = 0;\n sp10.vz = D_80126B66;\n sp18.vx = 0x11;\n sp18.vz = -0x3AA;\n sp18.vy = 0;\n\n t = func_800132BC((s32)&sp10, (s32)&sp18);\n }\n\n a0 = D_8018EF84;\n if (t >= a0) {\n func_8002D4C8(4, 0x5E5);\n } else {\n t = a0 - t;\n t -= 0x10000;\n\n if (t >= 0) {\n v0 = t << 7;\n } else {\n t = -t;\n v0 = t << 7;\n }\n v0 = v0 - t;\n t = v0 / a0;\n\n if (t <= 0) {\n t = 0;\n } else if (t >= 0x80) {\n t = 0x7F;\n }\n\n if (D_801270C8 != 0) {\n a1 = (u32)t >> 31;\n v0 = t + a1;\n t = v0 >> 1;\n }\n\n a0 = 0x5E5;\n __asm__ __volatile__(\"\" : \"=r\"(a0) : \"0\"(a0)); // !FAKE: launder \u2014 NEEDED DIFFERS (P36 rung B tus8)\n a1p = t | 0x1000;\n a1p = (u16)a1p;\n func_8002D4C8(a0, a1p);\n }\n }\n}\n", "after_text": "void func_801842E0(void) {\n s32 t;\n s32 y;\n\n t = D_800B99DA;\n if (t != D_8018EF88) {\n y = (s16)D_80126B62;\n D_8018EF88 = t;\n\n if (y >= -0x7FF) {\n t = D_8018EF84;\n } else if (y < -0xC00) {\n t = D_8018EF84;\n } else {\n SV3_8012CC88 sp10;\n SV3_8012CC88 sp18;\n\n sp10.vx = D_80126B5E;\n sp10.vy = 0;\n sp10.vz = D_80126B66;\n sp18.vx = 0x11;\n sp18.vz = -0x3AA;\n sp18.vy = 0;\n\n t = func_800132BC((s32)&sp10, (s32)&sp18);\n }\n\n if (t >= D_8018EF84) {\n func_8002D4C8(4, 0x5E5);\n } else {\n t = D_8018EF84 - t - 0x10000;\n if (t < 0) {\n t = -t;\n }\n t = t * 127 / D_8018EF84;\n\n if (t <= 0) {\n t = 0;\n } else if (t >= 0x80) {\n t = 0x7F;\n }\n\n if (D_801270C8 != 0) {\n t /= 2;\n }\n\n func_8002D4C8(0x5E5, (u16)(t | 0x1000));\n }\n }\n}\n"} +{"ts": "2026-09-11 02:14:31", "label": "s104_e11", "rung": "E", "calib": {"head": "860ed9153", "stamp": "15956e4a96c4"}, "tu": "src/ov_SC04_011/ov_SC04_011_jr_8017D494.c", "fn": "func_80183E20", "addr": 2149072416, "aliases": null, "header": false, "includers": 0, "nhash_before": "e58fc156bf0394cf9c2d78148df0a8015d1cd7a7", "nhash_after": "2346bbbe0d370caa86a647c595e42a557989f555", "source": ".run/P36/agents/ov_SC04_011__func_80183E20/body.c", "verdict": "LEVER-FREE", "sites": [], "compiles": 1, "seconds": 0.389, "objects": ["build/src/ov_SC04_011/ov_SC04_011_jr_8017D494.o"], "before_text": "void aF80183E20(void *a0) {\n extern u16 D_801EFD40;\n extern u16 aEFD24 __asm__(\"D_801EFD24\");\n u16 flags;\n u16 cnt;\n s32 v0;\n s32 v1;\n s32 v1b;\n flags = D_801EFD40;\n if (flags & 0x80) {\n v0 = *(u16 *)((u8 *)a0 + 0x76);\n v1 = *(u16 *)((u8 *)a0 + 0x60);\n *(u16 *)((u8 *)a0 + 0x60) = 0;\n v0 = v0 - v1;\n *(u16 *)((u8 *)a0 + 0x76) = v0;\n __asm__(\"\":::\"memory\"); // !FAKE: barrier memory \u2014 NEEDED DIFFERS (P36 rung B t3_tus1)\n v1b = *(s16 *)((u8 *)a0 + 0x76);\n D_801EFD40 = flags & 0xFF7F;\n if (v1b < 0) {\n *(u16 *)((u8 *)a0 + 0x76) = 0;\n }\n }\n cnt = aEFD24;\n if (cnt != 0) {\n cnt = cnt - 1;\n aEFD24 = cnt;\n if (cnt == 0) {\n func_80186AB8();\n }\n }\n}\n", "after_text": "void aF80183E20(void *a0) {\n extern u16 D_801EFD40;\n extern u16 aEFD24 __asm__(\"D_801EFD24\");\n u16 flags;\n u16 cnt;\n s32 v0;\n s32 v1;\n flags = D_801EFD40;\n if (flags & 0x80) {\n v0 = *(u16 *)((u8 *)a0 + 0x76);\n v1 = *(u16 *)((u8 *)a0 + 0x60);\n *(u16 *)((u8 *)a0 + 0x60) = 0;\n *(u16 *)((u8 *)a0 + 0x76) = v0 - v1;\n D_801EFD40 = flags & 0xFF7F;\n if (((s16 *)a0)[0x3B] < 0) {\n *(u16 *)((u8 *)a0 + 0x76) = 0;\n }\n }\n cnt = aEFD24;\n if (cnt != 0) {\n cnt = cnt - 1;\n aEFD24 = cnt;\n if (cnt == 0) {\n func_80186AB8();\n }\n }\n}\n"} diff --git a/src/ov_SC04_011/ov_SC04_011_jr_8017D494.c b/src/ov_SC04_011/ov_SC04_011_jr_8017D494.c index eb071b5b3e..2717b3f152 100644 --- a/src/ov_SC04_011/ov_SC04_011_jr_8017D494.c +++ b/src/ov_SC04_011/ov_SC04_011_jr_8017D494.c @@ -7397,18 +7397,14 @@ void aF80183E20(void *a0) { u16 cnt; s32 v0; s32 v1; - s32 v1b; flags = D_801EFD40; if (flags & 0x80) { v0 = *(u16 *)((u8 *)a0 + 0x76); v1 = *(u16 *)((u8 *)a0 + 0x60); *(u16 *)((u8 *)a0 + 0x60) = 0; - v0 = v0 - v1; - *(u16 *)((u8 *)a0 + 0x76) = v0; - __asm__("":::"memory"); // !FAKE: barrier memory — NEEDED DIFFERS (P36 rung B t3_tus1) - v1b = *(s16 *)((u8 *)a0 + 0x76); + *(u16 *)((u8 *)a0 + 0x76) = v0 - v1; D_801EFD40 = flags & 0xFF7F; - if (v1b < 0) { + if (((s16 *)a0)[0x3B] < 0) { *(u16 *)((u8 *)a0 + 0x76) = 0; } }