phase-36: S104 s104_e11 — func_80183E20 banked at 0 through the whole-object gate + propagated — memory barrier → 0: a global store between (cse.c:7539-7578) + the reload as an array index (expr.c:4569-4577; sched.c:817-839)

This commit is contained in:
Drew T
2026-09-11 02:14:32 -06:00
parent 860ed9153e
commit 5c561effc1
11 changed files with 638 additions and 188 deletions
@@ -0,0 +1,65 @@
s32 impl_801833D4(s32 a0, s32 a1)
{
/* $a1 is call-saved across func_801852BC/func_801853D0/func_80185648 until its one use
far below; gcc puts its param->hardreg move in the branch's delay slot rather than up
front. The copy is a STATEMENT after the first load, not the declaration's initialiser:
an initialiser sits right after the parameter copy, cse retargets that copy
(cse.c:7454-7480) and sched1 never moves it (sched.c:3188-3205); after a real insn,
combine folds the parameter copy into it and it is scheduled like any other insn
(P36 S104 e11, the $19 pin removed). */
s32 r1;
u16 s0;
s32 obj;
s32 s4;
/* v[2..4] is a 3-word (vel-like x/y/z) struct passed by address to func_800484EC; the
real local apparently has 2 leading words of other data ahead of it in the frame (its
address-taken struct forces gcc to reserve stack starting 2 words earlier than our x
field) -- v[0]/v[1] are unused padding needed only to reproduce the frame layout. */
s32 v[5];
s0 = D_801EFD20;
s4 = 0;
r1 = a1;
obj = aFC4C[s0];
if (D_801EFD40 & 0x4) {
func_801852BC(s0);
func_801853D0(s0);
func_80185648(s0);
v[3] = 0;
v[2] = 0;
if (D_801EFD40 & 0x20) {
v[4] = 0xFFD80000;
} else {
v[4] = 0xFFEC0000;
}
func_800484EC(*(s32 *)(obj + 0x20) + 0x34, (s32)&v[2], (s32)&v[2]);
*(s32 *)(obj + 0x4) += v[2];
*(s32 *)(obj + 0x8) += v[3];
*(s32 *)(obj + 0xC) += v[4];
*(u16 *)(*(s32 *)(obj + 0x20) + 0x14) =
*(u16 *)(*(s32 *)(obj + 0x20) + 0x14) + r1;
func_80183BF0(a0);
*(u16 *)(*(s32 *)(obj + 0x20) + 0x10) =
*(u16 *)(a0 + 0xFE) + *(u16 *)(a0 + 0x102);
if (func_801836D4((void *)a0, (void *)obj) != 0) {
s4 = 1;
} else {
if (D_801EFD40 & 0x20) {
s0 = 0x40;
} else {
s0 = 0x20;
}
func_80185960(*(s16 *)(a0 + 0x106), (u16 *)(*(s32 *)(obj + 0x20) + 0x12), s0);
func_80185960(*(s16 *)(a0 + 0x100), (u16 *)(a0 + 0xFE), s0);
}
}
return s4;
}
@@ -0,0 +1,62 @@
# func_801833D4 (impl_801833D4, ov_SC04_011_jr_8017D494.c): mechanism (P36 T7 S104, agent e11)
**Result: score 0 in plain C.** No pin, no asm, no added volatile. Levers go from 1 to 0 (the `$19` pin on `r1`).
Signature unchanged. No copy of this class in another TU (the 0xFFD80000 / `obj + 0x20) + 0x12), s0)` greps find
unrelated bodies).
## (a) The residual in one sentence
Same count, same registers. Two moves of the first block are swapped. The target does `sw s4; move s4,zero` early
and fills the `beqz` delay slot with `move s3,a1`. Mine does `sw s3; move s3,a1` early and puts `move s4,zero` in
the slot.
## (b) The pass and the decision (proven on dumps and bytes)
- **The initialiser `s32 r1 = a1;` sits directly after the parameter copy** `(set 74 $5)`. cse's
`(set REG0 REG1)` swap (`cse.c:7454-7480`: `NEXT_INSN (PREV_INSN (insn)) == insn`, the previous insn sets REG1,
REG0 is the canonical reg) retargets the parameter copy to set `r1` directly. The `.cse` dump of
`scratch/v_first.c` shows `(insn 6 (set (reg/v 74) (reg 5 a1)))`.
- **sched1 never moves a parameter copy.** Before reload, block 0's leading SETs before `NOTE_INSN_FUNCTION_BEG`
are removed from scheduling (`sched.c:3188-3205`: "don't delay getting parameters from hard registers into
pseudo registers"), so the copy keeps the lowest LUID. In sched2 all first-block insns have priority 1, and
`rank_for_schedule`'s last tie-break is `INSN_LUID` (`sched.c:2425-2428`). The `s4 = 0` insn (higher LUID) is
therefore scheduled closer to the branch, and reorg's backward scan puts it in the delay slot.
- **Written as a statement after a real insn** (`s0 = D_801EFD20;` first), the copy `r1 = <param pseudo>` stays a
separate insn at its own position. Combine then folds the dying parameter pseudo into it: in the `.lreg` of
`scratch/v_before_s4.c`, reg 73 is gone and `(insn 16 (set (reg/v 74) (reg 5 a1)) REG_DEAD a1)` sits after the
`D_801EFD20` load. sched1 now schedules it with `LAUNCH_PRIORITY` (`sched.c:187`, the 0x7f000001 in the trace),
ahead of `s4 = 0` (priority 1). That places it after `s4 = 0`, sched2's LUID tie-break keeps the order, and reorg
takes `move s3,a1` into the delay slot, which is the target.
## (c) The move that closed it
`s32 r1 = a1;` becomes `s32 r1;` plus the statement `r1 = a1;` after `s4 = 0;`. Every position after the first
real insn scores 0: after `s0 = …` (`v_before_s4`), after `s4 = 0` (`body.c`), after `obj = …` (`v_after_obj`),
and as the declaration list `u16 s0 = D_801EFD20; s32 r1 = a1;` (`v_declinit`). The body comment that explained
the pin now explains this.
## (d) Generator proposal
When a register residual is two first-block moves swapped, and one of them is a parameter copy (`move sN,aK`, which
the target has later or in a delay slot), turn `T x = aK;` at the top of the body into a plain `x = aK;` statement
placed after the first real statement (and try each later position). This takes the copy out of cse's
adjacent-producer swap and sched1's parameter-copy exclusion.
## (e) What did NOT work (byte evidence)
- `r1 = a1;` as the FIRST statement (`v_first.c`): score 5, the same residual. It is still adjacent to the
parameter copy, so cse still swaps it.
- Deleting `r1` and using `a1` directly (`v_noR1.c`): score 5. The parameter copy is the only copy and stays
unscheduled.
- `s4 = 0;` moved first (`v_s4_first.c`): score 5.
- `r1 = a1;` inside the `if` (`v_in_if.c`): score 8. The copy lands in block 1, which is too late.
- The sweep's R6 (inline r1) and R9 (swap statements) stayed at 5. No generator splits an initialiser off its
declaration.
## (f) Where the method fell short
The residual's "register pairs zero->a1, s4->s3" line reads like an allocation defect, but it was purely order
(ORDER class, same registers). The sched1 trace (the 0x7f000001 LAUNCH_PRIORITY on the copy) and sched.c:3188 gave
the reason. The allocation table would have been a detour.
## (g) Structs question
No. The lever was about where a register copy sits (cse adjacency, sched1's parameter-copy exclusion), not about
memory. A struct for `obj`/`a0` would not change it. The `v[5]` padding array could plausibly become a real struct
local (2 leading words + an x/y/z vector) in the structs phase, but it is not part of this lever.
Files: `body.c` (score 0), `scratch/v_*.c` (variants above), `scratch/dumps_{free,first,before_s4}/`,
`scratch/splice.py`.
@@ -0,0 +1,27 @@
void aF80183E20(void *a0) {
extern u16 D_801EFD40;
extern u16 aEFD24 __asm__("D_801EFD24");
u16 flags;
u16 cnt;
s32 v0;
s32 v1;
flags = D_801EFD40;
if (flags & 0x80) {
v0 = *(u16 *)((u8 *)a0 + 0x76);
v1 = *(u16 *)((u8 *)a0 + 0x60);
*(u16 *)((u8 *)a0 + 0x60) = 0;
*(u16 *)((u8 *)a0 + 0x76) = v0 - v1;
D_801EFD40 = flags & 0xFF7F;
if (((s16 *)a0)[0x3B] < 0) {
*(u16 *)((u8 *)a0 + 0x76) = 0;
}
}
cnt = aEFD24;
if (cnt != 0) {
cnt = cnt - 1;
aEFD24 = cnt;
if (cnt == 0) {
func_80186AB8();
}
}
}
@@ -0,0 +1,59 @@
# func_80183E20 (ov_SC04_011_jr_8017D494.c): mechanism (P36 T7 S104, agent e11)
**Result: score 0 in plain C, on the first `--try`.** No pin, no asm, no added volatile. Levers go from 1 to 0 (the
`__asm__("":::"memory")` barrier at tree line 7408). Signature unchanged. No other copy of the class in `src/` (grep of
`a0 + 0x60) = 0;` with a 0x76/0xFF7F neighbour finds only this one).
## (a) The residual in one sentence
Same count (34/34). The target RELOADS the field it has just stored (`sh v0,118(a0); lh v1,118(a0)`). Mine has
`sll v0,v0,16` instead: cse forwarded the stored value into the `(s16)` test, so no load is left.
## (b) The pass and the decision (proven on bytes)
The original stored the global flag BEFORE it read the field back. Two passes are involved:
- **cse forgets the field at the scalar store.** A store to the fixed address `D_801EFD40` goes through
`note_mem_written` (`cse.c:7539-7578`). Its address does not vary, so it only sets `writes.var`.
`invalidate_memory` (`cse.c:1701-1720`) then removes every in-memory table entry whose address DOES vary
(`cse_rtx_addr_varies_p`, `cse.c:2474`). `a0+0x76` varies, so its entry is removed. The load after the flag
store now finds nothing to forward, and the `lh` survives.
- **sched1 hoists the load back above the flag store.** `true_dependence` (`sched.c:817-839`) drops the
store→load dependence when the load is `MEM_IN_STRUCT_P` with a varying address and the store is a scalar with a
fixed address (the second `! (...)` clause). expand_expr sets `MEM_IN_STRUCT_P` on an INDIRECT_REF whose operand
is a PLUS_EXPR (`expr.c:4569-4577`). `((s16 *)a0)[0x3B]` gives exactly that. `*(s16 *)((u8 *)a0 + 0x76)` is a
NOP_EXPR around the PLUS, so it is not marked, and the load stays below the store.
## (c) The move that closed it
```c
*(u16 *)((u8 *)a0 + 0x76) = v0 - v1;
D_801EFD40 = flags & 0xFF7F; /* flag store moved BEFORE the reload */
if (((s16 *)a0)[0x3B] < 0) { /* array-indexed reload: in-struct MEM */
*(u16 *)((u8 *)a0 + 0x76) = 0;
}
```
The `v1b` temp is gone, and so is the `v0 = v0 - v1;` self-update (it is now stored directly). Both moves are needed:
- cast-style reload after the flag store (`scratch/v_cast.c`): score 5, 35 ins. The `lh` is kept but sched cannot
lift it above `sh D_801EFD40`.
- indexed reload read into a temp BEFORE the flag store (`scratch/v_before.c`): score 5. This is the free body's
residual exactly: cse forwards.
## (d) Generator proposal
When the target has `sw/sh X,K(rA); l[hw] rB,K(rA)` (a store immediately reloaded) and mine forwards the value
(a missing load, often an extra `sll`/`sra` or `move`), look for a later store to a GLOBAL scalar in the same block.
Move that store textually between the field store and the field read, and respell the read as an array index
`((T *)p)[K/sizeof(T)]` (or a struct field). This replaces the `"memory"` barrier lever.
## (e) What did NOT work
Recorded under (c). The sweep's best was 2 (R7 do-while + R9 swap). Neither move reaches this, because no generator
changes a cast deref into an index expression.
## (f) Where the method fell short
Nothing blocked. What found it was the S103 c11 note in METHOD step 3 ("`p[i]` is an aggregate access that
`true_dependence` treats as independent of a scalar store; a cast-wrapped byte-offset read is not marked") plus
reading the target order (`lh` BEFORE the flag's `andi`/`sh`). No dumps were needed.
## (g) Structs question
Yes, and here it is the whole answer. A struct type for `a0` (`struct { … u16 f60 @0x60; … s16 f76 @0x76; }`, with
`a0->f76` as a COMPONENT_REF) makes the reload an in-struct MEM: the COMPONENT_REF case sets `MEM_IN_STRUCT_P`
(`expr.c:4873/4888`). The array index sets the same flag through `expr.c:4569`, so it is the same channel. Not tested with a struct declaration. The array index already proves the channel
on bytes, and a struct field would take the same `true_dependence` clause.
Files: `body.c` (score 0), `scratch/v_cast.c`, `scratch/v_before.c` (both score 5, the two halves of the proof).
@@ -0,0 +1,20 @@
void func_80185214(s32 a0)
{
extern s32 D_801EFC48;
extern u8 D_80194724[];
extern void func_80185A18(s32 idx, s32 val);
s32 p;
s32 i;
s32 c;
p = *(s32 *)(D_801EFC48 + 0xCC);
for (i = 0; i < 5; i++) {
c = *(u8 *)(p + i * 8 + 6);
if (c < 0x15 && D_80194724[c] == 0) {
func_80185A18(c, (s16)(*(u16 *)(p + i * 8 + 2) + a0));
}
if (*(s16 *)(p + i * 8 + 6) & 0x8000) {
break;
}
}
}
@@ -0,0 +1,69 @@
# func_80185214 (ov_SC04_011_jr_8017D494.c): mechanism (P36 T7 S104, agent e11)
**Result: score 0 in plain C.** No pin, no asm, no added volatile, no goto. Levers go from 1 to 0 (the `$16` pin on
`s0`). Signature unchanged. No copy of this class elsewhere: `c = *(u8 *)(s0 + 6);` occurs only here. The
same-name functions in ov_SC02_005 and ov_SC06_029 are different bodies, and the similar walk in `func_80184CCC`
(same TU) is already lever-free.
## (a) The residual in one sentence
One extra instruction. Mine walks a pointer to the entry's `+6` field (`addiu s0,v0,6` and `addiu s1,v0,46`, then
fields at `0(s0)`/`-4(s0)`). The target walks the entry base (`lw s0,204(v0)` straight into s0, `addiu s1,s0,40`,
fields at `6(s0)`/`2(s0)`).
## (b) The pass and the decision (proven on the `-dL` dumps and bytes)
loop.c strength reduction. In the free body `s0` is a pointer biv, and the three field ADDRESSES are its givs:
`Insn 26/42/59: dest address src reg 73 … mult 1 add 6 / add 2 / add 6` (`record_giv`, `loop.c:4341`, dump text
`:4508`). `combine_givs` (`loop.c:5494`, `:5527`) merges them onto the add-6 giv ("giv at 42 combined with giv at
59", "giv at 26 combined with giv at 59"). That giv is reduced to a new register holding `s0+6`, and "biv 73 was
eliminated", so the exit test is rewritten against `s1+6`. The pin hid `s0` from loop.c (a hard register is never
a biv), which is why the lever "worked".
## (c) The move that closed it
Index the entries by a counter instead of walking a pointer (the S103 c2 / S104 d16/d18 family):
```c
p = *(s32 *)(D_801EFC48 + 0xCC);
for (i = 0; i < 5; i++) {
c = *(u8 *)(p + i * 8 + 6);
if (c < 0x15 && D_80194724[c] == 0) {
func_80185A18(c, (s16)(*(u16 *)(p + i * 8 + 2) + a0));
}
if (*(s16 *)(p + i * 8 + 6) & 0x8000) {
break;
}
}
```
Now the givs are the SUMS `p + i*8` (`Insn 31/78: giv reg … mult 8 add (reg 73)`, a DEST_REG giv with
`add_val = p`). The reduced register holds the entry base itself (initial value `p`), and the field offsets stay in
the addressing mode (`6(s0)`, `2(s0)`). Biv `i` is eliminated against `p + 40`, which gives `addiu s1,s0,40` and
the signed `slt`. `s2 = a0` is gone (`a0` is used directly). Also proven:
- `scratch/v_do_i.c` (the same loop as a do-while with `i++`): 0.
- `scratch/v_goto.c` (the ORIGINAL pointer walk as a goto loop, no LOOP notes, so loop.c never runs): 0.
- `scratch/v_struct.c` (a body-local `struct Ent { u16 f0, f2, f4; union { u8 id; s16 flags; } f6; } *e;` indexed
as `e[i]`): 0.
## (d) Generator proposal
When a loop's residual shows a walked pointer whose reduced register sits at a field offset (fields read at
`0(sN)`/negative offsets, the end pointer `base+K+off`), and the target reads the same fields at `off(sN)` with
`sN` = the loaded base: rewrite `for/do (p…; p < end; p += S)` as `for (i = 0; i < (end-base)/S; i++)` with every
`*(T *)(p + K)` becoming `*(T *)(base + i * S + K)`. The DEST_REG giv `base + i*S` then becomes the reduced register.
## (e) What did NOT work (byte evidence)
- `for (s1 = s0 + 0x28; s0 < s1; s0 += 8)` (`scratch/v_for_ptr.c`): 11. It keeps the entry test and the
offset giv.
- The sweep (R4 declaration moves, R6/R10 parameter moves, R7/R8/R9) stayed at 6. No generator turns a pointer walk
into an indexed loop.
## (f) Where the method fell short
METHOD step 3 (S103 c2) and step 14 (d16/d18) already name this family. The `-dL` dump's "giv … combined … reduced
to" lines settle it in one compile. The residual's "register pairs v0->s0" line pointed at allocation, but the
defect was the +6 bias of the reduced register.
## (g) Structs question
Yes, plausibly for readability, but it is not needed for the bytes. `D_801EFC48 + 0xCC` points to an array of
8-byte entries (`+2` u16 value, `+6` u8 id whose halfword also carries a 0x8000 end flag). With a struct array
indexed `e[i]` (`scratch/v_struct.c`) the loop also scores 0, because the channel is the loop's giv shape (an
indexed sum vs a walked pointer), not MEM_IN_STRUCT_P. A struct POINTER walked with `e++` would presumably
reproduce the free body's defect. Not tested.
Files: `body.c` (score 0), `scratch/v_{for_i,do_i,goto,struct}.c` (0), `scratch/v_for_ptr.c` (11),
`scratch/dumps_{free,for_i}/` (`.loop` dumps).
@@ -0,0 +1,45 @@
void func_80186020(void *a0) {
s32 p;
if ((D_801EFD40 & 2) != 0) {
s32 a;
s32 b;
a = (s16)D_801EFD48;
if (a >= 0x120) {
*(s16 *)&D_801EFD4C = -0x60;
} else if (a < -0x5F) {
*(s16 *)&D_801EFD4C = 0x60;
}
a = D_801EFD48;
b = D_801EFD4C;
p = *(s32 *)((s32)a0 + 0x20);
a = a + b;
b = D_801EFD44;
D_801EFD48 = a;
b = b + a;
*(u16 *)(p + 0x10) = b;
} else {
u16 x;
u16 y;
if ((D_801EFD40 & 1) != 0) {
x = D_801EFD48;
if ((s16)x >= 0x120) {
return;
}
y = x + 0x60;
} else {
x = D_801EFD48;
if ((s16)x < -0x5F) {
return;
}
y = x - 0x60;
}
x = D_801EFD44;
p = *(s32 *)((s32)a0 + 0x20);
D_801EFD48 = y;
x = x + y;
*(u16 *)(p + 0x10) = x;
}
}
@@ -0,0 +1,106 @@
# func_80186020 (ov_SC04_011_jr_8017D494.c): mechanism (P36 T7 S104, agent e11)
**Result: score 0 in plain C.** No pin, no asm, no added volatile. Levers go from 1 to 0 (the `$2` pin on `y`).
Signature unchanged. No copy of this class elsewhere (greps of the `-0x60`/`0x120` pair and of the pin find only
this body).
## (a) The residual in one sentence
Same count (60/60). Every value the target keeps in `v0` is in `v1` in mine and the other way round, in BOTH arms.
The free body's shared `u16 x, y` get x→v0 and y→v1; the target has x→v1 and y→v0.
## (b) The pass and the decision (proven on the `.lreg`/`.greg` dumps and bytes)
global.c allocates in priority order (`allocno_compare`, `global.c:594-610`). `find_reg` gives each allocno the
first register that is free of conflicts and preferences (`global.c:945-990`). In the free body:
- In arm 1, `x = x + y` is HImode arithmetic. It expands into an SI temp (r91) and a HI copy into `x`. cse then
routes the two later readers (the `D_801EFD48` store and `y + x`) through r91, so the copy into `x` dies and
r91 is a LOCAL. local-alloc allocates it before global runs and gives it `v0`, its first choice.
- Global then sees `y` (reloaded with `D_801EFD44` while r91 is live) as conflicting with `v0`. `x`, the
highest-priority allocno (26666 vs y's 21428), has no `v0` conflict and a copy preference for r91's `v0`, so it
takes `v0`. `y` falls to `v1`. The pin forced `y` into `v0`.
What the target's allocation needs (read off the working body's `.greg`: order `x a b y`, dispositions
x→3, a→3, b→2, y→2):
- The two `v1` roles (arm 1's D48 accumulator, arm 2's D48/D44 value) must CONFLICT with `v0`. That happens when
the variable is live while a `v0` compare temp (`slti v0,…`) is live, so `find_reg` skips `v0` for them.
- The two `v0` roles must be global and must not overlap any local that sits in `v0`.
- Arm 1's sum must not be a separate local temp.
## (c) The moves that closed it (joint; each alone scores worse)
```c
if ((D_801EFD40 & 2) != 0) {
s32 a; /* arm-1 locals, SImode */
s32 b;
a = (s16)D_801EFD48; /* the test operand hoisted into `a` */
if (a >= 0x120) { … } else if (a < -0x5F) { … }
a = D_801EFD48;
b = D_801EFD4C; p = …; a = a + b; b = D_801EFD44; D_801EFD48 = a; b = b + a; … = b;
} else {
u16 x; /* arm-2 locals, HImode */
u16 y;
if (D_801EFD40 & 1) { x = D_801EFD48; if ((s16)x >= 0x120) return; y = x + 0x60; }
else { x = D_801EFD48; if ((s16)x < -0x5F) return; y = x - 0x60; }
x = D_801EFD44; p = …; D_801EFD48 = y; x = x + y; … = x;
}
```
1. **Split the shared variables per arm** (S104 d7/d22: one variable per case), and give each arm the width its
arithmetic proves. Arm 1 as `s32` makes `a = a + b` a plain SImode add into `a`, so no HI→SI temp and no local
r91. Arm 2 stays `u16`: its `move v1,v0` (the HI copy of the `lh`) and the 16-byte frame (the dead
sign-extension intermediate r96/r100 with class `ST_REGS or none` that reload spills; see `.lreg`) are
HImode-only artefacts. With `s32` in arm 2 both vanish (`scratch/c3/w_s32_s32.c`, 56 ins).
2. **Hoist the arm-1 test operand into `a`** (`a = (s16)D_801EFD48; if (a >= 0x120) … else if (a < -0x5F)`).
`a` is now set in one block and read in the else-if block, so it is block-global
(`local-alloc.c:469-476` requires a single block and `reg_n_deaths == 1`). It is also live across the
`slti v0` temps, so it conflicts with `v0` and takes `v1`. The target shows the same thing: the `lh v1,D48` of
the test and the later `lhu v1,D48` share `v1`.
3. **Arm 2 reads `D_801EFD48` into `x` in each sub-block** (`x = D_801EFD48; if ((s16)x …) …; y = x ± 0x60;`).
`x` now spans three blocks, and its HI copy is born while the `lh` result (local `v0`) is still read by
`slti`, so `x` conflicts with `v0` and takes `v1`. `y` takes `v0`. This is what the target's
`lh v0; move v1,v0; slti v0,v0,…` is.
Proven on bytes:
- `body.c` (block-scoped declarations) and `scratch/c10/g_abb_s32s32_v.c` (function-scope declarations): 0.
- Without move 2 (`scratch/c11/n3_nohoist.c`): 15, because cross-jump merges the arm tails once the registers
line up.
- Without move 3 (`n1_arm2free.c`): 14.
- `a` as `s16` (`n4_a_s16.c`): 22.
- Without the `*(s16 *)&D_801EFD4C` store casts (`n5_nocast.c`): 1.
## (d) Generator proposal
When a register residual swaps `v0`/`v1` (or any pair) between variables SHARED by two if/else arms, first split
them per arm (one variable per role per arm, at the widths the instructions prove: `s32` where the target adds in
a register with no HI copy, `u16` where it keeps a `move` of an `lh`). Then, for each variable the target puts in
the register that also holds an earlier TEST operand (`lh vN,G; slti …,vN,K` followed by a later reload of `G`
into `vN`), hoist the test operand into that variable (`v = (s16)G; if (v >= K)`). This makes it block-global and
makes it conflict with the compare temp's register.
## (e) What did NOT work (byte evidence, scratch/c*/ with their scores)
- The free body: 22. The regen sweep's 122 candidates: best 10 (`scratch/regen_scores.txt`). The history's "7" path
could not be reproduced from its label.
- Renaming or splitting per arm with `u16` everywhere (`c/`, `c2/`): 10-24. Arm 1's cse temp r91 always takes `v0`.
- Widths alone (`c3/`, 16 combinations): best 15. `s32` in arm 2 deletes the HI `move` and the frame.
- Arm 1 `s32` + arm 2 `u16` `y2` with a shared `s32 x` (`c4/t_s32s32_x_s32u16.c`): 4. The `x + y2` add
zero-extends (`andi 0xffff`). Folding the add into the store removes the `andi` but lets cross-jump merge the
tails (`c9/`: 7).
- A single-set D44 temp to get the scheduler's birthing boost (`c8/`): 10. The D44 load is then hoisted above the
sum (`sched.c:2477-2490` boosts it, `sched.c:2425-2428` breaks the tie), so it overlaps `y`.
## (f) Where the method fell short
- The allocation table (`tools/alloc_table.py`) was the deciding read. It showed r91 as a LOCAL in `v0` and
`y` "conflicts v0". What it cannot show is WHICH local causes a conflict. Grepping `.lreg` for the local's
block and life settled it.
- The winning move 2 (hoisting a test operand into an existing variable, so that it conflicts with the compare
temp) is not in METHOD. It is the inverse of S103 c18's "same register, same role → one variable": here the
target's `lh v1,G … lhu v1,G` said "same register, same variable". A block-by-block read of the target for
"which values share a register" found it. The hunk view hid it.
- Enumeration was effective: about 300 bodies in 8 batches through `--try`. Each batch was shaped by the previous
batch's residual.
## (g) Structs question
No. The lever came from register-allocation conflicts between user variables (local-alloc/global), with no memory
ordering involved. The globals `D_801EFD44/48/4C` are three separate scalars (x-offset, position, velocity-like).
Making them fields of one struct would put `MEM_IN_STRUCT_P` on the accesses. Nothing here depends on alias
decisions, although cse's store-forwarding into `x = x + y` (`cse.c` invalidation of varying-address entries,
`cse.c:1701-1720`) would be worth re-checking in the structs phase. Not tested.
Files: `body.c` (score 0), `scratch/c{,2..11}/` (candidate batches), `scratch/tryall.sh`, `scratch/splice.py`,
`scratch/dumps_{free,i1,m_xyc_u,k1_u16,match}/`.
+182 -182
View File
@@ -1,7 +1,7 @@
{
"head": "580d7bfd8",
"head": "860ed9153",
"stamp": "15956e4a96c4",
"generated": "2026-09-11 02:05",
"generated": "2026-09-11 02:14",
"aliases": [
"main",
"ov_SC03_014",
@@ -21,207 +21,207 @@
"main": {
"objects": 85,
"identical": 85,
"seconds": 6.937999999999999,
"mean_s": 0.082
"seconds": 6.653,
"mean_s": 0.078
},
"ov_SC03_014": {
"objects": 32,
"identical": 32,
"seconds": 4.733,
"mean_s": 0.148
"seconds": 4.586,
"mean_s": 0.143
},
"ov_SC03_015": {
"objects": 32,
"identical": 32,
"seconds": 4.500000000000001,
"mean_s": 0.141
"seconds": 4.437,
"mean_s": 0.139
},
"ov_SC04_011": {
"objects": 28,
"identical": 28,
"seconds": 3.8809999999999993,
"mean_s": 0.139
"seconds": 3.7390000000000008,
"mean_s": 0.134
}
},
"per_object_seconds": {
"build/src/800.o": 0.742,
"build/src/800_b.o": 0.077,
"build/src/800_b_2.o": 0.303,
"build/src/800_b_o0a.o": 0.077,
"build/src/800_c.o": 0.203,
"build/src/800b2.o": 0.083,
"build/src/apicard1.o": 0.08,
"build/src/apicard2.o": 0.072,
"build/src/apicard3.o": 0.079,
"build/src/apicard4.o": 0.067,
"build/src/apicard5.o": 0.087,
"build/src/apicard6.o": 0.076,
"build/src/apicard7.o": 0.082,
"build/src/boot.o": 0.117,
"build/src/gap.o": 0.07,
"build/src/libapi1.o": 0.071,
"build/src/libapi2.o": 0.061,
"build/src/libc2_1.o": 0.064,
"build/src/libc2_2.o": 0.059,
"build/src/libcd1.o": 0.066,
"build/src/libcd2.o": 0.064,
"build/src/libetc.o": 0.058,
"build/src/libgpu.o": 0.07,
"build/src/libgpu2.o": 0.07,
"build/src/libgs1.o": 0.081,
"build/src/libgs2.o": 0.048,
"build/src/libgs3.o": 0.066,
"build/src/libgs4.o": 0.066,
"build/src/libgs5.o": 0.06,
"build/src/libgs6.o": 0.077,
"build/src/libgs7.o": 0.059,
"build/src/libgs8.o": 0.068,
"build/src/libgte1.o": 0.06,
"build/src/libgte10.o": 0.058,
"build/src/libgte11.o": 0.061,
"build/src/libgte12.o": 0.064,
"build/src/800.o": 0.712,
"build/src/800_b.o": 0.087,
"build/src/800_b_2.o": 0.276,
"build/src/800_b_o0a.o": 0.062,
"build/src/800_c.o": 0.197,
"build/src/800b2.o": 0.072,
"build/src/apicard1.o": 0.072,
"build/src/apicard2.o": 0.071,
"build/src/apicard3.o": 0.068,
"build/src/apicard4.o": 0.062,
"build/src/apicard5.o": 0.078,
"build/src/apicard6.o": 0.066,
"build/src/apicard7.o": 0.079,
"build/src/boot.o": 0.086,
"build/src/gap.o": 0.06,
"build/src/libapi1.o": 0.08,
"build/src/libapi2.o": 0.057,
"build/src/libc2_1.o": 0.071,
"build/src/libc2_2.o": 0.067,
"build/src/libcd1.o": 0.078,
"build/src/libcd2.o": 0.061,
"build/src/libetc.o": 0.056,
"build/src/libgpu.o": 0.063,
"build/src/libgpu2.o": 0.067,
"build/src/libgs1.o": 0.055,
"build/src/libgs2.o": 0.079,
"build/src/libgs3.o": 0.063,
"build/src/libgs4.o": 0.05,
"build/src/libgs5.o": 0.061,
"build/src/libgs6.o": 0.072,
"build/src/libgs7.o": 0.075,
"build/src/libgs8.o": 0.062,
"build/src/libgte1.o": 0.059,
"build/src/libgte10.o": 0.061,
"build/src/libgte11.o": 0.057,
"build/src/libgte12.o": 0.067,
"build/src/libgte13.o": 0.064,
"build/src/libgte14.o": 0.064,
"build/src/libgte15.o": 0.066,
"build/src/libgte16.o": 0.06,
"build/src/libgte17.o": 0.062,
"build/src/libgte18.o": 0.061,
"build/src/libgte19.o": 0.069,
"build/src/libgte2.o": 0.061,
"build/src/libgte20.o": 0.071,
"build/src/libgte21.o": 0.064,
"build/src/libgte22.o": 0.065,
"build/src/libgte23.o": 0.061,
"build/src/libgte24.o": 0.069,
"build/src/libgte25.o": 0.067,
"build/src/libgte26.o": 0.068,
"build/src/libgte27.o": 0.065,
"build/src/libgte28.o": 0.073,
"build/src/libgte29.o": 0.065,
"build/src/libgte3.o": 0.069,
"build/src/libgte30.o": 0.071,
"build/src/libgte4.o": 0.063,
"build/src/libgte14.o": 0.049,
"build/src/libgte15.o": 0.063,
"build/src/libgte16.o": 0.07,
"build/src/libgte17.o": 0.069,
"build/src/libgte18.o": 0.063,
"build/src/libgte19.o": 0.06,
"build/src/libgte2.o": 0.077,
"build/src/libgte20.o": 0.065,
"build/src/libgte21.o": 0.063,
"build/src/libgte22.o": 0.059,
"build/src/libgte23.o": 0.063,
"build/src/libgte24.o": 0.063,
"build/src/libgte25.o": 0.064,
"build/src/libgte26.o": 0.067,
"build/src/libgte27.o": 0.077,
"build/src/libgte28.o": 0.053,
"build/src/libgte29.o": 0.067,
"build/src/libgte3.o": 0.072,
"build/src/libgte30.o": 0.075,
"build/src/libgte4.o": 0.066,
"build/src/libgte5.o": 0.066,
"build/src/libgte6.o": 0.072,
"build/src/libgte7.o": 0.076,
"build/src/libgte8.o": 0.068,
"build/src/libgte9.o": 0.076,
"build/src/libmcrd1.o": 0.079,
"build/src/libmcrd2.o": 0.075,
"build/src/libpad1.o": 0.073,
"build/src/libgte6.o": 0.059,
"build/src/libgte7.o": 0.069,
"build/src/libgte8.o": 0.063,
"build/src/libgte9.o": 0.062,
"build/src/libmcrd1.o": 0.077,
"build/src/libmcrd2.o": 0.063,
"build/src/libpad1.o": 0.068,
"build/src/libpad2.o": 0.073,
"build/src/sgap.o": 0.084,
"build/src/sgap_2.o": 0.072,
"build/src/sgap_3.o": 0.067,
"build/src/sgap_4.o": 0.072,
"build/src/sgap_5.o": 0.067,
"build/src/sgap_6.o": 0.063,
"build/src/sgap_8.o": 0.067,
"build/src/snd1.o": 0.077,
"build/src/snd10.o": 0.065,
"build/src/sgap.o": 0.071,
"build/src/sgap_2.o": 0.055,
"build/src/sgap_3.o": 0.071,
"build/src/sgap_4.o": 0.075,
"build/src/sgap_5.o": 0.064,
"build/src/sgap_6.o": 0.059,
"build/src/sgap_8.o": 0.069,
"build/src/snd1.o": 0.068,
"build/src/snd10.o": 0.072,
"build/src/snd11.o": 0.061,
"build/src/snd12.o": 0.066,
"build/src/snd2.o": 0.071,
"build/src/snd3.o": 0.07,
"build/src/snd4.o": 0.068,
"build/src/snd5.o": 0.072,
"build/src/snd6.o": 0.081,
"build/src/snd7.o": 0.068,
"build/src/snd8.o": 0.069,
"build/src/snd9.o": 0.076,
"build/src/ov_SC03_014/ov_SC03_014.o": 0.132,
"build/src/ov_SC03_014/ov_SC03_014_after.o": 0.519,
"build/src/ov_SC03_014/ov_SC03_014_jr_8012ACE0.o": 0.406,
"build/src/ov_SC03_014/ov_SC03_014_jr_80135888.o": 0.059,
"build/src/ov_SC03_014/ov_SC03_014_jr_80135A4C.o": 0.063,
"build/src/ov_SC03_014/ov_SC03_014_jr_80135D20.o": 0.121,
"build/src/ov_SC03_014/ov_SC03_014_jr_801380E0.o": 0.162,
"build/src/ov_SC03_014/ov_SC03_014_jr_8013C98C.o": 0.122,
"build/src/ov_SC03_014/ov_SC03_014_jr_8013F350.o": 0.109,
"build/src/ov_SC03_014/ov_SC03_014_jr_8013FFD8.o": 0.078,
"build/src/ov_SC03_014/ov_SC03_014_jr_80140608.o": 0.183,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015444C.o": 0.076,
"build/src/ov_SC03_014/ov_SC03_014_jr_80154C24.o": 0.166,
"build/src/ov_SC03_014/ov_SC03_014_jr_801588CC.o": 0.096,
"build/src/ov_SC03_014/ov_SC03_014_jr_80159C84.o": 0.074,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015A3C8.o": 0.069,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015AE2C.o": 0.102,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015C32C.o": 0.473,
"build/src/ov_SC03_014/ov_SC03_014_jr_8016AB6C.o": 0.255,
"build/src/ov_SC03_014/ov_SC03_014_jr_80171B4C.o": 0.103,
"build/src/ov_SC03_014/ov_SC03_014_jr_801734BC.o": 0.2,
"build/src/ov_SC03_014/ov_SC03_014_jr_801789AC.o": 0.061,
"build/src/ov_SC03_014/ov_SC03_014_jr_80178D40.o": 0.103,
"build/src/ov_SC03_014/ov_SC03_014_jr_8017A4AC.o": 0.073,
"build/src/ov_SC03_014/ov_SC03_014_jr_8017AE2C.o": 0.169,
"build/src/ov_SC03_014/ov_SC03_014_jr_8017EB7C.o": 0.205,
"build/src/ov_SC03_014/ov_SC03_014_jr_80184440.o": 0.052,
"build/src/ov_SC03_014/ov_SC03_014_jr_801848E4.o": 0.268,
"build/src/ov_SC03_014/ov_SC03_014_o0b.o": 0.057,
"build/src/ov_SC03_014/ov_SC03_014_o0c.o": 0.059,
"build/src/ov_SC03_014/ov_SC03_014_o0d.o": 0.051,
"build/src/snd12.o": 0.059,
"build/src/snd2.o": 0.076,
"build/src/snd3.o": 0.074,
"build/src/snd4.o": 0.066,
"build/src/snd5.o": 0.068,
"build/src/snd6.o": 0.073,
"build/src/snd7.o": 0.066,
"build/src/snd8.o": 0.065,
"build/src/snd9.o": 0.063,
"build/src/ov_SC03_014/ov_SC03_014.o": 0.122,
"build/src/ov_SC03_014/ov_SC03_014_after.o": 0.501,
"build/src/ov_SC03_014/ov_SC03_014_jr_8012ACE0.o": 0.393,
"build/src/ov_SC03_014/ov_SC03_014_jr_80135888.o": 0.065,
"build/src/ov_SC03_014/ov_SC03_014_jr_80135A4C.o": 0.061,
"build/src/ov_SC03_014/ov_SC03_014_jr_80135D20.o": 0.124,
"build/src/ov_SC03_014/ov_SC03_014_jr_801380E0.o": 0.156,
"build/src/ov_SC03_014/ov_SC03_014_jr_8013C98C.o": 0.101,
"build/src/ov_SC03_014/ov_SC03_014_jr_8013F350.o": 0.091,
"build/src/ov_SC03_014/ov_SC03_014_jr_8013FFD8.o": 0.099,
"build/src/ov_SC03_014/ov_SC03_014_jr_80140608.o": 0.18,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015444C.o": 0.084,
"build/src/ov_SC03_014/ov_SC03_014_jr_80154C24.o": 0.176,
"build/src/ov_SC03_014/ov_SC03_014_jr_801588CC.o": 0.092,
"build/src/ov_SC03_014/ov_SC03_014_jr_80159C84.o": 0.068,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015A3C8.o": 0.063,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015AE2C.o": 0.082,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015C32C.o": 0.463,
"build/src/ov_SC03_014/ov_SC03_014_jr_8016AB6C.o": 0.258,
"build/src/ov_SC03_014/ov_SC03_014_jr_80171B4C.o": 0.099,
"build/src/ov_SC03_014/ov_SC03_014_jr_801734BC.o": 0.186,
"build/src/ov_SC03_014/ov_SC03_014_jr_801789AC.o": 0.058,
"build/src/ov_SC03_014/ov_SC03_014_jr_80178D40.o": 0.094,
"build/src/ov_SC03_014/ov_SC03_014_jr_8017A4AC.o": 0.064,
"build/src/ov_SC03_014/ov_SC03_014_jr_8017AE2C.o": 0.172,
"build/src/ov_SC03_014/ov_SC03_014_jr_8017EB7C.o": 0.188,
"build/src/ov_SC03_014/ov_SC03_014_jr_80184440.o": 0.05,
"build/src/ov_SC03_014/ov_SC03_014_jr_801848E4.o": 0.253,
"build/src/ov_SC03_014/ov_SC03_014_o0b.o": 0.063,
"build/src/ov_SC03_014/ov_SC03_014_o0c.o": 0.056,
"build/src/ov_SC03_014/ov_SC03_014_o0d.o": 0.057,
"build/src/ov_SC03_014/ov_SC03_014_o0e.o": 0.067,
"build/src/ov_SC03_015/ov_SC03_015.o": 0.112,
"build/src/ov_SC03_015/ov_SC03_015_after.o": 0.509,
"build/src/ov_SC03_015/ov_SC03_015_jr_8012ACE0.o": 0.391,
"build/src/ov_SC03_015/ov_SC03_015_jr_80135888.o": 0.054,
"build/src/ov_SC03_015/ov_SC03_015_jr_80135A4C.o": 0.052,
"build/src/ov_SC03_015/ov_SC03_015_jr_80135D20.o": 0.1,
"build/src/ov_SC03_015/ov_SC03_015_jr_801380E0.o": 0.143,
"build/src/ov_SC03_015/ov_SC03_015_jr_8013C98C.o": 0.108,
"build/src/ov_SC03_015/ov_SC03_015_jr_8013F350.o": 0.076,
"build/src/ov_SC03_015/ov_SC03_015_jr_8013FFD8.o": 0.063,
"build/src/ov_SC03_015/ov_SC03_015_jr_80140608.o": 0.182,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015444C.o": 0.076,
"build/src/ov_SC03_015/ov_SC03_015_jr_80154C24.o": 0.158,
"build/src/ov_SC03_015/ov_SC03_015_jr_801588CC.o": 0.091,
"build/src/ov_SC03_015/ov_SC03_015_jr_80159C84.o": 0.07,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015A3C8.o": 0.068,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015AE2C.o": 0.081,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015C32C.o": 0.443,
"build/src/ov_SC03_015/ov_SC03_015_jr_8016AB6C.o": 0.262,
"build/src/ov_SC03_015/ov_SC03_015_jr_80171B4C.o": 0.095,
"build/src/ov_SC03_015/ov_SC03_015_jr_801734BC.o": 0.21,
"build/src/ov_SC03_015/ov_SC03_015_jr_801789AC.o": 0.057,
"build/src/ov_SC03_015/ov_SC03_015_jr_80178D40.o": 0.094,
"build/src/ov_SC03_015/ov_SC03_015_jr_8017A4AC.o": 0.068,
"build/src/ov_SC03_015/ov_SC03_015_jr_8017AE2C.o": 0.168,
"build/src/ov_SC03_015/ov_SC03_015_jr_8017EB7C.o": 0.209,
"build/src/ov_SC03_015/ov_SC03_015_jr_80184440.o": 0.048,
"build/src/ov_SC03_015/ov_SC03_015.o": 0.131,
"build/src/ov_SC03_015/ov_SC03_015_after.o": 0.494,
"build/src/ov_SC03_015/ov_SC03_015_jr_8012ACE0.o": 0.407,
"build/src/ov_SC03_015/ov_SC03_015_jr_80135888.o": 0.046,
"build/src/ov_SC03_015/ov_SC03_015_jr_80135A4C.o": 0.05,
"build/src/ov_SC03_015/ov_SC03_015_jr_80135D20.o": 0.108,
"build/src/ov_SC03_015/ov_SC03_015_jr_801380E0.o": 0.149,
"build/src/ov_SC03_015/ov_SC03_015_jr_8013C98C.o": 0.109,
"build/src/ov_SC03_015/ov_SC03_015_jr_8013F350.o": 0.073,
"build/src/ov_SC03_015/ov_SC03_015_jr_8013FFD8.o": 0.059,
"build/src/ov_SC03_015/ov_SC03_015_jr_80140608.o": 0.174,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015444C.o": 0.072,
"build/src/ov_SC03_015/ov_SC03_015_jr_80154C24.o": 0.159,
"build/src/ov_SC03_015/ov_SC03_015_jr_801588CC.o": 0.077,
"build/src/ov_SC03_015/ov_SC03_015_jr_80159C84.o": 0.059,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015A3C8.o": 0.066,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015AE2C.o": 0.095,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015C32C.o": 0.44,
"build/src/ov_SC03_015/ov_SC03_015_jr_8016AB6C.o": 0.272,
"build/src/ov_SC03_015/ov_SC03_015_jr_80171B4C.o": 0.094,
"build/src/ov_SC03_015/ov_SC03_015_jr_801734BC.o": 0.199,
"build/src/ov_SC03_015/ov_SC03_015_jr_801789AC.o": 0.062,
"build/src/ov_SC03_015/ov_SC03_015_jr_80178D40.o": 0.1,
"build/src/ov_SC03_015/ov_SC03_015_jr_8017A4AC.o": 0.056,
"build/src/ov_SC03_015/ov_SC03_015_jr_8017AE2C.o": 0.16,
"build/src/ov_SC03_015/ov_SC03_015_jr_8017EB7C.o": 0.161,
"build/src/ov_SC03_015/ov_SC03_015_jr_80184440.o": 0.053,
"build/src/ov_SC03_015/ov_SC03_015_jr_801848E4.o": 0.27,
"build/src/ov_SC03_015/ov_SC03_015_o0b.o": 0.062,
"build/src/ov_SC03_015/ov_SC03_015_o0c.o": 0.058,
"build/src/ov_SC03_015/ov_SC03_015_o0d.o": 0.058,
"build/src/ov_SC03_015/ov_SC03_015_o0e.o": 0.064,
"build/src/ov_SC04_011/ov_SC04_011.o": 0.115,
"build/src/ov_SC04_011/ov_SC04_011_after.o": 0.419,
"build/src/ov_SC04_011/ov_SC04_011_jr_8012ACE0.o": 0.346,
"build/src/ov_SC04_011/ov_SC04_011_jr_80135888.o": 0.047,
"build/src/ov_SC04_011/ov_SC04_011_jr_80135A4C.o": 0.056,
"build/src/ov_SC04_011/ov_SC04_011_jr_80135D20.o": 0.114,
"build/src/ov_SC04_011/ov_SC04_011_jr_801380E0.o": 0.147,
"build/src/ov_SC04_011/ov_SC04_011_jr_8013C98C.o": 0.107,
"build/src/ov_SC04_011/ov_SC04_011_jr_8013F350.o": 0.076,
"build/src/ov_SC04_011/ov_SC04_011_jr_8013FFD8.o": 0.063,
"build/src/ov_SC04_011/ov_SC04_011_jr_80140608.o": 0.172,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015444C.o": 0.071,
"build/src/ov_SC04_011/ov_SC04_011_jr_80154C24.o": 0.17,
"build/src/ov_SC04_011/ov_SC04_011_jr_801588CC.o": 0.092,
"build/src/ov_SC04_011/ov_SC04_011_jr_80159C84.o": 0.066,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015A3C8.o": 0.07,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015AE2C.o": 0.092,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015C32C.o": 0.34,
"build/src/ov_SC04_011/ov_SC04_011_jr_8016AB6C.o": 0.214,
"build/src/ov_SC04_011/ov_SC04_011_jr_80171B4C.o": 0.095,
"build/src/ov_SC04_011/ov_SC04_011_jr_801734BC.o": 0.175,
"build/src/ov_SC04_011/ov_SC04_011_jr_801789AC.o": 0.057,
"build/src/ov_SC04_011/ov_SC04_011_jr_80178D40.o": 0.09,
"build/src/ov_SC04_011/ov_SC04_011_jr_8017A4AC.o": 0.061,
"build/src/ov_SC04_011/ov_SC04_011_jr_8017AE2C.o": 0.085,
"build/src/ov_SC04_011/ov_SC04_011_jr_8017D494.o": 0.426,
"build/src/ov_SC04_011/ov_SC04_011_o0b.o": 0.05,
"build/src/ov_SC04_011/ov_SC04_011_o0c.o": 0.065
"build/src/ov_SC03_015/ov_SC03_015_o0b.o": 0.056,
"build/src/ov_SC03_015/ov_SC03_015_o0c.o": 0.064,
"build/src/ov_SC03_015/ov_SC03_015_o0d.o": 0.054,
"build/src/ov_SC03_015/ov_SC03_015_o0e.o": 0.068,
"build/src/ov_SC04_011/ov_SC04_011.o": 0.128,
"build/src/ov_SC04_011/ov_SC04_011_after.o": 0.411,
"build/src/ov_SC04_011/ov_SC04_011_jr_8012ACE0.o": 0.329,
"build/src/ov_SC04_011/ov_SC04_011_jr_80135888.o": 0.049,
"build/src/ov_SC04_011/ov_SC04_011_jr_80135A4C.o": 0.052,
"build/src/ov_SC04_011/ov_SC04_011_jr_80135D20.o": 0.113,
"build/src/ov_SC04_011/ov_SC04_011_jr_801380E0.o": 0.138,
"build/src/ov_SC04_011/ov_SC04_011_jr_8013C98C.o": 0.101,
"build/src/ov_SC04_011/ov_SC04_011_jr_8013F350.o": 0.069,
"build/src/ov_SC04_011/ov_SC04_011_jr_8013FFD8.o": 0.061,
"build/src/ov_SC04_011/ov_SC04_011_jr_80140608.o": 0.16,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015444C.o": 0.062,
"build/src/ov_SC04_011/ov_SC04_011_jr_80154C24.o": 0.159,
"build/src/ov_SC04_011/ov_SC04_011_jr_801588CC.o": 0.078,
"build/src/ov_SC04_011/ov_SC04_011_jr_80159C84.o": 0.061,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015A3C8.o": 0.059,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015AE2C.o": 0.084,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015C32C.o": 0.335,
"build/src/ov_SC04_011/ov_SC04_011_jr_8016AB6C.o": 0.211,
"build/src/ov_SC04_011/ov_SC04_011_jr_80171B4C.o": 0.098,
"build/src/ov_SC04_011/ov_SC04_011_jr_801734BC.o": 0.177,
"build/src/ov_SC04_011/ov_SC04_011_jr_801789AC.o": 0.052,
"build/src/ov_SC04_011/ov_SC04_011_jr_80178D40.o": 0.087,
"build/src/ov_SC04_011/ov_SC04_011_jr_8017A4AC.o": 0.064,
"build/src/ov_SC04_011/ov_SC04_011_jr_8017AE2C.o": 0.092,
"build/src/ov_SC04_011/ov_SC04_011_jr_8017D494.o": 0.414,
"build/src/ov_SC04_011/ov_SC04_011_o0b.o": 0.044,
"build/src/ov_SC04_011/ov_SC04_011_o0c.o": 0.051
},
"ok": true,
"seconds": 2.3
"seconds": 2.2
}
+1
View File
@@ -29418,3 +29418,4 @@
{"ts": "2026-09-11 02:04:50", "label": "s104_e10", "rung": "E", "calib": {"head": "7e7d9c523", "stamp": "15956e4a96c4"}, "tu": "src/800.c", "fn": "func_80021174", "addr": 2147619188, "aliases": null, "header": false, "includers": 0, "nhash_before": "ff01f6d8bf282c431329206e903f5ab8fb9a952c", "nhash_after": "27b3d5dffa268ac2b379b13c87142ab329756edf", "source": ".run/P36/agents/main__func_80021174/body.c", "verdict": "LEVER-FREE", "sites": [], "compiles": 1, "seconds": 0.438, "objects": ["build/src/800.o"], "before_text": "s32 func_80021174(s32 a0, s32 a1)\n{\n s32 sp[4];\n s32 lim;\ns32 *p_a3;\nregister s32 ret __asm__(\"$2\"); // !FAKE: pin $2 \u2014 NEEDED DIFFERS (P36 rung B tus9)\n if (a0 == 0x7FFF7FFF)\n {\n return 1;\n }\n p_a3 = D_800AE688;\n__asm__ volatile ( \"lw $12, 0( %0 );\" \"lw $13, 4( %0 );\" \"ctc2 $12, $0;\" \"ctc2 $13, $1;\" \"lw $12, 8( %0 );\" \"lw $13, 12( %0 );\" \"lw $14, 16( %0 );\" \"ctc2 $12, $2;\" \"ctc2 $13, $3;\" \"ctc2 $14, $4;\" \"lw $12, 20( %0 );\" \"lw $13, 24( %0 );\" \"ctc2 $12, $5;\" \"lw $14, 28( %0 );\" \"ctc2 $13, $6;\" \"ctc2 $14, $7\" : : \"r\"( p_a3 ) : \"$12\", \"$13\", \"$14\", \"memory\" ); // !FAKE: gte direct \u2014 clobbers ['memory'] (gte_SetRotTransMatrix_m) beyond Sony's (P36 T5 gte1)\n__asm__ volatile ( \"lhu $13, 4( %0 );\" \"lhu $12, 0( %0 );\" \"sll $13, $13, 16;\" \"or $12, $12, $13;\" \"mtc2 $12, $0;\" \"lwc2 $1, 8( %0 )\" : : \"r\"( (SV_80021174 *)a1 ) : \"$12\", \"$13\", \"memory\" ); // !FAKE: gte direct \u2014 clobbers ['memory'] (gte_ldlv0_m) beyond Sony's (P36 T5 gte1)\n__asm__ volatile ( \"nop;\" \"nop;\" \"rtps\" : : : \"memory\" ); // !FAKE: gte direct \u2014 clobbers ['memory'] (gte_rtps_m) beyond Sony's (P36 T5 gte1)\ngte_stsxy(sp);\ngte_stflg(&sp[1]);\ngte_stszotz(&sp[2]);\n if (sp[1] < 0)\n {\n return 0;\n }\n ret = 0;\n lim = (s16) a0;\n ;\n if ((*((s16 *) sp)) <= (-lim))\n {\n return ret;\n }\n if (lim < (*((s16 *) sp)))\n {\n return ret;\n }\n a0 = a0 >> 16;\n a1 = *((s16 *) (((char *) sp) + 2));\n if (a1 <= (-a0))\n {\n return ret;\n }\n ret = !(a0 < a1);\n return ret;\n}\n", "after_text": "s32 func_80021174(s32 a0, s32 a1)\n{\n struct {\n s16 sx, sy;\n s32 flag;\n s32 otz;\n } r;\n s32 lim;\n\n if (a0 == 0x7FFF7FFF) {\n return 1;\n }\n gte_SetRotTransMatrix(D_800AE688);\n gte_ldlv0((SV_80021174 *)a1);\n gte_rtps();\n gte_stsxy(&r.sx);\n gte_stflg(&r.flag);\n gte_stszotz(&r.otz);\n if (r.flag < 0) {\n return 0;\n }\n lim = (s16)a0;\n if (-lim < r.sx && r.sx <= lim && -(a0 >> 16) < r.sy && r.sy <= (a0 >> 16)) {\n return 1;\n }\n return 0;\n}\n"}
{"ts": "2026-09-11 02:05:20", "label": "s104_e13", "rung": "E", "calib": {"head": "c37ef9c15", "stamp": "15956e4a96c4"}, "tu": "src/ov_SC02_017/ov_SC02_017_jr_8017DF34.c", "fn": "func_80185574", "addr": 2149078388, "aliases": null, "header": false, "includers": 0, "nhash_before": "9d423d4275920c5121fd435c05fc402c1fba561c", "nhash_after": "0bee5bbfb3d4a6371c187915f491596b1b83bab8", "source": ".run/P36/agents/ov_SC02_017__func_80185574/body.c", "verdict": "LEVER-FREE", "sites": [], "compiles": 1, "seconds": 0.202, "objects": ["build/src/ov_SC02_017/ov_SC02_017_jr_8017DF34.o"], "before_text": "void func_80185574(s32 arg0) {\n s32 self;\n register s32 zr __asm__(\"$0\"); // !FAKE: pin $0 \u2014 NEEDED DIFFERS (P36 rung B tus8)\n s32 mask;\n s32 flag;\n\n self = arg0 + zr;\n flag = *(s16 *)(self + 0xAA);\n *(s16 *)(self + 0x5C) = 0;\n *(s16 *)(self + 0x98) = 0;\n *(s32 *)(self + 0x1C) = 0;\n if (flag == 0) {\n s32 p;\n register s32 f __asm__(\"$2\"); // !FAKE: pin $2 \u2014 NEEDED DIFFERS (P36 rung B tus8)\n mask = ~0x80;\n p = *(s32 *)(self + 0x20);\n f = *(s32 *)(self + 0xDC);\n p = *(u16 *)(p + 0x18);\n *(s32 *)(self + 0xDC) = f & mask;\n *(s16 *)(self + 0x100) = p;\n } else {\n s32 q;\n register s32 g __asm__(\"$3\"); // !FAKE: pin $3 \u2014 NEEDED DIFFERS (P36 rung B tus8)\n q = *(s32 *)(self + 0x20);\n g = *(s32 *)(self + 0xDC);\n q = *(u16 *)(q + 0x18);\n g |= 0x80;\n *(s32 *)(self + 0xDC) = g;\n *(s16 *)(self + 0x104) = q;\n }\n}\n", "after_text": "void func_80185574(s32 arg0) {\n *(s16 *)(arg0 + 0x5C) = 0;\n *(s16 *)(arg0 + 0x98) = 0;\n *(s32 *)(arg0 + 0x1C) = 0;\n if (*(s16 *)(arg0 + 0xAA) == 0) {\n *(s16 *)(arg0 + 0x100) = *(u16 *)(*(s32 *)(arg0 + 0x20) + 0x18);\n *(s32 *)(arg0 + 0xDC) &= ~0x80;\n } else {\n *(s16 *)(arg0 + 0x104) = *(u16 *)(*(s32 *)(arg0 + 0x20) + 0x18);\n *(s32 *)(arg0 + 0xDC) |= 0x80;\n }\n}\n"}
{"ts": "2026-09-11 02:05:38", "label": "s104_e13", "rung": "E", "calib": {"head": "580d7bfd8", "stamp": "15956e4a96c4"}, "tu": "src/ov_SC02_017/ov_SC02_017_jr_8017DF34.c", "fn": "func_801842E0", "addr": 2149073632, "aliases": null, "header": false, "includers": 0, "nhash_before": "5a24bef2d7747b0700b66f9dd4d611717951cb5d", "nhash_after": "11c0fffaf83ae81cbc2d75d40ee56dacc3c3679b", "source": ".run/P36/agents/ov_SC02_017__func_801842E0/body.c", "verdict": "LEVER-FREE", "sites": [], "compiles": 1, "seconds": 0.195, "objects": ["build/src/ov_SC02_017/ov_SC02_017_jr_8017DF34.o"], "before_text": "void func_801842E0(void) {\n register s32 t __asm__(\"$3\"); // !FAKE: pin $3 \u2014 NEEDED DIFFERS (P36 rung B tus8)\n s32 a0;\n s32 v0;\n s32 a1;\n register s32 a1p __asm__(\"$5\"); // !FAKE: pin $5 \u2014 NEEDED DIFFERS (P36 rung B tus8)\n\n t = D_800B99DA;\n if (t != D_8018EF88) {\n a0 = (s16)D_80126B62;\n D_8018EF88 = t;\n\n if (a0 >= -0x7FF) {\n t = D_8018EF84;\n } else if (a0 < -0xC00) {\n t = D_8018EF84;\n } else {\n SV3_8012CC88 sp10;\n SV3_8012CC88 sp18;\n\n sp10.vx = D_80126B5E;\n sp10.vy = 0;\n sp10.vz = D_80126B66;\n sp18.vx = 0x11;\n sp18.vz = -0x3AA;\n sp18.vy = 0;\n\n t = func_800132BC((s32)&sp10, (s32)&sp18);\n }\n\n a0 = D_8018EF84;\n if (t >= a0) {\n func_8002D4C8(4, 0x5E5);\n } else {\n t = a0 - t;\n t -= 0x10000;\n\n if (t >= 0) {\n v0 = t << 7;\n } else {\n t = -t;\n v0 = t << 7;\n }\n v0 = v0 - t;\n t = v0 / a0;\n\n if (t <= 0) {\n t = 0;\n } else if (t >= 0x80) {\n t = 0x7F;\n }\n\n if (D_801270C8 != 0) {\n a1 = (u32)t >> 31;\n v0 = t + a1;\n t = v0 >> 1;\n }\n\n a0 = 0x5E5;\n __asm__ __volatile__(\"\" : \"=r\"(a0) : \"0\"(a0)); // !FAKE: launder \u2014 NEEDED DIFFERS (P36 rung B tus8)\n a1p = t | 0x1000;\n a1p = (u16)a1p;\n func_8002D4C8(a0, a1p);\n }\n }\n}\n", "after_text": "void func_801842E0(void) {\n s32 t;\n s32 y;\n\n t = D_800B99DA;\n if (t != D_8018EF88) {\n y = (s16)D_80126B62;\n D_8018EF88 = t;\n\n if (y >= -0x7FF) {\n t = D_8018EF84;\n } else if (y < -0xC00) {\n t = D_8018EF84;\n } else {\n SV3_8012CC88 sp10;\n SV3_8012CC88 sp18;\n\n sp10.vx = D_80126B5E;\n sp10.vy = 0;\n sp10.vz = D_80126B66;\n sp18.vx = 0x11;\n sp18.vz = -0x3AA;\n sp18.vy = 0;\n\n t = func_800132BC((s32)&sp10, (s32)&sp18);\n }\n\n if (t >= D_8018EF84) {\n func_8002D4C8(4, 0x5E5);\n } else {\n t = D_8018EF84 - t - 0x10000;\n if (t < 0) {\n t = -t;\n }\n t = t * 127 / D_8018EF84;\n\n if (t <= 0) {\n t = 0;\n } else if (t >= 0x80) {\n t = 0x7F;\n }\n\n if (D_801270C8 != 0) {\n t /= 2;\n }\n\n func_8002D4C8(0x5E5, (u16)(t | 0x1000));\n }\n }\n}\n"}
{"ts": "2026-09-11 02:14:31", "label": "s104_e11", "rung": "E", "calib": {"head": "860ed9153", "stamp": "15956e4a96c4"}, "tu": "src/ov_SC04_011/ov_SC04_011_jr_8017D494.c", "fn": "func_80183E20", "addr": 2149072416, "aliases": null, "header": false, "includers": 0, "nhash_before": "e58fc156bf0394cf9c2d78148df0a8015d1cd7a7", "nhash_after": "2346bbbe0d370caa86a647c595e42a557989f555", "source": ".run/P36/agents/ov_SC04_011__func_80183E20/body.c", "verdict": "LEVER-FREE", "sites": [], "compiles": 1, "seconds": 0.389, "objects": ["build/src/ov_SC04_011/ov_SC04_011_jr_8017D494.o"], "before_text": "void aF80183E20(void *a0) {\n extern u16 D_801EFD40;\n extern u16 aEFD24 __asm__(\"D_801EFD24\");\n u16 flags;\n u16 cnt;\n s32 v0;\n s32 v1;\n s32 v1b;\n flags = D_801EFD40;\n if (flags & 0x80) {\n v0 = *(u16 *)((u8 *)a0 + 0x76);\n v1 = *(u16 *)((u8 *)a0 + 0x60);\n *(u16 *)((u8 *)a0 + 0x60) = 0;\n v0 = v0 - v1;\n *(u16 *)((u8 *)a0 + 0x76) = v0;\n __asm__(\"\":::\"memory\"); // !FAKE: barrier memory \u2014 NEEDED DIFFERS (P36 rung B t3_tus1)\n v1b = *(s16 *)((u8 *)a0 + 0x76);\n D_801EFD40 = flags & 0xFF7F;\n if (v1b < 0) {\n *(u16 *)((u8 *)a0 + 0x76) = 0;\n }\n }\n cnt = aEFD24;\n if (cnt != 0) {\n cnt = cnt - 1;\n aEFD24 = cnt;\n if (cnt == 0) {\n func_80186AB8();\n }\n }\n}\n", "after_text": "void aF80183E20(void *a0) {\n extern u16 D_801EFD40;\n extern u16 aEFD24 __asm__(\"D_801EFD24\");\n u16 flags;\n u16 cnt;\n s32 v0;\n s32 v1;\n flags = D_801EFD40;\n if (flags & 0x80) {\n v0 = *(u16 *)((u8 *)a0 + 0x76);\n v1 = *(u16 *)((u8 *)a0 + 0x60);\n *(u16 *)((u8 *)a0 + 0x60) = 0;\n *(u16 *)((u8 *)a0 + 0x76) = v0 - v1;\n D_801EFD40 = flags & 0xFF7F;\n if (((s16 *)a0)[0x3B] < 0) {\n *(u16 *)((u8 *)a0 + 0x76) = 0;\n }\n }\n cnt = aEFD24;\n if (cnt != 0) {\n cnt = cnt - 1;\n aEFD24 = cnt;\n if (cnt == 0) {\n func_80186AB8();\n }\n }\n}\n"}
+2 -6
View File
@@ -7397,18 +7397,14 @@ void aF80183E20(void *a0) {
u16 cnt;
s32 v0;
s32 v1;
s32 v1b;
flags = D_801EFD40;
if (flags & 0x80) {
v0 = *(u16 *)((u8 *)a0 + 0x76);
v1 = *(u16 *)((u8 *)a0 + 0x60);
*(u16 *)((u8 *)a0 + 0x60) = 0;
v0 = v0 - v1;
*(u16 *)((u8 *)a0 + 0x76) = v0;
__asm__("":::"memory"); // !FAKE: barrier memory — NEEDED DIFFERS (P36 rung B t3_tus1)
v1b = *(s16 *)((u8 *)a0 + 0x76);
*(u16 *)((u8 *)a0 + 0x76) = v0 - v1;
D_801EFD40 = flags & 0xFF7F;
if (v1b < 0) {
if (((s16 *)a0)[0x3B] < 0) {
*(u16 *)((u8 *)a0 + 0x76) = 0;
}
}