phase-36: S104 s104_e26 — func_8017F024 banked at 0 through the whole-object gate + propagated — 2 levers → 0: the actor passed to the dispatch handler (sched.c:2469-2545; jump.c:425-462)

This commit is contained in:
Drew T
2026-09-11 03:58:29 -06:00
parent 2b9250a50e
commit 195dee8a0c
11 changed files with 847 additions and 193 deletions
@@ -0,0 +1,195 @@
void func_8017E764(s32 param_1)
{
s16 pos[4];
s16 adv[4];
s16 mv[4];
s32 src3[4];
s32 dst2[4];
s32 mid[4];
u16 angle;
s32 angle2;
s32 t;
s32 i;
s32 j;
s32 k;
s32 m;
s32 self;
s32 found;
s32 flag;
s32 val;
s32 hundred;
u16 *row;
u16 *np;
s32 *midp;
u8 *ent;
i = *(s16 *)&D_80126B66;
if ((u32)(i - 0x44E) >= 0x32C5U) {
if (*(s16 *)(param_1 + 0x84) != 0) {
*(s16 *)(param_1 + 0x84) = 0;
func_8002D4C8(4, 0x931);
}
} else {
*(s16 *)(param_1 + 0x84) = 1;
if (i < 0x5CE) {
func_8002D4C8(0x931, (((i - 0x44E) * 0x7F / 0x180) | 0x1000) & 0xFFFF);
} else if (i >= 0x3593) {
func_8002D4C8(0x931, (((i - 0x3592) * 0x7F / 0x180) | 0x1000) & 0xFFFF);
} else {
func_8002D4C8(0x931, 0x107F);
}
}
if (*(u16 *)(param_1 + 2) == 0) {
hundred = 100;
row = &D_8018A5A2;
for (i = 0x13F; i >= 0; i--) {
*row = hundred;
row += 4;
}
*(s32 *)(param_1 + 0xCC) = func_8012C658(0x238, 0xA, param_1);
*(u16 *)(param_1 + 2) += 1;
*(u16 *)(param_1 + 0xE) += 0x5F0;
}
ent = D_801202A0;
i = 0;
pos[0] = *(u16 *)(param_1 + 6);
{
u16 tt;
/* `val` is the loop's table-value temp reused here (both live in $v1 in the target): its second set keeps
* this address load from sched1's birthing priority (sched.c:2469-2545), so it is scheduled above the pos[0]
* store and global-alloc gives it $v1 (P36 S104 e26). */
val = (s32)&D_80126B66;
tt = *(volatile u16 *)val; /* a byte-needed volatile, kept as ordinary C (gate-1 decision 3): a volatile MEM fails combine's recog (combine.c:491 init_recog_no_volatile, recog.c:807), so the address stays in a register (la + lhu 0(reg)) (P36 S104 e26 minimum-lever) */
pos[1] = *(u16 *)(param_1 + 0xA);
pos[2] = tt + 8;
}
np = (u16 *)D_801152A8;
t = (ratan2(*(s16 *)&D_80126B62 - *(s16 *)(param_1 + 0xA),
*(s16 *)&D_80126B5E - *(s16 *)(param_1 + 6)) - 0x400) & 0xFFF;
angle = t;
adv[0] = pos[0] - (func_8004787C(t) >> 3);
adv[1] = pos[1] + (func_80047948(t) >> 3);
adv[2] = pos[2];
self = *(s32 *)(param_1 + 0xCC);
angle2 = t;
*(u16 *)(self + 0x5C) = 0;
for (; i < 0x60; ent += 0x10C, i++) {
if (*(u16 *)ent != 0x238) {
continue;
}
flag = *(u16 *)(ent + 0x34);
if (flag != 1) {
continue;
}
if (func_80135888(*(s32 *)(ent + 0x20), *(s32 *)(ent + 0x58), (s32)pos, (s32)adv) == 0) {
continue;
}
found = 0;
*(u16 *)(self + 0x5C) = 0x4C08;
if (*(u8 *)(self + 0x74) != 0) {
*(s32 *)(param_1 + 0xDC) = flag;
if (*(s16 *)(ent + 0x70) == 0) {
mv[0] = D_80126B5E;
mv[1] = D_80126B62;
mv[2] = D_80126B66;
np[0] = -np[0];
np[1] = -np[1];
np[2] = -np[2];
func_8012F568(1, 0x4018, 0, 0x20, (s32)mv, (s32)np);
}
j = *(s16 *)&D_80126B66 - *(s16 *)(param_1 + 0xE) + 0x98;
j = j / 0x130;
val = (&D_8018A5A0)[j * 0x20];
if (val == flag) {
goto Lup;
}
if (val < 2) {
if (val == 0) {
*(u16 *)(*(s32 *)(self + 0x20) + 0x14) = angle;
}
goto Ldone;
}
if (val == 2) {
goto Ldown;
}
goto Ldone;
Lup:
*(s16 *)(*(s32 *)(self + 0x20) + 0x14) =
*(u16 *)(*(s32 *)(self + 0x20) + 0x14) + 0x10;
found = 1;
*(u16 *)(*(s32 *)(self + 0x20) + 0x14) &= 0xFFF;
goto Ldone;
Ldown:
*(s16 *)(*(s32 *)(self + 0x20) + 0x14) =
*(u16 *)(*(s32 *)(self + 0x20) + 0x14) - 0x10;
found = 1;
*(u16 *)(*(s32 *)(self + 0x20) + 0x14) &= 0xFFF;
} else {
*(u16 *)(*(s32 *)(self + 0x20) + 0x14) = angle;
}
Ldone:
*(s32 *)(self + 4) = *(s32 *)(param_1 + 4)
- func_8004787C(*(s16 *)(*(s32 *)(self + 0x20) + 0x14)) * 0x1A40;
*(s32 *)(self + 8) = *(s32 *)(param_1 + 8)
+ func_80047948(*(s16 *)(*(s32 *)(self + 0x20) + 0x14)) * 0x1A40;
*(s32 *)(self + 0xC) = D_80126B64;
if (found != 0) {
src3[0] = *(s32 *)&D_80126B5C;
src3[1] = D_80126B60;
src3[2] = *(s32 *)(&D_80126B5C + 8);
midp = mid;
func_8012F0BC((s32 *)(*(s32 *)(self + 0x20) + 0x34), src3, midp);
midp = 0;
func_8012B2CC(self);
func_8012F1A4((s32 *)(*(s32 *)(self + 0x20) + 0x34), mid, dst2);
*(s32 *)&D_80126B5C = dst2[0];
D_80126B60 = dst2[1];
*(u16 *)(*(s32 *)(self + 0x20) + 0x14) = angle;
*(s32 *)(self + 4) = *(s32 *)(param_1 + 4)
- func_8004787C(*(s16 *)(*(s32 *)(self + 0x20) + 0x14)) * 0x1A40;
*(s32 *)(self + 8) = *(s32 *)(param_1 + 8)
+ func_80047948(*(s16 *)(*(s32 *)(self + 0x20) + 0x14)) * 0x1A40;
func_8012B2CC(self);
}
if (*(s32 *)(param_1 + 0xDC) != 0) {
i = *(s16 *)&D_80126B5E - *(s16 *)(param_1 + 6);
j = *(s16 *)&D_80126B62 - *(s16 *)(param_1 + 0xA);
if (i * i + j * j > 0x14D10) {
*(s32 *)&D_80126B5C = *(s32 *)(param_1 + 4) - func_8004787C(angle2) * 0x1240;
D_80126B60 = *(s32 *)(param_1 + 8) + func_80047948(angle2) * 0x1240;
}
}
}
if (*(u16 *)(self + 0x5C) == 0) {
*(s32 *)(param_1 + 0xDC) = 0;
}
*(u16 *)(param_1 + 0xFC) = (*(u16 *)(param_1 + 0xFC) + 0x10) & 0xFFF;
i = *(s16 *)&D_80126B66 - *(s16 *)(param_1 + 0xE);
j = i / 0x130;
j -= 2;
row = &D_8018A59C + j * 0x20;
for (k = 0; k < 7; k++, j++) {
if ((u32)j < 0x28) {
for (m = 0; m < 8; m++) {
row += 3;
if (*row == 100) {
row -= 3;
ent = (u8 *)func_8012C658(0x238, *row, param_1);
if (ent == 0) {
return;
}
*(s32 *)(ent + 0xCC) = (s32)row;
row += 3;
}
*row = j;
row++;
}
} else {
row += 0x20;
}
}
}
@@ -0,0 +1,91 @@
# func_8017E764 (ov_SC06_010_jr_8017A4AC.c) — T7 e26, P36 S104
**Score 0 with ONE marked lever** (minimum-lever, METHOD step 9): levers 2 -> 1. The `$3` pin on `vp` is gone; the
`*(volatile u16 *)` cast (present in body_free.c, so the brief's rule lets it stay; sites.txt calls it NEEDED) is kept
and now marked `// !FAKE: volatile cast …`. Free start 6 = regen best 6. No copies (`0x14D10`, `D_8018A5A2` occur only
here).
The two moves (both needed; each alone scores worse — METHOD step 5):
```
ent = D_801202A0;
+ i = 0;
pos[0] = *(u16 *)(param_1 + 6);
{
- register u16 *vp __asm__("$3");
- u16 t;
- vp = (u16 *)&D_80126B66;
- t = *(volatile u16 *)vp;
+ u16 tt;
+ val = (s32)&D_80126B66; /* the loop's `val` temp, reused */
+ tt = *(volatile u16 *)val; /* !FAKE: volatile cast (kept) */
pos[1] = *(u16 *)(param_1 + 0xA);
- pos[2] = t + 8;
+ pos[2] = tt + 8;
}
- i = 0;
np = (u16 *)D_801152A8;
```
## (a) The residual
Same 438 instructions; the `pos[2] = camZ + 8` load: the target has `move s4,zero; lhu v0,6(s3); la v1,D_80126B66;
sh v0,24(sp); lhu v0,0(v1)` — the address in `$v1`, set ABOVE the pos[0] store. Pin-free: `la v0` after the store,
`lhu v0,0(v0)`, and `move s4,zero` in the load-delay slot.
## (b) The passes and decisions (dumps: scratch/dumps_free, dumps_vB, dumps_tree, dumps_w1, dumps_w4)
1. **Why the address is in a register at all — the volatile.** expand forces the constant address into a pseudo
(`(set r130 (symbol_ref))`, `.rtl`); combine folds that set into its single use `(mem r130)` for every non-volatile
spelling (vA/vB/vC/w5: 437 ins, `lui v0; lhu v0,%lo(v0)`). With a volatile MEM the combined insn cannot be
recognised: combine runs under `init_recog_no_volatile` (combine.c:491) and `general_operand` refuses volatile MEMs
(recog.c:807). No other can_combine_p/try_combine refusal applies to a constant-address load in one block, and the
bytes show no block boundary between the `la` and the load. So the volatile stays (the marked lever).
2. **Why `$v0` without the pin — sched1's birthing boost.** The address set births a pseudo set once
(`birthing_insn_p`, sched.c:2469-2545: `reg_n_sets == 1`), so `adjust_priority` raises it to the block maximum and the
reverse list scheduler places it immediately before the load, AFTER the pos[0] store (free `.sched` T-33: 213 at
`7f000001` over 208). local-alloc then finds `$v0` free (the pos[0] temp died at the store).
3. **Reusing `val`** (the table-value temp of the entity loop, `val = (&D_8018A5A0)[j * 0x20]`, which the target keeps in
`$v1` — `lhu v1,0(at)` at 0x45b4): `reg_n_sets` = 2, no birthing boost (w4 `.sched`: priority 1), and `val` is now
ONE global allocno spanning the loop: `.greg` "83 conflicts: … 2 29" — it conflicts with `$v0`, so global-alloc gives
it `$v1`. Alone (w1) this leaves an ORDER residual of 4: the unboosted set sinks to the TOP of the block in sched1
(lowest LUID), and sched2's LUID tie-break (`rank_for_schedule`, sched.c:2425-2428) keeps `la v1` above
`lhu v0,6(s3)` and `move s4,zero`.
4. **`i = 0;` written before `pos[0]`** gives that insn (231) a LUID below the address set's, so both schedulers break
the priority-1 tie the target's way: `move s4,zero; lhu; la; sh` (w4 = 0). `i = 0` alone (w6): 6.
Proven on bytes (w1 4, w6 6, w4 0, w5 = w4 without the volatile 10) and on dumps (w4 `.sched` priority, `.greg`
conflicts).
## (c) The moves
- Reuse the function-scope `val` for the address (`val = (s32)&D_80126B66; tt = *(volatile u16 *)val;`), deleting `vp`
and its `$3` pin (d14's rule: the target keeps both values in the same register -> one original variable).
- Move `i = 0;` above `pos[0] = …` (statement order = LUID tie-break).
- Kept, marked: the `volatile u16 *` cast (structs phase).
## (d) GENERATOR PROPOSAL
**R-reuse-late-temp**: when a register residual puts an early short-lived pseudo in `$vN` too LOW (e.g. `$v0` where the
target has `$v1`) and the target's `$v1` also holds a function-scope temp later in the body (same register, disjoint
live ranges), rename the early temp to that later variable (cast as needed) — the second set removes sched1's birthing
boost (sched.c:2489-2490) and makes the pair one global allocno whose conflicts pick the target register; then
enumerate the position of each independent constant statement in the block (`i = 0;`, `ent = …`) for the LUID
tie-break. Enumerate: every function-scope local the target keeps in the wanted register.
## (e) What did not work (bytes)
- body_free (volatile, no pin): 6. Statement orders with a block-local `vp` (o1–o4): 6 each — sched1 re-sinks the
boosted set regardless of source order.
- No volatile: vA/vB (plain or pointer local) 10, vC 17 (437 ins, address folded).
- Reusing `j` instead of `val` (w2): 5 (`j` is `$s1`). Inline `pos[2] = *(volatile u16 *)val + 8` with `i = 0` early (w3): 4.
- A 400-order enumeration of the region was started and stopped once w4 was found by reading the sched2 tie.
## (f) Where the method fell short / what helped
- The residual looked like a pure register choice (a pin), but half of it is COUNT-in-disguise: the volatile is what
creates the `la`; METHOD should say "a `la`+`lw/lhu 0(reg)` pair on a global the target reads elsewhere with `%lo` =
a combine refusal; only a volatile MEM (or a block boundary) produces it at -O2".
- METHOD 12 (d3)/14 (d14) were the right families: variable reuse + a statement-order tie-break. The `.sched` ready-list
priorities (`7f000001` = birthing) located both decisions in one reading.
## (g) Structs
No, not for this lever. The kept lever decides combine's RECOG of a volatile MEM; a struct type on the camera vector
(`D_80126B5C` as `{ s32 vx, vy, vz; }`, reading the upper half of `vz`) would still be a non-volatile MEM that combine
folds (and it would relocate against `D_80126B5C+10`, identical only after LINKING). The `expr.c:4568-4577` aggregate
channel concerns scheduling/cse memory ordering, not combine's operand predicate. Plausibly the original declared this
global (or a macro over it) volatile — if the structs phase types the D_80126B5C block, try `volatile` on that single
member and drop the cast (not tested: other accesses to D_80126B66 in this body fold, so a volatile member would have to
be accessed through a separate declaration there).
@@ -0,0 +1,14 @@
void func_8017F024(s32 a0) {
s32 *p;
/* The table's handlers take the actor (func_8017EE3C(s32 a0) is entry 1): passing it keeps a second set of $a0
* in the RTL (cse deletes the copy — $a0 still holds the parameter), so the later argument copy is not a
* "birthing" insn for sched1 (sched.c:2469-2545) and stays at the top of the block, where it keeps the mask
* constant out of $a0 (P36 S104 e26). */
((void (*)(s32))D_8018AFBC[*(u16 *)(a0 + 2)])(a0);
if (*(u16 *)a0 != 0) {
p = *(s32 **)(a0 + 0x20);
p[1] |= 0x80000000;
func_8012B2CC(a0);
}
}
@@ -0,0 +1,71 @@
# func_8017F024 (ov_SC06_010_jr_8017A4AC.c) — T7 e26, P36 S104
**Score 0, ZERO levers** (free start 5 = regen best 5). Tree levers: 2 (the `$2` pin on `p` + the `self` launder) -> 0.
No copies elsewhere (the `p[1] |= 0x80000000` after a `D_8018AFBC` dispatch is unique; the other `func_8017F024`s
in src/ are different functions of other overlays).
The move (one): the dispatch-table call passes the actor, as its handlers take it (`func_8017EE3C(s32 a0)` is entry 1
of `D_8018AFBC` = {func_801804B0, func_8017EE3C, func_80180598}, asm/ov_SC06_010/data/tail.data.s:8754):
```
- D_8018AFBC[*(u16 *)(a0 + 2)]();
+ ((void (*)(s32))D_8018AFBC[*(u16 *)(a0 + 2)])(a0);
```
(`self`, the launder and the pin deleted.) body.c uses the cast because the table's `extern` is at file scope
(TU line 5317) and D_8018AFBC has no other user in this TU. **The cleaner spelling, also 0** (scratch/tu_decl.c,
whole-TU `--try`): change the file-scope declaration to `extern void (*D_8018AFBC[])(s32);` and call
`D_8018AFBC[*(u16 *)(a0 + 2)](a0);` — recommended for the bank (a declaration change outside the body, not a
signature change of this function).
## (a) The residual
Count 30 vs 29 (COUNT class, but really a register decision): the mask constant `0x80000000` took `$a0` (`lui a0` in
the beqz slot, then `move a0,s0` before the jal = one extra insn); the target keeps the argument copy `move a0,s0`
first in the block (reorg puts it in the beqz slot) and loads the constant into `$a1`.
## (b) The passes and decisions (dumps: scratch/dumps_free, dumps_l1, dumps_a1)
1. **sched1 `adjust_priority` / `birthing_insn_p` (sched.c:2469-2545).** In the free body the argument copy
`(set (reg a0) (reg 72))` (insn 47) births `$a0` and `reg_n_sets[4] == 1` (the only set of `$a0` in the function),
so its priority is raised to the block maximum (`7f000001` in `.sched`) and the reverse list scheduler places it
immediately before the call — after the constant load (order 36 39 41 42 44 47). local-alloc then sees `$a0` free
across the constant's life and block 1's three quantities (p, the load/ior pair, the constant — allocated in BIRTH
order, local-alloc.c:1486-1500) get v0, v1, **a0**.
2. **Passing `a0` to the handler adds a second set of `$a0`** (insn 23, `(set a0 r72)` before the jalr). cse keeps it
(the `.cse` still has it), flow counts it (`reg_n_sets[4] = 2`, flow.c:2047), so `birthing_insn_p` returns 0
(sched.c:2489-2490) and the argument copy keeps priority 1: `.sched` "ready list at T-2: 43 (3) 46 (1)" ... it is
placed LAST in the reverse schedule = first in the block. `$a0` is now live across the constant, which takes the
next free register after v0/v1: **a1** — the target.
3. **The handler's copy costs zero bytes**: after reload insn 23 is `(set a0 s0)` right after the prologue's
`(set s0 a0)`; jump2's no-op move scan (`find_equiv_reg`, jump.c:425-462) finds `$a0` still equal to `$s0` and deletes
it (`.sched2` has insn 23, `.jump2` does not). So the jalr's delay slot stays a nop, as in the target.
The launder was faking (2): an asm output tied to `$a0` at the block top made `$a0` live over the constant; the `$2`
pin then restored the p/v0 assignment the extra asm quantity had disturbed (the launder alone gives p=a1, dumps_l1).
Proven on bytes (5 -> 0) and on dumps (.sched priorities, .sched2 vs .jump2 insn 23).
## (c) The move that closed it
- Pass the actor to the dispatch-table handler (`(a0)`), deleting the launder, `self` and the pin.
## (d) GENERATOR PROPOSAL
**R-dispatch-arg**: for every call through a function-pointer TABLE declared `(void)` whose index is read from a
parameter's struct (`D_x[*(u16 *)(aN + K)]()`), emit the variant that passes that parameter (`(void (*)(s32))` cast or
the widened extern): it is free in bytes when the parameter is still in its incoming register (jump.c:425-462 deletes the
copy) and it demotes every LATER argument copy of that register from sched1's birthing priority (sched.c:2489-2490).
Trigger: a residual where a later `move aK,sN` argument copy sits right before its `jal` in mine but at the top of its
block in the target (or a constant/temp takes `$aK` in mine). Leads with levers still in the tree:
scratch/dispatch_sites_with_levers.txt (func_80180100 ov_SC03_105, func_80180AA0 ov_SC05_010 — not tried).
## (e) What did not work
- body_free: 5. All R2–R38 families: 5 (history.txt: inline/width/param-alias/decl-move of `self`/`p`). No family
changes a call's ARITY at a different call site from the one that differs.
- The launder alone (no pin), scratch/l1.c: 5 (REG: p=a1, load=v0, const=v1 — four quantities, sorted order).
## (f) Where the method fell short / what helped
- METHOD S103 c3/c12 ("a dispatch-table handler READS `$aN` implicitly — pass it") is exactly this family, but its
trigger is a MISSING `move sK,$aN`; here nothing is missing at the dispatch site — the effect shows at a DIFFERENT call
(the second `$a0` set demotes the later copy's sched1 priority). Worth a line: "a later argument copy placed too late
= its hard register is set only once in the function (birthing priority); pass the parameter to an earlier call".
- Reading `.sched`'s ready list (the `7f000001` priority on a hard-register copy) settled it in one step.
## (g) Structs
No. The decision is a hard-register set count (`reg_n_sets[$a0]`) feeding sched1's birthing priority; typing `a0` as an
actor struct (`a0->state`, `a0->model->flags |= 0x80000000`) changes no set of `$a0` and no quantity count, and the
`expr.c:4568-4577` aggregate channel (memory-access ordering) is not involved. The table's element type (a handler
taking the actor) is the "type" fix here — a function-pointer prototype, not a struct.
@@ -0,0 +1,47 @@
void func_8017F438(s32 param_1) {
u8 *s0 = (u8 *)param_1;
s32 t1;
s32 t2;
s32 p;
t1 = *(s16 *)(s0 + 6);
t2 = *(s16 *)(s0 + 0xA);
if (t1 * t1 + (t2 + 0x482) * (t2 + 0x482) > 0x1323F) {
func_8012ADE4(param_1);
*(s32 *)(s0 + 0x14) = 0;
}
func_8002D4C8(0xB32, 0);
if (*(u16 *)s0 != 0) {
/* Copy the model's position/rotation into both attached objects and mirror the model's bit 31 into theirs.
* The bit is TESTED with the mask (not `< 0`): the mask register is loaded before the branch, cse reuses it
* for the `|=` (so the store's constant sits in the branch delay slot), and combine's sign test leaves a
* `(use)` of the dead AND result that reload gives a stack slot — the frame's 16 bytes (P36 S104 e26). */
p = *(s32 *)(s0 + 0xCC);
*(u16 *)(p + 0x8) = *(u16 *)(s0 + 0x6);
*(u16 *)(p + 0xA) = *(u16 *)(s0 + 0xA);
*(u16 *)(p + 0xC) = *(u16 *)(s0 + 0xE);
*(u16 *)(p + 0x10) = *(u16 *)(*(s32 *)(s0 + 0x20) + 0x10);
*(u16 *)(p + 0x12) = *(u16 *)(*(s32 *)(s0 + 0x20) + 0x12) + *(u16 *)(s0 + 0xFC);
*(u16 *)(p + 0x14) = *(u16 *)(*(s32 *)(s0 + 0x20) + 0x14);
if (*(u32 *)(*(s32 *)(s0 + 0x20) + 4) & 0x80000000) {
*(u32 *)(p + 4) |= 0x80000000;
} else {
*(u32 *)(p + 4) &= 0x7FFFFFFF;
}
p = *(s32 *)(s0 + 0xD0);
*(u16 *)(p + 0x8) = *(u16 *)(s0 + 0x6);
*(u16 *)(p + 0xA) = *(u16 *)(s0 + 0xA);
*(u16 *)(p + 0xC) = *(u16 *)(s0 + 0xE);
*(u16 *)(p + 0x10) = *(u16 *)(*(s32 *)(s0 + 0x20) + 0x10);
*(u16 *)(p + 0x12) = *(u16 *)(*(s32 *)(s0 + 0x20) + 0x12) + *(u16 *)(s0 + 0xFE);
*(u16 *)(p + 0x14) = *(u16 *)(*(s32 *)(s0 + 0x20) + 0x14);
if (*(u32 *)(*(s32 *)(s0 + 0x20) + 4) & 0x80000000) {
*(u32 *)(p + 4) |= 0x80000000;
} else {
*(u32 *)(p + 4) &= 0x7FFFFFFF;
}
}
}
@@ -0,0 +1,89 @@
# func_8017F438 (ov_SC06_010_jr_8017A4AC.c) — T7 e26, P36 S104
**Score 0, ZERO levers** (free start 18 = regen best 18). Tree levers: 4 (two `$2` pins on `val`, two `asm("")`
barriers) + 1 unmarked dead `s32 pad_[4]` -> 0 (no pin, no asm, no pad). The same text closes func_8017F600 (its own
pack) and the same-TU siblings listed in scratch/siblings.md.
The move (twice, once per attached object), plus the dead `pad_[4]`, `q`, `uVar1` and `val` deleted:
```
- q = *(s32 *)(s0 + 0x20);
- if (*(s32 *)(q + 4) < 0) {
- u32 val;
- val = *(u32 *)(p + 4);
- uVar1 = val | 0x80000000;
- } else {
- uVar1 = *(u32 *)(p + 4) & 0x7FFFFFFF;
- }
- *(u32 *)(p + 4) = uVar1;
+ if (*(u32 *)(*(s32 *)(s0 + 0x20) + 4) & 0x80000000) {
+ *(u32 *)(p + 4) |= 0x80000000;
+ } else {
+ *(u32 *)(p + 4) &= 0x7FFFFFFF;
+ }
```
The one move that matters is the CONDITION: `x & 0x80000000` instead of `(s32)x < 0`. The per-arm compound stores are
readability; the free body's `uVar1`/`val` form with only the condition changed (and the pad deleted) also scores 0
(scratch/c4.c). With the pad kept, both score 6 (frame 56 vs 40).
## (a) The residual
Same 114 instructions, two identical defects: in the `< 0` arm the loaded word took `$v1` and the constant `$v0`
(`lw v1; lui v0; j; or v0,v1,v0`), where the target has `bgez …; lui v1,0x8000` (the constant in the branch DELAY SLOT)
then `lw v0; j; or v0,v0,v1`.
## (b) The passes and decisions (dumps: scratch/dumps_free, dumps_c1, dumps_c3, dumps_tree)
1. **With `< 0` the constant is born inside the arm.** Block 4 (the or-arm) has two quantities, the load r104 and the
constant r105; sched1 must put the load first (its latency queues it: `.sched` "launching 117 before 120"), so the
constant's life is shorter, `qty_compare` (local-alloc.c:1486-1507, 2-qty case) ranks it first -> `$v0`, load -> `$v1`.
Even with the register order fixed (c1: the store written in each arm, so the load ties to the local ior result)
reorg fills the `bgez` slot from the TARGET thread (`lui v1,0x7fff` of the else arm): a `ge 0` test is predicted
taken (`mostly_true_jump`, reorg.c:1335-1420, GE vs const0 -> 1) and `fill_eager_delay_slots` tries the target
thread first (reorg.c:3700-3712). Score 8.
2. **`x & 0x80000000` loads the mask BEFORE the branch.** The `.flow` has insn 111 `(set r104 (const_int 0x80000000))`
and insn 112 `(set r103 (and r102 r104))` feeding the `eq 0` jump; cse gives the arm's `|= 0x80000000` the same
pseudo r104 (the constant is already in a register). combine rewrites the `(eq (and x 0x80000000) 0)` test as the sign
test `bgez`, deleting the AND, but r104 still feeds the arm's `or`, so its load stays in the block before the branch.
sched2 places it last, and `fill_simple_delay_slots` moves it into the `bgez` slot from BEFORE the branch
(reorg.c:2799ff., always safe) — the target's `bgez v0,…; lui v1,0x8000`. The arm now has one quantity (the load)
-> `$v0`; r104 is block-global -> `$v1`.
3. **The frame's 16 bytes are the dead AND result, not a pad.** combine plants `(use (reg:SI 103))` for the eliminated
AND result (combine.c:10831-10845; `.combine` insns 223/224, one per if); r103 then has a reference but no set,
gets no hard register, and reload gives it a stack slot (`.greg`: `(use (mem:SI (plus sp 16)))`) -> frame 40, the
target's. The tree's unmarked `s32 pad_[4]` was faking exactly these slots (with the bit test AND the pad: frame 56).
Same artefact as d34's func_80181408 (METHOD S103 c35 / d34 (b)3).
The tree's levers faked (1)-(2): the `$2` pin forced the load into `$v0`; the `asm("")` at the head of the else arm
stops reorg's target-thread search (`stop_search_p` on an asm insn), so the slot came from the fall-through thread.
Proven on bytes (18 -> 8 c1 -> 6 c2 -> 0 c3; c4 = condition only, 0) and on dumps (`.flow` insns 111/112, `.combine`
`(use)`, `.greg` stack slot).
## (c) The moves
- Test the bit with its mask: `if (*(u32 *)(q + 4) & 0x80000000)` (both copies).
- Delete the dead `s32 pad_[4];` (the dead AND results size the frame now).
- Readability (also 0): the arms as `|= 0x80000000` / `&= 0x7FFFFFFF` on the field; `q`, `uVar1`, `val` deleted.
## (d) GENERATOR PROPOSAL
**R-bittest-mask**: when a body tests `(s32)X < 0` (or `>= 0`) and then sets or clears bit 31 (`| 0x80000000`,
`& 0x7FFFFFFF`) in the same if, emit the variant `X & 0x80000000` for the test (unsigned read) — and, if the body carries a
dead `pad[N]`, the variant without it. Trigger: a `bgez`/`bltz` whose delay slot in the target holds `lui rK,0x8000`
used later in one arm, or a frame larger than the locals explain. Generalises to any sign test followed by a use of the
sign-bit mask (combine turns the mask test into the sign branch but keeps the mask register cse shared).
## (e) What did not work (bytes)
- body_free: 18. All R2–R38 families: 18 (history.txt: temp/block/do-while/inline `q`).
- c1 = `< 0` with a compound store per arm: 8 (registers right, delay slot filled from the else arm, see (b)1).
- c2 = c1 + the bit test, pad kept: 6 (frame 56). c5 = free body + bit test, pad kept: 6 (frame 56).
## (f) Where the method fell short / what helped
- The residual looked like a local-alloc register swap (METHOD step 4); the register half WAS a 2-quantity ranking,
but the delay-slot half is invisible in the residual hunk (the `lui` sits in the slot) and only the whole objdump
(METHOD S103 step 1) shows that the constant lives BEFORE the branch in the target — which no spelling of a `< 0`
test can produce. "A constant in a branch delay slot that one arm uses = the constant was computed before the
branch — look for a test that uses it" is worth a line in step 3.
- The dead `pad_[4]` was an unmarked lever: re-check any pad against `.greg`'s `(use (mem sp…))` artefacts whenever
the condition text changes (d34 said the same).
## (g) Structs
No. The decisive fact is the CONDITION's spelling (a mask test vs a sign test), which decides whether the mask constant
exists before the branch; a struct type for the model/object (`obj->flags |= 0x80000000`, `model->flags & 0x80000000`)
keeps every insn and would match only if the test is still written with the mask. The `expr.c:4568-4577` aggregate
channel (memory ordering) is not involved. The five sibling copies in this TU (scratch/siblings.md) suggest a shared
inline/macro "copy transform + mirror bit 31" in the original — worth a named helper in the structs phase.
@@ -0,0 +1,97 @@
void func_8017F600(s32 param_1) {
extern s32 D_801151D4;
u8 *s0 = (u8 *)param_1;
u16 nv[4];
s32 g;
s32 iv;
s32 idx;
s32 t1;
s32 t2;
s32 p;
g = D_801151D4;
if (*(u16 *)s0 == 0) {
return;
}
t1 = *(s16 *)(s0 + 6);
t2 = *(s16 *)(s0 + 0xA);
if (t1 * t1 + (t2 + 0x482) * (t2 + 0x482) > 0x1323F) {
func_8012ADE4(param_1);
*(s32 *)(s0 + 0x14) = 0;
}
idx = *(s32 *)(s0 + 0x1C) & 3;
switch (idx) {
case 0:
iv = func_8012C588(0x281, (s32)s0);
if (iv != 0) {
*(s32 *)(iv + 0x1C) = 2;
*(s16 *)(iv + 0x12) = (rand() & 0x1F) - 0x10;
*(s16 *)(iv + 0x16) = -((rand() & 0xF) + 0x10);
*(s16 *)(iv + 0x1A) = (rand() & 0x1F) - 0x10;
}
break;
case 1:
break;
case 2:
case 3:
iv = (s32)func_8012913C(0x23);
if (iv != 0) {
*(s16 *)(iv + 0x6) = *(u16 *)(s0 + 0x6) + (rand() & 0x3F) - 0x20;
*(s16 *)(iv + 0xA) = *(u16 *)(s0 + 0xA) + (rand() & 0x3F) - 0x30;
{
s32 r = rand();
s32 t = *(u16 *)(s0 + 0xE);
*(s32 *)(iv + 0x18) = 0;
*(s32 *)(iv + 0x14) = 0;
*(s32 *)(iv + 0x10) = 0;
*(s16 *)(iv + 0xE) = t + (r & 0x3F) - 0x20;
}
*(s16 *)(iv + 0x34) = (rand() & 0x17FF) + 0x800;
nv[0] = *(s32 *)(g + 0x5C) - *(u16 *)(iv + 0x6);
nv[1] = *(s32 *)(g + 0x60) - *(u16 *)(iv + 0xA);
nv[2] = *(s32 *)(g + 0x64) - *(u16 *)(iv + 0xE);
VectorNormalSS(nv, nv);
*(s16 *)(iv + 0x6) = *(u16 *)(iv + 0x6) + ((s16)nv[0] >> 6);
*(s16 *)(iv + 0xA) = *(u16 *)(iv + 0xA) + ((s16)nv[1] >> 6);
*(s16 *)(iv + 0xE) = *(u16 *)(iv + 0xE) + ((s16)nv[2] >> 6);
}
break;
}
if (*(u16 *)s0 != 0) {
/* Mirror the model's bit 31 into both attached objects. The bit is TESTED with the mask (not `< 0`): the
* mask register is loaded before the branch and cse reuses it for the `|=` (its load fills the branch delay
* slot), and combine's sign test leaves a `(use)` of the dead AND result that reload gives a stack slot — the
* frame's 16 bytes (P36 S104 e26; same text as func_8017F438). */
p = *(s32 *)(s0 + 0xCC);
*(u16 *)(p + 0x8) = *(u16 *)(s0 + 0x6);
*(u16 *)(p + 0xA) = *(u16 *)(s0 + 0xA);
*(u16 *)(p + 0xC) = *(u16 *)(s0 + 0xE);
*(u16 *)(p + 0x10) = *(u16 *)(*(s32 *)(s0 + 0x20) + 0x10);
*(u16 *)(p + 0x12) = *(u16 *)(*(s32 *)(s0 + 0x20) + 0x12) + *(u16 *)(s0 + 0xFC);
*(u16 *)(p + 0x14) = *(u16 *)(*(s32 *)(s0 + 0x20) + 0x14);
if (*(u32 *)(*(s32 *)(s0 + 0x20) + 4) & 0x80000000) {
*(u32 *)(p + 4) |= 0x80000000;
} else {
*(u32 *)(p + 4) &= 0x7FFFFFFF;
}
p = *(s32 *)(s0 + 0xD0);
*(u16 *)(p + 0x8) = *(u16 *)(s0 + 0x6);
*(u16 *)(p + 0xA) = *(u16 *)(s0 + 0xA);
*(u16 *)(p + 0xC) = *(u16 *)(s0 + 0xE);
*(u16 *)(p + 0x10) = *(u16 *)(*(s32 *)(s0 + 0x20) + 0x10);
*(u16 *)(p + 0x12) = *(u16 *)(*(s32 *)(s0 + 0x20) + 0x12) + *(u16 *)(s0 + 0xFE);
*(u16 *)(p + 0x14) = *(u16 *)(*(s32 *)(s0 + 0x20) + 0x14);
if (*(u32 *)(*(s32 *)(s0 + 0x20) + 4) & 0x80000000) {
*(u32 *)(p + 4) |= 0x80000000;
} else {
*(u32 *)(p + 4) &= 0x7FFFFFFF;
}
}
}
@@ -0,0 +1,47 @@
# func_8017F600 (ov_SC06_010_jr_8017A4AC.c) — T7 e26, P36 S104
**Score 0, ZERO levers** (free start 18 = regen best 18). Tree levers: 4 (two `$2` pins on `val`, two `asm("")`
barriers) + 1 unmarked dead `s32 pad_[4]` -> 0. **Same mechanism and same text as func_8017F438** (its pack's
mechanism.md has the full reading: `.run/P36/agents/ov_SC06_010__func_8017F438/mechanism.md`).
The move (both flag-mirror blocks at the end of the function), plus `pad_[4]`, `q`, `uVar1`, `val` deleted:
```
- q = *(s32 *)(s0 + 0x20);
- if (*(s32 *)(q + 4) < 0) { u32 val; val = *(u32 *)(p + 4); uVar1 = val | 0x80000000; }
- else { uVar1 = *(u32 *)(p + 4) & 0x7FFFFFFF; }
- *(u32 *)(p + 4) = uVar1;
+ if (*(u32 *)(*(s32 *)(s0 + 0x20) + 4) & 0x80000000) {
+ *(u32 *)(p + 4) |= 0x80000000;
+ } else {
+ *(u32 *)(p + 4) &= 0x7FFFFFFF;
+ }
```
## (a) The residual
222/222 instructions; in each `< 0` arm the load took `$v1` and the constant `$v0`; the target has the constant
`lui v1,0x8000` in the `bgez` delay slot and the load in `$v0`.
## (b) The pass and the decision (proven on bytes; dumps taken on func_8017F438, the same block text)
With the mask test the constant is loaded BEFORE the branch (the AND's operand), cse hands the arm's `|=` the same
pseudo, combine turns the `(eq (and x 0x80000000) 0)` test into `bgez` but keeps the constant (still used in the arm),
and reorg's `fill_simple_delay_slots` puts it in the slot from before the branch (reorg.c:2799ff.); the arm keeps one
quantity (the load) -> `$v0`. combine's `(use)` of the deleted AND result (combine.c:10831-10845) gets a stack slot in
reload: the 16 frame bytes the dead `pad_[4]` was faking (with the pad: frame 72 vs 56, score 10 — scratch/v_pad.c).
## (c) The moves
- `if (*(u32 *)(*(s32 *)(s0 + 0x20) + 4) & 0x80000000)` for both tests; delete `pad_[4]`.
- Readability (0 as well): `|=` / `&=` on the field; `q`, `uVar1`, `val` deleted.
## (d) GENERATOR PROPOSAL
R-bittest-mask (see func_8017F438): a sign test followed by a set/clear of bit 31 in the same if -> test with the mask
`& 0x80000000`; also emit the variant without a dead `pad[N]`.
## (e) What did not work
- body_free: 18; every R2–R38 family: 18 (history.txt). The bit test with the pad kept: 10 (frame only).
## (f) Where the method fell short
Same as func_8017F438 — the delay-slot `lui` is invisible in the residual's hunk; the whole objdump shows it.
## (g) Structs
No — the condition's spelling decides it (see func_8017F438 (g)). `nv[4]` is the VECTOR-like normal buffer for
`VectorNormalSS`; typing it as an SVECTOR would not touch the flag-mirror blocks.
+188 -188
View File
@@ -1,7 +1,7 @@
{
"head": "777dc340f",
"head": "2b9250a50",
"stamp": "15956e4a96c4",
"generated": "2026-09-11 03:53",
"generated": "2026-09-11 03:58",
"aliases": [
"main",
"ov_SC03_014",
@@ -21,207 +21,207 @@
"main": {
"objects": 85,
"identical": 85,
"seconds": 6.8469999999999995,
"mean_s": 0.081
"seconds": 9.305000000000001,
"mean_s": 0.109
},
"ov_SC03_014": {
"objects": 32,
"identical": 32,
"seconds": 4.631,
"mean_s": 0.145
"seconds": 5.863,
"mean_s": 0.183
},
"ov_SC03_015": {
"objects": 32,
"identical": 32,
"seconds": 4.4140000000000015,
"mean_s": 0.138
"seconds": 5.91,
"mean_s": 0.185
},
"ov_SC04_011": {
"objects": 28,
"identical": 28,
"seconds": 3.8380000000000005,
"mean_s": 0.137
"seconds": 5.112,
"mean_s": 0.183
}
},
"per_object_seconds": {
"build/src/800.o": 0.711,
"build/src/800_b.o": 0.099,
"build/src/800_b_2.o": 0.29,
"build/src/800_b_o0a.o": 0.093,
"build/src/800_c.o": 0.2,
"build/src/800b2.o": 0.094,
"build/src/apicard1.o": 0.067,
"build/src/apicard2.o": 0.059,
"build/src/apicard3.o": 0.072,
"build/src/apicard4.o": 0.076,
"build/src/apicard5.o": 0.063,
"build/src/apicard6.o": 0.073,
"build/src/apicard7.o": 0.087,
"build/src/boot.o": 0.081,
"build/src/gap.o": 0.083,
"build/src/libapi1.o": 0.068,
"build/src/libapi2.o": 0.06,
"build/src/libc2_1.o": 0.06,
"build/src/libc2_2.o": 0.064,
"build/src/libcd1.o": 0.066,
"build/src/libcd2.o": 0.067,
"build/src/libetc.o": 0.066,
"build/src/libgpu.o": 0.07,
"build/src/libgpu2.o": 0.071,
"build/src/libgs1.o": 0.066,
"build/src/libgs2.o": 0.059,
"build/src/libgs3.o": 0.063,
"build/src/libgs4.o": 0.066,
"build/src/libgs5.o": 0.061,
"build/src/libgs6.o": 0.079,
"build/src/libgs7.o": 0.061,
"build/src/libgs8.o": 0.062,
"build/src/libgte1.o": 0.07,
"build/src/libgte10.o": 0.063,
"build/src/libgte11.o": 0.063,
"build/src/libgte12.o": 0.088,
"build/src/libgte13.o": 0.058,
"build/src/libgte14.o": 0.067,
"build/src/libgte15.o": 0.051,
"build/src/libgte16.o": 0.063,
"build/src/libgte17.o": 0.066,
"build/src/libgte18.o": 0.067,
"build/src/libgte19.o": 0.056,
"build/src/libgte2.o": 0.063,
"build/src/libgte20.o": 0.063,
"build/src/libgte21.o": 0.068,
"build/src/libgte22.o": 0.065,
"build/src/libgte23.o": 0.06,
"build/src/libgte24.o": 0.056,
"build/src/libgte25.o": 0.072,
"build/src/libgte26.o": 0.101,
"build/src/libgte27.o": 0.059,
"build/src/libgte28.o": 0.068,
"build/src/libgte29.o": 0.067,
"build/src/libgte3.o": 0.066,
"build/src/libgte30.o": 0.071,
"build/src/libgte4.o": 0.066,
"build/src/libgte5.o": 0.061,
"build/src/libgte6.o": 0.068,
"build/src/libgte7.o": 0.059,
"build/src/libgte8.o": 0.07,
"build/src/libgte9.o": 0.066,
"build/src/libmcrd1.o": 0.094,
"build/src/libmcrd2.o": 0.065,
"build/src/libpad1.o": 0.072,
"build/src/libpad2.o": 0.075,
"build/src/sgap.o": 0.068,
"build/src/sgap_2.o": 0.071,
"build/src/sgap_3.o": 0.065,
"build/src/sgap_4.o": 0.076,
"build/src/sgap_5.o": 0.067,
"build/src/sgap_6.o": 0.063,
"build/src/sgap_8.o": 0.08,
"build/src/snd1.o": 0.072,
"build/src/snd10.o": 0.068,
"build/src/snd11.o": 0.067,
"build/src/snd12.o": 0.064,
"build/src/snd2.o": 0.068,
"build/src/snd3.o": 0.065,
"build/src/snd4.o": 0.078,
"build/src/snd5.o": 0.063,
"build/src/snd6.o": 0.076,
"build/src/snd7.o": 0.06,
"build/src/snd8.o": 0.065,
"build/src/snd9.o": 0.067,
"build/src/ov_SC03_014/ov_SC03_014.o": 0.127,
"build/src/ov_SC03_014/ov_SC03_014_after.o": 0.493,
"build/src/ov_SC03_014/ov_SC03_014_jr_8012ACE0.o": 0.403,
"build/src/ov_SC03_014/ov_SC03_014_jr_80135888.o": 0.057,
"build/src/ov_SC03_014/ov_SC03_014_jr_80135A4C.o": 0.06,
"build/src/ov_SC03_014/ov_SC03_014_jr_80135D20.o": 0.136,
"build/src/ov_SC03_014/ov_SC03_014_jr_801380E0.o": 0.162,
"build/src/ov_SC03_014/ov_SC03_014_jr_8013C98C.o": 0.115,
"build/src/ov_SC03_014/ov_SC03_014_jr_8013F350.o": 0.09,
"build/src/ov_SC03_014/ov_SC03_014_jr_8013FFD8.o": 0.072,
"build/src/ov_SC03_014/ov_SC03_014_jr_80140608.o": 0.166,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015444C.o": 0.074,
"build/src/ov_SC03_014/ov_SC03_014_jr_80154C24.o": 0.18,
"build/src/ov_SC03_014/ov_SC03_014_jr_801588CC.o": 0.095,
"build/src/ov_SC03_014/ov_SC03_014_jr_80159C84.o": 0.061,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015A3C8.o": 0.066,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015AE2C.o": 0.096,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015C32C.o": 0.447,
"build/src/ov_SC03_014/ov_SC03_014_jr_8016AB6C.o": 0.266,
"build/src/ov_SC03_014/ov_SC03_014_jr_80171B4C.o": 0.106,
"build/src/ov_SC03_014/ov_SC03_014_jr_801734BC.o": 0.198,
"build/src/ov_SC03_014/ov_SC03_014_jr_801789AC.o": 0.057,
"build/src/ov_SC03_014/ov_SC03_014_jr_80178D40.o": 0.092,
"build/src/ov_SC03_014/ov_SC03_014_jr_8017A4AC.o": 0.065,
"build/src/ov_SC03_014/ov_SC03_014_jr_8017AE2C.o": 0.178,
"build/src/ov_SC03_014/ov_SC03_014_jr_8017EB7C.o": 0.2,
"build/src/ov_SC03_014/ov_SC03_014_jr_80184440.o": 0.055,
"build/src/ov_SC03_014/ov_SC03_014_jr_801848E4.o": 0.266,
"build/src/ov_SC03_014/ov_SC03_014_o0b.o": 0.062,
"build/src/ov_SC03_014/ov_SC03_014_o0c.o": 0.061,
"build/src/ov_SC03_014/ov_SC03_014_o0d.o": 0.054,
"build/src/ov_SC03_014/ov_SC03_014_o0e.o": 0.071,
"build/src/ov_SC03_015/ov_SC03_015.o": 0.114,
"build/src/ov_SC03_015/ov_SC03_015_after.o": 0.5,
"build/src/ov_SC03_015/ov_SC03_015_jr_8012ACE0.o": 0.378,
"build/src/ov_SC03_015/ov_SC03_015_jr_80135888.o": 0.052,
"build/src/ov_SC03_015/ov_SC03_015_jr_80135A4C.o": 0.043,
"build/src/ov_SC03_015/ov_SC03_015_jr_80135D20.o": 0.119,
"build/src/ov_SC03_015/ov_SC03_015_jr_801380E0.o": 0.15,
"build/src/ov_SC03_015/ov_SC03_015_jr_8013C98C.o": 0.105,
"build/src/ov_SC03_015/ov_SC03_015_jr_8013F350.o": 0.063,
"build/src/ov_SC03_015/ov_SC03_015_jr_8013FFD8.o": 0.063,
"build/src/ov_SC03_015/ov_SC03_015_jr_80140608.o": 0.168,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015444C.o": 0.067,
"build/src/ov_SC03_015/ov_SC03_015_jr_80154C24.o": 0.167,
"build/src/ov_SC03_015/ov_SC03_015_jr_801588CC.o": 0.081,
"build/src/ov_SC03_015/ov_SC03_015_jr_80159C84.o": 0.059,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015A3C8.o": 0.066,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015AE2C.o": 0.087,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015C32C.o": 0.44,
"build/src/ov_SC03_015/ov_SC03_015_jr_8016AB6C.o": 0.262,
"build/src/ov_SC03_015/ov_SC03_015_jr_80171B4C.o": 0.096,
"build/src/ov_SC03_015/ov_SC03_015_jr_801734BC.o": 0.206,
"build/src/ov_SC03_015/ov_SC03_015_jr_801789AC.o": 0.054,
"build/src/ov_SC03_015/ov_SC03_015_jr_80178D40.o": 0.092,
"build/src/ov_SC03_015/ov_SC03_015_jr_8017A4AC.o": 0.063,
"build/src/ov_SC03_015/ov_SC03_015_jr_8017AE2C.o": 0.159,
"build/src/ov_SC03_015/ov_SC03_015_jr_8017EB7C.o": 0.196,
"build/src/ov_SC03_015/ov_SC03_015_jr_80184440.o": 0.051,
"build/src/ov_SC03_015/ov_SC03_015_jr_801848E4.o": 0.271,
"build/src/ov_SC03_015/ov_SC03_015_o0b.o": 0.062,
"build/src/ov_SC03_015/ov_SC03_015_o0c.o": 0.056,
"build/src/ov_SC03_015/ov_SC03_015_o0d.o": 0.059,
"build/src/ov_SC03_015/ov_SC03_015_o0e.o": 0.065,
"build/src/ov_SC04_011/ov_SC04_011.o": 0.125,
"build/src/ov_SC04_011/ov_SC04_011_after.o": 0.407,
"build/src/ov_SC04_011/ov_SC04_011_jr_8012ACE0.o": 0.331,
"build/src/ov_SC04_011/ov_SC04_011_jr_80135888.o": 0.05,
"build/src/ov_SC04_011/ov_SC04_011_jr_80135A4C.o": 0.051,
"build/src/ov_SC04_011/ov_SC04_011_jr_80135D20.o": 0.107,
"build/src/ov_SC04_011/ov_SC04_011_jr_801380E0.o": 0.151,
"build/src/ov_SC04_011/ov_SC04_011_jr_8013C98C.o": 0.114,
"build/src/ov_SC04_011/ov_SC04_011_jr_8013F350.o": 0.077,
"build/src/ov_SC04_011/ov_SC04_011_jr_8013FFD8.o": 0.061,
"build/src/ov_SC04_011/ov_SC04_011_jr_80140608.o": 0.167,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015444C.o": 0.071,
"build/src/ov_SC04_011/ov_SC04_011_jr_80154C24.o": 0.159,
"build/src/ov_SC04_011/ov_SC04_011_jr_801588CC.o": 0.088,
"build/src/ov_SC04_011/ov_SC04_011_jr_80159C84.o": 0.061,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015A3C8.o": 0.062,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015AE2C.o": 0.091,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015C32C.o": 0.343,
"build/src/ov_SC04_011/ov_SC04_011_jr_8016AB6C.o": 0.205,
"build/src/ov_SC04_011/ov_SC04_011_jr_80171B4C.o": 0.089,
"build/src/ov_SC04_011/ov_SC04_011_jr_801734BC.o": 0.184,
"build/src/ov_SC04_011/ov_SC04_011_jr_801789AC.o": 0.056,
"build/src/ov_SC04_011/ov_SC04_011_jr_80178D40.o": 0.088,
"build/src/ov_SC04_011/ov_SC04_011_jr_8017A4AC.o": 0.062,
"build/src/ov_SC04_011/ov_SC04_011_jr_8017AE2C.o": 0.101,
"build/src/ov_SC04_011/ov_SC04_011_jr_8017D494.o": 0.437,
"build/src/ov_SC04_011/ov_SC04_011_o0b.o": 0.047,
"build/src/ov_SC04_011/ov_SC04_011_o0c.o": 0.053
"build/src/800.o": 0.848,
"build/src/800_b.o": 0.135,
"build/src/800_b_2.o": 0.372,
"build/src/800_b_o0a.o": 0.091,
"build/src/800_c.o": 0.234,
"build/src/800b2.o": 0.142,
"build/src/apicard1.o": 0.09,
"build/src/apicard2.o": 0.069,
"build/src/apicard3.o": 0.115,
"build/src/apicard4.o": 0.087,
"build/src/apicard5.o": 0.121,
"build/src/apicard6.o": 0.093,
"build/src/apicard7.o": 0.095,
"build/src/boot.o": 0.138,
"build/src/gap.o": 0.112,
"build/src/libapi1.o": 0.085,
"build/src/libapi2.o": 0.086,
"build/src/libc2_1.o": 0.098,
"build/src/libc2_2.o": 0.077,
"build/src/libcd1.o": 0.09,
"build/src/libcd2.o": 0.091,
"build/src/libetc.o": 0.079,
"build/src/libgpu.o": 0.096,
"build/src/libgpu2.o": 0.101,
"build/src/libgs1.o": 0.098,
"build/src/libgs2.o": 0.1,
"build/src/libgs3.o": 0.1,
"build/src/libgs4.o": 0.093,
"build/src/libgs5.o": 0.084,
"build/src/libgs6.o": 0.138,
"build/src/libgs7.o": 0.085,
"build/src/libgs8.o": 0.072,
"build/src/libgte1.o": 0.115,
"build/src/libgte10.o": 0.077,
"build/src/libgte11.o": 0.082,
"build/src/libgte12.o": 0.075,
"build/src/libgte13.o": 0.085,
"build/src/libgte14.o": 0.116,
"build/src/libgte15.o": 0.093,
"build/src/libgte16.o": 0.093,
"build/src/libgte17.o": 0.064,
"build/src/libgte18.o": 0.107,
"build/src/libgte19.o": 0.08,
"build/src/libgte2.o": 0.101,
"build/src/libgte20.o": 0.076,
"build/src/libgte21.o": 0.073,
"build/src/libgte22.o": 0.093,
"build/src/libgte23.o": 0.114,
"build/src/libgte24.o": 0.12,
"build/src/libgte25.o": 0.084,
"build/src/libgte26.o": 0.129,
"build/src/libgte27.o": 0.071,
"build/src/libgte28.o": 0.141,
"build/src/libgte29.o": 0.111,
"build/src/libgte3.o": 0.089,
"build/src/libgte30.o": 0.082,
"build/src/libgte4.o": 0.083,
"build/src/libgte5.o": 0.138,
"build/src/libgte6.o": 0.091,
"build/src/libgte7.o": 0.145,
"build/src/libgte8.o": 0.115,
"build/src/libgte9.o": 0.082,
"build/src/libmcrd1.o": 0.097,
"build/src/libmcrd2.o": 0.072,
"build/src/libpad1.o": 0.074,
"build/src/libpad2.o": 0.079,
"build/src/sgap.o": 0.09,
"build/src/sgap_2.o": 0.073,
"build/src/sgap_3.o": 0.086,
"build/src/sgap_4.o": 0.072,
"build/src/sgap_5.o": 0.114,
"build/src/sgap_6.o": 0.084,
"build/src/sgap_8.o": 0.096,
"build/src/snd1.o": 0.106,
"build/src/snd10.o": 0.107,
"build/src/snd11.o": 0.083,
"build/src/snd12.o": 0.11,
"build/src/snd2.o": 0.085,
"build/src/snd3.o": 0.083,
"build/src/snd4.o": 0.089,
"build/src/snd5.o": 0.093,
"build/src/snd6.o": 0.084,
"build/src/snd7.o": 0.106,
"build/src/snd8.o": 0.089,
"build/src/snd9.o": 0.093,
"build/src/ov_SC03_014/ov_SC03_014.o": 0.15,
"build/src/ov_SC03_014/ov_SC03_014_after.o": 0.587,
"build/src/ov_SC03_014/ov_SC03_014_jr_8012ACE0.o": 0.487,
"build/src/ov_SC03_014/ov_SC03_014_jr_80135888.o": 0.073,
"build/src/ov_SC03_014/ov_SC03_014_jr_80135A4C.o": 0.095,
"build/src/ov_SC03_014/ov_SC03_014_jr_80135D20.o": 0.147,
"build/src/ov_SC03_014/ov_SC03_014_jr_801380E0.o": 0.215,
"build/src/ov_SC03_014/ov_SC03_014_jr_8013C98C.o": 0.146,
"build/src/ov_SC03_014/ov_SC03_014_jr_8013F350.o": 0.117,
"build/src/ov_SC03_014/ov_SC03_014_jr_8013FFD8.o": 0.099,
"build/src/ov_SC03_014/ov_SC03_014_jr_80140608.o": 0.221,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015444C.o": 0.095,
"build/src/ov_SC03_014/ov_SC03_014_jr_80154C24.o": 0.223,
"build/src/ov_SC03_014/ov_SC03_014_jr_801588CC.o": 0.156,
"build/src/ov_SC03_014/ov_SC03_014_jr_80159C84.o": 0.081,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015A3C8.o": 0.075,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015AE2C.o": 0.101,
"build/src/ov_SC03_014/ov_SC03_014_jr_8015C32C.o": 0.553,
"build/src/ov_SC03_014/ov_SC03_014_jr_8016AB6C.o": 0.34,
"build/src/ov_SC03_014/ov_SC03_014_jr_80171B4C.o": 0.13,
"build/src/ov_SC03_014/ov_SC03_014_jr_801734BC.o": 0.243,
"build/src/ov_SC03_014/ov_SC03_014_jr_801789AC.o": 0.074,
"build/src/ov_SC03_014/ov_SC03_014_jr_80178D40.o": 0.134,
"build/src/ov_SC03_014/ov_SC03_014_jr_8017A4AC.o": 0.098,
"build/src/ov_SC03_014/ov_SC03_014_jr_8017AE2C.o": 0.227,
"build/src/ov_SC03_014/ov_SC03_014_jr_8017EB7C.o": 0.267,
"build/src/ov_SC03_014/ov_SC03_014_jr_80184440.o": 0.063,
"build/src/ov_SC03_014/ov_SC03_014_jr_801848E4.o": 0.323,
"build/src/ov_SC03_014/ov_SC03_014_o0b.o": 0.074,
"build/src/ov_SC03_014/ov_SC03_014_o0c.o": 0.079,
"build/src/ov_SC03_014/ov_SC03_014_o0d.o": 0.1,
"build/src/ov_SC03_014/ov_SC03_014_o0e.o": 0.09,
"build/src/ov_SC03_015/ov_SC03_015.o": 0.149,
"build/src/ov_SC03_015/ov_SC03_015_after.o": 0.608,
"build/src/ov_SC03_015/ov_SC03_015_jr_8012ACE0.o": 0.514,
"build/src/ov_SC03_015/ov_SC03_015_jr_80135888.o": 0.104,
"build/src/ov_SC03_015/ov_SC03_015_jr_80135A4C.o": 0.068,
"build/src/ov_SC03_015/ov_SC03_015_jr_80135D20.o": 0.149,
"build/src/ov_SC03_015/ov_SC03_015_jr_801380E0.o": 0.191,
"build/src/ov_SC03_015/ov_SC03_015_jr_8013C98C.o": 0.141,
"build/src/ov_SC03_015/ov_SC03_015_jr_8013F350.o": 0.085,
"build/src/ov_SC03_015/ov_SC03_015_jr_8013FFD8.o": 0.073,
"build/src/ov_SC03_015/ov_SC03_015_jr_80140608.o": 0.216,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015444C.o": 0.111,
"build/src/ov_SC03_015/ov_SC03_015_jr_80154C24.o": 0.197,
"build/src/ov_SC03_015/ov_SC03_015_jr_801588CC.o": 0.095,
"build/src/ov_SC03_015/ov_SC03_015_jr_80159C84.o": 0.136,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015A3C8.o": 0.105,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015AE2C.o": 0.169,
"build/src/ov_SC03_015/ov_SC03_015_jr_8015C32C.o": 0.55,
"build/src/ov_SC03_015/ov_SC03_015_jr_8016AB6C.o": 0.342,
"build/src/ov_SC03_015/ov_SC03_015_jr_80171B4C.o": 0.136,
"build/src/ov_SC03_015/ov_SC03_015_jr_801734BC.o": 0.259,
"build/src/ov_SC03_015/ov_SC03_015_jr_801789AC.o": 0.077,
"build/src/ov_SC03_015/ov_SC03_015_jr_80178D40.o": 0.13,
"build/src/ov_SC03_015/ov_SC03_015_jr_8017A4AC.o": 0.122,
"build/src/ov_SC03_015/ov_SC03_015_jr_8017AE2C.o": 0.204,
"build/src/ov_SC03_015/ov_SC03_015_jr_8017EB7C.o": 0.242,
"build/src/ov_SC03_015/ov_SC03_015_jr_80184440.o": 0.068,
"build/src/ov_SC03_015/ov_SC03_015_jr_801848E4.o": 0.357,
"build/src/ov_SC03_015/ov_SC03_015_o0b.o": 0.074,
"build/src/ov_SC03_015/ov_SC03_015_o0c.o": 0.085,
"build/src/ov_SC03_015/ov_SC03_015_o0d.o": 0.066,
"build/src/ov_SC03_015/ov_SC03_015_o0e.o": 0.087,
"build/src/ov_SC04_011/ov_SC04_011.o": 0.167,
"build/src/ov_SC04_011/ov_SC04_011_after.o": 0.543,
"build/src/ov_SC04_011/ov_SC04_011_jr_8012ACE0.o": 0.415,
"build/src/ov_SC04_011/ov_SC04_011_jr_80135888.o": 0.104,
"build/src/ov_SC04_011/ov_SC04_011_jr_80135A4C.o": 0.094,
"build/src/ov_SC04_011/ov_SC04_011_jr_80135D20.o": 0.181,
"build/src/ov_SC04_011/ov_SC04_011_jr_801380E0.o": 0.2,
"build/src/ov_SC04_011/ov_SC04_011_jr_8013C98C.o": 0.126,
"build/src/ov_SC04_011/ov_SC04_011_jr_8013F350.o": 0.103,
"build/src/ov_SC04_011/ov_SC04_011_jr_8013FFD8.o": 0.08,
"build/src/ov_SC04_011/ov_SC04_011_jr_80140608.o": 0.229,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015444C.o": 0.11,
"build/src/ov_SC04_011/ov_SC04_011_jr_80154C24.o": 0.225,
"build/src/ov_SC04_011/ov_SC04_011_jr_801588CC.o": 0.103,
"build/src/ov_SC04_011/ov_SC04_011_jr_80159C84.o": 0.107,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015A3C8.o": 0.08,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015AE2C.o": 0.11,
"build/src/ov_SC04_011/ov_SC04_011_jr_8015C32C.o": 0.405,
"build/src/ov_SC04_011/ov_SC04_011_jr_8016AB6C.o": 0.264,
"build/src/ov_SC04_011/ov_SC04_011_jr_80171B4C.o": 0.152,
"build/src/ov_SC04_011/ov_SC04_011_jr_801734BC.o": 0.217,
"build/src/ov_SC04_011/ov_SC04_011_jr_801789AC.o": 0.078,
"build/src/ov_SC04_011/ov_SC04_011_jr_80178D40.o": 0.122,
"build/src/ov_SC04_011/ov_SC04_011_jr_8017A4AC.o": 0.076,
"build/src/ov_SC04_011/ov_SC04_011_jr_8017AE2C.o": 0.128,
"build/src/ov_SC04_011/ov_SC04_011_jr_8017D494.o": 0.542,
"build/src/ov_SC04_011/ov_SC04_011_o0b.o": 0.059,
"build/src/ov_SC04_011/ov_SC04_011_o0c.o": 0.092
},
"ok": true,
"seconds": 2.3
"seconds": 2.9
}
File diff suppressed because one or more lines are too long
+7 -5
View File
@@ -5318,15 +5318,17 @@ extern void (*D_8018AFBC[])(void);
extern void func_8012B2CC(s32 a0);
void func_8017F024(s32 a0) {
register s32 *p __asm__("$2"); // !FAKE: pin $2 — NEEDED DIFFERS (P36 rung B tus7)
s32 self;
s32 *p;
D_8018AFBC[*(u16 *)(a0 + 2)]();
/* The table's handlers take the actor (func_8017EE3C(s32 a0) is entry 1): passing it keeps a second set of $a0
* in the RTL (cse deletes the copy — $a0 still holds the parameter), so the later argument copy is not a
* "birthing" insn for sched1 (sched.c:2469-2545) and stays at the top of the block, where it keeps the mask
* constant out of $a0 (P36 S104 e26). */
((void (*)(s32))D_8018AFBC[*(u16 *)(a0 + 2)])(a0);
if (*(u16 *)a0 != 0) {
__asm__ __volatile__("" : "=r"(self) : "0"(a0)); // !FAKE: launder — NEEDED DIFFERS (P36 rung B tus7)
p = *(s32 **)(a0 + 0x20);
p[1] |= 0x80000000;
func_8012B2CC(self);
func_8012B2CC(a0);
}
}