phase11: merge 39 — 4 rows (A's 0x80038D48, 0x800F6D60; D's 0x8009B56C, 0x80018458)
Worker A's two: the addition operand-order row (a1[i]+a0[i] vs a0[i]+a1[i] -- same length, 16 bytes apart, because cc1 evaluates the right-hand operand first) and the unconditional p[0]=0 that lands in a branch delay slot. Worker D's two: 0x8009B56C closed on cookbook 43 trigger 1 after D had nearly written the row off, and 0x80018458.
This commit is contained in:
+1017
-1022
File diff suppressed because it is too large
Load Diff
@@ -61,6 +61,7 @@
|
||||
0x800182D4 0x800182F4 src/func_800182D4.c
|
||||
0x80018384 0x800183B8 src/func_80018384.c
|
||||
0x800183B8 0x800183EC src/func_800183B8.c
|
||||
0x80018458 0x80018560 src/func_80018458.c
|
||||
0x800196B4 0x80019700 src/func_800196B4.c
|
||||
0x800198C0 0x800198F4 src/func_800198C0.c
|
||||
0x80019B6C 0x80019B94 src/func_80019B6C.c
|
||||
@@ -172,6 +173,7 @@
|
||||
0x8003768C 0x800376CC src/func_8003768C.c
|
||||
0x80038788 0x80038790 src/func_80038788.c
|
||||
0x80038790 0x8003879C src/func_80038790.c
|
||||
0x80038D48 0x80038DD8 src/func_80038D48.c
|
||||
0x8003AAE8 0x8003AB20 src/func_8003AAE8.c
|
||||
0x8003ABB4 0x8003AC00 src/func_8003ABB4.c
|
||||
0x8003B2F0 0x8003B320 src/func_8003B2F0.c
|
||||
@@ -385,6 +387,7 @@
|
||||
0x80099E14 0x80099E34 src/func_80099E14.c
|
||||
0x8009AC08 0x8009AC28 src/func_8009AC08.c
|
||||
0x8009B464 0x8009B4A0 src/func_8009B464.c
|
||||
0x8009B56C 0x8009B638 src/func_8009B56C.c
|
||||
0x8009C904 0x8009CB28 src/func_8009C904.c
|
||||
0x8009D8A0 0x8009D8E0 src/func_8009D8A0.c
|
||||
0x8009E8D0 0x8009E95C src/func_8009E8D0.c
|
||||
@@ -480,6 +483,7 @@
|
||||
0x800F6330 0x800F6364 src/func_800F6330.c
|
||||
0x800F6570 0x800F6594 src/func_800F6570.c
|
||||
0x800F66B8 0x800F6710 src/func_800F66B8.c
|
||||
0x800F6D60 0x800F6DD0 src/func_800F6D60.c
|
||||
0x800F75D0 0x800F760C src/func_800F75D0.c
|
||||
0x800F7990 0x800F79C0 src/func_800F7990.c
|
||||
0x800F79C0 0x800F79F0 src/func_800F79C0.c
|
||||
|
||||
|
@@ -0,0 +1,53 @@
|
||||
/*
|
||||
* func_80018458 — 264 bytes at 0x80018458..0x80018560
|
||||
*
|
||||
* Goal B, Phase 11. **First attempt.** Found by the size-capped redundancy rank (0.67).
|
||||
*
|
||||
* Two 3-element signed-short differences, fed to the GTE transform 0x800F3E18, then rounded through
|
||||
* 0x8009C56C and stored as three shorts:
|
||||
*
|
||||
* short *s1 = *(short **)(a0 + 8);
|
||||
* int a[3], b[3], r[3];
|
||||
* a[i] = *(short *)(*(int *)(a0 + 32) + 2i) - *(short *)(*(int *)(a0 + 28) + 2i); i = 0,1,2
|
||||
* b[i] = *(short *)(*(int *)(a0 + 24) + 2i) - *(short *)(*(int *)(a0 + 28) + 2i); i = 0,1,2
|
||||
* func_800F3E18(b, a, r); <- the GTE rotation/translation helper
|
||||
* func_8009C56C(r, r);
|
||||
* s1[0] = r[0]; s1[1] = r[1]; s1[2] = r[2];
|
||||
*
|
||||
* Both difference blocks are **fully unrolled** (three statements each) and both re-dereference the
|
||||
* record's pointers at every site — `*(int *)(a0 + 32)` / `+28` / `+24` are re-loaded per element,
|
||||
* which is cookbook 105's per-statement re-read setting again. The frame is 80 bytes with the two
|
||||
* scratch vectors at sp+0x10 and sp+0x20 and the result at sp+0x30; the stores to `s1` are `sh` fed
|
||||
* by 32-bit loads.
|
||||
*
|
||||
* LIMITS: the function name, the two callees, the meaning of the record at a0 and of the offsets
|
||||
* 8/24/28/32 are hypotheses reconstructed from the disassembly; only the compiled bytes are evidence.
|
||||
* The three int vectors are declared `int[3]` because the slots are word-sized and the callees take
|
||||
* `int *`; the sources are signed halfword loads. `s1` is the only callee-saved value, so only s0/s1
|
||||
* are saved.
|
||||
*/
|
||||
|
||||
extern void func_800F3E18(int *a0, int *a1, int *a2);
|
||||
extern void func_8009C56C(int *a0, int *a1);
|
||||
|
||||
void func_80018458(int a0)
|
||||
{
|
||||
short *s1 = *(short **)(a0 + 8);
|
||||
int a[3];
|
||||
int b[3];
|
||||
int r[3];
|
||||
|
||||
a[0] = *(short *)(*(int *)(a0 + 32) + 0) - *(short *)(*(int *)(a0 + 28) + 0);
|
||||
a[1] = *(short *)(*(int *)(a0 + 32) + 2) - *(short *)(*(int *)(a0 + 28) + 2);
|
||||
a[2] = *(short *)(*(int *)(a0 + 32) + 4) - *(short *)(*(int *)(a0 + 28) + 4);
|
||||
b[0] = *(short *)(*(int *)(a0 + 24) + 0) - *(short *)(*(int *)(a0 + 28) + 0);
|
||||
b[1] = *(short *)(*(int *)(a0 + 24) + 2) - *(short *)(*(int *)(a0 + 28) + 2);
|
||||
b[2] = *(short *)(*(int *)(a0 + 24) + 4) - *(short *)(*(int *)(a0 + 28) + 4);
|
||||
|
||||
func_800F3E18(b, a, r);
|
||||
func_8009C56C(r, r);
|
||||
|
||||
s1[0] = r[0];
|
||||
s1[1] = r[1];
|
||||
s1[2] = r[2];
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
/*
|
||||
* func_80028CE0 — 244 bytes at 0x80028CE0..0x80028DD4
|
||||
*
|
||||
* Builds a basis triple from a source object and passes it to func_800289B4.
|
||||
*
|
||||
* Reconstructed structure (every claim below is read off the emitted bytes):
|
||||
*
|
||||
* addiu sp,sp,-96 / sw ra,88(sp) frame 96; the four locals are
|
||||
* FOUR 16-byte vectors at
|
||||
* sp+24/40/56/72 (16-byte stride,
|
||||
* cookbook 96), and sp+16 is the
|
||||
* outgoing 5th-argument slot.
|
||||
* lw v0,20(a1) .. lw t0,32(a1) four loads ...
|
||||
* sw v0,24(sp) .. sw t0,36(sp) ... then four stores: a single
|
||||
* 16-byte STRUCT ASSIGNMENT
|
||||
* (cookbook 28/102), not element
|
||||
* stores, which would interleave
|
||||
* and cost 8 bytes per copy.
|
||||
* lw t6,24(sp) / lw t4,28(sp) / lw t2,32(sp) the copy is then RE-READ from
|
||||
* memory (cookbook 45's re-read
|
||||
* direction), so the source reads
|
||||
* the elements, not a named local.
|
||||
* subu / sw -> 40,44,48 three differences against a1[9..11]
|
||||
* subu / sw -> 56,60,64 three differences against a1[1..3]
|
||||
* lw v0,4/8/0(a0) -> sw 72,76,80 three rotated reads of a0's elements
|
||||
* sw 24,28,32 / 40,44,48 / 56,60,64 a SECOND write to all nine slots:
|
||||
* the three vectors are each shifted
|
||||
* by one element, in place, through a
|
||||
* scalar temporary.
|
||||
* jal 0x800289B4 / addiu a2,sp,40 call(&z, &t, &u, &w, a2)
|
||||
* andi v0,v0,0xff the 0xff mask is the return type
|
||||
*
|
||||
* The load-delay `nop` positions are the discriminator for the statement ORDER: the
|
||||
* z-vector block must be written BEFORE the three shifts. With the shifts first, cc1
|
||||
* has an independent store available for the `lw v1,12(a1)` and `lw v0,4(a0)` delay
|
||||
* slots, fills both from the shift block, and the region comes out 236 bytes
|
||||
* (LENGTH-MISMATCH). Ordering the source [differences] [z] [shifts] reproduces the
|
||||
* original's two `nop`s and matches exactly. This is cookbook 100's diagnostic in a
|
||||
* new form: when the instruction COUNT is short and the difference is *delay-slot
|
||||
* filling*, the cause is statement order, not the code inside the statements.
|
||||
*
|
||||
* LIMITS: the structure above is evidence; the names are not.
|
||||
* * `struct V` (four ints) is inferred from the 16-byte stride and the 16-byte
|
||||
* frame step plus the 4-word batched copy. The fourth element is copied and
|
||||
* never read (it survives in sp+36), so the original's vector type has a
|
||||
* fourth member that this routine ignores.
|
||||
* * `a1` is indexed as an `int *` at offsets 1/2/3, 5/6/7/8 and 9/10/11; whether
|
||||
* the original spelled those as fields of a struct is not recoverable from the
|
||||
* bytes, and does not affect the output.
|
||||
* * `a2` is forwarded as the callee's fifth argument; its type is unknown.
|
||||
* * The three in-place shifts are what the bytes require, but WHY the original
|
||||
* permutes its own inputs before the call (an axis swizzle, most plausibly) is a
|
||||
* hypothesis. It is not needed to match, and it is not asserted.
|
||||
* * func_800289B4 is a cross-reference by address only (ADDRESS_SYMBOL); its
|
||||
* purpose is not established here.
|
||||
*/
|
||||
|
||||
struct V { int v[4]; };
|
||||
|
||||
extern int func_800289B4(struct V *a0, struct V *a1, struct V *a2, struct V *a3, int a4);
|
||||
|
||||
int func_80028CE0(struct V *a0, int *a1, int a2)
|
||||
{
|
||||
struct V t, u, w, z;
|
||||
int tmp;
|
||||
|
||||
t = *(struct V *)(a1 + 5);
|
||||
|
||||
u.v[0] = a1[9] - t.v[0];
|
||||
u.v[1] = a1[10] - t.v[1];
|
||||
u.v[2] = a1[11] - t.v[2];
|
||||
|
||||
w.v[0] = a1[1] - t.v[0];
|
||||
w.v[1] = a1[2] - t.v[1];
|
||||
w.v[2] = a1[3] - t.v[2];
|
||||
|
||||
z.v[0] = a0->v[1];
|
||||
z.v[1] = a0->v[2];
|
||||
z.v[2] = a0->v[0];
|
||||
|
||||
tmp = t.v[0];
|
||||
t.v[0] = t.v[1];
|
||||
t.v[1] = t.v[2];
|
||||
t.v[2] = tmp;
|
||||
|
||||
tmp = u.v[0];
|
||||
u.v[0] = u.v[1];
|
||||
u.v[1] = u.v[2];
|
||||
u.v[2] = tmp;
|
||||
|
||||
tmp = w.v[0];
|
||||
w.v[0] = w.v[1];
|
||||
w.v[1] = w.v[2];
|
||||
w.v[2] = tmp;
|
||||
|
||||
return func_800289B4(&z, &t, &u, &w, a2) & 0xff;
|
||||
}
|
||||
Reference in New Issue
Block a user