From 35d00fe3ec9250671923d800dcbe248d475ee4c6 Mon Sep 17 00:00:00 2001 From: Drew T <50529377+Druthulu@users.noreply.github.com> Date: Wed, 5 Aug 2026 14:54:59 -0600 Subject: [PATCH] chore(phase-30 S42): PRESERVE the two serial NEAR drafts + log them; flag a draft-less ledger row MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Answering "did you bank the results": the two serial functions did NOT match, so there was nothing to bank (G3 -- NEAR is not a match). Everything that DID match this session is already banked and committed (7 from the S4 redo, 24 wave exemplars + propagations, both giants x138). But the drafts were about to be LOST, which is worse than not banking them: .run/s42/ov_SC01_077/func_8017C294.c NEAR(12) of 246 ~245k subagent tokens .run/s42/ov_SC03_126/func_8017C6F4.c NEAR(63) of 947 ~434k subagent tokens .run/s42/ov_SC03_126/func_8017C6F4.pin-t5.c NEAR(47), pinned variant All three were gitignored -- one `git clean` from gone (R20: commit irreplaceable work). Added a curated /.run/s42/ allowlist and committed them. They are the best base any future attempt has: func_8017C6F4 has frame 0x120 + vars=232 EXACT with only a register rotation left, and its permuter has never been aimed at it (make_base_c fails on the gte_ macro block -- demacroize first). Both logged to the backlog with today's MEASURED values, class, reach and draft path. ⚠️ LEDGER INTEGRITY, flagged not silently fixed: the backlog already held `func_8017C6F4 closeness=14` (2026-07-01, ov_SC03_010, source=bulk-harvest) -- BETTER than today's 63, but with **draft: None, klass: None, nins: None, reach: None**. There is no artifact behind it and no draft of it survives on disk (today's agent scanned every stored draft and found two, both junk). `load_best` takes the LOWEST closeness per address, so this unverifiable row will out-rank today's real, reproducible 63 in every future target selection. This is the Phase-28 defect class (`func_80178004` recorded close=0 when it was 91). It is left in place rather than deleted because deciding between "a lost good draft" and "a bad number" needs evidence I do not have. **Whoever picks this up: treat the 14 as UNVERIFIED, start from the committed 63/47 drafts, and if the 14 cannot be reproduced, purge the row.** The general rule this argues for: a backlog row with no draft artifact is a rumour, not a result -- `backlog.py log` should require a draft path (or mark the row unverifiable) so an artifact-less number cannot outrank a reproducible one. --- .gitignore | 6 + .run/backlog.jsonl | 2 + .run/s42/ov_SC01_077/func_8017C294.c | 194 ++++++++ .run/s42/ov_SC03_126/func_8017C6F4.c | 507 ++++++++++++++++++++ .run/s42/ov_SC03_126/func_8017C6F4.pin-t5.c | 488 +++++++++++++++++++ docs/backlog.md | 4 +- 6 files changed, 1199 insertions(+), 2 deletions(-) create mode 100644 .run/s42/ov_SC01_077/func_8017C294.c create mode 100644 .run/s42/ov_SC03_126/func_8017C6F4.c create mode 100644 .run/s42/ov_SC03_126/func_8017C6F4.pin-t5.c diff --git a/.gitignore b/.gitignore index 3a17de1b2..31b94f134 100644 --- a/.gitignore +++ b/.gitignore @@ -37,6 +37,12 @@ !/.run/giants/*.md !/.run/giants/*.sh !/.run/giants/*.py +# P30 S42 serial crack drafts — ~680k subagent tokens of work, both NEAR and both the best +# base any future attempt has. They were gitignored, i.e. one `git clean` from gone (R20). +!/.run/s42/ +/.run/s42/* +!/.run/s42/*/ +!/.run/s42/*/*.c !/.run/probe_jtbl/ /.run/probe_jtbl/* !/.run/probe_jtbl/verdict.md diff --git a/.run/backlog.jsonl b/.run/backlog.jsonl index f1274c856..a1bdd8e2e 100644 --- a/.run/backlog.jsonl +++ b/.run/backlog.jsonl @@ -834,3 +834,5 @@ {"ts": "2026-08-04 23:16:43", "addr": "0x80183834", "name": "func_80183834", "reach": 1, "klass": "STRUCT", "nins": 135, "status": "near", "closeness": 35, "where_stuck": "STRUCT: 35 mismatch", "best_draft": ".run/backlog_drafts/func_80183834.c", "binary": "ov_SC01_077", "source": "worker", "residual": [[28, "3c040000 lui\ta0,0x0", "a6020002 sh $v0, 0x2($s0)"], [29, "84840000 lh\ta0,0(a0)", "2402003c addiu $v0, $zero, 0x3C"], [30, "a6020002 sh\tv0,2(s0)", "a6000098 sh $zero, 0x98($s0)"], [31, "2402003c li\tv0,60", "ae02001c sw $v0, 0x1C($s0)"], [32, "ae02001c sw\tv0,28(s0)", "a600005c sh $zero, 0x5C($s0)"], [33, "3c020000 lui\tv0,0x0", "3c048019 lui $a0, %hi(D_8018AAD0)"], [34, "84420000 lh\tv0,0(v0)", "8484aad0 lh $a0, %lo(D_8018AAD0)($a0)"], [35, "86050006 lh\ta1,6(s0)", "3c028019 lui $v0, %hi(D_8018AACC)"], [36, "a6000098 sh\tzero,152(s0)", "8442aacc lh $v0, %lo(D_8018AACC)($v0)"], [37, "a600005c sh\tzero,92(s0)", "86050006 lh $a1, 0x6($s0)"], [52, "8e110020 lw\ts1,32(s0)", "8e120020 lw $s2, 0x20($s0)"], [54, "96320012 lhu\ts2,18(s1)", "86510012 lh $s1, 0x12($s2)"], [59, "2642fc00 addiu\tv0,s2,-1024", "2622fc00 addiu $v0, $s1, -0x400"], [60, "26420400 addiu\tv0,s2,1024", "26220400 addiu $v0, $s1, 0x400"], [61, "a6220012 sh\tv0,18(s1)", "a6420012 sh $v0, 0x12($s2)"], [76, "14620031 bne\tv1,v0,1f8 ", "14620033 bne $v1, $v0, .L80183A34"], [95, "1440001e bnez\tv0,1f8 ", "14400020 bnez $v0, .L80183A34"], [99, "1040001a beqz\tv0,1f8 ", "1040001c beqz $v0, .L80183A34"], [106, "10400013 beqz\tv0,1f8 ", "10400015 beqz $v0, .L80183A34"], [118, "3c020000 lui\tv0,0x0", "08060e8b j .L80183A2C"], [119, "8c420000 lw\tv0,0(v0)", "02002021 addu $a0, $s0, $zero"], [120, "00000000 nop", "3c028012 lui $v0, %hi(D_801270CC)"], [121, "2442ffff addiu\tv0,v0,-1", "8c4270cc lw $v0, %lo(D_801270CC)($v0)"], [122, "3c010000 lui\tat,0x0", "02002021 addu $a0, $s0, $zero"]], "passes_tried": null} {"ts": "2026-08-04 23:17:32", "addr": "0x80186078", "name": "func_80186078", "reach": 1, "klass": "WAVE", "nins": 174, "status": "near", "closeness": 141, "where_stuck": "WAVE: 141 mismatch", "best_draft": ".run/backlog_drafts/func_80186078.c", "binary": "ov_SC01_077", "source": "worker", "residual": [[15, "a7a00014 sh\tzero,20(sp)", "8e02001c lw $v0, 0x1C($s0)"], [16, "a7a00012 sh\tzero,18(sp)", "02603021 addu $a2, $s3, $zero"], [17, "8e02001c lw\tv0,28(s0)", "a7a00014 sh $zero, 0x14($sp)"], [18, "02603021 move\ta2,s3", "a7a00012 sh $zero, 0x12($sp)"], [24, "8e030090 lw\tv1,144(s0)", "8e040090 lw $a0, 0x90($s0)"], [25, "8e040044 lw\ta0,68(s0)", "8e030044 lw $v1, 0x44($s0)"], [26, "8e050048 lw\ta1,72(s0)", "8e05004c lw $a1, 0x4C($s0)"], [27, "a6000098 sh\tzero,152(s0)", "a6020006 sh $v0, 0x6($s0)"], [28, "a6020006 sh\tv0,6(s0)", "97a2001c lhu $v0, 0x1C($sp)"], [29, "97a2001c lhu\tv0,28(sp)", "a6000098 sh $zero, 0x98($s0)"], [30, "00000000 nop", "a602000e sh $v0, 0xE($s0)"], [31, "a602000e sh\tv0,14(s0)", "8e020010 lw $v0, 0x10($s0)"], [32, "8c660000 lw\ta2,0(v1)", "8c860000 lw $a2, 0x0($a0)"], [33, "8e020010 lw\tv0,16(s0)", "8e040048 lw $a0, 0x48($s0)"], [34, "8e030014 lw\tv1,20(s0)", "00431021 addu $v0, $v0, $v1"], [35, "00441021 addu\tv0,v0,a0", "ae020010 sw $v0, 0x10($s0)"], [36, "ae020010 sw\tv0,16(s0)", "8e020014 lw $v0, 0x14($s0)"], [37, "8e020018 lw\tv0,24(s0)", "8e030018 lw $v1, 0x18($s0)"], [38, "8e04004c lw\ta0,76(s0)", "00441021 addu $v0, $v0, $a0"], [40, "ae030014 sw\tv1,20(s0)", "3c04801e lui $a0, %hi(D_801DA988)"], [41, "3c030000 lui\tv1,0x0", "2484a988 addiu $a0, $a0, %lo(D_801DA988)"], [42, "8c630000 lw\tv1,0(v1)", "ae020014 sw $v0, 0x14($s0)"], [43, "8e050010 lw\ta1,16(s0)", "ae030018 sw $v1, 0x18($s0)"], [44, "00441021 addu\tv0,v0,a0", "8c820000 lw $v0, 0x0($a0)"]], "passes_tried": null} {"ts": "2026-08-04 23:17:33", "addr": "0x801865ec", "name": "func_801865EC", "reach": 1, "klass": "WAVE", "nins": 125, "status": "near", "closeness": 65, "where_stuck": "WAVE: 65 mismatch", "best_draft": ".run/backlog_drafts/func_801865EC.c", "binary": "ov_SC01_077", "source": "worker", "residual": [[36, "00023023 negu\ta2,v0", "00022823 negu $a1, $v0"], [37, "24c40400 addiu\ta0,a2,1024", "24a30400 addiu $v1, $a1, 0x400"], [38, "00041040 sll\tv0,a0,0x1", "00031040 sll $v0, $v1, 1"], [39, "00442821 addu\ta1,v0,a0", "00432021 addu $a0, $v0, $v1"], [40, "00051100 sll\tv0,a1,0x4", "00041100 sll $v0, $a0, 4"], [42, "00401821 move\tv1,v0", "26260024 addiu $a2, $s1, 0x24"], [43, "244307ff addiu\tv1,v0,2047", "244207ff addiu $v0, $v0, 0x7FF"], [44, "000312c3 sra\tv0,v1,0xb", "000212c3 sra $v0, $v0, 11"], [46, "00051140 sll\tv0,a1,0x5", "00041140 sll $v0, $a0, 5"], [52, "00801021 move\tv0,a0", "00601021 addu $v0, $v1, $zero"], [55, "24c2043f addiu\tv0,a2,1087", "24a2043f addiu $v0, $a1, 0x43F"], [63, "8e230020 lw\tv1,32(s1)", "8e220020 lw $v0, 0x20($s1)"], [64, "26220024 addiu\tv0,s1,36", "00000000 nop"], [65, "ac620080 sw\tv0,128(v1)", "ac460080 sw $a2, 0x80($v0)"], [67, "00000000 nop", "02202021 addu $a0, $s1, $zero"], [69, "00000000 nop", "3c058018 lui $a1, %hi(D_80186E48)"], [70, "34420010 ori\tv0,v0,0x10", "24a56e48 addiu $a1, $a1, %lo(D_80186E48)"], [71, "a462002c sh\tv0,44(v1)", "34420010 ori $v0, $v0, 0x10"], [72, "8e230020 lw\tv1,32(s1)", "a462002c sh $v0, 0x2C($v1)"], [73, "24022000 li\tv0,8192", "8e230020 lw $v1, 0x20($s1)"], [74, "a462001c sh\tv0,28(v1)", "24022000 addiu $v0, $zero, 0x2000"], [75, "8e230020 lw\tv1,32(s1)", "a462001c sh $v0, 0x1C($v1)"], [76, "02202021 move\ta0,s1", "a462001a sh $v0, 0x1A($v1)"], [77, "a462001a sh\tv0,26(v1)", "0c04aa0a jal func_8012A828"]], "passes_tried": null} +{"ts": "2026-08-05 14:54:19", "addr": "0x8017C294", "name": "func_8017C294", "reach": 16, "klass": "regalloc: qty_compare tie + stratum-3 frame slot (cookbook 147)", "nins": 246, "status": "near", "closeness": 12, "where_stuck": "ov_SC01_077", "best_draft": ".run/s42/ov_SC01_077/func_8017C294.c", "binary": null, "source": "s42-serial", "residual": null, "passes_tried": null} +{"ts": "2026-08-05 14:54:22", "addr": "0x8017C6F4", "name": "func_8017C6F4", "reach": 3, "klass": "regalloc: one register rotation; permuter UNTRIED (make_base_c fails on gte_ macros) - demacroize first (cookbook 148)", "nins": 947, "status": "near", "closeness": 63, "where_stuck": "ov_SC03_126", "best_draft": ".run/s42/ov_SC03_126/func_8017C6F4.c", "binary": null, "source": "s42-serial", "residual": null, "passes_tried": null} diff --git a/.run/s42/ov_SC01_077/func_8017C294.c b/.run/s42/ov_SC01_077/func_8017C294.c new file mode 100644 index 000000000..b03257470 --- /dev/null +++ b/.run/s42/ov_SC01_077/func_8017C294.c @@ -0,0 +1,194 @@ +/* + * func_8017C294 -- ov_SC01_077 (exemplar of a 16-member structural cluster), 246 ins. + * STATUS: NEAR -- 246/246 ins, 12 mismatched (95.1% index-wise exact), ZERO structural + * divergence (no opcode/length drift; every residual is a register grant or a stack-slot + * offset). Session S42. + * + * WHAT THE FUNCTION IS + * Screen-space AABB of a world rect: build 4 corner points from (x,y,w,h), rotate/project + * each through func_8017C66C/func_8017C710/func_8017C908, clamp each projected corner to + * +/- (min(w,h)/20) of the rect, then min/max the 4 results (plus the origin `base`) into + * out[0..3] = {minx, minz, width, depth}. + * + * THE 12 RESIDUALS (all measured, none guessed) + * idx 12,226 a1's spill slot: mine 0xA8, target 0x80 (offset only, reg correct) + * idx 14,15,18,19,23 head reg grant: `mh` lands in $a2, target $a0; and the `lh 0x6($a0)` + * / `lh 0x0($a0)` pair is emitted in the other order. See NOTE 2. + * idx 31,33,34,106,108 pEnd's storage: mine 0xA0 in $v0, target 0x108 in $t8. See NOTE 1. + * + * NOTE 1 -- THE 0x108 SLOT IS *NOT* REACHABLE FROM A DECLARED LOCAL (byte-measured, S42) + * Frame map of the target (vars=256, 0x10..0x10F): + * 0x10 pos[4][4] | 0x30 mat[4][4] | 0x50 outp[4][4] | 0x70 zero[4] | 0x78 base[4] + * 0x80 = the spilled `a1` parameter | 0x88..0x107 = 128 B never referenced + * 0x108 = the loop-start pointer (stored once, reloaded for `p < pStart+0x20`) + * gcc-2.7.2 lays the frame out in three strata, in this order: + * (1) every DECLARED local, in declaration order, ascending from 0x10; + * (2) reload spill slots; + * (3) a trailing ~96 B block that the tail's `?:` chain allocates and never touches. + * Proven by ablation: deleting the min/max tail drops vars by exactly 96 while leaving the + * spill slot where it was; `volatile`, plain, and inner-block declarations of pEnd ALL land + * in stratum (1)/(2) and therefore always BELOW the block. The target's 0x108 sits at the + * TOP of the block, so its storage is allocated after everything -- i.e. it is not a C local + * in the original at all. Every source-level lever for it was byte-refuted: + * - `volatile` local + filler array -> slot tracks the filler linearly, frame grows too + * - plain local, natural reload spill -> right slot stratum, WRONG code shape (76-97 diffs) + * - `__asm__` opacity on the pointer -> 184 diffs + * - inner-block declaration -> slot unchanged (gcc walks the whole BLOCK tree in + * expand_function_start, so nesting does NOT delay it) + * - zero-temp tail (if/else or operands bound to locals) -> kills the 96 B block, but the + * tail's shape DEPENDS on the memory operands (each MIN re-reads outp[i][j]: lhu+lh), + * so a zero-temp tail is 219/234 ins, not 246. The block and the code are the same fact. + * `s32 dead[7]` below is therefore a DELIBERATE frame-size dial, not a real variable -- it is + * the 32 B that makes vars come out at 256. (Cookbook 83c warns this may be gcc's own spill + * area; here the measurement says stratum (1) is genuinely 32 B larger than the arrays.) + * + * NOTE 2 -- THE HEAD IS A qty_compare TIE (local-alloc.c), not a spelling problem + * Target: lh 0x4 -> $v0 (w), lh 0x0 -> $s5 (x), lh 0x6 -> $v1 (h), lh 0x2 -> $s4 (y); + * then mw=w ($a1), xw=x+w ($s7), mh=h ($a0), slt w,h. + * Writing the body in that order (mw; xw; mh; yh) reproduces the ORDER and puts mh in $a0 + * correctly, but flips w/h to $v1/$v0 -- because QTY_CMP_PRI = log2(nrefs)*nrefs*size / + * (death-birth) and with that order w's live range is one insn LONGER than h's, so h wins + * $v0. Writing it as (mw; mh; xw; yh) -- the form below -- keeps w in $v0 but costs the + * emission order and puts mh in $a2. Both cost exactly 5 diffs; 72 statement permutations, + * 24 declaration orders, s16/s32 retypings, `?:`-MAX spellings, ref-count shifts + * (`yh = y + mh`), and $v0/$v1 pins were all swept -- the floor is 12 in every direction. + * tools/permuter_ils.py (REGALLOC profile, 5 warm restarts x 220 s x -j10) also plateaus + * at exactly 12. This is a genuine local optimum for source-level mutation. + * + * VERIFY: + * .venv/bin/python tools/match_one.py func_8017C294 --c * --asm-subdir asm/ov_SC01_077/nonmatchings/ov_SC01_077_jr_8017AE2C + */ +#include "common.h" + +#ifndef MIN +#define MIN(a, b) ((a) < (b) ? (a) : (b)) +#endif +#ifndef MAX +#define MAX(a, b) ((a) > (b) ? (a) : (b)) +#endif + +extern s32 func_800491EC(void); +extern void func_8017C66C(u16 *a0, void *a1); +extern int func_8017C710(short *param_1, short *param_2, short *param_3, int param_4); +extern void func_8017C908(s32 a0, s32 a1); + +void func_8017C294(short *a0, short *a1, s32 a2) +{ + short pos[4][4]; + short mat[4][4]; + short outp[4][4]; + short zero[4]; + short base[4]; + s32 dead[7]; + short *volatile pEnd; + short *p; + short *m; + short *o; + short *e; + short *dst; + s32 x, y, w, h, xw, yh, d, z; + s16 mw, mh; + s32 t, r; + register s32 n __asm__("$4"); + register s32 hi __asm__("$2"); + s32 bx, bz; + s32 minx, minz, maxx, maxz; + + (void)&dead; + w = a0[2]; + x = a0[0]; + h = a0[3]; + y = a0[1]; + mw = w; + mh = h; + xw = x + w; + yh = y + h; + if (w < h) { + mw = mh; + } + o = outp[0]; + m = mat[0]; + pEnd = pos[0]; + p = pEnd; + d = (s16)(mw / 20); + z = func_800491EC(); + pos[0][0] = x; + pos[0][1] = y; + pos[0][2] = z; + pos[1][0] = xw; + pos[1][1] = y; + pos[1][2] = z; + pos[2][0] = x; + pos[2][1] = yh; + pos[2][2] = z; + pos[3][0] = xw; + pos[3][1] = yh; + pos[3][2] = z; + zero[0] = 0; + zero[1] = 0; + zero[2] = 0; + func_8017C66C((u16 *)zero, base); + do { + func_8017C66C((u16 *)p, m); + if (func_8017C710(base, m, o, a2) != 0) { + func_8017C908((s32)o, (s32)p); + t = p[0]; + dst = p; + n = t; + r = x - d; + if (t >= r) { + hi = xw + d; + r = hi; + __asm__ __volatile__("" : "=r"(r) : "0"(r)); + if (t <= hi) { + r = n; + } + } + *dst = (short)r; + t = p[1]; + dst = p; + n = t; + r = y - d; + if (t >= r) { + hi = yh + d; + r = hi; + __asm__ __volatile__("" : "=r"(r) : "0"(r)); + if (t <= hi) { + r = n; + } + } + dst[1] = (short)r; + func_8017C66C((u16 *)p, m); + func_8017C710(base, m, o, a2); + } + o += 4; + e = pEnd; + __asm__ __volatile__(""); + p += 4; + m += 4; + } while ((s32)p < (s32)(e + 0x10)); + + minx = MIN(MIN(outp[0][0], outp[1][0]), MIN(outp[2][0], outp[3][0])); + bx = base[0]; + if (bx < minx) { + minx = bx; + } + minz = MIN(MIN(outp[0][2], outp[1][2]), MIN(outp[2][2], outp[3][2])); + bz = base[2]; + if (bz < minz) { + minz = bz; + } + maxx = MAX(MAX(outp[0][0], outp[1][0]), MAX(outp[2][0], outp[3][0])); + if (maxx < bx) { + maxx = bx; + } + maxz = MAX(MAX(outp[0][2], outp[1][2]), MAX(outp[2][2], outp[3][2])); + if (maxz < bz) { + maxz = bz; + } + + a1[2] = maxx - minx; + a1[0] = minx; + a1[1] = minz; + a1[3] = maxz - minz; +} diff --git a/.run/s42/ov_SC03_126/func_8017C6F4.c b/.run/s42/ov_SC03_126/func_8017C6F4.c new file mode 100644 index 000000000..b86004cca --- /dev/null +++ b/.run/s42/ov_SC03_126/func_8017C6F4.c @@ -0,0 +1,507 @@ +#include "common.h" + +/* func_8017C6F4 — ov_SC03_126_jr_8017AE2C — MAP-TILE model renderer (947 ins). + * Sibling of the matched func_8017BEBC (952 ins, §47) and func_8017CDF0. + * Outer: screen rect -> 64x64 cell grid window -> per-cell bbox RTPT/RTPS cull. + * Inner: per-prim RTPT -> flag/nclip/opz cull -> switch(w & 0xF): + * 0,1=POLY_F4 / 2,3=POLY_FT4 / 4,5=POLY_F3 / 6,7=POLY_FT3 -> OT insert. + * NOTE: the F3 arm bbox-tests the packet through PolyFT3 offsets (8/0x10/0x18) + * — a source-level copy/paste quirk of this variant, reproduced verbatim. + * + * ===== S42 STATUS: NEAR(63) — 947/947 ins, frame 0x120 EXACT, vars=232 EXACT ===== + * Every instruction, opcode, immediate, stack offset and branch target matches. + * The whole residual is a REGISTER-NAME ROTATION confined to the per-cell 8-corner + * screen-bbox min/max block + the two pointers that feed it: + * cell $t1 (mine) vs $t3 (target) prim $t3 vs $t5 + * xmx1 $t0 vs $t1 xmn1 $a3 vs $a2 + * xmx2 $a2 vs $a3 mnc $a0 vs $t0 mxc $v1 vs $a2 + * Root: `mnc` gets $t0 in the target (so xmx1 is pushed to $t1, so `cell` is pushed + * to $t3, so `prim` is pushed to $t5); in mine `mnc` takes the lower-numbered $a0. + * Swept with no movement off 63: declaration order (7 permutations), inner-block + * scope for the bbox temps / cell / prim / code, separate X vs Y variables, + * mnc/mxc statement order (4), mnc/mxc aliased onto mn/mx/xa32/t32, in-place-min + * spellings, direct-field-read spellings, and 6 register-pin probes. + * `register s32 prim __asm__("$13")` alone drops it to 47 (see the .pin-t5.c + * variant beside this file) but nothing composes past that. + * The permuter cannot be pointed at this draft as-is: run_masked reports + * "Function func_8017C6F4 not found in base.c" (the GTE #define block defeats + * make_base_c) — demacroize the gte_* macros first if someone retries it. + */ + +typedef struct { u32 w0, w1, w2; } Prim126; +typedef struct { u8 *vtx; u32 f4; u32 xx, yy, zz; Prim126 *prim, *end; } Cell126; + +typedef struct { s16 vx, vy; } DVECTOR2; +typedef struct { s16 vx, vy, vz, pad; } SVECTOR2; +typedef struct { s16 m[3][3]; s32 t[3]; } MATRIX2; +typedef struct { u32 tag, rgbc; s16 x0, y0, x1, y1, x2, y2; } PolyF3; +typedef struct { u32 tag, rgbc; s16 x0, y0, x1, y1, x2, y2, x3, y3; } PolyF4; +typedef struct { u32 tag, rgbc; s16 x0, y0; u32 uvc0; s16 x1, y1; u32 uvp1; s16 x2, y2; u16 uv2, p2; } PolyFT3; +typedef struct { u32 tag, rgbc; s16 x0, y0; u32 uvc0; s16 x1, y1; u32 uvp1; s16 x2, y2; u16 uv2, p2; s16 x3, y3; u16 uv3, p3; } PolyFT4; + +#define gte_ldv0(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_ldv3(r0, r1, r2) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 0( %1 );" \ + "lwc2 $3, 4( %1 );" \ + "lwc2 $4, 0( %2 );" \ + "lwc2 $5, 4( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) ) + +#define gte_ldv3c(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 8( %0 );" \ + "lwc2 $3, 12( %0 );" \ + "lwc2 $4, 16( %0 );" \ + "lwc2 $5, 20( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_rtps() __asm__ volatile ("nop;nop;rtps") +#define gte_rtpt() __asm__ volatile ("nop;nop;rtpt") +#define gte_nclip() __asm__ volatile ("nop;nop;nclip") +#define gte_avsz3() __asm__ volatile ("nop;nop;avsz3") +#define gte_avsz4() __asm__ volatile ("nop;nop;avsz4") + +#define gte_stsxy(r0) __asm__ volatile ( \ + "swc2 $14, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 0( %1 );" \ + "swc2 $14, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsxy3c(r0) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 4( %0 );" \ + "swc2 $14, 8( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_f3(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 12( %0 );" \ + "swc2 $14, 16( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsz3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $17, 0( %0 );" \ + "swc2 $18, 0( %1 );" \ + "swc2 $19, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsz4(r0, r1, r2, r3) __asm__ volatile ( \ + "swc2 $16, 0( %0 );" \ + "swc2 $17, 0( %1 );" \ + "swc2 $18, 0( %2 );" \ + "swc2 $19, 0( %3 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ), "r"( r3 ) \ + : "memory" ) + +#define gte_stflg(r0) __asm__ volatile ( \ + "cfc2 $12, $31;" \ + "nop;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stopz(r0) __asm__ volatile ( \ + "swc2 $24, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +void func_8017C6F4(s32 arg0) +{ + extern s32 func_800491EC(void); + extern void func_800547D8(s32, MATRIX2 *); + extern void func_80052E38(MATRIX2 *); + extern void func_8017C014(void *, void *, s32); + extern u8 D_8018F9C0[]; + extern u8 *D_800A5E60; + extern u8 D_800A6610[]; + extern short D_800B9A02; + + s16 rect[4]; + DVECTOR2 tmpxy[4]; + SVECTOR2 box[8]; + SVECTOR2 sxy[8]; + MATRIX2 mtx; + struct { long flag, opz, sz0, sz1, sz2, sz3; } g; + + s32 cx0, cx1, cy0, cy1, y, col; + Cell126 **rowptr; + Cell126 **p; + Cell126 *cell; + Prim126 *prim; + Prim126 *end; + u8 *pkt; + u32 ot; + u8 *vtx; + u8 *va, *vb, *vc, *vd; + u32 w, code; + u32 wx, wy, wz; + u32 xlo, xhi, ylo, yhi, zlo, zhi; + s32 xa32, xb32, t32; + s32 xmn1, xmx1, xmn2, xmx2; + s32 mnc, mxc; + s32 my, mny, mx, mn; + + func_800491EC(); + func_800547D8(arg0 + 0x10, &mtx); + func_80052E38(&mtx); + func_8017C014(D_8018F9C0, rect, *(s32 *)(arg0 + 0x60)); + + pkt = D_800A5E60; + ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14]; + + cx0 = (rect[0] + 0x4000) / 512; + cx1 = (rect[0] + rect[2] + 0x4000) / 512 + 2; + cx0 = (cx0 < 0) ? 0 : ((cx0 > 0x3F) ? 0x3F : cx0); + cx1 = (cx1 < 0) ? 0 : ((cx1 > 0x3F) ? 0x3F : cx1); + cy0 = (rect[1] + 0x4000) / 512 - 1; + cy1 = (rect[1] + rect[3] + 0x4000) / 512 + 2; + cy0 = (cy0 < 0) ? 0 : ((cy0 > 0x3F) ? 0x3F : cy0); + cy1 = (cy1 < 0) ? 0 : ((cy1 > 0x3F) ? 0x3F : cy1); + + rowptr = (Cell126 **)(*(s32 *)(arg0 + 0xC)) + (cy0 * 64 + cx0); + + for (y = cy0; y < cy1; y++, rowptr += 0x40) { + for (col = cx0, p = rowptr; col < cx1; col++, p++) { + cell = *p; + if (cell == 0) continue; + + wx = cell->xx; + xlo = wx & 0xFFFF; + xhi = wx >> 16; + wy = cell->yy; + ylo = wy & 0xFFFF; + yhi = wy >> 16; + wz = cell->zz; + zlo = wz & 0xFFFF; + zhi = wz >> 16; + + box[0].vx = xlo; box[0].vy = ylo; box[0].vz = zlo; + box[1].vx = xhi; box[1].vy = ylo; box[1].vz = zlo; + box[2].vx = xlo; box[2].vy = ylo; box[2].vz = zhi; + box[3].vx = xhi; box[3].vy = ylo; box[3].vz = zhi; + box[4].vx = xlo; box[4].vy = yhi; box[4].vz = zlo; + box[5].vx = xhi; box[5].vy = yhi; box[5].vz = zlo; + box[6].vx = xlo; box[6].vy = yhi; box[6].vz = zhi; + box[7].vx = xhi; box[7].vy = yhi; box[7].vz = zhi; + + gte_ldv3c(&box[0]); + gte_rtpt(); + gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]); + gte_ldv0(&box[3]); + gte_rtps(); + gte_stsxy(&sxy[3]); + gte_ldv3c(&box[4]); + gte_rtpt(); + gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]); + gte_ldv0(&box[7]); + gte_rtps(); + gte_stsxy(&sxy[7]); + + xa32 = sxy[0].vx; + xb32 = sxy[1].vx; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vx; + xb32 = sxy[5].vx; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < mnc) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if (mxc < -0xA0) continue; + if (!(mnc < 0xA1)) continue; + + xa32 = sxy[0].vy; + xb32 = sxy[1].vy; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vy; + xb32 = sxy[5].vy; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < mnc) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if (mxc < -0x78) continue; + if (!(mnc < 0x79)) continue; + + prim = cell->prim; + end = cell->end; + vtx = cell->vtx; + while (prim < end) { + w = prim->w1; + va = vtx + (w & 0xFFFF); + vb = vtx + (w >> 16); + w = prim->w2; + vc = vtx + (w & 0xFFFF); + w = w >> 16; + gte_ldv3(va, vb, vc); + gte_rtpt(); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_nclip(); + code = w & 0xF; + vd = vtx + ((w & 0xFFF0) >> 1); + gte_stopz(&g.opz); + if (g.opz > 0) { + switch (code) { + case 4: + case 5: + gte_stsxy3_f3(pkt); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (((PolyFT3 *)pkt)->x0 > ((PolyFT3 *)pkt)->x1) { + mx = ((PolyFT3 *)pkt)->x0; + mn = ((PolyFT3 *)pkt)->x1; + } else { + mn = ((PolyFT3 *)pkt)->x0; + mx = ((PolyFT3 *)pkt)->x1; + } + if (((PolyFT3 *)pkt)->x2 > mx) mx = ((PolyFT3 *)pkt)->x2; + else if (((PolyFT3 *)pkt)->x2 < mn) mn = ((PolyFT3 *)pkt)->x2; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT3 *)pkt)->y0 > ((PolyFT3 *)pkt)->y1) { + my = ((PolyFT3 *)pkt)->y0; + mny = ((PolyFT3 *)pkt)->y1; + } else { + mny = ((PolyFT3 *)pkt)->y0; + my = ((PolyFT3 *)pkt)->y1; + } + if (((PolyFT3 *)pkt)->y2 > my) my = ((PolyFT3 *)pkt)->y2; + else if (((PolyFT3 *)pkt)->y2 < mny) mny = ((PolyFT3 *)pkt)->y2; + if (my >= -0x78 && mny < 0x79) { + s32 za; + u32 *otp; + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code != 4) g.opz = za + 0x200; + ((PolyF3 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x4000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x14; + } + } + break; + case 6: + case 7: + gte_stsxy3c(&tmpxy[0]); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (mx >= -0xA0 && mn < 0xA1) { + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + if (my >= -0x78 && mny < 0x79) { + s32 za; + u32 *otp; + u32 *tp; + gte_avsz3(); + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code != 6) g.opz = za + 0x200; + *(u32 *)&((PolyFT3 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT3 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT3 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + tp = (u32 *)prim->w0; + ((PolyFT3 *)pkt)->rgbc = tp[0]; + ((PolyFT3 *)pkt)->uvc0 = tp[1]; + ((PolyFT3 *)pkt)->uvp1 = tp[2]; + ((PolyFT3 *)pkt)->uv2 = tp[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x20; + } + } + break; + case 0: + case 1: + gte_stsxy3_f3(pkt); + gte_ldv0(vd); + gte_rtps(); + if (((PolyF4 *)pkt)->x0 > ((PolyF4 *)pkt)->x1) { + mx = ((PolyF4 *)pkt)->x0; + mn = ((PolyF4 *)pkt)->x1; + } else { + mn = ((PolyF4 *)pkt)->x0; + mx = ((PolyF4 *)pkt)->x1; + } + if (((PolyF4 *)pkt)->x2 > mx) mx = ((PolyF4 *)pkt)->x2; + else if (((PolyF4 *)pkt)->x2 < mn) mn = ((PolyF4 *)pkt)->x2; + if (((PolyF4 *)pkt)->y0 > ((PolyF4 *)pkt)->y1) { + my = ((PolyF4 *)pkt)->y0; + mny = ((PolyF4 *)pkt)->y1; + } else { + mny = ((PolyF4 *)pkt)->y0; + my = ((PolyF4 *)pkt)->y1; + } + if (((PolyF4 *)pkt)->y2 > my) my = ((PolyF4 *)pkt)->y2; + else if (((PolyF4 *)pkt)->y2 < mny) mny = ((PolyF4 *)pkt)->y2; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyF4 *)pkt)->x3); + if (((PolyF4 *)pkt)->x3 < mn) mn = ((PolyF4 *)pkt)->x3; + else if (mx < ((PolyF4 *)pkt)->x3) mx = ((PolyF4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyF4 *)pkt)->y3 < mny) mny = ((PolyF4 *)pkt)->y3; + else if (my < ((PolyF4 *)pkt)->y3) my = ((PolyF4 *)pkt)->y3; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + if (code != 0) g.opz = za + 0x200; + ((PolyF4 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x5000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x18; + } + } + } + break; + case 2: + case 3: + gte_stsxy3c(&tmpxy[0]); + gte_ldv0(vd); + gte_rtps(); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsxy(&tmpxy[3]); + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[3].vx > mx) mx = tmpxy[3].vx; + else if (tmpxy[3].vx < mn) mn = tmpxy[3].vx; + if (mx >= -0xA0 && mn < 0xA1) { + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + if (tmpxy[3].vx > my) my = tmpxy[3].vx; + else if (tmpxy[3].vx < mny) mny = tmpxy[3].vx; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + u32 uvw; + gte_avsz4(); + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + if (code != 2) g.opz = za + 0x200; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + *(u32 *)&((PolyFT4 *)pkt)->x3 = *(u32 *)&tmpxy[3]; + tp = (u32 *)prim->w0; + ((PolyFT4 *)pkt)->rgbc = tp[0]; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + } + } + } + break; + } + } + } + prim++; + } + } + } + D_800A5E60 = pkt; +} diff --git a/.run/s42/ov_SC03_126/func_8017C6F4.pin-t5.c b/.run/s42/ov_SC03_126/func_8017C6F4.pin-t5.c new file mode 100644 index 000000000..ad8eecebc --- /dev/null +++ b/.run/s42/ov_SC03_126/func_8017C6F4.pin-t5.c @@ -0,0 +1,488 @@ +#include "common.h" + +/* func_8017C6F4 — ov_SC03_126_jr_8017AE2C — MAP-TILE model renderer (947 ins). + * Sibling of the matched func_8017BEBC (952 ins, §47) and func_8017CDF0. + * Outer: screen rect -> 64x64 cell grid window -> per-cell bbox RTPT/RTPS cull. + * Inner: per-prim RTPT -> flag/nclip/opz cull -> switch(w & 0xF): + * 0,1=POLY_F4 / 2,3=POLY_FT4 / 4,5=POLY_F3 / 6,7=POLY_FT3 -> OT insert. + * NOTE: the F3 arm bbox-tests the packet through PolyFT3 offsets (8/0x10/0x18) + * — a source-level copy/paste quirk of this variant, reproduced verbatim. + */ + +typedef struct { u32 w0, w1, w2; } Prim126; +typedef struct { u8 *vtx; u32 f4; u32 xx, yy, zz; Prim126 *prim, *end; } Cell126; + +typedef struct { s16 vx, vy; } DVECTOR2; +typedef struct { s16 vx, vy, vz, pad; } SVECTOR2; +typedef struct { s16 m[3][3]; s32 t[3]; } MATRIX2; +typedef struct { u32 tag, rgbc; s16 x0, y0, x1, y1, x2, y2; } PolyF3; +typedef struct { u32 tag, rgbc; s16 x0, y0, x1, y1, x2, y2, x3, y3; } PolyF4; +typedef struct { u32 tag, rgbc; s16 x0, y0; u32 uvc0; s16 x1, y1; u32 uvp1; s16 x2, y2; u16 uv2, p2; } PolyFT3; +typedef struct { u32 tag, rgbc; s16 x0, y0; u32 uvc0; s16 x1, y1; u32 uvp1; s16 x2, y2; u16 uv2, p2; s16 x3, y3; u16 uv3, p3; } PolyFT4; + +#define gte_ldv0(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_ldv3(r0, r1, r2) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 0( %1 );" \ + "lwc2 $3, 4( %1 );" \ + "lwc2 $4, 0( %2 );" \ + "lwc2 $5, 4( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) ) + +#define gte_ldv3c(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 8( %0 );" \ + "lwc2 $3, 12( %0 );" \ + "lwc2 $4, 16( %0 );" \ + "lwc2 $5, 20( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_rtps() __asm__ volatile ("nop;nop;rtps") +#define gte_rtpt() __asm__ volatile ("nop;nop;rtpt") +#define gte_nclip() __asm__ volatile ("nop;nop;nclip") +#define gte_avsz3() __asm__ volatile ("nop;nop;avsz3") +#define gte_avsz4() __asm__ volatile ("nop;nop;avsz4") + +#define gte_stsxy(r0) __asm__ volatile ( \ + "swc2 $14, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 0( %1 );" \ + "swc2 $14, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsxy3c(r0) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 4( %0 );" \ + "swc2 $14, 8( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3_f3(r0) __asm__ volatile ( \ + "swc2 $12, 8( %0 );" \ + "swc2 $13, 12( %0 );" \ + "swc2 $14, 16( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsz3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $17, 0( %0 );" \ + "swc2 $18, 0( %1 );" \ + "swc2 $19, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsz4(r0, r1, r2, r3) __asm__ volatile ( \ + "swc2 $16, 0( %0 );" \ + "swc2 $17, 0( %1 );" \ + "swc2 $18, 0( %2 );" \ + "swc2 $19, 0( %3 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ), "r"( r3 ) \ + : "memory" ) + +#define gte_stflg(r0) __asm__ volatile ( \ + "cfc2 $12, $31;" \ + "nop;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stopz(r0) __asm__ volatile ( \ + "swc2 $24, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +void func_8017C6F4(s32 arg0) +{ + extern s32 func_800491EC(void); + extern void func_800547D8(s32, MATRIX2 *); + extern void func_80052E38(MATRIX2 *); + extern void func_8017C014(void *, void *, s32); + extern u8 D_8018F9C0[]; + extern u8 *D_800A5E60; + extern u8 D_800A6610[]; + extern short D_800B9A02; + + s16 rect[4]; + DVECTOR2 tmpxy[4]; + SVECTOR2 box[8]; + SVECTOR2 sxy[8]; + MATRIX2 mtx; + struct { long flag, opz, sz0, sz1, sz2, sz3; } g; + + s32 cx0, cx1, cy0, cy1, y, col; + Cell126 **rowptr; + Cell126 **p; + Cell126 *cell; + register Prim126 *prim __asm__("$13"); + Prim126 *end; + u8 *pkt; + u32 ot; + u8 *vtx; + u8 *va, *vb, *vc, *vd; + u32 w, code; + u32 wx, wy, wz; + u32 xlo, xhi, ylo, yhi, zlo, zhi; + s32 xa32, xb32, t32; + s32 xmn1, xmx1, xmn2, xmx2; + s32 mnc, mxc; + s32 my, mny, mx, mn; + + func_800491EC(); + func_800547D8(arg0 + 0x10, &mtx); + func_80052E38(&mtx); + func_8017C014(D_8018F9C0, rect, *(s32 *)(arg0 + 0x60)); + + pkt = D_800A5E60; + ot = (u32)&D_800A6610[(*(u16 *)&D_800B9A02) << 14]; + + cx0 = (rect[0] + 0x4000) / 512; + cx1 = (rect[0] + rect[2] + 0x4000) / 512 + 2; + cx0 = (cx0 < 0) ? 0 : ((cx0 > 0x3F) ? 0x3F : cx0); + cx1 = (cx1 < 0) ? 0 : ((cx1 > 0x3F) ? 0x3F : cx1); + cy0 = (rect[1] + 0x4000) / 512 - 1; + cy1 = (rect[1] + rect[3] + 0x4000) / 512 + 2; + cy0 = (cy0 < 0) ? 0 : ((cy0 > 0x3F) ? 0x3F : cy0); + cy1 = (cy1 < 0) ? 0 : ((cy1 > 0x3F) ? 0x3F : cy1); + + rowptr = (Cell126 **)(*(s32 *)(arg0 + 0xC)) + (cy0 * 64 + cx0); + + for (y = cy0; y < cy1; y++, rowptr += 0x40) { + for (col = cx0, p = rowptr; col < cx1; col++, p++) { + cell = *p; + if (cell == 0) continue; + + wx = cell->xx; + xlo = wx & 0xFFFF; + xhi = wx >> 16; + wy = cell->yy; + ylo = wy & 0xFFFF; + yhi = wy >> 16; + wz = cell->zz; + zlo = wz & 0xFFFF; + zhi = wz >> 16; + + box[0].vx = xlo; box[0].vy = ylo; box[0].vz = zlo; + box[1].vx = xhi; box[1].vy = ylo; box[1].vz = zlo; + box[2].vx = xlo; box[2].vy = ylo; box[2].vz = zhi; + box[3].vx = xhi; box[3].vy = ylo; box[3].vz = zhi; + box[4].vx = xlo; box[4].vy = yhi; box[4].vz = zlo; + box[5].vx = xhi; box[5].vy = yhi; box[5].vz = zlo; + box[6].vx = xlo; box[6].vy = yhi; box[6].vz = zhi; + box[7].vx = xhi; box[7].vy = yhi; box[7].vz = zhi; + + gte_ldv3c(&box[0]); + gte_rtpt(); + gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]); + gte_ldv0(&box[3]); + gte_rtps(); + gte_stsxy(&sxy[3]); + gte_ldv3c(&box[4]); + gte_rtpt(); + gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]); + gte_ldv0(&box[7]); + gte_rtps(); + gte_stsxy(&sxy[7]); + + xa32 = sxy[0].vx; + xb32 = sxy[1].vx; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vx; + xb32 = sxy[5].vx; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < mnc) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if (mxc < -0xA0) continue; + if (!(mnc < 0xA1)) continue; + + xa32 = sxy[0].vy; + xb32 = sxy[1].vy; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vy; + xb32 = sxy[5].vy; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < mnc) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if (mxc < -0x78) continue; + if (!(mnc < 0x79)) continue; + + prim = cell->prim; + end = cell->end; + vtx = cell->vtx; + while (prim < end) { + w = prim->w1; + va = vtx + (w & 0xFFFF); + vb = vtx + (w >> 16); + w = prim->w2; + vc = vtx + (w & 0xFFFF); + w = w >> 16; + gte_ldv3(va, vb, vc); + gte_rtpt(); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_nclip(); + code = w & 0xF; + vd = vtx + ((w & 0xFFF0) >> 1); + gte_stopz(&g.opz); + if (g.opz > 0) { + switch (code) { + case 4: + case 5: + gte_stsxy3_f3(pkt); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (((PolyFT3 *)pkt)->x0 > ((PolyFT3 *)pkt)->x1) { + mx = ((PolyFT3 *)pkt)->x0; + mn = ((PolyFT3 *)pkt)->x1; + } else { + mn = ((PolyFT3 *)pkt)->x0; + mx = ((PolyFT3 *)pkt)->x1; + } + if (((PolyFT3 *)pkt)->x2 > mx) mx = ((PolyFT3 *)pkt)->x2; + else if (((PolyFT3 *)pkt)->x2 < mn) mn = ((PolyFT3 *)pkt)->x2; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT3 *)pkt)->y0 > ((PolyFT3 *)pkt)->y1) { + my = ((PolyFT3 *)pkt)->y0; + mny = ((PolyFT3 *)pkt)->y1; + } else { + mny = ((PolyFT3 *)pkt)->y0; + my = ((PolyFT3 *)pkt)->y1; + } + if (((PolyFT3 *)pkt)->y2 > my) my = ((PolyFT3 *)pkt)->y2; + else if (((PolyFT3 *)pkt)->y2 < mny) mny = ((PolyFT3 *)pkt)->y2; + if (my >= -0x78 && mny < 0x79) { + s32 za; + u32 *otp; + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code != 4) g.opz = za + 0x200; + ((PolyF3 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x4000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x14; + } + } + break; + case 6: + case 7: + gte_stsxy3c(&tmpxy[0]); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (mx >= -0xA0 && mn < 0xA1) { + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + if (my >= -0x78 && mny < 0x79) { + s32 za; + u32 *otp; + u32 *tp; + gte_avsz3(); + if (g.sz0 > g.sz1) { + za = g.sz0; + if (za < g.sz2) za = g.sz2; + } else { + za = g.sz1; + if (za < g.sz2) za = g.sz2; + } + g.opz = za; + if (code != 6) g.opz = za + 0x200; + *(u32 *)&((PolyFT3 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT3 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT3 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + tp = (u32 *)prim->w0; + ((PolyFT3 *)pkt)->rgbc = tp[0]; + ((PolyFT3 *)pkt)->uvc0 = tp[1]; + ((PolyFT3 *)pkt)->uvp1 = tp[2]; + ((PolyFT3 *)pkt)->uv2 = tp[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x20; + } + } + break; + case 0: + case 1: + gte_stsxy3_f3(pkt); + gte_ldv0(vd); + gte_rtps(); + if (((PolyF4 *)pkt)->x0 > ((PolyF4 *)pkt)->x1) { + mx = ((PolyF4 *)pkt)->x0; + mn = ((PolyF4 *)pkt)->x1; + } else { + mn = ((PolyF4 *)pkt)->x0; + mx = ((PolyF4 *)pkt)->x1; + } + if (((PolyF4 *)pkt)->x2 > mx) mx = ((PolyF4 *)pkt)->x2; + else if (((PolyF4 *)pkt)->x2 < mn) mn = ((PolyF4 *)pkt)->x2; + if (((PolyF4 *)pkt)->y0 > ((PolyF4 *)pkt)->y1) { + my = ((PolyF4 *)pkt)->y0; + mny = ((PolyF4 *)pkt)->y1; + } else { + mny = ((PolyF4 *)pkt)->y0; + my = ((PolyF4 *)pkt)->y1; + } + if (((PolyF4 *)pkt)->y2 > my) my = ((PolyF4 *)pkt)->y2; + else if (((PolyF4 *)pkt)->y2 < mny) mny = ((PolyF4 *)pkt)->y2; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyF4 *)pkt)->x3); + if (((PolyF4 *)pkt)->x3 < mn) mn = ((PolyF4 *)pkt)->x3; + else if (mx < ((PolyF4 *)pkt)->x3) mx = ((PolyF4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyF4 *)pkt)->y3 < mny) mny = ((PolyF4 *)pkt)->y3; + else if (my < ((PolyF4 *)pkt)->y3) my = ((PolyF4 *)pkt)->y3; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + if (code != 0) g.opz = za + 0x200; + ((PolyF4 *)pkt)->rgbc = prim->w0; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x5000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x18; + } + } + } + break; + case 2: + case 3: + gte_stsxy3c(&tmpxy[0]); + gte_ldv0(vd); + gte_rtps(); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsxy(&tmpxy[3]); + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + if (tmpxy[0].vx > tmpxy[1].vx) { + mx = tmpxy[0].vx; + mn = tmpxy[1].vx; + } else { + mn = tmpxy[0].vx; + mx = tmpxy[1].vx; + } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[3].vx > mx) mx = tmpxy[3].vx; + else if (tmpxy[3].vx < mn) mn = tmpxy[3].vx; + if (mx >= -0xA0 && mn < 0xA1) { + if (tmpxy[0].vy > tmpxy[1].vy) { + my = tmpxy[0].vy; + mny = tmpxy[1].vy; + } else { + mny = tmpxy[0].vy; + my = tmpxy[1].vy; + } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + if (tmpxy[3].vx > my) my = tmpxy[3].vx; + else if (tmpxy[3].vx < mny) mny = tmpxy[3].vx; + if (my >= -0x78 && mny < 0x79) { + s32 za, zb; + u32 *otp; + u32 *tp; + u32 uvw; + gte_avsz4(); + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + if (code != 2) g.opz = za + 0x200; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + *(u32 *)&((PolyFT4 *)pkt)->x3 = *(u32 *)&tmpxy[3]; + tp = (u32 *)prim->w0; + ((PolyFT4 *)pkt)->rgbc = tp[0]; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + } + } + } + break; + } + } + } + prim++; + } + } + } + D_800A5E60 = pkt; +} diff --git a/docs/backlog.md b/docs/backlog.md index 0d8c53bee..dc993d247 100644 --- a/docs/backlog.md +++ b/docs/backlog.md @@ -2,11 +2,11 @@ > Generated by `tools/backlog.py render` from `.run/backlog.jsonl`. These are functions the Phase-21 automation got **close** on but did NOT byte-match. The whole-binary byte-gate is the sole arbiter (G3/P9): **byte-matches bank and are NOT listed here** — only genuine near-misses/blockers are. Ranked by hand-session priority: **reach** (×N propagation leverage) → **closeness** (match_one mismatch count, lower = closer) → **size**. Each row's `best_draft` is the closest C the machine reached — resume from there. -**Open near-misses:** 836 · by status {'near': 796, 'failed': 40} · by class {'regalloc-order': 4, 'plumbing': 3, 'struct': 4, 'schedule': 2, 'loose-typing': 1, 'WAVE': 8, None: 805, 'STRUCT': 9} +**Open near-misses:** 836 · by status {'near': 796, 'failed': 40} · by class {'regalloc: qty_compare tie + stratum-3 frame slot (cookbook 147)': 1, 'plumbing': 3, 'struct': 4, 'regalloc-order': 3, 'schedule': 2, 'loose-typing': 1, 'WAVE': 8, None: 805, 'STRUCT': 9} | # | addr | reach | class | nins | status | closeness | where it stuck | best draft | |--:|------|------:|-------|-----:|--------|----------:|----------------|------------| -| 1 | func_8017C294 | 2 | regalloc-order | 246 | near | 230 | structurally matched (246 ins, param_3 in $fp, spurious psVar7[1] giv killed via | `.run/backlog_drafts/func_8017C294.c` | +| 1 | func_8017C294 | 16 | regalloc: qty_compare tie + stratum-3 frame slot (cookbook 147) | 246 | near | 12 | ov_SC01_077 | `.run/s42/ov_SC01_077/func_8017C294.c` | | 2 | func_8017F714 | 1 | plumbing | 27 | near | 0 | none — MATCH (27/27 ins, match_one verified) | `.run/backlog_drafts/func_8017F714.c` | | 3 | func_8017E224 | 1 | struct | 29 | near | 0 | none — MATCH (unaligned 8-byte memcpy of global onto stack + cond byte incr) | `.run/backlog_drafts/func_8017E224.c` | | 4 | func_80184A68 | 1 | regalloc-order | 33 | near | 0 | none — MATCH | `.run/backlog_drafts/func_80184A68.c` |