mirror of
https://github.com/Druthulu/BFM-decomp
synced 2026-10-08 17:51:46 -04:00
feat(phase-29): BEHEMOTH func_8017BF14 CLOSED — 45 -> 0 (4,763 ins, the largest match yet)
- 45 -> 37 -> 33 -> 21 -> 11 -> 3 -> 2 -> 0, reproduced 3x from independent work dirs. Verified independently before believing it (R14): match_one MATCH (4763 ins), then harvest_verify --binary ov_SC03_116 BYTE-IDENTICAL, then R22 clean-fleet 140 passed, 0 failed of 140. distinct-code 3,838,143 -> 3,842,906 = 68.1% -> 68.2%. instr 80.5%. Agent was interrupted by a weekly API limit and RESUMED FROM ITS TRANSCRIPT -- its round-2 harness survived, nothing was re-derived. - §80 THE PROCESS CORRECTION, worth more than the match: A DO-NOT-RE-BUY ENTRY IS SCOPED TO ITS BASE, NOT TO THE FUNCTION. Three of round 1's ~40 measured negatives INVERTED on round 2's base -- the same edit (qsingle23) measured 1,040 mismatched on the 45-base and 11 on the 21-base. Re-testing the round-1 negative list cost ~20s and produced THREE of the seven winning levers. Such a table records (edit, base) -> result, NOT edit -> useless; after any lever that moves the base materially, RE-RUN THE NEGATIVE LIST. This retroactively qualifies every do-not-re-buy table in the cookbook (§45, §60b, §75a, §76, §78, §79). Concrete: round 1 measured "removing the va->$t2 pin costs 4% elsewhere" => keep the pin; on a base with c0..c3 at function scope, removing those pins is worth 21->13. Same experiment, opposite conclusion. - MY FLAGGED "#1 MOVE" LOST, and the failure is the finding. I briefed variable REUSE (§45-A / RC-14) as the top lever because it took func_8017F510 from 97->10. Swept in full here: EVERY merge lost, 43-3294 across 8 merges. Reason: the TRI and QUAD grants did not differ by RANK but by IDENTITY -- two independent allocno sets, and re-ranking inside one set cannot fix a two-set problem. Diagnose ranking-vs-identity before reaching for a merge. The actual fix (c0..c3 at FUNCTION scope, 33->21) was read off the two matched relatives (b5:310, b4:338) and confirmed against the target -- the 4th time today that reading a matched relative beat the clever lever. - PIN'S HIDDEN COST, cited: combine_regs' hard-register branch (local-alloc.c:1795, reached from :1295 with already_dead==0) records the pinned reg in qty_phys_sugg UNCONDITIONALLY -- no death guard. A pin invites local-alloc to tie producer chains into it. New cure R7: a zero-byte __asm__ ref keeping the pinned value live past the temp so find_free_reg can't honour the suggestion -- closed the last 2 ins (c1->$a0 is uniquely load-bearing; every alternative pin lost 64 ins). - §78's attribution primitive RUN and REPRODUCED: under -fno-schedule-insns, -fno-schedule-insns2 and both, order unchanged => the rgb transposition was never a sched.c decision. - Cold-start economics complete: round 1 = decode + exact length + exact frame + 99.06%; round 2 = the last 45, and cheaper. Budget TWO passes at this size. 5th source copy-paste artefact found.
This commit is contained in:
@@ -0,0 +1,10 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Splice a new dossier header (lines 4..135 of the b1-derived draft) into the
|
||||
winning round-2 draft. usage: bf14_hdr.py <src.c> <hdr.txt> <out.c>"""
|
||||
import sys
|
||||
src = open(sys.argv[1]).read().splitlines(True)
|
||||
hdr = open(sys.argv[2]).read()
|
||||
assert src[3].startswith('/* ====='), src[3]
|
||||
assert src[134].rstrip().endswith('=== */'), src[134]
|
||||
open(sys.argv[3], 'w').write(''.join(src[:3]) + hdr + ''.join(src[135:]))
|
||||
print('wrote', sys.argv[3])
|
||||
@@ -0,0 +1,696 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Round-2 lever generator for func_8017BF14.
|
||||
|
||||
Same contract as bf14_mk.py: EVERY transformation asserts it actually applied,
|
||||
so a "neutral" reading can never be a silent no-op.
|
||||
|
||||
usage: bf14_mk2.py <base.c> <out.c> lever [lever ...]
|
||||
"""
|
||||
import sys, re
|
||||
|
||||
LEVERS = {}
|
||||
def lever(fn):
|
||||
LEVERS[fn.__name__] = fn
|
||||
return fn
|
||||
|
||||
def _one(src, old, new, n=1):
|
||||
c = src.count(old)
|
||||
assert c == n, 'expected %d of %r, found %d' % (n, old, c)
|
||||
return src.replace(old, new)
|
||||
|
||||
def _all(src, old, new, n):
|
||||
c = src.count(old)
|
||||
assert c == n, 'expected %d of %r, found %d' % (n, old, c)
|
||||
return src.replace(old, new)
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# UNPIN levers -- the b1 base carries four pins baked in.
|
||||
# --------------------------------------------------------------------------
|
||||
@lever
|
||||
def unpin_va(src):
|
||||
return _one(src, ' register u8 *va __asm__("$10");\n u8 *vb, *vc;\n',
|
||||
' u8 *va, *vb, *vc;\n')
|
||||
@lever
|
||||
def unpin_w(src):
|
||||
return _one(src, ' register u32 w __asm__("$5");\n s32 code;\n',
|
||||
' u32 w;\n s32 code;\n')
|
||||
@lever
|
||||
def unpin_f0(src):
|
||||
return _one(src, ' register s32 f0 __asm__("$19");\n s32 f1, f2, f3;\n',
|
||||
' s32 f0, f1, f2, f3;\n')
|
||||
@lever
|
||||
def unpin_c1(src):
|
||||
return _all(src, 'register s32 c1 __asm__("$4"); s32 c0, c2, c3;',
|
||||
's32 c0, c1, c2, c3;', 2)
|
||||
|
||||
# --- re-pin at other registers -------------------------------------------
|
||||
def _repin_c(src, spec):
|
||||
"""spec like 'c0=12,c2=10' -> pin those, leave the rest plain."""
|
||||
want = dict(kv.split('=') for kv in spec.split(','))
|
||||
out = []
|
||||
for nm in ('c0', 'c1', 'c2', 'c3'):
|
||||
if nm in want:
|
||||
out.append('register s32 %s __asm__("$%s");' % (nm, want[nm]))
|
||||
plain = [nm for nm in ('c0', 'c1', 'c2', 'c3') if nm not in want]
|
||||
if plain:
|
||||
out.append('s32 %s;' % ', '.join(plain))
|
||||
new = ' '.join(out)
|
||||
for old in ('register s32 c1 __asm__("$4"); s32 c0, c2, c3;',
|
||||
's32 c0, c1, c2, c3;'):
|
||||
if src.count(old) in (1, 2):
|
||||
return src.replace(old, new)
|
||||
raise AssertionError('no c-decl found')
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# rgb emit-word FORM levers
|
||||
# --------------------------------------------------------------------------
|
||||
_CHAIN_Q = [
|
||||
('rgbw = c0 | cb; rgbw |= c0 << 8; rgbw |= c0 << 16;',
|
||||
'rgbw = (c0 | cb) | (c0 << 8) | (c0 << 16);'),
|
||||
('rgbw = c1 | cb; rgbw |= c1 << 8; rgbw |= c1 << 16;',
|
||||
'rgbw = (c1 | cb) | (c1 << 8) | (c1 << 16);'),
|
||||
('rgbw = c2 | cb; rgbw |= c2 << 8; rgbw |= c1 << 16;',
|
||||
'rgbw = (c2 | cb) | (c2 << 8) | (c1 << 16);'),
|
||||
('rgbw = c3 | cb; rgbw |= c3 << 8; rgbw |= c1 << 16;',
|
||||
'rgbw = (c3 | cb) | (c3 << 8) | (c1 << 16);'),
|
||||
]
|
||||
|
||||
@lever
|
||||
def qsingle(src):
|
||||
"""QUAD lit arm: revert the 3-statement accumulator to ONE expression."""
|
||||
for a, b in _CHAIN_Q:
|
||||
src = _one(src, a, b)
|
||||
return src
|
||||
|
||||
@lever
|
||||
def qsingle23(src):
|
||||
"""QUAD lit arm: single-expression for rgb2/rgb3 ONLY (keep 0/1 chained)."""
|
||||
for a, b in _CHAIN_Q[2:]:
|
||||
src = _one(src, a, b)
|
||||
return src
|
||||
|
||||
@lever
|
||||
def qsingle01(src):
|
||||
for a, b in _CHAIN_Q[:2]:
|
||||
src = _one(src, a, b)
|
||||
return src
|
||||
|
||||
_RGBW_RE = re.compile(
|
||||
r'rgbw = (?P<e>[^;]+?);(?P<mid>\s*(?:rgbw \|= [^;]+;\s*)*)'
|
||||
r'\(\(PolyGT(?P<n>[34]) \*\)pkt\)->rgb(?P<k>\d) = rgbw;')
|
||||
|
||||
def _direct(src, poly):
|
||||
"""collapse `rgbw = ...; rgbw |= ...; pkt->rgbN = rgbw;` into one store."""
|
||||
hits = [0]
|
||||
def rep(m):
|
||||
if m.group('n') != poly:
|
||||
return m.group(0)
|
||||
expr = m.group('e')
|
||||
for extra in re.findall(r'rgbw \|= ([^;]+);', m.group('mid')):
|
||||
expr = '(%s) | %s' % (expr, extra)
|
||||
hits[0] += 1
|
||||
return '((PolyGT%s *)pkt)->rgb%s = %s;' % (poly, m.group('k'), expr)
|
||||
out = _RGBW_RE.sub(rep, src)
|
||||
assert hits[0] == int(poly), 'direct(GT%s): %d sites' % (poly, hits[0])
|
||||
return out
|
||||
|
||||
@lever
|
||||
def qdirect(src):
|
||||
"""QUAD lit arm: no rgbw temp at all -- store the expression directly.
|
||||
Each rgb word then becomes a 1-death LOCAL temp, so local-alloc places
|
||||
them independently and they can alternate $v0/$v1 the way the target does."""
|
||||
return _direct(src, '4')
|
||||
|
||||
@lever
|
||||
def tdirect2(src):
|
||||
"""TRI lit arm: same, no rgbw temp."""
|
||||
return _direct(src, '3')
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# residual (c) part 1: the UNLIT rgbc word.
|
||||
# target: and $a1,$v0,$a2 / or $v1,$a1,$v1 / sw $v1
|
||||
# mine : and $a1,$v0,$a2 / or $a1,$a1,$v1 / sw $a1
|
||||
# `cb` is a function-scope global allocno ($a1). `cb |= 0x101010` writes it
|
||||
# in place. The target instead stores `cb | 0x101010` as a 1-death LOCAL
|
||||
# temp, which combine_regs ties to the DYING constant register ($v1).
|
||||
# --------------------------------------------------------------------------
|
||||
@lever
|
||||
def cb_expr(src):
|
||||
"""unlit arms: `pkt->rgbc = cb | 0x101010;` (drop the `cb |=` statement)."""
|
||||
out, n = re.subn(
|
||||
r'cb = tp\[0\] & 0xFF000000;\s*\n\s*cb \|= 0x101010;\s*\n(\s*)'
|
||||
r'\(\(PolyFT(\d) \*\)pkt\)->rgbc = cb;',
|
||||
lambda m: ('cb = tp[0] & 0xFF000000;\n%s((PolyFT%s *)pkt)->rgbc'
|
||||
' = cb | 0x101010;' % (m.group(1), m.group(2))), src)
|
||||
assert n == 2, n
|
||||
return out
|
||||
|
||||
@lever
|
||||
def cb_expr_one(src):
|
||||
"""unlit arms: the whole thing as ONE expression, no `cb` at all."""
|
||||
out, n = re.subn(
|
||||
r'cb = tp\[0\] & 0xFF000000;\s*\n\s*cb \|= 0x101010;\s*\n(\s*)'
|
||||
r'\(\(PolyFT(\d) \*\)pkt\)->rgbc = cb;',
|
||||
lambda m: ('((PolyFT%s *)pkt)->rgbc = (tp[0] & 0xFF000000) | 0x101010;'
|
||||
% m.group(2)), src)
|
||||
assert n == 2, n
|
||||
return out
|
||||
|
||||
_CHAIN_T = [
|
||||
('rgbw = (c0 | cb) | (c0 << 8) | (c0 << 16);',
|
||||
'rgbw = c0 | cb; rgbw |= c0 << 8; rgbw |= c0 << 16;'),
|
||||
('rgbw = (c1 | cb) | (c1 << 8) | (c1 << 16);',
|
||||
'rgbw = c1 | cb; rgbw |= c1 << 8; rgbw |= c1 << 16;'),
|
||||
('rgbw = (c2 | cb) | (c2 << 8) | (c2 << 16);',
|
||||
'rgbw = c2 | cb; rgbw |= c2 << 8; rgbw |= c2 << 16;'),
|
||||
]
|
||||
@lever
|
||||
def tchain(src):
|
||||
"""TRI lit arm: 3-statement accumulator (base is single-expression)."""
|
||||
for a, b in _CHAIN_T:
|
||||
src = _one(src, a, b)
|
||||
return src
|
||||
|
||||
@lever
|
||||
def tdirect(src):
|
||||
"""TRI lit arm: store the expression directly, no rgbw."""
|
||||
for k in range(3):
|
||||
old = 'rgbw = (c%d | cb) | (c%d << 8) | (c%d << 16);' % (k, k, k)
|
||||
assert src.count(old) == 1, old
|
||||
expr = old.split('= ', 1)[1].rstrip(';')
|
||||
src = src.replace(old, '')
|
||||
st = '((PolyGT3 *)pkt)->rgb%d = rgbw;' % k
|
||||
assert src.count(st) == 1, st
|
||||
src = src.replace(st, '((PolyGT3 *)pkt)->rgb%d = %s;' % (k, expr))
|
||||
return src
|
||||
|
||||
@lever
|
||||
def rgbw_fn(src):
|
||||
"""ONE function-scope `u32 rgbw;` (matched-relative style) instead of per-arm."""
|
||||
n = src.count(' u32 rgbw;\n')
|
||||
m = src.count(' u32 rgbw;\n')
|
||||
assert n + m == 4, (n, m)
|
||||
src = src.replace(' u32 rgbw;\n', '')
|
||||
src = src.replace(' u32 rgbw;\n', '')
|
||||
return _one(src, ' u32 cb;\n', ' u32 cb;\n u32 rgbw;\n')
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# residual (a): named producer-offset temps.
|
||||
# --------------------------------------------------------------------------
|
||||
_PROD = """ w = prim->w1;
|
||||
va = vtx + (w & 0xFFFF);
|
||||
vb = vtx + (w >> 16);
|
||||
w = prim->w2;
|
||||
vc = vtx + (w & 0xFFFF);
|
||||
w = w >> 16;
|
||||
"""
|
||||
|
||||
@lever
|
||||
def prod1(src):
|
||||
"""ONE shared offset variable `vo` for all 3 head offsets -> 3 deaths."""
|
||||
new = """ w = prim->w1;
|
||||
vo = w & 0xFFFF;
|
||||
va = vtx + vo;
|
||||
vo = w >> 16;
|
||||
vb = vtx + vo;
|
||||
w = prim->w2;
|
||||
vo = w & 0xFFFF;
|
||||
vc = vtx + vo;
|
||||
w = w >> 16;
|
||||
"""
|
||||
src = _one(src, _PROD, new)
|
||||
return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n')
|
||||
|
||||
@lever
|
||||
def prod2(src):
|
||||
"""TWO offset variables: `vo` (masked, 2 deaths) and `vs` (shifted, 1)."""
|
||||
new = """ w = prim->w1;
|
||||
vo = w & 0xFFFF;
|
||||
va = vtx + vo;
|
||||
vb = vtx + (w >> 16);
|
||||
w = prim->w2;
|
||||
vo = w & 0xFFFF;
|
||||
vc = vtx + vo;
|
||||
w = w >> 16;
|
||||
"""
|
||||
src = _one(src, _PROD, new)
|
||||
return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n')
|
||||
|
||||
@lever
|
||||
def prod3(src):
|
||||
"""`vo` shared by the head offsets AND the vd offset (4 deaths)."""
|
||||
src = prod1(src)
|
||||
return _one(src, 'vd = vtx + (w & 0xFFF8);',
|
||||
'vo = w & 0xFFF8; vd = vtx + vo;')
|
||||
|
||||
@lever
|
||||
def prodvd(src):
|
||||
"""only the vd offset gets a named temp (shared with nothing)."""
|
||||
src = _one(src, 'vd = vtx + (w & 0xFFF8);',
|
||||
'vo = w & 0xFFF8; vd = vtx + vo;')
|
||||
return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n')
|
||||
|
||||
@lever
|
||||
def prodswap(src):
|
||||
"""commute the producer adds: (w & 0xFFFF) + vtx."""
|
||||
new = """ w = prim->w1;
|
||||
va = (w & 0xFFFF) + vtx;
|
||||
vb = (w >> 16) + vtx;
|
||||
w = prim->w2;
|
||||
vc = (w & 0xFFFF) + vtx;
|
||||
w = w >> 16;
|
||||
"""
|
||||
return _one(src, _PROD, new)
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# residual (b): variable REUSE merges (RC-14 MERGE / cookbook 45-A).
|
||||
# The four colours are simultaneously live, so they cannot merge with each
|
||||
# other -- merge them with values that are DEAD by then instead.
|
||||
# --------------------------------------------------------------------------
|
||||
def _arm(src, which):
|
||||
"""return (start,end) slice of the TRI or QUAD lit arm."""
|
||||
if which == 'tri':
|
||||
a = src.index('/* ---------------- TRI')
|
||||
b = src.index('case 2:')
|
||||
else:
|
||||
a = src.index('/* ---------------- QUAD')
|
||||
b = src.index('D_800A5E60 = pkt;')
|
||||
return a, b
|
||||
|
||||
def _merge(src, victim, survivor, which, ndecl):
|
||||
"""rename `victim` -> `survivor` inside one arm, and drop victim's decl."""
|
||||
a, b = _arm(src, which)
|
||||
seg = src[a:b]
|
||||
new, n = re.subn(r'\b%s\b' % victim, survivor, seg)
|
||||
assert n == ndecl, 'merge %s->%s in %s: %d hits' % (victim, survivor, which, n)
|
||||
return src[:a] + new + src[b:]
|
||||
|
||||
def _mrg(src, which, victim, survivor, dropdecl):
|
||||
a, b = _arm(src, which)
|
||||
seg = src[a:b]
|
||||
seg2 = seg.replace(*dropdecl)
|
||||
assert seg2 != seg, 'decl %r not found in %s arm' % (dropdecl[0], which)
|
||||
seg2, n = re.subn(r'\b%s\b' % victim, survivor, seg2)
|
||||
assert n > 0, 'no %s in %s arm' % (victim, which)
|
||||
return src[:a] + seg2 + src[b:]
|
||||
|
||||
@lever
|
||||
def merge_za_c0_t(src):
|
||||
"""TRI: the max-z temp `za` and `c0` never overlap -> one variable."""
|
||||
return _mrg(src, 'tri', 'za', 'c0', ('s32 za, zb;', 's32 zb;'))
|
||||
|
||||
@lever
|
||||
def merge_za_c0_q(src):
|
||||
return _mrg(src, 'quad', 'za', 'c0', ('s32 za, zb;', 's32 zb;'))
|
||||
|
||||
@lever
|
||||
def merge_zb_c1_q(src):
|
||||
return _mrg(src, 'quad', 'zb', 'c1', ('s32 za, zb;', 's32 za;'))
|
||||
|
||||
@lever
|
||||
def merge_zb_c0_q(src):
|
||||
return _mrg(src, 'quad', 'zb', 'c0', ('s32 za, zb;', 's32 za;'))
|
||||
|
||||
def _mrg_tail(src, which, anchor, victim, survivor, decls):
|
||||
a, b = _arm(src, which)
|
||||
seg = src[a:b]
|
||||
i = seg.index(anchor)
|
||||
head, tail = seg[:i], seg[i:]
|
||||
tail2, n = re.subn(r'\b%s\b' % victim, survivor, tail)
|
||||
assert n > 0, (victim, n)
|
||||
for old, new in decls:
|
||||
if old in head:
|
||||
head = head.replace(old, new)
|
||||
break
|
||||
else:
|
||||
raise AssertionError('no c-decl in %s arm' % which)
|
||||
return src[:a] + head + tail2 + src[b:]
|
||||
|
||||
@lever
|
||||
def merge_f0_c3_q(src):
|
||||
"""QUAD: f0..f3 are dead once the last ATTEN3 has run -> f0 doubles as c3."""
|
||||
return _mrg_tail(src, 'quad', 'CLAMP80(c3', 'c3', 'f0',
|
||||
[('s32 c0, c2, c3;', 's32 c0, c2;'),
|
||||
('s32 c0, c1, c2, c3;', 's32 c0, c1, c2;')])
|
||||
|
||||
@lever
|
||||
def merge_f0_c2_t(src):
|
||||
"""TRI: f0..f3 dead after the last ATTEN3 -> f0 doubles as c2."""
|
||||
return _mrg_tail(src, 'tri', 'CLAMP80(c2', 'c2', 'f0',
|
||||
[('s32 c0, c2, c3;', 's32 c0, c3;'),
|
||||
('s32 c0, c1, c2, c3;', 's32 c0, c1, c3;')])
|
||||
|
||||
@lever
|
||||
def merge_f1_c2_t(src):
|
||||
return _mrg_tail(src, 'tri', 'CLAMP80(c2', 'c2', 'f1',
|
||||
[('s32 c0, c2, c3;', 's32 c0, c3;'),
|
||||
('s32 c0, c1, c2, c3;', 's32 c0, c1, c3;')])
|
||||
|
||||
@lever
|
||||
def merge_f1_c3_q(src):
|
||||
return _mrg_tail(src, 'quad', 'CLAMP80(c3', 'c3', 'f1',
|
||||
[('s32 c0, c2, c3;', 's32 c0, c2;'),
|
||||
('s32 c0, c1, c2, c3;', 's32 c0, c1, c2;')])
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# residual (b2): the CLAMP80 sum accumulator.
|
||||
# target: addu $v0,.. addu $v0,.. addu $v0,.. addiu <c>,$v0,0x10
|
||||
# mine : the whole chain is tied INTO the pinned c1 ($a0) by
|
||||
# combine_regs' `sreg < FIRST_PSEUDO_REGISTER` phys_sugg path.
|
||||
# fix : give the sum its own NAMED variable with >1 death, so
|
||||
# local-alloc.c:472 refuses it a qty and combine_regs bails at its
|
||||
# very first test (`reg_qty[ureg] < 0`).
|
||||
# --------------------------------------------------------------------------
|
||||
_CL = ('#define CLAMP80(C, A0, A1, A2, A3) C = (A0) + (A1) + (A2) + (A3)'
|
||||
' + 0x10; if ((C) > 0x80) C = 0x80\n')
|
||||
_CLS = (_CL +
|
||||
'#define CLAMP80S(C, A0, A1, A2, A3) sv = (A0) + (A1) + (A2) + (A3);'
|
||||
' C = sv + 0x10; if ((C) > 0x80) C = 0x80\n')
|
||||
|
||||
@lever
|
||||
def sumvar_c1(src):
|
||||
"""shared sum variable `sv` on the two c1 CLAMP80 sites only (2 deaths)."""
|
||||
src = _one(src, _CL, _CLS)
|
||||
src = _all(src, 'CLAMP80(c1,', 'CLAMP80S(c1,', 2)
|
||||
return _one(src, ' u32 cb;\n', ' u32 cb;\n s32 sv;\n')
|
||||
|
||||
@lever
|
||||
def sumvar_all(src):
|
||||
"""shared sum variable `sv` on ALL seven CLAMP80 sites."""
|
||||
src = _one(src, _CL, _CLS)
|
||||
n = src.count('CLAMP80(c')
|
||||
assert n == 7, n
|
||||
src = src.replace('CLAMP80(c', 'CLAMP80S(c')
|
||||
return _one(src, ' u32 cb;\n', ' u32 cb;\n s32 sv;\n')
|
||||
|
||||
@lever
|
||||
def sumvar_q(src):
|
||||
"""shared sum variable on the four QUAD CLAMP80 sites."""
|
||||
src = _one(src, _CL, _CLS)
|
||||
a, b = _arm(src, 'quad')
|
||||
seg = src[a:b]
|
||||
n = seg.count('CLAMP80(c')
|
||||
assert n == 4, n
|
||||
src = src[:a] + seg.replace('CLAMP80(c', 'CLAMP80S(c') + src[b:]
|
||||
return _one(src, ' u32 cb;\n', ' u32 cb;\n s32 sv;\n')
|
||||
|
||||
@lever
|
||||
def sumvar_t(src):
|
||||
"""shared sum variable on the three TRI CLAMP80 sites."""
|
||||
src = _one(src, _CL, _CLS)
|
||||
a, b = _arm(src, 'tri')
|
||||
seg = src[a:b]
|
||||
n = seg.count('CLAMP80(c')
|
||||
assert n == 3, n
|
||||
src = src[:a] + seg.replace('CLAMP80(c', 'CLAMP80S(c') + src[b:]
|
||||
return _one(src, ' u32 cb;\n', ' u32 cb;\n s32 sv;\n')
|
||||
|
||||
# --- more producer-offset partitions -------------------------------------
|
||||
@lever
|
||||
def prod_cd(src):
|
||||
"""`vo` shared by the vc offset and the vd offset (target puts both in $v0)."""
|
||||
new = """ w = prim->w1;
|
||||
va = vtx + (w & 0xFFFF);
|
||||
vb = vtx + (w >> 16);
|
||||
w = prim->w2;
|
||||
vo = w & 0xFFFF;
|
||||
vc = vtx + vo;
|
||||
w = w >> 16;
|
||||
"""
|
||||
src = _one(src, _PROD, new)
|
||||
src = _one(src, 'vd = vtx + (w & 0xFFF8);', 'vo = w & 0xFFF8; vd = vtx + vo;')
|
||||
return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n')
|
||||
|
||||
@lever
|
||||
def prod_ad(src):
|
||||
"""`vo` shared by the va offset and the vd offset."""
|
||||
new = """ w = prim->w1;
|
||||
vo = w & 0xFFFF;
|
||||
va = vtx + vo;
|
||||
vb = vtx + (w >> 16);
|
||||
w = prim->w2;
|
||||
vc = vtx + (w & 0xFFFF);
|
||||
w = w >> 16;
|
||||
"""
|
||||
src = _one(src, _PROD, new)
|
||||
src = _one(src, 'vd = vtx + (w & 0xFFF8);', 'vo = w & 0xFFF8; vd = vtx + vo;')
|
||||
return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n')
|
||||
|
||||
@lever
|
||||
def prod_a(src):
|
||||
"""`vo` on the va offset only, but made multi-death by also carrying vd."""
|
||||
return prod_ad(src)
|
||||
|
||||
@lever
|
||||
def cdrop3_t(src):
|
||||
"""TRI arm declares c3 but never uses it -- drop it."""
|
||||
a, b = _arm(src, 'tri')
|
||||
seg = src[a:b]
|
||||
for old, new in (('s32 c0, c2, c3;', 's32 c0, c2;'),
|
||||
('s32 c0, c1, c2, c3;', 's32 c0, c1, c2;')):
|
||||
if old in seg:
|
||||
return src[:a] + seg.replace(old, new) + src[b:]
|
||||
raise AssertionError('no c-decl in tri arm')
|
||||
|
||||
@lever
|
||||
def cdropzb_t(src):
|
||||
"""TRI arm declares zb but never uses it -- drop it."""
|
||||
a, b = _arm(src, 'tri')
|
||||
seg = src[a:b]
|
||||
assert 's32 za, zb;' in seg
|
||||
return src[:a] + seg.replace('s32 za, zb;', 's32 za;') + src[b:]
|
||||
|
||||
# --- c declaration ORDER inside the arm ----------------------------------
|
||||
def _corder(src, order):
|
||||
plain = [c for c in order]
|
||||
txt = 's32 %s;' % ', '.join(plain)
|
||||
for old in ('register s32 c1 __asm__("$4"); s32 c0, c2, c3;',
|
||||
's32 c0, c1, c2, c3;'):
|
||||
if src.count(old) == 2:
|
||||
if 'register' in old:
|
||||
txt = ('register s32 c1 __asm__("$4"); s32 %s;'
|
||||
% ', '.join([c for c in order if c != 'c1']))
|
||||
return src.replace(old, txt)
|
||||
raise AssertionError('no c-decl')
|
||||
|
||||
# --- ref dials -----------------------------------------------------------
|
||||
def _dial(src, name, which, n=1):
|
||||
"""insert n zero-byte ref bumps on `name` right after the c-decl of an arm."""
|
||||
a, b = _arm(src, which)
|
||||
seg = src[a:b]
|
||||
m = re.search(r'( *)(register s32 c1 __asm__\("\$4"\); s32 c0, c2, c3;|s32 c0, c1, c2, c3;)\n', seg)
|
||||
assert m, 'no anchor'
|
||||
ins = ''.join('%s__asm__ __volatile__ ("" :: "r" (%s));\n' % (m.group(1), name)
|
||||
for _ in range(n))
|
||||
seg = seg[:m.end()] + ins + seg[m.end():]
|
||||
return src[:a] + seg + src[b:]
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# residual (b1): the TRI arm's c0/c2 grants.
|
||||
# The target's TRI grants (c0=$t4, c1=$a0, c2=$t2) are IDENTICAL to its QUAD
|
||||
# grants -- i.e. c0..c3 are ONE set of function-scope variables shared by
|
||||
# both arms, exactly as in the matched relatives func_8017D960 /
|
||||
# func_8017F510. Per-cull-block scope splits them into two independent
|
||||
# allocno sets, which is why the TRI set drifts.
|
||||
# --------------------------------------------------------------------------
|
||||
_CDECLS = ('register s32 c1 __asm__("$4"); s32 c0, c2, c3;',
|
||||
's32 c0, c1, c2, c3;')
|
||||
|
||||
def _cfn(src, anchor, pinned):
|
||||
for old in _CDECLS:
|
||||
if src.count(old) == 2:
|
||||
break
|
||||
else:
|
||||
raise AssertionError('no per-arm c-decl')
|
||||
# drop the two per-arm declarations (whole lines)
|
||||
out, n = re.subn(r'[ \t]*%s\n' % re.escape(old), '', src)
|
||||
assert n == 2, n
|
||||
decl = ('register s32 c1 __asm__("$4"); s32 c0, c2, c3;' if pinned
|
||||
else 's32 c0, c1, c2, c3;')
|
||||
assert out.count(anchor) == 1, anchor
|
||||
return out.replace(anchor, anchor + ' %s\n' % decl)
|
||||
|
||||
@lever
|
||||
def cfn(src):
|
||||
"""c0..c3 at FUNCTION scope, c1 still pinned to $a0."""
|
||||
return _cfn(src, ' s32 a0v, a1v, a2v, a3v;\n', True)
|
||||
|
||||
@lever
|
||||
def cfn_np(src):
|
||||
"""c0..c3 at FUNCTION scope, PIN-FREE (matched-relative style)."""
|
||||
return _cfn(src, ' s32 a0v, a1v, a2v, a3v;\n', False)
|
||||
|
||||
@lever
|
||||
def cfn_cb(src):
|
||||
"""c0..c3 at function scope, declared just before `u32 cb;`."""
|
||||
return _cfn(src, ' u32 uvw;\n', True)
|
||||
|
||||
@lever
|
||||
def cfn_top(src):
|
||||
"""c0..c3 at function scope, declared early (before the r/lo/hi block)."""
|
||||
return _cfn(src, ' Part *part;\n', True)
|
||||
|
||||
@lever
|
||||
def cfn_end(src):
|
||||
"""c0..c3 at function scope, declared LAST."""
|
||||
return _cfn(src, ' u32 cb;\n', True)
|
||||
|
||||
@lever
|
||||
def cfn_d(src):
|
||||
"""c0..c3 at function scope, declared right after `s32 d;`."""
|
||||
return _cfn(src, ' s32 d;\n', True)
|
||||
|
||||
|
||||
@lever
|
||||
def unpin_c1n(src):
|
||||
"""drop the c1 pin (function-scope c-decl form, single occurrence)."""
|
||||
return _one(src, 'register s32 c1 __asm__("$4"); s32 c0, c2, c3;',
|
||||
's32 c0, c1, c2, c3;')
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# residual 3889: one CLAMP80 site sums its attenuations in a different
|
||||
# ORDER (a0v+a1v+a3v+a2v). Its `sra` therefore lands in a3v's own register
|
||||
# ($a3) instead of being written in place over a2v's. Another hand-edit
|
||||
# copy-paste artefact, of the same family as report-1 artefacts 1-4.
|
||||
# --------------------------------------------------------------------------
|
||||
_CLAMP_RE = re.compile(r'CLAMP80S?\((c\d), (a0v), (a1v), (a2v), (a3v)\);')
|
||||
|
||||
def _clampswap(src, which, sites):
|
||||
a, b = _arm(src, which)
|
||||
seg = src[a:b]
|
||||
hits = [-1]
|
||||
def rep(m):
|
||||
hits[0] += 1
|
||||
if hits[0] not in sites:
|
||||
return m.group(0)
|
||||
return m.group(0).replace('a2v, a3v', 'a3v, a2v')
|
||||
seg2 = _CLAMP_RE.sub(rep, seg)
|
||||
assert seg2 != seg, 'clampswap %s %s: no site changed' % (which, sites)
|
||||
return src[:a] + seg2 + src[b:]
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# residual 3889 -- a FIFTH copy-paste artefact.
|
||||
# At ONE of the seven ATTEN3 sites the y-axis `else if` branch accumulates
|
||||
# into a2v instead of a3v, while the y-axis KILL branch still says a3v.
|
||||
# Byte-evidence in the target:
|
||||
# 3875 addu $a2,$zero,$zero <- kill branch writes a3v ($a2) [matches]
|
||||
# 3889 sra $a3,$s2,7 <- else branch writes a2v ($a3) [differs]
|
||||
# Cost: zero instructions. Same shape as report-1 artefacts 1-4.
|
||||
# --------------------------------------------------------------------------
|
||||
_A3 = """#define ATTEN3(A, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \\
|
||||
"""
|
||||
_A3W = """#define ATTEN3W(A, AW, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \\
|
||||
A = 0; \\
|
||||
if (F) { \\
|
||||
d = (X) - (CX); if (d < 0) d = (CX) - (X); \\
|
||||
if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \\
|
||||
d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \\
|
||||
if ((RZ) < d) A = 0; \\
|
||||
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \\
|
||||
d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \\
|
||||
if ((RY) < d) A = 0; \\
|
||||
else if ((RLO) < d) AW = (A * (((R) - d) / 4)) >> 7; \\
|
||||
}
|
||||
|
||||
"""
|
||||
_A3CALL = re.compile(r'ATTEN3\((a\dv), (f\d),')
|
||||
|
||||
def _atten3w(src, which, sites, dst):
|
||||
if 'ATTEN3W' not in src:
|
||||
src = _one(src, _A3, _A3W + _A3)
|
||||
a, b = _arm(src, which)
|
||||
seg = src[a:b]
|
||||
hits = [-1]
|
||||
def rep(m):
|
||||
hits[0] += 1
|
||||
if hits[0] not in sites:
|
||||
return m.group(0)
|
||||
return 'ATTEN3W(%s, %s, %s,' % (m.group(1), dst, m.group(2))
|
||||
seg2 = _A3CALL.sub(rep, seg)
|
||||
assert seg2 != seg, 'atten3w %s %s: no site changed' % (which, sites)
|
||||
return src[:a] + seg2 + src[b:]
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# last residual (2258/2259): `c1 << 16` in the TRI arm.
|
||||
# c1 is pinned, so it is a HARD reg from the start; it DIES at the `sll`, so
|
||||
# combine_regs takes its `sreg < FIRST_PSEUDO_REGISTER` branch and records
|
||||
# $a0 in qty_phys_sugg for the sll's temp -> the temp lands in $a0 and the
|
||||
# shift is done in place. The target keeps the temp in $v0.
|
||||
# Cure family: make the temp NOT a 1-death local (a named var used in both
|
||||
# arms), or move the pin off c1 onto a colour whose grant we already match.
|
||||
# --------------------------------------------------------------------------
|
||||
@lever
|
||||
def hivar(src):
|
||||
"""named `hi` for the `c1 << 16` term in BOTH arms -> 2 deaths, no qty."""
|
||||
out, n = re.subn(r'\(c1 << 16\)', 'hi', src)
|
||||
assert n >= 1, n
|
||||
out, m = re.subn(r'rgbw \|= c1 << 16;', 'rgbw |= hi;', out)
|
||||
tri = out.index('/* ---------------- TRI')
|
||||
quad = out.index('/* ---------------- QUAD')
|
||||
# one `hi = c1 << 16;` immediately before the first rgb store of each arm
|
||||
def ins(s, marker):
|
||||
i = s.index(marker)
|
||||
j = s.rindex('\n', 0, s.rindex('rgbw', 0, i) if 'rgbw' in s[:i] else i)
|
||||
return s
|
||||
for anchor in ('rgbw = (c1 | cb)', 'rgbw = c1 | cb'):
|
||||
while anchor in out:
|
||||
k = out.index(anchor)
|
||||
ln = out.rindex('\n', 0, k) + 1
|
||||
pad = out[ln:k]
|
||||
out = out[:ln] + pad + 'hi = c1 << 16;\n' + out[ln:]
|
||||
k2 = out.index(anchor, ln + len(pad) + 14)
|
||||
out = out[:k2] + anchor.replace('rgbw', 'rgbw$') + out[k2 + len(anchor):]
|
||||
out = out.replace('rgbw$', 'rgbw')
|
||||
assert 'hi = c1 << 16;' in out
|
||||
return _one(out, ' u32 cb;\n', ' u32 cb;\n u32 hi;\n')
|
||||
|
||||
@lever
|
||||
def _noop(src):
|
||||
return src
|
||||
|
||||
def _c1live(src, which, anchor):
|
||||
"""RC-15 zero-byte ref that keeps the PINNED c1 ($a0) live past the
|
||||
`sll` of `c1 << 16`. combine_regs records $a0 in qty_phys_sugg
|
||||
unconditionally (local-alloc.c:1798, no death guard), but find_free_reg
|
||||
can only honour a suggestion whose hard reg is actually free over the
|
||||
temp's live range -- so extending c1 past the shift is what refuses it."""
|
||||
a, b = _arm(src, which)
|
||||
seg = src[a:b]
|
||||
assert seg.count(anchor) == 1, (anchor, seg.count(anchor))
|
||||
k = seg.index(anchor) + len(anchor)
|
||||
ln = seg.rindex('\n', 0, seg.index(anchor)) + 1
|
||||
pad = seg[ln:seg.index(anchor)]
|
||||
seg = seg[:k] + '\n' + pad + '__asm__ __volatile__ ("" :: "r" (c1));' + seg[k:]
|
||||
return src[:a] + seg + src[b:]
|
||||
|
||||
def main():
|
||||
base, out, levers = sys.argv[1], sys.argv[2], sys.argv[3:]
|
||||
src = open(base).read()
|
||||
for lv in levers:
|
||||
if lv.startswith('RC:'): # RC:c0=12,c2=10
|
||||
src = _repin_c(src, lv[3:]); continue
|
||||
if lv.startswith('CO:'): # CO:c2,c0,c1,c3
|
||||
src = _corder(src, lv[3:].split(',')); continue
|
||||
if lv.startswith('CL:'): # CL:tri:rgb1
|
||||
_, wh, tag = lv.split(':')
|
||||
poly = '3' if wh == 'tri' else '4'
|
||||
anc = {'rgb0': '((PolyGT%s *)pkt)->rgb0 = rgbw;' % poly,
|
||||
'rgb1': '((PolyGT%s *)pkt)->rgb1 = rgbw;' % poly,
|
||||
'rgb2': '((PolyGT%s *)pkt)->rgb2 = rgbw;' % poly,
|
||||
'uv0': '((PolyGT%s *)pkt)->uv0 = tp[1];' % poly,
|
||||
'end': 'pkt += 0x%s;' % ('28' if wh == 'tri' else '34')}[tag]
|
||||
src = _c1live(src, wh, anc); continue
|
||||
if lv.startswith('AW:'): # AW:quad:1:a2v
|
||||
_, wh, ix, dst = lv.split(':')
|
||||
src = _atten3w(src, wh, set(int(x) for x in ix.split(',')), dst); continue
|
||||
if lv.startswith('CS:'): # CS:quad:1 or CS:tri:0,2
|
||||
_, wh, ix = lv.split(':')
|
||||
src = _clampswap(src, wh, set(int(x) for x in ix.split(','))); continue
|
||||
if lv.startswith('DL:'): # DL:name:tri[:n]
|
||||
p = lv[3:].split(':')
|
||||
src = _dial(src, p[0], p[1], int(p[2]) if len(p) > 2 else 1); continue
|
||||
src = LEVERS[lv](src)
|
||||
open(out, 'w').write(src)
|
||||
|
||||
main()
|
||||
@@ -0,0 +1,15 @@
|
||||
#!/bin/bash
|
||||
# bf14_sw2.sh <base.c> "<tag>|<lever> [lever...]" ... -> one line per variant, parallel
|
||||
# uses bf14_mk2.py (round-2 levers). Reports raw mismatch count + length.
|
||||
cd /home/musashi/bfm-decomp
|
||||
BASE="$1"; shift
|
||||
run() {
|
||||
local spec="$1"; local tag="${spec%%|*}"; local lv="${spec#*|}"
|
||||
local wd=".run/giants/bf14_sw2/$tag"
|
||||
mkdir -p "$wd"
|
||||
python3 .run/giants/bf14_mk2.py "$BASE" "$wd/v.c" $lv 2>"$wd/mk.err" || { printf '%-22s :: MK-FAIL %s\n' "$tag" "$(tail -2 $wd/mk.err|tr '\n' ' ')"; return; }
|
||||
bash .run/giants/bf14_cc.sh "$wd/v.c" "$wd/w" >/dev/null 2>"$wd/cc.err" || { printf '%-22s :: COMPILE-FAIL %s\n' "$tag" "$(tail -3 $wd/cc.err|tr '\n' ' ')"; return; }
|
||||
printf '%-22s :: %s\n' "$tag" "$(python3 .run/giants/bf14_full.py "$wd/w/t.o" --count)"
|
||||
}
|
||||
export -f run; export BASE
|
||||
printf '%s\n' "$@" | xargs -P 8 -I{} bash -c 'run "$@"' _ {}
|
||||
@@ -0,0 +1,383 @@
|
||||
# `func_8017BF14` — behemoth #4, 4,763 ins, `ov_SC03_116` — ROUND 2: **MATCH**
|
||||
|
||||
**Session 20 round 2, 2026-07-25.** Round 1 handed over 45/4763 mismatched.
|
||||
Round 2 closed it.
|
||||
|
||||
---
|
||||
|
||||
## 1. FINAL NUMBER (measured, `tools/match_one.py`, the CANDIDATE gate)
|
||||
|
||||
```
|
||||
python3 tools/match_one.py func_8017BF14 --c .run/giants/s19_func_8017BF14_b2.c \
|
||||
--asm-subdir asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C
|
||||
-> MATCH (4763 ins) func_8017BF14
|
||||
```
|
||||
|
||||
Reproduced **3×** from independent private work dirs. Independent confirmations:
|
||||
|
||||
| check | result |
|
||||
|---|---|
|
||||
| `bf14_full.py` masked index-wise diff | **0 / 4763 mismatched** |
|
||||
| `bf14_hist.py` opcode histogram | `len+0 L1=0` |
|
||||
| `bf14_slots.py` stack-slot census | **127 / 127**, all at the target's offsets |
|
||||
| compile warnings (`-Wall`) | none |
|
||||
|
||||
**The whole-binary SHA1 arbiter (G3/P9) was NOT run** — the task forbade touching
|
||||
the build tree. `match_one` is the candidate check only; the coordinator's
|
||||
whole-binary gate is the sole arbiter (G3/P9).
|
||||
|
||||
Trajectory, every step measured: **45 → 37 → 33 → 21 → 11 → 3 → 2 → 0.**
|
||||
|
||||
---
|
||||
|
||||
## 2. PER-RESIDUAL OUTCOME
|
||||
|
||||
Round 1 split the 45 into (a) producer temps ≈8, (b) `c0`/`c2` grants ≈15,
|
||||
(c) quad-lit rgb accumulator ≈20. All three fell. The exact split of the 45,
|
||||
recounted from the diff, was **(a) 8 + (b) 23 + (c) 14**.
|
||||
|
||||
### (a) Prim-word producer temps — **FELL** (8 ins, idx 508–512, 539–542)
|
||||
|
||||
Round 1's diagnosis (`combine_regs` tying the chain into the pinned `va`) was
|
||||
**correct**, and its prescribed cure (named MULTI-death offset variables) does
|
||||
work — `prod2` broke the `$t2` tie exactly as predicted, byte-visible:
|
||||
|
||||
```
|
||||
base 508 andi $t2,$a1,0xffff 510 addu $t2,$t6,$t2 <- tied, in place
|
||||
prod2 508 andi $v1,$a1,0xFFFF 510 addu $t2,$t6,$v1 <- MATCHES target
|
||||
```
|
||||
|
||||
But it is a **conservation law, not a fix**: one shared `vo` gives one register
|
||||
for all its sites, whereas the target uses three distinct temps ($v1, $a0, $v0,
|
||||
$v0 for the four offsets). `prod2` fixed 508/510 and broke 513/515 — 45 → 45.
|
||||
Every partition of the four offsets across named variables was swept
|
||||
(`prod1/2/3`, `prod_ad`, `prod_cd`, `prodvd`, `prodswap`): best is neutral.
|
||||
|
||||
**What actually fixed it: R4 — dropping the `va→$t2` and `w→$a1` pins.**
|
||||
Round 1 measured those pins as worth 4% and the brief said not to re-buy their
|
||||
removal. That was true *of the round-1 base* and is **false** once `c0..c3` sit
|
||||
at function scope (R3): on the 21-base, `unpin_va` = 17, `unpin_va unpin_w` =
|
||||
**13**. This is base-dependence, not a contradiction — and it is the single
|
||||
most important methodological lesson of the round (§5.1).
|
||||
|
||||
### (b) `c0`/`c2` grants — **FELL** (23 ins)
|
||||
|
||||
Round 1 pointed at `allocno_compare` order and prescribed a variable-REUSE
|
||||
merge. The **reuse sweep was run in full and every merge lost** (§4.3): merging
|
||||
`za`/`zb`/`f0`/`f1` into `c0..c3` scored 43–3294 against a 33/37 base. Reuse is
|
||||
now a measured dead end on this function.
|
||||
|
||||
The real lever was found by **reading the target and the matched relatives**,
|
||||
which is where round 1 said the value was:
|
||||
|
||||
* The target's TRI grants are `c0→$t4, c1→$a0, c2→$t2` — **identical to its
|
||||
QUAD grants**. Two independently-scoped allocno sets cannot coincide by
|
||||
chance; one shared set can.
|
||||
* Both matched relatives declare the colours at **function scope**:
|
||||
`.run/giants/s19_func_8017D960_b5.c:310` and `s19_func_8017F510_b4.c:338`.
|
||||
|
||||
`s32 c0, c1, c2, c3;` at function scope: **33 → 21**. Declaration *position* is
|
||||
neutral (5 anchors swept, all 21) — consistent with round 1's L4.2 oracle, since
|
||||
these never take stack slots.
|
||||
|
||||
This **reverses round-1's L4**, which put them per-cull-block. L4 was right on
|
||||
its own base — it was supplying the extra local allocno that spills `r1lo` — but
|
||||
R1 and R2 supply that pressure now.
|
||||
|
||||
A second, separable part of (b) was the c1 **sum accumulator** (9 ins, idx
|
||||
1876-1879 / 3890-3893), which round 1 had classified with the grants. It is a
|
||||
pin artefact, and `CLAMP80S` (R1) fixed it: **45 → 37**.
|
||||
|
||||
### (c) Quad-lit rgb accumulator — **FELL** (14 ins)
|
||||
|
||||
Two independent pieces:
|
||||
|
||||
* **Unlit `rgbc` (4 ins, idx 2310/2311, 4699/4700).** `cb` is a function-scope
|
||||
global allocno, so `cb |= 0x101010` writes it in place ($a1). The target
|
||||
stores the *expression* `cb | 0x101010`, a 1-death local that `combine_regs`
|
||||
ties to the **dying constant register** $v1. **37 → 33.**
|
||||
* **Quad lit rgb2/rgb3 (10 ins, idx 4643-4652).** Reverting those two to the
|
||||
single-expression form (rgb0/rgb1 keep the 3-statement accumulator) makes the
|
||||
intermediates 1-death local temps that alternate $v0/$v1 — which is what lets
|
||||
the target's store of the previous rgb word sit one slot later. **21 → 11.**
|
||||
All-four = 23, rgb0/rgb1-only = 21, direct-store = 28. The split really is
|
||||
2-and-2, matching artefact 4 (rgb2/rgb3 are the two that take `c1 << 16`).
|
||||
|
||||
### THE ATTRIBUTION PRIMITIVE FOR (c) — run, and decisive
|
||||
|
||||
```
|
||||
-fno-schedule-insns -> my order unchanged
|
||||
-fno-schedule-insns2 -> my order unchanged
|
||||
both -> my order unchanged
|
||||
```
|
||||
|
||||
In all three builds `sw v1,-20(t3)` still precedes `or v1,t2,a1`. **The
|
||||
transposition was never a `sched.c` decision.** With a 3-statement accumulator
|
||||
the value is pinned to one register, so the next `or` clobbers $v1 and *no*
|
||||
scheduler could hoist it above the store — the ordering is a consequence of the
|
||||
register grant. Changing the grant (R5) fixed the order for free. This is
|
||||
**§78 reproduced exactly**, and it is the second time on this family that a
|
||||
"scheduling" residual was really a register grant. Do not reason about `sched.c`
|
||||
on this family until this primitive has been run.
|
||||
|
||||
---
|
||||
|
||||
## 3. THE ONE MECHANISM BEHIND FOUR OF THE SEVEN LEVERS
|
||||
|
||||
R1, R2, R4 and R7 are all the same compiler fact, and it is worth a cookbook
|
||||
entry because it is the *cost* of the pin technique:
|
||||
|
||||
> A `register __asm__` pin makes the variable a **hard register in the RTL from
|
||||
> the start**. When a 1-death local temp is produced from — or consumed into —
|
||||
> that hard reg, `local-alloc.c`'s `combine_regs` takes its hard-register branch
|
||||
> (`local-alloc.c:1795-1820`) and records the pinned register in
|
||||
> `qty_phys_sugg` for the temp's quantity. **That path has no death guard and no
|
||||
> cost model — it fires unconditionally.** The temp then lands in the pinned
|
||||
> register and the operation is performed IN PLACE. An ordinary pseudo never
|
||||
> gets that suggestion, because anything crossing a basic block has
|
||||
> `reg_qty == -1` and `combine_regs` bails at its very first test.
|
||||
|
||||
Verified in source, `tools/reference/gcc-2.7.2/local-alloc.c`:
|
||||
|
||||
* `:472` — a pseudo is local iff `reg_basic_block[i] >= 0 && reg_n_deaths[i] == 1`.
|
||||
* `:1763` — `combine_regs` returns 0 immediately if `reg_qty[ureg] < 0`.
|
||||
* `:1795` — `if (ureg < FIRST_PSEUDO_REGISTER) { ... qty_phys_sugg |= ureg; return 0; }`
|
||||
— the branch that costs us, reached from `block_alloc` at `:1295` with
|
||||
`already_dead = 0`.
|
||||
|
||||
**Three separable cures, all three used in this match:**
|
||||
|
||||
| cure | how | used by |
|
||||
|---|---|---|
|
||||
| (i) refuse the temp a quantity | give it a NAMED variable with **>1 death**, so `:472` rejects it and `combine_regs` bails at `:1763` | **R1** (`CLAMP80S`/`sv`) |
|
||||
| (ii) remove the pin | only if the pin is not load-bearing — **re-measure, it is base-dependent** | **R4** (`va`, `w`) |
|
||||
| (iii) starve the suggestion | keep the pinned value **LIVE past the temp**, so `find_free_reg` cannot honour the suggestion | **R7** (zero-byte ref on `c1`) |
|
||||
|
||||
Cure (iii) is new and is the one that closed the function. The last two
|
||||
instructions were `sll $a0,$a0,16` (mine, in place over the pinned `c1`) vs
|
||||
`sll $v0,$a0,16` (target). `c1` could not be unpinned — it is what spills
|
||||
`r1lo`, and dropping it or moving it to any other colour costs −64 length
|
||||
(measured, §4.5). So instead:
|
||||
|
||||
```c
|
||||
((PolyGT3 *)pkt)->rgb1 = rgbw;
|
||||
__asm__ __volatile__ ("" :: "r" (c1)); /* RC-15, zero bytes */
|
||||
```
|
||||
|
||||
`c1` is now live past the shift, `$a0` is unavailable, the temp falls to `$v0`.
|
||||
**2 → 0.** Placing it after the rgb2 store also matches; after `uv0` or at the
|
||||
arm's end does not (they perturb length).
|
||||
|
||||
---
|
||||
|
||||
## 4. EVERY LEVER MEASURED THIS ROUND
|
||||
|
||||
Metric is the `match_one` **mismatch count** (length is exact throughout, so the
|
||||
raw count is honest — the round-1 metric trap of §5.3 does not apply). Baselines
|
||||
are stated per block because the base moved as levers landed.
|
||||
|
||||
### 4.1 The winning chain
|
||||
|
||||
| # | lever | base → result |
|
||||
|---|---|---|
|
||||
| **R1** | `CLAMP80S` — named `sv` sum variable on the **two c1 sites only** | 45 → **37** |
|
||||
| **R2** | unlit arms: `pkt->rgbc = cb \| 0x101010;` (expression, not `cb \|=`) | 37 → **33** |
|
||||
| **R3** | **`s32 c0, c1, c2, c3;` at FUNCTION scope** | 33 → **21** |
|
||||
| **R4** | drop the `va→$t2` and `w→$a1` pins | 21 → 17 → **13** |
|
||||
| **R5** | quad-lit **rgb2/rgb3 only** revert to single-expression | 21 → **11**; with R4 → **3** |
|
||||
| **R6** | artefact 5 — `ATTEN3W(a3v, a2v, …)` at the QUAD vertex-1 site | 3 → **2** |
|
||||
| **R7** | RC-15 zero-byte ref on pinned `c1` after the TRI rgb1 store | 2 → **0** |
|
||||
|
||||
### 4.2 Neutral — measured, no effect at all
|
||||
|
||||
| lever | base | result |
|
||||
|---|---|---|
|
||||
| `prod1` / `prod2` / `prodvd` / `prodswap` (named producer-offset temps) | 45 | 45 (fixes 508/510, breaks 513/515) |
|
||||
| same four | 21 | 21 |
|
||||
| `cdrop3_t` (drop the unused `c3` from the TRI arm) | 45 / 37 | 45 / 37 |
|
||||
| `cdropzb_t` (drop the unused `zb` from the TRI arm) | 45 | 45 |
|
||||
| `qsingle01` (single-expression for quad rgb0/rgb1) | 21 | 21 |
|
||||
| `prod2` on top of `sumvar_c1` | 45 | 37 (= R1 alone) |
|
||||
| declaration POSITION of the function-scope `c0..c3` — 5 anchors: before `a0v`, before `cb`, before `d`, before `part`, last | 33 | **21 at every anchor** |
|
||||
| `qsingle23 + prodvd` / `+ prodswap` / `+ prod1` / `+ prod2` | 11 | 11 |
|
||||
|
||||
### 4.3 The REUSE sweep — run in full, every merge LOST
|
||||
|
||||
This was round 1's flagged #1 move. It is now a measured dead end here.
|
||||
|
||||
| merge | base | result |
|
||||
|---|---|---|
|
||||
| `za` → `c0` (QUAD) | 45 | 55 |
|
||||
| `za` → `c0` (TRI) | 45 | 3294 |
|
||||
| `zb` → `c1` (QUAD) | 45 | 55 |
|
||||
| `zb` → `c0` (QUAD) | 45 / 37 | 51 / 43 |
|
||||
| `f0` → `c3` (QUAD) | 45 / 37 | 51 / 43 |
|
||||
| `f0` → `c2` (TRI) | 45 | 49 |
|
||||
| `f1` → `c2` (TRI) | 45 | 838 |
|
||||
| `f1` → `c3` (QUAD) | 45 | 841 |
|
||||
|
||||
**Why it failed, and this generalises:** a REUSE merge raises the survivor's
|
||||
`reg_n_refs` to move `allocno_compare` priority. But the TRI/QUAD `c0..c3`
|
||||
grants did not disagree because of *priority* — they disagreed because the two
|
||||
arms had **two independent allocno sets** at all. No amount of re-ranking inside
|
||||
one set can make it agree with a different set; only merging the sets can (R3).
|
||||
**Diagnose whether two grants differ by RANK or by IDENTITY before reaching for
|
||||
a ref-count lever.**
|
||||
|
||||
### 4.4 rgb emit-word forms
|
||||
|
||||
| lever | base | result |
|
||||
|---|---|---|
|
||||
| `qsingle` (all four quad words single-expression) | 45 / 37 / 21 | 47 / 39 / 23 |
|
||||
| `qsingle23` (**rgb2/rgb3 only**) | 45 / 21 | **1040 / 11** ← extreme base-dependence |
|
||||
| `qsingle01` (rgb0/rgb1 only) | 45 / 21 | 45 / 21 |
|
||||
| `qdirect` (no `rgbw`, store the expression) | 37 / 21 | 44 / 28 |
|
||||
| `tdirect2` (TRI, no `rgbw`) | 37 / 21 | 47 / 37 |
|
||||
| `qdirect + tdirect2` | 37 | 54 |
|
||||
| `tchain` (TRI 3-statement accumulator) | 45 | 2514 (len 4762) |
|
||||
| `cb_expr_one` (unlit, drop `cb` entirely) | 37 | 2423 (len 4767) |
|
||||
| one function-scope `u32 rgbw;` (relative style) | 37 | n/a — 4 per-arm decls, lever refused |
|
||||
|
||||
### 4.5 The pins — re-measured on every base
|
||||
|
||||
| lever | base | result |
|
||||
|---|---|---|
|
||||
| `unpin_va` | 45 / 21 | 174 / **17** |
|
||||
| `unpin_w` | 45 / 21 | 47 / 23 |
|
||||
| `unpin_va + unpin_w` | 45 / 21 | 170 / **13** |
|
||||
| `unpin_f0` | 45 / 21 | 88 / 64 — **f0→$s3 stays** |
|
||||
| `unpin_c1` | 45 / 21 / 3 | 654 / — / **len 4699 (−64)** |
|
||||
| all four unpinned | 45 | 789 (the round-1 pin-free fallback) |
|
||||
| c0..c3 at function scope, **pin-free** | 33 | len 4699 (−64) |
|
||||
| re-pin `c0→$t4` and/or `c2→$t2`/`c3→$a2` **instead of** c1 | 2 | **len 4699 every time** |
|
||||
| re-pin `c0→$t4` **plus** c1 | 2 | 14 |
|
||||
| re-pin `c1 + c2` / `c0+c1+c2` / all four / `c1+c3` | 2 | 60 / 71 / 320 / 307 |
|
||||
|
||||
**The `c1→$a0` pin is uniquely load-bearing: it is the register pressure that
|
||||
spills `r1lo`.** No other colour, and no combination without it, reproduces the
|
||||
spill. Round-1 §5.2's "pinning c0 and c2 *in addition to* c1 is worse" is
|
||||
confirmed and extended: pinning them *instead of* c1 does not even preserve the
|
||||
frame.
|
||||
|
||||
### 4.6 Artefact-5 localisation (the `sra $a2` vs `$a3` residual)
|
||||
|
||||
| lever | base | result |
|
||||
|---|---|---|
|
||||
| `ATTEN3W(a3v, a2v, …)` at QUAD site **1** | 3 | **2** |
|
||||
| same at QUAD sites 0 / 2 / 3 | 3 | 4 / 4 / 4 |
|
||||
| same at TRI sites 0 / 1 / 2 | 3 | 4 / 4 / 4 |
|
||||
| `ATTEN3W(a3v, a1v, …)` / `(a3v, a0v, …)` at QUAD site 1 | 3 | 3 / 3 |
|
||||
| swap the last two `CLAMP80` args (a2v↔a3v) — **all 7 sites tried one at a time** | 3 | 5 at every site |
|
||||
|
||||
The `CLAMP80` argument-order hypothesis is **refuted**: the sum order is
|
||||
identical (3890-3893 are byte-identical in both), and the target's kill branch
|
||||
at idx 3875 (`addu $a2,$zero,$zero`) already agrees. Only the y-axis **`else if`
|
||||
destination** differs — a genuine fifth copy-paste artefact, costing zero
|
||||
instructions.
|
||||
|
||||
### 4.7 Other
|
||||
|
||||
| lever | base | result |
|
||||
|---|---|---|
|
||||
| `sumvar_all` (shared `sv` on all 7 CLAMP sites) | 45 | 4568, **len 4699** |
|
||||
| `sumvar_q` / `sumvar_t` (per-arm) | 45 | 41 / 41 |
|
||||
| `prod3` (`vo` on all four offsets) | 45 / 21 | 47 / 23 |
|
||||
| `prod_ad` / `prod_cd` | 45 / 21 | 50, 52 / 26, 28 |
|
||||
| `hivar` (named `hi` for `c1 << 16`, both arms) | 2 | 226 (len 4765) |
|
||||
| `hivar + unpin_c1` | 2 | 4549 (len 4697) |
|
||||
| zero-byte `c1` ref after the TRI **uv0** store / at arm **end** | 2 | 2471 (len 4764) / 2446 (len 4766) |
|
||||
|
||||
---
|
||||
|
||||
## 5. LESSONS TO FEED BACK (cookbook candidates)
|
||||
|
||||
### 5.1 A "do-not-re-buy" entry is scoped to the BASE that measured it
|
||||
|
||||
Three of round 1's measured, correctly-recorded findings inverted once the base
|
||||
moved:
|
||||
|
||||
| round-1 finding | round-2 measurement |
|
||||
|---|---|
|
||||
| L4: `c0..c3` per cull block (52% → 93%) | function scope is **strictly better**, 33 → 21 |
|
||||
| L9: removing the `va` pin costs 4% | removing `va` **and** `w` is worth 21 → 13 |
|
||||
| L8: quad rgb 3-statement accumulator is +0.02% | single-expression for rgb2/rgb3 is worth 21 → 11 |
|
||||
|
||||
`qsingle23` is the extreme case: **1040 mismatched on the 45-base, 11 on the
|
||||
21-base** — the same edit, two orders of magnitude apart. None of these were
|
||||
errors in round 1; they were correct readings of a different base.
|
||||
|
||||
**Rule candidate:** a do-not-re-buy table must record *the base it was measured
|
||||
against*, and any entry measured against a base that has since moved by a
|
||||
structural lever is **stale, not settled** — re-measure the cheap ones (one
|
||||
0.28 s probe each) rather than inheriting them. Re-testing the whole round-1
|
||||
"negative" list on the new base cost about 20 seconds of compute and produced
|
||||
three of the seven winning levers.
|
||||
|
||||
### 5.2 The pin's hidden cost: `combine_regs`' unconditional `qty_phys_sugg`
|
||||
|
||||
§3 above, with source citations. Worth its own cookbook section: it explains a
|
||||
whole *class* of 2-instruction "in-place vs not" residuals, and it gives three
|
||||
separable cures. Cure (iii) — the zero-byte liveness extension — is new, and it
|
||||
is how a pin can be kept for its allocation pressure while its tie is refused.
|
||||
|
||||
### 5.3 Sibling grants are an IDENTITY oracle, not just a hint
|
||||
|
||||
If two code paths in the target show *identical* register grants for
|
||||
corresponding variables, those variables are **one set of allocnos** — i.e. one
|
||||
declaration at a scope enclosing both. That is a positive structural inference
|
||||
from register numbers alone, and it beat an exhaustive scope sweep plus a full
|
||||
reuse sweep. It also agreed with what the two matched relatives already showed,
|
||||
which is the round-1 report's own advice (§"MATCHED RELATIVES") paying off again:
|
||||
**5 of 9 winning levers on the last behemoth, and 2 of 7 here (R3, and R5's
|
||||
2-and-2 split), came straight off a matched relative or off the target's own
|
||||
register numbering.**
|
||||
|
||||
### 5.4 Run the attribution primitive before any scheduling reasoning
|
||||
|
||||
§2. Two for two on this family: an apparent scheduling residual that was a
|
||||
register grant. Cost: three compiles.
|
||||
|
||||
---
|
||||
|
||||
## 6. WHAT CHANGED IN THE SOURCE (7 hunks vs `s19_func_8017BF14_b1.c`)
|
||||
|
||||
1. `ATTEN3W` macro added (artefact 5, y-axis `else` destination).
|
||||
2. `CLAMP80S` macro added (named `sv` sum).
|
||||
3. `va` / `w` pins removed; `c0..c3` (with the `c1` pin) moved to function
|
||||
scope; `s32 sv;` declared.
|
||||
4. `CLAMP80(c1, …)` → `CLAMP80S(c1, …)` in both arms.
|
||||
5. TRI arm: zero-byte `c1` ref after the rgb1 store; unlit `rgbc` written as an
|
||||
expression.
|
||||
6. QUAD arm: `ATTEN3W` at the vertex-1 site; rgb2/rgb3 single-expression; unlit
|
||||
`rgbc` written as an expression.
|
||||
7. The two per-arm `register s32 c1 …; s32 c0, c2, c3;` declarations removed.
|
||||
|
||||
---
|
||||
|
||||
## 7. FILES
|
||||
|
||||
* `.run/giants/s19_func_8017BF14_b2.c` — **the match**, full updated dossier.
|
||||
* `.run/giants/s19_func_8017BF14_b1.c` — round-1 draft, 45/4763 (kept).
|
||||
* `.run/giants/s19_func_8017BF14_b1_pinfree.c` — round-1 pin-free, 789/4763 (kept).
|
||||
* `.run/giants/bf14_mk2.py` — round-2 lever generator. Same contract as
|
||||
`bf14_mk.py`: **every transformation asserts it applied**, so a "neutral"
|
||||
reading can never be a silent no-op. Levers: `unpin_*`, `RC:` (re-pin),
|
||||
`sumvar_*`, `cb_expr*`, `cfn*`, `qsingle*`, `qdirect`, `tdirect2`, `prod*`,
|
||||
`merge_*`, `CS:` (CLAMP arg swap), `AW:` (ATTEN3 y-else destination),
|
||||
`CL:` (zero-byte `c1` liveness), `hivar`.
|
||||
* `.run/giants/bf14_sw2.sh` — 8-way parallel sweep over `bf14_mk2.py`, reporting
|
||||
the raw mismatch count.
|
||||
* `.run/giants/bf14_hdr.py` + `bf14_hdr.txt` — dossier-header splicer.
|
||||
* `.run/giants/r2/` — the promoted bases: `b2base` 37, `b3base` 33, `b4base` 21,
|
||||
`b5base` 3, `b6base` 2, `b7base` **0**; `sch_*` the attribution-primitive
|
||||
builds; `da/` the `-da` RTL dumps of the 45-base.
|
||||
* Round-1 harness (`bf14_cc/score/probe/full/side/win/hist/ali/slots/…`) used
|
||||
unchanged.
|
||||
|
||||
---
|
||||
|
||||
## 8. STATUS FOR THE COORDINATOR
|
||||
|
||||
`match_one` reports **MATCH (4763 ins)**. That is the candidate gate only.
|
||||
**The whole-binary SHA1 byte-gate (G3/P9) has not been run and is the sole
|
||||
arbiter** — this is not a confirmed match until that is green.
|
||||
@@ -0,0 +1,839 @@
|
||||
#include "common.h"
|
||||
#include "/home/musashi/bfm-decomp/src/shared/engine_types.h"
|
||||
|
||||
/* ===========================================================================
|
||||
* func_8017BF14 -- 4,763 ins, ov_SC03_116 (behemoth #4). *** MATCH ***
|
||||
*
|
||||
* STATUS (2026-07-25, session 20 round 2, gcc-2.7.2 pinned triple):
|
||||
* python3 tools/match_one.py func_8017BF14 --c <this file> \
|
||||
* --asm-subdir asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C
|
||||
* -> MATCH (4763 ins) func_8017BF14 [reproduced 3x, private work dirs]
|
||||
* LENGTH EXACT | OPCODE HISTOGRAM EXACT (L1 = 0) | STACK FRAME EXACT
|
||||
* (all 127 slots at the target's offsets, frame 0x360)
|
||||
* masked index-wise diff: 0 / 4763 mismatched.
|
||||
* `match_one` is the CANDIDATE gate. The whole-binary SHA1 arbiter (G3/P9)
|
||||
* is run by the coordinator and is the only thing that makes this final.
|
||||
*
|
||||
* Round 1 closed at 45/4763 mismatched. Round 2 took 45 -> 37 -> 33 -> 21
|
||||
* -> 11 -> 3 -> 2 -> 0. See .run/giants/s19_bf14_report2.md.
|
||||
*
|
||||
* WHAT IT IS
|
||||
* The *four*-light-box variant of the volumetric-light renderer whose
|
||||
* 3-box sibling func_8017D960 (3,338 ins, ov_SC03_090) is MATCHED, and whose
|
||||
* unlit ancestor func_8017BEBC (ov_SC03_099) is MATCHED. Same family, same
|
||||
* skeleton; this is the biggest member.
|
||||
*
|
||||
* Signature: func_8017BF14(s32 arg0, s32 lim). Unlike every other member of
|
||||
* the family this one is a LEAF -- 0 callees. The 3-call prologue
|
||||
* (func_800491EC / func_800547D8 / func_80052E38) of the siblings is gone;
|
||||
* `lim` arrives as arg1 (spilled to 0xB0). That is why the frame has no
|
||||
* 0x10 argument area (tmpxy[] starts at sp+0x00) and no $ra save.
|
||||
*
|
||||
* Per part (stride 0x14, outer loop): build the 8-corner AABB in box[],
|
||||
* rtpt/rtpt + rtps/rtps -> sxy[8], stszotz -> g.otz, reject on
|
||||
* `lim >= g.otz`, then screen-space bbox reject on X (-0xA0..0xA1) and
|
||||
* Y (-0x6E..0x6F).
|
||||
* Per prim (stride 0xC, inner loop): rtpt the 3 vertices, stflg mask
|
||||
* 0x7F85E000, nclip, stopz > 0, then a 4-way range tree on `code = w & 7`
|
||||
* that keeps ONLY codes 6,7 (tri) and 2,3 (quad); 0,1,4,5 fall through to
|
||||
* the loop tail.
|
||||
* Per drawn poly: screen bbox reject, then each vertex is tested against
|
||||
* FOUR axis-aligned light boxes (flags f0..f3), and if any is lit a
|
||||
* 0x00..0x80 attenuation per active box is computed, summed, biased +0x10
|
||||
* and clamped to 0x80 -> a grey gouraud vertex colour.
|
||||
* lit -> POLY_GT3 (0x28, tag 0x34000000, OT 0x9000000)
|
||||
* POLY_GT4 (0x34, tag 0x3C000000, OT 0xC000000)
|
||||
* unlit -> POLY_FT3 (0x20, OT 0x7000000) / POLY_FT4 (0x28, OT 0x9000000)
|
||||
* with rgbc = (tp[0] & 0xFF000000) | 0x101010 <-- NOT black,
|
||||
* unlike func_8017D960 where the unlit colour is plain black.
|
||||
*
|
||||
* THE FOUR LIGHT BOXES (stride 0x1C, {s32 enable; u16 cx,cy,cz; s32 range})
|
||||
* D_80197C28 / D_80197C44 / D_80197C60 / D_80197C7C.
|
||||
* Falloff geometry differs from the 3-box sibling: RLO = R - 0x200 (not
|
||||
* -0x80) and the ramp is ((R - d) / 4) (not (R - d)), so the 0x80 ceiling is
|
||||
* reached over a 0x200-wide band instead of 0x80. The `/ 4` is a SIGNED
|
||||
* divide -- `bgez / addiu 3 / sra 2` -- not a shift.
|
||||
*
|
||||
* FIVE ORIGINAL-SOURCE COPY-PASTE ARTEFACTS, all byte-proven
|
||||
* The 4th light box was bolted onto a copy of the 3-box source BY HAND and
|
||||
* the hand edit was incomplete in five places. Each was read off the target
|
||||
* and each removed a measured delta.
|
||||
* (1) `r3lo = r2 - 0x200;` -- box 3's low radius is derived from box 2's
|
||||
* RANGE VARIABLE, not from its own D_80197C88. Proven by the target's
|
||||
* `addiu $t6, $s0, -0x200` reusing the register that box 2's `lw` filled;
|
||||
* spelling it `D_80197C6C - 0x200` re-loads the global (+2 ins).
|
||||
* (2) Only SIX of the eight radius variables are zero-initialised
|
||||
* (r0,r1,r2,r0lo,r1lo,r2lo) -- r3/r3lo are left uninitialised, exactly
|
||||
* the init list the 3-box version needed.
|
||||
* (3) The ATTEN body is written out LONGHAND 7 times (3 tri vertices +
|
||||
* 4 quad vertices). When box 3 was bolted on, the `R` of the z- and
|
||||
* y-axis KILL tests was left as r2 in three of those copies:
|
||||
* tri v0: z and y use r2 tri v2: z uses r2 all others use r3.
|
||||
* Proven by the 24 `sll $v0,$s0,16` sites: 3 per group in box 2 plus
|
||||
* exactly three extra at idx 1471, 1504 (tri v0) and 2178 (tri v2).
|
||||
* (4) In the QUAD lit arm only, rgb2 and rgb3 take their `<< 16` term from
|
||||
* c1, not from c2/c3. Proven by the target CSE-ing ONE
|
||||
* `sll $a0, $a0, 16` and re-using $a0 for all three stores.
|
||||
* (5) *** ROUND 2 *** In the QUAD arm's VERTEX-1 group only, the box-3
|
||||
* ATTEN's Y-axis `else if` branch accumulates into a2v instead of a3v,
|
||||
* while that same test's KILL branch still says a3v. Modelled by the
|
||||
* ATTEN3W macro below (`AW` = the y-else destination). Byte-proof:
|
||||
* idx 3875 addu $a2,$zero,$zero kill branch -> a3v ($a2) [agreed]
|
||||
* idx 3889 sra $a3,$s2,7 else branch -> a2v ($a3) [was the
|
||||
* last structural residual]
|
||||
* Costs zero instructions; the four other quad/tri sites are NOT like
|
||||
* this (each was measured -- putting the artefact anywhere else is +2).
|
||||
*
|
||||
* FRAME (0x360, leaf -- no $ra, no argument area)
|
||||
* 0x000 tmpxy[4] | 0x010 box[8] | 0x050 sxy[8] |
|
||||
* 0x090 g{otz,flag,opz,sz0..sz3} | 0x0B0 lim | 0x0B8 j | 0x0C0 i |
|
||||
* 0x0C8 vd | 0x0D0 ot | 0x0D8 pkt | 0x0E0 f2 | 0x0E8 f3 |
|
||||
* 0x0F0/0x0F8/0x100 x3,z3,y3 | 0x108 prim | 0x110 nprim | 0x118 vtx |
|
||||
* 0x120 nparts | 0x128 part | 0x130..0x1D0 lo/hi bounds (21 s16 slots) |
|
||||
* 0x1D8..0x230 cx0..cz3 (12) | 0x238 r0 | 0x240 r1lo | 0x248 r2lo |
|
||||
* 0x250 r3lo | 0x288..0x2D0 the LICM-hoisted sign-extended bounds |
|
||||
* 0x328/0x330 spilled vertex coords | 0x338..0x358 s0-s7,fp.
|
||||
* *** THE SLOT ORDER IS THE DECLARATION-ORDER ORACLE (see L5). ***
|
||||
*
|
||||
* ---------------------------------------------------------------------------
|
||||
* ROUND-1 LEVERS (kept; measured effect is byte-identical %, anchored)
|
||||
*
|
||||
* L1 `cb = (tp[0] & 0xFF000000) | 0x101010;` in both UNLIT arms.
|
||||
* L2 box-3 ATTEN kill-register per copy (artefact 3) + `r3lo = r2 - 0x200`
|
||||
* (artefact 1). 89.15% -> 96.96% shape; killed sra+11 / sll+9.
|
||||
* L3 quad-lit rgb2/rgb3 use `c1 << 16` (artefact 4). Killed the last sll+2.
|
||||
* L4 [SUPERSEDED BY R3] `s32 c0..c3` declared inside the two CULL blocks.
|
||||
* L5 `s32 f0, f1, f2, f3;` MOVED TO IMMEDIATELY AFTER `u8 *pkt;`.
|
||||
* Spilled pseudos get stack slots in PSEUDO-NUMBER order and pseudo
|
||||
* numbers are handed out in DECLARATION order, so the target's stack
|
||||
* layout is a direct read-out of its declaration order. After this one
|
||||
* move ALL 127 stack slots agree with the target exactly.
|
||||
* L6 `u32 rgbw;` per emit arm.
|
||||
* L7 RC-15 zero-byte ref dial on `mny`, first statement of the TRI cull
|
||||
* block -- flips my->$a3 / mny->$a2 to the target's grant.
|
||||
* L8 `cb` 2-statement accumulator [SUPERSEDED BY R2]; quad-lit rgb word as a
|
||||
* 3-statement accumulator [kept for rgb0/rgb1, SUPERSEDED for rgb2/rgb3
|
||||
* by R5].
|
||||
* L9 FOUR REGISTER PINS: va->$t2, w->$a1, f0->$s3, c1->$a0.
|
||||
* Round 2 removed va and w (see R4); f0 and c1 REMAIN and are both
|
||||
* load-bearing. This function has NO `jal`, so Sec.74's caller-saved
|
||||
* pin-spanning-a-call hazard cannot arise -- that is why pins are usable
|
||||
* on this family member and were a trap on the others.
|
||||
*
|
||||
* ---------------------------------------------------------------------------
|
||||
* ROUND-2 LEVERS -- 45 -> 0. Metric is `match_one` MISMATCH COUNT (length is
|
||||
* exact throughout, so the raw count is honest). Every number is measured.
|
||||
*
|
||||
* THE ONE MECHANISM BEHIND R1/R2/R4/R7. A `register __asm__` pin makes the
|
||||
* variable a HARD REG in the RTL from the start. When a 1-death local temp is
|
||||
* produced from, or consumed into, that hard reg, local-alloc.c's
|
||||
* `combine_regs` takes its hard-register branch (local-alloc.c:1795-1820) and
|
||||
* records the pinned register in `qty_phys_sugg` for the temp's quantity --
|
||||
* UNCONDITIONALLY, there is no death guard on that path. The temp then lands
|
||||
* in the pinned register and the operation is done IN PLACE. The target,
|
||||
* whose variable is an ordinary pseudo (reg_qty == -1 for anything crossing a
|
||||
* block), never gets that suggestion and keeps the temp in $v0/$v1.
|
||||
* Three independent cures, all used here:
|
||||
* (i) give the temp a NAMED variable with >1 death, so local-alloc.c:472
|
||||
* (`reg_basic_block >= 0 && reg_n_deaths == 1`) refuses it a quantity
|
||||
* and combine_regs bails at its very first test -> R1
|
||||
* (ii) drop the pin, if the pin is not load-bearing -> R4
|
||||
* (iii) keep the pinned value LIVE past the temp, so that
|
||||
* find_free_reg cannot honour the suggestion -> R7
|
||||
*
|
||||
* R1 `CLAMP80S` -- the four-way attenuation sum gets its own named variable
|
||||
* `sv`, used at BOTH c1 sites (2 deaths). Cure (i). 45 -> 37
|
||||
* Target: `addu $v0,..; addu $v0,..; addu $v0,..; addiu <c>,$v0,0x10`.
|
||||
* Without it the whole chain is tied into the pinned c1 ($a0).
|
||||
* Only the two c1 sites: `sv` on all 7 sites collapses the frame (-64).
|
||||
* R2 UNLIT arms store `cb | 0x101010` as an expression instead of doing
|
||||
* `cb |= 0x101010` in place. `cb` is a function-scope global allocno, so
|
||||
* the in-place form writes $a1; the expression form is a 1-death local
|
||||
* that combine_regs ties to the DYING constant register $v1, which is
|
||||
* what the target does. 37 -> 33
|
||||
* R3 *** `s32 c0, c1, c2, c3;` AT FUNCTION SCOPE, not per cull block. ***
|
||||
* Read straight off the two MATCHED relatives (func_8017D960 line 310,
|
||||
* func_8017F510 line 338), and confirmed by the target itself: its TRI
|
||||
* grants (c0=$t4, c1=$a0, c2=$t2) are IDENTICAL to its QUAD grants, which
|
||||
* is only possible if both arms share one set of allocnos. Per-cull-block
|
||||
* scope splits them into two independent allocno sets and the TRI set
|
||||
* drifts. 33 -> 21
|
||||
* Declaration POSITION is neutral (5 anchors swept, all 21).
|
||||
* NOTE this REVERSES round-1's L4. L4 was correct on the round-1 base --
|
||||
* it was supplying the extra local allocno that spills r1lo -- but R1+R2
|
||||
* supply that pressure now, and the c1 pin does the rest.
|
||||
* R4 DROP the `va->$t2` and `w->$a1` pins. With c0..c3 at function scope
|
||||
* they are no longer load-bearing, and they were the sole cause of the
|
||||
* prim-word producer ties (`andi`/`srl` written straight into $t2/$a1).
|
||||
* Cure (ii). 21 -> 17 -> 13 (pair)
|
||||
* Round 1 measured these as worth 4%; that was true of the round-1 base
|
||||
* and is FALSE here. Base-dependence, not a contradiction.
|
||||
* R5 QUAD lit rgb2/rgb3 revert to the SINGLE-EXPRESSION form (rgb0/rgb1 keep
|
||||
* the 3-statement accumulator). The intermediates then become 1-death
|
||||
* local temps that alternate $v0/$v1, which is what lets the target's
|
||||
* store of the previous rgb word sit one slot LATER. 21 -> 11
|
||||
* Doing it to all four, or to rgb0/rgb1 only, is worse (23 / 21).
|
||||
* R6 Artefact 5 -- `ATTEN3W(a3v, a2v, ...)` at the QUAD vertex-1 site. 3 -> 2
|
||||
* R7 RC-15 zero-byte ref `__asm__ __volatile__ ("" :: "r" (c1));` placed
|
||||
* immediately after the TRI arm's rgb1 store. Cure (iii): it keeps the
|
||||
* pinned c1 ($a0) live past `c1 << 16`, so find_free_reg cannot honour
|
||||
* combine_regs' $a0 suggestion and the shift goes to $v0. 2 -> 0
|
||||
* The c1 pin CANNOT simply be removed: it is what spills r1lo (dropping
|
||||
* it, or moving it to any other colour, costs -64 length). Measured.
|
||||
*
|
||||
* SCHEDULER ATTRIBUTION (Sec.76 primitive, run in round 2, never run before on
|
||||
* this function). Compiled with -fno-schedule-insns, with
|
||||
* -fno-schedule-insns2, and with both. The store/shift transposition at
|
||||
* idx 4643-4652 kept MY source order under all three. It was therefore never
|
||||
* a `sched.c` decision: the ordering is a CONSEQUENCE of the register grant
|
||||
* (a 3-statement accumulator pins the value in one register, so no scheduler
|
||||
* could hoist the next `or` above the `sw`). R5 fixed it by changing the
|
||||
* grant, exactly as Sec.78 predicts.
|
||||
* =========================================================================== */
|
||||
|
||||
#define gte_ldv0(r0) __asm__ volatile ( \
|
||||
"lwc2 $0, 0( %0 );" \
|
||||
"lwc2 $1, 4( %0 )" \
|
||||
: \
|
||||
: "r"( r0 ) )
|
||||
|
||||
#define gte_ldv3(r0, r1, r2) __asm__ volatile ( \
|
||||
"lwc2 $0, 0( %0 );" \
|
||||
"lwc2 $1, 4( %0 );" \
|
||||
"lwc2 $2, 0( %1 );" \
|
||||
"lwc2 $3, 4( %1 );" \
|
||||
"lwc2 $4, 0( %2 );" \
|
||||
"lwc2 $5, 4( %2 )" \
|
||||
: \
|
||||
: "r"( r0 ), "r"( r1 ), "r"( r2 ) )
|
||||
|
||||
#define gte_ldv3c(r0) __asm__ volatile ( \
|
||||
"lwc2 $0, 0( %0 );" \
|
||||
"lwc2 $1, 4( %0 );" \
|
||||
"lwc2 $2, 8( %0 );" \
|
||||
"lwc2 $3, 12( %0 );" \
|
||||
"lwc2 $4, 16( %0 );" \
|
||||
"lwc2 $5, 20( %0 )" \
|
||||
: \
|
||||
: "r"( r0 ) )
|
||||
|
||||
#define gte_rtps() __asm__ volatile ("nop;nop;rtps")
|
||||
#define gte_rtpt() __asm__ volatile ("nop;nop;rtpt")
|
||||
#define gte_nclip() __asm__ volatile ("nop;nop;nclip")
|
||||
|
||||
#define gte_stsxy(r0) __asm__ volatile ( \
|
||||
"swc2 $14, 0( %0 )" \
|
||||
: \
|
||||
: "r"( r0 ) \
|
||||
: "memory" )
|
||||
|
||||
#define gte_stsxy3(r0, r1, r2) __asm__ volatile ( \
|
||||
"swc2 $12, 0( %0 );" \
|
||||
"swc2 $13, 0( %1 );" \
|
||||
"swc2 $14, 0( %2 )" \
|
||||
: \
|
||||
: "r"( r0 ), "r"( r1 ), "r"( r2 ) \
|
||||
: "memory" )
|
||||
|
||||
#define gte_stsxy3c(r0) __asm__ volatile ( \
|
||||
"swc2 $12, 0( %0 );" \
|
||||
"swc2 $13, 4( %0 );" \
|
||||
"swc2 $14, 8( %0 )" \
|
||||
: \
|
||||
: "r"( r0 ) \
|
||||
: "memory" )
|
||||
|
||||
#define gte_stsz3(r0, r1, r2) __asm__ volatile ( \
|
||||
"swc2 $17, 0( %0 );" \
|
||||
"swc2 $18, 0( %1 );" \
|
||||
"swc2 $19, 0( %2 )" \
|
||||
: \
|
||||
: "r"( r0 ), "r"( r1 ), "r"( r2 ) \
|
||||
: "memory" )
|
||||
|
||||
#define gte_stsz4(r0, r1, r2, r3) __asm__ volatile ( \
|
||||
"swc2 $16, 0( %0 );" \
|
||||
"swc2 $17, 0( %1 );" \
|
||||
"swc2 $18, 0( %2 );" \
|
||||
"swc2 $19, 0( %3 )" \
|
||||
: \
|
||||
: "r"( r0 ), "r"( r1 ), "r"( r2 ), "r"( r3 ) \
|
||||
: "memory" )
|
||||
|
||||
#define gte_stszotz(r0) __asm__ volatile ( \
|
||||
"mfc2 $12, $19;" \
|
||||
"nop;" \
|
||||
"sra $12, $12, 2;" \
|
||||
"sw $12, 0( %0 )" \
|
||||
: \
|
||||
: "r"( r0 ) \
|
||||
: "$12", "memory" )
|
||||
|
||||
#define gte_stflg(r0) __asm__ volatile ( \
|
||||
"cfc2 $12, $31;" \
|
||||
"nop;" \
|
||||
"sw $12, 0( %0 )" \
|
||||
: \
|
||||
: "r"( r0 ) \
|
||||
: "$12", "memory" )
|
||||
|
||||
#define gte_stopz(r0) __asm__ volatile ( \
|
||||
"swc2 $24, 0( %0 )" \
|
||||
: \
|
||||
: "r"( r0 ) \
|
||||
: "memory" )
|
||||
|
||||
/* ---- the two gouraud-textured packet layouts this function emits ---------- */
|
||||
typedef struct {
|
||||
u32 tag;
|
||||
u32 rgb0; s16 x0, y0; u32 uv0;
|
||||
u32 rgb1; s16 x1, y1; u32 uv1;
|
||||
u32 rgb2; s16 x2, y2; u16 uv2, p2;
|
||||
} PolyGT3; /* 0x28 */
|
||||
|
||||
typedef struct {
|
||||
u32 tag;
|
||||
u32 rgb0; s16 x0, y0; u32 uv0;
|
||||
u32 rgb1; s16 x1, y1; u32 uv1;
|
||||
u32 rgb2; s16 x2, y2; u16 uv2, p2;
|
||||
u32 rgb3; s16 x3, y3; u16 uv3, p3;
|
||||
} PolyGT4; /* 0x34 */
|
||||
|
||||
|
||||
/* ---- the four light-volume descriptors (stride 0x1C) --------------------- */
|
||||
extern s32 D_80197C28;
|
||||
extern u16 D_80197C2C, D_80197C2E, D_80197C30;
|
||||
extern s32 D_80197C34;
|
||||
extern s32 D_80197C44;
|
||||
extern u16 D_80197C48, D_80197C4A, D_80197C4C;
|
||||
extern s32 D_80197C50;
|
||||
extern s32 D_80197C60;
|
||||
extern u16 D_80197C64, D_80197C66, D_80197C68;
|
||||
extern s32 D_80197C6C;
|
||||
extern s32 D_80197C7C;
|
||||
extern u16 D_80197C80, D_80197C82, D_80197C84;
|
||||
extern u16 D_80197C88;
|
||||
|
||||
/* ---- the box-containment test for one vertex against one light box ------- */
|
||||
#define BOXTEST(F, X, Y, Z, LX, HX, LY, HY, LZ, HZ) \
|
||||
if ((LX) < (X) && (X) < (HX) && (LY) < (Y) && (Y) < (HY) && (LZ) < (Z) && (Z) < (HZ)) F = 1
|
||||
|
||||
/* ---- the separable per-axis falloff, visited in x, z, y order ------------ */
|
||||
#define ATTEN(A, F, X, Y, Z, CX, CY, CZ, R, RLO) \
|
||||
A = 0; \
|
||||
if (F) { \
|
||||
d = (X) - (CX); if (d < 0) d = (CX) - (X); \
|
||||
if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \
|
||||
d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \
|
||||
if ((R) < d) A = 0; \
|
||||
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
|
||||
d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \
|
||||
if ((R) < d) A = 0; \
|
||||
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
|
||||
}
|
||||
|
||||
#define ATTEN3W(A, AW, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \
|
||||
A = 0; \
|
||||
if (F) { \
|
||||
d = (X) - (CX); if (d < 0) d = (CX) - (X); \
|
||||
if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \
|
||||
d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \
|
||||
if ((RZ) < d) A = 0; \
|
||||
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
|
||||
d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \
|
||||
if ((RY) < d) A = 0; \
|
||||
else if ((RLO) < d) AW = (A * (((R) - d) / 4)) >> 7; \
|
||||
}
|
||||
|
||||
#define ATTEN3(A, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \
|
||||
A = 0; \
|
||||
if (F) { \
|
||||
d = (X) - (CX); if (d < 0) d = (CX) - (X); \
|
||||
if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \
|
||||
d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \
|
||||
if ((RZ) < d) A = 0; \
|
||||
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
|
||||
d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \
|
||||
if ((RY) < d) A = 0; \
|
||||
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
|
||||
}
|
||||
|
||||
#define CLAMP80(C, A0, A1, A2, A3) C = (A0) + (A1) + (A2) + (A3) + 0x10; if ((C) > 0x80) C = 0x80
|
||||
#define CLAMP80S(C, A0, A1, A2, A3) sv = (A0) + (A1) + (A2) + (A3); C = sv + 0x10; if ((C) > 0x80) C = 0x80
|
||||
|
||||
|
||||
void func_8017BF14(s32 arg0, s32 lim)
|
||||
{
|
||||
typedef struct { u32 w0, w1, w2; } Prim;
|
||||
|
||||
extern u8 *D_800A5E60;
|
||||
extern u8 D_800A6610[];
|
||||
extern u8 D_800AF630[];
|
||||
|
||||
DVECTOR2 tmpxy[4];
|
||||
SVECTOR2 box[8];
|
||||
SVECTOR2 sxy[8];
|
||||
struct { long otz, flag, opz, sz0, sz1, sz2, sz3; } g;
|
||||
|
||||
s32 j;
|
||||
u32 i;
|
||||
u8 *vd;
|
||||
u32 ot;
|
||||
u8 *pkt;
|
||||
register s32 f0 __asm__("$19");
|
||||
s32 f1, f2, f3;
|
||||
s16 x3, z3, y3;
|
||||
Prim *prim;
|
||||
u32 nprim;
|
||||
u8 *vtx;
|
||||
s32 nparts;
|
||||
Part *part;
|
||||
s16 lo0x, hi0x, lo0y, hi0y, lo0z, hi0z;
|
||||
s16 lo1x, hi1x, lo1y, hi1y, lo1z, hi1z;
|
||||
s16 lo2x, hi2x, lo2y, hi2y, lo2z, hi2z;
|
||||
s16 lo3x, hi3x, lo3y, hi3y, lo3z, hi3z;
|
||||
s16 cx0, cy0, cz0, cx1, cy1, cz1, cx2, cy2, cz2, cx3, cy3, cz3;
|
||||
s16 r0;
|
||||
s16 r1;
|
||||
s16 r2;
|
||||
s16 r3;
|
||||
s16 r0lo;
|
||||
s16 r1lo;
|
||||
s16 r2lo;
|
||||
s16 r3lo;
|
||||
u8 *va, *vb, *vc;
|
||||
u32 w;
|
||||
s32 code;
|
||||
u32 vw, vzw;
|
||||
u32 wx, wy, wz;
|
||||
s32 xa32, xb32, t32;
|
||||
s32 xmn1, xmx1, xmn2, xmx2;
|
||||
s32 mnc, mxc;
|
||||
s16 my, mny, mx, mn;
|
||||
u8 *base;
|
||||
s16 x0, y0, z0, x1, y1, z1, x2, y2, z2;
|
||||
s32 a0v, a1v, a2v, a3v;
|
||||
register s32 c1 __asm__("$4"); s32 c0, c2, c3;
|
||||
s32 d;
|
||||
u32 *tp;
|
||||
u32 uvw;
|
||||
u32 cb;
|
||||
s32 sv;
|
||||
|
||||
base = D_800AF630;
|
||||
|
||||
r2lo = 0;
|
||||
r1lo = 0;
|
||||
r0lo = 0;
|
||||
r2 = 0;
|
||||
r1 = 0;
|
||||
r0 = 0;
|
||||
|
||||
if (D_80197C28) {
|
||||
cx0 = D_80197C2C;
|
||||
cy0 = D_80197C2E;
|
||||
r0lo = D_80197C34 - 0x200;
|
||||
r0 = D_80197C34;
|
||||
cz0 = D_80197C30;
|
||||
} else {
|
||||
cz0 = 0x6000;
|
||||
cy0 = 0x6000;
|
||||
cx0 = 0x6000;
|
||||
}
|
||||
if (D_80197C44) {
|
||||
cx1 = D_80197C48;
|
||||
cy1 = D_80197C4A;
|
||||
r1 = D_80197C50;
|
||||
r1lo = D_80197C50 - 0x200;
|
||||
cz1 = D_80197C4C;
|
||||
} else {
|
||||
cz1 = 0x6000;
|
||||
cy1 = 0x6000;
|
||||
cx1 = 0x6000;
|
||||
}
|
||||
if (D_80197C60) {
|
||||
cx2 = D_80197C64;
|
||||
cy2 = D_80197C66;
|
||||
r2 = D_80197C6C;
|
||||
r2lo = D_80197C6C - 0x200;
|
||||
cz2 = D_80197C68;
|
||||
} else {
|
||||
cz2 = 0x6000;
|
||||
cy2 = 0x6000;
|
||||
cx2 = 0x6000;
|
||||
}
|
||||
if (D_80197C7C) {
|
||||
cx3 = D_80197C80;
|
||||
cy3 = D_80197C82;
|
||||
r3lo = r2 - 0x200;
|
||||
r3 = D_80197C88;
|
||||
cz3 = D_80197C84;
|
||||
} else {
|
||||
cz3 = 0x6000;
|
||||
cy3 = 0x6000;
|
||||
cx3 = 0x6000;
|
||||
}
|
||||
|
||||
lo0x = cx0 - r0; hi0x = cx0 + r0;
|
||||
lo0y = cy0 - r0; hi0y = cy0 + r0;
|
||||
lo0z = cz0 - r0; hi0z = cz0 + r0;
|
||||
lo1x = cx1 - r1; hi1x = cx1 + r1;
|
||||
lo1y = cy1 - r1; hi1y = cy1 + r1;
|
||||
lo1z = cz1 - r1; hi1z = cz1 + r1;
|
||||
lo2x = cx2 - r2; hi2x = cx2 + r2;
|
||||
lo2y = cy2 - r2; hi2y = cy2 + r2;
|
||||
lo2z = cz2 - r2; hi2z = cz2 + r2;
|
||||
lo3x = cx3 - r3; hi3x = cx3 + r3;
|
||||
lo3y = cy3 - r3; hi3y = cy3 + r3;
|
||||
lo3z = cz3 - r3; hi3z = cz3 + r3;
|
||||
|
||||
pkt = D_800A5E60;
|
||||
part = *(Part **)(arg0 + 0xC);
|
||||
nparts = *(s32 *)(*(s32 *)(arg0 + 8) + 8);
|
||||
vtx = *(u8 **)(*(s32 *)(arg0 + 8) + 0x10);
|
||||
ot = (u32)&D_800A6610[(*(u16 *)(base + 0xA3D2)) << 14];
|
||||
|
||||
for (j = 0; j < nparts; j++, part++) {
|
||||
wx = part->xx;
|
||||
mn = wx;
|
||||
mx = wx >> 16;
|
||||
wy = part->yy;
|
||||
mny = wy;
|
||||
my = wy >> 16;
|
||||
wz = part->zz;
|
||||
box[0].vx = mn; box[0].vy = mny;
|
||||
box[1].vx = mx; box[1].vy = mny;
|
||||
box[2].vx = mn; box[2].vy = mny;
|
||||
box[3].vx = mx; box[3].vy = mny;
|
||||
box[4].vx = mn; box[4].vy = my;
|
||||
box[5].vx = mx; box[5].vy = my;
|
||||
box[6].vx = mn; box[6].vy = my;
|
||||
box[7].vx = mx; box[7].vy = my;
|
||||
wy = wz >> 16;
|
||||
box[0].vz = wz;
|
||||
box[1].vz = wz;
|
||||
box[4].vz = wz;
|
||||
box[5].vz = wz;
|
||||
box[2].vz = wy;
|
||||
box[3].vz = wy;
|
||||
box[6].vz = wy;
|
||||
box[7].vz = wy;
|
||||
|
||||
gte_ldv3c(&box[0]);
|
||||
gte_rtpt();
|
||||
gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]);
|
||||
gte_ldv0(&box[3]);
|
||||
gte_rtps();
|
||||
gte_stsxy(&sxy[3]);
|
||||
gte_ldv3c(&box[4]);
|
||||
gte_rtpt();
|
||||
gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]);
|
||||
gte_ldv0(&box[7]);
|
||||
gte_rtps();
|
||||
gte_stsxy(&sxy[7]);
|
||||
gte_stszotz(&g.otz);
|
||||
|
||||
if (lim >= g.otz) {
|
||||
xa32 = sxy[0].vx;
|
||||
xb32 = sxy[1].vx;
|
||||
if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; }
|
||||
t32 = sxy[2].vx;
|
||||
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
|
||||
t32 = sxy[3].vx;
|
||||
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
|
||||
xa32 = sxy[4].vx;
|
||||
xb32 = sxy[5].vx;
|
||||
if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; }
|
||||
t32 = sxy[6].vx;
|
||||
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
|
||||
t32 = sxy[7].vx;
|
||||
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
|
||||
mnc = xmn1;
|
||||
if (xmn2 < xmn1) mnc = xmn2;
|
||||
mxc = xmx1;
|
||||
if (mxc < xmx2) mxc = xmx2;
|
||||
if ((s16)mxc >= -0xA0 && (s16)mnc < 0xA1) {
|
||||
xa32 = sxy[0].vy;
|
||||
xb32 = sxy[1].vy;
|
||||
if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; }
|
||||
t32 = sxy[2].vy;
|
||||
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
|
||||
t32 = sxy[3].vy;
|
||||
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
|
||||
xa32 = sxy[4].vy;
|
||||
xb32 = sxy[5].vy;
|
||||
if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; }
|
||||
t32 = sxy[6].vy;
|
||||
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
|
||||
t32 = sxy[7].vy;
|
||||
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
|
||||
mnc = xmn1;
|
||||
if (xmn2 < xmn1) mnc = xmn2;
|
||||
mxc = xmx1;
|
||||
if (mxc < xmx2) mxc = xmx2;
|
||||
if ((s16)mxc >= -0x6E && (s16)mnc < 0x6F) {
|
||||
nprim = part->nprim;
|
||||
prim = (Prim *)part->prim;
|
||||
for (i = 0; i < nprim; i++, prim++) {
|
||||
w = prim->w1;
|
||||
va = vtx + (w & 0xFFFF);
|
||||
vb = vtx + (w >> 16);
|
||||
w = prim->w2;
|
||||
vc = vtx + (w & 0xFFFF);
|
||||
w = w >> 16;
|
||||
gte_ldv3(va, vb, vc);
|
||||
gte_rtpt();
|
||||
gte_stflg(&g.flag);
|
||||
if (!(g.flag & 0x7F85E000)) {
|
||||
gte_nclip();
|
||||
code = w & 7;
|
||||
vd = vtx + (w & 0xFFF8);
|
||||
gte_stopz(&g.opz);
|
||||
if (g.opz > 0) {
|
||||
switch (code) {
|
||||
case 6:
|
||||
case 7:
|
||||
/* ---------------- TRI (FT3 / GT3) ---------------- */
|
||||
gte_stsxy3c(&tmpxy[0]);
|
||||
gte_stsz3(&g.sz0, &g.sz1, &g.sz2);
|
||||
if (tmpxy[0].vx > tmpxy[1].vx) { mx = tmpxy[0].vx; mn = tmpxy[1].vx; }
|
||||
else { mn = tmpxy[0].vx; mx = tmpxy[1].vx; }
|
||||
if (tmpxy[2].vx > mx) mx = tmpxy[2].vx;
|
||||
else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx;
|
||||
if (mx >= -0xA0 && mn < 0xA1) {
|
||||
if (tmpxy[0].vy > tmpxy[1].vy) { my = tmpxy[0].vy; mny = tmpxy[1].vy; }
|
||||
else { mny = tmpxy[0].vy; my = tmpxy[1].vy; }
|
||||
if (tmpxy[2].vy > my) my = tmpxy[2].vy;
|
||||
else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy;
|
||||
if (my >= -0x6E && mny < 0x6F) {
|
||||
s32 za, zb;
|
||||
__asm__ __volatile__ ("" :: "r" (mny));
|
||||
if (g.sz0 > g.sz1) { za = g.sz0; if (za < g.sz2) za = g.sz2; }
|
||||
else { za = g.sz1; if (za < g.sz2) za = g.sz2; }
|
||||
g.opz = za;
|
||||
|
||||
f3 = 0; f2 = 0; f1 = 0; f0 = 0;
|
||||
|
||||
vw = *(u32 *)va;
|
||||
vzw = *(u32 *)(va + 4);
|
||||
x0 = vw; y0 = vw >> 16; z0 = vzw;
|
||||
BOXTEST(f0, x0, y0, z0, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
|
||||
BOXTEST(f1, x0, y0, z0, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
|
||||
BOXTEST(f2, x0, y0, z0, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
|
||||
BOXTEST(f3, x0, y0, z0, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
|
||||
vw = *(u32 *)vb;
|
||||
vzw = *(u32 *)(vb + 4);
|
||||
x1 = vw; y1 = vw >> 16; z1 = vzw;
|
||||
BOXTEST(f0, x1, y1, z1, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
|
||||
BOXTEST(f1, x1, y1, z1, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
|
||||
BOXTEST(f2, x1, y1, z1, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
|
||||
BOXTEST(f3, x1, y1, z1, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
|
||||
vw = *(u32 *)vc;
|
||||
vzw = *(u32 *)(vc + 4);
|
||||
x2 = vw; y2 = vw >> 16; z2 = vzw;
|
||||
BOXTEST(f0, x2, y2, z2, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
|
||||
BOXTEST(f1, x2, y2, z2, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
|
||||
BOXTEST(f2, x2, y2, z2, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
|
||||
BOXTEST(f3, x2, y2, z2, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
|
||||
|
||||
if (f0 | f1 | f2 | f3) {
|
||||
u32 *otp;
|
||||
u32 rgbw;
|
||||
ATTEN(a0v, f0, x0, y0, z0, cx0, cy0, cz0, r0, r0lo);
|
||||
ATTEN(a1v, f1, x0, y0, z0, cx1, cy1, cz1, r1, r1lo);
|
||||
ATTEN(a2v, f2, x0, y0, z0, cx2, cy2, cz2, r2, r2lo);
|
||||
ATTEN3(a3v, f3, x0, y0, z0, cx3, cy3, cz3, r3, r3lo, r2, r2);
|
||||
CLAMP80(c0, a0v, a1v, a2v, a3v);
|
||||
ATTEN(a0v, f0, x1, y1, z1, cx0, cy0, cz0, r0, r0lo);
|
||||
ATTEN(a1v, f1, x1, y1, z1, cx1, cy1, cz1, r1, r1lo);
|
||||
ATTEN(a2v, f2, x1, y1, z1, cx2, cy2, cz2, r2, r2lo);
|
||||
ATTEN3(a3v, f3, x1, y1, z1, cx3, cy3, cz3, r3, r3lo, r3, r3);
|
||||
CLAMP80S(c1, a0v, a1v, a2v, a3v);
|
||||
ATTEN(a0v, f0, x2, y2, z2, cx0, cy0, cz0, r0, r0lo);
|
||||
ATTEN(a1v, f1, x2, y2, z2, cx1, cy1, cz1, r1, r1lo);
|
||||
ATTEN(a2v, f2, x2, y2, z2, cx2, cy2, cz2, r2, r2lo);
|
||||
ATTEN3(a3v, f3, x2, y2, z2, cx3, cy3, cz3, r3, r3lo, r2, r3);
|
||||
CLAMP80(c2, a0v, a1v, a2v, a3v);
|
||||
|
||||
*(u32 *)&((PolyGT3 *)pkt)->x0 = *(u32 *)&tmpxy[0];
|
||||
*(u32 *)&((PolyGT3 *)pkt)->x1 = *(u32 *)&tmpxy[1];
|
||||
*(u32 *)&((PolyGT3 *)pkt)->x2 = *(u32 *)&tmpxy[2];
|
||||
tp = (u32 *)prim->w0;
|
||||
cb = 0x34000000;
|
||||
rgbw = (c0 | cb) | (c0 << 8) | (c0 << 16);
|
||||
((PolyGT3 *)pkt)->rgb0 = rgbw;
|
||||
rgbw = (c1 | cb) | (c1 << 8) | (c1 << 16);
|
||||
((PolyGT3 *)pkt)->rgb1 = rgbw;
|
||||
__asm__ __volatile__ ("" :: "r" (c1));
|
||||
rgbw = (c2 | cb) | (c2 << 8) | (c2 << 16);
|
||||
((PolyGT3 *)pkt)->rgb2 = rgbw;
|
||||
((PolyGT3 *)pkt)->uv0 = tp[1];
|
||||
((PolyGT3 *)pkt)->uv1 = tp[2];
|
||||
((PolyGT3 *)pkt)->uv2 = tp[3];
|
||||
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
|
||||
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000;
|
||||
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
|
||||
pkt += 0x28;
|
||||
} else {
|
||||
u32 *otp;
|
||||
u32 rgbw;
|
||||
*(u32 *)&((PolyFT3 *)pkt)->x0 = *(u32 *)&tmpxy[0];
|
||||
*(u32 *)&((PolyFT3 *)pkt)->x1 = *(u32 *)&tmpxy[1];
|
||||
*(u32 *)&((PolyFT3 *)pkt)->x2 = *(u32 *)&tmpxy[2];
|
||||
tp = (u32 *)prim->w0;
|
||||
cb = tp[0] & 0xFF000000;
|
||||
((PolyFT3 *)pkt)->rgbc = cb | 0x101010;
|
||||
((PolyFT3 *)pkt)->uvc0 = tp[1];
|
||||
((PolyFT3 *)pkt)->uvp1 = tp[2];
|
||||
((PolyFT3 *)pkt)->uv2 = tp[3];
|
||||
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
|
||||
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000;
|
||||
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
|
||||
pkt += 0x20;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
case 2:
|
||||
case 3:
|
||||
/* ---------------- QUAD (FT4 / GT4) ---------------- */
|
||||
gte_stsxy3c(&tmpxy[0]);
|
||||
gte_ldv0(vd);
|
||||
gte_rtps();
|
||||
if (tmpxy[0].vx > tmpxy[1].vx) { mx = tmpxy[0].vx; mn = tmpxy[1].vx; }
|
||||
else { mn = tmpxy[0].vx; mx = tmpxy[1].vx; }
|
||||
if (tmpxy[2].vx > mx) mx = tmpxy[2].vx;
|
||||
else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx;
|
||||
if (tmpxy[0].vy > tmpxy[1].vy) { my = tmpxy[0].vy; mny = tmpxy[1].vy; }
|
||||
else { mny = tmpxy[0].vy; my = tmpxy[1].vy; }
|
||||
if (tmpxy[2].vy > my) my = tmpxy[2].vy;
|
||||
else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy;
|
||||
gte_stflg(&g.flag);
|
||||
if (!(g.flag & 0x7F85E000)) {
|
||||
gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3);
|
||||
gte_stsxy((long *)&((PolyFT4 *)pkt)->x3);
|
||||
if (((PolyFT4 *)pkt)->x3 < mn) mn = ((PolyFT4 *)pkt)->x3;
|
||||
else if (mx < ((PolyFT4 *)pkt)->x3) mx = ((PolyFT4 *)pkt)->x3;
|
||||
if (mx >= -0xA0 && mn < 0xA1) {
|
||||
if (((PolyFT4 *)pkt)->y3 < mny) mny = ((PolyFT4 *)pkt)->y3;
|
||||
else if (my < ((PolyFT4 *)pkt)->y3) my = ((PolyFT4 *)pkt)->y3;
|
||||
if (my >= -0x6E && mny < 0x6F) {
|
||||
s32 za, zb;
|
||||
zb = g.sz2;
|
||||
if (zb < g.sz3) zb = g.sz3;
|
||||
za = g.sz0;
|
||||
if (za < g.sz1) za = g.sz1;
|
||||
if (za < zb) za = zb;
|
||||
g.opz = za;
|
||||
|
||||
f3 = 0; f2 = 0; f1 = 0; f0 = 0;
|
||||
|
||||
vw = *(u32 *)va;
|
||||
vzw = *(u32 *)(va + 4);
|
||||
x0 = vw; y0 = vw >> 16; z0 = vzw;
|
||||
BOXTEST(f0, x0, y0, z0, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
|
||||
BOXTEST(f1, x0, y0, z0, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
|
||||
BOXTEST(f2, x0, y0, z0, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
|
||||
BOXTEST(f3, x0, y0, z0, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
|
||||
vw = *(u32 *)vb;
|
||||
vzw = *(u32 *)(vb + 4);
|
||||
x1 = vw; y1 = vw >> 16; z1 = vzw;
|
||||
BOXTEST(f0, x1, y1, z1, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
|
||||
BOXTEST(f1, x1, y1, z1, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
|
||||
BOXTEST(f2, x1, y1, z1, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
|
||||
BOXTEST(f3, x1, y1, z1, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
|
||||
vw = *(u32 *)vc;
|
||||
vzw = *(u32 *)(vc + 4);
|
||||
x2 = vw; y2 = vw >> 16; z2 = vzw;
|
||||
BOXTEST(f0, x2, y2, z2, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
|
||||
BOXTEST(f1, x2, y2, z2, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
|
||||
BOXTEST(f2, x2, y2, z2, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
|
||||
BOXTEST(f3, x2, y2, z2, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
|
||||
vw = *(u32 *)vd;
|
||||
vzw = *(u32 *)(vd + 4);
|
||||
x3 = vw; y3 = vw >> 16; z3 = vzw;
|
||||
BOXTEST(f0, x3, y3, z3, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
|
||||
BOXTEST(f1, x3, y3, z3, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
|
||||
BOXTEST(f2, x3, y3, z3, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
|
||||
BOXTEST(f3, x3, y3, z3, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
|
||||
|
||||
if (f0 | f1 | f2 | f3) {
|
||||
u32 *otp;
|
||||
u32 rgbw;
|
||||
ATTEN(a0v, f0, x0, y0, z0, cx0, cy0, cz0, r0, r0lo);
|
||||
ATTEN(a1v, f1, x0, y0, z0, cx1, cy1, cz1, r1, r1lo);
|
||||
ATTEN(a2v, f2, x0, y0, z0, cx2, cy2, cz2, r2, r2lo);
|
||||
ATTEN3(a3v, f3, x0, y0, z0, cx3, cy3, cz3, r3, r3lo, r3, r3);
|
||||
CLAMP80(c0, a0v, a1v, a2v, a3v);
|
||||
ATTEN(a0v, f0, x1, y1, z1, cx0, cy0, cz0, r0, r0lo);
|
||||
ATTEN(a1v, f1, x1, y1, z1, cx1, cy1, cz1, r1, r1lo);
|
||||
ATTEN(a2v, f2, x1, y1, z1, cx2, cy2, cz2, r2, r2lo);
|
||||
ATTEN3W(a3v, a2v, f3, x1, y1, z1, cx3, cy3, cz3, r3, r3lo, r3, r3);
|
||||
CLAMP80S(c1, a0v, a1v, a2v, a3v);
|
||||
ATTEN(a0v, f0, x2, y2, z2, cx0, cy0, cz0, r0, r0lo);
|
||||
ATTEN(a1v, f1, x2, y2, z2, cx1, cy1, cz1, r1, r1lo);
|
||||
ATTEN(a2v, f2, x2, y2, z2, cx2, cy2, cz2, r2, r2lo);
|
||||
ATTEN3(a3v, f3, x2, y2, z2, cx3, cy3, cz3, r3, r3lo, r3, r3);
|
||||
CLAMP80(c2, a0v, a1v, a2v, a3v);
|
||||
ATTEN(a0v, f0, x3, y3, z3, cx0, cy0, cz0, r0, r0lo);
|
||||
ATTEN(a1v, f1, x3, y3, z3, cx1, cy1, cz1, r1, r1lo);
|
||||
ATTEN(a2v, f2, x3, y3, z3, cx2, cy2, cz2, r2, r2lo);
|
||||
ATTEN3(a3v, f3, x3, y3, z3, cx3, cy3, cz3, r3, r3lo, r3, r3);
|
||||
CLAMP80(c3, a0v, a1v, a2v, a3v);
|
||||
|
||||
*(u32 *)&((PolyGT4 *)pkt)->x0 = *(u32 *)&tmpxy[0];
|
||||
*(u32 *)&((PolyGT4 *)pkt)->x1 = *(u32 *)&tmpxy[1];
|
||||
*(u32 *)&((PolyGT4 *)pkt)->x2 = *(u32 *)&tmpxy[2];
|
||||
gte_stsxy((long *)&((PolyGT4 *)pkt)->x3);
|
||||
tp = (u32 *)prim->w0;
|
||||
cb = 0x3C000000;
|
||||
rgbw = c0 | cb; rgbw |= c0 << 8; rgbw |= c0 << 16;
|
||||
((PolyGT4 *)pkt)->rgb0 = rgbw;
|
||||
rgbw = c1 | cb; rgbw |= c1 << 8; rgbw |= c1 << 16;
|
||||
((PolyGT4 *)pkt)->rgb1 = rgbw;
|
||||
rgbw = (c2 | cb) | (c2 << 8) | (c1 << 16);
|
||||
((PolyGT4 *)pkt)->rgb2 = rgbw;
|
||||
rgbw = (c3 | cb) | (c3 << 8) | (c1 << 16);
|
||||
((PolyGT4 *)pkt)->rgb3 = rgbw;
|
||||
((PolyGT4 *)pkt)->uv0 = tp[1];
|
||||
((PolyGT4 *)pkt)->uv1 = tp[2];
|
||||
uvw = tp[3];
|
||||
((PolyGT4 *)pkt)->uv2 = uvw;
|
||||
((PolyGT4 *)pkt)->uv3 = uvw >> 16;
|
||||
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
|
||||
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0xC000000;
|
||||
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
|
||||
pkt += 0x34;
|
||||
} else {
|
||||
u32 *otp;
|
||||
u32 rgbw;
|
||||
*(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0];
|
||||
*(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1];
|
||||
*(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2];
|
||||
tp = (u32 *)prim->w0;
|
||||
cb = tp[0] & 0xFF000000;
|
||||
((PolyFT4 *)pkt)->rgbc = cb | 0x101010;
|
||||
((PolyFT4 *)pkt)->uvc0 = tp[1];
|
||||
((PolyFT4 *)pkt)->uvp1 = tp[2];
|
||||
uvw = tp[3];
|
||||
((PolyFT4 *)pkt)->uv2 = uvw;
|
||||
((PolyFT4 *)pkt)->uv3 = uvw >> 16;
|
||||
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
|
||||
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000;
|
||||
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
|
||||
pkt += 0x28;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
D_800A5E60 = pkt;
|
||||
}
|
||||
@@ -6194,3 +6194,56 @@ Artifacts: `.run/giants/s19_func_8017BF14_b1.c` (45/4763), a **pin-free fallback
|
||||
**Cold-start economics, measured:** a 4,763-instruction leaf giant with a *findable* matched relative
|
||||
reached 99.06% in one pass but did not close. Budget a second pass for anything this size; the first
|
||||
pass buys the decode, the frame, and the length — the last ~1% is register grants.
|
||||
|
||||
## §80 — A do-not-re-buy entry is scoped to its BASE, not to the function; and the pin's hidden cost is an unconditional `qty_phys_sugg` (Phase 29 SESSION-19, `func_8017BF14` 45 → 0)
|
||||
|
||||
Round 2 closed the 4,763-instruction behemoth (`45 → 37 → 33 → 21 → 11 → 3 → 2 → 0`, reproduced 3×
|
||||
from independent work dirs, banked whole-binary BYTE-IDENTICAL). The route matters more than the win.
|
||||
|
||||
### ⚠️ THE PROCESS CORRECTION: a measured negative is relative to the draft it was measured on
|
||||
Round 1 left a careful ~40-row do-not-re-buy table. **Three of its entries INVERTED on round 2's base.**
|
||||
The same edit (`qsingle23`) measured **1,040 mismatched on the 45-base and 11 on the 21-base**.
|
||||
Re-testing the round-1 negative list cost **~20 seconds** and produced **three of the seven winning
|
||||
levers**.
|
||||
|
||||
**So: a do-not-re-buy table is a record of `(edit, base) → result`, NOT `edit → useless`.** After any
|
||||
lever that moves the base materially, **re-run the negative list** — it is seconds with a real harness
|
||||
and it is where the next levers hide. This retroactively qualifies every such table in this cookbook
|
||||
(§45, §60b, §75a, §76, §78, §79 and round 1 of this function): treat them as *starting hypotheses at
|
||||
the base where they were taken*, not as closed questions.
|
||||
|
||||
Corollary already seen: round 1 measured "removing the `va→$t2` pin costs 4% elsewhere" and concluded
|
||||
*keep the pin*. On a base where `c0..c3` sit at function scope, **removing those pins is worth 21→13**
|
||||
— the opposite conclusion from the same experiment.
|
||||
|
||||
### The pin's hidden cost, with the citation
|
||||
`combine_regs`' hard-register branch (`local-alloc.c:1795`, reached from `:1295` with
|
||||
`already_dead == 0`) records the pinned register in **`qty_phys_sugg` unconditionally — there is no
|
||||
death guard.** So a `register __asm__` pin does not merely *prefer* a register: it actively invites
|
||||
local-alloc to tie producer chains into it, which is exactly the residual-(a) tie round 1 diagnosed
|
||||
but mis-cured. Three separable cures exist; the new one is worth knowing:
|
||||
- **R7 — a zero-byte `__asm__` ref that keeps the pinned value LIVE PAST the temp**, so
|
||||
`find_free_reg` cannot honour the suggestion. That closed the final 2 instructions, and was
|
||||
*necessary* because `c1→$a0` proved uniquely load-bearing (it is what spills `r1lo`; every
|
||||
alternative pin lost 64 instructions).
|
||||
|
||||
### The flagged "#1 move" LOST — and why the failure is informative
|
||||
Variable REUSE (§45-A / RC-14 MERGE) was swept in full: **every merge lost, 43–3294 across 8 merges.**
|
||||
It was the right lever class for the sibling `func_8017F510` (97 → 10) and the wrong one here, for a
|
||||
structural reason worth carrying: **the TRI and QUAD grants did not differ by RANK, they differed by
|
||||
IDENTITY — two independent allocno sets.** Re-ranking inside one set cannot fix a two-set problem.
|
||||
**Diagnose whether you have a ranking problem or an identity problem before reaching for a merge.**
|
||||
The actual fix was `s32 c0,c1,c2,c3;` at **function** scope (33 → 21), read off the two matched
|
||||
relatives (`b5:310`, `b4:338`) and confirmed against the target itself: its TRI grants are *identical*
|
||||
to its QUAD grants.
|
||||
|
||||
### §78's attribution primitive, run and reproduced
|
||||
Under `-fno-schedule-insns`, `-fno-schedule-insns2`, and both, the draft's order was **unchanged** ⇒
|
||||
the rgb-accumulator transposition was never a `sched.c` decision. A 3-statement accumulator pins the
|
||||
value to one register, so no scheduler *could* hoist the `or` above the `sw`. Changing the grant fixed
|
||||
the order for free — §78 reproduced on a second function.
|
||||
|
||||
### Cold-start economics, now complete
|
||||
A 4,763-instruction leaf giant with a findable matched relative: **round 1 = decode + exact length +
|
||||
exact frame + 99.06%; round 2 = the last 45.** Two passes, and the second was far cheaper than the
|
||||
first. Budget two passes at this size and do not read a 99% round-1 result as a stall.
|
||||
|
||||
@@ -4,16 +4,16 @@
|
||||
# cross-binary collapsible-byte leverage: docs/duplicates.cross.md.
|
||||
|
||||
# THREE progress metrics (all matter — see the labels):
|
||||
FLEET fn-count byte-ident: 315457 / 353717 = 89.18% (REAL+LINKED+empties; FUNCTION-count, ×134-inflated — one crack counts per overlay)
|
||||
FLEET instr-weighted : 10580590 / 13141652 = 80.5% (shipped .text across main + resident + 138 overlays; the decomp.dev-DISPLAY number)
|
||||
FLEET distinct-code(uniq): 3838143 / 5634875 = 68.1% (64905/87459 unique fns; the DISTINCT-RE number)
|
||||
FLEET fn-count byte-ident: 315458 / 353717 = 89.18% (REAL+LINKED+empties; FUNCTION-count, ×134-inflated — one crack counts per overlay)
|
||||
FLEET instr-weighted : 10585353 / 13141652 = 80.5% (shipped .text across main + resident + 138 overlays; the decomp.dev-DISPLAY number)
|
||||
FLEET distinct-code(uniq): 3842906 / 5634875 = 68.2% (64906/87459 unique fns; the DISTINCT-RE number)
|
||||
MAIN game-code weighted : 436 / 60201 = 0.7% (INCLUDED in the fleet numbers above since 2026-07-22 — roadmap §1 metrics contract; LINKED-excluding Ghidra sig dated 2026-06-14; caveat is R34: no independent second oracle for a PS-X EXE, NOT drift)
|
||||
(fleet EXCLUDING main, for continuity with pre-2026-07-22 readings: 10580154 / 13081451 = 80.9%)
|
||||
(fleet EXCLUDING main, for continuity with pre-2026-07-22 readings: 10584917 / 13081451 = 80.9%)
|
||||
|
||||
FLEET REAL substantive : 313602 (of which dedup-shared 239530 via 1886 groups / 239604 instances)
|
||||
FLEET REAL substantive : 313603 (of which dedup-shared 239530 via 1886 groups / 239604 instances)
|
||||
FLEET LINKED PsyQ objs : 959
|
||||
FLEET NON_MATCHING : 7 (0 in any default build — G4)
|
||||
FLEET INCLUDE_ASM stubs : 38253
|
||||
FLEET INCLUDE_ASM stubs : 38252
|
||||
FLEET matchable : 353717
|
||||
|
||||
| binary | REAL | shared | LINKED | byte-ident | matchable | byte-ident % |
|
||||
@@ -89,7 +89,7 @@ FLEET matchable : 353717
|
||||
| ov_SC03_113 | 2252 | 1740 | 0 | 2255 | 2468 | 91.4% |
|
||||
| ov_SC03_114 | 2242 | 1738 | 0 | 2244 | 2415 | 92.9% |
|
||||
| ov_SC03_115 | 2259 | 1738 | 0 | 2261 | 2472 | 91.5% |
|
||||
| ov_SC03_116 | 2249 | 1738 | 0 | 2252 | 2438 | 92.4% |
|
||||
| ov_SC03_116 | 2250 | 1738 | 0 | 2253 | 2438 | 92.4% |
|
||||
| ov_SC03_117 | 2277 | 1738 | 0 | 2283 | 2557 | 89.3% |
|
||||
| ov_SC03_118 | 2323 | 1762 | 0 | 2324 | 2685 | 86.6% |
|
||||
| ov_SC03_119 | 2322 | 1762 | 0 | 2323 | 2685 | 86.5% |
|
||||
|
||||
@@ -3871,6 +3871,48 @@ conditional) · main-EXE/B9 + GLM/B6 + resident's 14 walls (P30) · behemoths B7
|
||||
Artifacts: `s19_func_8017BF14_b1.c` (45/4763) + a **pin-free fallback at 789/4763 that is 100%
|
||||
structural** + `s19_bf14_report.md` (~40-row do-not-re-buy table, 4 refuted diagnoses).
|
||||
|
||||
- **🏆🏆🏆 2026-07-25 (SESSION-19) — `func_8017BF14` (4,763 ins) CLOSED IN ROUND 2: 45 → 0.
|
||||
The 4th behemoth of the session, and the largest single function matched in the project.**
|
||||
`45 → 37 → 33 → 21 → 11 → 3 → 2 → 0`, reproduced 3× from independent work dirs. **Verified
|
||||
independently (R14):** `match_one` → **MATCH (4763 ins)**; `harvest_verify --binary ov_SC03_116`
|
||||
→ **BYTE-IDENTICAL**. (Agent was interrupted mid-run by a weekly API limit and RESUMED FROM ITS
|
||||
TRANSCRIPT — its round-2 harness `bf14_mk2.py`/`bf14_sw2.sh` survived intact, so nothing was
|
||||
re-derived.)
|
||||
**⚠️ THE PROCESS CORRECTION THAT MATTERS MORE THAN THE MATCH (→ §80): A DO-NOT-RE-BUY ENTRY IS
|
||||
SCOPED TO ITS BASE, NOT TO THE FUNCTION.** Three of round 1's ~40 carefully-measured negatives
|
||||
**INVERTED** on round 2's base — the same edit (`qsingle23`) measured **1,040 mismatched on the
|
||||
45-base and 11 on the 21-base**. Re-testing the round-1 negative list cost **~20 seconds** and
|
||||
produced **three of the seven winning levers**. **A do-not-re-buy table records `(edit, base) →
|
||||
result`, NOT `edit → useless`; after any lever that moves the base materially, RE-RUN THE NEGATIVE
|
||||
LIST.** This retroactively qualifies every such table in the cookbook (§45, §60b, §75a, §76, §78,
|
||||
§79) — they are starting hypotheses at the base where they were taken, not closed questions.
|
||||
Concrete instance: round 1 measured "removing the `va→$t2` pin costs 4% elsewhere" ⇒ *keep the pin*;
|
||||
on a base with `c0..c3` at function scope, **removing those pins is worth 21→13** — opposite
|
||||
conclusion, same experiment.
|
||||
**MY FLAGGED "#1 MOVE" LOST, AND THE FAILURE IS THE FINDING.** I briefed variable REUSE (§45-A /
|
||||
RC-14) as the #1 lever because it took `func_8017F510` from 97→10. Swept in full here: **every
|
||||
merge lost, 43–3294 across 8 merges.** Reason: **the TRI and QUAD grants did not differ by RANK,
|
||||
they differed by IDENTITY — two independent allocno sets, and re-ranking inside one set cannot fix
|
||||
a two-set problem.** Diagnose ranking-vs-identity BEFORE reaching for a merge. The actual fix
|
||||
(`s32 c0,c1,c2,c3;` at **function** scope, 33→21) was **read off the two matched relatives**
|
||||
(`b5:310`, `b4:338`) and confirmed against the target (its TRI grants are identical to its QUAD
|
||||
grants) — **the 4th time today that reading a matched relative beat the clever lever.**
|
||||
**THE PIN'S HIDDEN COST, WITH A CITATION:** `combine_regs`' hard-register branch
|
||||
(`local-alloc.c:1795`, reached from `:1295` with `already_dead == 0`) records the pinned register in
|
||||
**`qty_phys_sugg` UNCONDITIONALLY — no death guard**. A pin doesn't merely *prefer* a register, it
|
||||
invites local-alloc to tie producer chains into it. New cure **R7**: a zero-byte `__asm__` ref that
|
||||
keeps the pinned value LIVE PAST the temp so `find_free_reg` can't honour the suggestion — closed
|
||||
the last 2 ins, and was necessary because `c1→$a0` is uniquely load-bearing (it is what spills
|
||||
`r1lo`; every alternative pin lost 64 ins).
|
||||
**§78's ATTRIBUTION PRIMITIVE RUN AND REPRODUCED:** under `-fno-schedule-insns`,
|
||||
`-fno-schedule-insns2`, and both, the order was **unchanged** ⇒ the rgb-accumulator transposition
|
||||
was never a `sched.c` decision (a 3-statement accumulator pins the value, so no scheduler *could*
|
||||
hoist the `or` above the `sw`). Changing the grant fixed the order for free.
|
||||
**COLD-START ECONOMICS, NOW COMPLETE:** round 1 = decode + exact length + exact frame + 99.06%;
|
||||
round 2 = the last 45, and far cheaper than round 1. **Budget TWO passes at this size; do not read a
|
||||
99% round-1 result as a stall.** Also found a **5th original-source copy-paste artefact** (QUAD
|
||||
vertex-1 box-3 y-axis accumulates into `a2v` while its kill branch still says `a3v`).
|
||||
|
||||
> **🛑 SESSION-19 CLOSING CHECKPOINT (2026-07-25, Opus 5 @ High) — REFRESHED mid-session; supersedes
|
||||
> both the SESSION-18 block and the earlier SESSION-19 block (which was written before the
|
||||
> ENGINE_SHB / class-B / dedup_extend-bug work and went stale). Fresh session safe here.**
|
||||
|
||||
@@ -3276,7 +3276,846 @@ DEFINE_func_8017BEB4() /* dedup: shared engine-core @0x8017BEB4 (src/shared) */
|
||||
|
||||
INCLUDE_ASM("asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C", func_8017BEBC);
|
||||
|
||||
INCLUDE_ASM("asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C", func_8017BF14);
|
||||
#include "common.h"
|
||||
#include "/home/musashi/bfm-decomp/src/shared/engine_types.h"
|
||||
|
||||
/* ===========================================================================
|
||||
* func_8017BF14 -- 4,763 ins, ov_SC03_116 (behemoth #4). *** MATCH ***
|
||||
*
|
||||
* STATUS (2026-07-25, session 20 round 2, gcc-2.7.2 pinned triple):
|
||||
* python3 tools/match_one.py func_8017BF14 --c <this file> \
|
||||
* --asm-subdir asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C
|
||||
* -> MATCH (4763 ins) func_8017BF14 [reproduced 3x, private work dirs]
|
||||
* LENGTH EXACT | OPCODE HISTOGRAM EXACT (L1 = 0) | STACK FRAME EXACT
|
||||
* (all 127 slots at the target's offsets, frame 0x360)
|
||||
* masked index-wise diff: 0 / 4763 mismatched.
|
||||
* `match_one` is the CANDIDATE gate. The whole-binary SHA1 arbiter (G3/P9)
|
||||
* is run by the coordinator and is the only thing that makes this final.
|
||||
*
|
||||
* Round 1 closed at 45/4763 mismatched. Round 2 took 45 -> 37 -> 33 -> 21
|
||||
* -> 11 -> 3 -> 2 -> 0. See .run/giants/s19_bf14_report2.md.
|
||||
*
|
||||
* WHAT IT IS
|
||||
* The *four*-light-box variant of the volumetric-light renderer whose
|
||||
* 3-box sibling func_8017D960 (3,338 ins, ov_SC03_090) is MATCHED, and whose
|
||||
* unlit ancestor func_8017BEBC (ov_SC03_099) is MATCHED. Same family, same
|
||||
* skeleton; this is the biggest member.
|
||||
*
|
||||
* Signature: func_8017BF14(s32 arg0, s32 lim). Unlike every other member of
|
||||
* the family this one is a LEAF -- 0 callees. The 3-call prologue
|
||||
* (func_800491EC / func_800547D8 / func_80052E38) of the siblings is gone;
|
||||
* `lim` arrives as arg1 (spilled to 0xB0). That is why the frame has no
|
||||
* 0x10 argument area (tmpxy[] starts at sp+0x00) and no $ra save.
|
||||
*
|
||||
* Per part (stride 0x14, outer loop): build the 8-corner AABB in box[],
|
||||
* rtpt/rtpt + rtps/rtps -> sxy[8], stszotz -> g.otz, reject on
|
||||
* `lim >= g.otz`, then screen-space bbox reject on X (-0xA0..0xA1) and
|
||||
* Y (-0x6E..0x6F).
|
||||
* Per prim (stride 0xC, inner loop): rtpt the 3 vertices, stflg mask
|
||||
* 0x7F85E000, nclip, stopz > 0, then a 4-way range tree on `code = w & 7`
|
||||
* that keeps ONLY codes 6,7 (tri) and 2,3 (quad); 0,1,4,5 fall through to
|
||||
* the loop tail.
|
||||
* Per drawn poly: screen bbox reject, then each vertex is tested against
|
||||
* FOUR axis-aligned light boxes (flags f0..f3), and if any is lit a
|
||||
* 0x00..0x80 attenuation per active box is computed, summed, biased +0x10
|
||||
* and clamped to 0x80 -> a grey gouraud vertex colour.
|
||||
* lit -> POLY_GT3 (0x28, tag 0x34000000, OT 0x9000000)
|
||||
* POLY_GT4 (0x34, tag 0x3C000000, OT 0xC000000)
|
||||
* unlit -> POLY_FT3 (0x20, OT 0x7000000) / POLY_FT4 (0x28, OT 0x9000000)
|
||||
* with rgbc = (tp[0] & 0xFF000000) | 0x101010 <-- NOT black,
|
||||
* unlike func_8017D960 where the unlit colour is plain black.
|
||||
*
|
||||
* THE FOUR LIGHT BOXES (stride 0x1C, {s32 enable; u16 cx,cy,cz; s32 range})
|
||||
* D_80197C28 / D_80197C44 / D_80197C60 / D_80197C7C.
|
||||
* Falloff geometry differs from the 3-box sibling: RLO = R - 0x200 (not
|
||||
* -0x80) and the ramp is ((R - d) / 4) (not (R - d)), so the 0x80 ceiling is
|
||||
* reached over a 0x200-wide band instead of 0x80. The `/ 4` is a SIGNED
|
||||
* divide -- `bgez / addiu 3 / sra 2` -- not a shift.
|
||||
*
|
||||
* FIVE ORIGINAL-SOURCE COPY-PASTE ARTEFACTS, all byte-proven
|
||||
* The 4th light box was bolted onto a copy of the 3-box source BY HAND and
|
||||
* the hand edit was incomplete in five places. Each was read off the target
|
||||
* and each removed a measured delta.
|
||||
* (1) `r3lo = r2 - 0x200;` -- box 3's low radius is derived from box 2's
|
||||
* RANGE VARIABLE, not from its own D_80197C88. Proven by the target's
|
||||
* `addiu $t6, $s0, -0x200` reusing the register that box 2's `lw` filled;
|
||||
* spelling it `D_80197C6C - 0x200` re-loads the global (+2 ins).
|
||||
* (2) Only SIX of the eight radius variables are zero-initialised
|
||||
* (r0,r1,r2,r0lo,r1lo,r2lo) -- r3/r3lo are left uninitialised, exactly
|
||||
* the init list the 3-box version needed.
|
||||
* (3) The ATTEN body is written out LONGHAND 7 times (3 tri vertices +
|
||||
* 4 quad vertices). When box 3 was bolted on, the `R` of the z- and
|
||||
* y-axis KILL tests was left as r2 in three of those copies:
|
||||
* tri v0: z and y use r2 tri v2: z uses r2 all others use r3.
|
||||
* Proven by the 24 `sll $v0,$s0,16` sites: 3 per group in box 2 plus
|
||||
* exactly three extra at idx 1471, 1504 (tri v0) and 2178 (tri v2).
|
||||
* (4) In the QUAD lit arm only, rgb2 and rgb3 take their `<< 16` term from
|
||||
* c1, not from c2/c3. Proven by the target CSE-ing ONE
|
||||
* `sll $a0, $a0, 16` and re-using $a0 for all three stores.
|
||||
* (5) *** ROUND 2 *** In the QUAD arm's VERTEX-1 group only, the box-3
|
||||
* ATTEN's Y-axis `else if` branch accumulates into a2v instead of a3v,
|
||||
* while that same test's KILL branch still says a3v. Modelled by the
|
||||
* ATTEN3W macro below (`AW` = the y-else destination). Byte-proof:
|
||||
* idx 3875 addu $a2,$zero,$zero kill branch -> a3v ($a2) [agreed]
|
||||
* idx 3889 sra $a3,$s2,7 else branch -> a2v ($a3) [was the
|
||||
* last structural residual]
|
||||
* Costs zero instructions; the four other quad/tri sites are NOT like
|
||||
* this (each was measured -- putting the artefact anywhere else is +2).
|
||||
*
|
||||
* FRAME (0x360, leaf -- no $ra, no argument area)
|
||||
* 0x000 tmpxy[4] | 0x010 box[8] | 0x050 sxy[8] |
|
||||
* 0x090 g{otz,flag,opz,sz0..sz3} | 0x0B0 lim | 0x0B8 j | 0x0C0 i |
|
||||
* 0x0C8 vd | 0x0D0 ot | 0x0D8 pkt | 0x0E0 f2 | 0x0E8 f3 |
|
||||
* 0x0F0/0x0F8/0x100 x3,z3,y3 | 0x108 prim | 0x110 nprim | 0x118 vtx |
|
||||
* 0x120 nparts | 0x128 part | 0x130..0x1D0 lo/hi bounds (21 s16 slots) |
|
||||
* 0x1D8..0x230 cx0..cz3 (12) | 0x238 r0 | 0x240 r1lo | 0x248 r2lo |
|
||||
* 0x250 r3lo | 0x288..0x2D0 the LICM-hoisted sign-extended bounds |
|
||||
* 0x328/0x330 spilled vertex coords | 0x338..0x358 s0-s7,fp.
|
||||
* *** THE SLOT ORDER IS THE DECLARATION-ORDER ORACLE (see L5). ***
|
||||
*
|
||||
* ---------------------------------------------------------------------------
|
||||
* ROUND-1 LEVERS (kept; measured effect is byte-identical %, anchored)
|
||||
*
|
||||
* L1 `cb = (tp[0] & 0xFF000000) | 0x101010;` in both UNLIT arms.
|
||||
* L2 box-3 ATTEN kill-register per copy (artefact 3) + `r3lo = r2 - 0x200`
|
||||
* (artefact 1). 89.15% -> 96.96% shape; killed sra+11 / sll+9.
|
||||
* L3 quad-lit rgb2/rgb3 use `c1 << 16` (artefact 4). Killed the last sll+2.
|
||||
* L4 [SUPERSEDED BY R3] `s32 c0..c3` declared inside the two CULL blocks.
|
||||
* L5 `s32 f0, f1, f2, f3;` MOVED TO IMMEDIATELY AFTER `u8 *pkt;`.
|
||||
* Spilled pseudos get stack slots in PSEUDO-NUMBER order and pseudo
|
||||
* numbers are handed out in DECLARATION order, so the target's stack
|
||||
* layout is a direct read-out of its declaration order. After this one
|
||||
* move ALL 127 stack slots agree with the target exactly.
|
||||
* L6 `u32 rgbw;` per emit arm.
|
||||
* L7 RC-15 zero-byte ref dial on `mny`, first statement of the TRI cull
|
||||
* block -- flips my->$a3 / mny->$a2 to the target's grant.
|
||||
* L8 `cb` 2-statement accumulator [SUPERSEDED BY R2]; quad-lit rgb word as a
|
||||
* 3-statement accumulator [kept for rgb0/rgb1, SUPERSEDED for rgb2/rgb3
|
||||
* by R5].
|
||||
* L9 FOUR REGISTER PINS: va->$t2, w->$a1, f0->$s3, c1->$a0.
|
||||
* Round 2 removed va and w (see R4); f0 and c1 REMAIN and are both
|
||||
* load-bearing. This function has NO `jal`, so Sec.74's caller-saved
|
||||
* pin-spanning-a-call hazard cannot arise -- that is why pins are usable
|
||||
* on this family member and were a trap on the others.
|
||||
*
|
||||
* ---------------------------------------------------------------------------
|
||||
* ROUND-2 LEVERS -- 45 -> 0. Metric is `match_one` MISMATCH COUNT (length is
|
||||
* exact throughout, so the raw count is honest). Every number is measured.
|
||||
*
|
||||
* THE ONE MECHANISM BEHIND R1/R2/R4/R7. A `register __asm__` pin makes the
|
||||
* variable a HARD REG in the RTL from the start. When a 1-death local temp is
|
||||
* produced from, or consumed into, that hard reg, local-alloc.c's
|
||||
* `combine_regs` takes its hard-register branch (local-alloc.c:1795-1820) and
|
||||
* records the pinned register in `qty_phys_sugg` for the temp's quantity --
|
||||
* UNCONDITIONALLY, there is no death guard on that path. The temp then lands
|
||||
* in the pinned register and the operation is done IN PLACE. The target,
|
||||
* whose variable is an ordinary pseudo (reg_qty == -1 for anything crossing a
|
||||
* block), never gets that suggestion and keeps the temp in $v0/$v1.
|
||||
* Three independent cures, all used here:
|
||||
* (i) give the temp a NAMED variable with >1 death, so local-alloc.c:472
|
||||
* (`reg_basic_block >= 0 && reg_n_deaths == 1`) refuses it a quantity
|
||||
* and combine_regs bails at its very first test -> R1
|
||||
* (ii) drop the pin, if the pin is not load-bearing -> R4
|
||||
* (iii) keep the pinned value LIVE past the temp, so that
|
||||
* find_free_reg cannot honour the suggestion -> R7
|
||||
*
|
||||
* R1 `CLAMP80S` -- the four-way attenuation sum gets its own named variable
|
||||
* `sv`, used at BOTH c1 sites (2 deaths). Cure (i). 45 -> 37
|
||||
* Target: `addu $v0,..; addu $v0,..; addu $v0,..; addiu <c>,$v0,0x10`.
|
||||
* Without it the whole chain is tied into the pinned c1 ($a0).
|
||||
* Only the two c1 sites: `sv` on all 7 sites collapses the frame (-64).
|
||||
* R2 UNLIT arms store `cb | 0x101010` as an expression instead of doing
|
||||
* `cb |= 0x101010` in place. `cb` is a function-scope global allocno, so
|
||||
* the in-place form writes $a1; the expression form is a 1-death local
|
||||
* that combine_regs ties to the DYING constant register $v1, which is
|
||||
* what the target does. 37 -> 33
|
||||
* R3 *** `s32 c0, c1, c2, c3;` AT FUNCTION SCOPE, not per cull block. ***
|
||||
* Read straight off the two MATCHED relatives (func_8017D960 line 310,
|
||||
* func_8017F510 line 338), and confirmed by the target itself: its TRI
|
||||
* grants (c0=$t4, c1=$a0, c2=$t2) are IDENTICAL to its QUAD grants, which
|
||||
* is only possible if both arms share one set of allocnos. Per-cull-block
|
||||
* scope splits them into two independent allocno sets and the TRI set
|
||||
* drifts. 33 -> 21
|
||||
* Declaration POSITION is neutral (5 anchors swept, all 21).
|
||||
* NOTE this REVERSES round-1's L4. L4 was correct on the round-1 base --
|
||||
* it was supplying the extra local allocno that spills r1lo -- but R1+R2
|
||||
* supply that pressure now, and the c1 pin does the rest.
|
||||
* R4 DROP the `va->$t2` and `w->$a1` pins. With c0..c3 at function scope
|
||||
* they are no longer load-bearing, and they were the sole cause of the
|
||||
* prim-word producer ties (`andi`/`srl` written straight into $t2/$a1).
|
||||
* Cure (ii). 21 -> 17 -> 13 (pair)
|
||||
* Round 1 measured these as worth 4%; that was true of the round-1 base
|
||||
* and is FALSE here. Base-dependence, not a contradiction.
|
||||
* R5 QUAD lit rgb2/rgb3 revert to the SINGLE-EXPRESSION form (rgb0/rgb1 keep
|
||||
* the 3-statement accumulator). The intermediates then become 1-death
|
||||
* local temps that alternate $v0/$v1, which is what lets the target's
|
||||
* store of the previous rgb word sit one slot LATER. 21 -> 11
|
||||
* Doing it to all four, or to rgb0/rgb1 only, is worse (23 / 21).
|
||||
* R6 Artefact 5 -- `ATTEN3W(a3v, a2v, ...)` at the QUAD vertex-1 site. 3 -> 2
|
||||
* R7 RC-15 zero-byte ref `__asm__ __volatile__ ("" :: "r" (c1));` placed
|
||||
* immediately after the TRI arm's rgb1 store. Cure (iii): it keeps the
|
||||
* pinned c1 ($a0) live past `c1 << 16`, so find_free_reg cannot honour
|
||||
* combine_regs' $a0 suggestion and the shift goes to $v0. 2 -> 0
|
||||
* The c1 pin CANNOT simply be removed: it is what spills r1lo (dropping
|
||||
* it, or moving it to any other colour, costs -64 length). Measured.
|
||||
*
|
||||
* SCHEDULER ATTRIBUTION (Sec.76 primitive, run in round 2, never run before on
|
||||
* this function). Compiled with -fno-schedule-insns, with
|
||||
* -fno-schedule-insns2, and with both. The store/shift transposition at
|
||||
* idx 4643-4652 kept MY source order under all three. It was therefore never
|
||||
* a `sched.c` decision: the ordering is a CONSEQUENCE of the register grant
|
||||
* (a 3-statement accumulator pins the value in one register, so no scheduler
|
||||
* could hoist the next `or` above the `sw`). R5 fixed it by changing the
|
||||
* grant, exactly as Sec.78 predicts.
|
||||
* =========================================================================== */
|
||||
|
||||
#define gte_ldv0(r0) __asm__ volatile ( \
|
||||
"lwc2 $0, 0( %0 );" \
|
||||
"lwc2 $1, 4( %0 )" \
|
||||
: \
|
||||
: "r"( r0 ) )
|
||||
|
||||
#define gte_ldv3(r0, r1, r2) __asm__ volatile ( \
|
||||
"lwc2 $0, 0( %0 );" \
|
||||
"lwc2 $1, 4( %0 );" \
|
||||
"lwc2 $2, 0( %1 );" \
|
||||
"lwc2 $3, 4( %1 );" \
|
||||
"lwc2 $4, 0( %2 );" \
|
||||
"lwc2 $5, 4( %2 )" \
|
||||
: \
|
||||
: "r"( r0 ), "r"( r1 ), "r"( r2 ) )
|
||||
|
||||
#define gte_ldv3c(r0) __asm__ volatile ( \
|
||||
"lwc2 $0, 0( %0 );" \
|
||||
"lwc2 $1, 4( %0 );" \
|
||||
"lwc2 $2, 8( %0 );" \
|
||||
"lwc2 $3, 12( %0 );" \
|
||||
"lwc2 $4, 16( %0 );" \
|
||||
"lwc2 $5, 20( %0 )" \
|
||||
: \
|
||||
: "r"( r0 ) )
|
||||
|
||||
#define gte_rtps() __asm__ volatile ("nop;nop;rtps")
|
||||
#define gte_rtpt() __asm__ volatile ("nop;nop;rtpt")
|
||||
#define gte_nclip() __asm__ volatile ("nop;nop;nclip")
|
||||
|
||||
#define gte_stsxy(r0) __asm__ volatile ( \
|
||||
"swc2 $14, 0( %0 )" \
|
||||
: \
|
||||
: "r"( r0 ) \
|
||||
: "memory" )
|
||||
|
||||
#define gte_stsxy3(r0, r1, r2) __asm__ volatile ( \
|
||||
"swc2 $12, 0( %0 );" \
|
||||
"swc2 $13, 0( %1 );" \
|
||||
"swc2 $14, 0( %2 )" \
|
||||
: \
|
||||
: "r"( r0 ), "r"( r1 ), "r"( r2 ) \
|
||||
: "memory" )
|
||||
|
||||
#define gte_stsxy3c(r0) __asm__ volatile ( \
|
||||
"swc2 $12, 0( %0 );" \
|
||||
"swc2 $13, 4( %0 );" \
|
||||
"swc2 $14, 8( %0 )" \
|
||||
: \
|
||||
: "r"( r0 ) \
|
||||
: "memory" )
|
||||
|
||||
#define gte_stsz3(r0, r1, r2) __asm__ volatile ( \
|
||||
"swc2 $17, 0( %0 );" \
|
||||
"swc2 $18, 0( %1 );" \
|
||||
"swc2 $19, 0( %2 )" \
|
||||
: \
|
||||
: "r"( r0 ), "r"( r1 ), "r"( r2 ) \
|
||||
: "memory" )
|
||||
|
||||
#define gte_stsz4(r0, r1, r2, r3) __asm__ volatile ( \
|
||||
"swc2 $16, 0( %0 );" \
|
||||
"swc2 $17, 0( %1 );" \
|
||||
"swc2 $18, 0( %2 );" \
|
||||
"swc2 $19, 0( %3 )" \
|
||||
: \
|
||||
: "r"( r0 ), "r"( r1 ), "r"( r2 ), "r"( r3 ) \
|
||||
: "memory" )
|
||||
|
||||
#define gte_stszotz(r0) __asm__ volatile ( \
|
||||
"mfc2 $12, $19;" \
|
||||
"nop;" \
|
||||
"sra $12, $12, 2;" \
|
||||
"sw $12, 0( %0 )" \
|
||||
: \
|
||||
: "r"( r0 ) \
|
||||
: "$12", "memory" )
|
||||
|
||||
#define gte_stflg(r0) __asm__ volatile ( \
|
||||
"cfc2 $12, $31;" \
|
||||
"nop;" \
|
||||
"sw $12, 0( %0 )" \
|
||||
: \
|
||||
: "r"( r0 ) \
|
||||
: "$12", "memory" )
|
||||
|
||||
#define gte_stopz(r0) __asm__ volatile ( \
|
||||
"swc2 $24, 0( %0 )" \
|
||||
: \
|
||||
: "r"( r0 ) \
|
||||
: "memory" )
|
||||
|
||||
/* ---- the two gouraud-textured packet layouts this function emits ---------- */
|
||||
typedef struct {
|
||||
u32 tag;
|
||||
u32 rgb0; s16 x0, y0; u32 uv0;
|
||||
u32 rgb1; s16 x1, y1; u32 uv1;
|
||||
u32 rgb2; s16 x2, y2; u16 uv2, p2;
|
||||
} PolyGT3; /* 0x28 */
|
||||
|
||||
typedef struct {
|
||||
u32 tag;
|
||||
u32 rgb0; s16 x0, y0; u32 uv0;
|
||||
u32 rgb1; s16 x1, y1; u32 uv1;
|
||||
u32 rgb2; s16 x2, y2; u16 uv2, p2;
|
||||
u32 rgb3; s16 x3, y3; u16 uv3, p3;
|
||||
} PolyGT4; /* 0x34 */
|
||||
|
||||
|
||||
/* ---- the four light-volume descriptors (stride 0x1C) --------------------- */
|
||||
extern s32 D_80197C28;
|
||||
extern u16 D_80197C2C, D_80197C2E, D_80197C30;
|
||||
extern s32 D_80197C34;
|
||||
extern s32 D_80197C44;
|
||||
extern u16 D_80197C48, D_80197C4A, D_80197C4C;
|
||||
extern s32 D_80197C50;
|
||||
extern s32 D_80197C60;
|
||||
extern u16 D_80197C64, D_80197C66, D_80197C68;
|
||||
extern s32 D_80197C6C;
|
||||
extern s32 D_80197C7C;
|
||||
extern u16 D_80197C80, D_80197C82, D_80197C84;
|
||||
extern u16 D_80197C88;
|
||||
|
||||
/* ---- the box-containment test for one vertex against one light box ------- */
|
||||
#define BOXTEST(F, X, Y, Z, LX, HX, LY, HY, LZ, HZ) \
|
||||
if ((LX) < (X) && (X) < (HX) && (LY) < (Y) && (Y) < (HY) && (LZ) < (Z) && (Z) < (HZ)) F = 1
|
||||
|
||||
/* ---- the separable per-axis falloff, visited in x, z, y order ------------ */
|
||||
#define ATTEN(A, F, X, Y, Z, CX, CY, CZ, R, RLO) \
|
||||
A = 0; \
|
||||
if (F) { \
|
||||
d = (X) - (CX); if (d < 0) d = (CX) - (X); \
|
||||
if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \
|
||||
d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \
|
||||
if ((R) < d) A = 0; \
|
||||
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
|
||||
d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \
|
||||
if ((R) < d) A = 0; \
|
||||
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
|
||||
}
|
||||
|
||||
#define ATTEN3W(A, AW, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \
|
||||
A = 0; \
|
||||
if (F) { \
|
||||
d = (X) - (CX); if (d < 0) d = (CX) - (X); \
|
||||
if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \
|
||||
d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \
|
||||
if ((RZ) < d) A = 0; \
|
||||
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
|
||||
d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \
|
||||
if ((RY) < d) A = 0; \
|
||||
else if ((RLO) < d) AW = (A * (((R) - d) / 4)) >> 7; \
|
||||
}
|
||||
|
||||
#define ATTEN3(A, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \
|
||||
A = 0; \
|
||||
if (F) { \
|
||||
d = (X) - (CX); if (d < 0) d = (CX) - (X); \
|
||||
if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \
|
||||
d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \
|
||||
if ((RZ) < d) A = 0; \
|
||||
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
|
||||
d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \
|
||||
if ((RY) < d) A = 0; \
|
||||
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
|
||||
}
|
||||
|
||||
#define CLAMP80(C, A0, A1, A2, A3) C = (A0) + (A1) + (A2) + (A3) + 0x10; if ((C) > 0x80) C = 0x80
|
||||
#define CLAMP80S(C, A0, A1, A2, A3) sv = (A0) + (A1) + (A2) + (A3); C = sv + 0x10; if ((C) > 0x80) C = 0x80
|
||||
|
||||
|
||||
void func_8017BF14(s32 arg0, s32 lim)
|
||||
{
|
||||
typedef struct { u32 w0, w1, w2; } Prim;
|
||||
|
||||
extern u8 *D_800A5E60;
|
||||
extern u8 D_800A6610[];
|
||||
extern u8 D_800AF630[];
|
||||
|
||||
DVECTOR2 tmpxy[4];
|
||||
SVECTOR2 box[8];
|
||||
SVECTOR2 sxy[8];
|
||||
struct { long otz, flag, opz, sz0, sz1, sz2, sz3; } g;
|
||||
|
||||
s32 j;
|
||||
u32 i;
|
||||
u8 *vd;
|
||||
u32 ot;
|
||||
u8 *pkt;
|
||||
register s32 f0 __asm__("$19");
|
||||
s32 f1, f2, f3;
|
||||
s16 x3, z3, y3;
|
||||
Prim *prim;
|
||||
u32 nprim;
|
||||
u8 *vtx;
|
||||
s32 nparts;
|
||||
Part *part;
|
||||
s16 lo0x, hi0x, lo0y, hi0y, lo0z, hi0z;
|
||||
s16 lo1x, hi1x, lo1y, hi1y, lo1z, hi1z;
|
||||
s16 lo2x, hi2x, lo2y, hi2y, lo2z, hi2z;
|
||||
s16 lo3x, hi3x, lo3y, hi3y, lo3z, hi3z;
|
||||
s16 cx0, cy0, cz0, cx1, cy1, cz1, cx2, cy2, cz2, cx3, cy3, cz3;
|
||||
s16 r0;
|
||||
s16 r1;
|
||||
s16 r2;
|
||||
s16 r3;
|
||||
s16 r0lo;
|
||||
s16 r1lo;
|
||||
s16 r2lo;
|
||||
s16 r3lo;
|
||||
u8 *va, *vb, *vc;
|
||||
u32 w;
|
||||
s32 code;
|
||||
u32 vw, vzw;
|
||||
u32 wx, wy, wz;
|
||||
s32 xa32, xb32, t32;
|
||||
s32 xmn1, xmx1, xmn2, xmx2;
|
||||
s32 mnc, mxc;
|
||||
s16 my, mny, mx, mn;
|
||||
u8 *base;
|
||||
s16 x0, y0, z0, x1, y1, z1, x2, y2, z2;
|
||||
s32 a0v, a1v, a2v, a3v;
|
||||
register s32 c1 __asm__("$4"); s32 c0, c2, c3;
|
||||
s32 d;
|
||||
u32 *tp;
|
||||
u32 uvw;
|
||||
u32 cb;
|
||||
s32 sv;
|
||||
|
||||
base = D_800AF630;
|
||||
|
||||
r2lo = 0;
|
||||
r1lo = 0;
|
||||
r0lo = 0;
|
||||
r2 = 0;
|
||||
r1 = 0;
|
||||
r0 = 0;
|
||||
|
||||
if (D_80197C28) {
|
||||
cx0 = D_80197C2C;
|
||||
cy0 = D_80197C2E;
|
||||
r0lo = D_80197C34 - 0x200;
|
||||
r0 = D_80197C34;
|
||||
cz0 = D_80197C30;
|
||||
} else {
|
||||
cz0 = 0x6000;
|
||||
cy0 = 0x6000;
|
||||
cx0 = 0x6000;
|
||||
}
|
||||
if (D_80197C44) {
|
||||
cx1 = D_80197C48;
|
||||
cy1 = D_80197C4A;
|
||||
r1 = D_80197C50;
|
||||
r1lo = D_80197C50 - 0x200;
|
||||
cz1 = D_80197C4C;
|
||||
} else {
|
||||
cz1 = 0x6000;
|
||||
cy1 = 0x6000;
|
||||
cx1 = 0x6000;
|
||||
}
|
||||
if (D_80197C60) {
|
||||
cx2 = D_80197C64;
|
||||
cy2 = D_80197C66;
|
||||
r2 = D_80197C6C;
|
||||
r2lo = D_80197C6C - 0x200;
|
||||
cz2 = D_80197C68;
|
||||
} else {
|
||||
cz2 = 0x6000;
|
||||
cy2 = 0x6000;
|
||||
cx2 = 0x6000;
|
||||
}
|
||||
if (D_80197C7C) {
|
||||
cx3 = D_80197C80;
|
||||
cy3 = D_80197C82;
|
||||
r3lo = r2 - 0x200;
|
||||
r3 = D_80197C88;
|
||||
cz3 = D_80197C84;
|
||||
} else {
|
||||
cz3 = 0x6000;
|
||||
cy3 = 0x6000;
|
||||
cx3 = 0x6000;
|
||||
}
|
||||
|
||||
lo0x = cx0 - r0; hi0x = cx0 + r0;
|
||||
lo0y = cy0 - r0; hi0y = cy0 + r0;
|
||||
lo0z = cz0 - r0; hi0z = cz0 + r0;
|
||||
lo1x = cx1 - r1; hi1x = cx1 + r1;
|
||||
lo1y = cy1 - r1; hi1y = cy1 + r1;
|
||||
lo1z = cz1 - r1; hi1z = cz1 + r1;
|
||||
lo2x = cx2 - r2; hi2x = cx2 + r2;
|
||||
lo2y = cy2 - r2; hi2y = cy2 + r2;
|
||||
lo2z = cz2 - r2; hi2z = cz2 + r2;
|
||||
lo3x = cx3 - r3; hi3x = cx3 + r3;
|
||||
lo3y = cy3 - r3; hi3y = cy3 + r3;
|
||||
lo3z = cz3 - r3; hi3z = cz3 + r3;
|
||||
|
||||
pkt = D_800A5E60;
|
||||
part = *(Part **)(arg0 + 0xC);
|
||||
nparts = *(s32 *)(*(s32 *)(arg0 + 8) + 8);
|
||||
vtx = *(u8 **)(*(s32 *)(arg0 + 8) + 0x10);
|
||||
ot = (u32)&D_800A6610[(*(u16 *)(base + 0xA3D2)) << 14];
|
||||
|
||||
for (j = 0; j < nparts; j++, part++) {
|
||||
wx = part->xx;
|
||||
mn = wx;
|
||||
mx = wx >> 16;
|
||||
wy = part->yy;
|
||||
mny = wy;
|
||||
my = wy >> 16;
|
||||
wz = part->zz;
|
||||
box[0].vx = mn; box[0].vy = mny;
|
||||
box[1].vx = mx; box[1].vy = mny;
|
||||
box[2].vx = mn; box[2].vy = mny;
|
||||
box[3].vx = mx; box[3].vy = mny;
|
||||
box[4].vx = mn; box[4].vy = my;
|
||||
box[5].vx = mx; box[5].vy = my;
|
||||
box[6].vx = mn; box[6].vy = my;
|
||||
box[7].vx = mx; box[7].vy = my;
|
||||
wy = wz >> 16;
|
||||
box[0].vz = wz;
|
||||
box[1].vz = wz;
|
||||
box[4].vz = wz;
|
||||
box[5].vz = wz;
|
||||
box[2].vz = wy;
|
||||
box[3].vz = wy;
|
||||
box[6].vz = wy;
|
||||
box[7].vz = wy;
|
||||
|
||||
gte_ldv3c(&box[0]);
|
||||
gte_rtpt();
|
||||
gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]);
|
||||
gte_ldv0(&box[3]);
|
||||
gte_rtps();
|
||||
gte_stsxy(&sxy[3]);
|
||||
gte_ldv3c(&box[4]);
|
||||
gte_rtpt();
|
||||
gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]);
|
||||
gte_ldv0(&box[7]);
|
||||
gte_rtps();
|
||||
gte_stsxy(&sxy[7]);
|
||||
gte_stszotz(&g.otz);
|
||||
|
||||
if (lim >= g.otz) {
|
||||
xa32 = sxy[0].vx;
|
||||
xb32 = sxy[1].vx;
|
||||
if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; }
|
||||
t32 = sxy[2].vx;
|
||||
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
|
||||
t32 = sxy[3].vx;
|
||||
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
|
||||
xa32 = sxy[4].vx;
|
||||
xb32 = sxy[5].vx;
|
||||
if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; }
|
||||
t32 = sxy[6].vx;
|
||||
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
|
||||
t32 = sxy[7].vx;
|
||||
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
|
||||
mnc = xmn1;
|
||||
if (xmn2 < xmn1) mnc = xmn2;
|
||||
mxc = xmx1;
|
||||
if (mxc < xmx2) mxc = xmx2;
|
||||
if ((s16)mxc >= -0xA0 && (s16)mnc < 0xA1) {
|
||||
xa32 = sxy[0].vy;
|
||||
xb32 = sxy[1].vy;
|
||||
if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; }
|
||||
t32 = sxy[2].vy;
|
||||
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
|
||||
t32 = sxy[3].vy;
|
||||
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
|
||||
xa32 = sxy[4].vy;
|
||||
xb32 = sxy[5].vy;
|
||||
if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; }
|
||||
t32 = sxy[6].vy;
|
||||
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
|
||||
t32 = sxy[7].vy;
|
||||
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
|
||||
mnc = xmn1;
|
||||
if (xmn2 < xmn1) mnc = xmn2;
|
||||
mxc = xmx1;
|
||||
if (mxc < xmx2) mxc = xmx2;
|
||||
if ((s16)mxc >= -0x6E && (s16)mnc < 0x6F) {
|
||||
nprim = part->nprim;
|
||||
prim = (Prim *)part->prim;
|
||||
for (i = 0; i < nprim; i++, prim++) {
|
||||
w = prim->w1;
|
||||
va = vtx + (w & 0xFFFF);
|
||||
vb = vtx + (w >> 16);
|
||||
w = prim->w2;
|
||||
vc = vtx + (w & 0xFFFF);
|
||||
w = w >> 16;
|
||||
gte_ldv3(va, vb, vc);
|
||||
gte_rtpt();
|
||||
gte_stflg(&g.flag);
|
||||
if (!(g.flag & 0x7F85E000)) {
|
||||
gte_nclip();
|
||||
code = w & 7;
|
||||
vd = vtx + (w & 0xFFF8);
|
||||
gte_stopz(&g.opz);
|
||||
if (g.opz > 0) {
|
||||
switch (code) {
|
||||
case 6:
|
||||
case 7:
|
||||
/* ---------------- TRI (FT3 / GT3) ---------------- */
|
||||
gte_stsxy3c(&tmpxy[0]);
|
||||
gte_stsz3(&g.sz0, &g.sz1, &g.sz2);
|
||||
if (tmpxy[0].vx > tmpxy[1].vx) { mx = tmpxy[0].vx; mn = tmpxy[1].vx; }
|
||||
else { mn = tmpxy[0].vx; mx = tmpxy[1].vx; }
|
||||
if (tmpxy[2].vx > mx) mx = tmpxy[2].vx;
|
||||
else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx;
|
||||
if (mx >= -0xA0 && mn < 0xA1) {
|
||||
if (tmpxy[0].vy > tmpxy[1].vy) { my = tmpxy[0].vy; mny = tmpxy[1].vy; }
|
||||
else { mny = tmpxy[0].vy; my = tmpxy[1].vy; }
|
||||
if (tmpxy[2].vy > my) my = tmpxy[2].vy;
|
||||
else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy;
|
||||
if (my >= -0x6E && mny < 0x6F) {
|
||||
s32 za, zb;
|
||||
__asm__ __volatile__ ("" :: "r" (mny));
|
||||
if (g.sz0 > g.sz1) { za = g.sz0; if (za < g.sz2) za = g.sz2; }
|
||||
else { za = g.sz1; if (za < g.sz2) za = g.sz2; }
|
||||
g.opz = za;
|
||||
|
||||
f3 = 0; f2 = 0; f1 = 0; f0 = 0;
|
||||
|
||||
vw = *(u32 *)va;
|
||||
vzw = *(u32 *)(va + 4);
|
||||
x0 = vw; y0 = vw >> 16; z0 = vzw;
|
||||
BOXTEST(f0, x0, y0, z0, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
|
||||
BOXTEST(f1, x0, y0, z0, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
|
||||
BOXTEST(f2, x0, y0, z0, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
|
||||
BOXTEST(f3, x0, y0, z0, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
|
||||
vw = *(u32 *)vb;
|
||||
vzw = *(u32 *)(vb + 4);
|
||||
x1 = vw; y1 = vw >> 16; z1 = vzw;
|
||||
BOXTEST(f0, x1, y1, z1, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
|
||||
BOXTEST(f1, x1, y1, z1, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
|
||||
BOXTEST(f2, x1, y1, z1, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
|
||||
BOXTEST(f3, x1, y1, z1, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
|
||||
vw = *(u32 *)vc;
|
||||
vzw = *(u32 *)(vc + 4);
|
||||
x2 = vw; y2 = vw >> 16; z2 = vzw;
|
||||
BOXTEST(f0, x2, y2, z2, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
|
||||
BOXTEST(f1, x2, y2, z2, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
|
||||
BOXTEST(f2, x2, y2, z2, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
|
||||
BOXTEST(f3, x2, y2, z2, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
|
||||
|
||||
if (f0 | f1 | f2 | f3) {
|
||||
u32 *otp;
|
||||
u32 rgbw;
|
||||
ATTEN(a0v, f0, x0, y0, z0, cx0, cy0, cz0, r0, r0lo);
|
||||
ATTEN(a1v, f1, x0, y0, z0, cx1, cy1, cz1, r1, r1lo);
|
||||
ATTEN(a2v, f2, x0, y0, z0, cx2, cy2, cz2, r2, r2lo);
|
||||
ATTEN3(a3v, f3, x0, y0, z0, cx3, cy3, cz3, r3, r3lo, r2, r2);
|
||||
CLAMP80(c0, a0v, a1v, a2v, a3v);
|
||||
ATTEN(a0v, f0, x1, y1, z1, cx0, cy0, cz0, r0, r0lo);
|
||||
ATTEN(a1v, f1, x1, y1, z1, cx1, cy1, cz1, r1, r1lo);
|
||||
ATTEN(a2v, f2, x1, y1, z1, cx2, cy2, cz2, r2, r2lo);
|
||||
ATTEN3(a3v, f3, x1, y1, z1, cx3, cy3, cz3, r3, r3lo, r3, r3);
|
||||
CLAMP80S(c1, a0v, a1v, a2v, a3v);
|
||||
ATTEN(a0v, f0, x2, y2, z2, cx0, cy0, cz0, r0, r0lo);
|
||||
ATTEN(a1v, f1, x2, y2, z2, cx1, cy1, cz1, r1, r1lo);
|
||||
ATTEN(a2v, f2, x2, y2, z2, cx2, cy2, cz2, r2, r2lo);
|
||||
ATTEN3(a3v, f3, x2, y2, z2, cx3, cy3, cz3, r3, r3lo, r2, r3);
|
||||
CLAMP80(c2, a0v, a1v, a2v, a3v);
|
||||
|
||||
*(u32 *)&((PolyGT3 *)pkt)->x0 = *(u32 *)&tmpxy[0];
|
||||
*(u32 *)&((PolyGT3 *)pkt)->x1 = *(u32 *)&tmpxy[1];
|
||||
*(u32 *)&((PolyGT3 *)pkt)->x2 = *(u32 *)&tmpxy[2];
|
||||
tp = (u32 *)prim->w0;
|
||||
cb = 0x34000000;
|
||||
rgbw = (c0 | cb) | (c0 << 8) | (c0 << 16);
|
||||
((PolyGT3 *)pkt)->rgb0 = rgbw;
|
||||
rgbw = (c1 | cb) | (c1 << 8) | (c1 << 16);
|
||||
((PolyGT3 *)pkt)->rgb1 = rgbw;
|
||||
__asm__ __volatile__ ("" :: "r" (c1));
|
||||
rgbw = (c2 | cb) | (c2 << 8) | (c2 << 16);
|
||||
((PolyGT3 *)pkt)->rgb2 = rgbw;
|
||||
((PolyGT3 *)pkt)->uv0 = tp[1];
|
||||
((PolyGT3 *)pkt)->uv1 = tp[2];
|
||||
((PolyGT3 *)pkt)->uv2 = tp[3];
|
||||
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
|
||||
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000;
|
||||
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
|
||||
pkt += 0x28;
|
||||
} else {
|
||||
u32 *otp;
|
||||
u32 rgbw;
|
||||
*(u32 *)&((PolyFT3 *)pkt)->x0 = *(u32 *)&tmpxy[0];
|
||||
*(u32 *)&((PolyFT3 *)pkt)->x1 = *(u32 *)&tmpxy[1];
|
||||
*(u32 *)&((PolyFT3 *)pkt)->x2 = *(u32 *)&tmpxy[2];
|
||||
tp = (u32 *)prim->w0;
|
||||
cb = tp[0] & 0xFF000000;
|
||||
((PolyFT3 *)pkt)->rgbc = cb | 0x101010;
|
||||
((PolyFT3 *)pkt)->uvc0 = tp[1];
|
||||
((PolyFT3 *)pkt)->uvp1 = tp[2];
|
||||
((PolyFT3 *)pkt)->uv2 = tp[3];
|
||||
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
|
||||
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000;
|
||||
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
|
||||
pkt += 0x20;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
case 2:
|
||||
case 3:
|
||||
/* ---------------- QUAD (FT4 / GT4) ---------------- */
|
||||
gte_stsxy3c(&tmpxy[0]);
|
||||
gte_ldv0(vd);
|
||||
gte_rtps();
|
||||
if (tmpxy[0].vx > tmpxy[1].vx) { mx = tmpxy[0].vx; mn = tmpxy[1].vx; }
|
||||
else { mn = tmpxy[0].vx; mx = tmpxy[1].vx; }
|
||||
if (tmpxy[2].vx > mx) mx = tmpxy[2].vx;
|
||||
else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx;
|
||||
if (tmpxy[0].vy > tmpxy[1].vy) { my = tmpxy[0].vy; mny = tmpxy[1].vy; }
|
||||
else { mny = tmpxy[0].vy; my = tmpxy[1].vy; }
|
||||
if (tmpxy[2].vy > my) my = tmpxy[2].vy;
|
||||
else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy;
|
||||
gte_stflg(&g.flag);
|
||||
if (!(g.flag & 0x7F85E000)) {
|
||||
gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3);
|
||||
gte_stsxy((long *)&((PolyFT4 *)pkt)->x3);
|
||||
if (((PolyFT4 *)pkt)->x3 < mn) mn = ((PolyFT4 *)pkt)->x3;
|
||||
else if (mx < ((PolyFT4 *)pkt)->x3) mx = ((PolyFT4 *)pkt)->x3;
|
||||
if (mx >= -0xA0 && mn < 0xA1) {
|
||||
if (((PolyFT4 *)pkt)->y3 < mny) mny = ((PolyFT4 *)pkt)->y3;
|
||||
else if (my < ((PolyFT4 *)pkt)->y3) my = ((PolyFT4 *)pkt)->y3;
|
||||
if (my >= -0x6E && mny < 0x6F) {
|
||||
s32 za, zb;
|
||||
zb = g.sz2;
|
||||
if (zb < g.sz3) zb = g.sz3;
|
||||
za = g.sz0;
|
||||
if (za < g.sz1) za = g.sz1;
|
||||
if (za < zb) za = zb;
|
||||
g.opz = za;
|
||||
|
||||
f3 = 0; f2 = 0; f1 = 0; f0 = 0;
|
||||
|
||||
vw = *(u32 *)va;
|
||||
vzw = *(u32 *)(va + 4);
|
||||
x0 = vw; y0 = vw >> 16; z0 = vzw;
|
||||
BOXTEST(f0, x0, y0, z0, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
|
||||
BOXTEST(f1, x0, y0, z0, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
|
||||
BOXTEST(f2, x0, y0, z0, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
|
||||
BOXTEST(f3, x0, y0, z0, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
|
||||
vw = *(u32 *)vb;
|
||||
vzw = *(u32 *)(vb + 4);
|
||||
x1 = vw; y1 = vw >> 16; z1 = vzw;
|
||||
BOXTEST(f0, x1, y1, z1, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
|
||||
BOXTEST(f1, x1, y1, z1, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
|
||||
BOXTEST(f2, x1, y1, z1, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
|
||||
BOXTEST(f3, x1, y1, z1, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
|
||||
vw = *(u32 *)vc;
|
||||
vzw = *(u32 *)(vc + 4);
|
||||
x2 = vw; y2 = vw >> 16; z2 = vzw;
|
||||
BOXTEST(f0, x2, y2, z2, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
|
||||
BOXTEST(f1, x2, y2, z2, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
|
||||
BOXTEST(f2, x2, y2, z2, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
|
||||
BOXTEST(f3, x2, y2, z2, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
|
||||
vw = *(u32 *)vd;
|
||||
vzw = *(u32 *)(vd + 4);
|
||||
x3 = vw; y3 = vw >> 16; z3 = vzw;
|
||||
BOXTEST(f0, x3, y3, z3, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
|
||||
BOXTEST(f1, x3, y3, z3, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
|
||||
BOXTEST(f2, x3, y3, z3, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
|
||||
BOXTEST(f3, x3, y3, z3, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
|
||||
|
||||
if (f0 | f1 | f2 | f3) {
|
||||
u32 *otp;
|
||||
u32 rgbw;
|
||||
ATTEN(a0v, f0, x0, y0, z0, cx0, cy0, cz0, r0, r0lo);
|
||||
ATTEN(a1v, f1, x0, y0, z0, cx1, cy1, cz1, r1, r1lo);
|
||||
ATTEN(a2v, f2, x0, y0, z0, cx2, cy2, cz2, r2, r2lo);
|
||||
ATTEN3(a3v, f3, x0, y0, z0, cx3, cy3, cz3, r3, r3lo, r3, r3);
|
||||
CLAMP80(c0, a0v, a1v, a2v, a3v);
|
||||
ATTEN(a0v, f0, x1, y1, z1, cx0, cy0, cz0, r0, r0lo);
|
||||
ATTEN(a1v, f1, x1, y1, z1, cx1, cy1, cz1, r1, r1lo);
|
||||
ATTEN(a2v, f2, x1, y1, z1, cx2, cy2, cz2, r2, r2lo);
|
||||
ATTEN3W(a3v, a2v, f3, x1, y1, z1, cx3, cy3, cz3, r3, r3lo, r3, r3);
|
||||
CLAMP80S(c1, a0v, a1v, a2v, a3v);
|
||||
ATTEN(a0v, f0, x2, y2, z2, cx0, cy0, cz0, r0, r0lo);
|
||||
ATTEN(a1v, f1, x2, y2, z2, cx1, cy1, cz1, r1, r1lo);
|
||||
ATTEN(a2v, f2, x2, y2, z2, cx2, cy2, cz2, r2, r2lo);
|
||||
ATTEN3(a3v, f3, x2, y2, z2, cx3, cy3, cz3, r3, r3lo, r3, r3);
|
||||
CLAMP80(c2, a0v, a1v, a2v, a3v);
|
||||
ATTEN(a0v, f0, x3, y3, z3, cx0, cy0, cz0, r0, r0lo);
|
||||
ATTEN(a1v, f1, x3, y3, z3, cx1, cy1, cz1, r1, r1lo);
|
||||
ATTEN(a2v, f2, x3, y3, z3, cx2, cy2, cz2, r2, r2lo);
|
||||
ATTEN3(a3v, f3, x3, y3, z3, cx3, cy3, cz3, r3, r3lo, r3, r3);
|
||||
CLAMP80(c3, a0v, a1v, a2v, a3v);
|
||||
|
||||
*(u32 *)&((PolyGT4 *)pkt)->x0 = *(u32 *)&tmpxy[0];
|
||||
*(u32 *)&((PolyGT4 *)pkt)->x1 = *(u32 *)&tmpxy[1];
|
||||
*(u32 *)&((PolyGT4 *)pkt)->x2 = *(u32 *)&tmpxy[2];
|
||||
gte_stsxy((long *)&((PolyGT4 *)pkt)->x3);
|
||||
tp = (u32 *)prim->w0;
|
||||
cb = 0x3C000000;
|
||||
rgbw = c0 | cb; rgbw |= c0 << 8; rgbw |= c0 << 16;
|
||||
((PolyGT4 *)pkt)->rgb0 = rgbw;
|
||||
rgbw = c1 | cb; rgbw |= c1 << 8; rgbw |= c1 << 16;
|
||||
((PolyGT4 *)pkt)->rgb1 = rgbw;
|
||||
rgbw = (c2 | cb) | (c2 << 8) | (c1 << 16);
|
||||
((PolyGT4 *)pkt)->rgb2 = rgbw;
|
||||
rgbw = (c3 | cb) | (c3 << 8) | (c1 << 16);
|
||||
((PolyGT4 *)pkt)->rgb3 = rgbw;
|
||||
((PolyGT4 *)pkt)->uv0 = tp[1];
|
||||
((PolyGT4 *)pkt)->uv1 = tp[2];
|
||||
uvw = tp[3];
|
||||
((PolyGT4 *)pkt)->uv2 = uvw;
|
||||
((PolyGT4 *)pkt)->uv3 = uvw >> 16;
|
||||
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
|
||||
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0xC000000;
|
||||
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
|
||||
pkt += 0x34;
|
||||
} else {
|
||||
u32 *otp;
|
||||
u32 rgbw;
|
||||
*(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0];
|
||||
*(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1];
|
||||
*(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2];
|
||||
tp = (u32 *)prim->w0;
|
||||
cb = tp[0] & 0xFF000000;
|
||||
((PolyFT4 *)pkt)->rgbc = cb | 0x101010;
|
||||
((PolyFT4 *)pkt)->uvc0 = tp[1];
|
||||
((PolyFT4 *)pkt)->uvp1 = tp[2];
|
||||
uvw = tp[3];
|
||||
((PolyFT4 *)pkt)->uv2 = uvw;
|
||||
((PolyFT4 *)pkt)->uv3 = uvw >> 16;
|
||||
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
|
||||
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000;
|
||||
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
|
||||
pkt += 0x28;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
D_800A5E60 = pkt;
|
||||
}
|
||||
|
||||
|
||||
|
||||
extern s32 D_8012704C;
|
||||
|
||||
Reference in New Issue
Block a user