feat(phase-29): BEHEMOTH func_8017BF14 CLOSED — 45 -> 0 (4,763 ins, the largest match yet)

- 45 -> 37 -> 33 -> 21 -> 11 -> 3 -> 2 -> 0, reproduced 3x from independent work dirs. Verified
  independently before believing it (R14): match_one MATCH (4763 ins), then harvest_verify
  --binary ov_SC03_116 BYTE-IDENTICAL, then R22 clean-fleet 140 passed, 0 failed of 140.
  distinct-code 3,838,143 -> 3,842,906 = 68.1% -> 68.2%. instr 80.5%. Agent was interrupted by a
  weekly API limit and RESUMED FROM ITS TRANSCRIPT -- its round-2 harness survived, nothing was
  re-derived.
- §80 THE PROCESS CORRECTION, worth more than the match: A DO-NOT-RE-BUY ENTRY IS SCOPED TO ITS
  BASE, NOT TO THE FUNCTION. Three of round 1's ~40 measured negatives INVERTED on round 2's
  base -- the same edit (qsingle23) measured 1,040 mismatched on the 45-base and 11 on the
  21-base. Re-testing the round-1 negative list cost ~20s and produced THREE of the seven winning
  levers. Such a table records (edit, base) -> result, NOT edit -> useless; after any lever that
  moves the base materially, RE-RUN THE NEGATIVE LIST. This retroactively qualifies every
  do-not-re-buy table in the cookbook (§45, §60b, §75a, §76, §78, §79). Concrete: round 1 measured
  "removing the va->$t2 pin costs 4% elsewhere" => keep the pin; on a base with c0..c3 at function
  scope, removing those pins is worth 21->13. Same experiment, opposite conclusion.
- MY FLAGGED "#1 MOVE" LOST, and the failure is the finding. I briefed variable REUSE (§45-A /
  RC-14) as the top lever because it took func_8017F510 from 97->10. Swept in full here: EVERY
  merge lost, 43-3294 across 8 merges. Reason: the TRI and QUAD grants did not differ by RANK but
  by IDENTITY -- two independent allocno sets, and re-ranking inside one set cannot fix a two-set
  problem. Diagnose ranking-vs-identity before reaching for a merge. The actual fix (c0..c3 at
  FUNCTION scope, 33->21) was read off the two matched relatives (b5:310, b4:338) and confirmed
  against the target -- the 4th time today that reading a matched relative beat the clever lever.
- PIN'S HIDDEN COST, cited: combine_regs' hard-register branch (local-alloc.c:1795, reached from
  :1295 with already_dead==0) records the pinned reg in qty_phys_sugg UNCONDITIONALLY -- no death
  guard. A pin invites local-alloc to tie producer chains into it. New cure R7: a zero-byte
  __asm__ ref keeping the pinned value live past the temp so find_free_reg can't honour the
  suggestion -- closed the last 2 ins (c1->$a0 is uniquely load-bearing; every alternative pin
  lost 64 ins).
- §78's attribution primitive RUN and REPRODUCED: under -fno-schedule-insns, -fno-schedule-insns2
  and both, order unchanged => the rgb transposition was never a sched.c decision.
- Cold-start economics complete: round 1 = decode + exact length + exact frame + 99.06%; round 2 =
  the last 45, and cheaper. Budget TWO passes at this size. 5th source copy-paste artefact found.
This commit is contained in:
Drew T
2026-07-25 20:29:10 -06:00
parent bd768d8a4e
commit 8b828f1ea6
9 changed files with 2885 additions and 8 deletions
+10
View File
@@ -0,0 +1,10 @@
#!/usr/bin/env python3
"""Splice a new dossier header (lines 4..135 of the b1-derived draft) into the
winning round-2 draft. usage: bf14_hdr.py <src.c> <hdr.txt> <out.c>"""
import sys
src = open(sys.argv[1]).read().splitlines(True)
hdr = open(sys.argv[2]).read()
assert src[3].startswith('/* ====='), src[3]
assert src[134].rstrip().endswith('=== */'), src[134]
open(sys.argv[3], 'w').write(''.join(src[:3]) + hdr + ''.join(src[135:]))
print('wrote', sys.argv[3])
+696
View File
@@ -0,0 +1,696 @@
#!/usr/bin/env python3
"""Round-2 lever generator for func_8017BF14.
Same contract as bf14_mk.py: EVERY transformation asserts it actually applied,
so a "neutral" reading can never be a silent no-op.
usage: bf14_mk2.py <base.c> <out.c> lever [lever ...]
"""
import sys, re
LEVERS = {}
def lever(fn):
LEVERS[fn.__name__] = fn
return fn
def _one(src, old, new, n=1):
c = src.count(old)
assert c == n, 'expected %d of %r, found %d' % (n, old, c)
return src.replace(old, new)
def _all(src, old, new, n):
c = src.count(old)
assert c == n, 'expected %d of %r, found %d' % (n, old, c)
return src.replace(old, new)
# --------------------------------------------------------------------------
# UNPIN levers -- the b1 base carries four pins baked in.
# --------------------------------------------------------------------------
@lever
def unpin_va(src):
return _one(src, ' register u8 *va __asm__("$10");\n u8 *vb, *vc;\n',
' u8 *va, *vb, *vc;\n')
@lever
def unpin_w(src):
return _one(src, ' register u32 w __asm__("$5");\n s32 code;\n',
' u32 w;\n s32 code;\n')
@lever
def unpin_f0(src):
return _one(src, ' register s32 f0 __asm__("$19");\n s32 f1, f2, f3;\n',
' s32 f0, f1, f2, f3;\n')
@lever
def unpin_c1(src):
return _all(src, 'register s32 c1 __asm__("$4"); s32 c0, c2, c3;',
's32 c0, c1, c2, c3;', 2)
# --- re-pin at other registers -------------------------------------------
def _repin_c(src, spec):
"""spec like 'c0=12,c2=10' -> pin those, leave the rest plain."""
want = dict(kv.split('=') for kv in spec.split(','))
out = []
for nm in ('c0', 'c1', 'c2', 'c3'):
if nm in want:
out.append('register s32 %s __asm__("$%s");' % (nm, want[nm]))
plain = [nm for nm in ('c0', 'c1', 'c2', 'c3') if nm not in want]
if plain:
out.append('s32 %s;' % ', '.join(plain))
new = ' '.join(out)
for old in ('register s32 c1 __asm__("$4"); s32 c0, c2, c3;',
's32 c0, c1, c2, c3;'):
if src.count(old) in (1, 2):
return src.replace(old, new)
raise AssertionError('no c-decl found')
# --------------------------------------------------------------------------
# rgb emit-word FORM levers
# --------------------------------------------------------------------------
_CHAIN_Q = [
('rgbw = c0 | cb; rgbw |= c0 << 8; rgbw |= c0 << 16;',
'rgbw = (c0 | cb) | (c0 << 8) | (c0 << 16);'),
('rgbw = c1 | cb; rgbw |= c1 << 8; rgbw |= c1 << 16;',
'rgbw = (c1 | cb) | (c1 << 8) | (c1 << 16);'),
('rgbw = c2 | cb; rgbw |= c2 << 8; rgbw |= c1 << 16;',
'rgbw = (c2 | cb) | (c2 << 8) | (c1 << 16);'),
('rgbw = c3 | cb; rgbw |= c3 << 8; rgbw |= c1 << 16;',
'rgbw = (c3 | cb) | (c3 << 8) | (c1 << 16);'),
]
@lever
def qsingle(src):
"""QUAD lit arm: revert the 3-statement accumulator to ONE expression."""
for a, b in _CHAIN_Q:
src = _one(src, a, b)
return src
@lever
def qsingle23(src):
"""QUAD lit arm: single-expression for rgb2/rgb3 ONLY (keep 0/1 chained)."""
for a, b in _CHAIN_Q[2:]:
src = _one(src, a, b)
return src
@lever
def qsingle01(src):
for a, b in _CHAIN_Q[:2]:
src = _one(src, a, b)
return src
_RGBW_RE = re.compile(
r'rgbw = (?P<e>[^;]+?);(?P<mid>\s*(?:rgbw \|= [^;]+;\s*)*)'
r'\(\(PolyGT(?P<n>[34]) \*\)pkt\)->rgb(?P<k>\d) = rgbw;')
def _direct(src, poly):
"""collapse `rgbw = ...; rgbw |= ...; pkt->rgbN = rgbw;` into one store."""
hits = [0]
def rep(m):
if m.group('n') != poly:
return m.group(0)
expr = m.group('e')
for extra in re.findall(r'rgbw \|= ([^;]+);', m.group('mid')):
expr = '(%s) | %s' % (expr, extra)
hits[0] += 1
return '((PolyGT%s *)pkt)->rgb%s = %s;' % (poly, m.group('k'), expr)
out = _RGBW_RE.sub(rep, src)
assert hits[0] == int(poly), 'direct(GT%s): %d sites' % (poly, hits[0])
return out
@lever
def qdirect(src):
"""QUAD lit arm: no rgbw temp at all -- store the expression directly.
Each rgb word then becomes a 1-death LOCAL temp, so local-alloc places
them independently and they can alternate $v0/$v1 the way the target does."""
return _direct(src, '4')
@lever
def tdirect2(src):
"""TRI lit arm: same, no rgbw temp."""
return _direct(src, '3')
# --------------------------------------------------------------------------
# residual (c) part 1: the UNLIT rgbc word.
# target: and $a1,$v0,$a2 / or $v1,$a1,$v1 / sw $v1
# mine : and $a1,$v0,$a2 / or $a1,$a1,$v1 / sw $a1
# `cb` is a function-scope global allocno ($a1). `cb |= 0x101010` writes it
# in place. The target instead stores `cb | 0x101010` as a 1-death LOCAL
# temp, which combine_regs ties to the DYING constant register ($v1).
# --------------------------------------------------------------------------
@lever
def cb_expr(src):
"""unlit arms: `pkt->rgbc = cb | 0x101010;` (drop the `cb |=` statement)."""
out, n = re.subn(
r'cb = tp\[0\] & 0xFF000000;\s*\n\s*cb \|= 0x101010;\s*\n(\s*)'
r'\(\(PolyFT(\d) \*\)pkt\)->rgbc = cb;',
lambda m: ('cb = tp[0] & 0xFF000000;\n%s((PolyFT%s *)pkt)->rgbc'
' = cb | 0x101010;' % (m.group(1), m.group(2))), src)
assert n == 2, n
return out
@lever
def cb_expr_one(src):
"""unlit arms: the whole thing as ONE expression, no `cb` at all."""
out, n = re.subn(
r'cb = tp\[0\] & 0xFF000000;\s*\n\s*cb \|= 0x101010;\s*\n(\s*)'
r'\(\(PolyFT(\d) \*\)pkt\)->rgbc = cb;',
lambda m: ('((PolyFT%s *)pkt)->rgbc = (tp[0] & 0xFF000000) | 0x101010;'
% m.group(2)), src)
assert n == 2, n
return out
_CHAIN_T = [
('rgbw = (c0 | cb) | (c0 << 8) | (c0 << 16);',
'rgbw = c0 | cb; rgbw |= c0 << 8; rgbw |= c0 << 16;'),
('rgbw = (c1 | cb) | (c1 << 8) | (c1 << 16);',
'rgbw = c1 | cb; rgbw |= c1 << 8; rgbw |= c1 << 16;'),
('rgbw = (c2 | cb) | (c2 << 8) | (c2 << 16);',
'rgbw = c2 | cb; rgbw |= c2 << 8; rgbw |= c2 << 16;'),
]
@lever
def tchain(src):
"""TRI lit arm: 3-statement accumulator (base is single-expression)."""
for a, b in _CHAIN_T:
src = _one(src, a, b)
return src
@lever
def tdirect(src):
"""TRI lit arm: store the expression directly, no rgbw."""
for k in range(3):
old = 'rgbw = (c%d | cb) | (c%d << 8) | (c%d << 16);' % (k, k, k)
assert src.count(old) == 1, old
expr = old.split('= ', 1)[1].rstrip(';')
src = src.replace(old, '')
st = '((PolyGT3 *)pkt)->rgb%d = rgbw;' % k
assert src.count(st) == 1, st
src = src.replace(st, '((PolyGT3 *)pkt)->rgb%d = %s;' % (k, expr))
return src
@lever
def rgbw_fn(src):
"""ONE function-scope `u32 rgbw;` (matched-relative style) instead of per-arm."""
n = src.count(' u32 rgbw;\n')
m = src.count(' u32 rgbw;\n')
assert n + m == 4, (n, m)
src = src.replace(' u32 rgbw;\n', '')
src = src.replace(' u32 rgbw;\n', '')
return _one(src, ' u32 cb;\n', ' u32 cb;\n u32 rgbw;\n')
# --------------------------------------------------------------------------
# residual (a): named producer-offset temps.
# --------------------------------------------------------------------------
_PROD = """ w = prim->w1;
va = vtx + (w & 0xFFFF);
vb = vtx + (w >> 16);
w = prim->w2;
vc = vtx + (w & 0xFFFF);
w = w >> 16;
"""
@lever
def prod1(src):
"""ONE shared offset variable `vo` for all 3 head offsets -> 3 deaths."""
new = """ w = prim->w1;
vo = w & 0xFFFF;
va = vtx + vo;
vo = w >> 16;
vb = vtx + vo;
w = prim->w2;
vo = w & 0xFFFF;
vc = vtx + vo;
w = w >> 16;
"""
src = _one(src, _PROD, new)
return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n')
@lever
def prod2(src):
"""TWO offset variables: `vo` (masked, 2 deaths) and `vs` (shifted, 1)."""
new = """ w = prim->w1;
vo = w & 0xFFFF;
va = vtx + vo;
vb = vtx + (w >> 16);
w = prim->w2;
vo = w & 0xFFFF;
vc = vtx + vo;
w = w >> 16;
"""
src = _one(src, _PROD, new)
return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n')
@lever
def prod3(src):
"""`vo` shared by the head offsets AND the vd offset (4 deaths)."""
src = prod1(src)
return _one(src, 'vd = vtx + (w & 0xFFF8);',
'vo = w & 0xFFF8; vd = vtx + vo;')
@lever
def prodvd(src):
"""only the vd offset gets a named temp (shared with nothing)."""
src = _one(src, 'vd = vtx + (w & 0xFFF8);',
'vo = w & 0xFFF8; vd = vtx + vo;')
return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n')
@lever
def prodswap(src):
"""commute the producer adds: (w & 0xFFFF) + vtx."""
new = """ w = prim->w1;
va = (w & 0xFFFF) + vtx;
vb = (w >> 16) + vtx;
w = prim->w2;
vc = (w & 0xFFFF) + vtx;
w = w >> 16;
"""
return _one(src, _PROD, new)
# --------------------------------------------------------------------------
# residual (b): variable REUSE merges (RC-14 MERGE / cookbook 45-A).
# The four colours are simultaneously live, so they cannot merge with each
# other -- merge them with values that are DEAD by then instead.
# --------------------------------------------------------------------------
def _arm(src, which):
"""return (start,end) slice of the TRI or QUAD lit arm."""
if which == 'tri':
a = src.index('/* ---------------- TRI')
b = src.index('case 2:')
else:
a = src.index('/* ---------------- QUAD')
b = src.index('D_800A5E60 = pkt;')
return a, b
def _merge(src, victim, survivor, which, ndecl):
"""rename `victim` -> `survivor` inside one arm, and drop victim's decl."""
a, b = _arm(src, which)
seg = src[a:b]
new, n = re.subn(r'\b%s\b' % victim, survivor, seg)
assert n == ndecl, 'merge %s->%s in %s: %d hits' % (victim, survivor, which, n)
return src[:a] + new + src[b:]
def _mrg(src, which, victim, survivor, dropdecl):
a, b = _arm(src, which)
seg = src[a:b]
seg2 = seg.replace(*dropdecl)
assert seg2 != seg, 'decl %r not found in %s arm' % (dropdecl[0], which)
seg2, n = re.subn(r'\b%s\b' % victim, survivor, seg2)
assert n > 0, 'no %s in %s arm' % (victim, which)
return src[:a] + seg2 + src[b:]
@lever
def merge_za_c0_t(src):
"""TRI: the max-z temp `za` and `c0` never overlap -> one variable."""
return _mrg(src, 'tri', 'za', 'c0', ('s32 za, zb;', 's32 zb;'))
@lever
def merge_za_c0_q(src):
return _mrg(src, 'quad', 'za', 'c0', ('s32 za, zb;', 's32 zb;'))
@lever
def merge_zb_c1_q(src):
return _mrg(src, 'quad', 'zb', 'c1', ('s32 za, zb;', 's32 za;'))
@lever
def merge_zb_c0_q(src):
return _mrg(src, 'quad', 'zb', 'c0', ('s32 za, zb;', 's32 za;'))
def _mrg_tail(src, which, anchor, victim, survivor, decls):
a, b = _arm(src, which)
seg = src[a:b]
i = seg.index(anchor)
head, tail = seg[:i], seg[i:]
tail2, n = re.subn(r'\b%s\b' % victim, survivor, tail)
assert n > 0, (victim, n)
for old, new in decls:
if old in head:
head = head.replace(old, new)
break
else:
raise AssertionError('no c-decl in %s arm' % which)
return src[:a] + head + tail2 + src[b:]
@lever
def merge_f0_c3_q(src):
"""QUAD: f0..f3 are dead once the last ATTEN3 has run -> f0 doubles as c3."""
return _mrg_tail(src, 'quad', 'CLAMP80(c3', 'c3', 'f0',
[('s32 c0, c2, c3;', 's32 c0, c2;'),
('s32 c0, c1, c2, c3;', 's32 c0, c1, c2;')])
@lever
def merge_f0_c2_t(src):
"""TRI: f0..f3 dead after the last ATTEN3 -> f0 doubles as c2."""
return _mrg_tail(src, 'tri', 'CLAMP80(c2', 'c2', 'f0',
[('s32 c0, c2, c3;', 's32 c0, c3;'),
('s32 c0, c1, c2, c3;', 's32 c0, c1, c3;')])
@lever
def merge_f1_c2_t(src):
return _mrg_tail(src, 'tri', 'CLAMP80(c2', 'c2', 'f1',
[('s32 c0, c2, c3;', 's32 c0, c3;'),
('s32 c0, c1, c2, c3;', 's32 c0, c1, c3;')])
@lever
def merge_f1_c3_q(src):
return _mrg_tail(src, 'quad', 'CLAMP80(c3', 'c3', 'f1',
[('s32 c0, c2, c3;', 's32 c0, c2;'),
('s32 c0, c1, c2, c3;', 's32 c0, c1, c2;')])
# --------------------------------------------------------------------------
# residual (b2): the CLAMP80 sum accumulator.
# target: addu $v0,.. addu $v0,.. addu $v0,.. addiu <c>,$v0,0x10
# mine : the whole chain is tied INTO the pinned c1 ($a0) by
# combine_regs' `sreg < FIRST_PSEUDO_REGISTER` phys_sugg path.
# fix : give the sum its own NAMED variable with >1 death, so
# local-alloc.c:472 refuses it a qty and combine_regs bails at its
# very first test (`reg_qty[ureg] < 0`).
# --------------------------------------------------------------------------
_CL = ('#define CLAMP80(C, A0, A1, A2, A3) C = (A0) + (A1) + (A2) + (A3)'
' + 0x10; if ((C) > 0x80) C = 0x80\n')
_CLS = (_CL +
'#define CLAMP80S(C, A0, A1, A2, A3) sv = (A0) + (A1) + (A2) + (A3);'
' C = sv + 0x10; if ((C) > 0x80) C = 0x80\n')
@lever
def sumvar_c1(src):
"""shared sum variable `sv` on the two c1 CLAMP80 sites only (2 deaths)."""
src = _one(src, _CL, _CLS)
src = _all(src, 'CLAMP80(c1,', 'CLAMP80S(c1,', 2)
return _one(src, ' u32 cb;\n', ' u32 cb;\n s32 sv;\n')
@lever
def sumvar_all(src):
"""shared sum variable `sv` on ALL seven CLAMP80 sites."""
src = _one(src, _CL, _CLS)
n = src.count('CLAMP80(c')
assert n == 7, n
src = src.replace('CLAMP80(c', 'CLAMP80S(c')
return _one(src, ' u32 cb;\n', ' u32 cb;\n s32 sv;\n')
@lever
def sumvar_q(src):
"""shared sum variable on the four QUAD CLAMP80 sites."""
src = _one(src, _CL, _CLS)
a, b = _arm(src, 'quad')
seg = src[a:b]
n = seg.count('CLAMP80(c')
assert n == 4, n
src = src[:a] + seg.replace('CLAMP80(c', 'CLAMP80S(c') + src[b:]
return _one(src, ' u32 cb;\n', ' u32 cb;\n s32 sv;\n')
@lever
def sumvar_t(src):
"""shared sum variable on the three TRI CLAMP80 sites."""
src = _one(src, _CL, _CLS)
a, b = _arm(src, 'tri')
seg = src[a:b]
n = seg.count('CLAMP80(c')
assert n == 3, n
src = src[:a] + seg.replace('CLAMP80(c', 'CLAMP80S(c') + src[b:]
return _one(src, ' u32 cb;\n', ' u32 cb;\n s32 sv;\n')
# --- more producer-offset partitions -------------------------------------
@lever
def prod_cd(src):
"""`vo` shared by the vc offset and the vd offset (target puts both in $v0)."""
new = """ w = prim->w1;
va = vtx + (w & 0xFFFF);
vb = vtx + (w >> 16);
w = prim->w2;
vo = w & 0xFFFF;
vc = vtx + vo;
w = w >> 16;
"""
src = _one(src, _PROD, new)
src = _one(src, 'vd = vtx + (w & 0xFFF8);', 'vo = w & 0xFFF8; vd = vtx + vo;')
return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n')
@lever
def prod_ad(src):
"""`vo` shared by the va offset and the vd offset."""
new = """ w = prim->w1;
vo = w & 0xFFFF;
va = vtx + vo;
vb = vtx + (w >> 16);
w = prim->w2;
vc = vtx + (w & 0xFFFF);
w = w >> 16;
"""
src = _one(src, _PROD, new)
src = _one(src, 'vd = vtx + (w & 0xFFF8);', 'vo = w & 0xFFF8; vd = vtx + vo;')
return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n')
@lever
def prod_a(src):
"""`vo` on the va offset only, but made multi-death by also carrying vd."""
return prod_ad(src)
@lever
def cdrop3_t(src):
"""TRI arm declares c3 but never uses it -- drop it."""
a, b = _arm(src, 'tri')
seg = src[a:b]
for old, new in (('s32 c0, c2, c3;', 's32 c0, c2;'),
('s32 c0, c1, c2, c3;', 's32 c0, c1, c2;')):
if old in seg:
return src[:a] + seg.replace(old, new) + src[b:]
raise AssertionError('no c-decl in tri arm')
@lever
def cdropzb_t(src):
"""TRI arm declares zb but never uses it -- drop it."""
a, b = _arm(src, 'tri')
seg = src[a:b]
assert 's32 za, zb;' in seg
return src[:a] + seg.replace('s32 za, zb;', 's32 za;') + src[b:]
# --- c declaration ORDER inside the arm ----------------------------------
def _corder(src, order):
plain = [c for c in order]
txt = 's32 %s;' % ', '.join(plain)
for old in ('register s32 c1 __asm__("$4"); s32 c0, c2, c3;',
's32 c0, c1, c2, c3;'):
if src.count(old) == 2:
if 'register' in old:
txt = ('register s32 c1 __asm__("$4"); s32 %s;'
% ', '.join([c for c in order if c != 'c1']))
return src.replace(old, txt)
raise AssertionError('no c-decl')
# --- ref dials -----------------------------------------------------------
def _dial(src, name, which, n=1):
"""insert n zero-byte ref bumps on `name` right after the c-decl of an arm."""
a, b = _arm(src, which)
seg = src[a:b]
m = re.search(r'( *)(register s32 c1 __asm__\("\$4"\); s32 c0, c2, c3;|s32 c0, c1, c2, c3;)\n', seg)
assert m, 'no anchor'
ins = ''.join('%s__asm__ __volatile__ ("" :: "r" (%s));\n' % (m.group(1), name)
for _ in range(n))
seg = seg[:m.end()] + ins + seg[m.end():]
return src[:a] + seg + src[b:]
# --------------------------------------------------------------------------
# residual (b1): the TRI arm's c0/c2 grants.
# The target's TRI grants (c0=$t4, c1=$a0, c2=$t2) are IDENTICAL to its QUAD
# grants -- i.e. c0..c3 are ONE set of function-scope variables shared by
# both arms, exactly as in the matched relatives func_8017D960 /
# func_8017F510. Per-cull-block scope splits them into two independent
# allocno sets, which is why the TRI set drifts.
# --------------------------------------------------------------------------
_CDECLS = ('register s32 c1 __asm__("$4"); s32 c0, c2, c3;',
's32 c0, c1, c2, c3;')
def _cfn(src, anchor, pinned):
for old in _CDECLS:
if src.count(old) == 2:
break
else:
raise AssertionError('no per-arm c-decl')
# drop the two per-arm declarations (whole lines)
out, n = re.subn(r'[ \t]*%s\n' % re.escape(old), '', src)
assert n == 2, n
decl = ('register s32 c1 __asm__("$4"); s32 c0, c2, c3;' if pinned
else 's32 c0, c1, c2, c3;')
assert out.count(anchor) == 1, anchor
return out.replace(anchor, anchor + ' %s\n' % decl)
@lever
def cfn(src):
"""c0..c3 at FUNCTION scope, c1 still pinned to $a0."""
return _cfn(src, ' s32 a0v, a1v, a2v, a3v;\n', True)
@lever
def cfn_np(src):
"""c0..c3 at FUNCTION scope, PIN-FREE (matched-relative style)."""
return _cfn(src, ' s32 a0v, a1v, a2v, a3v;\n', False)
@lever
def cfn_cb(src):
"""c0..c3 at function scope, declared just before `u32 cb;`."""
return _cfn(src, ' u32 uvw;\n', True)
@lever
def cfn_top(src):
"""c0..c3 at function scope, declared early (before the r/lo/hi block)."""
return _cfn(src, ' Part *part;\n', True)
@lever
def cfn_end(src):
"""c0..c3 at function scope, declared LAST."""
return _cfn(src, ' u32 cb;\n', True)
@lever
def cfn_d(src):
"""c0..c3 at function scope, declared right after `s32 d;`."""
return _cfn(src, ' s32 d;\n', True)
@lever
def unpin_c1n(src):
"""drop the c1 pin (function-scope c-decl form, single occurrence)."""
return _one(src, 'register s32 c1 __asm__("$4"); s32 c0, c2, c3;',
's32 c0, c1, c2, c3;')
# --------------------------------------------------------------------------
# residual 3889: one CLAMP80 site sums its attenuations in a different
# ORDER (a0v+a1v+a3v+a2v). Its `sra` therefore lands in a3v's own register
# ($a3) instead of being written in place over a2v's. Another hand-edit
# copy-paste artefact, of the same family as report-1 artefacts 1-4.
# --------------------------------------------------------------------------
_CLAMP_RE = re.compile(r'CLAMP80S?\((c\d), (a0v), (a1v), (a2v), (a3v)\);')
def _clampswap(src, which, sites):
a, b = _arm(src, which)
seg = src[a:b]
hits = [-1]
def rep(m):
hits[0] += 1
if hits[0] not in sites:
return m.group(0)
return m.group(0).replace('a2v, a3v', 'a3v, a2v')
seg2 = _CLAMP_RE.sub(rep, seg)
assert seg2 != seg, 'clampswap %s %s: no site changed' % (which, sites)
return src[:a] + seg2 + src[b:]
# --------------------------------------------------------------------------
# residual 3889 -- a FIFTH copy-paste artefact.
# At ONE of the seven ATTEN3 sites the y-axis `else if` branch accumulates
# into a2v instead of a3v, while the y-axis KILL branch still says a3v.
# Byte-evidence in the target:
# 3875 addu $a2,$zero,$zero <- kill branch writes a3v ($a2) [matches]
# 3889 sra $a3,$s2,7 <- else branch writes a2v ($a3) [differs]
# Cost: zero instructions. Same shape as report-1 artefacts 1-4.
# --------------------------------------------------------------------------
_A3 = """#define ATTEN3(A, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \\
"""
_A3W = """#define ATTEN3W(A, AW, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \\
A = 0; \\
if (F) { \\
d = (X) - (CX); if (d < 0) d = (CX) - (X); \\
if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \\
d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \\
if ((RZ) < d) A = 0; \\
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \\
d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \\
if ((RY) < d) A = 0; \\
else if ((RLO) < d) AW = (A * (((R) - d) / 4)) >> 7; \\
}
"""
_A3CALL = re.compile(r'ATTEN3\((a\dv), (f\d),')
def _atten3w(src, which, sites, dst):
if 'ATTEN3W' not in src:
src = _one(src, _A3, _A3W + _A3)
a, b = _arm(src, which)
seg = src[a:b]
hits = [-1]
def rep(m):
hits[0] += 1
if hits[0] not in sites:
return m.group(0)
return 'ATTEN3W(%s, %s, %s,' % (m.group(1), dst, m.group(2))
seg2 = _A3CALL.sub(rep, seg)
assert seg2 != seg, 'atten3w %s %s: no site changed' % (which, sites)
return src[:a] + seg2 + src[b:]
# --------------------------------------------------------------------------
# last residual (2258/2259): `c1 << 16` in the TRI arm.
# c1 is pinned, so it is a HARD reg from the start; it DIES at the `sll`, so
# combine_regs takes its `sreg < FIRST_PSEUDO_REGISTER` branch and records
# $a0 in qty_phys_sugg for the sll's temp -> the temp lands in $a0 and the
# shift is done in place. The target keeps the temp in $v0.
# Cure family: make the temp NOT a 1-death local (a named var used in both
# arms), or move the pin off c1 onto a colour whose grant we already match.
# --------------------------------------------------------------------------
@lever
def hivar(src):
"""named `hi` for the `c1 << 16` term in BOTH arms -> 2 deaths, no qty."""
out, n = re.subn(r'\(c1 << 16\)', 'hi', src)
assert n >= 1, n
out, m = re.subn(r'rgbw \|= c1 << 16;', 'rgbw |= hi;', out)
tri = out.index('/* ---------------- TRI')
quad = out.index('/* ---------------- QUAD')
# one `hi = c1 << 16;` immediately before the first rgb store of each arm
def ins(s, marker):
i = s.index(marker)
j = s.rindex('\n', 0, s.rindex('rgbw', 0, i) if 'rgbw' in s[:i] else i)
return s
for anchor in ('rgbw = (c1 | cb)', 'rgbw = c1 | cb'):
while anchor in out:
k = out.index(anchor)
ln = out.rindex('\n', 0, k) + 1
pad = out[ln:k]
out = out[:ln] + pad + 'hi = c1 << 16;\n' + out[ln:]
k2 = out.index(anchor, ln + len(pad) + 14)
out = out[:k2] + anchor.replace('rgbw', 'rgbw$') + out[k2 + len(anchor):]
out = out.replace('rgbw$', 'rgbw')
assert 'hi = c1 << 16;' in out
return _one(out, ' u32 cb;\n', ' u32 cb;\n u32 hi;\n')
@lever
def _noop(src):
return src
def _c1live(src, which, anchor):
"""RC-15 zero-byte ref that keeps the PINNED c1 ($a0) live past the
`sll` of `c1 << 16`. combine_regs records $a0 in qty_phys_sugg
unconditionally (local-alloc.c:1798, no death guard), but find_free_reg
can only honour a suggestion whose hard reg is actually free over the
temp's live range -- so extending c1 past the shift is what refuses it."""
a, b = _arm(src, which)
seg = src[a:b]
assert seg.count(anchor) == 1, (anchor, seg.count(anchor))
k = seg.index(anchor) + len(anchor)
ln = seg.rindex('\n', 0, seg.index(anchor)) + 1
pad = seg[ln:seg.index(anchor)]
seg = seg[:k] + '\n' + pad + '__asm__ __volatile__ ("" :: "r" (c1));' + seg[k:]
return src[:a] + seg + src[b:]
def main():
base, out, levers = sys.argv[1], sys.argv[2], sys.argv[3:]
src = open(base).read()
for lv in levers:
if lv.startswith('RC:'): # RC:c0=12,c2=10
src = _repin_c(src, lv[3:]); continue
if lv.startswith('CO:'): # CO:c2,c0,c1,c3
src = _corder(src, lv[3:].split(',')); continue
if lv.startswith('CL:'): # CL:tri:rgb1
_, wh, tag = lv.split(':')
poly = '3' if wh == 'tri' else '4'
anc = {'rgb0': '((PolyGT%s *)pkt)->rgb0 = rgbw;' % poly,
'rgb1': '((PolyGT%s *)pkt)->rgb1 = rgbw;' % poly,
'rgb2': '((PolyGT%s *)pkt)->rgb2 = rgbw;' % poly,
'uv0': '((PolyGT%s *)pkt)->uv0 = tp[1];' % poly,
'end': 'pkt += 0x%s;' % ('28' if wh == 'tri' else '34')}[tag]
src = _c1live(src, wh, anc); continue
if lv.startswith('AW:'): # AW:quad:1:a2v
_, wh, ix, dst = lv.split(':')
src = _atten3w(src, wh, set(int(x) for x in ix.split(',')), dst); continue
if lv.startswith('CS:'): # CS:quad:1 or CS:tri:0,2
_, wh, ix = lv.split(':')
src = _clampswap(src, wh, set(int(x) for x in ix.split(','))); continue
if lv.startswith('DL:'): # DL:name:tri[:n]
p = lv[3:].split(':')
src = _dial(src, p[0], p[1], int(p[2]) if len(p) > 2 else 1); continue
src = LEVERS[lv](src)
open(out, 'w').write(src)
main()
+15
View File
@@ -0,0 +1,15 @@
#!/bin/bash
# bf14_sw2.sh <base.c> "<tag>|<lever> [lever...]" ... -> one line per variant, parallel
# uses bf14_mk2.py (round-2 levers). Reports raw mismatch count + length.
cd /home/musashi/bfm-decomp
BASE="$1"; shift
run() {
local spec="$1"; local tag="${spec%%|*}"; local lv="${spec#*|}"
local wd=".run/giants/bf14_sw2/$tag"
mkdir -p "$wd"
python3 .run/giants/bf14_mk2.py "$BASE" "$wd/v.c" $lv 2>"$wd/mk.err" || { printf '%-22s :: MK-FAIL %s\n' "$tag" "$(tail -2 $wd/mk.err|tr '\n' ' ')"; return; }
bash .run/giants/bf14_cc.sh "$wd/v.c" "$wd/w" >/dev/null 2>"$wd/cc.err" || { printf '%-22s :: COMPILE-FAIL %s\n' "$tag" "$(tail -3 $wd/cc.err|tr '\n' ' ')"; return; }
printf '%-22s :: %s\n' "$tag" "$(python3 .run/giants/bf14_full.py "$wd/w/t.o" --count)"
}
export -f run; export BASE
printf '%s\n' "$@" | xargs -P 8 -I{} bash -c 'run "$@"' _ {}
+383
View File
@@ -0,0 +1,383 @@
# `func_8017BF14` — behemoth #4, 4,763 ins, `ov_SC03_116` — ROUND 2: **MATCH**
**Session 20 round 2, 2026-07-25.** Round 1 handed over 45/4763 mismatched.
Round 2 closed it.
---
## 1. FINAL NUMBER (measured, `tools/match_one.py`, the CANDIDATE gate)
```
python3 tools/match_one.py func_8017BF14 --c .run/giants/s19_func_8017BF14_b2.c \
--asm-subdir asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C
-> MATCH (4763 ins) func_8017BF14
```
Reproduced **3×** from independent private work dirs. Independent confirmations:
| check | result |
|---|---|
| `bf14_full.py` masked index-wise diff | **0 / 4763 mismatched** |
| `bf14_hist.py` opcode histogram | `len+0 L1=0` |
| `bf14_slots.py` stack-slot census | **127 / 127**, all at the target's offsets |
| compile warnings (`-Wall`) | none |
**The whole-binary SHA1 arbiter (G3/P9) was NOT run** — the task forbade touching
the build tree. `match_one` is the candidate check only; the coordinator's
whole-binary gate is the sole arbiter (G3/P9).
Trajectory, every step measured: **45 → 37 → 33 → 21 → 11 → 3 → 2 → 0.**
---
## 2. PER-RESIDUAL OUTCOME
Round 1 split the 45 into (a) producer temps ≈8, (b) `c0`/`c2` grants ≈15,
(c) quad-lit rgb accumulator ≈20. All three fell. The exact split of the 45,
recounted from the diff, was **(a) 8 + (b) 23 + (c) 14**.
### (a) Prim-word producer temps — **FELL** (8 ins, idx 508–512, 539–542)
Round 1's diagnosis (`combine_regs` tying the chain into the pinned `va`) was
**correct**, and its prescribed cure (named MULTI-death offset variables) does
work — `prod2` broke the `$t2` tie exactly as predicted, byte-visible:
```
base 508 andi $t2,$a1,0xffff 510 addu $t2,$t6,$t2 <- tied, in place
prod2 508 andi $v1,$a1,0xFFFF 510 addu $t2,$t6,$v1 <- MATCHES target
```
But it is a **conservation law, not a fix**: one shared `vo` gives one register
for all its sites, whereas the target uses three distinct temps ($v1, $a0, $v0,
$v0 for the four offsets). `prod2` fixed 508/510 and broke 513/515 — 45 → 45.
Every partition of the four offsets across named variables was swept
(`prod1/2/3`, `prod_ad`, `prod_cd`, `prodvd`, `prodswap`): best is neutral.
**What actually fixed it: R4 — dropping the `va→$t2` and `w→$a1` pins.**
Round 1 measured those pins as worth 4% and the brief said not to re-buy their
removal. That was true *of the round-1 base* and is **false** once `c0..c3` sit
at function scope (R3): on the 21-base, `unpin_va` = 17, `unpin_va unpin_w` =
**13**. This is base-dependence, not a contradiction — and it is the single
most important methodological lesson of the round (§5.1).
### (b) `c0`/`c2` grants — **FELL** (23 ins)
Round 1 pointed at `allocno_compare` order and prescribed a variable-REUSE
merge. The **reuse sweep was run in full and every merge lost** (§4.3): merging
`za`/`zb`/`f0`/`f1` into `c0..c3` scored 43–3294 against a 33/37 base. Reuse is
now a measured dead end on this function.
The real lever was found by **reading the target and the matched relatives**,
which is where round 1 said the value was:
* The target's TRI grants are `c0→$t4, c1→$a0, c2→$t2` — **identical to its
QUAD grants**. Two independently-scoped allocno sets cannot coincide by
chance; one shared set can.
* Both matched relatives declare the colours at **function scope**:
`.run/giants/s19_func_8017D960_b5.c:310` and `s19_func_8017F510_b4.c:338`.
`s32 c0, c1, c2, c3;` at function scope: **33 → 21**. Declaration *position* is
neutral (5 anchors swept, all 21) — consistent with round 1's L4.2 oracle, since
these never take stack slots.
This **reverses round-1's L4**, which put them per-cull-block. L4 was right on
its own base — it was supplying the extra local allocno that spills `r1lo` — but
R1 and R2 supply that pressure now.
A second, separable part of (b) was the c1 **sum accumulator** (9 ins, idx
1876-1879 / 3890-3893), which round 1 had classified with the grants. It is a
pin artefact, and `CLAMP80S` (R1) fixed it: **45 → 37**.
### (c) Quad-lit rgb accumulator — **FELL** (14 ins)
Two independent pieces:
* **Unlit `rgbc` (4 ins, idx 2310/2311, 4699/4700).** `cb` is a function-scope
global allocno, so `cb |= 0x101010` writes it in place ($a1). The target
stores the *expression* `cb | 0x101010`, a 1-death local that `combine_regs`
ties to the **dying constant register** $v1. **37 → 33.**
* **Quad lit rgb2/rgb3 (10 ins, idx 4643-4652).** Reverting those two to the
single-expression form (rgb0/rgb1 keep the 3-statement accumulator) makes the
intermediates 1-death local temps that alternate $v0/$v1 — which is what lets
the target's store of the previous rgb word sit one slot later. **21 → 11.**
All-four = 23, rgb0/rgb1-only = 21, direct-store = 28. The split really is
2-and-2, matching artefact 4 (rgb2/rgb3 are the two that take `c1 << 16`).
### THE ATTRIBUTION PRIMITIVE FOR (c) — run, and decisive
```
-fno-schedule-insns -> my order unchanged
-fno-schedule-insns2 -> my order unchanged
both -> my order unchanged
```
In all three builds `sw v1,-20(t3)` still precedes `or v1,t2,a1`. **The
transposition was never a `sched.c` decision.** With a 3-statement accumulator
the value is pinned to one register, so the next `or` clobbers $v1 and *no*
scheduler could hoist it above the store — the ordering is a consequence of the
register grant. Changing the grant (R5) fixed the order for free. This is
**§78 reproduced exactly**, and it is the second time on this family that a
"scheduling" residual was really a register grant. Do not reason about `sched.c`
on this family until this primitive has been run.
---
## 3. THE ONE MECHANISM BEHIND FOUR OF THE SEVEN LEVERS
R1, R2, R4 and R7 are all the same compiler fact, and it is worth a cookbook
entry because it is the *cost* of the pin technique:
> A `register __asm__` pin makes the variable a **hard register in the RTL from
> the start**. When a 1-death local temp is produced from — or consumed into —
> that hard reg, `local-alloc.c`'s `combine_regs` takes its hard-register branch
> (`local-alloc.c:1795-1820`) and records the pinned register in
> `qty_phys_sugg` for the temp's quantity. **That path has no death guard and no
> cost model — it fires unconditionally.** The temp then lands in the pinned
> register and the operation is performed IN PLACE. An ordinary pseudo never
> gets that suggestion, because anything crossing a basic block has
> `reg_qty == -1` and `combine_regs` bails at its very first test.
Verified in source, `tools/reference/gcc-2.7.2/local-alloc.c`:
* `:472` — a pseudo is local iff `reg_basic_block[i] >= 0 && reg_n_deaths[i] == 1`.
* `:1763` — `combine_regs` returns 0 immediately if `reg_qty[ureg] < 0`.
* `:1795` — `if (ureg < FIRST_PSEUDO_REGISTER) { ... qty_phys_sugg |= ureg; return 0; }`
— the branch that costs us, reached from `block_alloc` at `:1295` with
`already_dead = 0`.
**Three separable cures, all three used in this match:**
| cure | how | used by |
|---|---|---|
| (i) refuse the temp a quantity | give it a NAMED variable with **>1 death**, so `:472` rejects it and `combine_regs` bails at `:1763` | **R1** (`CLAMP80S`/`sv`) |
| (ii) remove the pin | only if the pin is not load-bearing — **re-measure, it is base-dependent** | **R4** (`va`, `w`) |
| (iii) starve the suggestion | keep the pinned value **LIVE past the temp**, so `find_free_reg` cannot honour the suggestion | **R7** (zero-byte ref on `c1`) |
Cure (iii) is new and is the one that closed the function. The last two
instructions were `sll $a0,$a0,16` (mine, in place over the pinned `c1`) vs
`sll $v0,$a0,16` (target). `c1` could not be unpinned — it is what spills
`r1lo`, and dropping it or moving it to any other colour costs −64 length
(measured, §4.5). So instead:
```c
((PolyGT3 *)pkt)->rgb1 = rgbw;
__asm__ __volatile__ ("" :: "r" (c1)); /* RC-15, zero bytes */
```
`c1` is now live past the shift, `$a0` is unavailable, the temp falls to `$v0`.
**2 → 0.** Placing it after the rgb2 store also matches; after `uv0` or at the
arm's end does not (they perturb length).
---
## 4. EVERY LEVER MEASURED THIS ROUND
Metric is the `match_one` **mismatch count** (length is exact throughout, so the
raw count is honest — the round-1 metric trap of §5.3 does not apply). Baselines
are stated per block because the base moved as levers landed.
### 4.1 The winning chain
| # | lever | base → result |
|---|---|---|
| **R1** | `CLAMP80S` — named `sv` sum variable on the **two c1 sites only** | 45 → **37** |
| **R2** | unlit arms: `pkt->rgbc = cb \| 0x101010;` (expression, not `cb \|=`) | 37 → **33** |
| **R3** | **`s32 c0, c1, c2, c3;` at FUNCTION scope** | 33 → **21** |
| **R4** | drop the `va→$t2` and `w→$a1` pins | 21 → 17 → **13** |
| **R5** | quad-lit **rgb2/rgb3 only** revert to single-expression | 21 → **11**; with R4 → **3** |
| **R6** | artefact 5 — `ATTEN3W(a3v, a2v, …)` at the QUAD vertex-1 site | 3 → **2** |
| **R7** | RC-15 zero-byte ref on pinned `c1` after the TRI rgb1 store | 2 → **0** |
### 4.2 Neutral — measured, no effect at all
| lever | base | result |
|---|---|---|
| `prod1` / `prod2` / `prodvd` / `prodswap` (named producer-offset temps) | 45 | 45 (fixes 508/510, breaks 513/515) |
| same four | 21 | 21 |
| `cdrop3_t` (drop the unused `c3` from the TRI arm) | 45 / 37 | 45 / 37 |
| `cdropzb_t` (drop the unused `zb` from the TRI arm) | 45 | 45 |
| `qsingle01` (single-expression for quad rgb0/rgb1) | 21 | 21 |
| `prod2` on top of `sumvar_c1` | 45 | 37 (= R1 alone) |
| declaration POSITION of the function-scope `c0..c3` — 5 anchors: before `a0v`, before `cb`, before `d`, before `part`, last | 33 | **21 at every anchor** |
| `qsingle23 + prodvd` / `+ prodswap` / `+ prod1` / `+ prod2` | 11 | 11 |
### 4.3 The REUSE sweep — run in full, every merge LOST
This was round 1's flagged #1 move. It is now a measured dead end here.
| merge | base | result |
|---|---|---|
| `za` → `c0` (QUAD) | 45 | 55 |
| `za` → `c0` (TRI) | 45 | 3294 |
| `zb` → `c1` (QUAD) | 45 | 55 |
| `zb` → `c0` (QUAD) | 45 / 37 | 51 / 43 |
| `f0` → `c3` (QUAD) | 45 / 37 | 51 / 43 |
| `f0` → `c2` (TRI) | 45 | 49 |
| `f1` → `c2` (TRI) | 45 | 838 |
| `f1` → `c3` (QUAD) | 45 | 841 |
**Why it failed, and this generalises:** a REUSE merge raises the survivor's
`reg_n_refs` to move `allocno_compare` priority. But the TRI/QUAD `c0..c3`
grants did not disagree because of *priority* — they disagreed because the two
arms had **two independent allocno sets** at all. No amount of re-ranking inside
one set can make it agree with a different set; only merging the sets can (R3).
**Diagnose whether two grants differ by RANK or by IDENTITY before reaching for
a ref-count lever.**
### 4.4 rgb emit-word forms
| lever | base | result |
|---|---|---|
| `qsingle` (all four quad words single-expression) | 45 / 37 / 21 | 47 / 39 / 23 |
| `qsingle23` (**rgb2/rgb3 only**) | 45 / 21 | **1040 / 11** ← extreme base-dependence |
| `qsingle01` (rgb0/rgb1 only) | 45 / 21 | 45 / 21 |
| `qdirect` (no `rgbw`, store the expression) | 37 / 21 | 44 / 28 |
| `tdirect2` (TRI, no `rgbw`) | 37 / 21 | 47 / 37 |
| `qdirect + tdirect2` | 37 | 54 |
| `tchain` (TRI 3-statement accumulator) | 45 | 2514 (len 4762) |
| `cb_expr_one` (unlit, drop `cb` entirely) | 37 | 2423 (len 4767) |
| one function-scope `u32 rgbw;` (relative style) | 37 | n/a — 4 per-arm decls, lever refused |
### 4.5 The pins — re-measured on every base
| lever | base | result |
|---|---|---|
| `unpin_va` | 45 / 21 | 174 / **17** |
| `unpin_w` | 45 / 21 | 47 / 23 |
| `unpin_va + unpin_w` | 45 / 21 | 170 / **13** |
| `unpin_f0` | 45 / 21 | 88 / 64 — **f0→$s3 stays** |
| `unpin_c1` | 45 / 21 / 3 | 654 / — / **len 4699 (−64)** |
| all four unpinned | 45 | 789 (the round-1 pin-free fallback) |
| c0..c3 at function scope, **pin-free** | 33 | len 4699 (−64) |
| re-pin `c0→$t4` and/or `c2→$t2`/`c3→$a2` **instead of** c1 | 2 | **len 4699 every time** |
| re-pin `c0→$t4` **plus** c1 | 2 | 14 |
| re-pin `c1 + c2` / `c0+c1+c2` / all four / `c1+c3` | 2 | 60 / 71 / 320 / 307 |
**The `c1→$a0` pin is uniquely load-bearing: it is the register pressure that
spills `r1lo`.** No other colour, and no combination without it, reproduces the
spill. Round-1 §5.2's "pinning c0 and c2 *in addition to* c1 is worse" is
confirmed and extended: pinning them *instead of* c1 does not even preserve the
frame.
### 4.6 Artefact-5 localisation (the `sra $a2` vs `$a3` residual)
| lever | base | result |
|---|---|---|
| `ATTEN3W(a3v, a2v, …)` at QUAD site **1** | 3 | **2** |
| same at QUAD sites 0 / 2 / 3 | 3 | 4 / 4 / 4 |
| same at TRI sites 0 / 1 / 2 | 3 | 4 / 4 / 4 |
| `ATTEN3W(a3v, a1v, …)` / `(a3v, a0v, …)` at QUAD site 1 | 3 | 3 / 3 |
| swap the last two `CLAMP80` args (a2v↔a3v) — **all 7 sites tried one at a time** | 3 | 5 at every site |
The `CLAMP80` argument-order hypothesis is **refuted**: the sum order is
identical (3890-3893 are byte-identical in both), and the target's kill branch
at idx 3875 (`addu $a2,$zero,$zero`) already agrees. Only the y-axis **`else if`
destination** differs — a genuine fifth copy-paste artefact, costing zero
instructions.
### 4.7 Other
| lever | base | result |
|---|---|---|
| `sumvar_all` (shared `sv` on all 7 CLAMP sites) | 45 | 4568, **len 4699** |
| `sumvar_q` / `sumvar_t` (per-arm) | 45 | 41 / 41 |
| `prod3` (`vo` on all four offsets) | 45 / 21 | 47 / 23 |
| `prod_ad` / `prod_cd` | 45 / 21 | 50, 52 / 26, 28 |
| `hivar` (named `hi` for `c1 << 16`, both arms) | 2 | 226 (len 4765) |
| `hivar + unpin_c1` | 2 | 4549 (len 4697) |
| zero-byte `c1` ref after the TRI **uv0** store / at arm **end** | 2 | 2471 (len 4764) / 2446 (len 4766) |
---
## 5. LESSONS TO FEED BACK (cookbook candidates)
### 5.1 A "do-not-re-buy" entry is scoped to the BASE that measured it
Three of round 1's measured, correctly-recorded findings inverted once the base
moved:
| round-1 finding | round-2 measurement |
|---|---|
| L4: `c0..c3` per cull block (52% → 93%) | function scope is **strictly better**, 33 → 21 |
| L9: removing the `va` pin costs 4% | removing `va` **and** `w` is worth 21 → 13 |
| L8: quad rgb 3-statement accumulator is +0.02% | single-expression for rgb2/rgb3 is worth 21 → 11 |
`qsingle23` is the extreme case: **1040 mismatched on the 45-base, 11 on the
21-base** — the same edit, two orders of magnitude apart. None of these were
errors in round 1; they were correct readings of a different base.
**Rule candidate:** a do-not-re-buy table must record *the base it was measured
against*, and any entry measured against a base that has since moved by a
structural lever is **stale, not settled** — re-measure the cheap ones (one
0.28 s probe each) rather than inheriting them. Re-testing the whole round-1
"negative" list on the new base cost about 20 seconds of compute and produced
three of the seven winning levers.
### 5.2 The pin's hidden cost: `combine_regs`' unconditional `qty_phys_sugg`
§3 above, with source citations. Worth its own cookbook section: it explains a
whole *class* of 2-instruction "in-place vs not" residuals, and it gives three
separable cures. Cure (iii) — the zero-byte liveness extension — is new, and it
is how a pin can be kept for its allocation pressure while its tie is refused.
### 5.3 Sibling grants are an IDENTITY oracle, not just a hint
If two code paths in the target show *identical* register grants for
corresponding variables, those variables are **one set of allocnos** — i.e. one
declaration at a scope enclosing both. That is a positive structural inference
from register numbers alone, and it beat an exhaustive scope sweep plus a full
reuse sweep. It also agreed with what the two matched relatives already showed,
which is the round-1 report's own advice (§"MATCHED RELATIVES") paying off again:
**5 of 9 winning levers on the last behemoth, and 2 of 7 here (R3, and R5's
2-and-2 split), came straight off a matched relative or off the target's own
register numbering.**
### 5.4 Run the attribution primitive before any scheduling reasoning
§2. Two for two on this family: an apparent scheduling residual that was a
register grant. Cost: three compiles.
---
## 6. WHAT CHANGED IN THE SOURCE (7 hunks vs `s19_func_8017BF14_b1.c`)
1. `ATTEN3W` macro added (artefact 5, y-axis `else` destination).
2. `CLAMP80S` macro added (named `sv` sum).
3. `va` / `w` pins removed; `c0..c3` (with the `c1` pin) moved to function
scope; `s32 sv;` declared.
4. `CLAMP80(c1, …)` → `CLAMP80S(c1, …)` in both arms.
5. TRI arm: zero-byte `c1` ref after the rgb1 store; unlit `rgbc` written as an
expression.
6. QUAD arm: `ATTEN3W` at the vertex-1 site; rgb2/rgb3 single-expression; unlit
`rgbc` written as an expression.
7. The two per-arm `register s32 c1 …; s32 c0, c2, c3;` declarations removed.
---
## 7. FILES
* `.run/giants/s19_func_8017BF14_b2.c` — **the match**, full updated dossier.
* `.run/giants/s19_func_8017BF14_b1.c` — round-1 draft, 45/4763 (kept).
* `.run/giants/s19_func_8017BF14_b1_pinfree.c` — round-1 pin-free, 789/4763 (kept).
* `.run/giants/bf14_mk2.py` — round-2 lever generator. Same contract as
`bf14_mk.py`: **every transformation asserts it applied**, so a "neutral"
reading can never be a silent no-op. Levers: `unpin_*`, `RC:` (re-pin),
`sumvar_*`, `cb_expr*`, `cfn*`, `qsingle*`, `qdirect`, `tdirect2`, `prod*`,
`merge_*`, `CS:` (CLAMP arg swap), `AW:` (ATTEN3 y-else destination),
`CL:` (zero-byte `c1` liveness), `hivar`.
* `.run/giants/bf14_sw2.sh` — 8-way parallel sweep over `bf14_mk2.py`, reporting
the raw mismatch count.
* `.run/giants/bf14_hdr.py` + `bf14_hdr.txt` — dossier-header splicer.
* `.run/giants/r2/` — the promoted bases: `b2base` 37, `b3base` 33, `b4base` 21,
`b5base` 3, `b6base` 2, `b7base` **0**; `sch_*` the attribution-primitive
builds; `da/` the `-da` RTL dumps of the 45-base.
* Round-1 harness (`bf14_cc/score/probe/full/side/win/hist/ali/slots/…`) used
unchanged.
---
## 8. STATUS FOR THE COORDINATOR
`match_one` reports **MATCH (4763 ins)**. That is the candidate gate only.
**The whole-binary SHA1 byte-gate (G3/P9) has not been run and is the sole
arbiter** — this is not a confirmed match until that is green.
+839
View File
@@ -0,0 +1,839 @@
#include "common.h"
#include "/home/musashi/bfm-decomp/src/shared/engine_types.h"
/* ===========================================================================
* func_8017BF14 -- 4,763 ins, ov_SC03_116 (behemoth #4). *** MATCH ***
*
* STATUS (2026-07-25, session 20 round 2, gcc-2.7.2 pinned triple):
* python3 tools/match_one.py func_8017BF14 --c <this file> \
* --asm-subdir asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C
* -> MATCH (4763 ins) func_8017BF14 [reproduced 3x, private work dirs]
* LENGTH EXACT | OPCODE HISTOGRAM EXACT (L1 = 0) | STACK FRAME EXACT
* (all 127 slots at the target's offsets, frame 0x360)
* masked index-wise diff: 0 / 4763 mismatched.
* `match_one` is the CANDIDATE gate. The whole-binary SHA1 arbiter (G3/P9)
* is run by the coordinator and is the only thing that makes this final.
*
* Round 1 closed at 45/4763 mismatched. Round 2 took 45 -> 37 -> 33 -> 21
* -> 11 -> 3 -> 2 -> 0. See .run/giants/s19_bf14_report2.md.
*
* WHAT IT IS
* The *four*-light-box variant of the volumetric-light renderer whose
* 3-box sibling func_8017D960 (3,338 ins, ov_SC03_090) is MATCHED, and whose
* unlit ancestor func_8017BEBC (ov_SC03_099) is MATCHED. Same family, same
* skeleton; this is the biggest member.
*
* Signature: func_8017BF14(s32 arg0, s32 lim). Unlike every other member of
* the family this one is a LEAF -- 0 callees. The 3-call prologue
* (func_800491EC / func_800547D8 / func_80052E38) of the siblings is gone;
* `lim` arrives as arg1 (spilled to 0xB0). That is why the frame has no
* 0x10 argument area (tmpxy[] starts at sp+0x00) and no $ra save.
*
* Per part (stride 0x14, outer loop): build the 8-corner AABB in box[],
* rtpt/rtpt + rtps/rtps -> sxy[8], stszotz -> g.otz, reject on
* `lim >= g.otz`, then screen-space bbox reject on X (-0xA0..0xA1) and
* Y (-0x6E..0x6F).
* Per prim (stride 0xC, inner loop): rtpt the 3 vertices, stflg mask
* 0x7F85E000, nclip, stopz > 0, then a 4-way range tree on `code = w & 7`
* that keeps ONLY codes 6,7 (tri) and 2,3 (quad); 0,1,4,5 fall through to
* the loop tail.
* Per drawn poly: screen bbox reject, then each vertex is tested against
* FOUR axis-aligned light boxes (flags f0..f3), and if any is lit a
* 0x00..0x80 attenuation per active box is computed, summed, biased +0x10
* and clamped to 0x80 -> a grey gouraud vertex colour.
* lit -> POLY_GT3 (0x28, tag 0x34000000, OT 0x9000000)
* POLY_GT4 (0x34, tag 0x3C000000, OT 0xC000000)
* unlit -> POLY_FT3 (0x20, OT 0x7000000) / POLY_FT4 (0x28, OT 0x9000000)
* with rgbc = (tp[0] & 0xFF000000) | 0x101010 <-- NOT black,
* unlike func_8017D960 where the unlit colour is plain black.
*
* THE FOUR LIGHT BOXES (stride 0x1C, {s32 enable; u16 cx,cy,cz; s32 range})
* D_80197C28 / D_80197C44 / D_80197C60 / D_80197C7C.
* Falloff geometry differs from the 3-box sibling: RLO = R - 0x200 (not
* -0x80) and the ramp is ((R - d) / 4) (not (R - d)), so the 0x80 ceiling is
* reached over a 0x200-wide band instead of 0x80. The `/ 4` is a SIGNED
* divide -- `bgez / addiu 3 / sra 2` -- not a shift.
*
* FIVE ORIGINAL-SOURCE COPY-PASTE ARTEFACTS, all byte-proven
* The 4th light box was bolted onto a copy of the 3-box source BY HAND and
* the hand edit was incomplete in five places. Each was read off the target
* and each removed a measured delta.
* (1) `r3lo = r2 - 0x200;` -- box 3's low radius is derived from box 2's
* RANGE VARIABLE, not from its own D_80197C88. Proven by the target's
* `addiu $t6, $s0, -0x200` reusing the register that box 2's `lw` filled;
* spelling it `D_80197C6C - 0x200` re-loads the global (+2 ins).
* (2) Only SIX of the eight radius variables are zero-initialised
* (r0,r1,r2,r0lo,r1lo,r2lo) -- r3/r3lo are left uninitialised, exactly
* the init list the 3-box version needed.
* (3) The ATTEN body is written out LONGHAND 7 times (3 tri vertices +
* 4 quad vertices). When box 3 was bolted on, the `R` of the z- and
* y-axis KILL tests was left as r2 in three of those copies:
* tri v0: z and y use r2 tri v2: z uses r2 all others use r3.
* Proven by the 24 `sll $v0,$s0,16` sites: 3 per group in box 2 plus
* exactly three extra at idx 1471, 1504 (tri v0) and 2178 (tri v2).
* (4) In the QUAD lit arm only, rgb2 and rgb3 take their `<< 16` term from
* c1, not from c2/c3. Proven by the target CSE-ing ONE
* `sll $a0, $a0, 16` and re-using $a0 for all three stores.
* (5) *** ROUND 2 *** In the QUAD arm's VERTEX-1 group only, the box-3
* ATTEN's Y-axis `else if` branch accumulates into a2v instead of a3v,
* while that same test's KILL branch still says a3v. Modelled by the
* ATTEN3W macro below (`AW` = the y-else destination). Byte-proof:
* idx 3875 addu $a2,$zero,$zero kill branch -> a3v ($a2) [agreed]
* idx 3889 sra $a3,$s2,7 else branch -> a2v ($a3) [was the
* last structural residual]
* Costs zero instructions; the four other quad/tri sites are NOT like
* this (each was measured -- putting the artefact anywhere else is +2).
*
* FRAME (0x360, leaf -- no $ra, no argument area)
* 0x000 tmpxy[4] | 0x010 box[8] | 0x050 sxy[8] |
* 0x090 g{otz,flag,opz,sz0..sz3} | 0x0B0 lim | 0x0B8 j | 0x0C0 i |
* 0x0C8 vd | 0x0D0 ot | 0x0D8 pkt | 0x0E0 f2 | 0x0E8 f3 |
* 0x0F0/0x0F8/0x100 x3,z3,y3 | 0x108 prim | 0x110 nprim | 0x118 vtx |
* 0x120 nparts | 0x128 part | 0x130..0x1D0 lo/hi bounds (21 s16 slots) |
* 0x1D8..0x230 cx0..cz3 (12) | 0x238 r0 | 0x240 r1lo | 0x248 r2lo |
* 0x250 r3lo | 0x288..0x2D0 the LICM-hoisted sign-extended bounds |
* 0x328/0x330 spilled vertex coords | 0x338..0x358 s0-s7,fp.
* *** THE SLOT ORDER IS THE DECLARATION-ORDER ORACLE (see L5). ***
*
* ---------------------------------------------------------------------------
* ROUND-1 LEVERS (kept; measured effect is byte-identical %, anchored)
*
* L1 `cb = (tp[0] & 0xFF000000) | 0x101010;` in both UNLIT arms.
* L2 box-3 ATTEN kill-register per copy (artefact 3) + `r3lo = r2 - 0x200`
* (artefact 1). 89.15% -> 96.96% shape; killed sra+11 / sll+9.
* L3 quad-lit rgb2/rgb3 use `c1 << 16` (artefact 4). Killed the last sll+2.
* L4 [SUPERSEDED BY R3] `s32 c0..c3` declared inside the two CULL blocks.
* L5 `s32 f0, f1, f2, f3;` MOVED TO IMMEDIATELY AFTER `u8 *pkt;`.
* Spilled pseudos get stack slots in PSEUDO-NUMBER order and pseudo
* numbers are handed out in DECLARATION order, so the target's stack
* layout is a direct read-out of its declaration order. After this one
* move ALL 127 stack slots agree with the target exactly.
* L6 `u32 rgbw;` per emit arm.
* L7 RC-15 zero-byte ref dial on `mny`, first statement of the TRI cull
* block -- flips my->$a3 / mny->$a2 to the target's grant.
* L8 `cb` 2-statement accumulator [SUPERSEDED BY R2]; quad-lit rgb word as a
* 3-statement accumulator [kept for rgb0/rgb1, SUPERSEDED for rgb2/rgb3
* by R5].
* L9 FOUR REGISTER PINS: va->$t2, w->$a1, f0->$s3, c1->$a0.
* Round 2 removed va and w (see R4); f0 and c1 REMAIN and are both
* load-bearing. This function has NO `jal`, so Sec.74's caller-saved
* pin-spanning-a-call hazard cannot arise -- that is why pins are usable
* on this family member and were a trap on the others.
*
* ---------------------------------------------------------------------------
* ROUND-2 LEVERS -- 45 -> 0. Metric is `match_one` MISMATCH COUNT (length is
* exact throughout, so the raw count is honest). Every number is measured.
*
* THE ONE MECHANISM BEHIND R1/R2/R4/R7. A `register __asm__` pin makes the
* variable a HARD REG in the RTL from the start. When a 1-death local temp is
* produced from, or consumed into, that hard reg, local-alloc.c's
* `combine_regs` takes its hard-register branch (local-alloc.c:1795-1820) and
* records the pinned register in `qty_phys_sugg` for the temp's quantity --
* UNCONDITIONALLY, there is no death guard on that path. The temp then lands
* in the pinned register and the operation is done IN PLACE. The target,
* whose variable is an ordinary pseudo (reg_qty == -1 for anything crossing a
* block), never gets that suggestion and keeps the temp in $v0/$v1.
* Three independent cures, all used here:
* (i) give the temp a NAMED variable with >1 death, so local-alloc.c:472
* (`reg_basic_block >= 0 && reg_n_deaths == 1`) refuses it a quantity
* and combine_regs bails at its very first test -> R1
* (ii) drop the pin, if the pin is not load-bearing -> R4
* (iii) keep the pinned value LIVE past the temp, so that
* find_free_reg cannot honour the suggestion -> R7
*
* R1 `CLAMP80S` -- the four-way attenuation sum gets its own named variable
* `sv`, used at BOTH c1 sites (2 deaths). Cure (i). 45 -> 37
* Target: `addu $v0,..; addu $v0,..; addu $v0,..; addiu <c>,$v0,0x10`.
* Without it the whole chain is tied into the pinned c1 ($a0).
* Only the two c1 sites: `sv` on all 7 sites collapses the frame (-64).
* R2 UNLIT arms store `cb | 0x101010` as an expression instead of doing
* `cb |= 0x101010` in place. `cb` is a function-scope global allocno, so
* the in-place form writes $a1; the expression form is a 1-death local
* that combine_regs ties to the DYING constant register $v1, which is
* what the target does. 37 -> 33
* R3 *** `s32 c0, c1, c2, c3;` AT FUNCTION SCOPE, not per cull block. ***
* Read straight off the two MATCHED relatives (func_8017D960 line 310,
* func_8017F510 line 338), and confirmed by the target itself: its TRI
* grants (c0=$t4, c1=$a0, c2=$t2) are IDENTICAL to its QUAD grants, which
* is only possible if both arms share one set of allocnos. Per-cull-block
* scope splits them into two independent allocno sets and the TRI set
* drifts. 33 -> 21
* Declaration POSITION is neutral (5 anchors swept, all 21).
* NOTE this REVERSES round-1's L4. L4 was correct on the round-1 base --
* it was supplying the extra local allocno that spills r1lo -- but R1+R2
* supply that pressure now, and the c1 pin does the rest.
* R4 DROP the `va->$t2` and `w->$a1` pins. With c0..c3 at function scope
* they are no longer load-bearing, and they were the sole cause of the
* prim-word producer ties (`andi`/`srl` written straight into $t2/$a1).
* Cure (ii). 21 -> 17 -> 13 (pair)
* Round 1 measured these as worth 4%; that was true of the round-1 base
* and is FALSE here. Base-dependence, not a contradiction.
* R5 QUAD lit rgb2/rgb3 revert to the SINGLE-EXPRESSION form (rgb0/rgb1 keep
* the 3-statement accumulator). The intermediates then become 1-death
* local temps that alternate $v0/$v1, which is what lets the target's
* store of the previous rgb word sit one slot LATER. 21 -> 11
* Doing it to all four, or to rgb0/rgb1 only, is worse (23 / 21).
* R6 Artefact 5 -- `ATTEN3W(a3v, a2v, ...)` at the QUAD vertex-1 site. 3 -> 2
* R7 RC-15 zero-byte ref `__asm__ __volatile__ ("" :: "r" (c1));` placed
* immediately after the TRI arm's rgb1 store. Cure (iii): it keeps the
* pinned c1 ($a0) live past `c1 << 16`, so find_free_reg cannot honour
* combine_regs' $a0 suggestion and the shift goes to $v0. 2 -> 0
* The c1 pin CANNOT simply be removed: it is what spills r1lo (dropping
* it, or moving it to any other colour, costs -64 length). Measured.
*
* SCHEDULER ATTRIBUTION (Sec.76 primitive, run in round 2, never run before on
* this function). Compiled with -fno-schedule-insns, with
* -fno-schedule-insns2, and with both. The store/shift transposition at
* idx 4643-4652 kept MY source order under all three. It was therefore never
* a `sched.c` decision: the ordering is a CONSEQUENCE of the register grant
* (a 3-statement accumulator pins the value in one register, so no scheduler
* could hoist the next `or` above the `sw`). R5 fixed it by changing the
* grant, exactly as Sec.78 predicts.
* =========================================================================== */
#define gte_ldv0(r0) __asm__ volatile ( \
"lwc2 $0, 0( %0 );" \
"lwc2 $1, 4( %0 )" \
: \
: "r"( r0 ) )
#define gte_ldv3(r0, r1, r2) __asm__ volatile ( \
"lwc2 $0, 0( %0 );" \
"lwc2 $1, 4( %0 );" \
"lwc2 $2, 0( %1 );" \
"lwc2 $3, 4( %1 );" \
"lwc2 $4, 0( %2 );" \
"lwc2 $5, 4( %2 )" \
: \
: "r"( r0 ), "r"( r1 ), "r"( r2 ) )
#define gte_ldv3c(r0) __asm__ volatile ( \
"lwc2 $0, 0( %0 );" \
"lwc2 $1, 4( %0 );" \
"lwc2 $2, 8( %0 );" \
"lwc2 $3, 12( %0 );" \
"lwc2 $4, 16( %0 );" \
"lwc2 $5, 20( %0 )" \
: \
: "r"( r0 ) )
#define gte_rtps() __asm__ volatile ("nop;nop;rtps")
#define gte_rtpt() __asm__ volatile ("nop;nop;rtpt")
#define gte_nclip() __asm__ volatile ("nop;nop;nclip")
#define gte_stsxy(r0) __asm__ volatile ( \
"swc2 $14, 0( %0 )" \
: \
: "r"( r0 ) \
: "memory" )
#define gte_stsxy3(r0, r1, r2) __asm__ volatile ( \
"swc2 $12, 0( %0 );" \
"swc2 $13, 0( %1 );" \
"swc2 $14, 0( %2 )" \
: \
: "r"( r0 ), "r"( r1 ), "r"( r2 ) \
: "memory" )
#define gte_stsxy3c(r0) __asm__ volatile ( \
"swc2 $12, 0( %0 );" \
"swc2 $13, 4( %0 );" \
"swc2 $14, 8( %0 )" \
: \
: "r"( r0 ) \
: "memory" )
#define gte_stsz3(r0, r1, r2) __asm__ volatile ( \
"swc2 $17, 0( %0 );" \
"swc2 $18, 0( %1 );" \
"swc2 $19, 0( %2 )" \
: \
: "r"( r0 ), "r"( r1 ), "r"( r2 ) \
: "memory" )
#define gte_stsz4(r0, r1, r2, r3) __asm__ volatile ( \
"swc2 $16, 0( %0 );" \
"swc2 $17, 0( %1 );" \
"swc2 $18, 0( %2 );" \
"swc2 $19, 0( %3 )" \
: \
: "r"( r0 ), "r"( r1 ), "r"( r2 ), "r"( r3 ) \
: "memory" )
#define gte_stszotz(r0) __asm__ volatile ( \
"mfc2 $12, $19;" \
"nop;" \
"sra $12, $12, 2;" \
"sw $12, 0( %0 )" \
: \
: "r"( r0 ) \
: "$12", "memory" )
#define gte_stflg(r0) __asm__ volatile ( \
"cfc2 $12, $31;" \
"nop;" \
"sw $12, 0( %0 )" \
: \
: "r"( r0 ) \
: "$12", "memory" )
#define gte_stopz(r0) __asm__ volatile ( \
"swc2 $24, 0( %0 )" \
: \
: "r"( r0 ) \
: "memory" )
/* ---- the two gouraud-textured packet layouts this function emits ---------- */
typedef struct {
u32 tag;
u32 rgb0; s16 x0, y0; u32 uv0;
u32 rgb1; s16 x1, y1; u32 uv1;
u32 rgb2; s16 x2, y2; u16 uv2, p2;
} PolyGT3; /* 0x28 */
typedef struct {
u32 tag;
u32 rgb0; s16 x0, y0; u32 uv0;
u32 rgb1; s16 x1, y1; u32 uv1;
u32 rgb2; s16 x2, y2; u16 uv2, p2;
u32 rgb3; s16 x3, y3; u16 uv3, p3;
} PolyGT4; /* 0x34 */
/* ---- the four light-volume descriptors (stride 0x1C) --------------------- */
extern s32 D_80197C28;
extern u16 D_80197C2C, D_80197C2E, D_80197C30;
extern s32 D_80197C34;
extern s32 D_80197C44;
extern u16 D_80197C48, D_80197C4A, D_80197C4C;
extern s32 D_80197C50;
extern s32 D_80197C60;
extern u16 D_80197C64, D_80197C66, D_80197C68;
extern s32 D_80197C6C;
extern s32 D_80197C7C;
extern u16 D_80197C80, D_80197C82, D_80197C84;
extern u16 D_80197C88;
/* ---- the box-containment test for one vertex against one light box ------- */
#define BOXTEST(F, X, Y, Z, LX, HX, LY, HY, LZ, HZ) \
if ((LX) < (X) && (X) < (HX) && (LY) < (Y) && (Y) < (HY) && (LZ) < (Z) && (Z) < (HZ)) F = 1
/* ---- the separable per-axis falloff, visited in x, z, y order ------------ */
#define ATTEN(A, F, X, Y, Z, CX, CY, CZ, R, RLO) \
A = 0; \
if (F) { \
d = (X) - (CX); if (d < 0) d = (CX) - (X); \
if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \
d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \
if ((R) < d) A = 0; \
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \
if ((R) < d) A = 0; \
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
}
#define ATTEN3W(A, AW, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \
A = 0; \
if (F) { \
d = (X) - (CX); if (d < 0) d = (CX) - (X); \
if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \
d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \
if ((RZ) < d) A = 0; \
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \
if ((RY) < d) A = 0; \
else if ((RLO) < d) AW = (A * (((R) - d) / 4)) >> 7; \
}
#define ATTEN3(A, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \
A = 0; \
if (F) { \
d = (X) - (CX); if (d < 0) d = (CX) - (X); \
if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \
d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \
if ((RZ) < d) A = 0; \
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \
if ((RY) < d) A = 0; \
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
}
#define CLAMP80(C, A0, A1, A2, A3) C = (A0) + (A1) + (A2) + (A3) + 0x10; if ((C) > 0x80) C = 0x80
#define CLAMP80S(C, A0, A1, A2, A3) sv = (A0) + (A1) + (A2) + (A3); C = sv + 0x10; if ((C) > 0x80) C = 0x80
void func_8017BF14(s32 arg0, s32 lim)
{
typedef struct { u32 w0, w1, w2; } Prim;
extern u8 *D_800A5E60;
extern u8 D_800A6610[];
extern u8 D_800AF630[];
DVECTOR2 tmpxy[4];
SVECTOR2 box[8];
SVECTOR2 sxy[8];
struct { long otz, flag, opz, sz0, sz1, sz2, sz3; } g;
s32 j;
u32 i;
u8 *vd;
u32 ot;
u8 *pkt;
register s32 f0 __asm__("$19");
s32 f1, f2, f3;
s16 x3, z3, y3;
Prim *prim;
u32 nprim;
u8 *vtx;
s32 nparts;
Part *part;
s16 lo0x, hi0x, lo0y, hi0y, lo0z, hi0z;
s16 lo1x, hi1x, lo1y, hi1y, lo1z, hi1z;
s16 lo2x, hi2x, lo2y, hi2y, lo2z, hi2z;
s16 lo3x, hi3x, lo3y, hi3y, lo3z, hi3z;
s16 cx0, cy0, cz0, cx1, cy1, cz1, cx2, cy2, cz2, cx3, cy3, cz3;
s16 r0;
s16 r1;
s16 r2;
s16 r3;
s16 r0lo;
s16 r1lo;
s16 r2lo;
s16 r3lo;
u8 *va, *vb, *vc;
u32 w;
s32 code;
u32 vw, vzw;
u32 wx, wy, wz;
s32 xa32, xb32, t32;
s32 xmn1, xmx1, xmn2, xmx2;
s32 mnc, mxc;
s16 my, mny, mx, mn;
u8 *base;
s16 x0, y0, z0, x1, y1, z1, x2, y2, z2;
s32 a0v, a1v, a2v, a3v;
register s32 c1 __asm__("$4"); s32 c0, c2, c3;
s32 d;
u32 *tp;
u32 uvw;
u32 cb;
s32 sv;
base = D_800AF630;
r2lo = 0;
r1lo = 0;
r0lo = 0;
r2 = 0;
r1 = 0;
r0 = 0;
if (D_80197C28) {
cx0 = D_80197C2C;
cy0 = D_80197C2E;
r0lo = D_80197C34 - 0x200;
r0 = D_80197C34;
cz0 = D_80197C30;
} else {
cz0 = 0x6000;
cy0 = 0x6000;
cx0 = 0x6000;
}
if (D_80197C44) {
cx1 = D_80197C48;
cy1 = D_80197C4A;
r1 = D_80197C50;
r1lo = D_80197C50 - 0x200;
cz1 = D_80197C4C;
} else {
cz1 = 0x6000;
cy1 = 0x6000;
cx1 = 0x6000;
}
if (D_80197C60) {
cx2 = D_80197C64;
cy2 = D_80197C66;
r2 = D_80197C6C;
r2lo = D_80197C6C - 0x200;
cz2 = D_80197C68;
} else {
cz2 = 0x6000;
cy2 = 0x6000;
cx2 = 0x6000;
}
if (D_80197C7C) {
cx3 = D_80197C80;
cy3 = D_80197C82;
r3lo = r2 - 0x200;
r3 = D_80197C88;
cz3 = D_80197C84;
} else {
cz3 = 0x6000;
cy3 = 0x6000;
cx3 = 0x6000;
}
lo0x = cx0 - r0; hi0x = cx0 + r0;
lo0y = cy0 - r0; hi0y = cy0 + r0;
lo0z = cz0 - r0; hi0z = cz0 + r0;
lo1x = cx1 - r1; hi1x = cx1 + r1;
lo1y = cy1 - r1; hi1y = cy1 + r1;
lo1z = cz1 - r1; hi1z = cz1 + r1;
lo2x = cx2 - r2; hi2x = cx2 + r2;
lo2y = cy2 - r2; hi2y = cy2 + r2;
lo2z = cz2 - r2; hi2z = cz2 + r2;
lo3x = cx3 - r3; hi3x = cx3 + r3;
lo3y = cy3 - r3; hi3y = cy3 + r3;
lo3z = cz3 - r3; hi3z = cz3 + r3;
pkt = D_800A5E60;
part = *(Part **)(arg0 + 0xC);
nparts = *(s32 *)(*(s32 *)(arg0 + 8) + 8);
vtx = *(u8 **)(*(s32 *)(arg0 + 8) + 0x10);
ot = (u32)&D_800A6610[(*(u16 *)(base + 0xA3D2)) << 14];
for (j = 0; j < nparts; j++, part++) {
wx = part->xx;
mn = wx;
mx = wx >> 16;
wy = part->yy;
mny = wy;
my = wy >> 16;
wz = part->zz;
box[0].vx = mn; box[0].vy = mny;
box[1].vx = mx; box[1].vy = mny;
box[2].vx = mn; box[2].vy = mny;
box[3].vx = mx; box[3].vy = mny;
box[4].vx = mn; box[4].vy = my;
box[5].vx = mx; box[5].vy = my;
box[6].vx = mn; box[6].vy = my;
box[7].vx = mx; box[7].vy = my;
wy = wz >> 16;
box[0].vz = wz;
box[1].vz = wz;
box[4].vz = wz;
box[5].vz = wz;
box[2].vz = wy;
box[3].vz = wy;
box[6].vz = wy;
box[7].vz = wy;
gte_ldv3c(&box[0]);
gte_rtpt();
gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]);
gte_ldv0(&box[3]);
gte_rtps();
gte_stsxy(&sxy[3]);
gte_ldv3c(&box[4]);
gte_rtpt();
gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]);
gte_ldv0(&box[7]);
gte_rtps();
gte_stsxy(&sxy[7]);
gte_stszotz(&g.otz);
if (lim >= g.otz) {
xa32 = sxy[0].vx;
xb32 = sxy[1].vx;
if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; }
t32 = sxy[2].vx;
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
t32 = sxy[3].vx;
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
xa32 = sxy[4].vx;
xb32 = sxy[5].vx;
if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; }
t32 = sxy[6].vx;
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
t32 = sxy[7].vx;
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
mnc = xmn1;
if (xmn2 < xmn1) mnc = xmn2;
mxc = xmx1;
if (mxc < xmx2) mxc = xmx2;
if ((s16)mxc >= -0xA0 && (s16)mnc < 0xA1) {
xa32 = sxy[0].vy;
xb32 = sxy[1].vy;
if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; }
t32 = sxy[2].vy;
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
t32 = sxy[3].vy;
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
xa32 = sxy[4].vy;
xb32 = sxy[5].vy;
if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; }
t32 = sxy[6].vy;
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
t32 = sxy[7].vy;
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
mnc = xmn1;
if (xmn2 < xmn1) mnc = xmn2;
mxc = xmx1;
if (mxc < xmx2) mxc = xmx2;
if ((s16)mxc >= -0x6E && (s16)mnc < 0x6F) {
nprim = part->nprim;
prim = (Prim *)part->prim;
for (i = 0; i < nprim; i++, prim++) {
w = prim->w1;
va = vtx + (w & 0xFFFF);
vb = vtx + (w >> 16);
w = prim->w2;
vc = vtx + (w & 0xFFFF);
w = w >> 16;
gte_ldv3(va, vb, vc);
gte_rtpt();
gte_stflg(&g.flag);
if (!(g.flag & 0x7F85E000)) {
gte_nclip();
code = w & 7;
vd = vtx + (w & 0xFFF8);
gte_stopz(&g.opz);
if (g.opz > 0) {
switch (code) {
case 6:
case 7:
/* ---------------- TRI (FT3 / GT3) ---------------- */
gte_stsxy3c(&tmpxy[0]);
gte_stsz3(&g.sz0, &g.sz1, &g.sz2);
if (tmpxy[0].vx > tmpxy[1].vx) { mx = tmpxy[0].vx; mn = tmpxy[1].vx; }
else { mn = tmpxy[0].vx; mx = tmpxy[1].vx; }
if (tmpxy[2].vx > mx) mx = tmpxy[2].vx;
else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx;
if (mx >= -0xA0 && mn < 0xA1) {
if (tmpxy[0].vy > tmpxy[1].vy) { my = tmpxy[0].vy; mny = tmpxy[1].vy; }
else { mny = tmpxy[0].vy; my = tmpxy[1].vy; }
if (tmpxy[2].vy > my) my = tmpxy[2].vy;
else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy;
if (my >= -0x6E && mny < 0x6F) {
s32 za, zb;
__asm__ __volatile__ ("" :: "r" (mny));
if (g.sz0 > g.sz1) { za = g.sz0; if (za < g.sz2) za = g.sz2; }
else { za = g.sz1; if (za < g.sz2) za = g.sz2; }
g.opz = za;
f3 = 0; f2 = 0; f1 = 0; f0 = 0;
vw = *(u32 *)va;
vzw = *(u32 *)(va + 4);
x0 = vw; y0 = vw >> 16; z0 = vzw;
BOXTEST(f0, x0, y0, z0, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
BOXTEST(f1, x0, y0, z0, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
BOXTEST(f2, x0, y0, z0, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
BOXTEST(f3, x0, y0, z0, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
vw = *(u32 *)vb;
vzw = *(u32 *)(vb + 4);
x1 = vw; y1 = vw >> 16; z1 = vzw;
BOXTEST(f0, x1, y1, z1, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
BOXTEST(f1, x1, y1, z1, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
BOXTEST(f2, x1, y1, z1, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
BOXTEST(f3, x1, y1, z1, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
vw = *(u32 *)vc;
vzw = *(u32 *)(vc + 4);
x2 = vw; y2 = vw >> 16; z2 = vzw;
BOXTEST(f0, x2, y2, z2, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
BOXTEST(f1, x2, y2, z2, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
BOXTEST(f2, x2, y2, z2, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
BOXTEST(f3, x2, y2, z2, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
if (f0 | f1 | f2 | f3) {
u32 *otp;
u32 rgbw;
ATTEN(a0v, f0, x0, y0, z0, cx0, cy0, cz0, r0, r0lo);
ATTEN(a1v, f1, x0, y0, z0, cx1, cy1, cz1, r1, r1lo);
ATTEN(a2v, f2, x0, y0, z0, cx2, cy2, cz2, r2, r2lo);
ATTEN3(a3v, f3, x0, y0, z0, cx3, cy3, cz3, r3, r3lo, r2, r2);
CLAMP80(c0, a0v, a1v, a2v, a3v);
ATTEN(a0v, f0, x1, y1, z1, cx0, cy0, cz0, r0, r0lo);
ATTEN(a1v, f1, x1, y1, z1, cx1, cy1, cz1, r1, r1lo);
ATTEN(a2v, f2, x1, y1, z1, cx2, cy2, cz2, r2, r2lo);
ATTEN3(a3v, f3, x1, y1, z1, cx3, cy3, cz3, r3, r3lo, r3, r3);
CLAMP80S(c1, a0v, a1v, a2v, a3v);
ATTEN(a0v, f0, x2, y2, z2, cx0, cy0, cz0, r0, r0lo);
ATTEN(a1v, f1, x2, y2, z2, cx1, cy1, cz1, r1, r1lo);
ATTEN(a2v, f2, x2, y2, z2, cx2, cy2, cz2, r2, r2lo);
ATTEN3(a3v, f3, x2, y2, z2, cx3, cy3, cz3, r3, r3lo, r2, r3);
CLAMP80(c2, a0v, a1v, a2v, a3v);
*(u32 *)&((PolyGT3 *)pkt)->x0 = *(u32 *)&tmpxy[0];
*(u32 *)&((PolyGT3 *)pkt)->x1 = *(u32 *)&tmpxy[1];
*(u32 *)&((PolyGT3 *)pkt)->x2 = *(u32 *)&tmpxy[2];
tp = (u32 *)prim->w0;
cb = 0x34000000;
rgbw = (c0 | cb) | (c0 << 8) | (c0 << 16);
((PolyGT3 *)pkt)->rgb0 = rgbw;
rgbw = (c1 | cb) | (c1 << 8) | (c1 << 16);
((PolyGT3 *)pkt)->rgb1 = rgbw;
__asm__ __volatile__ ("" :: "r" (c1));
rgbw = (c2 | cb) | (c2 << 8) | (c2 << 16);
((PolyGT3 *)pkt)->rgb2 = rgbw;
((PolyGT3 *)pkt)->uv0 = tp[1];
((PolyGT3 *)pkt)->uv1 = tp[2];
((PolyGT3 *)pkt)->uv2 = tp[3];
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000;
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
pkt += 0x28;
} else {
u32 *otp;
u32 rgbw;
*(u32 *)&((PolyFT3 *)pkt)->x0 = *(u32 *)&tmpxy[0];
*(u32 *)&((PolyFT3 *)pkt)->x1 = *(u32 *)&tmpxy[1];
*(u32 *)&((PolyFT3 *)pkt)->x2 = *(u32 *)&tmpxy[2];
tp = (u32 *)prim->w0;
cb = tp[0] & 0xFF000000;
((PolyFT3 *)pkt)->rgbc = cb | 0x101010;
((PolyFT3 *)pkt)->uvc0 = tp[1];
((PolyFT3 *)pkt)->uvp1 = tp[2];
((PolyFT3 *)pkt)->uv2 = tp[3];
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000;
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
pkt += 0x20;
}
}
}
break;
case 2:
case 3:
/* ---------------- QUAD (FT4 / GT4) ---------------- */
gte_stsxy3c(&tmpxy[0]);
gte_ldv0(vd);
gte_rtps();
if (tmpxy[0].vx > tmpxy[1].vx) { mx = tmpxy[0].vx; mn = tmpxy[1].vx; }
else { mn = tmpxy[0].vx; mx = tmpxy[1].vx; }
if (tmpxy[2].vx > mx) mx = tmpxy[2].vx;
else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx;
if (tmpxy[0].vy > tmpxy[1].vy) { my = tmpxy[0].vy; mny = tmpxy[1].vy; }
else { mny = tmpxy[0].vy; my = tmpxy[1].vy; }
if (tmpxy[2].vy > my) my = tmpxy[2].vy;
else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy;
gte_stflg(&g.flag);
if (!(g.flag & 0x7F85E000)) {
gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3);
gte_stsxy((long *)&((PolyFT4 *)pkt)->x3);
if (((PolyFT4 *)pkt)->x3 < mn) mn = ((PolyFT4 *)pkt)->x3;
else if (mx < ((PolyFT4 *)pkt)->x3) mx = ((PolyFT4 *)pkt)->x3;
if (mx >= -0xA0 && mn < 0xA1) {
if (((PolyFT4 *)pkt)->y3 < mny) mny = ((PolyFT4 *)pkt)->y3;
else if (my < ((PolyFT4 *)pkt)->y3) my = ((PolyFT4 *)pkt)->y3;
if (my >= -0x6E && mny < 0x6F) {
s32 za, zb;
zb = g.sz2;
if (zb < g.sz3) zb = g.sz3;
za = g.sz0;
if (za < g.sz1) za = g.sz1;
if (za < zb) za = zb;
g.opz = za;
f3 = 0; f2 = 0; f1 = 0; f0 = 0;
vw = *(u32 *)va;
vzw = *(u32 *)(va + 4);
x0 = vw; y0 = vw >> 16; z0 = vzw;
BOXTEST(f0, x0, y0, z0, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
BOXTEST(f1, x0, y0, z0, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
BOXTEST(f2, x0, y0, z0, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
BOXTEST(f3, x0, y0, z0, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
vw = *(u32 *)vb;
vzw = *(u32 *)(vb + 4);
x1 = vw; y1 = vw >> 16; z1 = vzw;
BOXTEST(f0, x1, y1, z1, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
BOXTEST(f1, x1, y1, z1, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
BOXTEST(f2, x1, y1, z1, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
BOXTEST(f3, x1, y1, z1, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
vw = *(u32 *)vc;
vzw = *(u32 *)(vc + 4);
x2 = vw; y2 = vw >> 16; z2 = vzw;
BOXTEST(f0, x2, y2, z2, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
BOXTEST(f1, x2, y2, z2, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
BOXTEST(f2, x2, y2, z2, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
BOXTEST(f3, x2, y2, z2, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
vw = *(u32 *)vd;
vzw = *(u32 *)(vd + 4);
x3 = vw; y3 = vw >> 16; z3 = vzw;
BOXTEST(f0, x3, y3, z3, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
BOXTEST(f1, x3, y3, z3, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
BOXTEST(f2, x3, y3, z3, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
BOXTEST(f3, x3, y3, z3, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
if (f0 | f1 | f2 | f3) {
u32 *otp;
u32 rgbw;
ATTEN(a0v, f0, x0, y0, z0, cx0, cy0, cz0, r0, r0lo);
ATTEN(a1v, f1, x0, y0, z0, cx1, cy1, cz1, r1, r1lo);
ATTEN(a2v, f2, x0, y0, z0, cx2, cy2, cz2, r2, r2lo);
ATTEN3(a3v, f3, x0, y0, z0, cx3, cy3, cz3, r3, r3lo, r3, r3);
CLAMP80(c0, a0v, a1v, a2v, a3v);
ATTEN(a0v, f0, x1, y1, z1, cx0, cy0, cz0, r0, r0lo);
ATTEN(a1v, f1, x1, y1, z1, cx1, cy1, cz1, r1, r1lo);
ATTEN(a2v, f2, x1, y1, z1, cx2, cy2, cz2, r2, r2lo);
ATTEN3W(a3v, a2v, f3, x1, y1, z1, cx3, cy3, cz3, r3, r3lo, r3, r3);
CLAMP80S(c1, a0v, a1v, a2v, a3v);
ATTEN(a0v, f0, x2, y2, z2, cx0, cy0, cz0, r0, r0lo);
ATTEN(a1v, f1, x2, y2, z2, cx1, cy1, cz1, r1, r1lo);
ATTEN(a2v, f2, x2, y2, z2, cx2, cy2, cz2, r2, r2lo);
ATTEN3(a3v, f3, x2, y2, z2, cx3, cy3, cz3, r3, r3lo, r3, r3);
CLAMP80(c2, a0v, a1v, a2v, a3v);
ATTEN(a0v, f0, x3, y3, z3, cx0, cy0, cz0, r0, r0lo);
ATTEN(a1v, f1, x3, y3, z3, cx1, cy1, cz1, r1, r1lo);
ATTEN(a2v, f2, x3, y3, z3, cx2, cy2, cz2, r2, r2lo);
ATTEN3(a3v, f3, x3, y3, z3, cx3, cy3, cz3, r3, r3lo, r3, r3);
CLAMP80(c3, a0v, a1v, a2v, a3v);
*(u32 *)&((PolyGT4 *)pkt)->x0 = *(u32 *)&tmpxy[0];
*(u32 *)&((PolyGT4 *)pkt)->x1 = *(u32 *)&tmpxy[1];
*(u32 *)&((PolyGT4 *)pkt)->x2 = *(u32 *)&tmpxy[2];
gte_stsxy((long *)&((PolyGT4 *)pkt)->x3);
tp = (u32 *)prim->w0;
cb = 0x3C000000;
rgbw = c0 | cb; rgbw |= c0 << 8; rgbw |= c0 << 16;
((PolyGT4 *)pkt)->rgb0 = rgbw;
rgbw = c1 | cb; rgbw |= c1 << 8; rgbw |= c1 << 16;
((PolyGT4 *)pkt)->rgb1 = rgbw;
rgbw = (c2 | cb) | (c2 << 8) | (c1 << 16);
((PolyGT4 *)pkt)->rgb2 = rgbw;
rgbw = (c3 | cb) | (c3 << 8) | (c1 << 16);
((PolyGT4 *)pkt)->rgb3 = rgbw;
((PolyGT4 *)pkt)->uv0 = tp[1];
((PolyGT4 *)pkt)->uv1 = tp[2];
uvw = tp[3];
((PolyGT4 *)pkt)->uv2 = uvw;
((PolyGT4 *)pkt)->uv3 = uvw >> 16;
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0xC000000;
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
pkt += 0x34;
} else {
u32 *otp;
u32 rgbw;
*(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0];
*(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1];
*(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2];
tp = (u32 *)prim->w0;
cb = tp[0] & 0xFF000000;
((PolyFT4 *)pkt)->rgbc = cb | 0x101010;
((PolyFT4 *)pkt)->uvc0 = tp[1];
((PolyFT4 *)pkt)->uvp1 = tp[2];
uvw = tp[3];
((PolyFT4 *)pkt)->uv2 = uvw;
((PolyFT4 *)pkt)->uv3 = uvw >> 16;
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000;
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
pkt += 0x28;
}
}
}
}
break;
}
}
}
}
}
}
}
}
D_800A5E60 = pkt;
}
+53
View File
@@ -6194,3 +6194,56 @@ Artifacts: `.run/giants/s19_func_8017BF14_b1.c` (45/4763), a **pin-free fallback
**Cold-start economics, measured:** a 4,763-instruction leaf giant with a *findable* matched relative
reached 99.06% in one pass but did not close. Budget a second pass for anything this size; the first
pass buys the decode, the frame, and the length — the last ~1% is register grants.
## §80 — A do-not-re-buy entry is scoped to its BASE, not to the function; and the pin's hidden cost is an unconditional `qty_phys_sugg` (Phase 29 SESSION-19, `func_8017BF14` 45 → 0)
Round 2 closed the 4,763-instruction behemoth (`45 → 37 → 33 → 21 → 11 → 3 → 2 → 0`, reproduced 3×
from independent work dirs, banked whole-binary BYTE-IDENTICAL). The route matters more than the win.
### ⚠️ THE PROCESS CORRECTION: a measured negative is relative to the draft it was measured on
Round 1 left a careful ~40-row do-not-re-buy table. **Three of its entries INVERTED on round 2's base.**
The same edit (`qsingle23`) measured **1,040 mismatched on the 45-base and 11 on the 21-base**.
Re-testing the round-1 negative list cost **~20 seconds** and produced **three of the seven winning
levers**.
**So: a do-not-re-buy table is a record of `(edit, base) → result`, NOT `edit → useless`.** After any
lever that moves the base materially, **re-run the negative list** — it is seconds with a real harness
and it is where the next levers hide. This retroactively qualifies every such table in this cookbook
(§45, §60b, §75a, §76, §78, §79 and round 1 of this function): treat them as *starting hypotheses at
the base where they were taken*, not as closed questions.
Corollary already seen: round 1 measured "removing the `va→$t2` pin costs 4% elsewhere" and concluded
*keep the pin*. On a base where `c0..c3` sit at function scope, **removing those pins is worth 21→13**
— the opposite conclusion from the same experiment.
### The pin's hidden cost, with the citation
`combine_regs`' hard-register branch (`local-alloc.c:1795`, reached from `:1295` with
`already_dead == 0`) records the pinned register in **`qty_phys_sugg` unconditionally — there is no
death guard.** So a `register __asm__` pin does not merely *prefer* a register: it actively invites
local-alloc to tie producer chains into it, which is exactly the residual-(a) tie round 1 diagnosed
but mis-cured. Three separable cures exist; the new one is worth knowing:
- **R7 — a zero-byte `__asm__` ref that keeps the pinned value LIVE PAST the temp**, so
`find_free_reg` cannot honour the suggestion. That closed the final 2 instructions, and was
*necessary* because `c1→$a0` proved uniquely load-bearing (it is what spills `r1lo`; every
alternative pin lost 64 instructions).
### The flagged "#1 move" LOST — and why the failure is informative
Variable REUSE (§45-A / RC-14 MERGE) was swept in full: **every merge lost, 43–3294 across 8 merges.**
It was the right lever class for the sibling `func_8017F510` (97 → 10) and the wrong one here, for a
structural reason worth carrying: **the TRI and QUAD grants did not differ by RANK, they differed by
IDENTITY — two independent allocno sets.** Re-ranking inside one set cannot fix a two-set problem.
**Diagnose whether you have a ranking problem or an identity problem before reaching for a merge.**
The actual fix was `s32 c0,c1,c2,c3;` at **function** scope (33 → 21), read off the two matched
relatives (`b5:310`, `b4:338`) and confirmed against the target itself: its TRI grants are *identical*
to its QUAD grants.
### §78's attribution primitive, run and reproduced
Under `-fno-schedule-insns`, `-fno-schedule-insns2`, and both, the draft's order was **unchanged** ⇒
the rgb-accumulator transposition was never a `sched.c` decision. A 3-statement accumulator pins the
value to one register, so no scheduler *could* hoist the `or` above the `sw`. Changing the grant fixed
the order for free — §78 reproduced on a second function.
### Cold-start economics, now complete
A 4,763-instruction leaf giant with a findable matched relative: **round 1 = decode + exact length +
exact frame + 99.06%; round 2 = the last 45.** Two passes, and the second was far cheaper than the
first. Budget two passes at this size and do not read a 99% round-1 result as a stall.
+7 -7
View File
@@ -4,16 +4,16 @@
# cross-binary collapsible-byte leverage: docs/duplicates.cross.md.
# THREE progress metrics (all matter — see the labels):
FLEET fn-count byte-ident: 315457 / 353717 = 89.18% (REAL+LINKED+empties; FUNCTION-count, ×134-inflated — one crack counts per overlay)
FLEET instr-weighted : 10580590 / 13141652 = 80.5% (shipped .text across main + resident + 138 overlays; the decomp.dev-DISPLAY number)
FLEET distinct-code(uniq): 3838143 / 5634875 = 68.1% (64905/87459 unique fns; the DISTINCT-RE number)
FLEET fn-count byte-ident: 315458 / 353717 = 89.18% (REAL+LINKED+empties; FUNCTION-count, ×134-inflated — one crack counts per overlay)
FLEET instr-weighted : 10585353 / 13141652 = 80.5% (shipped .text across main + resident + 138 overlays; the decomp.dev-DISPLAY number)
FLEET distinct-code(uniq): 3842906 / 5634875 = 68.2% (64906/87459 unique fns; the DISTINCT-RE number)
MAIN game-code weighted : 436 / 60201 = 0.7% (INCLUDED in the fleet numbers above since 2026-07-22 — roadmap §1 metrics contract; LINKED-excluding Ghidra sig dated 2026-06-14; caveat is R34: no independent second oracle for a PS-X EXE, NOT drift)
(fleet EXCLUDING main, for continuity with pre-2026-07-22 readings: 10580154 / 13081451 = 80.9%)
(fleet EXCLUDING main, for continuity with pre-2026-07-22 readings: 10584917 / 13081451 = 80.9%)
FLEET REAL substantive : 313602 (of which dedup-shared 239530 via 1886 groups / 239604 instances)
FLEET REAL substantive : 313603 (of which dedup-shared 239530 via 1886 groups / 239604 instances)
FLEET LINKED PsyQ objs : 959
FLEET NON_MATCHING : 7 (0 in any default build — G4)
FLEET INCLUDE_ASM stubs : 38253
FLEET INCLUDE_ASM stubs : 38252
FLEET matchable : 353717
| binary | REAL | shared | LINKED | byte-ident | matchable | byte-ident % |
@@ -89,7 +89,7 @@ FLEET matchable : 353717
| ov_SC03_113 | 2252 | 1740 | 0 | 2255 | 2468 | 91.4% |
| ov_SC03_114 | 2242 | 1738 | 0 | 2244 | 2415 | 92.9% |
| ov_SC03_115 | 2259 | 1738 | 0 | 2261 | 2472 | 91.5% |
| ov_SC03_116 | 2249 | 1738 | 0 | 2252 | 2438 | 92.4% |
| ov_SC03_116 | 2250 | 1738 | 0 | 2253 | 2438 | 92.4% |
| ov_SC03_117 | 2277 | 1738 | 0 | 2283 | 2557 | 89.3% |
| ov_SC03_118 | 2323 | 1762 | 0 | 2324 | 2685 | 86.6% |
| ov_SC03_119 | 2322 | 1762 | 0 | 2323 | 2685 | 86.5% |
+42
View File
@@ -3871,6 +3871,48 @@ conditional) · main-EXE/B9 + GLM/B6 + resident's 14 walls (P30) · behemoths B7
Artifacts: `s19_func_8017BF14_b1.c` (45/4763) + a **pin-free fallback at 789/4763 that is 100%
structural** + `s19_bf14_report.md` (~40-row do-not-re-buy table, 4 refuted diagnoses).
- **🏆🏆🏆 2026-07-25 (SESSION-19) — `func_8017BF14` (4,763 ins) CLOSED IN ROUND 2: 45 → 0.
The 4th behemoth of the session, and the largest single function matched in the project.**
`45 → 37 → 33 → 21 → 11 → 3 → 2 → 0`, reproduced 3× from independent work dirs. **Verified
independently (R14):** `match_one` → **MATCH (4763 ins)**; `harvest_verify --binary ov_SC03_116`
→ **BYTE-IDENTICAL**. (Agent was interrupted mid-run by a weekly API limit and RESUMED FROM ITS
TRANSCRIPT — its round-2 harness `bf14_mk2.py`/`bf14_sw2.sh` survived intact, so nothing was
re-derived.)
**⚠️ THE PROCESS CORRECTION THAT MATTERS MORE THAN THE MATCH (→ §80): A DO-NOT-RE-BUY ENTRY IS
SCOPED TO ITS BASE, NOT TO THE FUNCTION.** Three of round 1's ~40 carefully-measured negatives
**INVERTED** on round 2's base — the same edit (`qsingle23`) measured **1,040 mismatched on the
45-base and 11 on the 21-base**. Re-testing the round-1 negative list cost **~20 seconds** and
produced **three of the seven winning levers**. **A do-not-re-buy table records `(edit, base) →
result`, NOT `edit → useless`; after any lever that moves the base materially, RE-RUN THE NEGATIVE
LIST.** This retroactively qualifies every such table in the cookbook (§45, §60b, §75a, §76, §78,
§79) — they are starting hypotheses at the base where they were taken, not closed questions.
Concrete instance: round 1 measured "removing the `va→$t2` pin costs 4% elsewhere" ⇒ *keep the pin*;
on a base with `c0..c3` at function scope, **removing those pins is worth 21→13** — opposite
conclusion, same experiment.
**MY FLAGGED "#1 MOVE" LOST, AND THE FAILURE IS THE FINDING.** I briefed variable REUSE (§45-A /
RC-14) as the #1 lever because it took `func_8017F510` from 97→10. Swept in full here: **every
merge lost, 43–3294 across 8 merges.** Reason: **the TRI and QUAD grants did not differ by RANK,
they differed by IDENTITY — two independent allocno sets, and re-ranking inside one set cannot fix
a two-set problem.** Diagnose ranking-vs-identity BEFORE reaching for a merge. The actual fix
(`s32 c0,c1,c2,c3;` at **function** scope, 33→21) was **read off the two matched relatives**
(`b5:310`, `b4:338`) and confirmed against the target (its TRI grants are identical to its QUAD
grants) — **the 4th time today that reading a matched relative beat the clever lever.**
**THE PIN'S HIDDEN COST, WITH A CITATION:** `combine_regs`' hard-register branch
(`local-alloc.c:1795`, reached from `:1295` with `already_dead == 0`) records the pinned register in
**`qty_phys_sugg` UNCONDITIONALLY — no death guard**. A pin doesn't merely *prefer* a register, it
invites local-alloc to tie producer chains into it. New cure **R7**: a zero-byte `__asm__` ref that
keeps the pinned value LIVE PAST the temp so `find_free_reg` can't honour the suggestion — closed
the last 2 ins, and was necessary because `c1→$a0` is uniquely load-bearing (it is what spills
`r1lo`; every alternative pin lost 64 ins).
**§78's ATTRIBUTION PRIMITIVE RUN AND REPRODUCED:** under `-fno-schedule-insns`,
`-fno-schedule-insns2`, and both, the order was **unchanged** ⇒ the rgb-accumulator transposition
was never a `sched.c` decision (a 3-statement accumulator pins the value, so no scheduler *could*
hoist the `or` above the `sw`). Changing the grant fixed the order for free.
**COLD-START ECONOMICS, NOW COMPLETE:** round 1 = decode + exact length + exact frame + 99.06%;
round 2 = the last 45, and far cheaper than round 1. **Budget TWO passes at this size; do not read a
99% round-1 result as a stall.** Also found a **5th original-source copy-paste artefact** (QUAD
vertex-1 box-3 y-axis accumulates into `a2v` while its kill branch still says `a3v`).
> **🛑 SESSION-19 CLOSING CHECKPOINT (2026-07-25, Opus 5 @ High) — REFRESHED mid-session; supersedes
> both the SESSION-18 block and the earlier SESSION-19 block (which was written before the
> ENGINE_SHB / class-B / dedup_extend-bug work and went stale). Fresh session safe here.**
+840 -1
View File
@@ -3276,7 +3276,846 @@ DEFINE_func_8017BEB4() /* dedup: shared engine-core @0x8017BEB4 (src/shared) */
INCLUDE_ASM("asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C", func_8017BEBC);
INCLUDE_ASM("asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C", func_8017BF14);
#include "common.h"
#include "/home/musashi/bfm-decomp/src/shared/engine_types.h"
/* ===========================================================================
* func_8017BF14 -- 4,763 ins, ov_SC03_116 (behemoth #4). *** MATCH ***
*
* STATUS (2026-07-25, session 20 round 2, gcc-2.7.2 pinned triple):
* python3 tools/match_one.py func_8017BF14 --c <this file> \
* --asm-subdir asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C
* -> MATCH (4763 ins) func_8017BF14 [reproduced 3x, private work dirs]
* LENGTH EXACT | OPCODE HISTOGRAM EXACT (L1 = 0) | STACK FRAME EXACT
* (all 127 slots at the target's offsets, frame 0x360)
* masked index-wise diff: 0 / 4763 mismatched.
* `match_one` is the CANDIDATE gate. The whole-binary SHA1 arbiter (G3/P9)
* is run by the coordinator and is the only thing that makes this final.
*
* Round 1 closed at 45/4763 mismatched. Round 2 took 45 -> 37 -> 33 -> 21
* -> 11 -> 3 -> 2 -> 0. See .run/giants/s19_bf14_report2.md.
*
* WHAT IT IS
* The *four*-light-box variant of the volumetric-light renderer whose
* 3-box sibling func_8017D960 (3,338 ins, ov_SC03_090) is MATCHED, and whose
* unlit ancestor func_8017BEBC (ov_SC03_099) is MATCHED. Same family, same
* skeleton; this is the biggest member.
*
* Signature: func_8017BF14(s32 arg0, s32 lim). Unlike every other member of
* the family this one is a LEAF -- 0 callees. The 3-call prologue
* (func_800491EC / func_800547D8 / func_80052E38) of the siblings is gone;
* `lim` arrives as arg1 (spilled to 0xB0). That is why the frame has no
* 0x10 argument area (tmpxy[] starts at sp+0x00) and no $ra save.
*
* Per part (stride 0x14, outer loop): build the 8-corner AABB in box[],
* rtpt/rtpt + rtps/rtps -> sxy[8], stszotz -> g.otz, reject on
* `lim >= g.otz`, then screen-space bbox reject on X (-0xA0..0xA1) and
* Y (-0x6E..0x6F).
* Per prim (stride 0xC, inner loop): rtpt the 3 vertices, stflg mask
* 0x7F85E000, nclip, stopz > 0, then a 4-way range tree on `code = w & 7`
* that keeps ONLY codes 6,7 (tri) and 2,3 (quad); 0,1,4,5 fall through to
* the loop tail.
* Per drawn poly: screen bbox reject, then each vertex is tested against
* FOUR axis-aligned light boxes (flags f0..f3), and if any is lit a
* 0x00..0x80 attenuation per active box is computed, summed, biased +0x10
* and clamped to 0x80 -> a grey gouraud vertex colour.
* lit -> POLY_GT3 (0x28, tag 0x34000000, OT 0x9000000)
* POLY_GT4 (0x34, tag 0x3C000000, OT 0xC000000)
* unlit -> POLY_FT3 (0x20, OT 0x7000000) / POLY_FT4 (0x28, OT 0x9000000)
* with rgbc = (tp[0] & 0xFF000000) | 0x101010 <-- NOT black,
* unlike func_8017D960 where the unlit colour is plain black.
*
* THE FOUR LIGHT BOXES (stride 0x1C, {s32 enable; u16 cx,cy,cz; s32 range})
* D_80197C28 / D_80197C44 / D_80197C60 / D_80197C7C.
* Falloff geometry differs from the 3-box sibling: RLO = R - 0x200 (not
* -0x80) and the ramp is ((R - d) / 4) (not (R - d)), so the 0x80 ceiling is
* reached over a 0x200-wide band instead of 0x80. The `/ 4` is a SIGNED
* divide -- `bgez / addiu 3 / sra 2` -- not a shift.
*
* FIVE ORIGINAL-SOURCE COPY-PASTE ARTEFACTS, all byte-proven
* The 4th light box was bolted onto a copy of the 3-box source BY HAND and
* the hand edit was incomplete in five places. Each was read off the target
* and each removed a measured delta.
* (1) `r3lo = r2 - 0x200;` -- box 3's low radius is derived from box 2's
* RANGE VARIABLE, not from its own D_80197C88. Proven by the target's
* `addiu $t6, $s0, -0x200` reusing the register that box 2's `lw` filled;
* spelling it `D_80197C6C - 0x200` re-loads the global (+2 ins).
* (2) Only SIX of the eight radius variables are zero-initialised
* (r0,r1,r2,r0lo,r1lo,r2lo) -- r3/r3lo are left uninitialised, exactly
* the init list the 3-box version needed.
* (3) The ATTEN body is written out LONGHAND 7 times (3 tri vertices +
* 4 quad vertices). When box 3 was bolted on, the `R` of the z- and
* y-axis KILL tests was left as r2 in three of those copies:
* tri v0: z and y use r2 tri v2: z uses r2 all others use r3.
* Proven by the 24 `sll $v0,$s0,16` sites: 3 per group in box 2 plus
* exactly three extra at idx 1471, 1504 (tri v0) and 2178 (tri v2).
* (4) In the QUAD lit arm only, rgb2 and rgb3 take their `<< 16` term from
* c1, not from c2/c3. Proven by the target CSE-ing ONE
* `sll $a0, $a0, 16` and re-using $a0 for all three stores.
* (5) *** ROUND 2 *** In the QUAD arm's VERTEX-1 group only, the box-3
* ATTEN's Y-axis `else if` branch accumulates into a2v instead of a3v,
* while that same test's KILL branch still says a3v. Modelled by the
* ATTEN3W macro below (`AW` = the y-else destination). Byte-proof:
* idx 3875 addu $a2,$zero,$zero kill branch -> a3v ($a2) [agreed]
* idx 3889 sra $a3,$s2,7 else branch -> a2v ($a3) [was the
* last structural residual]
* Costs zero instructions; the four other quad/tri sites are NOT like
* this (each was measured -- putting the artefact anywhere else is +2).
*
* FRAME (0x360, leaf -- no $ra, no argument area)
* 0x000 tmpxy[4] | 0x010 box[8] | 0x050 sxy[8] |
* 0x090 g{otz,flag,opz,sz0..sz3} | 0x0B0 lim | 0x0B8 j | 0x0C0 i |
* 0x0C8 vd | 0x0D0 ot | 0x0D8 pkt | 0x0E0 f2 | 0x0E8 f3 |
* 0x0F0/0x0F8/0x100 x3,z3,y3 | 0x108 prim | 0x110 nprim | 0x118 vtx |
* 0x120 nparts | 0x128 part | 0x130..0x1D0 lo/hi bounds (21 s16 slots) |
* 0x1D8..0x230 cx0..cz3 (12) | 0x238 r0 | 0x240 r1lo | 0x248 r2lo |
* 0x250 r3lo | 0x288..0x2D0 the LICM-hoisted sign-extended bounds |
* 0x328/0x330 spilled vertex coords | 0x338..0x358 s0-s7,fp.
* *** THE SLOT ORDER IS THE DECLARATION-ORDER ORACLE (see L5). ***
*
* ---------------------------------------------------------------------------
* ROUND-1 LEVERS (kept; measured effect is byte-identical %, anchored)
*
* L1 `cb = (tp[0] & 0xFF000000) | 0x101010;` in both UNLIT arms.
* L2 box-3 ATTEN kill-register per copy (artefact 3) + `r3lo = r2 - 0x200`
* (artefact 1). 89.15% -> 96.96% shape; killed sra+11 / sll+9.
* L3 quad-lit rgb2/rgb3 use `c1 << 16` (artefact 4). Killed the last sll+2.
* L4 [SUPERSEDED BY R3] `s32 c0..c3` declared inside the two CULL blocks.
* L5 `s32 f0, f1, f2, f3;` MOVED TO IMMEDIATELY AFTER `u8 *pkt;`.
* Spilled pseudos get stack slots in PSEUDO-NUMBER order and pseudo
* numbers are handed out in DECLARATION order, so the target's stack
* layout is a direct read-out of its declaration order. After this one
* move ALL 127 stack slots agree with the target exactly.
* L6 `u32 rgbw;` per emit arm.
* L7 RC-15 zero-byte ref dial on `mny`, first statement of the TRI cull
* block -- flips my->$a3 / mny->$a2 to the target's grant.
* L8 `cb` 2-statement accumulator [SUPERSEDED BY R2]; quad-lit rgb word as a
* 3-statement accumulator [kept for rgb0/rgb1, SUPERSEDED for rgb2/rgb3
* by R5].
* L9 FOUR REGISTER PINS: va->$t2, w->$a1, f0->$s3, c1->$a0.
* Round 2 removed va and w (see R4); f0 and c1 REMAIN and are both
* load-bearing. This function has NO `jal`, so Sec.74's caller-saved
* pin-spanning-a-call hazard cannot arise -- that is why pins are usable
* on this family member and were a trap on the others.
*
* ---------------------------------------------------------------------------
* ROUND-2 LEVERS -- 45 -> 0. Metric is `match_one` MISMATCH COUNT (length is
* exact throughout, so the raw count is honest). Every number is measured.
*
* THE ONE MECHANISM BEHIND R1/R2/R4/R7. A `register __asm__` pin makes the
* variable a HARD REG in the RTL from the start. When a 1-death local temp is
* produced from, or consumed into, that hard reg, local-alloc.c's
* `combine_regs` takes its hard-register branch (local-alloc.c:1795-1820) and
* records the pinned register in `qty_phys_sugg` for the temp's quantity --
* UNCONDITIONALLY, there is no death guard on that path. The temp then lands
* in the pinned register and the operation is done IN PLACE. The target,
* whose variable is an ordinary pseudo (reg_qty == -1 for anything crossing a
* block), never gets that suggestion and keeps the temp in $v0/$v1.
* Three independent cures, all used here:
* (i) give the temp a NAMED variable with >1 death, so local-alloc.c:472
* (`reg_basic_block >= 0 && reg_n_deaths == 1`) refuses it a quantity
* and combine_regs bails at its very first test -> R1
* (ii) drop the pin, if the pin is not load-bearing -> R4
* (iii) keep the pinned value LIVE past the temp, so that
* find_free_reg cannot honour the suggestion -> R7
*
* R1 `CLAMP80S` -- the four-way attenuation sum gets its own named variable
* `sv`, used at BOTH c1 sites (2 deaths). Cure (i). 45 -> 37
* Target: `addu $v0,..; addu $v0,..; addu $v0,..; addiu <c>,$v0,0x10`.
* Without it the whole chain is tied into the pinned c1 ($a0).
* Only the two c1 sites: `sv` on all 7 sites collapses the frame (-64).
* R2 UNLIT arms store `cb | 0x101010` as an expression instead of doing
* `cb |= 0x101010` in place. `cb` is a function-scope global allocno, so
* the in-place form writes $a1; the expression form is a 1-death local
* that combine_regs ties to the DYING constant register $v1, which is
* what the target does. 37 -> 33
* R3 *** `s32 c0, c1, c2, c3;` AT FUNCTION SCOPE, not per cull block. ***
* Read straight off the two MATCHED relatives (func_8017D960 line 310,
* func_8017F510 line 338), and confirmed by the target itself: its TRI
* grants (c0=$t4, c1=$a0, c2=$t2) are IDENTICAL to its QUAD grants, which
* is only possible if both arms share one set of allocnos. Per-cull-block
* scope splits them into two independent allocno sets and the TRI set
* drifts. 33 -> 21
* Declaration POSITION is neutral (5 anchors swept, all 21).
* NOTE this REVERSES round-1's L4. L4 was correct on the round-1 base --
* it was supplying the extra local allocno that spills r1lo -- but R1+R2
* supply that pressure now, and the c1 pin does the rest.
* R4 DROP the `va->$t2` and `w->$a1` pins. With c0..c3 at function scope
* they are no longer load-bearing, and they were the sole cause of the
* prim-word producer ties (`andi`/`srl` written straight into $t2/$a1).
* Cure (ii). 21 -> 17 -> 13 (pair)
* Round 1 measured these as worth 4%; that was true of the round-1 base
* and is FALSE here. Base-dependence, not a contradiction.
* R5 QUAD lit rgb2/rgb3 revert to the SINGLE-EXPRESSION form (rgb0/rgb1 keep
* the 3-statement accumulator). The intermediates then become 1-death
* local temps that alternate $v0/$v1, which is what lets the target's
* store of the previous rgb word sit one slot LATER. 21 -> 11
* Doing it to all four, or to rgb0/rgb1 only, is worse (23 / 21).
* R6 Artefact 5 -- `ATTEN3W(a3v, a2v, ...)` at the QUAD vertex-1 site. 3 -> 2
* R7 RC-15 zero-byte ref `__asm__ __volatile__ ("" :: "r" (c1));` placed
* immediately after the TRI arm's rgb1 store. Cure (iii): it keeps the
* pinned c1 ($a0) live past `c1 << 16`, so find_free_reg cannot honour
* combine_regs' $a0 suggestion and the shift goes to $v0. 2 -> 0
* The c1 pin CANNOT simply be removed: it is what spills r1lo (dropping
* it, or moving it to any other colour, costs -64 length). Measured.
*
* SCHEDULER ATTRIBUTION (Sec.76 primitive, run in round 2, never run before on
* this function). Compiled with -fno-schedule-insns, with
* -fno-schedule-insns2, and with both. The store/shift transposition at
* idx 4643-4652 kept MY source order under all three. It was therefore never
* a `sched.c` decision: the ordering is a CONSEQUENCE of the register grant
* (a 3-statement accumulator pins the value in one register, so no scheduler
* could hoist the next `or` above the `sw`). R5 fixed it by changing the
* grant, exactly as Sec.78 predicts.
* =========================================================================== */
#define gte_ldv0(r0) __asm__ volatile ( \
"lwc2 $0, 0( %0 );" \
"lwc2 $1, 4( %0 )" \
: \
: "r"( r0 ) )
#define gte_ldv3(r0, r1, r2) __asm__ volatile ( \
"lwc2 $0, 0( %0 );" \
"lwc2 $1, 4( %0 );" \
"lwc2 $2, 0( %1 );" \
"lwc2 $3, 4( %1 );" \
"lwc2 $4, 0( %2 );" \
"lwc2 $5, 4( %2 )" \
: \
: "r"( r0 ), "r"( r1 ), "r"( r2 ) )
#define gte_ldv3c(r0) __asm__ volatile ( \
"lwc2 $0, 0( %0 );" \
"lwc2 $1, 4( %0 );" \
"lwc2 $2, 8( %0 );" \
"lwc2 $3, 12( %0 );" \
"lwc2 $4, 16( %0 );" \
"lwc2 $5, 20( %0 )" \
: \
: "r"( r0 ) )
#define gte_rtps() __asm__ volatile ("nop;nop;rtps")
#define gte_rtpt() __asm__ volatile ("nop;nop;rtpt")
#define gte_nclip() __asm__ volatile ("nop;nop;nclip")
#define gte_stsxy(r0) __asm__ volatile ( \
"swc2 $14, 0( %0 )" \
: \
: "r"( r0 ) \
: "memory" )
#define gte_stsxy3(r0, r1, r2) __asm__ volatile ( \
"swc2 $12, 0( %0 );" \
"swc2 $13, 0( %1 );" \
"swc2 $14, 0( %2 )" \
: \
: "r"( r0 ), "r"( r1 ), "r"( r2 ) \
: "memory" )
#define gte_stsxy3c(r0) __asm__ volatile ( \
"swc2 $12, 0( %0 );" \
"swc2 $13, 4( %0 );" \
"swc2 $14, 8( %0 )" \
: \
: "r"( r0 ) \
: "memory" )
#define gte_stsz3(r0, r1, r2) __asm__ volatile ( \
"swc2 $17, 0( %0 );" \
"swc2 $18, 0( %1 );" \
"swc2 $19, 0( %2 )" \
: \
: "r"( r0 ), "r"( r1 ), "r"( r2 ) \
: "memory" )
#define gte_stsz4(r0, r1, r2, r3) __asm__ volatile ( \
"swc2 $16, 0( %0 );" \
"swc2 $17, 0( %1 );" \
"swc2 $18, 0( %2 );" \
"swc2 $19, 0( %3 )" \
: \
: "r"( r0 ), "r"( r1 ), "r"( r2 ), "r"( r3 ) \
: "memory" )
#define gte_stszotz(r0) __asm__ volatile ( \
"mfc2 $12, $19;" \
"nop;" \
"sra $12, $12, 2;" \
"sw $12, 0( %0 )" \
: \
: "r"( r0 ) \
: "$12", "memory" )
#define gte_stflg(r0) __asm__ volatile ( \
"cfc2 $12, $31;" \
"nop;" \
"sw $12, 0( %0 )" \
: \
: "r"( r0 ) \
: "$12", "memory" )
#define gte_stopz(r0) __asm__ volatile ( \
"swc2 $24, 0( %0 )" \
: \
: "r"( r0 ) \
: "memory" )
/* ---- the two gouraud-textured packet layouts this function emits ---------- */
typedef struct {
u32 tag;
u32 rgb0; s16 x0, y0; u32 uv0;
u32 rgb1; s16 x1, y1; u32 uv1;
u32 rgb2; s16 x2, y2; u16 uv2, p2;
} PolyGT3; /* 0x28 */
typedef struct {
u32 tag;
u32 rgb0; s16 x0, y0; u32 uv0;
u32 rgb1; s16 x1, y1; u32 uv1;
u32 rgb2; s16 x2, y2; u16 uv2, p2;
u32 rgb3; s16 x3, y3; u16 uv3, p3;
} PolyGT4; /* 0x34 */
/* ---- the four light-volume descriptors (stride 0x1C) --------------------- */
extern s32 D_80197C28;
extern u16 D_80197C2C, D_80197C2E, D_80197C30;
extern s32 D_80197C34;
extern s32 D_80197C44;
extern u16 D_80197C48, D_80197C4A, D_80197C4C;
extern s32 D_80197C50;
extern s32 D_80197C60;
extern u16 D_80197C64, D_80197C66, D_80197C68;
extern s32 D_80197C6C;
extern s32 D_80197C7C;
extern u16 D_80197C80, D_80197C82, D_80197C84;
extern u16 D_80197C88;
/* ---- the box-containment test for one vertex against one light box ------- */
#define BOXTEST(F, X, Y, Z, LX, HX, LY, HY, LZ, HZ) \
if ((LX) < (X) && (X) < (HX) && (LY) < (Y) && (Y) < (HY) && (LZ) < (Z) && (Z) < (HZ)) F = 1
/* ---- the separable per-axis falloff, visited in x, z, y order ------------ */
#define ATTEN(A, F, X, Y, Z, CX, CY, CZ, R, RLO) \
A = 0; \
if (F) { \
d = (X) - (CX); if (d < 0) d = (CX) - (X); \
if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \
d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \
if ((R) < d) A = 0; \
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \
if ((R) < d) A = 0; \
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
}
#define ATTEN3W(A, AW, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \
A = 0; \
if (F) { \
d = (X) - (CX); if (d < 0) d = (CX) - (X); \
if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \
d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \
if ((RZ) < d) A = 0; \
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \
if ((RY) < d) A = 0; \
else if ((RLO) < d) AW = (A * (((R) - d) / 4)) >> 7; \
}
#define ATTEN3(A, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \
A = 0; \
if (F) { \
d = (X) - (CX); if (d < 0) d = (CX) - (X); \
if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \
d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \
if ((RZ) < d) A = 0; \
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \
if ((RY) < d) A = 0; \
else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \
}
#define CLAMP80(C, A0, A1, A2, A3) C = (A0) + (A1) + (A2) + (A3) + 0x10; if ((C) > 0x80) C = 0x80
#define CLAMP80S(C, A0, A1, A2, A3) sv = (A0) + (A1) + (A2) + (A3); C = sv + 0x10; if ((C) > 0x80) C = 0x80
void func_8017BF14(s32 arg0, s32 lim)
{
typedef struct { u32 w0, w1, w2; } Prim;
extern u8 *D_800A5E60;
extern u8 D_800A6610[];
extern u8 D_800AF630[];
DVECTOR2 tmpxy[4];
SVECTOR2 box[8];
SVECTOR2 sxy[8];
struct { long otz, flag, opz, sz0, sz1, sz2, sz3; } g;
s32 j;
u32 i;
u8 *vd;
u32 ot;
u8 *pkt;
register s32 f0 __asm__("$19");
s32 f1, f2, f3;
s16 x3, z3, y3;
Prim *prim;
u32 nprim;
u8 *vtx;
s32 nparts;
Part *part;
s16 lo0x, hi0x, lo0y, hi0y, lo0z, hi0z;
s16 lo1x, hi1x, lo1y, hi1y, lo1z, hi1z;
s16 lo2x, hi2x, lo2y, hi2y, lo2z, hi2z;
s16 lo3x, hi3x, lo3y, hi3y, lo3z, hi3z;
s16 cx0, cy0, cz0, cx1, cy1, cz1, cx2, cy2, cz2, cx3, cy3, cz3;
s16 r0;
s16 r1;
s16 r2;
s16 r3;
s16 r0lo;
s16 r1lo;
s16 r2lo;
s16 r3lo;
u8 *va, *vb, *vc;
u32 w;
s32 code;
u32 vw, vzw;
u32 wx, wy, wz;
s32 xa32, xb32, t32;
s32 xmn1, xmx1, xmn2, xmx2;
s32 mnc, mxc;
s16 my, mny, mx, mn;
u8 *base;
s16 x0, y0, z0, x1, y1, z1, x2, y2, z2;
s32 a0v, a1v, a2v, a3v;
register s32 c1 __asm__("$4"); s32 c0, c2, c3;
s32 d;
u32 *tp;
u32 uvw;
u32 cb;
s32 sv;
base = D_800AF630;
r2lo = 0;
r1lo = 0;
r0lo = 0;
r2 = 0;
r1 = 0;
r0 = 0;
if (D_80197C28) {
cx0 = D_80197C2C;
cy0 = D_80197C2E;
r0lo = D_80197C34 - 0x200;
r0 = D_80197C34;
cz0 = D_80197C30;
} else {
cz0 = 0x6000;
cy0 = 0x6000;
cx0 = 0x6000;
}
if (D_80197C44) {
cx1 = D_80197C48;
cy1 = D_80197C4A;
r1 = D_80197C50;
r1lo = D_80197C50 - 0x200;
cz1 = D_80197C4C;
} else {
cz1 = 0x6000;
cy1 = 0x6000;
cx1 = 0x6000;
}
if (D_80197C60) {
cx2 = D_80197C64;
cy2 = D_80197C66;
r2 = D_80197C6C;
r2lo = D_80197C6C - 0x200;
cz2 = D_80197C68;
} else {
cz2 = 0x6000;
cy2 = 0x6000;
cx2 = 0x6000;
}
if (D_80197C7C) {
cx3 = D_80197C80;
cy3 = D_80197C82;
r3lo = r2 - 0x200;
r3 = D_80197C88;
cz3 = D_80197C84;
} else {
cz3 = 0x6000;
cy3 = 0x6000;
cx3 = 0x6000;
}
lo0x = cx0 - r0; hi0x = cx0 + r0;
lo0y = cy0 - r0; hi0y = cy0 + r0;
lo0z = cz0 - r0; hi0z = cz0 + r0;
lo1x = cx1 - r1; hi1x = cx1 + r1;
lo1y = cy1 - r1; hi1y = cy1 + r1;
lo1z = cz1 - r1; hi1z = cz1 + r1;
lo2x = cx2 - r2; hi2x = cx2 + r2;
lo2y = cy2 - r2; hi2y = cy2 + r2;
lo2z = cz2 - r2; hi2z = cz2 + r2;
lo3x = cx3 - r3; hi3x = cx3 + r3;
lo3y = cy3 - r3; hi3y = cy3 + r3;
lo3z = cz3 - r3; hi3z = cz3 + r3;
pkt = D_800A5E60;
part = *(Part **)(arg0 + 0xC);
nparts = *(s32 *)(*(s32 *)(arg0 + 8) + 8);
vtx = *(u8 **)(*(s32 *)(arg0 + 8) + 0x10);
ot = (u32)&D_800A6610[(*(u16 *)(base + 0xA3D2)) << 14];
for (j = 0; j < nparts; j++, part++) {
wx = part->xx;
mn = wx;
mx = wx >> 16;
wy = part->yy;
mny = wy;
my = wy >> 16;
wz = part->zz;
box[0].vx = mn; box[0].vy = mny;
box[1].vx = mx; box[1].vy = mny;
box[2].vx = mn; box[2].vy = mny;
box[3].vx = mx; box[3].vy = mny;
box[4].vx = mn; box[4].vy = my;
box[5].vx = mx; box[5].vy = my;
box[6].vx = mn; box[6].vy = my;
box[7].vx = mx; box[7].vy = my;
wy = wz >> 16;
box[0].vz = wz;
box[1].vz = wz;
box[4].vz = wz;
box[5].vz = wz;
box[2].vz = wy;
box[3].vz = wy;
box[6].vz = wy;
box[7].vz = wy;
gte_ldv3c(&box[0]);
gte_rtpt();
gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]);
gte_ldv0(&box[3]);
gte_rtps();
gte_stsxy(&sxy[3]);
gte_ldv3c(&box[4]);
gte_rtpt();
gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]);
gte_ldv0(&box[7]);
gte_rtps();
gte_stsxy(&sxy[7]);
gte_stszotz(&g.otz);
if (lim >= g.otz) {
xa32 = sxy[0].vx;
xb32 = sxy[1].vx;
if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; }
t32 = sxy[2].vx;
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
t32 = sxy[3].vx;
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
xa32 = sxy[4].vx;
xb32 = sxy[5].vx;
if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; }
t32 = sxy[6].vx;
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
t32 = sxy[7].vx;
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
mnc = xmn1;
if (xmn2 < xmn1) mnc = xmn2;
mxc = xmx1;
if (mxc < xmx2) mxc = xmx2;
if ((s16)mxc >= -0xA0 && (s16)mnc < 0xA1) {
xa32 = sxy[0].vy;
xb32 = sxy[1].vy;
if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; }
t32 = sxy[2].vy;
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
t32 = sxy[3].vy;
if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32;
xa32 = sxy[4].vy;
xb32 = sxy[5].vy;
if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; }
t32 = sxy[6].vy;
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
t32 = sxy[7].vy;
if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32;
mnc = xmn1;
if (xmn2 < xmn1) mnc = xmn2;
mxc = xmx1;
if (mxc < xmx2) mxc = xmx2;
if ((s16)mxc >= -0x6E && (s16)mnc < 0x6F) {
nprim = part->nprim;
prim = (Prim *)part->prim;
for (i = 0; i < nprim; i++, prim++) {
w = prim->w1;
va = vtx + (w & 0xFFFF);
vb = vtx + (w >> 16);
w = prim->w2;
vc = vtx + (w & 0xFFFF);
w = w >> 16;
gte_ldv3(va, vb, vc);
gte_rtpt();
gte_stflg(&g.flag);
if (!(g.flag & 0x7F85E000)) {
gte_nclip();
code = w & 7;
vd = vtx + (w & 0xFFF8);
gte_stopz(&g.opz);
if (g.opz > 0) {
switch (code) {
case 6:
case 7:
/* ---------------- TRI (FT3 / GT3) ---------------- */
gte_stsxy3c(&tmpxy[0]);
gte_stsz3(&g.sz0, &g.sz1, &g.sz2);
if (tmpxy[0].vx > tmpxy[1].vx) { mx = tmpxy[0].vx; mn = tmpxy[1].vx; }
else { mn = tmpxy[0].vx; mx = tmpxy[1].vx; }
if (tmpxy[2].vx > mx) mx = tmpxy[2].vx;
else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx;
if (mx >= -0xA0 && mn < 0xA1) {
if (tmpxy[0].vy > tmpxy[1].vy) { my = tmpxy[0].vy; mny = tmpxy[1].vy; }
else { mny = tmpxy[0].vy; my = tmpxy[1].vy; }
if (tmpxy[2].vy > my) my = tmpxy[2].vy;
else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy;
if (my >= -0x6E && mny < 0x6F) {
s32 za, zb;
__asm__ __volatile__ ("" :: "r" (mny));
if (g.sz0 > g.sz1) { za = g.sz0; if (za < g.sz2) za = g.sz2; }
else { za = g.sz1; if (za < g.sz2) za = g.sz2; }
g.opz = za;
f3 = 0; f2 = 0; f1 = 0; f0 = 0;
vw = *(u32 *)va;
vzw = *(u32 *)(va + 4);
x0 = vw; y0 = vw >> 16; z0 = vzw;
BOXTEST(f0, x0, y0, z0, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
BOXTEST(f1, x0, y0, z0, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
BOXTEST(f2, x0, y0, z0, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
BOXTEST(f3, x0, y0, z0, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
vw = *(u32 *)vb;
vzw = *(u32 *)(vb + 4);
x1 = vw; y1 = vw >> 16; z1 = vzw;
BOXTEST(f0, x1, y1, z1, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
BOXTEST(f1, x1, y1, z1, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
BOXTEST(f2, x1, y1, z1, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
BOXTEST(f3, x1, y1, z1, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
vw = *(u32 *)vc;
vzw = *(u32 *)(vc + 4);
x2 = vw; y2 = vw >> 16; z2 = vzw;
BOXTEST(f0, x2, y2, z2, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
BOXTEST(f1, x2, y2, z2, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
BOXTEST(f2, x2, y2, z2, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
BOXTEST(f3, x2, y2, z2, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
if (f0 | f1 | f2 | f3) {
u32 *otp;
u32 rgbw;
ATTEN(a0v, f0, x0, y0, z0, cx0, cy0, cz0, r0, r0lo);
ATTEN(a1v, f1, x0, y0, z0, cx1, cy1, cz1, r1, r1lo);
ATTEN(a2v, f2, x0, y0, z0, cx2, cy2, cz2, r2, r2lo);
ATTEN3(a3v, f3, x0, y0, z0, cx3, cy3, cz3, r3, r3lo, r2, r2);
CLAMP80(c0, a0v, a1v, a2v, a3v);
ATTEN(a0v, f0, x1, y1, z1, cx0, cy0, cz0, r0, r0lo);
ATTEN(a1v, f1, x1, y1, z1, cx1, cy1, cz1, r1, r1lo);
ATTEN(a2v, f2, x1, y1, z1, cx2, cy2, cz2, r2, r2lo);
ATTEN3(a3v, f3, x1, y1, z1, cx3, cy3, cz3, r3, r3lo, r3, r3);
CLAMP80S(c1, a0v, a1v, a2v, a3v);
ATTEN(a0v, f0, x2, y2, z2, cx0, cy0, cz0, r0, r0lo);
ATTEN(a1v, f1, x2, y2, z2, cx1, cy1, cz1, r1, r1lo);
ATTEN(a2v, f2, x2, y2, z2, cx2, cy2, cz2, r2, r2lo);
ATTEN3(a3v, f3, x2, y2, z2, cx3, cy3, cz3, r3, r3lo, r2, r3);
CLAMP80(c2, a0v, a1v, a2v, a3v);
*(u32 *)&((PolyGT3 *)pkt)->x0 = *(u32 *)&tmpxy[0];
*(u32 *)&((PolyGT3 *)pkt)->x1 = *(u32 *)&tmpxy[1];
*(u32 *)&((PolyGT3 *)pkt)->x2 = *(u32 *)&tmpxy[2];
tp = (u32 *)prim->w0;
cb = 0x34000000;
rgbw = (c0 | cb) | (c0 << 8) | (c0 << 16);
((PolyGT3 *)pkt)->rgb0 = rgbw;
rgbw = (c1 | cb) | (c1 << 8) | (c1 << 16);
((PolyGT3 *)pkt)->rgb1 = rgbw;
__asm__ __volatile__ ("" :: "r" (c1));
rgbw = (c2 | cb) | (c2 << 8) | (c2 << 16);
((PolyGT3 *)pkt)->rgb2 = rgbw;
((PolyGT3 *)pkt)->uv0 = tp[1];
((PolyGT3 *)pkt)->uv1 = tp[2];
((PolyGT3 *)pkt)->uv2 = tp[3];
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000;
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
pkt += 0x28;
} else {
u32 *otp;
u32 rgbw;
*(u32 *)&((PolyFT3 *)pkt)->x0 = *(u32 *)&tmpxy[0];
*(u32 *)&((PolyFT3 *)pkt)->x1 = *(u32 *)&tmpxy[1];
*(u32 *)&((PolyFT3 *)pkt)->x2 = *(u32 *)&tmpxy[2];
tp = (u32 *)prim->w0;
cb = tp[0] & 0xFF000000;
((PolyFT3 *)pkt)->rgbc = cb | 0x101010;
((PolyFT3 *)pkt)->uvc0 = tp[1];
((PolyFT3 *)pkt)->uvp1 = tp[2];
((PolyFT3 *)pkt)->uv2 = tp[3];
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000;
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
pkt += 0x20;
}
}
}
break;
case 2:
case 3:
/* ---------------- QUAD (FT4 / GT4) ---------------- */
gte_stsxy3c(&tmpxy[0]);
gte_ldv0(vd);
gte_rtps();
if (tmpxy[0].vx > tmpxy[1].vx) { mx = tmpxy[0].vx; mn = tmpxy[1].vx; }
else { mn = tmpxy[0].vx; mx = tmpxy[1].vx; }
if (tmpxy[2].vx > mx) mx = tmpxy[2].vx;
else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx;
if (tmpxy[0].vy > tmpxy[1].vy) { my = tmpxy[0].vy; mny = tmpxy[1].vy; }
else { mny = tmpxy[0].vy; my = tmpxy[1].vy; }
if (tmpxy[2].vy > my) my = tmpxy[2].vy;
else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy;
gte_stflg(&g.flag);
if (!(g.flag & 0x7F85E000)) {
gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3);
gte_stsxy((long *)&((PolyFT4 *)pkt)->x3);
if (((PolyFT4 *)pkt)->x3 < mn) mn = ((PolyFT4 *)pkt)->x3;
else if (mx < ((PolyFT4 *)pkt)->x3) mx = ((PolyFT4 *)pkt)->x3;
if (mx >= -0xA0 && mn < 0xA1) {
if (((PolyFT4 *)pkt)->y3 < mny) mny = ((PolyFT4 *)pkt)->y3;
else if (my < ((PolyFT4 *)pkt)->y3) my = ((PolyFT4 *)pkt)->y3;
if (my >= -0x6E && mny < 0x6F) {
s32 za, zb;
zb = g.sz2;
if (zb < g.sz3) zb = g.sz3;
za = g.sz0;
if (za < g.sz1) za = g.sz1;
if (za < zb) za = zb;
g.opz = za;
f3 = 0; f2 = 0; f1 = 0; f0 = 0;
vw = *(u32 *)va;
vzw = *(u32 *)(va + 4);
x0 = vw; y0 = vw >> 16; z0 = vzw;
BOXTEST(f0, x0, y0, z0, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
BOXTEST(f1, x0, y0, z0, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
BOXTEST(f2, x0, y0, z0, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
BOXTEST(f3, x0, y0, z0, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
vw = *(u32 *)vb;
vzw = *(u32 *)(vb + 4);
x1 = vw; y1 = vw >> 16; z1 = vzw;
BOXTEST(f0, x1, y1, z1, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
BOXTEST(f1, x1, y1, z1, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
BOXTEST(f2, x1, y1, z1, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
BOXTEST(f3, x1, y1, z1, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
vw = *(u32 *)vc;
vzw = *(u32 *)(vc + 4);
x2 = vw; y2 = vw >> 16; z2 = vzw;
BOXTEST(f0, x2, y2, z2, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
BOXTEST(f1, x2, y2, z2, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
BOXTEST(f2, x2, y2, z2, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
BOXTEST(f3, x2, y2, z2, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
vw = *(u32 *)vd;
vzw = *(u32 *)(vd + 4);
x3 = vw; y3 = vw >> 16; z3 = vzw;
BOXTEST(f0, x3, y3, z3, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z);
BOXTEST(f1, x3, y3, z3, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z);
BOXTEST(f2, x3, y3, z3, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z);
BOXTEST(f3, x3, y3, z3, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z);
if (f0 | f1 | f2 | f3) {
u32 *otp;
u32 rgbw;
ATTEN(a0v, f0, x0, y0, z0, cx0, cy0, cz0, r0, r0lo);
ATTEN(a1v, f1, x0, y0, z0, cx1, cy1, cz1, r1, r1lo);
ATTEN(a2v, f2, x0, y0, z0, cx2, cy2, cz2, r2, r2lo);
ATTEN3(a3v, f3, x0, y0, z0, cx3, cy3, cz3, r3, r3lo, r3, r3);
CLAMP80(c0, a0v, a1v, a2v, a3v);
ATTEN(a0v, f0, x1, y1, z1, cx0, cy0, cz0, r0, r0lo);
ATTEN(a1v, f1, x1, y1, z1, cx1, cy1, cz1, r1, r1lo);
ATTEN(a2v, f2, x1, y1, z1, cx2, cy2, cz2, r2, r2lo);
ATTEN3W(a3v, a2v, f3, x1, y1, z1, cx3, cy3, cz3, r3, r3lo, r3, r3);
CLAMP80S(c1, a0v, a1v, a2v, a3v);
ATTEN(a0v, f0, x2, y2, z2, cx0, cy0, cz0, r0, r0lo);
ATTEN(a1v, f1, x2, y2, z2, cx1, cy1, cz1, r1, r1lo);
ATTEN(a2v, f2, x2, y2, z2, cx2, cy2, cz2, r2, r2lo);
ATTEN3(a3v, f3, x2, y2, z2, cx3, cy3, cz3, r3, r3lo, r3, r3);
CLAMP80(c2, a0v, a1v, a2v, a3v);
ATTEN(a0v, f0, x3, y3, z3, cx0, cy0, cz0, r0, r0lo);
ATTEN(a1v, f1, x3, y3, z3, cx1, cy1, cz1, r1, r1lo);
ATTEN(a2v, f2, x3, y3, z3, cx2, cy2, cz2, r2, r2lo);
ATTEN3(a3v, f3, x3, y3, z3, cx3, cy3, cz3, r3, r3lo, r3, r3);
CLAMP80(c3, a0v, a1v, a2v, a3v);
*(u32 *)&((PolyGT4 *)pkt)->x0 = *(u32 *)&tmpxy[0];
*(u32 *)&((PolyGT4 *)pkt)->x1 = *(u32 *)&tmpxy[1];
*(u32 *)&((PolyGT4 *)pkt)->x2 = *(u32 *)&tmpxy[2];
gte_stsxy((long *)&((PolyGT4 *)pkt)->x3);
tp = (u32 *)prim->w0;
cb = 0x3C000000;
rgbw = c0 | cb; rgbw |= c0 << 8; rgbw |= c0 << 16;
((PolyGT4 *)pkt)->rgb0 = rgbw;
rgbw = c1 | cb; rgbw |= c1 << 8; rgbw |= c1 << 16;
((PolyGT4 *)pkt)->rgb1 = rgbw;
rgbw = (c2 | cb) | (c2 << 8) | (c1 << 16);
((PolyGT4 *)pkt)->rgb2 = rgbw;
rgbw = (c3 | cb) | (c3 << 8) | (c1 << 16);
((PolyGT4 *)pkt)->rgb3 = rgbw;
((PolyGT4 *)pkt)->uv0 = tp[1];
((PolyGT4 *)pkt)->uv1 = tp[2];
uvw = tp[3];
((PolyGT4 *)pkt)->uv2 = uvw;
((PolyGT4 *)pkt)->uv3 = uvw >> 16;
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0xC000000;
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
pkt += 0x34;
} else {
u32 *otp;
u32 rgbw;
*(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0];
*(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1];
*(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2];
tp = (u32 *)prim->w0;
cb = tp[0] & 0xFF000000;
((PolyFT4 *)pkt)->rgbc = cb | 0x101010;
((PolyFT4 *)pkt)->uvc0 = tp[1];
((PolyFT4 *)pkt)->uvp1 = tp[2];
uvw = tp[3];
((PolyFT4 *)pkt)->uv2 = uvw;
((PolyFT4 *)pkt)->uv3 = uvw >> 16;
otp = (u32 *)(((g.opz >> 2) << 2) + ot);
*(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000;
*otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF);
pkt += 0x28;
}
}
}
}
break;
}
}
}
}
}
}
}
}
D_800A5E60 = pkt;
}
extern s32 D_8012704C;