diff --git a/.run/giants/bf14_hdr.py b/.run/giants/bf14_hdr.py new file mode 100644 index 0000000000..885ffb5899 --- /dev/null +++ b/.run/giants/bf14_hdr.py @@ -0,0 +1,10 @@ +#!/usr/bin/env python3 +"""Splice a new dossier header (lines 4..135 of the b1-derived draft) into the +winning round-2 draft. usage: bf14_hdr.py """ +import sys +src = open(sys.argv[1]).read().splitlines(True) +hdr = open(sys.argv[2]).read() +assert src[3].startswith('/* ====='), src[3] +assert src[134].rstrip().endswith('=== */'), src[134] +open(sys.argv[3], 'w').write(''.join(src[:3]) + hdr + ''.join(src[135:])) +print('wrote', sys.argv[3]) diff --git a/.run/giants/bf14_mk2.py b/.run/giants/bf14_mk2.py new file mode 100644 index 0000000000..ef5d646acf --- /dev/null +++ b/.run/giants/bf14_mk2.py @@ -0,0 +1,696 @@ +#!/usr/bin/env python3 +"""Round-2 lever generator for func_8017BF14. + +Same contract as bf14_mk.py: EVERY transformation asserts it actually applied, +so a "neutral" reading can never be a silent no-op. + +usage: bf14_mk2.py lever [lever ...] +""" +import sys, re + +LEVERS = {} +def lever(fn): + LEVERS[fn.__name__] = fn + return fn + +def _one(src, old, new, n=1): + c = src.count(old) + assert c == n, 'expected %d of %r, found %d' % (n, old, c) + return src.replace(old, new) + +def _all(src, old, new, n): + c = src.count(old) + assert c == n, 'expected %d of %r, found %d' % (n, old, c) + return src.replace(old, new) + +# -------------------------------------------------------------------------- +# UNPIN levers -- the b1 base carries four pins baked in. +# -------------------------------------------------------------------------- +@lever +def unpin_va(src): + return _one(src, ' register u8 *va __asm__("$10");\n u8 *vb, *vc;\n', + ' u8 *va, *vb, *vc;\n') +@lever +def unpin_w(src): + return _one(src, ' register u32 w __asm__("$5");\n s32 code;\n', + ' u32 w;\n s32 code;\n') +@lever +def unpin_f0(src): + return _one(src, ' register s32 f0 __asm__("$19");\n s32 f1, f2, f3;\n', + ' s32 f0, f1, f2, f3;\n') +@lever +def unpin_c1(src): + return _all(src, 'register s32 c1 __asm__("$4"); s32 c0, c2, c3;', + 's32 c0, c1, c2, c3;', 2) + +# --- re-pin at other registers ------------------------------------------- +def _repin_c(src, spec): + """spec like 'c0=12,c2=10' -> pin those, leave the rest plain.""" + want = dict(kv.split('=') for kv in spec.split(',')) + out = [] + for nm in ('c0', 'c1', 'c2', 'c3'): + if nm in want: + out.append('register s32 %s __asm__("$%s");' % (nm, want[nm])) + plain = [nm for nm in ('c0', 'c1', 'c2', 'c3') if nm not in want] + if plain: + out.append('s32 %s;' % ', '.join(plain)) + new = ' '.join(out) + for old in ('register s32 c1 __asm__("$4"); s32 c0, c2, c3;', + 's32 c0, c1, c2, c3;'): + if src.count(old) in (1, 2): + return src.replace(old, new) + raise AssertionError('no c-decl found') + +# -------------------------------------------------------------------------- +# rgb emit-word FORM levers +# -------------------------------------------------------------------------- +_CHAIN_Q = [ + ('rgbw = c0 | cb; rgbw |= c0 << 8; rgbw |= c0 << 16;', + 'rgbw = (c0 | cb) | (c0 << 8) | (c0 << 16);'), + ('rgbw = c1 | cb; rgbw |= c1 << 8; rgbw |= c1 << 16;', + 'rgbw = (c1 | cb) | (c1 << 8) | (c1 << 16);'), + ('rgbw = c2 | cb; rgbw |= c2 << 8; rgbw |= c1 << 16;', + 'rgbw = (c2 | cb) | (c2 << 8) | (c1 << 16);'), + ('rgbw = c3 | cb; rgbw |= c3 << 8; rgbw |= c1 << 16;', + 'rgbw = (c3 | cb) | (c3 << 8) | (c1 << 16);'), +] + +@lever +def qsingle(src): + """QUAD lit arm: revert the 3-statement accumulator to ONE expression.""" + for a, b in _CHAIN_Q: + src = _one(src, a, b) + return src + +@lever +def qsingle23(src): + """QUAD lit arm: single-expression for rgb2/rgb3 ONLY (keep 0/1 chained).""" + for a, b in _CHAIN_Q[2:]: + src = _one(src, a, b) + return src + +@lever +def qsingle01(src): + for a, b in _CHAIN_Q[:2]: + src = _one(src, a, b) + return src + +_RGBW_RE = re.compile( + r'rgbw = (?P[^;]+?);(?P\s*(?:rgbw \|= [^;]+;\s*)*)' + r'\(\(PolyGT(?P[34]) \*\)pkt\)->rgb(?P\d) = rgbw;') + +def _direct(src, poly): + """collapse `rgbw = ...; rgbw |= ...; pkt->rgbN = rgbw;` into one store.""" + hits = [0] + def rep(m): + if m.group('n') != poly: + return m.group(0) + expr = m.group('e') + for extra in re.findall(r'rgbw \|= ([^;]+);', m.group('mid')): + expr = '(%s) | %s' % (expr, extra) + hits[0] += 1 + return '((PolyGT%s *)pkt)->rgb%s = %s;' % (poly, m.group('k'), expr) + out = _RGBW_RE.sub(rep, src) + assert hits[0] == int(poly), 'direct(GT%s): %d sites' % (poly, hits[0]) + return out + +@lever +def qdirect(src): + """QUAD lit arm: no rgbw temp at all -- store the expression directly. + Each rgb word then becomes a 1-death LOCAL temp, so local-alloc places + them independently and they can alternate $v0/$v1 the way the target does.""" + return _direct(src, '4') + +@lever +def tdirect2(src): + """TRI lit arm: same, no rgbw temp.""" + return _direct(src, '3') + +# -------------------------------------------------------------------------- +# residual (c) part 1: the UNLIT rgbc word. +# target: and $a1,$v0,$a2 / or $v1,$a1,$v1 / sw $v1 +# mine : and $a1,$v0,$a2 / or $a1,$a1,$v1 / sw $a1 +# `cb` is a function-scope global allocno ($a1). `cb |= 0x101010` writes it +# in place. The target instead stores `cb | 0x101010` as a 1-death LOCAL +# temp, which combine_regs ties to the DYING constant register ($v1). +# -------------------------------------------------------------------------- +@lever +def cb_expr(src): + """unlit arms: `pkt->rgbc = cb | 0x101010;` (drop the `cb |=` statement).""" + out, n = re.subn( + r'cb = tp\[0\] & 0xFF000000;\s*\n\s*cb \|= 0x101010;\s*\n(\s*)' + r'\(\(PolyFT(\d) \*\)pkt\)->rgbc = cb;', + lambda m: ('cb = tp[0] & 0xFF000000;\n%s((PolyFT%s *)pkt)->rgbc' + ' = cb | 0x101010;' % (m.group(1), m.group(2))), src) + assert n == 2, n + return out + +@lever +def cb_expr_one(src): + """unlit arms: the whole thing as ONE expression, no `cb` at all.""" + out, n = re.subn( + r'cb = tp\[0\] & 0xFF000000;\s*\n\s*cb \|= 0x101010;\s*\n(\s*)' + r'\(\(PolyFT(\d) \*\)pkt\)->rgbc = cb;', + lambda m: ('((PolyFT%s *)pkt)->rgbc = (tp[0] & 0xFF000000) | 0x101010;' + % m.group(2)), src) + assert n == 2, n + return out + +_CHAIN_T = [ + ('rgbw = (c0 | cb) | (c0 << 8) | (c0 << 16);', + 'rgbw = c0 | cb; rgbw |= c0 << 8; rgbw |= c0 << 16;'), + ('rgbw = (c1 | cb) | (c1 << 8) | (c1 << 16);', + 'rgbw = c1 | cb; rgbw |= c1 << 8; rgbw |= c1 << 16;'), + ('rgbw = (c2 | cb) | (c2 << 8) | (c2 << 16);', + 'rgbw = c2 | cb; rgbw |= c2 << 8; rgbw |= c2 << 16;'), +] +@lever +def tchain(src): + """TRI lit arm: 3-statement accumulator (base is single-expression).""" + for a, b in _CHAIN_T: + src = _one(src, a, b) + return src + +@lever +def tdirect(src): + """TRI lit arm: store the expression directly, no rgbw.""" + for k in range(3): + old = 'rgbw = (c%d | cb) | (c%d << 8) | (c%d << 16);' % (k, k, k) + assert src.count(old) == 1, old + expr = old.split('= ', 1)[1].rstrip(';') + src = src.replace(old, '') + st = '((PolyGT3 *)pkt)->rgb%d = rgbw;' % k + assert src.count(st) == 1, st + src = src.replace(st, '((PolyGT3 *)pkt)->rgb%d = %s;' % (k, expr)) + return src + +@lever +def rgbw_fn(src): + """ONE function-scope `u32 rgbw;` (matched-relative style) instead of per-arm.""" + n = src.count(' u32 rgbw;\n') + m = src.count(' u32 rgbw;\n') + assert n + m == 4, (n, m) + src = src.replace(' u32 rgbw;\n', '') + src = src.replace(' u32 rgbw;\n', '') + return _one(src, ' u32 cb;\n', ' u32 cb;\n u32 rgbw;\n') + +# -------------------------------------------------------------------------- +# residual (a): named producer-offset temps. +# -------------------------------------------------------------------------- +_PROD = """ w = prim->w1; + va = vtx + (w & 0xFFFF); + vb = vtx + (w >> 16); + w = prim->w2; + vc = vtx + (w & 0xFFFF); + w = w >> 16; +""" + +@lever +def prod1(src): + """ONE shared offset variable `vo` for all 3 head offsets -> 3 deaths.""" + new = """ w = prim->w1; + vo = w & 0xFFFF; + va = vtx + vo; + vo = w >> 16; + vb = vtx + vo; + w = prim->w2; + vo = w & 0xFFFF; + vc = vtx + vo; + w = w >> 16; +""" + src = _one(src, _PROD, new) + return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n') + +@lever +def prod2(src): + """TWO offset variables: `vo` (masked, 2 deaths) and `vs` (shifted, 1).""" + new = """ w = prim->w1; + vo = w & 0xFFFF; + va = vtx + vo; + vb = vtx + (w >> 16); + w = prim->w2; + vo = w & 0xFFFF; + vc = vtx + vo; + w = w >> 16; +""" + src = _one(src, _PROD, new) + return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n') + +@lever +def prod3(src): + """`vo` shared by the head offsets AND the vd offset (4 deaths).""" + src = prod1(src) + return _one(src, 'vd = vtx + (w & 0xFFF8);', + 'vo = w & 0xFFF8; vd = vtx + vo;') + +@lever +def prodvd(src): + """only the vd offset gets a named temp (shared with nothing).""" + src = _one(src, 'vd = vtx + (w & 0xFFF8);', + 'vo = w & 0xFFF8; vd = vtx + vo;') + return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n') + +@lever +def prodswap(src): + """commute the producer adds: (w & 0xFFFF) + vtx.""" + new = """ w = prim->w1; + va = (w & 0xFFFF) + vtx; + vb = (w >> 16) + vtx; + w = prim->w2; + vc = (w & 0xFFFF) + vtx; + w = w >> 16; +""" + return _one(src, _PROD, new) + +# -------------------------------------------------------------------------- +# residual (b): variable REUSE merges (RC-14 MERGE / cookbook 45-A). +# The four colours are simultaneously live, so they cannot merge with each +# other -- merge them with values that are DEAD by then instead. +# -------------------------------------------------------------------------- +def _arm(src, which): + """return (start,end) slice of the TRI or QUAD lit arm.""" + if which == 'tri': + a = src.index('/* ---------------- TRI') + b = src.index('case 2:') + else: + a = src.index('/* ---------------- QUAD') + b = src.index('D_800A5E60 = pkt;') + return a, b + +def _merge(src, victim, survivor, which, ndecl): + """rename `victim` -> `survivor` inside one arm, and drop victim's decl.""" + a, b = _arm(src, which) + seg = src[a:b] + new, n = re.subn(r'\b%s\b' % victim, survivor, seg) + assert n == ndecl, 'merge %s->%s in %s: %d hits' % (victim, survivor, which, n) + return src[:a] + new + src[b:] + +def _mrg(src, which, victim, survivor, dropdecl): + a, b = _arm(src, which) + seg = src[a:b] + seg2 = seg.replace(*dropdecl) + assert seg2 != seg, 'decl %r not found in %s arm' % (dropdecl[0], which) + seg2, n = re.subn(r'\b%s\b' % victim, survivor, seg2) + assert n > 0, 'no %s in %s arm' % (victim, which) + return src[:a] + seg2 + src[b:] + +@lever +def merge_za_c0_t(src): + """TRI: the max-z temp `za` and `c0` never overlap -> one variable.""" + return _mrg(src, 'tri', 'za', 'c0', ('s32 za, zb;', 's32 zb;')) + +@lever +def merge_za_c0_q(src): + return _mrg(src, 'quad', 'za', 'c0', ('s32 za, zb;', 's32 zb;')) + +@lever +def merge_zb_c1_q(src): + return _mrg(src, 'quad', 'zb', 'c1', ('s32 za, zb;', 's32 za;')) + +@lever +def merge_zb_c0_q(src): + return _mrg(src, 'quad', 'zb', 'c0', ('s32 za, zb;', 's32 za;')) + +def _mrg_tail(src, which, anchor, victim, survivor, decls): + a, b = _arm(src, which) + seg = src[a:b] + i = seg.index(anchor) + head, tail = seg[:i], seg[i:] + tail2, n = re.subn(r'\b%s\b' % victim, survivor, tail) + assert n > 0, (victim, n) + for old, new in decls: + if old in head: + head = head.replace(old, new) + break + else: + raise AssertionError('no c-decl in %s arm' % which) + return src[:a] + head + tail2 + src[b:] + +@lever +def merge_f0_c3_q(src): + """QUAD: f0..f3 are dead once the last ATTEN3 has run -> f0 doubles as c3.""" + return _mrg_tail(src, 'quad', 'CLAMP80(c3', 'c3', 'f0', + [('s32 c0, c2, c3;', 's32 c0, c2;'), + ('s32 c0, c1, c2, c3;', 's32 c0, c1, c2;')]) + +@lever +def merge_f0_c2_t(src): + """TRI: f0..f3 dead after the last ATTEN3 -> f0 doubles as c2.""" + return _mrg_tail(src, 'tri', 'CLAMP80(c2', 'c2', 'f0', + [('s32 c0, c2, c3;', 's32 c0, c3;'), + ('s32 c0, c1, c2, c3;', 's32 c0, c1, c3;')]) + +@lever +def merge_f1_c2_t(src): + return _mrg_tail(src, 'tri', 'CLAMP80(c2', 'c2', 'f1', + [('s32 c0, c2, c3;', 's32 c0, c3;'), + ('s32 c0, c1, c2, c3;', 's32 c0, c1, c3;')]) + +@lever +def merge_f1_c3_q(src): + return _mrg_tail(src, 'quad', 'CLAMP80(c3', 'c3', 'f1', + [('s32 c0, c2, c3;', 's32 c0, c2;'), + ('s32 c0, c1, c2, c3;', 's32 c0, c1, c2;')]) + +# -------------------------------------------------------------------------- +# residual (b2): the CLAMP80 sum accumulator. +# target: addu $v0,.. addu $v0,.. addu $v0,.. addiu ,$v0,0x10 +# mine : the whole chain is tied INTO the pinned c1 ($a0) by +# combine_regs' `sreg < FIRST_PSEUDO_REGISTER` phys_sugg path. +# fix : give the sum its own NAMED variable with >1 death, so +# local-alloc.c:472 refuses it a qty and combine_regs bails at its +# very first test (`reg_qty[ureg] < 0`). +# -------------------------------------------------------------------------- +_CL = ('#define CLAMP80(C, A0, A1, A2, A3) C = (A0) + (A1) + (A2) + (A3)' + ' + 0x10; if ((C) > 0x80) C = 0x80\n') +_CLS = (_CL + + '#define CLAMP80S(C, A0, A1, A2, A3) sv = (A0) + (A1) + (A2) + (A3);' + ' C = sv + 0x10; if ((C) > 0x80) C = 0x80\n') + +@lever +def sumvar_c1(src): + """shared sum variable `sv` on the two c1 CLAMP80 sites only (2 deaths).""" + src = _one(src, _CL, _CLS) + src = _all(src, 'CLAMP80(c1,', 'CLAMP80S(c1,', 2) + return _one(src, ' u32 cb;\n', ' u32 cb;\n s32 sv;\n') + +@lever +def sumvar_all(src): + """shared sum variable `sv` on ALL seven CLAMP80 sites.""" + src = _one(src, _CL, _CLS) + n = src.count('CLAMP80(c') + assert n == 7, n + src = src.replace('CLAMP80(c', 'CLAMP80S(c') + return _one(src, ' u32 cb;\n', ' u32 cb;\n s32 sv;\n') + +@lever +def sumvar_q(src): + """shared sum variable on the four QUAD CLAMP80 sites.""" + src = _one(src, _CL, _CLS) + a, b = _arm(src, 'quad') + seg = src[a:b] + n = seg.count('CLAMP80(c') + assert n == 4, n + src = src[:a] + seg.replace('CLAMP80(c', 'CLAMP80S(c') + src[b:] + return _one(src, ' u32 cb;\n', ' u32 cb;\n s32 sv;\n') + +@lever +def sumvar_t(src): + """shared sum variable on the three TRI CLAMP80 sites.""" + src = _one(src, _CL, _CLS) + a, b = _arm(src, 'tri') + seg = src[a:b] + n = seg.count('CLAMP80(c') + assert n == 3, n + src = src[:a] + seg.replace('CLAMP80(c', 'CLAMP80S(c') + src[b:] + return _one(src, ' u32 cb;\n', ' u32 cb;\n s32 sv;\n') + +# --- more producer-offset partitions ------------------------------------- +@lever +def prod_cd(src): + """`vo` shared by the vc offset and the vd offset (target puts both in $v0).""" + new = """ w = prim->w1; + va = vtx + (w & 0xFFFF); + vb = vtx + (w >> 16); + w = prim->w2; + vo = w & 0xFFFF; + vc = vtx + vo; + w = w >> 16; +""" + src = _one(src, _PROD, new) + src = _one(src, 'vd = vtx + (w & 0xFFF8);', 'vo = w & 0xFFF8; vd = vtx + vo;') + return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n') + +@lever +def prod_ad(src): + """`vo` shared by the va offset and the vd offset.""" + new = """ w = prim->w1; + vo = w & 0xFFFF; + va = vtx + vo; + vb = vtx + (w >> 16); + w = prim->w2; + vc = vtx + (w & 0xFFFF); + w = w >> 16; +""" + src = _one(src, _PROD, new) + src = _one(src, 'vd = vtx + (w & 0xFFF8);', 'vo = w & 0xFFF8; vd = vtx + vo;') + return _one(src, ' u32 vw, vzw;\n', ' u32 vw, vzw;\n u32 vo;\n') + +@lever +def prod_a(src): + """`vo` on the va offset only, but made multi-death by also carrying vd.""" + return prod_ad(src) + +@lever +def cdrop3_t(src): + """TRI arm declares c3 but never uses it -- drop it.""" + a, b = _arm(src, 'tri') + seg = src[a:b] + for old, new in (('s32 c0, c2, c3;', 's32 c0, c2;'), + ('s32 c0, c1, c2, c3;', 's32 c0, c1, c2;')): + if old in seg: + return src[:a] + seg.replace(old, new) + src[b:] + raise AssertionError('no c-decl in tri arm') + +@lever +def cdropzb_t(src): + """TRI arm declares zb but never uses it -- drop it.""" + a, b = _arm(src, 'tri') + seg = src[a:b] + assert 's32 za, zb;' in seg + return src[:a] + seg.replace('s32 za, zb;', 's32 za;') + src[b:] + +# --- c declaration ORDER inside the arm ---------------------------------- +def _corder(src, order): + plain = [c for c in order] + txt = 's32 %s;' % ', '.join(plain) + for old in ('register s32 c1 __asm__("$4"); s32 c0, c2, c3;', + 's32 c0, c1, c2, c3;'): + if src.count(old) == 2: + if 'register' in old: + txt = ('register s32 c1 __asm__("$4"); s32 %s;' + % ', '.join([c for c in order if c != 'c1'])) + return src.replace(old, txt) + raise AssertionError('no c-decl') + +# --- ref dials ----------------------------------------------------------- +def _dial(src, name, which, n=1): + """insert n zero-byte ref bumps on `name` right after the c-decl of an arm.""" + a, b = _arm(src, which) + seg = src[a:b] + m = re.search(r'( *)(register s32 c1 __asm__\("\$4"\); s32 c0, c2, c3;|s32 c0, c1, c2, c3;)\n', seg) + assert m, 'no anchor' + ins = ''.join('%s__asm__ __volatile__ ("" :: "r" (%s));\n' % (m.group(1), name) + for _ in range(n)) + seg = seg[:m.end()] + ins + seg[m.end():] + return src[:a] + seg + src[b:] + + +# -------------------------------------------------------------------------- +# residual (b1): the TRI arm's c0/c2 grants. +# The target's TRI grants (c0=$t4, c1=$a0, c2=$t2) are IDENTICAL to its QUAD +# grants -- i.e. c0..c3 are ONE set of function-scope variables shared by +# both arms, exactly as in the matched relatives func_8017D960 / +# func_8017F510. Per-cull-block scope splits them into two independent +# allocno sets, which is why the TRI set drifts. +# -------------------------------------------------------------------------- +_CDECLS = ('register s32 c1 __asm__("$4"); s32 c0, c2, c3;', + 's32 c0, c1, c2, c3;') + +def _cfn(src, anchor, pinned): + for old in _CDECLS: + if src.count(old) == 2: + break + else: + raise AssertionError('no per-arm c-decl') + # drop the two per-arm declarations (whole lines) + out, n = re.subn(r'[ \t]*%s\n' % re.escape(old), '', src) + assert n == 2, n + decl = ('register s32 c1 __asm__("$4"); s32 c0, c2, c3;' if pinned + else 's32 c0, c1, c2, c3;') + assert out.count(anchor) == 1, anchor + return out.replace(anchor, anchor + ' %s\n' % decl) + +@lever +def cfn(src): + """c0..c3 at FUNCTION scope, c1 still pinned to $a0.""" + return _cfn(src, ' s32 a0v, a1v, a2v, a3v;\n', True) + +@lever +def cfn_np(src): + """c0..c3 at FUNCTION scope, PIN-FREE (matched-relative style).""" + return _cfn(src, ' s32 a0v, a1v, a2v, a3v;\n', False) + +@lever +def cfn_cb(src): + """c0..c3 at function scope, declared just before `u32 cb;`.""" + return _cfn(src, ' u32 uvw;\n', True) + +@lever +def cfn_top(src): + """c0..c3 at function scope, declared early (before the r/lo/hi block).""" + return _cfn(src, ' Part *part;\n', True) + +@lever +def cfn_end(src): + """c0..c3 at function scope, declared LAST.""" + return _cfn(src, ' u32 cb;\n', True) + +@lever +def cfn_d(src): + """c0..c3 at function scope, declared right after `s32 d;`.""" + return _cfn(src, ' s32 d;\n', True) + + +@lever +def unpin_c1n(src): + """drop the c1 pin (function-scope c-decl form, single occurrence).""" + return _one(src, 'register s32 c1 __asm__("$4"); s32 c0, c2, c3;', + 's32 c0, c1, c2, c3;') + +# -------------------------------------------------------------------------- +# residual 3889: one CLAMP80 site sums its attenuations in a different +# ORDER (a0v+a1v+a3v+a2v). Its `sra` therefore lands in a3v's own register +# ($a3) instead of being written in place over a2v's. Another hand-edit +# copy-paste artefact, of the same family as report-1 artefacts 1-4. +# -------------------------------------------------------------------------- +_CLAMP_RE = re.compile(r'CLAMP80S?\((c\d), (a0v), (a1v), (a2v), (a3v)\);') + +def _clampswap(src, which, sites): + a, b = _arm(src, which) + seg = src[a:b] + hits = [-1] + def rep(m): + hits[0] += 1 + if hits[0] not in sites: + return m.group(0) + return m.group(0).replace('a2v, a3v', 'a3v, a2v') + seg2 = _CLAMP_RE.sub(rep, seg) + assert seg2 != seg, 'clampswap %s %s: no site changed' % (which, sites) + return src[:a] + seg2 + src[b:] + +# -------------------------------------------------------------------------- +# residual 3889 -- a FIFTH copy-paste artefact. +# At ONE of the seven ATTEN3 sites the y-axis `else if` branch accumulates +# into a2v instead of a3v, while the y-axis KILL branch still says a3v. +# Byte-evidence in the target: +# 3875 addu $a2,$zero,$zero <- kill branch writes a3v ($a2) [matches] +# 3889 sra $a3,$s2,7 <- else branch writes a2v ($a3) [differs] +# Cost: zero instructions. Same shape as report-1 artefacts 1-4. +# -------------------------------------------------------------------------- +_A3 = """#define ATTEN3(A, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \\ +""" +_A3W = """#define ATTEN3W(A, AW, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \\ + A = 0; \\ + if (F) { \\ + d = (X) - (CX); if (d < 0) d = (CX) - (X); \\ + if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \\ + d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \\ + if ((RZ) < d) A = 0; \\ + else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \\ + d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \\ + if ((RY) < d) A = 0; \\ + else if ((RLO) < d) AW = (A * (((R) - d) / 4)) >> 7; \\ + } + +""" +_A3CALL = re.compile(r'ATTEN3\((a\dv), (f\d),') + +def _atten3w(src, which, sites, dst): + if 'ATTEN3W' not in src: + src = _one(src, _A3, _A3W + _A3) + a, b = _arm(src, which) + seg = src[a:b] + hits = [-1] + def rep(m): + hits[0] += 1 + if hits[0] not in sites: + return m.group(0) + return 'ATTEN3W(%s, %s, %s,' % (m.group(1), dst, m.group(2)) + seg2 = _A3CALL.sub(rep, seg) + assert seg2 != seg, 'atten3w %s %s: no site changed' % (which, sites) + return src[:a] + seg2 + src[b:] + +# -------------------------------------------------------------------------- +# last residual (2258/2259): `c1 << 16` in the TRI arm. +# c1 is pinned, so it is a HARD reg from the start; it DIES at the `sll`, so +# combine_regs takes its `sreg < FIRST_PSEUDO_REGISTER` branch and records +# $a0 in qty_phys_sugg for the sll's temp -> the temp lands in $a0 and the +# shift is done in place. The target keeps the temp in $v0. +# Cure family: make the temp NOT a 1-death local (a named var used in both +# arms), or move the pin off c1 onto a colour whose grant we already match. +# -------------------------------------------------------------------------- +@lever +def hivar(src): + """named `hi` for the `c1 << 16` term in BOTH arms -> 2 deaths, no qty.""" + out, n = re.subn(r'\(c1 << 16\)', 'hi', src) + assert n >= 1, n + out, m = re.subn(r'rgbw \|= c1 << 16;', 'rgbw |= hi;', out) + tri = out.index('/* ---------------- TRI') + quad = out.index('/* ---------------- QUAD') + # one `hi = c1 << 16;` immediately before the first rgb store of each arm + def ins(s, marker): + i = s.index(marker) + j = s.rindex('\n', 0, s.rindex('rgbw', 0, i) if 'rgbw' in s[:i] else i) + return s + for anchor in ('rgbw = (c1 | cb)', 'rgbw = c1 | cb'): + while anchor in out: + k = out.index(anchor) + ln = out.rindex('\n', 0, k) + 1 + pad = out[ln:k] + out = out[:ln] + pad + 'hi = c1 << 16;\n' + out[ln:] + k2 = out.index(anchor, ln + len(pad) + 14) + out = out[:k2] + anchor.replace('rgbw', 'rgbw$') + out[k2 + len(anchor):] + out = out.replace('rgbw$', 'rgbw') + assert 'hi = c1 << 16;' in out + return _one(out, ' u32 cb;\n', ' u32 cb;\n u32 hi;\n') + +@lever +def _noop(src): + return src + +def _c1live(src, which, anchor): + """RC-15 zero-byte ref that keeps the PINNED c1 ($a0) live past the + `sll` of `c1 << 16`. combine_regs records $a0 in qty_phys_sugg + unconditionally (local-alloc.c:1798, no death guard), but find_free_reg + can only honour a suggestion whose hard reg is actually free over the + temp's live range -- so extending c1 past the shift is what refuses it.""" + a, b = _arm(src, which) + seg = src[a:b] + assert seg.count(anchor) == 1, (anchor, seg.count(anchor)) + k = seg.index(anchor) + len(anchor) + ln = seg.rindex('\n', 0, seg.index(anchor)) + 1 + pad = seg[ln:seg.index(anchor)] + seg = seg[:k] + '\n' + pad + '__asm__ __volatile__ ("" :: "r" (c1));' + seg[k:] + return src[:a] + seg + src[b:] + +def main(): + base, out, levers = sys.argv[1], sys.argv[2], sys.argv[3:] + src = open(base).read() + for lv in levers: + if lv.startswith('RC:'): # RC:c0=12,c2=10 + src = _repin_c(src, lv[3:]); continue + if lv.startswith('CO:'): # CO:c2,c0,c1,c3 + src = _corder(src, lv[3:].split(',')); continue + if lv.startswith('CL:'): # CL:tri:rgb1 + _, wh, tag = lv.split(':') + poly = '3' if wh == 'tri' else '4' + anc = {'rgb0': '((PolyGT%s *)pkt)->rgb0 = rgbw;' % poly, + 'rgb1': '((PolyGT%s *)pkt)->rgb1 = rgbw;' % poly, + 'rgb2': '((PolyGT%s *)pkt)->rgb2 = rgbw;' % poly, + 'uv0': '((PolyGT%s *)pkt)->uv0 = tp[1];' % poly, + 'end': 'pkt += 0x%s;' % ('28' if wh == 'tri' else '34')}[tag] + src = _c1live(src, wh, anc); continue + if lv.startswith('AW:'): # AW:quad:1:a2v + _, wh, ix, dst = lv.split(':') + src = _atten3w(src, wh, set(int(x) for x in ix.split(',')), dst); continue + if lv.startswith('CS:'): # CS:quad:1 or CS:tri:0,2 + _, wh, ix = lv.split(':') + src = _clampswap(src, wh, set(int(x) for x in ix.split(','))); continue + if lv.startswith('DL:'): # DL:name:tri[:n] + p = lv[3:].split(':') + src = _dial(src, p[0], p[1], int(p[2]) if len(p) > 2 else 1); continue + src = LEVERS[lv](src) + open(out, 'w').write(src) + +main() diff --git a/.run/giants/bf14_sw2.sh b/.run/giants/bf14_sw2.sh new file mode 100644 index 0000000000..c371f3ab4c --- /dev/null +++ b/.run/giants/bf14_sw2.sh @@ -0,0 +1,15 @@ +#!/bin/bash +# bf14_sw2.sh "| [lever...]" ... -> one line per variant, parallel +# uses bf14_mk2.py (round-2 levers). Reports raw mismatch count + length. +cd /home/musashi/bfm-decomp +BASE="$1"; shift +run() { + local spec="$1"; local tag="${spec%%|*}"; local lv="${spec#*|}" + local wd=".run/giants/bf14_sw2/$tag" + mkdir -p "$wd" + python3 .run/giants/bf14_mk2.py "$BASE" "$wd/v.c" $lv 2>"$wd/mk.err" || { printf '%-22s :: MK-FAIL %s\n' "$tag" "$(tail -2 $wd/mk.err|tr '\n' ' ')"; return; } + bash .run/giants/bf14_cc.sh "$wd/v.c" "$wd/w" >/dev/null 2>"$wd/cc.err" || { printf '%-22s :: COMPILE-FAIL %s\n' "$tag" "$(tail -3 $wd/cc.err|tr '\n' ' ')"; return; } + printf '%-22s :: %s\n' "$tag" "$(python3 .run/giants/bf14_full.py "$wd/w/t.o" --count)" +} +export -f run; export BASE +printf '%s\n' "$@" | xargs -P 8 -I{} bash -c 'run "$@"' _ {} diff --git a/.run/giants/s19_bf14_report2.md b/.run/giants/s19_bf14_report2.md new file mode 100644 index 0000000000..1084d9d578 --- /dev/null +++ b/.run/giants/s19_bf14_report2.md @@ -0,0 +1,383 @@ +# `func_8017BF14` — behemoth #4, 4,763 ins, `ov_SC03_116` — ROUND 2: **MATCH** + +**Session 20 round 2, 2026-07-25.** Round 1 handed over 45/4763 mismatched. +Round 2 closed it. + +--- + +## 1. FINAL NUMBER (measured, `tools/match_one.py`, the CANDIDATE gate) + +``` +python3 tools/match_one.py func_8017BF14 --c .run/giants/s19_func_8017BF14_b2.c \ + --asm-subdir asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C +-> MATCH (4763 ins) func_8017BF14 +``` + +Reproduced **3×** from independent private work dirs. Independent confirmations: + +| check | result | +|---|---| +| `bf14_full.py` masked index-wise diff | **0 / 4763 mismatched** | +| `bf14_hist.py` opcode histogram | `len+0 L1=0` | +| `bf14_slots.py` stack-slot census | **127 / 127**, all at the target's offsets | +| compile warnings (`-Wall`) | none | + +**The whole-binary SHA1 arbiter (G3/P9) was NOT run** — the task forbade touching +the build tree. `match_one` is the candidate check only; the coordinator's +whole-binary gate is the sole arbiter (G3/P9). + +Trajectory, every step measured: **45 → 37 → 33 → 21 → 11 → 3 → 2 → 0.** + +--- + +## 2. PER-RESIDUAL OUTCOME + +Round 1 split the 45 into (a) producer temps ≈8, (b) `c0`/`c2` grants ≈15, +(c) quad-lit rgb accumulator ≈20. All three fell. The exact split of the 45, +recounted from the diff, was **(a) 8 + (b) 23 + (c) 14**. + +### (a) Prim-word producer temps — **FELL** (8 ins, idx 508–512, 539–542) + +Round 1's diagnosis (`combine_regs` tying the chain into the pinned `va`) was +**correct**, and its prescribed cure (named MULTI-death offset variables) does +work — `prod2` broke the `$t2` tie exactly as predicted, byte-visible: + +``` +base 508 andi $t2,$a1,0xffff 510 addu $t2,$t6,$t2 <- tied, in place +prod2 508 andi $v1,$a1,0xFFFF 510 addu $t2,$t6,$v1 <- MATCHES target +``` + +But it is a **conservation law, not a fix**: one shared `vo` gives one register +for all its sites, whereas the target uses three distinct temps ($v1, $a0, $v0, +$v0 for the four offsets). `prod2` fixed 508/510 and broke 513/515 — 45 → 45. +Every partition of the four offsets across named variables was swept +(`prod1/2/3`, `prod_ad`, `prod_cd`, `prodvd`, `prodswap`): best is neutral. + +**What actually fixed it: R4 — dropping the `va→$t2` and `w→$a1` pins.** +Round 1 measured those pins as worth 4% and the brief said not to re-buy their +removal. That was true *of the round-1 base* and is **false** once `c0..c3` sit +at function scope (R3): on the 21-base, `unpin_va` = 17, `unpin_va unpin_w` = +**13**. This is base-dependence, not a contradiction — and it is the single +most important methodological lesson of the round (§5.1). + +### (b) `c0`/`c2` grants — **FELL** (23 ins) + +Round 1 pointed at `allocno_compare` order and prescribed a variable-REUSE +merge. The **reuse sweep was run in full and every merge lost** (§4.3): merging +`za`/`zb`/`f0`/`f1` into `c0..c3` scored 43–3294 against a 33/37 base. Reuse is +now a measured dead end on this function. + +The real lever was found by **reading the target and the matched relatives**, +which is where round 1 said the value was: + +* The target's TRI grants are `c0→$t4, c1→$a0, c2→$t2` — **identical to its + QUAD grants**. Two independently-scoped allocno sets cannot coincide by + chance; one shared set can. +* Both matched relatives declare the colours at **function scope**: + `.run/giants/s19_func_8017D960_b5.c:310` and `s19_func_8017F510_b4.c:338`. + +`s32 c0, c1, c2, c3;` at function scope: **33 → 21**. Declaration *position* is +neutral (5 anchors swept, all 21) — consistent with round 1's L4.2 oracle, since +these never take stack slots. + +This **reverses round-1's L4**, which put them per-cull-block. L4 was right on +its own base — it was supplying the extra local allocno that spills `r1lo` — but +R1 and R2 supply that pressure now. + +A second, separable part of (b) was the c1 **sum accumulator** (9 ins, idx +1876-1879 / 3890-3893), which round 1 had classified with the grants. It is a +pin artefact, and `CLAMP80S` (R1) fixed it: **45 → 37**. + +### (c) Quad-lit rgb accumulator — **FELL** (14 ins) + +Two independent pieces: + +* **Unlit `rgbc` (4 ins, idx 2310/2311, 4699/4700).** `cb` is a function-scope + global allocno, so `cb |= 0x101010` writes it in place ($a1). The target + stores the *expression* `cb | 0x101010`, a 1-death local that `combine_regs` + ties to the **dying constant register** $v1. **37 → 33.** +* **Quad lit rgb2/rgb3 (10 ins, idx 4643-4652).** Reverting those two to the + single-expression form (rgb0/rgb1 keep the 3-statement accumulator) makes the + intermediates 1-death local temps that alternate $v0/$v1 — which is what lets + the target's store of the previous rgb word sit one slot later. **21 → 11.** + All-four = 23, rgb0/rgb1-only = 21, direct-store = 28. The split really is + 2-and-2, matching artefact 4 (rgb2/rgb3 are the two that take `c1 << 16`). + +### THE ATTRIBUTION PRIMITIVE FOR (c) — run, and decisive + +``` +-fno-schedule-insns -> my order unchanged +-fno-schedule-insns2 -> my order unchanged +both -> my order unchanged +``` + +In all three builds `sw v1,-20(t3)` still precedes `or v1,t2,a1`. **The +transposition was never a `sched.c` decision.** With a 3-statement accumulator +the value is pinned to one register, so the next `or` clobbers $v1 and *no* +scheduler could hoist it above the store — the ordering is a consequence of the +register grant. Changing the grant (R5) fixed the order for free. This is +**§78 reproduced exactly**, and it is the second time on this family that a +"scheduling" residual was really a register grant. Do not reason about `sched.c` +on this family until this primitive has been run. + +--- + +## 3. THE ONE MECHANISM BEHIND FOUR OF THE SEVEN LEVERS + +R1, R2, R4 and R7 are all the same compiler fact, and it is worth a cookbook +entry because it is the *cost* of the pin technique: + +> A `register __asm__` pin makes the variable a **hard register in the RTL from +> the start**. When a 1-death local temp is produced from — or consumed into — +> that hard reg, `local-alloc.c`'s `combine_regs` takes its hard-register branch +> (`local-alloc.c:1795-1820`) and records the pinned register in +> `qty_phys_sugg` for the temp's quantity. **That path has no death guard and no +> cost model — it fires unconditionally.** The temp then lands in the pinned +> register and the operation is performed IN PLACE. An ordinary pseudo never +> gets that suggestion, because anything crossing a basic block has +> `reg_qty == -1` and `combine_regs` bails at its very first test. + +Verified in source, `tools/reference/gcc-2.7.2/local-alloc.c`: + +* `:472` — a pseudo is local iff `reg_basic_block[i] >= 0 && reg_n_deaths[i] == 1`. +* `:1763` — `combine_regs` returns 0 immediately if `reg_qty[ureg] < 0`. +* `:1795` — `if (ureg < FIRST_PSEUDO_REGISTER) { ... qty_phys_sugg |= ureg; return 0; }` + — the branch that costs us, reached from `block_alloc` at `:1295` with + `already_dead = 0`. + +**Three separable cures, all three used in this match:** + +| cure | how | used by | +|---|---|---| +| (i) refuse the temp a quantity | give it a NAMED variable with **>1 death**, so `:472` rejects it and `combine_regs` bails at `:1763` | **R1** (`CLAMP80S`/`sv`) | +| (ii) remove the pin | only if the pin is not load-bearing — **re-measure, it is base-dependent** | **R4** (`va`, `w`) | +| (iii) starve the suggestion | keep the pinned value **LIVE past the temp**, so `find_free_reg` cannot honour the suggestion | **R7** (zero-byte ref on `c1`) | + +Cure (iii) is new and is the one that closed the function. The last two +instructions were `sll $a0,$a0,16` (mine, in place over the pinned `c1`) vs +`sll $v0,$a0,16` (target). `c1` could not be unpinned — it is what spills +`r1lo`, and dropping it or moving it to any other colour costs −64 length +(measured, §4.5). So instead: + +```c +((PolyGT3 *)pkt)->rgb1 = rgbw; +__asm__ __volatile__ ("" :: "r" (c1)); /* RC-15, zero bytes */ +``` + +`c1` is now live past the shift, `$a0` is unavailable, the temp falls to `$v0`. +**2 → 0.** Placing it after the rgb2 store also matches; after `uv0` or at the +arm's end does not (they perturb length). + +--- + +## 4. EVERY LEVER MEASURED THIS ROUND + +Metric is the `match_one` **mismatch count** (length is exact throughout, so the +raw count is honest — the round-1 metric trap of §5.3 does not apply). Baselines +are stated per block because the base moved as levers landed. + +### 4.1 The winning chain + +| # | lever | base → result | +|---|---|---| +| **R1** | `CLAMP80S` — named `sv` sum variable on the **two c1 sites only** | 45 → **37** | +| **R2** | unlit arms: `pkt->rgbc = cb \| 0x101010;` (expression, not `cb \|=`) | 37 → **33** | +| **R3** | **`s32 c0, c1, c2, c3;` at FUNCTION scope** | 33 → **21** | +| **R4** | drop the `va→$t2` and `w→$a1` pins | 21 → 17 → **13** | +| **R5** | quad-lit **rgb2/rgb3 only** revert to single-expression | 21 → **11**; with R4 → **3** | +| **R6** | artefact 5 — `ATTEN3W(a3v, a2v, …)` at the QUAD vertex-1 site | 3 → **2** | +| **R7** | RC-15 zero-byte ref on pinned `c1` after the TRI rgb1 store | 2 → **0** | + +### 4.2 Neutral — measured, no effect at all + +| lever | base | result | +|---|---|---| +| `prod1` / `prod2` / `prodvd` / `prodswap` (named producer-offset temps) | 45 | 45 (fixes 508/510, breaks 513/515) | +| same four | 21 | 21 | +| `cdrop3_t` (drop the unused `c3` from the TRI arm) | 45 / 37 | 45 / 37 | +| `cdropzb_t` (drop the unused `zb` from the TRI arm) | 45 | 45 | +| `qsingle01` (single-expression for quad rgb0/rgb1) | 21 | 21 | +| `prod2` on top of `sumvar_c1` | 45 | 37 (= R1 alone) | +| declaration POSITION of the function-scope `c0..c3` — 5 anchors: before `a0v`, before `cb`, before `d`, before `part`, last | 33 | **21 at every anchor** | +| `qsingle23 + prodvd` / `+ prodswap` / `+ prod1` / `+ prod2` | 11 | 11 | + +### 4.3 The REUSE sweep — run in full, every merge LOST + +This was round 1's flagged #1 move. It is now a measured dead end here. + +| merge | base | result | +|---|---|---| +| `za` → `c0` (QUAD) | 45 | 55 | +| `za` → `c0` (TRI) | 45 | 3294 | +| `zb` → `c1` (QUAD) | 45 | 55 | +| `zb` → `c0` (QUAD) | 45 / 37 | 51 / 43 | +| `f0` → `c3` (QUAD) | 45 / 37 | 51 / 43 | +| `f0` → `c2` (TRI) | 45 | 49 | +| `f1` → `c2` (TRI) | 45 | 838 | +| `f1` → `c3` (QUAD) | 45 | 841 | + +**Why it failed, and this generalises:** a REUSE merge raises the survivor's +`reg_n_refs` to move `allocno_compare` priority. But the TRI/QUAD `c0..c3` +grants did not disagree because of *priority* — they disagreed because the two +arms had **two independent allocno sets** at all. No amount of re-ranking inside +one set can make it agree with a different set; only merging the sets can (R3). +**Diagnose whether two grants differ by RANK or by IDENTITY before reaching for +a ref-count lever.** + +### 4.4 rgb emit-word forms + +| lever | base | result | +|---|---|---| +| `qsingle` (all four quad words single-expression) | 45 / 37 / 21 | 47 / 39 / 23 | +| `qsingle23` (**rgb2/rgb3 only**) | 45 / 21 | **1040 / 11** ← extreme base-dependence | +| `qsingle01` (rgb0/rgb1 only) | 45 / 21 | 45 / 21 | +| `qdirect` (no `rgbw`, store the expression) | 37 / 21 | 44 / 28 | +| `tdirect2` (TRI, no `rgbw`) | 37 / 21 | 47 / 37 | +| `qdirect + tdirect2` | 37 | 54 | +| `tchain` (TRI 3-statement accumulator) | 45 | 2514 (len 4762) | +| `cb_expr_one` (unlit, drop `cb` entirely) | 37 | 2423 (len 4767) | +| one function-scope `u32 rgbw;` (relative style) | 37 | n/a — 4 per-arm decls, lever refused | + +### 4.5 The pins — re-measured on every base + +| lever | base | result | +|---|---|---| +| `unpin_va` | 45 / 21 | 174 / **17** | +| `unpin_w` | 45 / 21 | 47 / 23 | +| `unpin_va + unpin_w` | 45 / 21 | 170 / **13** | +| `unpin_f0` | 45 / 21 | 88 / 64 — **f0→$s3 stays** | +| `unpin_c1` | 45 / 21 / 3 | 654 / — / **len 4699 (−64)** | +| all four unpinned | 45 | 789 (the round-1 pin-free fallback) | +| c0..c3 at function scope, **pin-free** | 33 | len 4699 (−64) | +| re-pin `c0→$t4` and/or `c2→$t2`/`c3→$a2` **instead of** c1 | 2 | **len 4699 every time** | +| re-pin `c0→$t4` **plus** c1 | 2 | 14 | +| re-pin `c1 + c2` / `c0+c1+c2` / all four / `c1+c3` | 2 | 60 / 71 / 320 / 307 | + +**The `c1→$a0` pin is uniquely load-bearing: it is the register pressure that +spills `r1lo`.** No other colour, and no combination without it, reproduces the +spill. Round-1 §5.2's "pinning c0 and c2 *in addition to* c1 is worse" is +confirmed and extended: pinning them *instead of* c1 does not even preserve the +frame. + +### 4.6 Artefact-5 localisation (the `sra $a2` vs `$a3` residual) + +| lever | base | result | +|---|---|---| +| `ATTEN3W(a3v, a2v, …)` at QUAD site **1** | 3 | **2** | +| same at QUAD sites 0 / 2 / 3 | 3 | 4 / 4 / 4 | +| same at TRI sites 0 / 1 / 2 | 3 | 4 / 4 / 4 | +| `ATTEN3W(a3v, a1v, …)` / `(a3v, a0v, …)` at QUAD site 1 | 3 | 3 / 3 | +| swap the last two `CLAMP80` args (a2v↔a3v) — **all 7 sites tried one at a time** | 3 | 5 at every site | + +The `CLAMP80` argument-order hypothesis is **refuted**: the sum order is +identical (3890-3893 are byte-identical in both), and the target's kill branch +at idx 3875 (`addu $a2,$zero,$zero`) already agrees. Only the y-axis **`else if` +destination** differs — a genuine fifth copy-paste artefact, costing zero +instructions. + +### 4.7 Other + +| lever | base | result | +|---|---|---| +| `sumvar_all` (shared `sv` on all 7 CLAMP sites) | 45 | 4568, **len 4699** | +| `sumvar_q` / `sumvar_t` (per-arm) | 45 | 41 / 41 | +| `prod3` (`vo` on all four offsets) | 45 / 21 | 47 / 23 | +| `prod_ad` / `prod_cd` | 45 / 21 | 50, 52 / 26, 28 | +| `hivar` (named `hi` for `c1 << 16`, both arms) | 2 | 226 (len 4765) | +| `hivar + unpin_c1` | 2 | 4549 (len 4697) | +| zero-byte `c1` ref after the TRI **uv0** store / at arm **end** | 2 | 2471 (len 4764) / 2446 (len 4766) | + +--- + +## 5. LESSONS TO FEED BACK (cookbook candidates) + +### 5.1 A "do-not-re-buy" entry is scoped to the BASE that measured it + +Three of round 1's measured, correctly-recorded findings inverted once the base +moved: + +| round-1 finding | round-2 measurement | +|---|---| +| L4: `c0..c3` per cull block (52% → 93%) | function scope is **strictly better**, 33 → 21 | +| L9: removing the `va` pin costs 4% | removing `va` **and** `w` is worth 21 → 13 | +| L8: quad rgb 3-statement accumulator is +0.02% | single-expression for rgb2/rgb3 is worth 21 → 11 | + +`qsingle23` is the extreme case: **1040 mismatched on the 45-base, 11 on the +21-base** — the same edit, two orders of magnitude apart. None of these were +errors in round 1; they were correct readings of a different base. + +**Rule candidate:** a do-not-re-buy table must record *the base it was measured +against*, and any entry measured against a base that has since moved by a +structural lever is **stale, not settled** — re-measure the cheap ones (one +0.28 s probe each) rather than inheriting them. Re-testing the whole round-1 +"negative" list on the new base cost about 20 seconds of compute and produced +three of the seven winning levers. + +### 5.2 The pin's hidden cost: `combine_regs`' unconditional `qty_phys_sugg` + +§3 above, with source citations. Worth its own cookbook section: it explains a +whole *class* of 2-instruction "in-place vs not" residuals, and it gives three +separable cures. Cure (iii) — the zero-byte liveness extension — is new, and it +is how a pin can be kept for its allocation pressure while its tie is refused. + +### 5.3 Sibling grants are an IDENTITY oracle, not just a hint + +If two code paths in the target show *identical* register grants for +corresponding variables, those variables are **one set of allocnos** — i.e. one +declaration at a scope enclosing both. That is a positive structural inference +from register numbers alone, and it beat an exhaustive scope sweep plus a full +reuse sweep. It also agreed with what the two matched relatives already showed, +which is the round-1 report's own advice (§"MATCHED RELATIVES") paying off again: +**5 of 9 winning levers on the last behemoth, and 2 of 7 here (R3, and R5's +2-and-2 split), came straight off a matched relative or off the target's own +register numbering.** + +### 5.4 Run the attribution primitive before any scheduling reasoning + +§2. Two for two on this family: an apparent scheduling residual that was a +register grant. Cost: three compiles. + +--- + +## 6. WHAT CHANGED IN THE SOURCE (7 hunks vs `s19_func_8017BF14_b1.c`) + +1. `ATTEN3W` macro added (artefact 5, y-axis `else` destination). +2. `CLAMP80S` macro added (named `sv` sum). +3. `va` / `w` pins removed; `c0..c3` (with the `c1` pin) moved to function + scope; `s32 sv;` declared. +4. `CLAMP80(c1, …)` → `CLAMP80S(c1, …)` in both arms. +5. TRI arm: zero-byte `c1` ref after the rgb1 store; unlit `rgbc` written as an + expression. +6. QUAD arm: `ATTEN3W` at the vertex-1 site; rgb2/rgb3 single-expression; unlit + `rgbc` written as an expression. +7. The two per-arm `register s32 c1 …; s32 c0, c2, c3;` declarations removed. + +--- + +## 7. FILES + +* `.run/giants/s19_func_8017BF14_b2.c` — **the match**, full updated dossier. +* `.run/giants/s19_func_8017BF14_b1.c` — round-1 draft, 45/4763 (kept). +* `.run/giants/s19_func_8017BF14_b1_pinfree.c` — round-1 pin-free, 789/4763 (kept). +* `.run/giants/bf14_mk2.py` — round-2 lever generator. Same contract as + `bf14_mk.py`: **every transformation asserts it applied**, so a "neutral" + reading can never be a silent no-op. Levers: `unpin_*`, `RC:` (re-pin), + `sumvar_*`, `cb_expr*`, `cfn*`, `qsingle*`, `qdirect`, `tdirect2`, `prod*`, + `merge_*`, `CS:` (CLAMP arg swap), `AW:` (ATTEN3 y-else destination), + `CL:` (zero-byte `c1` liveness), `hivar`. +* `.run/giants/bf14_sw2.sh` — 8-way parallel sweep over `bf14_mk2.py`, reporting + the raw mismatch count. +* `.run/giants/bf14_hdr.py` + `bf14_hdr.txt` — dossier-header splicer. +* `.run/giants/r2/` — the promoted bases: `b2base` 37, `b3base` 33, `b4base` 21, + `b5base` 3, `b6base` 2, `b7base` **0**; `sch_*` the attribution-primitive + builds; `da/` the `-da` RTL dumps of the 45-base. +* Round-1 harness (`bf14_cc/score/probe/full/side/win/hist/ali/slots/…`) used + unchanged. + +--- + +## 8. STATUS FOR THE COORDINATOR + +`match_one` reports **MATCH (4763 ins)**. That is the candidate gate only. +**The whole-binary SHA1 byte-gate (G3/P9) has not been run and is the sole +arbiter** — this is not a confirmed match until that is green. diff --git a/.run/giants/s19_func_8017BF14_b2.c b/.run/giants/s19_func_8017BF14_b2.c new file mode 100644 index 0000000000..97c06bbab3 --- /dev/null +++ b/.run/giants/s19_func_8017BF14_b2.c @@ -0,0 +1,839 @@ +#include "common.h" +#include "/home/musashi/bfm-decomp/src/shared/engine_types.h" + +/* =========================================================================== + * func_8017BF14 -- 4,763 ins, ov_SC03_116 (behemoth #4). *** MATCH *** + * + * STATUS (2026-07-25, session 20 round 2, gcc-2.7.2 pinned triple): + * python3 tools/match_one.py func_8017BF14 --c \ + * --asm-subdir asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C + * -> MATCH (4763 ins) func_8017BF14 [reproduced 3x, private work dirs] + * LENGTH EXACT | OPCODE HISTOGRAM EXACT (L1 = 0) | STACK FRAME EXACT + * (all 127 slots at the target's offsets, frame 0x360) + * masked index-wise diff: 0 / 4763 mismatched. + * `match_one` is the CANDIDATE gate. The whole-binary SHA1 arbiter (G3/P9) + * is run by the coordinator and is the only thing that makes this final. + * + * Round 1 closed at 45/4763 mismatched. Round 2 took 45 -> 37 -> 33 -> 21 + * -> 11 -> 3 -> 2 -> 0. See .run/giants/s19_bf14_report2.md. + * + * WHAT IT IS + * The *four*-light-box variant of the volumetric-light renderer whose + * 3-box sibling func_8017D960 (3,338 ins, ov_SC03_090) is MATCHED, and whose + * unlit ancestor func_8017BEBC (ov_SC03_099) is MATCHED. Same family, same + * skeleton; this is the biggest member. + * + * Signature: func_8017BF14(s32 arg0, s32 lim). Unlike every other member of + * the family this one is a LEAF -- 0 callees. The 3-call prologue + * (func_800491EC / func_800547D8 / func_80052E38) of the siblings is gone; + * `lim` arrives as arg1 (spilled to 0xB0). That is why the frame has no + * 0x10 argument area (tmpxy[] starts at sp+0x00) and no $ra save. + * + * Per part (stride 0x14, outer loop): build the 8-corner AABB in box[], + * rtpt/rtpt + rtps/rtps -> sxy[8], stszotz -> g.otz, reject on + * `lim >= g.otz`, then screen-space bbox reject on X (-0xA0..0xA1) and + * Y (-0x6E..0x6F). + * Per prim (stride 0xC, inner loop): rtpt the 3 vertices, stflg mask + * 0x7F85E000, nclip, stopz > 0, then a 4-way range tree on `code = w & 7` + * that keeps ONLY codes 6,7 (tri) and 2,3 (quad); 0,1,4,5 fall through to + * the loop tail. + * Per drawn poly: screen bbox reject, then each vertex is tested against + * FOUR axis-aligned light boxes (flags f0..f3), and if any is lit a + * 0x00..0x80 attenuation per active box is computed, summed, biased +0x10 + * and clamped to 0x80 -> a grey gouraud vertex colour. + * lit -> POLY_GT3 (0x28, tag 0x34000000, OT 0x9000000) + * POLY_GT4 (0x34, tag 0x3C000000, OT 0xC000000) + * unlit -> POLY_FT3 (0x20, OT 0x7000000) / POLY_FT4 (0x28, OT 0x9000000) + * with rgbc = (tp[0] & 0xFF000000) | 0x101010 <-- NOT black, + * unlike func_8017D960 where the unlit colour is plain black. + * + * THE FOUR LIGHT BOXES (stride 0x1C, {s32 enable; u16 cx,cy,cz; s32 range}) + * D_80197C28 / D_80197C44 / D_80197C60 / D_80197C7C. + * Falloff geometry differs from the 3-box sibling: RLO = R - 0x200 (not + * -0x80) and the ramp is ((R - d) / 4) (not (R - d)), so the 0x80 ceiling is + * reached over a 0x200-wide band instead of 0x80. The `/ 4` is a SIGNED + * divide -- `bgez / addiu 3 / sra 2` -- not a shift. + * + * FIVE ORIGINAL-SOURCE COPY-PASTE ARTEFACTS, all byte-proven + * The 4th light box was bolted onto a copy of the 3-box source BY HAND and + * the hand edit was incomplete in five places. Each was read off the target + * and each removed a measured delta. + * (1) `r3lo = r2 - 0x200;` -- box 3's low radius is derived from box 2's + * RANGE VARIABLE, not from its own D_80197C88. Proven by the target's + * `addiu $t6, $s0, -0x200` reusing the register that box 2's `lw` filled; + * spelling it `D_80197C6C - 0x200` re-loads the global (+2 ins). + * (2) Only SIX of the eight radius variables are zero-initialised + * (r0,r1,r2,r0lo,r1lo,r2lo) -- r3/r3lo are left uninitialised, exactly + * the init list the 3-box version needed. + * (3) The ATTEN body is written out LONGHAND 7 times (3 tri vertices + + * 4 quad vertices). When box 3 was bolted on, the `R` of the z- and + * y-axis KILL tests was left as r2 in three of those copies: + * tri v0: z and y use r2 tri v2: z uses r2 all others use r3. + * Proven by the 24 `sll $v0,$s0,16` sites: 3 per group in box 2 plus + * exactly three extra at idx 1471, 1504 (tri v0) and 2178 (tri v2). + * (4) In the QUAD lit arm only, rgb2 and rgb3 take their `<< 16` term from + * c1, not from c2/c3. Proven by the target CSE-ing ONE + * `sll $a0, $a0, 16` and re-using $a0 for all three stores. + * (5) *** ROUND 2 *** In the QUAD arm's VERTEX-1 group only, the box-3 + * ATTEN's Y-axis `else if` branch accumulates into a2v instead of a3v, + * while that same test's KILL branch still says a3v. Modelled by the + * ATTEN3W macro below (`AW` = the y-else destination). Byte-proof: + * idx 3875 addu $a2,$zero,$zero kill branch -> a3v ($a2) [agreed] + * idx 3889 sra $a3,$s2,7 else branch -> a2v ($a3) [was the + * last structural residual] + * Costs zero instructions; the four other quad/tri sites are NOT like + * this (each was measured -- putting the artefact anywhere else is +2). + * + * FRAME (0x360, leaf -- no $ra, no argument area) + * 0x000 tmpxy[4] | 0x010 box[8] | 0x050 sxy[8] | + * 0x090 g{otz,flag,opz,sz0..sz3} | 0x0B0 lim | 0x0B8 j | 0x0C0 i | + * 0x0C8 vd | 0x0D0 ot | 0x0D8 pkt | 0x0E0 f2 | 0x0E8 f3 | + * 0x0F0/0x0F8/0x100 x3,z3,y3 | 0x108 prim | 0x110 nprim | 0x118 vtx | + * 0x120 nparts | 0x128 part | 0x130..0x1D0 lo/hi bounds (21 s16 slots) | + * 0x1D8..0x230 cx0..cz3 (12) | 0x238 r0 | 0x240 r1lo | 0x248 r2lo | + * 0x250 r3lo | 0x288..0x2D0 the LICM-hoisted sign-extended bounds | + * 0x328/0x330 spilled vertex coords | 0x338..0x358 s0-s7,fp. + * *** THE SLOT ORDER IS THE DECLARATION-ORDER ORACLE (see L5). *** + * + * --------------------------------------------------------------------------- + * ROUND-1 LEVERS (kept; measured effect is byte-identical %, anchored) + * + * L1 `cb = (tp[0] & 0xFF000000) | 0x101010;` in both UNLIT arms. + * L2 box-3 ATTEN kill-register per copy (artefact 3) + `r3lo = r2 - 0x200` + * (artefact 1). 89.15% -> 96.96% shape; killed sra+11 / sll+9. + * L3 quad-lit rgb2/rgb3 use `c1 << 16` (artefact 4). Killed the last sll+2. + * L4 [SUPERSEDED BY R3] `s32 c0..c3` declared inside the two CULL blocks. + * L5 `s32 f0, f1, f2, f3;` MOVED TO IMMEDIATELY AFTER `u8 *pkt;`. + * Spilled pseudos get stack slots in PSEUDO-NUMBER order and pseudo + * numbers are handed out in DECLARATION order, so the target's stack + * layout is a direct read-out of its declaration order. After this one + * move ALL 127 stack slots agree with the target exactly. + * L6 `u32 rgbw;` per emit arm. + * L7 RC-15 zero-byte ref dial on `mny`, first statement of the TRI cull + * block -- flips my->$a3 / mny->$a2 to the target's grant. + * L8 `cb` 2-statement accumulator [SUPERSEDED BY R2]; quad-lit rgb word as a + * 3-statement accumulator [kept for rgb0/rgb1, SUPERSEDED for rgb2/rgb3 + * by R5]. + * L9 FOUR REGISTER PINS: va->$t2, w->$a1, f0->$s3, c1->$a0. + * Round 2 removed va and w (see R4); f0 and c1 REMAIN and are both + * load-bearing. This function has NO `jal`, so Sec.74's caller-saved + * pin-spanning-a-call hazard cannot arise -- that is why pins are usable + * on this family member and were a trap on the others. + * + * --------------------------------------------------------------------------- + * ROUND-2 LEVERS -- 45 -> 0. Metric is `match_one` MISMATCH COUNT (length is + * exact throughout, so the raw count is honest). Every number is measured. + * + * THE ONE MECHANISM BEHIND R1/R2/R4/R7. A `register __asm__` pin makes the + * variable a HARD REG in the RTL from the start. When a 1-death local temp is + * produced from, or consumed into, that hard reg, local-alloc.c's + * `combine_regs` takes its hard-register branch (local-alloc.c:1795-1820) and + * records the pinned register in `qty_phys_sugg` for the temp's quantity -- + * UNCONDITIONALLY, there is no death guard on that path. The temp then lands + * in the pinned register and the operation is done IN PLACE. The target, + * whose variable is an ordinary pseudo (reg_qty == -1 for anything crossing a + * block), never gets that suggestion and keeps the temp in $v0/$v1. + * Three independent cures, all used here: + * (i) give the temp a NAMED variable with >1 death, so local-alloc.c:472 + * (`reg_basic_block >= 0 && reg_n_deaths == 1`) refuses it a quantity + * and combine_regs bails at its very first test -> R1 + * (ii) drop the pin, if the pin is not load-bearing -> R4 + * (iii) keep the pinned value LIVE past the temp, so that + * find_free_reg cannot honour the suggestion -> R7 + * + * R1 `CLAMP80S` -- the four-way attenuation sum gets its own named variable + * `sv`, used at BOTH c1 sites (2 deaths). Cure (i). 45 -> 37 + * Target: `addu $v0,..; addu $v0,..; addu $v0,..; addiu ,$v0,0x10`. + * Without it the whole chain is tied into the pinned c1 ($a0). + * Only the two c1 sites: `sv` on all 7 sites collapses the frame (-64). + * R2 UNLIT arms store `cb | 0x101010` as an expression instead of doing + * `cb |= 0x101010` in place. `cb` is a function-scope global allocno, so + * the in-place form writes $a1; the expression form is a 1-death local + * that combine_regs ties to the DYING constant register $v1, which is + * what the target does. 37 -> 33 + * R3 *** `s32 c0, c1, c2, c3;` AT FUNCTION SCOPE, not per cull block. *** + * Read straight off the two MATCHED relatives (func_8017D960 line 310, + * func_8017F510 line 338), and confirmed by the target itself: its TRI + * grants (c0=$t4, c1=$a0, c2=$t2) are IDENTICAL to its QUAD grants, which + * is only possible if both arms share one set of allocnos. Per-cull-block + * scope splits them into two independent allocno sets and the TRI set + * drifts. 33 -> 21 + * Declaration POSITION is neutral (5 anchors swept, all 21). + * NOTE this REVERSES round-1's L4. L4 was correct on the round-1 base -- + * it was supplying the extra local allocno that spills r1lo -- but R1+R2 + * supply that pressure now, and the c1 pin does the rest. + * R4 DROP the `va->$t2` and `w->$a1` pins. With c0..c3 at function scope + * they are no longer load-bearing, and they were the sole cause of the + * prim-word producer ties (`andi`/`srl` written straight into $t2/$a1). + * Cure (ii). 21 -> 17 -> 13 (pair) + * Round 1 measured these as worth 4%; that was true of the round-1 base + * and is FALSE here. Base-dependence, not a contradiction. + * R5 QUAD lit rgb2/rgb3 revert to the SINGLE-EXPRESSION form (rgb0/rgb1 keep + * the 3-statement accumulator). The intermediates then become 1-death + * local temps that alternate $v0/$v1, which is what lets the target's + * store of the previous rgb word sit one slot LATER. 21 -> 11 + * Doing it to all four, or to rgb0/rgb1 only, is worse (23 / 21). + * R6 Artefact 5 -- `ATTEN3W(a3v, a2v, ...)` at the QUAD vertex-1 site. 3 -> 2 + * R7 RC-15 zero-byte ref `__asm__ __volatile__ ("" :: "r" (c1));` placed + * immediately after the TRI arm's rgb1 store. Cure (iii): it keeps the + * pinned c1 ($a0) live past `c1 << 16`, so find_free_reg cannot honour + * combine_regs' $a0 suggestion and the shift goes to $v0. 2 -> 0 + * The c1 pin CANNOT simply be removed: it is what spills r1lo (dropping + * it, or moving it to any other colour, costs -64 length). Measured. + * + * SCHEDULER ATTRIBUTION (Sec.76 primitive, run in round 2, never run before on + * this function). Compiled with -fno-schedule-insns, with + * -fno-schedule-insns2, and with both. The store/shift transposition at + * idx 4643-4652 kept MY source order under all three. It was therefore never + * a `sched.c` decision: the ordering is a CONSEQUENCE of the register grant + * (a 3-statement accumulator pins the value in one register, so no scheduler + * could hoist the next `or` above the `sw`). R5 fixed it by changing the + * grant, exactly as Sec.78 predicts. + * =========================================================================== */ + +#define gte_ldv0(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_ldv3(r0, r1, r2) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 0( %1 );" \ + "lwc2 $3, 4( %1 );" \ + "lwc2 $4, 0( %2 );" \ + "lwc2 $5, 4( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) ) + +#define gte_ldv3c(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 8( %0 );" \ + "lwc2 $3, 12( %0 );" \ + "lwc2 $4, 16( %0 );" \ + "lwc2 $5, 20( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_rtps() __asm__ volatile ("nop;nop;rtps") +#define gte_rtpt() __asm__ volatile ("nop;nop;rtpt") +#define gte_nclip() __asm__ volatile ("nop;nop;nclip") + +#define gte_stsxy(r0) __asm__ volatile ( \ + "swc2 $14, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 0( %1 );" \ + "swc2 $14, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsxy3c(r0) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 4( %0 );" \ + "swc2 $14, 8( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsz3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $17, 0( %0 );" \ + "swc2 $18, 0( %1 );" \ + "swc2 $19, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsz4(r0, r1, r2, r3) __asm__ volatile ( \ + "swc2 $16, 0( %0 );" \ + "swc2 $17, 0( %1 );" \ + "swc2 $18, 0( %2 );" \ + "swc2 $19, 0( %3 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ), "r"( r3 ) \ + : "memory" ) + +#define gte_stszotz(r0) __asm__ volatile ( \ + "mfc2 $12, $19;" \ + "nop;" \ + "sra $12, $12, 2;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stflg(r0) __asm__ volatile ( \ + "cfc2 $12, $31;" \ + "nop;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stopz(r0) __asm__ volatile ( \ + "swc2 $24, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +/* ---- the two gouraud-textured packet layouts this function emits ---------- */ +typedef struct { + u32 tag; + u32 rgb0; s16 x0, y0; u32 uv0; + u32 rgb1; s16 x1, y1; u32 uv1; + u32 rgb2; s16 x2, y2; u16 uv2, p2; +} PolyGT3; /* 0x28 */ + +typedef struct { + u32 tag; + u32 rgb0; s16 x0, y0; u32 uv0; + u32 rgb1; s16 x1, y1; u32 uv1; + u32 rgb2; s16 x2, y2; u16 uv2, p2; + u32 rgb3; s16 x3, y3; u16 uv3, p3; +} PolyGT4; /* 0x34 */ + + +/* ---- the four light-volume descriptors (stride 0x1C) --------------------- */ +extern s32 D_80197C28; +extern u16 D_80197C2C, D_80197C2E, D_80197C30; +extern s32 D_80197C34; +extern s32 D_80197C44; +extern u16 D_80197C48, D_80197C4A, D_80197C4C; +extern s32 D_80197C50; +extern s32 D_80197C60; +extern u16 D_80197C64, D_80197C66, D_80197C68; +extern s32 D_80197C6C; +extern s32 D_80197C7C; +extern u16 D_80197C80, D_80197C82, D_80197C84; +extern u16 D_80197C88; + +/* ---- the box-containment test for one vertex against one light box ------- */ +#define BOXTEST(F, X, Y, Z, LX, HX, LY, HY, LZ, HZ) \ + if ((LX) < (X) && (X) < (HX) && (LY) < (Y) && (Y) < (HY) && (LZ) < (Z) && (Z) < (HZ)) F = 1 + +/* ---- the separable per-axis falloff, visited in x, z, y order ------------ */ +#define ATTEN(A, F, X, Y, Z, CX, CY, CZ, R, RLO) \ + A = 0; \ + if (F) { \ + d = (X) - (CX); if (d < 0) d = (CX) - (X); \ + if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \ + d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \ + if ((R) < d) A = 0; \ + else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \ + d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \ + if ((R) < d) A = 0; \ + else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \ + } + +#define ATTEN3W(A, AW, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \ + A = 0; \ + if (F) { \ + d = (X) - (CX); if (d < 0) d = (CX) - (X); \ + if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \ + d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \ + if ((RZ) < d) A = 0; \ + else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \ + d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \ + if ((RY) < d) A = 0; \ + else if ((RLO) < d) AW = (A * (((R) - d) / 4)) >> 7; \ + } + +#define ATTEN3(A, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \ + A = 0; \ + if (F) { \ + d = (X) - (CX); if (d < 0) d = (CX) - (X); \ + if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \ + d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \ + if ((RZ) < d) A = 0; \ + else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \ + d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \ + if ((RY) < d) A = 0; \ + else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \ + } + +#define CLAMP80(C, A0, A1, A2, A3) C = (A0) + (A1) + (A2) + (A3) + 0x10; if ((C) > 0x80) C = 0x80 +#define CLAMP80S(C, A0, A1, A2, A3) sv = (A0) + (A1) + (A2) + (A3); C = sv + 0x10; if ((C) > 0x80) C = 0x80 + + +void func_8017BF14(s32 arg0, s32 lim) +{ + typedef struct { u32 w0, w1, w2; } Prim; + + extern u8 *D_800A5E60; + extern u8 D_800A6610[]; + extern u8 D_800AF630[]; + + DVECTOR2 tmpxy[4]; + SVECTOR2 box[8]; + SVECTOR2 sxy[8]; + struct { long otz, flag, opz, sz0, sz1, sz2, sz3; } g; + + s32 j; + u32 i; + u8 *vd; + u32 ot; + u8 *pkt; + register s32 f0 __asm__("$19"); + s32 f1, f2, f3; + s16 x3, z3, y3; + Prim *prim; + u32 nprim; + u8 *vtx; + s32 nparts; + Part *part; + s16 lo0x, hi0x, lo0y, hi0y, lo0z, hi0z; + s16 lo1x, hi1x, lo1y, hi1y, lo1z, hi1z; + s16 lo2x, hi2x, lo2y, hi2y, lo2z, hi2z; + s16 lo3x, hi3x, lo3y, hi3y, lo3z, hi3z; + s16 cx0, cy0, cz0, cx1, cy1, cz1, cx2, cy2, cz2, cx3, cy3, cz3; + s16 r0; + s16 r1; + s16 r2; + s16 r3; + s16 r0lo; + s16 r1lo; + s16 r2lo; + s16 r3lo; + u8 *va, *vb, *vc; + u32 w; + s32 code; + u32 vw, vzw; + u32 wx, wy, wz; + s32 xa32, xb32, t32; + s32 xmn1, xmx1, xmn2, xmx2; + s32 mnc, mxc; + s16 my, mny, mx, mn; + u8 *base; + s16 x0, y0, z0, x1, y1, z1, x2, y2, z2; + s32 a0v, a1v, a2v, a3v; + register s32 c1 __asm__("$4"); s32 c0, c2, c3; + s32 d; + u32 *tp; + u32 uvw; + u32 cb; + s32 sv; + + base = D_800AF630; + + r2lo = 0; + r1lo = 0; + r0lo = 0; + r2 = 0; + r1 = 0; + r0 = 0; + + if (D_80197C28) { + cx0 = D_80197C2C; + cy0 = D_80197C2E; + r0lo = D_80197C34 - 0x200; + r0 = D_80197C34; + cz0 = D_80197C30; + } else { + cz0 = 0x6000; + cy0 = 0x6000; + cx0 = 0x6000; + } + if (D_80197C44) { + cx1 = D_80197C48; + cy1 = D_80197C4A; + r1 = D_80197C50; + r1lo = D_80197C50 - 0x200; + cz1 = D_80197C4C; + } else { + cz1 = 0x6000; + cy1 = 0x6000; + cx1 = 0x6000; + } + if (D_80197C60) { + cx2 = D_80197C64; + cy2 = D_80197C66; + r2 = D_80197C6C; + r2lo = D_80197C6C - 0x200; + cz2 = D_80197C68; + } else { + cz2 = 0x6000; + cy2 = 0x6000; + cx2 = 0x6000; + } + if (D_80197C7C) { + cx3 = D_80197C80; + cy3 = D_80197C82; + r3lo = r2 - 0x200; + r3 = D_80197C88; + cz3 = D_80197C84; + } else { + cz3 = 0x6000; + cy3 = 0x6000; + cx3 = 0x6000; + } + + lo0x = cx0 - r0; hi0x = cx0 + r0; + lo0y = cy0 - r0; hi0y = cy0 + r0; + lo0z = cz0 - r0; hi0z = cz0 + r0; + lo1x = cx1 - r1; hi1x = cx1 + r1; + lo1y = cy1 - r1; hi1y = cy1 + r1; + lo1z = cz1 - r1; hi1z = cz1 + r1; + lo2x = cx2 - r2; hi2x = cx2 + r2; + lo2y = cy2 - r2; hi2y = cy2 + r2; + lo2z = cz2 - r2; hi2z = cz2 + r2; + lo3x = cx3 - r3; hi3x = cx3 + r3; + lo3y = cy3 - r3; hi3y = cy3 + r3; + lo3z = cz3 - r3; hi3z = cz3 + r3; + + pkt = D_800A5E60; + part = *(Part **)(arg0 + 0xC); + nparts = *(s32 *)(*(s32 *)(arg0 + 8) + 8); + vtx = *(u8 **)(*(s32 *)(arg0 + 8) + 0x10); + ot = (u32)&D_800A6610[(*(u16 *)(base + 0xA3D2)) << 14]; + + for (j = 0; j < nparts; j++, part++) { + wx = part->xx; + mn = wx; + mx = wx >> 16; + wy = part->yy; + mny = wy; + my = wy >> 16; + wz = part->zz; + box[0].vx = mn; box[0].vy = mny; + box[1].vx = mx; box[1].vy = mny; + box[2].vx = mn; box[2].vy = mny; + box[3].vx = mx; box[3].vy = mny; + box[4].vx = mn; box[4].vy = my; + box[5].vx = mx; box[5].vy = my; + box[6].vx = mn; box[6].vy = my; + box[7].vx = mx; box[7].vy = my; + wy = wz >> 16; + box[0].vz = wz; + box[1].vz = wz; + box[4].vz = wz; + box[5].vz = wz; + box[2].vz = wy; + box[3].vz = wy; + box[6].vz = wy; + box[7].vz = wy; + + gte_ldv3c(&box[0]); + gte_rtpt(); + gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]); + gte_ldv0(&box[3]); + gte_rtps(); + gte_stsxy(&sxy[3]); + gte_ldv3c(&box[4]); + gte_rtpt(); + gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]); + gte_ldv0(&box[7]); + gte_rtps(); + gte_stsxy(&sxy[7]); + gte_stszotz(&g.otz); + + if (lim >= g.otz) { + xa32 = sxy[0].vx; + xb32 = sxy[1].vx; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vx; + xb32 = sxy[5].vx; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0xA0 && (s16)mnc < 0xA1) { + xa32 = sxy[0].vy; + xb32 = sxy[1].vy; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vy; + xb32 = sxy[5].vy; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0x6E && (s16)mnc < 0x6F) { + nprim = part->nprim; + prim = (Prim *)part->prim; + for (i = 0; i < nprim; i++, prim++) { + w = prim->w1; + va = vtx + (w & 0xFFFF); + vb = vtx + (w >> 16); + w = prim->w2; + vc = vtx + (w & 0xFFFF); + w = w >> 16; + gte_ldv3(va, vb, vc); + gte_rtpt(); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_nclip(); + code = w & 7; + vd = vtx + (w & 0xFFF8); + gte_stopz(&g.opz); + if (g.opz > 0) { + switch (code) { + case 6: + case 7: + /* ---------------- TRI (FT3 / GT3) ---------------- */ + gte_stsxy3c(&tmpxy[0]); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (tmpxy[0].vx > tmpxy[1].vx) { mx = tmpxy[0].vx; mn = tmpxy[1].vx; } + else { mn = tmpxy[0].vx; mx = tmpxy[1].vx; } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (mx >= -0xA0 && mn < 0xA1) { + if (tmpxy[0].vy > tmpxy[1].vy) { my = tmpxy[0].vy; mny = tmpxy[1].vy; } + else { mny = tmpxy[0].vy; my = tmpxy[1].vy; } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + if (my >= -0x6E && mny < 0x6F) { + s32 za, zb; + __asm__ __volatile__ ("" :: "r" (mny)); + if (g.sz0 > g.sz1) { za = g.sz0; if (za < g.sz2) za = g.sz2; } + else { za = g.sz1; if (za < g.sz2) za = g.sz2; } + g.opz = za; + + f3 = 0; f2 = 0; f1 = 0; f0 = 0; + + vw = *(u32 *)va; + vzw = *(u32 *)(va + 4); + x0 = vw; y0 = vw >> 16; z0 = vzw; + BOXTEST(f0, x0, y0, z0, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z); + BOXTEST(f1, x0, y0, z0, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z); + BOXTEST(f2, x0, y0, z0, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z); + BOXTEST(f3, x0, y0, z0, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z); + vw = *(u32 *)vb; + vzw = *(u32 *)(vb + 4); + x1 = vw; y1 = vw >> 16; z1 = vzw; + BOXTEST(f0, x1, y1, z1, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z); + BOXTEST(f1, x1, y1, z1, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z); + BOXTEST(f2, x1, y1, z1, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z); + BOXTEST(f3, x1, y1, z1, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z); + vw = *(u32 *)vc; + vzw = *(u32 *)(vc + 4); + x2 = vw; y2 = vw >> 16; z2 = vzw; + BOXTEST(f0, x2, y2, z2, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z); + BOXTEST(f1, x2, y2, z2, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z); + BOXTEST(f2, x2, y2, z2, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z); + BOXTEST(f3, x2, y2, z2, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z); + + if (f0 | f1 | f2 | f3) { + u32 *otp; + u32 rgbw; + ATTEN(a0v, f0, x0, y0, z0, cx0, cy0, cz0, r0, r0lo); + ATTEN(a1v, f1, x0, y0, z0, cx1, cy1, cz1, r1, r1lo); + ATTEN(a2v, f2, x0, y0, z0, cx2, cy2, cz2, r2, r2lo); + ATTEN3(a3v, f3, x0, y0, z0, cx3, cy3, cz3, r3, r3lo, r2, r2); + CLAMP80(c0, a0v, a1v, a2v, a3v); + ATTEN(a0v, f0, x1, y1, z1, cx0, cy0, cz0, r0, r0lo); + ATTEN(a1v, f1, x1, y1, z1, cx1, cy1, cz1, r1, r1lo); + ATTEN(a2v, f2, x1, y1, z1, cx2, cy2, cz2, r2, r2lo); + ATTEN3(a3v, f3, x1, y1, z1, cx3, cy3, cz3, r3, r3lo, r3, r3); + CLAMP80S(c1, a0v, a1v, a2v, a3v); + ATTEN(a0v, f0, x2, y2, z2, cx0, cy0, cz0, r0, r0lo); + ATTEN(a1v, f1, x2, y2, z2, cx1, cy1, cz1, r1, r1lo); + ATTEN(a2v, f2, x2, y2, z2, cx2, cy2, cz2, r2, r2lo); + ATTEN3(a3v, f3, x2, y2, z2, cx3, cy3, cz3, r3, r3lo, r2, r3); + CLAMP80(c2, a0v, a1v, a2v, a3v); + + *(u32 *)&((PolyGT3 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyGT3 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyGT3 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + tp = (u32 *)prim->w0; + cb = 0x34000000; + rgbw = (c0 | cb) | (c0 << 8) | (c0 << 16); + ((PolyGT3 *)pkt)->rgb0 = rgbw; + rgbw = (c1 | cb) | (c1 << 8) | (c1 << 16); + ((PolyGT3 *)pkt)->rgb1 = rgbw; + __asm__ __volatile__ ("" :: "r" (c1)); + rgbw = (c2 | cb) | (c2 << 8) | (c2 << 16); + ((PolyGT3 *)pkt)->rgb2 = rgbw; + ((PolyGT3 *)pkt)->uv0 = tp[1]; + ((PolyGT3 *)pkt)->uv1 = tp[2]; + ((PolyGT3 *)pkt)->uv2 = tp[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + } else { + u32 *otp; + u32 rgbw; + *(u32 *)&((PolyFT3 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT3 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT3 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + tp = (u32 *)prim->w0; + cb = tp[0] & 0xFF000000; + ((PolyFT3 *)pkt)->rgbc = cb | 0x101010; + ((PolyFT3 *)pkt)->uvc0 = tp[1]; + ((PolyFT3 *)pkt)->uvp1 = tp[2]; + ((PolyFT3 *)pkt)->uv2 = tp[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x20; + } + } + } + break; + case 2: + case 3: + /* ---------------- QUAD (FT4 / GT4) ---------------- */ + gte_stsxy3c(&tmpxy[0]); + gte_ldv0(vd); + gte_rtps(); + if (tmpxy[0].vx > tmpxy[1].vx) { mx = tmpxy[0].vx; mn = tmpxy[1].vx; } + else { mn = tmpxy[0].vx; mx = tmpxy[1].vx; } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[0].vy > tmpxy[1].vy) { my = tmpxy[0].vy; mny = tmpxy[1].vy; } + else { mny = tmpxy[0].vy; my = tmpxy[1].vy; } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyFT4 *)pkt)->x3); + if (((PolyFT4 *)pkt)->x3 < mn) mn = ((PolyFT4 *)pkt)->x3; + else if (mx < ((PolyFT4 *)pkt)->x3) mx = ((PolyFT4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT4 *)pkt)->y3 < mny) mny = ((PolyFT4 *)pkt)->y3; + else if (my < ((PolyFT4 *)pkt)->y3) my = ((PolyFT4 *)pkt)->y3; + if (my >= -0x6E && mny < 0x6F) { + s32 za, zb; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + + f3 = 0; f2 = 0; f1 = 0; f0 = 0; + + vw = *(u32 *)va; + vzw = *(u32 *)(va + 4); + x0 = vw; y0 = vw >> 16; z0 = vzw; + BOXTEST(f0, x0, y0, z0, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z); + BOXTEST(f1, x0, y0, z0, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z); + BOXTEST(f2, x0, y0, z0, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z); + BOXTEST(f3, x0, y0, z0, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z); + vw = *(u32 *)vb; + vzw = *(u32 *)(vb + 4); + x1 = vw; y1 = vw >> 16; z1 = vzw; + BOXTEST(f0, x1, y1, z1, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z); + BOXTEST(f1, x1, y1, z1, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z); + BOXTEST(f2, x1, y1, z1, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z); + BOXTEST(f3, x1, y1, z1, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z); + vw = *(u32 *)vc; + vzw = *(u32 *)(vc + 4); + x2 = vw; y2 = vw >> 16; z2 = vzw; + BOXTEST(f0, x2, y2, z2, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z); + BOXTEST(f1, x2, y2, z2, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z); + BOXTEST(f2, x2, y2, z2, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z); + BOXTEST(f3, x2, y2, z2, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z); + vw = *(u32 *)vd; + vzw = *(u32 *)(vd + 4); + x3 = vw; y3 = vw >> 16; z3 = vzw; + BOXTEST(f0, x3, y3, z3, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z); + BOXTEST(f1, x3, y3, z3, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z); + BOXTEST(f2, x3, y3, z3, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z); + BOXTEST(f3, x3, y3, z3, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z); + + if (f0 | f1 | f2 | f3) { + u32 *otp; + u32 rgbw; + ATTEN(a0v, f0, x0, y0, z0, cx0, cy0, cz0, r0, r0lo); + ATTEN(a1v, f1, x0, y0, z0, cx1, cy1, cz1, r1, r1lo); + ATTEN(a2v, f2, x0, y0, z0, cx2, cy2, cz2, r2, r2lo); + ATTEN3(a3v, f3, x0, y0, z0, cx3, cy3, cz3, r3, r3lo, r3, r3); + CLAMP80(c0, a0v, a1v, a2v, a3v); + ATTEN(a0v, f0, x1, y1, z1, cx0, cy0, cz0, r0, r0lo); + ATTEN(a1v, f1, x1, y1, z1, cx1, cy1, cz1, r1, r1lo); + ATTEN(a2v, f2, x1, y1, z1, cx2, cy2, cz2, r2, r2lo); + ATTEN3W(a3v, a2v, f3, x1, y1, z1, cx3, cy3, cz3, r3, r3lo, r3, r3); + CLAMP80S(c1, a0v, a1v, a2v, a3v); + ATTEN(a0v, f0, x2, y2, z2, cx0, cy0, cz0, r0, r0lo); + ATTEN(a1v, f1, x2, y2, z2, cx1, cy1, cz1, r1, r1lo); + ATTEN(a2v, f2, x2, y2, z2, cx2, cy2, cz2, r2, r2lo); + ATTEN3(a3v, f3, x2, y2, z2, cx3, cy3, cz3, r3, r3lo, r3, r3); + CLAMP80(c2, a0v, a1v, a2v, a3v); + ATTEN(a0v, f0, x3, y3, z3, cx0, cy0, cz0, r0, r0lo); + ATTEN(a1v, f1, x3, y3, z3, cx1, cy1, cz1, r1, r1lo); + ATTEN(a2v, f2, x3, y3, z3, cx2, cy2, cz2, r2, r2lo); + ATTEN3(a3v, f3, x3, y3, z3, cx3, cy3, cz3, r3, r3lo, r3, r3); + CLAMP80(c3, a0v, a1v, a2v, a3v); + + *(u32 *)&((PolyGT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyGT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyGT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + gte_stsxy((long *)&((PolyGT4 *)pkt)->x3); + tp = (u32 *)prim->w0; + cb = 0x3C000000; + rgbw = c0 | cb; rgbw |= c0 << 8; rgbw |= c0 << 16; + ((PolyGT4 *)pkt)->rgb0 = rgbw; + rgbw = c1 | cb; rgbw |= c1 << 8; rgbw |= c1 << 16; + ((PolyGT4 *)pkt)->rgb1 = rgbw; + rgbw = (c2 | cb) | (c2 << 8) | (c1 << 16); + ((PolyGT4 *)pkt)->rgb2 = rgbw; + rgbw = (c3 | cb) | (c3 << 8) | (c1 << 16); + ((PolyGT4 *)pkt)->rgb3 = rgbw; + ((PolyGT4 *)pkt)->uv0 = tp[1]; + ((PolyGT4 *)pkt)->uv1 = tp[2]; + uvw = tp[3]; + ((PolyGT4 *)pkt)->uv2 = uvw; + ((PolyGT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0xC000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x34; + } else { + u32 *otp; + u32 rgbw; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + tp = (u32 *)prim->w0; + cb = tp[0] & 0xFF000000; + ((PolyFT4 *)pkt)->rgbc = cb | 0x101010; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + } + } + } + } + break; + } + } + } + } + } + } + } + } + D_800A5E60 = pkt; +} diff --git a/docs/matching-cookbook.md b/docs/matching-cookbook.md index a31efefc77..96f4f5d5c7 100644 --- a/docs/matching-cookbook.md +++ b/docs/matching-cookbook.md @@ -6194,3 +6194,56 @@ Artifacts: `.run/giants/s19_func_8017BF14_b1.c` (45/4763), a **pin-free fallback **Cold-start economics, measured:** a 4,763-instruction leaf giant with a *findable* matched relative reached 99.06% in one pass but did not close. Budget a second pass for anything this size; the first pass buys the decode, the frame, and the length — the last ~1% is register grants. + +## §80 — A do-not-re-buy entry is scoped to its BASE, not to the function; and the pin's hidden cost is an unconditional `qty_phys_sugg` (Phase 29 SESSION-19, `func_8017BF14` 45 → 0) + +Round 2 closed the 4,763-instruction behemoth (`45 → 37 → 33 → 21 → 11 → 3 → 2 → 0`, reproduced 3× +from independent work dirs, banked whole-binary BYTE-IDENTICAL). The route matters more than the win. + +### ⚠️ THE PROCESS CORRECTION: a measured negative is relative to the draft it was measured on +Round 1 left a careful ~40-row do-not-re-buy table. **Three of its entries INVERTED on round 2's base.** +The same edit (`qsingle23`) measured **1,040 mismatched on the 45-base and 11 on the 21-base**. +Re-testing the round-1 negative list cost **~20 seconds** and produced **three of the seven winning +levers**. + +**So: a do-not-re-buy table is a record of `(edit, base) → result`, NOT `edit → useless`.** After any +lever that moves the base materially, **re-run the negative list** — it is seconds with a real harness +and it is where the next levers hide. This retroactively qualifies every such table in this cookbook +(§45, §60b, §75a, §76, §78, §79 and round 1 of this function): treat them as *starting hypotheses at +the base where they were taken*, not as closed questions. + +Corollary already seen: round 1 measured "removing the `va→$t2` pin costs 4% elsewhere" and concluded +*keep the pin*. On a base where `c0..c3` sit at function scope, **removing those pins is worth 21→13** +— the opposite conclusion from the same experiment. + +### The pin's hidden cost, with the citation +`combine_regs`' hard-register branch (`local-alloc.c:1795`, reached from `:1295` with +`already_dead == 0`) records the pinned register in **`qty_phys_sugg` unconditionally — there is no +death guard.** So a `register __asm__` pin does not merely *prefer* a register: it actively invites +local-alloc to tie producer chains into it, which is exactly the residual-(a) tie round 1 diagnosed +but mis-cured. Three separable cures exist; the new one is worth knowing: +- **R7 — a zero-byte `__asm__` ref that keeps the pinned value LIVE PAST the temp**, so + `find_free_reg` cannot honour the suggestion. That closed the final 2 instructions, and was + *necessary* because `c1→$a0` proved uniquely load-bearing (it is what spills `r1lo`; every + alternative pin lost 64 instructions). + +### The flagged "#1 move" LOST — and why the failure is informative +Variable REUSE (§45-A / RC-14 MERGE) was swept in full: **every merge lost, 43–3294 across 8 merges.** +It was the right lever class for the sibling `func_8017F510` (97 → 10) and the wrong one here, for a +structural reason worth carrying: **the TRI and QUAD grants did not differ by RANK, they differed by +IDENTITY — two independent allocno sets.** Re-ranking inside one set cannot fix a two-set problem. +**Diagnose whether you have a ranking problem or an identity problem before reaching for a merge.** +The actual fix was `s32 c0,c1,c2,c3;` at **function** scope (33 → 21), read off the two matched +relatives (`b5:310`, `b4:338`) and confirmed against the target itself: its TRI grants are *identical* +to its QUAD grants. + +### §78's attribution primitive, run and reproduced +Under `-fno-schedule-insns`, `-fno-schedule-insns2`, and both, the draft's order was **unchanged** ⇒ +the rgb-accumulator transposition was never a `sched.c` decision. A 3-statement accumulator pins the +value to one register, so no scheduler *could* hoist the `or` above the `sw`. Changing the grant fixed +the order for free — §78 reproduced on a second function. + +### Cold-start economics, now complete +A 4,763-instruction leaf giant with a findable matched relative: **round 1 = decode + exact length + +exact frame + 99.06%; round 2 = the last 45.** Two passes, and the second was far cheaper than the +first. Budget two passes at this size and do not read a 99% round-1 result as a stall. diff --git a/docs/progress.fleet.md b/docs/progress.fleet.md index 7bc4601098..b604538fde 100644 --- a/docs/progress.fleet.md +++ b/docs/progress.fleet.md @@ -4,16 +4,16 @@ # cross-binary collapsible-byte leverage: docs/duplicates.cross.md. # THREE progress metrics (all matter — see the labels): -FLEET fn-count byte-ident: 315457 / 353717 = 89.18% (REAL+LINKED+empties; FUNCTION-count, ×134-inflated — one crack counts per overlay) -FLEET instr-weighted : 10580590 / 13141652 = 80.5% (shipped .text across main + resident + 138 overlays; the decomp.dev-DISPLAY number) -FLEET distinct-code(uniq): 3838143 / 5634875 = 68.1% (64905/87459 unique fns; the DISTINCT-RE number) +FLEET fn-count byte-ident: 315458 / 353717 = 89.18% (REAL+LINKED+empties; FUNCTION-count, ×134-inflated — one crack counts per overlay) +FLEET instr-weighted : 10585353 / 13141652 = 80.5% (shipped .text across main + resident + 138 overlays; the decomp.dev-DISPLAY number) +FLEET distinct-code(uniq): 3842906 / 5634875 = 68.2% (64906/87459 unique fns; the DISTINCT-RE number) MAIN game-code weighted : 436 / 60201 = 0.7% (INCLUDED in the fleet numbers above since 2026-07-22 — roadmap §1 metrics contract; LINKED-excluding Ghidra sig dated 2026-06-14; caveat is R34: no independent second oracle for a PS-X EXE, NOT drift) - (fleet EXCLUDING main, for continuity with pre-2026-07-22 readings: 10580154 / 13081451 = 80.9%) + (fleet EXCLUDING main, for continuity with pre-2026-07-22 readings: 10584917 / 13081451 = 80.9%) -FLEET REAL substantive : 313602 (of which dedup-shared 239530 via 1886 groups / 239604 instances) +FLEET REAL substantive : 313603 (of which dedup-shared 239530 via 1886 groups / 239604 instances) FLEET LINKED PsyQ objs : 959 FLEET NON_MATCHING : 7 (0 in any default build — G4) -FLEET INCLUDE_ASM stubs : 38253 +FLEET INCLUDE_ASM stubs : 38252 FLEET matchable : 353717 | binary | REAL | shared | LINKED | byte-ident | matchable | byte-ident % | @@ -89,7 +89,7 @@ FLEET matchable : 353717 | ov_SC03_113 | 2252 | 1740 | 0 | 2255 | 2468 | 91.4% | | ov_SC03_114 | 2242 | 1738 | 0 | 2244 | 2415 | 92.9% | | ov_SC03_115 | 2259 | 1738 | 0 | 2261 | 2472 | 91.5% | -| ov_SC03_116 | 2249 | 1738 | 0 | 2252 | 2438 | 92.4% | +| ov_SC03_116 | 2250 | 1738 | 0 | 2253 | 2438 | 92.4% | | ov_SC03_117 | 2277 | 1738 | 0 | 2283 | 2557 | 89.3% | | ov_SC03_118 | 2323 | 1762 | 0 | 2324 | 2685 | 86.6% | | ov_SC03_119 | 2322 | 1762 | 0 | 2323 | 2685 | 86.5% | diff --git a/phase-ends/CURRENT_PHASE.md b/phase-ends/CURRENT_PHASE.md index e8eb935f13..b51b616787 100644 --- a/phase-ends/CURRENT_PHASE.md +++ b/phase-ends/CURRENT_PHASE.md @@ -3871,6 +3871,48 @@ conditional) · main-EXE/B9 + GLM/B6 + resident's 14 walls (P30) · behemoths B7 Artifacts: `s19_func_8017BF14_b1.c` (45/4763) + a **pin-free fallback at 789/4763 that is 100% structural** + `s19_bf14_report.md` (~40-row do-not-re-buy table, 4 refuted diagnoses). +- **🏆🏆🏆 2026-07-25 (SESSION-19) — `func_8017BF14` (4,763 ins) CLOSED IN ROUND 2: 45 → 0. + The 4th behemoth of the session, and the largest single function matched in the project.** + `45 → 37 → 33 → 21 → 11 → 3 → 2 → 0`, reproduced 3× from independent work dirs. **Verified + independently (R14):** `match_one` → **MATCH (4763 ins)**; `harvest_verify --binary ov_SC03_116` + → **BYTE-IDENTICAL**. (Agent was interrupted mid-run by a weekly API limit and RESUMED FROM ITS + TRANSCRIPT — its round-2 harness `bf14_mk2.py`/`bf14_sw2.sh` survived intact, so nothing was + re-derived.) + **⚠️ THE PROCESS CORRECTION THAT MATTERS MORE THAN THE MATCH (→ §80): A DO-NOT-RE-BUY ENTRY IS + SCOPED TO ITS BASE, NOT TO THE FUNCTION.** Three of round 1's ~40 carefully-measured negatives + **INVERTED** on round 2's base — the same edit (`qsingle23`) measured **1,040 mismatched on the + 45-base and 11 on the 21-base**. Re-testing the round-1 negative list cost **~20 seconds** and + produced **three of the seven winning levers**. **A do-not-re-buy table records `(edit, base) → + result`, NOT `edit → useless`; after any lever that moves the base materially, RE-RUN THE NEGATIVE + LIST.** This retroactively qualifies every such table in the cookbook (§45, §60b, §75a, §76, §78, + §79) — they are starting hypotheses at the base where they were taken, not closed questions. + Concrete instance: round 1 measured "removing the `va→$t2` pin costs 4% elsewhere" ⇒ *keep the pin*; + on a base with `c0..c3` at function scope, **removing those pins is worth 21→13** — opposite + conclusion, same experiment. + **MY FLAGGED "#1 MOVE" LOST, AND THE FAILURE IS THE FINDING.** I briefed variable REUSE (§45-A / + RC-14) as the #1 lever because it took `func_8017F510` from 97→10. Swept in full here: **every + merge lost, 43–3294 across 8 merges.** Reason: **the TRI and QUAD grants did not differ by RANK, + they differed by IDENTITY — two independent allocno sets, and re-ranking inside one set cannot fix + a two-set problem.** Diagnose ranking-vs-identity BEFORE reaching for a merge. The actual fix + (`s32 c0,c1,c2,c3;` at **function** scope, 33→21) was **read off the two matched relatives** + (`b5:310`, `b4:338`) and confirmed against the target (its TRI grants are identical to its QUAD + grants) — **the 4th time today that reading a matched relative beat the clever lever.** + **THE PIN'S HIDDEN COST, WITH A CITATION:** `combine_regs`' hard-register branch + (`local-alloc.c:1795`, reached from `:1295` with `already_dead == 0`) records the pinned register in + **`qty_phys_sugg` UNCONDITIONALLY — no death guard**. A pin doesn't merely *prefer* a register, it + invites local-alloc to tie producer chains into it. New cure **R7**: a zero-byte `__asm__` ref that + keeps the pinned value LIVE PAST the temp so `find_free_reg` can't honour the suggestion — closed + the last 2 ins, and was necessary because `c1→$a0` is uniquely load-bearing (it is what spills + `r1lo`; every alternative pin lost 64 ins). + **§78's ATTRIBUTION PRIMITIVE RUN AND REPRODUCED:** under `-fno-schedule-insns`, + `-fno-schedule-insns2`, and both, the order was **unchanged** ⇒ the rgb-accumulator transposition + was never a `sched.c` decision (a 3-statement accumulator pins the value, so no scheduler *could* + hoist the `or` above the `sw`). Changing the grant fixed the order for free. + **COLD-START ECONOMICS, NOW COMPLETE:** round 1 = decode + exact length + exact frame + 99.06%; + round 2 = the last 45, and far cheaper than round 1. **Budget TWO passes at this size; do not read a + 99% round-1 result as a stall.** Also found a **5th original-source copy-paste artefact** (QUAD + vertex-1 box-3 y-axis accumulates into `a2v` while its kill branch still says `a3v`). + > **🛑 SESSION-19 CLOSING CHECKPOINT (2026-07-25, Opus 5 @ High) — REFRESHED mid-session; supersedes > both the SESSION-18 block and the earlier SESSION-19 block (which was written before the > ENGINE_SHB / class-B / dedup_extend-bug work and went stale). Fresh session safe here.** diff --git a/src/ov_SC03_116/ov_SC03_116_jr_8017AE2C.c b/src/ov_SC03_116/ov_SC03_116_jr_8017AE2C.c index cc48e3c820..aa399bbf29 100644 --- a/src/ov_SC03_116/ov_SC03_116_jr_8017AE2C.c +++ b/src/ov_SC03_116/ov_SC03_116_jr_8017AE2C.c @@ -3276,7 +3276,846 @@ DEFINE_func_8017BEB4() /* dedup: shared engine-core @0x8017BEB4 (src/shared) */ INCLUDE_ASM("asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C", func_8017BEBC); -INCLUDE_ASM("asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C", func_8017BF14); +#include "common.h" +#include "/home/musashi/bfm-decomp/src/shared/engine_types.h" + +/* =========================================================================== + * func_8017BF14 -- 4,763 ins, ov_SC03_116 (behemoth #4). *** MATCH *** + * + * STATUS (2026-07-25, session 20 round 2, gcc-2.7.2 pinned triple): + * python3 tools/match_one.py func_8017BF14 --c \ + * --asm-subdir asm/ov_SC03_116/nonmatchings/ov_SC03_116_jr_8017AE2C + * -> MATCH (4763 ins) func_8017BF14 [reproduced 3x, private work dirs] + * LENGTH EXACT | OPCODE HISTOGRAM EXACT (L1 = 0) | STACK FRAME EXACT + * (all 127 slots at the target's offsets, frame 0x360) + * masked index-wise diff: 0 / 4763 mismatched. + * `match_one` is the CANDIDATE gate. The whole-binary SHA1 arbiter (G3/P9) + * is run by the coordinator and is the only thing that makes this final. + * + * Round 1 closed at 45/4763 mismatched. Round 2 took 45 -> 37 -> 33 -> 21 + * -> 11 -> 3 -> 2 -> 0. See .run/giants/s19_bf14_report2.md. + * + * WHAT IT IS + * The *four*-light-box variant of the volumetric-light renderer whose + * 3-box sibling func_8017D960 (3,338 ins, ov_SC03_090) is MATCHED, and whose + * unlit ancestor func_8017BEBC (ov_SC03_099) is MATCHED. Same family, same + * skeleton; this is the biggest member. + * + * Signature: func_8017BF14(s32 arg0, s32 lim). Unlike every other member of + * the family this one is a LEAF -- 0 callees. The 3-call prologue + * (func_800491EC / func_800547D8 / func_80052E38) of the siblings is gone; + * `lim` arrives as arg1 (spilled to 0xB0). That is why the frame has no + * 0x10 argument area (tmpxy[] starts at sp+0x00) and no $ra save. + * + * Per part (stride 0x14, outer loop): build the 8-corner AABB in box[], + * rtpt/rtpt + rtps/rtps -> sxy[8], stszotz -> g.otz, reject on + * `lim >= g.otz`, then screen-space bbox reject on X (-0xA0..0xA1) and + * Y (-0x6E..0x6F). + * Per prim (stride 0xC, inner loop): rtpt the 3 vertices, stflg mask + * 0x7F85E000, nclip, stopz > 0, then a 4-way range tree on `code = w & 7` + * that keeps ONLY codes 6,7 (tri) and 2,3 (quad); 0,1,4,5 fall through to + * the loop tail. + * Per drawn poly: screen bbox reject, then each vertex is tested against + * FOUR axis-aligned light boxes (flags f0..f3), and if any is lit a + * 0x00..0x80 attenuation per active box is computed, summed, biased +0x10 + * and clamped to 0x80 -> a grey gouraud vertex colour. + * lit -> POLY_GT3 (0x28, tag 0x34000000, OT 0x9000000) + * POLY_GT4 (0x34, tag 0x3C000000, OT 0xC000000) + * unlit -> POLY_FT3 (0x20, OT 0x7000000) / POLY_FT4 (0x28, OT 0x9000000) + * with rgbc = (tp[0] & 0xFF000000) | 0x101010 <-- NOT black, + * unlike func_8017D960 where the unlit colour is plain black. + * + * THE FOUR LIGHT BOXES (stride 0x1C, {s32 enable; u16 cx,cy,cz; s32 range}) + * D_80197C28 / D_80197C44 / D_80197C60 / D_80197C7C. + * Falloff geometry differs from the 3-box sibling: RLO = R - 0x200 (not + * -0x80) and the ramp is ((R - d) / 4) (not (R - d)), so the 0x80 ceiling is + * reached over a 0x200-wide band instead of 0x80. The `/ 4` is a SIGNED + * divide -- `bgez / addiu 3 / sra 2` -- not a shift. + * + * FIVE ORIGINAL-SOURCE COPY-PASTE ARTEFACTS, all byte-proven + * The 4th light box was bolted onto a copy of the 3-box source BY HAND and + * the hand edit was incomplete in five places. Each was read off the target + * and each removed a measured delta. + * (1) `r3lo = r2 - 0x200;` -- box 3's low radius is derived from box 2's + * RANGE VARIABLE, not from its own D_80197C88. Proven by the target's + * `addiu $t6, $s0, -0x200` reusing the register that box 2's `lw` filled; + * spelling it `D_80197C6C - 0x200` re-loads the global (+2 ins). + * (2) Only SIX of the eight radius variables are zero-initialised + * (r0,r1,r2,r0lo,r1lo,r2lo) -- r3/r3lo are left uninitialised, exactly + * the init list the 3-box version needed. + * (3) The ATTEN body is written out LONGHAND 7 times (3 tri vertices + + * 4 quad vertices). When box 3 was bolted on, the `R` of the z- and + * y-axis KILL tests was left as r2 in three of those copies: + * tri v0: z and y use r2 tri v2: z uses r2 all others use r3. + * Proven by the 24 `sll $v0,$s0,16` sites: 3 per group in box 2 plus + * exactly three extra at idx 1471, 1504 (tri v0) and 2178 (tri v2). + * (4) In the QUAD lit arm only, rgb2 and rgb3 take their `<< 16` term from + * c1, not from c2/c3. Proven by the target CSE-ing ONE + * `sll $a0, $a0, 16` and re-using $a0 for all three stores. + * (5) *** ROUND 2 *** In the QUAD arm's VERTEX-1 group only, the box-3 + * ATTEN's Y-axis `else if` branch accumulates into a2v instead of a3v, + * while that same test's KILL branch still says a3v. Modelled by the + * ATTEN3W macro below (`AW` = the y-else destination). Byte-proof: + * idx 3875 addu $a2,$zero,$zero kill branch -> a3v ($a2) [agreed] + * idx 3889 sra $a3,$s2,7 else branch -> a2v ($a3) [was the + * last structural residual] + * Costs zero instructions; the four other quad/tri sites are NOT like + * this (each was measured -- putting the artefact anywhere else is +2). + * + * FRAME (0x360, leaf -- no $ra, no argument area) + * 0x000 tmpxy[4] | 0x010 box[8] | 0x050 sxy[8] | + * 0x090 g{otz,flag,opz,sz0..sz3} | 0x0B0 lim | 0x0B8 j | 0x0C0 i | + * 0x0C8 vd | 0x0D0 ot | 0x0D8 pkt | 0x0E0 f2 | 0x0E8 f3 | + * 0x0F0/0x0F8/0x100 x3,z3,y3 | 0x108 prim | 0x110 nprim | 0x118 vtx | + * 0x120 nparts | 0x128 part | 0x130..0x1D0 lo/hi bounds (21 s16 slots) | + * 0x1D8..0x230 cx0..cz3 (12) | 0x238 r0 | 0x240 r1lo | 0x248 r2lo | + * 0x250 r3lo | 0x288..0x2D0 the LICM-hoisted sign-extended bounds | + * 0x328/0x330 spilled vertex coords | 0x338..0x358 s0-s7,fp. + * *** THE SLOT ORDER IS THE DECLARATION-ORDER ORACLE (see L5). *** + * + * --------------------------------------------------------------------------- + * ROUND-1 LEVERS (kept; measured effect is byte-identical %, anchored) + * + * L1 `cb = (tp[0] & 0xFF000000) | 0x101010;` in both UNLIT arms. + * L2 box-3 ATTEN kill-register per copy (artefact 3) + `r3lo = r2 - 0x200` + * (artefact 1). 89.15% -> 96.96% shape; killed sra+11 / sll+9. + * L3 quad-lit rgb2/rgb3 use `c1 << 16` (artefact 4). Killed the last sll+2. + * L4 [SUPERSEDED BY R3] `s32 c0..c3` declared inside the two CULL blocks. + * L5 `s32 f0, f1, f2, f3;` MOVED TO IMMEDIATELY AFTER `u8 *pkt;`. + * Spilled pseudos get stack slots in PSEUDO-NUMBER order and pseudo + * numbers are handed out in DECLARATION order, so the target's stack + * layout is a direct read-out of its declaration order. After this one + * move ALL 127 stack slots agree with the target exactly. + * L6 `u32 rgbw;` per emit arm. + * L7 RC-15 zero-byte ref dial on `mny`, first statement of the TRI cull + * block -- flips my->$a3 / mny->$a2 to the target's grant. + * L8 `cb` 2-statement accumulator [SUPERSEDED BY R2]; quad-lit rgb word as a + * 3-statement accumulator [kept for rgb0/rgb1, SUPERSEDED for rgb2/rgb3 + * by R5]. + * L9 FOUR REGISTER PINS: va->$t2, w->$a1, f0->$s3, c1->$a0. + * Round 2 removed va and w (see R4); f0 and c1 REMAIN and are both + * load-bearing. This function has NO `jal`, so Sec.74's caller-saved + * pin-spanning-a-call hazard cannot arise -- that is why pins are usable + * on this family member and were a trap on the others. + * + * --------------------------------------------------------------------------- + * ROUND-2 LEVERS -- 45 -> 0. Metric is `match_one` MISMATCH COUNT (length is + * exact throughout, so the raw count is honest). Every number is measured. + * + * THE ONE MECHANISM BEHIND R1/R2/R4/R7. A `register __asm__` pin makes the + * variable a HARD REG in the RTL from the start. When a 1-death local temp is + * produced from, or consumed into, that hard reg, local-alloc.c's + * `combine_regs` takes its hard-register branch (local-alloc.c:1795-1820) and + * records the pinned register in `qty_phys_sugg` for the temp's quantity -- + * UNCONDITIONALLY, there is no death guard on that path. The temp then lands + * in the pinned register and the operation is done IN PLACE. The target, + * whose variable is an ordinary pseudo (reg_qty == -1 for anything crossing a + * block), never gets that suggestion and keeps the temp in $v0/$v1. + * Three independent cures, all used here: + * (i) give the temp a NAMED variable with >1 death, so local-alloc.c:472 + * (`reg_basic_block >= 0 && reg_n_deaths == 1`) refuses it a quantity + * and combine_regs bails at its very first test -> R1 + * (ii) drop the pin, if the pin is not load-bearing -> R4 + * (iii) keep the pinned value LIVE past the temp, so that + * find_free_reg cannot honour the suggestion -> R7 + * + * R1 `CLAMP80S` -- the four-way attenuation sum gets its own named variable + * `sv`, used at BOTH c1 sites (2 deaths). Cure (i). 45 -> 37 + * Target: `addu $v0,..; addu $v0,..; addu $v0,..; addiu ,$v0,0x10`. + * Without it the whole chain is tied into the pinned c1 ($a0). + * Only the two c1 sites: `sv` on all 7 sites collapses the frame (-64). + * R2 UNLIT arms store `cb | 0x101010` as an expression instead of doing + * `cb |= 0x101010` in place. `cb` is a function-scope global allocno, so + * the in-place form writes $a1; the expression form is a 1-death local + * that combine_regs ties to the DYING constant register $v1, which is + * what the target does. 37 -> 33 + * R3 *** `s32 c0, c1, c2, c3;` AT FUNCTION SCOPE, not per cull block. *** + * Read straight off the two MATCHED relatives (func_8017D960 line 310, + * func_8017F510 line 338), and confirmed by the target itself: its TRI + * grants (c0=$t4, c1=$a0, c2=$t2) are IDENTICAL to its QUAD grants, which + * is only possible if both arms share one set of allocnos. Per-cull-block + * scope splits them into two independent allocno sets and the TRI set + * drifts. 33 -> 21 + * Declaration POSITION is neutral (5 anchors swept, all 21). + * NOTE this REVERSES round-1's L4. L4 was correct on the round-1 base -- + * it was supplying the extra local allocno that spills r1lo -- but R1+R2 + * supply that pressure now, and the c1 pin does the rest. + * R4 DROP the `va->$t2` and `w->$a1` pins. With c0..c3 at function scope + * they are no longer load-bearing, and they were the sole cause of the + * prim-word producer ties (`andi`/`srl` written straight into $t2/$a1). + * Cure (ii). 21 -> 17 -> 13 (pair) + * Round 1 measured these as worth 4%; that was true of the round-1 base + * and is FALSE here. Base-dependence, not a contradiction. + * R5 QUAD lit rgb2/rgb3 revert to the SINGLE-EXPRESSION form (rgb0/rgb1 keep + * the 3-statement accumulator). The intermediates then become 1-death + * local temps that alternate $v0/$v1, which is what lets the target's + * store of the previous rgb word sit one slot LATER. 21 -> 11 + * Doing it to all four, or to rgb0/rgb1 only, is worse (23 / 21). + * R6 Artefact 5 -- `ATTEN3W(a3v, a2v, ...)` at the QUAD vertex-1 site. 3 -> 2 + * R7 RC-15 zero-byte ref `__asm__ __volatile__ ("" :: "r" (c1));` placed + * immediately after the TRI arm's rgb1 store. Cure (iii): it keeps the + * pinned c1 ($a0) live past `c1 << 16`, so find_free_reg cannot honour + * combine_regs' $a0 suggestion and the shift goes to $v0. 2 -> 0 + * The c1 pin CANNOT simply be removed: it is what spills r1lo (dropping + * it, or moving it to any other colour, costs -64 length). Measured. + * + * SCHEDULER ATTRIBUTION (Sec.76 primitive, run in round 2, never run before on + * this function). Compiled with -fno-schedule-insns, with + * -fno-schedule-insns2, and with both. The store/shift transposition at + * idx 4643-4652 kept MY source order under all three. It was therefore never + * a `sched.c` decision: the ordering is a CONSEQUENCE of the register grant + * (a 3-statement accumulator pins the value in one register, so no scheduler + * could hoist the next `or` above the `sw`). R5 fixed it by changing the + * grant, exactly as Sec.78 predicts. + * =========================================================================== */ + +#define gte_ldv0(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_ldv3(r0, r1, r2) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 0( %1 );" \ + "lwc2 $3, 4( %1 );" \ + "lwc2 $4, 0( %2 );" \ + "lwc2 $5, 4( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) ) + +#define gte_ldv3c(r0) __asm__ volatile ( \ + "lwc2 $0, 0( %0 );" \ + "lwc2 $1, 4( %0 );" \ + "lwc2 $2, 8( %0 );" \ + "lwc2 $3, 12( %0 );" \ + "lwc2 $4, 16( %0 );" \ + "lwc2 $5, 20( %0 )" \ + : \ + : "r"( r0 ) ) + +#define gte_rtps() __asm__ volatile ("nop;nop;rtps") +#define gte_rtpt() __asm__ volatile ("nop;nop;rtpt") +#define gte_nclip() __asm__ volatile ("nop;nop;nclip") + +#define gte_stsxy(r0) __asm__ volatile ( \ + "swc2 $14, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsxy3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 0( %1 );" \ + "swc2 $14, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsxy3c(r0) __asm__ volatile ( \ + "swc2 $12, 0( %0 );" \ + "swc2 $13, 4( %0 );" \ + "swc2 $14, 8( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +#define gte_stsz3(r0, r1, r2) __asm__ volatile ( \ + "swc2 $17, 0( %0 );" \ + "swc2 $18, 0( %1 );" \ + "swc2 $19, 0( %2 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ) \ + : "memory" ) + +#define gte_stsz4(r0, r1, r2, r3) __asm__ volatile ( \ + "swc2 $16, 0( %0 );" \ + "swc2 $17, 0( %1 );" \ + "swc2 $18, 0( %2 );" \ + "swc2 $19, 0( %3 )" \ + : \ + : "r"( r0 ), "r"( r1 ), "r"( r2 ), "r"( r3 ) \ + : "memory" ) + +#define gte_stszotz(r0) __asm__ volatile ( \ + "mfc2 $12, $19;" \ + "nop;" \ + "sra $12, $12, 2;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stflg(r0) __asm__ volatile ( \ + "cfc2 $12, $31;" \ + "nop;" \ + "sw $12, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "$12", "memory" ) + +#define gte_stopz(r0) __asm__ volatile ( \ + "swc2 $24, 0( %0 )" \ + : \ + : "r"( r0 ) \ + : "memory" ) + +/* ---- the two gouraud-textured packet layouts this function emits ---------- */ +typedef struct { + u32 tag; + u32 rgb0; s16 x0, y0; u32 uv0; + u32 rgb1; s16 x1, y1; u32 uv1; + u32 rgb2; s16 x2, y2; u16 uv2, p2; +} PolyGT3; /* 0x28 */ + +typedef struct { + u32 tag; + u32 rgb0; s16 x0, y0; u32 uv0; + u32 rgb1; s16 x1, y1; u32 uv1; + u32 rgb2; s16 x2, y2; u16 uv2, p2; + u32 rgb3; s16 x3, y3; u16 uv3, p3; +} PolyGT4; /* 0x34 */ + + +/* ---- the four light-volume descriptors (stride 0x1C) --------------------- */ +extern s32 D_80197C28; +extern u16 D_80197C2C, D_80197C2E, D_80197C30; +extern s32 D_80197C34; +extern s32 D_80197C44; +extern u16 D_80197C48, D_80197C4A, D_80197C4C; +extern s32 D_80197C50; +extern s32 D_80197C60; +extern u16 D_80197C64, D_80197C66, D_80197C68; +extern s32 D_80197C6C; +extern s32 D_80197C7C; +extern u16 D_80197C80, D_80197C82, D_80197C84; +extern u16 D_80197C88; + +/* ---- the box-containment test for one vertex against one light box ------- */ +#define BOXTEST(F, X, Y, Z, LX, HX, LY, HY, LZ, HZ) \ + if ((LX) < (X) && (X) < (HX) && (LY) < (Y) && (Y) < (HY) && (LZ) < (Z) && (Z) < (HZ)) F = 1 + +/* ---- the separable per-axis falloff, visited in x, z, y order ------------ */ +#define ATTEN(A, F, X, Y, Z, CX, CY, CZ, R, RLO) \ + A = 0; \ + if (F) { \ + d = (X) - (CX); if (d < 0) d = (CX) - (X); \ + if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \ + d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \ + if ((R) < d) A = 0; \ + else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \ + d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \ + if ((R) < d) A = 0; \ + else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \ + } + +#define ATTEN3W(A, AW, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \ + A = 0; \ + if (F) { \ + d = (X) - (CX); if (d < 0) d = (CX) - (X); \ + if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \ + d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \ + if ((RZ) < d) A = 0; \ + else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \ + d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \ + if ((RY) < d) A = 0; \ + else if ((RLO) < d) AW = (A * (((R) - d) / 4)) >> 7; \ + } + +#define ATTEN3(A, F, X, Y, Z, CX, CY, CZ, R, RLO, RZ, RY) \ + A = 0; \ + if (F) { \ + d = (X) - (CX); if (d < 0) d = (CX) - (X); \ + if (d < (R)) { A = 0x80; if (d >= (RLO)) A = ((R) - d) / 4; } \ + d = (Z) - (CZ); if (d < 0) d = (CZ) - (Z); \ + if ((RZ) < d) A = 0; \ + else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \ + d = (Y) - (CY); if (d < 0) d = (CY) - (Y); \ + if ((RY) < d) A = 0; \ + else if ((RLO) < d) A = (A * (((R) - d) / 4)) >> 7; \ + } + +#define CLAMP80(C, A0, A1, A2, A3) C = (A0) + (A1) + (A2) + (A3) + 0x10; if ((C) > 0x80) C = 0x80 +#define CLAMP80S(C, A0, A1, A2, A3) sv = (A0) + (A1) + (A2) + (A3); C = sv + 0x10; if ((C) > 0x80) C = 0x80 + + +void func_8017BF14(s32 arg0, s32 lim) +{ + typedef struct { u32 w0, w1, w2; } Prim; + + extern u8 *D_800A5E60; + extern u8 D_800A6610[]; + extern u8 D_800AF630[]; + + DVECTOR2 tmpxy[4]; + SVECTOR2 box[8]; + SVECTOR2 sxy[8]; + struct { long otz, flag, opz, sz0, sz1, sz2, sz3; } g; + + s32 j; + u32 i; + u8 *vd; + u32 ot; + u8 *pkt; + register s32 f0 __asm__("$19"); + s32 f1, f2, f3; + s16 x3, z3, y3; + Prim *prim; + u32 nprim; + u8 *vtx; + s32 nparts; + Part *part; + s16 lo0x, hi0x, lo0y, hi0y, lo0z, hi0z; + s16 lo1x, hi1x, lo1y, hi1y, lo1z, hi1z; + s16 lo2x, hi2x, lo2y, hi2y, lo2z, hi2z; + s16 lo3x, hi3x, lo3y, hi3y, lo3z, hi3z; + s16 cx0, cy0, cz0, cx1, cy1, cz1, cx2, cy2, cz2, cx3, cy3, cz3; + s16 r0; + s16 r1; + s16 r2; + s16 r3; + s16 r0lo; + s16 r1lo; + s16 r2lo; + s16 r3lo; + u8 *va, *vb, *vc; + u32 w; + s32 code; + u32 vw, vzw; + u32 wx, wy, wz; + s32 xa32, xb32, t32; + s32 xmn1, xmx1, xmn2, xmx2; + s32 mnc, mxc; + s16 my, mny, mx, mn; + u8 *base; + s16 x0, y0, z0, x1, y1, z1, x2, y2, z2; + s32 a0v, a1v, a2v, a3v; + register s32 c1 __asm__("$4"); s32 c0, c2, c3; + s32 d; + u32 *tp; + u32 uvw; + u32 cb; + s32 sv; + + base = D_800AF630; + + r2lo = 0; + r1lo = 0; + r0lo = 0; + r2 = 0; + r1 = 0; + r0 = 0; + + if (D_80197C28) { + cx0 = D_80197C2C; + cy0 = D_80197C2E; + r0lo = D_80197C34 - 0x200; + r0 = D_80197C34; + cz0 = D_80197C30; + } else { + cz0 = 0x6000; + cy0 = 0x6000; + cx0 = 0x6000; + } + if (D_80197C44) { + cx1 = D_80197C48; + cy1 = D_80197C4A; + r1 = D_80197C50; + r1lo = D_80197C50 - 0x200; + cz1 = D_80197C4C; + } else { + cz1 = 0x6000; + cy1 = 0x6000; + cx1 = 0x6000; + } + if (D_80197C60) { + cx2 = D_80197C64; + cy2 = D_80197C66; + r2 = D_80197C6C; + r2lo = D_80197C6C - 0x200; + cz2 = D_80197C68; + } else { + cz2 = 0x6000; + cy2 = 0x6000; + cx2 = 0x6000; + } + if (D_80197C7C) { + cx3 = D_80197C80; + cy3 = D_80197C82; + r3lo = r2 - 0x200; + r3 = D_80197C88; + cz3 = D_80197C84; + } else { + cz3 = 0x6000; + cy3 = 0x6000; + cx3 = 0x6000; + } + + lo0x = cx0 - r0; hi0x = cx0 + r0; + lo0y = cy0 - r0; hi0y = cy0 + r0; + lo0z = cz0 - r0; hi0z = cz0 + r0; + lo1x = cx1 - r1; hi1x = cx1 + r1; + lo1y = cy1 - r1; hi1y = cy1 + r1; + lo1z = cz1 - r1; hi1z = cz1 + r1; + lo2x = cx2 - r2; hi2x = cx2 + r2; + lo2y = cy2 - r2; hi2y = cy2 + r2; + lo2z = cz2 - r2; hi2z = cz2 + r2; + lo3x = cx3 - r3; hi3x = cx3 + r3; + lo3y = cy3 - r3; hi3y = cy3 + r3; + lo3z = cz3 - r3; hi3z = cz3 + r3; + + pkt = D_800A5E60; + part = *(Part **)(arg0 + 0xC); + nparts = *(s32 *)(*(s32 *)(arg0 + 8) + 8); + vtx = *(u8 **)(*(s32 *)(arg0 + 8) + 0x10); + ot = (u32)&D_800A6610[(*(u16 *)(base + 0xA3D2)) << 14]; + + for (j = 0; j < nparts; j++, part++) { + wx = part->xx; + mn = wx; + mx = wx >> 16; + wy = part->yy; + mny = wy; + my = wy >> 16; + wz = part->zz; + box[0].vx = mn; box[0].vy = mny; + box[1].vx = mx; box[1].vy = mny; + box[2].vx = mn; box[2].vy = mny; + box[3].vx = mx; box[3].vy = mny; + box[4].vx = mn; box[4].vy = my; + box[5].vx = mx; box[5].vy = my; + box[6].vx = mn; box[6].vy = my; + box[7].vx = mx; box[7].vy = my; + wy = wz >> 16; + box[0].vz = wz; + box[1].vz = wz; + box[4].vz = wz; + box[5].vz = wz; + box[2].vz = wy; + box[3].vz = wy; + box[6].vz = wy; + box[7].vz = wy; + + gte_ldv3c(&box[0]); + gte_rtpt(); + gte_stsxy3(&sxy[0], &sxy[1], &sxy[2]); + gte_ldv0(&box[3]); + gte_rtps(); + gte_stsxy(&sxy[3]); + gte_ldv3c(&box[4]); + gte_rtpt(); + gte_stsxy3(&sxy[4], &sxy[5], &sxy[6]); + gte_ldv0(&box[7]); + gte_rtps(); + gte_stsxy(&sxy[7]); + gte_stszotz(&g.otz); + + if (lim >= g.otz) { + xa32 = sxy[0].vx; + xb32 = sxy[1].vx; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vx; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vx; + xb32 = sxy[5].vx; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vx; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0xA0 && (s16)mnc < 0xA1) { + xa32 = sxy[0].vy; + xb32 = sxy[1].vy; + if (xb32 < xa32) { xmx1 = xa32; xmn1 = xb32; } else { xmn1 = xa32; xmx1 = xb32; } + t32 = sxy[2].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + t32 = sxy[3].vy; + if (xmx1 < t32) xmx1 = t32; else if (t32 < xmn1) xmn1 = t32; + xa32 = sxy[4].vy; + xb32 = sxy[5].vy; + if (xb32 < xa32) { xmx2 = xa32; xmn2 = xb32; } else { xmn2 = xa32; xmx2 = xb32; } + t32 = sxy[6].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + t32 = sxy[7].vy; + if (xmx2 < t32) xmx2 = t32; else if (t32 < xmn2) xmn2 = t32; + mnc = xmn1; + if (xmn2 < xmn1) mnc = xmn2; + mxc = xmx1; + if (mxc < xmx2) mxc = xmx2; + if ((s16)mxc >= -0x6E && (s16)mnc < 0x6F) { + nprim = part->nprim; + prim = (Prim *)part->prim; + for (i = 0; i < nprim; i++, prim++) { + w = prim->w1; + va = vtx + (w & 0xFFFF); + vb = vtx + (w >> 16); + w = prim->w2; + vc = vtx + (w & 0xFFFF); + w = w >> 16; + gte_ldv3(va, vb, vc); + gte_rtpt(); + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_nclip(); + code = w & 7; + vd = vtx + (w & 0xFFF8); + gte_stopz(&g.opz); + if (g.opz > 0) { + switch (code) { + case 6: + case 7: + /* ---------------- TRI (FT3 / GT3) ---------------- */ + gte_stsxy3c(&tmpxy[0]); + gte_stsz3(&g.sz0, &g.sz1, &g.sz2); + if (tmpxy[0].vx > tmpxy[1].vx) { mx = tmpxy[0].vx; mn = tmpxy[1].vx; } + else { mn = tmpxy[0].vx; mx = tmpxy[1].vx; } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (mx >= -0xA0 && mn < 0xA1) { + if (tmpxy[0].vy > tmpxy[1].vy) { my = tmpxy[0].vy; mny = tmpxy[1].vy; } + else { mny = tmpxy[0].vy; my = tmpxy[1].vy; } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + if (my >= -0x6E && mny < 0x6F) { + s32 za, zb; + __asm__ __volatile__ ("" :: "r" (mny)); + if (g.sz0 > g.sz1) { za = g.sz0; if (za < g.sz2) za = g.sz2; } + else { za = g.sz1; if (za < g.sz2) za = g.sz2; } + g.opz = za; + + f3 = 0; f2 = 0; f1 = 0; f0 = 0; + + vw = *(u32 *)va; + vzw = *(u32 *)(va + 4); + x0 = vw; y0 = vw >> 16; z0 = vzw; + BOXTEST(f0, x0, y0, z0, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z); + BOXTEST(f1, x0, y0, z0, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z); + BOXTEST(f2, x0, y0, z0, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z); + BOXTEST(f3, x0, y0, z0, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z); + vw = *(u32 *)vb; + vzw = *(u32 *)(vb + 4); + x1 = vw; y1 = vw >> 16; z1 = vzw; + BOXTEST(f0, x1, y1, z1, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z); + BOXTEST(f1, x1, y1, z1, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z); + BOXTEST(f2, x1, y1, z1, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z); + BOXTEST(f3, x1, y1, z1, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z); + vw = *(u32 *)vc; + vzw = *(u32 *)(vc + 4); + x2 = vw; y2 = vw >> 16; z2 = vzw; + BOXTEST(f0, x2, y2, z2, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z); + BOXTEST(f1, x2, y2, z2, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z); + BOXTEST(f2, x2, y2, z2, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z); + BOXTEST(f3, x2, y2, z2, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z); + + if (f0 | f1 | f2 | f3) { + u32 *otp; + u32 rgbw; + ATTEN(a0v, f0, x0, y0, z0, cx0, cy0, cz0, r0, r0lo); + ATTEN(a1v, f1, x0, y0, z0, cx1, cy1, cz1, r1, r1lo); + ATTEN(a2v, f2, x0, y0, z0, cx2, cy2, cz2, r2, r2lo); + ATTEN3(a3v, f3, x0, y0, z0, cx3, cy3, cz3, r3, r3lo, r2, r2); + CLAMP80(c0, a0v, a1v, a2v, a3v); + ATTEN(a0v, f0, x1, y1, z1, cx0, cy0, cz0, r0, r0lo); + ATTEN(a1v, f1, x1, y1, z1, cx1, cy1, cz1, r1, r1lo); + ATTEN(a2v, f2, x1, y1, z1, cx2, cy2, cz2, r2, r2lo); + ATTEN3(a3v, f3, x1, y1, z1, cx3, cy3, cz3, r3, r3lo, r3, r3); + CLAMP80S(c1, a0v, a1v, a2v, a3v); + ATTEN(a0v, f0, x2, y2, z2, cx0, cy0, cz0, r0, r0lo); + ATTEN(a1v, f1, x2, y2, z2, cx1, cy1, cz1, r1, r1lo); + ATTEN(a2v, f2, x2, y2, z2, cx2, cy2, cz2, r2, r2lo); + ATTEN3(a3v, f3, x2, y2, z2, cx3, cy3, cz3, r3, r3lo, r2, r3); + CLAMP80(c2, a0v, a1v, a2v, a3v); + + *(u32 *)&((PolyGT3 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyGT3 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyGT3 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + tp = (u32 *)prim->w0; + cb = 0x34000000; + rgbw = (c0 | cb) | (c0 << 8) | (c0 << 16); + ((PolyGT3 *)pkt)->rgb0 = rgbw; + rgbw = (c1 | cb) | (c1 << 8) | (c1 << 16); + ((PolyGT3 *)pkt)->rgb1 = rgbw; + __asm__ __volatile__ ("" :: "r" (c1)); + rgbw = (c2 | cb) | (c2 << 8) | (c2 << 16); + ((PolyGT3 *)pkt)->rgb2 = rgbw; + ((PolyGT3 *)pkt)->uv0 = tp[1]; + ((PolyGT3 *)pkt)->uv1 = tp[2]; + ((PolyGT3 *)pkt)->uv2 = tp[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + } else { + u32 *otp; + u32 rgbw; + *(u32 *)&((PolyFT3 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT3 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT3 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + tp = (u32 *)prim->w0; + cb = tp[0] & 0xFF000000; + ((PolyFT3 *)pkt)->rgbc = cb | 0x101010; + ((PolyFT3 *)pkt)->uvc0 = tp[1]; + ((PolyFT3 *)pkt)->uvp1 = tp[2]; + ((PolyFT3 *)pkt)->uv2 = tp[3]; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x7000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x20; + } + } + } + break; + case 2: + case 3: + /* ---------------- QUAD (FT4 / GT4) ---------------- */ + gte_stsxy3c(&tmpxy[0]); + gte_ldv0(vd); + gte_rtps(); + if (tmpxy[0].vx > tmpxy[1].vx) { mx = tmpxy[0].vx; mn = tmpxy[1].vx; } + else { mn = tmpxy[0].vx; mx = tmpxy[1].vx; } + if (tmpxy[2].vx > mx) mx = tmpxy[2].vx; + else if (tmpxy[2].vx < mn) mn = tmpxy[2].vx; + if (tmpxy[0].vy > tmpxy[1].vy) { my = tmpxy[0].vy; mny = tmpxy[1].vy; } + else { mny = tmpxy[0].vy; my = tmpxy[1].vy; } + if (tmpxy[2].vy > my) my = tmpxy[2].vy; + else if (tmpxy[2].vy < mny) mny = tmpxy[2].vy; + gte_stflg(&g.flag); + if (!(g.flag & 0x7F85E000)) { + gte_stsz4(&g.sz0, &g.sz1, &g.sz2, &g.sz3); + gte_stsxy((long *)&((PolyFT4 *)pkt)->x3); + if (((PolyFT4 *)pkt)->x3 < mn) mn = ((PolyFT4 *)pkt)->x3; + else if (mx < ((PolyFT4 *)pkt)->x3) mx = ((PolyFT4 *)pkt)->x3; + if (mx >= -0xA0 && mn < 0xA1) { + if (((PolyFT4 *)pkt)->y3 < mny) mny = ((PolyFT4 *)pkt)->y3; + else if (my < ((PolyFT4 *)pkt)->y3) my = ((PolyFT4 *)pkt)->y3; + if (my >= -0x6E && mny < 0x6F) { + s32 za, zb; + zb = g.sz2; + if (zb < g.sz3) zb = g.sz3; + za = g.sz0; + if (za < g.sz1) za = g.sz1; + if (za < zb) za = zb; + g.opz = za; + + f3 = 0; f2 = 0; f1 = 0; f0 = 0; + + vw = *(u32 *)va; + vzw = *(u32 *)(va + 4); + x0 = vw; y0 = vw >> 16; z0 = vzw; + BOXTEST(f0, x0, y0, z0, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z); + BOXTEST(f1, x0, y0, z0, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z); + BOXTEST(f2, x0, y0, z0, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z); + BOXTEST(f3, x0, y0, z0, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z); + vw = *(u32 *)vb; + vzw = *(u32 *)(vb + 4); + x1 = vw; y1 = vw >> 16; z1 = vzw; + BOXTEST(f0, x1, y1, z1, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z); + BOXTEST(f1, x1, y1, z1, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z); + BOXTEST(f2, x1, y1, z1, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z); + BOXTEST(f3, x1, y1, z1, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z); + vw = *(u32 *)vc; + vzw = *(u32 *)(vc + 4); + x2 = vw; y2 = vw >> 16; z2 = vzw; + BOXTEST(f0, x2, y2, z2, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z); + BOXTEST(f1, x2, y2, z2, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z); + BOXTEST(f2, x2, y2, z2, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z); + BOXTEST(f3, x2, y2, z2, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z); + vw = *(u32 *)vd; + vzw = *(u32 *)(vd + 4); + x3 = vw; y3 = vw >> 16; z3 = vzw; + BOXTEST(f0, x3, y3, z3, lo0x, hi0x, lo0y, hi0y, lo0z, hi0z); + BOXTEST(f1, x3, y3, z3, lo1x, hi1x, lo1y, hi1y, lo1z, hi1z); + BOXTEST(f2, x3, y3, z3, lo2x, hi2x, lo2y, hi2y, lo2z, hi2z); + BOXTEST(f3, x3, y3, z3, lo3x, hi3x, lo3y, hi3y, lo3z, hi3z); + + if (f0 | f1 | f2 | f3) { + u32 *otp; + u32 rgbw; + ATTEN(a0v, f0, x0, y0, z0, cx0, cy0, cz0, r0, r0lo); + ATTEN(a1v, f1, x0, y0, z0, cx1, cy1, cz1, r1, r1lo); + ATTEN(a2v, f2, x0, y0, z0, cx2, cy2, cz2, r2, r2lo); + ATTEN3(a3v, f3, x0, y0, z0, cx3, cy3, cz3, r3, r3lo, r3, r3); + CLAMP80(c0, a0v, a1v, a2v, a3v); + ATTEN(a0v, f0, x1, y1, z1, cx0, cy0, cz0, r0, r0lo); + ATTEN(a1v, f1, x1, y1, z1, cx1, cy1, cz1, r1, r1lo); + ATTEN(a2v, f2, x1, y1, z1, cx2, cy2, cz2, r2, r2lo); + ATTEN3W(a3v, a2v, f3, x1, y1, z1, cx3, cy3, cz3, r3, r3lo, r3, r3); + CLAMP80S(c1, a0v, a1v, a2v, a3v); + ATTEN(a0v, f0, x2, y2, z2, cx0, cy0, cz0, r0, r0lo); + ATTEN(a1v, f1, x2, y2, z2, cx1, cy1, cz1, r1, r1lo); + ATTEN(a2v, f2, x2, y2, z2, cx2, cy2, cz2, r2, r2lo); + ATTEN3(a3v, f3, x2, y2, z2, cx3, cy3, cz3, r3, r3lo, r3, r3); + CLAMP80(c2, a0v, a1v, a2v, a3v); + ATTEN(a0v, f0, x3, y3, z3, cx0, cy0, cz0, r0, r0lo); + ATTEN(a1v, f1, x3, y3, z3, cx1, cy1, cz1, r1, r1lo); + ATTEN(a2v, f2, x3, y3, z3, cx2, cy2, cz2, r2, r2lo); + ATTEN3(a3v, f3, x3, y3, z3, cx3, cy3, cz3, r3, r3lo, r3, r3); + CLAMP80(c3, a0v, a1v, a2v, a3v); + + *(u32 *)&((PolyGT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyGT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyGT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + gte_stsxy((long *)&((PolyGT4 *)pkt)->x3); + tp = (u32 *)prim->w0; + cb = 0x3C000000; + rgbw = c0 | cb; rgbw |= c0 << 8; rgbw |= c0 << 16; + ((PolyGT4 *)pkt)->rgb0 = rgbw; + rgbw = c1 | cb; rgbw |= c1 << 8; rgbw |= c1 << 16; + ((PolyGT4 *)pkt)->rgb1 = rgbw; + rgbw = (c2 | cb) | (c2 << 8) | (c1 << 16); + ((PolyGT4 *)pkt)->rgb2 = rgbw; + rgbw = (c3 | cb) | (c3 << 8) | (c1 << 16); + ((PolyGT4 *)pkt)->rgb3 = rgbw; + ((PolyGT4 *)pkt)->uv0 = tp[1]; + ((PolyGT4 *)pkt)->uv1 = tp[2]; + uvw = tp[3]; + ((PolyGT4 *)pkt)->uv2 = uvw; + ((PolyGT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0xC000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x34; + } else { + u32 *otp; + u32 rgbw; + *(u32 *)&((PolyFT4 *)pkt)->x0 = *(u32 *)&tmpxy[0]; + *(u32 *)&((PolyFT4 *)pkt)->x1 = *(u32 *)&tmpxy[1]; + *(u32 *)&((PolyFT4 *)pkt)->x2 = *(u32 *)&tmpxy[2]; + tp = (u32 *)prim->w0; + cb = tp[0] & 0xFF000000; + ((PolyFT4 *)pkt)->rgbc = cb | 0x101010; + ((PolyFT4 *)pkt)->uvc0 = tp[1]; + ((PolyFT4 *)pkt)->uvp1 = tp[2]; + uvw = tp[3]; + ((PolyFT4 *)pkt)->uv2 = uvw; + ((PolyFT4 *)pkt)->uv3 = uvw >> 16; + otp = (u32 *)(((g.opz >> 2) << 2) + ot); + *(u32 *)pkt = (*otp & 0xFFFFFF) | 0x9000000; + *otp = (*otp & 0xFF000000) | ((u32)pkt & 0xFFFFFF); + pkt += 0x28; + } + } + } + } + break; + } + } + } + } + } + } + } + } + D_800A5E60 = pkt; +} + extern s32 D_8012704C;