Files
BFM-decomp/src/800b2.c
T
Drew T 110570e3fc feat(phase-31): wave O — 46 main functions banked, main EXE byte-identical (S52)
36 fresh wave-O cracks + 10 recovered wave-J/K/L drafts, verified in ONE clean rebuild:
143dbb89f34491258bbc27810d0a12ec8b43a8dd BYTE-IDENTICAL. main stubs 1881 -> 1835.

Wave O was a 3-arm 49-card wave (6,266 ins): main head-crack, main UNKNOWN, overlay UNKNOWN.
47/49 standalone MATCH, independently re-verified by me (R14) at 47/49 -- exact agreement --
and reloc_identity reported 46 AGREE / 0 MISMATCH, the first wave of the campaign with zero
symbol errors. THE UNKNOWN LEVER DRAFTS LIKE ANY OTHER LANE, which matters strategically: it is
~138k ins fleet-wide (a quarter of everything open) and was routed as "needs its own lane".

FOUR gate_main defects fixed here, each of which had been silently costing prior waves drafts:
- typedef stripping walked drafts in SLATE order while substitution happens at ADDRESS order, so
  the surviving typedef could land BELOW a draft using it -> "syntax error before D_800A651C".
  Verified the two orders genuinely diverge for both destination files in this slate.
- a BUILD failure (sha None) fell through to a silent bisect -- a full clean rebuild per step to
  rediscover what the compiler had already printed and discarded. Now the error lines are shown
  and the offending drafts named for undefined-reference/redefinition/conflicting-types.
  (My first version of that printer TAILED a stderr+stdout concatenation and faithfully showed 25
  lines of make progress chatter instead of the error -- selecting by position, not by content.)
- typesig treated "short" and "s16" as different types (R39 over-refusal). Aliases now normalize;
  11/11 NC cases pass, with signedness, volatile and array-vs-scalar still conflicting correctly.
- conflict detection ignored shared headers: engine_core.h's DEFINE_ macros declare symbols in
  their own bodies, so a draft's file-scope array decl of D_800A651C was illegal. Block-scoping
  the draft's extern fixes it byte-identically.

DECLARATION RECONCILIATION took the slate from 5 dropped to 0, and three of the four conflicts
were load-bearing CODEGEN, not style: the array form of D_80078D88 blocks a sched1 hoist (scalar
users adopt [0] for free); "volatile" on D_800B9A02 is required by one draft and fatal to two
others (plain u16 loses 1 bank, volatile loses 2); D_800A651C needs block scope. Cookbook §176f.
2026-08-15 14:01:19 -06:00

680 lines
20 KiB
C

#include "common.h"
/*
* GsTMDfastG3GL (0x80057928) -- PsyQ libgs, HANDWRITTEN assembly (splat marks it
* "Handwritten function"). It is raw GTE code: cop2 compute ops (rtpt/nclip/avsz3/nccs),
* direct lwc2/swc2 to numbered cop2 data registers, cfc2 $31, hand-filled branch delay
* slots and explicit cop2/load-latency nops, 8 arguments (4 in $a0-$a3, 4 on the stack),
* and an allocation that uses only $t0-$t7. gcc-2.7.2 -O2 cannot emit any of this from C.
*
* So it is reproduced the way the project's other handwritten GTE bodies are
* (cookbook: DEFINE_func_8013E2C4 et al.): the whole body is one __asm__ __volatile__
* block. gcc supplies only the label and the zero-frame leaf epilogue (`jr $ra` + nop),
* which is exactly the target's tail. The GTE compute ops are written as `cop2 <imm>`
* because binutils has no rtpt/nclip/avsz3/nccs mnemonics (the splat .s gets them from
* include/gte_macros.inc, which is not reachable from inline asm):
* rtpt = 0x4A280030 -> cop2 0x0280030
* nclip = 0x4B400006 -> cop2 0x1400006
* avsz3 = 0x4B58002D -> cop2 0x158002D
* nccs = 0x4B08041B -> cop2 0x108041B
* No relocations and no symbol references at all: the function touches only its
* arguments and the GTE.
*/
void *GsTMDfastG3GL()
{
__asm__ __volatile__(
".set\tnoreorder\n"
"addiu $sp, $sp, -8\n"
"addu $10, $4, $0\n"
"lw $15, 24($sp)\n"
"lw $2, 28($sp)\n"
"lw $3, 32($sp)\n"
"lw $8, 36($sp)\n"
"lw $3, 4($3)\n"
"addu $11, $0, $0\n"
"sw $2, 12($8)\n"
"blez $15, 2f\n"
" sw $3, 16($8)\n"
"addiu $14, $8, 24\n"
"lui $13, 0xFF\n"
"ori $13, $13, 0xFFFF\n"
"addiu $9, $4, 24\n"
"1:\n"
"lhu $4, -6($9)\n"
"lhu $3, -2($9)\n"
"lhu $2, 2($9)\n"
"sll $4, $4, 3\n"
"addu $4, $5, $4\n"
"sll $3, $3, 3\n"
"addu $3, $5, $3\n"
"sll $2, $2, 3\n"
"addu $2, $5, $2\n"
"lwc2 $0, 0($4)\n"
"lwc2 $1, 4($4)\n"
"lwc2 $2, 0($3)\n"
"lwc2 $3, 4($3)\n"
"lwc2 $4, 0($2)\n"
"lwc2 $5, 4($2)\n"
"nop\n"
"nop\n"
"cop2 0x0280030\n" /* rtpt */
"addiu $2, $8, 36\n"
"cfc2 $12, $31\n"
"nop\n"
"sw $12, 0($2)\n"
"lw $2, 36($8)\n"
"nop\n"
"bltz $2, 3f\n"
" nop\n"
"nop\n"
"nop\n"
"cop2 0x1400006\n" /* nclip */
"swc2 $24, 0($14)\n"
"lw $2, 24($8)\n"
"nop\n"
"blez $2, 3f\n"
" nop\n"
"swc2 $12, 8($7)\n"
"swc2 $13, 16($7)\n"
"swc2 $14, 24($7)\n"
"nop\n"
"nop\n"
"cop2 0x158002D\n" /* avsz3 */
"swc2 $7, 0($14)\n"
"addiu $2, $10, 4\n"
"lwc2 $6, 0($2)\n"
"lhu $2, -8($9)\n"
"nop\n"
"sll $2, $2, 3\n"
"addu $2, $6, $2\n"
"lwc2 $0, 0($2)\n"
"lwc2 $1, 4($2)\n"
"nop\n"
"nop\n"
"cop2 0x108041B\n" /* nccs */
"lw $3, 24($8)\n"
"lw $2, 12($8)\n"
"nop\n"
"srav $3, $3, $2\n"
"lw $2, 16($8)\n"
"sll $3, $3, 2\n"
"addu $2, $2, $3\n"
"sw $2, 52($8)\n"
"addiu $2, $7, 4\n"
"swc2 $22, 0($2)\n"
"addiu $2, $10, 8\n"
"lwc2 $6, 0($2)\n"
"lhu $2, -4($9)\n"
"nop\n"
"sll $2, $2, 3\n"
"addu $2, $6, $2\n"
"lwc2 $0, 0($2)\n"
"lwc2 $1, 4($2)\n"
"nop\n"
"nop\n"
"cop2 0x108041B\n" /* nccs */
"lw $2, 52($8)\n"
"nop\n"
"lw $2, 0($2)\n"
"lui $3, 0x600\n"
"and $2, $2, $13\n"
"or $2, $2, $3\n"
"sw $2, 0($7)\n"
"addiu $2, $7, 12\n"
"swc2 $22, 0($2)\n"
"addiu $2, $10, 12\n"
"lwc2 $6, 0($2)\n"
"lhu $2, 0($9)\n"
"nop\n"
"sll $2, $2, 3\n"
"addu $2, $6, $2\n"
"lwc2 $0, 0($2)\n"
"lwc2 $1, 4($2)\n"
"nop\n"
"nop\n"
"cop2 0x108041B\n" /* nccs */
"lw $3, 52($8)\n"
"and $2, $7, $13\n"
"sw $2, 0($3)\n"
"addiu $2, $7, 20\n"
"swc2 $22, 0($2)\n"
"addiu $7, $7, 28\n"
"3:\n"
"addiu $11, $11, 1\n"
"addiu $9, $9, 28\n"
"slt $2, $11, $15\n"
"bne $2, $0, 1b\n"
" addiu $10, $10, 28\n"
"2:\n"
"addu $2, $7, $0\n"
"addiu $sp, $sp, 8\n"
".set\treorder\n"
: : : "memory");
}
INCLUDE_ASM("asm/nonmatchings/800b2", GsTMDfastG3GNL);
void *GsTMDfastF3GL(void *prim, void *vert, void *norm, void *pkt,
int n, int shift, void *ot, void *work)
{
__asm__ __volatile__(
".include \"gte_macros.inc\"\n"
".set noat\n"
".set\tnoreorder\n"
"addiu $sp, $sp, -8\n"
"addu $t2, $a0, $zero\n"
"lw $t7, 24($sp)\n"
"lw $v0, 28($sp)\n"
"lw $v1, 32($sp)\n"
"lw $t0, 36($sp)\n"
"lw $v1, 4($v1)\n"
"addu $t3, $zero, $zero\n"
"sw $v0, 12($t0)\n"
"blez $t7, .L80057E70\n"
"sw $v1, 16($t0)\n"
"addiu $t6, $t0, 24\n"
"lui $t5, 255\n"
"ori $t5, $t5, 65535\n"
"addiu $t1, $a0, 16\n"
".L80057CB8:\n"
"lhu $a0, 2($t1)\n"
"lhu $v1, 4($t1)\n"
"lhu $v0, 6($t1)\n"
"sll $a0, $a0, 3\n"
"addu $a0, $a1, $a0\n"
"sll $v1, $v1, 3\n"
"addu $v1, $a1, $v1\n"
"sll $v0, $v0, 3\n"
"addu $v0, $a1, $v0\n"
"lwc2 $0, 0($a0)\n"
"lwc2 $1, 4($a0)\n"
"lwc2 $2, 0($v1)\n"
"lwc2 $3, 4($v1)\n"
"lwc2 $4, 0($v0)\n"
"lwc2 $5, 4($v0)\n"
"nop\n"
"nop\n"
"rtpt\n"
"lw $v0, -12($t1)\n"
"nop\n"
"sw $v0, 0($t0)\n"
"lbu $v0, -13($t1)\n"
"nop\n"
"ori $v0, $v0, 16\n"
"sb $v0, 3($t0)\n"
"addiu $v0, $t0, 36\n"
"cfc2 $t4, $31\n"
"nop\n"
"sw $t4, 0($v0)\n"
"lw $v0, 36($t0)\n"
"nop\n"
"bltz $v0, .L80057E5C\n"
"nop\n"
"nop\n"
"nop\n"
"nclip\n"
"swc2 $24, 0($t6)\n"
"lw $v0, 24($t0)\n"
"nop\n"
"blez $v0, .L80057E5C\n"
"nop\n"
"swc2 $12, 8($a3)\n"
"swc2 $13, 16($a3)\n"
"swc2 $14, 24($a3)\n"
"nop\n"
"nop\n"
"avsz3\n"
"swc2 $7, 0($t6)\n"
"lwc2 $6, 0($t0)\n"
"lhu $v0, 0($t1)\n"
"nop\n"
"sll $v0, $v0, 3\n"
"addu $v0, $a2, $v0\n"
"lwc2 $0, 0($v0)\n"
"lwc2 $1, 4($v0)\n"
"nop\n"
"nop\n"
"nccs\n"
"lw $v1, 24($t0)\n"
"lw $v0, 12($t0)\n"
"nop\n"
"srav $v1, $v1, $v0\n"
"lw $v0, 16($t0)\n"
"sll $v1, $v1, 2\n"
"addu $v0, $v0, $v1\n"
"sw $v0, 52($t0)\n"
"addiu $v0, $a3, 4\n"
"swc2 $22, 0($v0)\n"
"addiu $v0, $t2, 8\n"
"lwc2 $6, 0($v0)\n"
"lhu $v0, 0($t1)\n"
"nop\n"
"sll $v0, $v0, 3\n"
"addu $v0, $a2, $v0\n"
"lwc2 $0, 0($v0)\n"
"lwc2 $1, 4($v0)\n"
"nop\n"
"nop\n"
"nccs\n"
"lw $v0, 52($t0)\n"
"nop\n"
"lw $v0, 0($v0)\n"
"lui $v1, 1536\n"
"and $v0, $v0, $t5\n"
"or $v0, $v0, $v1\n"
"sw $v0, 0($a3)\n"
"addiu $v0, $a3, 12\n"
"swc2 $22, 0($v0)\n"
"addiu $v0, $t2, 12\n"
"lwc2 $6, 0($v0)\n"
"lhu $v0, 0($t1)\n"
"nop\n"
"sll $v0, $v0, 3\n"
"addu $v0, $a2, $v0\n"
"lwc2 $0, 0($v0)\n"
"lwc2 $1, 4($v0)\n"
"nop\n"
"nop\n"
"nccs\n"
"lw $v1, 52($t0)\n"
"and $v0, $a3, $t5\n"
"sw $v0, 0($v1)\n"
"addiu $v0, $a3, 20\n"
"swc2 $22, 0($v0)\n"
"addiu $a3, $a3, 28\n"
".L80057E5C:\n"
"addiu $t3, $t3, 1\n"
"addiu $t1, $t1, 24\n"
"slt $v0, $t3, $t7\n"
"bnez $v0, .L80057CB8\n"
"addiu $t2, $t2, 24\n"
".L80057E70:\n"
"addu $v0, $a3, $zero\n"
"addiu $sp, $sp, 8\n"
".set\treorder\n"
);
}
INCLUDE_ASM("asm/nonmatchings/800b2", GsTMDfastF3GNL);
/* GsTMDfastF4GL - handwritten PsyQ libgs assembly (GTE cop2 ops, 12 args,
* explicit $t0-$t9 usage, hand-scheduled delay slots). Not expressible as
* compiler-generated C; encoded verbatim. gcc-2.7.2 -O2 emits no prologue for
* this frameless body and supplies the trailing "jr $ra / nop" epilogue. */
void *GsTMDfastF4GL(void)
{
__asm__ __volatile__(
".word 0x27BDFFF8\n"
".word 0x00805021\n"
".word 0x8FB90018\n"
".word 0x8FA2001C\n"
".word 0x8FA30020\n"
".word 0x8FA80024\n"
".word 0x8C630004\n"
".word 0x00005821\n"
".word 0xAD02000C\n"
".word 0x1B200094\n"
".word 0xAD030010\n"
".word 0x250F0028\n"
".word 0x3C188000\n"
".word 0x250E0018\n"
".word 0x3C0D00FF\n"
".word 0x35ADFFFF\n"
".word 0x24890014\n"
".word 0x95240002\n"
".word 0x95230004\n"
".word 0x95220006\n"
".word 0x000420C0\n"
".word 0x00A42021\n"
".word 0x000318C0\n"
".word 0x00A31821\n"
".word 0x000210C0\n"
".word 0x00A21021\n"
".word 0xC8800000\n"
".word 0xC8810004\n"
".word 0xC8620000\n"
".word 0xC8630004\n"
".word 0xC8440000\n"
".word 0xC8450004\n"
".word 0x00000000\n"
".word 0x00000000\n"
".word 0x4A280030\n"
".word 0x8D22FFF0\n"
".word 0x00000000\n"
".word 0xAD020000\n"
".word 0x9122FFEF\n"
".word 0x00000000\n"
".word 0x34420010\n"
".word 0xA1020003\n"
".word 0x484CF800\n"
".word 0x00000000\n"
".word 0xADEC0000\n"
".word 0x8D020028\n"
".word 0x00000000\n"
".word 0x00581024\n"
".word 0x14400068\n"
".word 0x00000000\n"
".word 0x00000000\n"
".word 0x00000000\n"
".word 0x4B400006\n"
".word 0xE9D80000\n"
".word 0x8D020018\n"
".word 0x00000000\n"
".word 0x18400060\n"
".word 0x00000000\n"
".word 0xE8EC0008\n"
".word 0xE8ED0010\n"
".word 0xE8EE0018\n"
".word 0x95220008\n"
".word 0x00000000\n"
".word 0x000210C0\n"
".word 0x00A21021\n"
".word 0xC8400000\n"
".word 0xC8410004\n"
".word 0x00000000\n"
".word 0x00000000\n"
".word 0x4A180001\n"
".word 0x484CF800\n"
".word 0x00000000\n"
".word 0xADEC0000\n"
".word 0x8D020028\n"
".word 0x00000000\n"
".word 0x00581024\n"
".word 0x1440004C\n"
".word 0x24E20020\n"
".word 0xE84E0000\n"
".word 0x00000000\n"
".word 0x00000000\n"
".word 0x4B68002E\n"
".word 0xE9C70000\n"
".word 0xC9060000\n"
".word 0x95220000\n"
".word 0x00000000\n"
".word 0x000210C0\n"
".word 0x00C21021\n"
".word 0xC8400000\n"
".word 0xC8410004\n"
".word 0x00000000\n"
".word 0x00000000\n"
".word 0x4B08041B\n"
".word 0x8D030018\n"
".word 0x8D02000C\n"
".word 0x00000000\n"
".word 0x00431807\n"
".word 0x8D020010\n"
".word 0x00031880\n"
".word 0x00431021\n"
".word 0xAD020038\n"
".word 0x24E20004\n"
".word 0xE8560000\n"
".word 0x25420008\n"
".word 0xC8460000\n"
".word 0x95220000\n"
".word 0x00000000\n"
".word 0x000210C0\n"
".word 0x00C21021\n"
".word 0xC8400000\n"
".word 0xC8410004\n"
".word 0x00000000\n"
".word 0x00000000\n"
".word 0x4B08041B\n"
".word 0x8D020038\n"
".word 0x00000000\n"
".word 0x8C420000\n"
".word 0x3C030800\n"
".word 0x004D1024\n"
".word 0x00431025\n"
".word 0xACE20000\n"
".word 0x24E2000C\n"
".word 0xE8560000\n"
".word 0x2542000C\n"
".word 0xC8460000\n"
".word 0x95220000\n"
".word 0x00000000\n"
".word 0x000210C0\n"
".word 0x00C21021\n"
".word 0xC8400000\n"
".word 0xC8410004\n"
".word 0x00000000\n"
".word 0x00000000\n"
".word 0x4B08041B\n"
".word 0x8D030038\n"
".word 0x00ED1024\n"
".word 0xAC620000\n"
".word 0x24E20014\n"
".word 0xE8560000\n"
".word 0x25420010\n"
".word 0xC8460000\n"
".word 0x95220000\n"
".word 0x00000000\n"
".word 0x000210C0\n"
".word 0x00C21021\n"
".word 0xC8400000\n"
".word 0xC8410004\n"
".word 0x00000000\n"
".word 0x00000000\n"
".word 0x4B08041B\n"
".word 0x24E2001C\n"
".word 0xE8560000\n"
".word 0x24E70024\n"
".word 0x256B0001\n"
".word 0x25290020\n"
".word 0x0179102A\n"
".word 0x1440FF74\n"
".word 0x254A0020\n"
".word 0x00E01021\n"
".word 0x27BD0008\n"
);
}
INCLUDE_ASM("asm/nonmatchings/800b2", GsTMDfastF4GNL);
/*
* GsTMDfastG4GL @ 0x8005845C -- HANDWRITTEN PsyQ libgs TMD primitive walker
* (gouraud, 4-vertex, lit, non-textured "fast" path).
*
* Splat marks this "Handwritten function": it is raw GTE assembly, not
* compiler output --
* - long-lived state in $t0..$t9 (never gcc's allocation order),
* - extra arguments read from the CALLER's frame at 0x18..0x24($sp) after a
* manual `addiu $sp,$sp,-8`, i.e. a private calling convention,
* - cop2 data-register traffic (lwc2/swc2 $0..$27), rtpt/rtps/nclip/avsz4/
* nccs, `cfc2 $t4,$31` flag reads and the mandatory cop2-latency `nop`s.
* None of that is reachable from C, so the body is written as one full inline
* asm block (cookbook: the func_801285E4 / func_80128714 full-inline-asm
* handwritten-wrapper idiom, and the §"handwritten GTE sqr" note).
*
* maspsx interaction (this is the whole trick):
* - the leading `.set<TAB>noreorder` is written with a TAB so maspsx's own
* is_reorder tracker flips to False -- otherwise maspsx auto-fills EVERY
* branch delay slot with a nop, and this function has REAL instructions in
* its delay slots (`sw $v1,16($t1)`, `addiu $v0,$a3,32`, `addiu $t2,..`).
* - maspsx's load-delay pass (_handle_nop_before_next_instruction) is NOT
* gated on is_reorder: it still injects a nop when the next instruction
* re-reads a load's destination through a base register. That fires exactly
* once here, on `lw $v0,56($t1)` -> `lw $v0,0($v0)`, so that one latency nop
* is deliberately NOT written by hand (maspsx emits it).
* - every memory offset is DECIMAL: maspsx's store path calls int(operand)
* base-10 and would raise on "0x38".
*
* Shape of the routine:
* for (i = 0; i < n; i++) // n = arg at 0x18($sp), prim stride 0x24
* RTPT the three leading vertex indices, bail on GTE flag bit31,
* NCLIP + backface/otz reject, RTPS the 4th vertex, AVSZ4 -> OT bucket
* (ot_base + ((otz >> shift) << 2)), then four NCCS colour passes writing
* the packet, and link the packet into the OT.
*/
void GsTMDfastG4GL(void)
{
__asm__ __volatile__(
".set\tnoreorder\n"
"addiu $sp, $sp, -8\n"
"addu $t2, $a0, $zero\n"
"lw $t9, 24($sp)\n"
"lw $v0, 28($sp)\n"
"lw $v1, 32($sp)\n"
"lw $t1, 36($sp)\n"
"lw $v1, 4($v1)\n"
"addu $t3, $zero, $zero\n"
"sw $v0, 12($t1)\n"
"blez $t9, 2f\n"
"sw $v1, 16($t1)\n"
"addiu $t7, $t1, 40\n"
"lui $t8, 0x8000\n"
"addiu $t6, $t1, 24\n"
"lui $t5, 0xff\n"
"ori $t5, $t5, 0xffff\n"
"addiu $t0, $a0, 32\n"
"1:\n"
"lhu $a0, -10($t0)\n"
"lhu $v1, -6($t0)\n"
"lhu $v0, -2($t0)\n"
"sll $a0, $a0, 3\n"
"addu $a0, $a1, $a0\n"
"sll $v1, $v1, 3\n"
"addu $v1, $a1, $v1\n"
"sll $v0, $v0, 3\n"
"addu $v0, $a1, $v0\n"
"lwc2 $0, 0($a0)\n"
"lwc2 $1, 4($a0)\n"
"lwc2 $2, 0($v1)\n"
"lwc2 $3, 4($v1)\n"
"lwc2 $4, 0($v0)\n"
"lwc2 $5, 4($v0)\n"
"nop\n"
"nop\n"
"rtpt\n"
"cfc2 $t4, $31\n"
"nop\n"
"sw $t4, 0($t7)\n"
"lw $v0, 40($t1)\n"
"nop\n"
"and $v0, $v0, $t8\n"
"bne $v0, $zero, 3f\n"
"nop\n"
"nop\n"
"nop\n"
"nclip\n"
"swc2 $24, 0($t6)\n"
"lw $v0, 24($t1)\n"
"nop\n"
"blez $v0, 3f\n"
"nop\n"
"swc2 $12, 8($a3)\n"
"swc2 $13, 16($a3)\n"
"swc2 $14, 24($a3)\n"
"lhu $v0, 2($t0)\n"
"nop\n"
"sll $v0, $v0, 3\n"
"addu $v0, $a1, $v0\n"
"lwc2 $0, 0($v0)\n"
"lwc2 $1, 4($v0)\n"
"nop\n"
"nop\n"
"rtps\n"
"cfc2 $t4, $31\n"
"nop\n"
"sw $t4, 0($t7)\n"
"lw $v0, 40($t1)\n"
"nop\n"
"and $v0, $v0, $t8\n"
"bne $v0, $zero, 3f\n"
"addiu $v0, $a3, 32\n"
"swc2 $14, 0($v0)\n"
"nop\n"
"nop\n"
"avsz4\n"
"swc2 $7, 0($t6)\n"
"addiu $v0, $t2, 4\n"
"lwc2 $6, 0($v0)\n"
"lhu $v0, -12($t0)\n"
"nop\n"
"sll $v0, $v0, 3\n"
"addu $v0, $a2, $v0\n"
"lwc2 $0, 0($v0)\n"
"lwc2 $1, 4($v0)\n"
"nop\n"
"nop\n"
"nccs\n"
"lw $v1, 24($t1)\n"
"lw $v0, 12($t1)\n"
"nop\n"
"srav $v1, $v1, $v0\n"
"lw $v0, 16($t1)\n"
"sll $v1, $v1, 2\n"
"addu $v0, $v0, $v1\n"
"sw $v0, 56($t1)\n"
"addiu $v0, $a3, 4\n"
"swc2 $22, 0($v0)\n"
"addiu $v0, $t2, 8\n"
"lwc2 $6, 0($v0)\n"
"lhu $v0, -8($t0)\n"
"nop\n"
"sll $v0, $v0, 3\n"
"addu $v0, $a2, $v0\n"
"lwc2 $0, 0($v0)\n"
"lwc2 $1, 4($v0)\n"
"nop\n"
"nop\n"
"nccs\n"
"lw $v0, 56($t1)\n"
"lw $v0, 0($v0)\n"
"lui $v1, 0x800\n"
"and $v0, $v0, $t5\n"
"or $v0, $v0, $v1\n"
"sw $v0, 0($a3)\n"
"addiu $v0, $a3, 12\n"
"swc2 $22, 0($v0)\n"
"addiu $v0, $t2, 12\n"
"lwc2 $6, 0($v0)\n"
"lhu $v0, -4($t0)\n"
"nop\n"
"sll $v0, $v0, 3\n"
"addu $v0, $a2, $v0\n"
"lwc2 $0, 0($v0)\n"
"lwc2 $1, 4($v0)\n"
"nop\n"
"nop\n"
"nccs\n"
"lw $v1, 56($t1)\n"
"and $v0, $a3, $t5\n"
"sw $v0, 0($v1)\n"
"addiu $v0, $a3, 20\n"
"swc2 $22, 0($v0)\n"
"addiu $v0, $t2, 16\n"
"lwc2 $6, 0($v0)\n"
"lhu $v0, 0($t0)\n"
"nop\n"
"sll $v0, $v0, 3\n"
"addu $v0, $a2, $v0\n"
"lwc2 $0, 0($v0)\n"
"lwc2 $1, 4($v0)\n"
"nop\n"
"nop\n"
"nccs\n"
"addiu $v0, $a3, 28\n"
"swc2 $22, 0($v0)\n"
"addiu $a3, $a3, 36\n"
"3:\n"
"addiu $t3, $t3, 1\n"
"addiu $t0, $t0, 36\n"
"slt $v0, $t3, $t9\n"
"bne $v0, $zero, 1b\n"
"addiu $t2, $t2, 36\n"
"2:\n"
"addu $v0, $a3, $zero\n"
"addiu $sp, $sp, 8\n"
/* Hand the tracker back: maspsx SWALLOWS .set lines (they never reach
* gas, which is already noreorder from the .ent hook), so this only
* flips is_reorder back to True -- just in time for maspsx to fill the
* delay slot of gcc's own epilogue `j $31` with the final nop. */
".set\treorder\n"
: : : "memory");
}
INCLUDE_ASM("asm/nonmatchings/800b2", GsTMDfastG4GNL);