diff --git a/game/CMakeLists.txt b/game/CMakeLists.txt index 48010647ae..66d88db0b1 100644 --- a/game/CMakeLists.txt +++ b/game/CMakeLists.txt @@ -162,8 +162,6 @@ set(RUNTIME_SOURCE mips2c/jak1_functions/collide_mesh.cpp mips2c/jak1_functions/collide_probe.cpp mips2c/jak1_functions/draw_string.cpp - mips2c/jak1_functions/generic_effect.cpp - mips2c/jak1_functions/generic_effect2.cpp mips2c/jak1_functions/generic_merc.cpp mips2c/jak1_functions/generic_tie.cpp mips2c/jak1_functions/merc_blend_shape.cpp diff --git a/game/mips2c/jak1_functions/generic_effect.cpp b/game/mips2c/jak1_functions/generic_effect.cpp deleted file mode 100644 index b7496aeb56..0000000000 --- a/game/mips2c/jak1_functions/generic_effect.cpp +++ /dev/null @@ -1,1954 +0,0 @@ - -//--------------------------MIPS2C--------------------- - -#include "game/kernel/jak1/kscheme.h" -#include "game/mips2c/mips2c_private.h" -using namespace jak1; -// clang-format off - -namespace Mips2C::jak1 { -namespace generic_prepare_dma_double { -struct Cache { - void* fake_scratchpad_data; // *fake-scratchpad-data* -} cache; - -u64 execute(void* ctxt) { - auto* c = (ExecutionContext*)ctxt; - bool bc = false; - c->daddiu(sp, sp, -128); // daddiu sp, sp, -128 - c->sd(ra, 12432, at); // sd ra, 12432(at) - c->sq(s0, 12448, at); // sq s0, 12448(at) - c->sq(s1, 12464, at); // sq s1, 12464(at) - c->sq(s2, 12480, at); // sq s2, 12480(at) - c->sq(s3, 12496, at); // sq s3, 12496(at) - c->sq(s4, 12512, at); // sq s4, 12512(at) - c->sq(s5, 12528, at); // sq s5, 12528(at) - c->sq(gp, 12544, at); // sq gp, 12544(at) - get_fake_spad_addr(at, cache.fake_scratchpad_data, 0, c);// lui at, 28672 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->lw(a3, 60, at); // lw a3, 60(at) - // nop // sll r0, r0, 0 - c->lbu(v1, 16, a3); // lbu v1, 16(a3) - // nop // sll r0, r0, 0 - c->lh(a0, 18, a3); // lh a0, 18(a3) - // nop // sll r0, r0, 0 - c->lw(t7, 40, at); // lw t7, 40(at) - c->sll(a2, v1, 2); // sll a2, v1, 2 - c->daddiu(a1, a0, 3); // daddiu a1, a0, 3 - c->daddu(a2, a2, v1); // daddu a2, a2, v1 - c->addiu(t0, r0, -4); // addiu t0, r0, -4 - c->and_(t6, a1, t0); // and t6, a1, t0 - c->sll(a1, a2, 4); // sll a1, a2, 4 - c->daddu(a2, t6, t6); // daddu a2, t6, t6 - c->daddiu(a1, a1, 112); // daddiu a1, a1, 112 - c->daddu(a2, a2, t6); // daddu a2, a2, t6 - c->dsll(t0, t6, 2); // dsll t0, t6, 2 - // nop // sll r0, r0, 0 - c->daddiu(t0, t0, 15); // daddiu t0, t0, 15 - c->dsll(a2, a2, 2); // dsll a2, a2, 2 - c->dsra(t0, t0, 4); // dsra t0, t0, 4 - c->daddiu(a2, a2, 15); // daddiu a2, a2, 15 - c->dsll(t0, t0, 4); // dsll t0, t0, 4 - c->dsra(t9, a2, 4); // dsra t9, a2, 4 - c->dsll(t2, t9, 4); // dsll t2, t9, 4 - c->mov64(a2, t7); // or a2, t7, r0 - c->dsra(t8, a1, 4); // dsra t8, a1, 4 - c->daddu(t1, a2, a1); // daddu t1, a2, a1 - // nop // sll r0, r0, 0 - c->daddu(gp, t1, t2); // daddu gp, t1, t2 - // nop // sll r0, r0, 0 - c->daddu(t2, gp, t0); // daddu t2, gp, t0 - // nop // sll r0, r0, 0 - c->daddu(t3, t2, t0); // daddu t3, t2, t0 - // nop // sll r0, r0, 0 - c->daddu(ra, t3, a1); // daddu ra, t3, a1 - // nop // sll r0, r0, 0 - c->daddu(t4, ra, t0); // daddu t4, ra, t0 - // nop // sll r0, r0, 0 - c->daddu(t5, t4, t0); // daddu t5, t4, t0 - c->daddiu(t0, t1, 32); // daddiu t0, t1, 32 - c->daddiu(t1, gp, 64); // daddiu t1, gp, 64 - c->daddiu(t2, t2, 80); // daddiu t2, t2, 80 - c->daddiu(a1, t3, 80); // daddiu a1, t3, 80 - c->daddiu(t3, ra, 112); // daddiu t3, ra, 112 - c->daddiu(t4, t4, 128); // daddiu t4, t4, 128 - c->daddiu(t5, t5, 128); // daddiu t5, t5, 128 - c->sq(r0, -16, t0); // sq r0, -16(t0) - c->sq(r0, -32, t1); // sq r0, -32(t1) - c->sq(r0, -16, t1); // sq r0, -16(t1) - c->sq(r0, -32, t2); // sq r0, -32(t2) - c->sq(r0, -16, t2); // sq r0, -16(t2) - c->sq(r0, -16, a1); // sq r0, -16(a1) - c->sq(r0, -16, t3); // sq r0, -16(t3) - c->sq(r0, -32, t4); // sq r0, -32(t4) - c->sq(r0, -16, t4); // sq r0, -16(t4) - c->sq(r0, -16, t5); // sq r0, -16(t5) - c->lq(ra, 11744, at); // lq ra, 11744(at) - c->lq(gp, 11760, at); // lq gp, 11760(at) - c->lq(s5, 11776, at); // lq s5, 11776(at) - c->lq(s4, 11792, at); // lq s4, 11792(at) - c->lw(s3, 11984, at); // lw s3, 11984(at) - c->sq(ra, 0, a2); // sq ra, 0(a2) - c->sq(gp, 0, a1); // sq gp, 0(a1) - c->sq(s5, 0, t5); // sq s5, 0(t5) - c->sh(t9, 0, t5); // sh t9, 0(t5) - c->sq(s4, 16, t5); // sq s4, 16(t5) - c->sw(s3, 24, t5); // sw s3, 24(t5) - c->subu(t9, t5, t7); // subu t9, t5, t7 - c->sra(t9, t9, 4); // sra t9, t9, 4 - c->daddiu(t9, t9, -1); // daddiu t9, t9, -1 - c->sh(t9, 0, t7); // sh t9, 0(t7) - c->daddiu(t7, t9, 3); // daddiu t7, t9, 3 - c->sw(t7, 56, at); // sw t7, 56(at) - c->lw(t7, 76, at); // lw t7, 76(at) - c->dsubu(t9, t0, a2); // dsubu t9, t0, a2 - c->daddu(t7, t7, t9); // daddu t7, t7, t9 - // nop // sll r0, r0, 0 - c->lw(t9, 12, a2); // lw t9, 12(a2) - c->sll(ra, t8, 16); // sll ra, t8, 16 - c->lw(t8, 84, at); // lw t8, 84(at) - // nop // sll r0, r0, 0 - c->lw(gp, 12, a1); // lw gp, 12(a1) - c->or_(s4, t9, t8); // or s4, t9, t8 - c->lw(t9, 88, at); // lw t9, 88(at) - c->xori(s5, t8, 38); // xori s5, t8, 38 - // nop // sll r0, r0, 0 - c->or_(s4, s4, ra); // or s4, s4, ra - c->lw(t8, 11968, at); // lw t8, 11968(at) - // nop // sll r0, r0, 0 - c->lw(s3, 11988, at); // lw s3, 11988(at) - c->or_(gp, gp, s5); // or gp, gp, s5 - c->sw(s4, 12, a2); // sw s4, 12(a2) - c->or_(s5, gp, ra); // or s5, gp, ra - c->lw(ra, 11972, at); // lw ra, 11972(at) - c->daddiu(gp, t9, 1); // daddiu gp, t9, 1 - c->sw(s5, 12, a1); // sw s5, 12(a1) - c->daddiu(s4, t9, 2); // daddiu s4, t9, 2 - c->lw(s3, 11976, at); // lw s3, 11976(at) - c->dsll(t6, t6, 16); // dsll t6, t6, 16 - c->lw(s5, 11980, at); // lw s5, 11980(at) - c->or_(s4, ra, s4); // or s4, ra, s4 - c->lw(ra, 11984, at); // lw ra, 11984(at) - c->or_(s3, s3, gp); // or s3, s3, gp - c->lw(gp, 11992, at); // lw gp, 11992(at) - c->mov64(s2, t9); // or s2, t9, r0 - c->sw(t8, -8, t0); // sw t8, -8(t0) - c->or_(s5, s5, s2); // or s5, s5, s2 - c->sw(gp, 4, a1); // sw gp, 4(a1) - c->or_(s4, s4, t6); // or s4, s4, t6 - c->sw(ra, 0, a1); // sw ra, 0(a1) - c->or_(s3, s3, t6); // or s3, s3, t6 - c->sw(s4, -4, t0); // sw s4, -4(t0) - c->or_(s5, s5, t6); // or s5, s5, t6 - c->sw(s3, -4, t1); // sw s3, -4(t1) - c->addiu(s4, r0, 567); // addiu s4, r0, 567 - c->sw(s5, -4, t2); // sw s5, -4(t2) - bc = c->sgpr64(t9) != c->sgpr64(s4); // bne t9, s4, L59 - c->daddiu(t9, t9, 279); // daddiu t9, t9, 279 - if (bc) {goto block_2;} // branch non-likely - - // nop // sll r0, r0, 0 - c->addiu(t9, r0, 9); // addiu t9, r0, 9 - - block_2: - c->daddiu(s1, t9, 1); // daddiu s1, t9, 1 - c->lw(s0, 11976, at); // lw s0, 11976(at) - c->mov64(s3, t9); // or s3, t9, r0 - c->lw(s2, 11980, at); // lw s2, 11980(at) - c->daddiu(s5, t9, 2); // daddiu s5, t9, 2 - c->lw(s4, 11972, at); // lw s4, 11972(at) - c->or_(s1, s0, s1); // or s1, s0, s1 - c->sw(t8, -8, t3); // sw t8, -8(t3) - c->or_(t8, s2, s3); // or t8, s2, s3 - c->sw(gp, 24, t5); // sw gp, 24(t5) - c->or_(gp, s4, s5); // or gp, s4, s5 - // nop // sll r0, r0, 0 - c->or_(s5, s1, t6); // or s5, s1, t6 - c->sw(ra, 28, t5); // sw ra, 28(t5) - c->or_(t8, t8, t6); // or t8, t8, t6 - c->sw(s5, -4, t3); // sw s5, -4(t3) - c->or_(ra, gp, t6); // or ra, gp, t6 - c->sw(t8, -4, t4); // sw t8, -4(t4) - c->addiu(t6, r0, 567); // addiu t6, r0, 567 - c->sw(t7, 4, t5); // sw t7, 4(t5) - // nop // sll r0, r0, 0 - c->sw(ra, 12, t5); // sw ra, 12(t5) - bc = c->sgpr64(t9) != c->sgpr64(t6); // bne t9, t6, L60 - c->daddiu(t5, t9, 279); // daddiu t5, t9, 279 - if (bc) {goto block_4;} // branch non-likely - - // nop // sll r0, r0, 0 - c->addiu(t5, r0, 9); // addiu t5, r0, 9 - - block_4: - // nop // sll r0, r0, 0 - c->sw(t5, 88, at); // sw t5, 88(at) - // nop // sll r0, r0, 0 - c->sw(t0, 20, at); // sw t0, 20(at) - // nop // sll r0, r0, 0 - c->sw(t1, 24, at); // sw t1, 24(at) - // nop // sll r0, r0, 0 - c->sw(t2, 28, at); // sw t2, 28(at) - // nop // sll r0, r0, 0 - c->sw(t3, 32, at); // sw t3, 32(at) - // nop // sll r0, r0, 0 - c->sw(t4, 36, at); // sw t4, 36(at) - // nop // sll r0, r0, 0 - c->lw(t0, 64, at); // lw t0, 64(at) - // nop // sll r0, r0, 0 - c->lq(t1, 11808, at); // lq t1, 11808(at) - bc = c->sgpr64(t0) == 0; // beq t0, r0, L63 - c->lq(t2, 11824, at); // lq t2, 11824(at) - if (bc) {goto block_10;} // branch non-likely - - // nop // sll r0, r0, 0 - c->lq(t3, 11840, at); // lq t3, 11840(at) - // nop // sll r0, r0, 0 - c->lq(t4, 11856, at); // lq t4, 11856(at) - // nop // sll r0, r0, 0 - c->sq(t1, 16, a2); // sq t1, 16(a2) - // nop // sll r0, r0, 0 - c->sq(t2, 32, a2); // sq t2, 32(a2) - // nop // sll r0, r0, 0 - c->sq(t3, 48, a2); // sq t3, 48(a2) - // nop // sll r0, r0, 0 - c->sq(t4, 64, a2); // sq t4, 64(a2) - // nop // sll r0, r0, 0 - c->sq(t1, 16, a1); // sq t1, 16(a1) - // nop // sll r0, r0, 0 - c->sq(t2, 32, a1); // sq t2, 32(a1) - // nop // sll r0, r0, 0 - c->sq(t3, 48, a1); // sq t3, 48(a1) - // nop // sll r0, r0, 0 - c->sq(t4, 64, a1); // sq t4, 64(a1) - // nop // sll r0, r0, 0 - c->lq(t1, 11872, at); // lq t1, 11872(at) - // nop // sll r0, r0, 0 - c->lq(t2, 11888, at); // lq t2, 11888(at) - // nop // sll r0, r0, 0 - c->lq(t3, 11936, at); // lq t3, 11936(at) - // nop // sll r0, r0, 0 - c->lq(t4, 12032, at); // lq t4, 12032(at) - // nop // sll r0, r0, 0 - c->sq(t1, 80, a2); // sq t1, 80(a2) - // nop // sll r0, r0, 0 - c->sq(t2, 96, a2); // sq t2, 96(a2) - // nop // sll r0, r0, 0 - c->sq(t3, 112, a2); // sq t3, 112(a2) - // nop // sll r0, r0, 0 - c->sq(t4, 80, a1); // sq t4, 80(a1) - // nop // sll r0, r0, 0 - c->sq(t3, 96, a1); // sq t3, 96(a1) - // nop // sll r0, r0, 0 - c->sq(t3, 112, a1); // sq t3, 112(a1) - c->daddiu(t2, a3, 22); // daddiu t2, a3, 22 - c->addiu(t3, r0, 0); // addiu t3, r0, 0 - c->mov64(t4, v1); // or t4, v1, r0 - c->addiu(t6, r0, 128); // addiu t6, r0, 128 - c->mov64(t1, a2); // or t1, a2, r0 - // nop // sll r0, r0, 0 - - block_6: - c->daddu(t1, t1, t6); // daddu t1, t1, t6 - c->lq(t6, 0, t0); // lq t6, 0(t0) - c->daddiu(t4, t4, -1); // daddiu t4, t4, -1 - c->lbu(t5, 0, t2); // lbu t5, 0(t2) - // nop // sll r0, r0, 0 - c->lq(t7, 16, t0); // lq t7, 16(t0) - c->daddu(t9, t5, t5); // daddu t9, t5, t5 - c->lq(t8, 32, t0); // lq t8, 32(t0) - c->daddu(ra, t9, t5); // daddu ra, t9, t5 - c->lq(t9, 48, t0); // lq t9, 48(t0) - c->daddiu(gp, ra, 9); // daddiu gp, ra, 9 - c->lq(ra, 64, t0); // lq ra, 64(t0) - c->daddiu(t0, t0, 80); // daddiu t0, t0, 80 - c->sq(t6, 0, t1); // sq t6, 0(t1) - // nop // sll r0, r0, 0 - c->sw(t3, 12, t1); // sw t3, 12(t1) - c->daddiu(t2, t2, 1); // daddiu t2, t2, 1 - c->sq(t7, 16, t1); // sq t7, 16(t1) - c->daddu(t3, t3, gp); // daddu t3, t3, gp - c->sw(t5, 28, t1); // sw t5, 28(t1) - // nop // sll r0, r0, 0 - c->sq(t8, 32, t1); // sq t8, 32(t1) - c->addiu(t6, r0, 80); // addiu t6, r0, 80 - c->sq(t9, 48, t1); // sq t9, 48(t1) - bc = ((s64)c->sgpr64(t4)) > 0; // bgtz t4, L61 - c->sq(ra, 64, t1); // sq ra, 64(t1) - if (bc) {goto block_6;} // branch non-likely - - c->ori(t0, t5, 32768); // ori t0, t5, 32768 - c->sw(a0, 52, at); // sw a0, 52(at) - // nop // sll r0, r0, 0 - c->sw(t0, 28, t1); // sw t0, 28(t1) - // nop // sll r0, r0, 0 - c->sw(v1, 92, a2); // sw v1, 92(a2) - // nop // sll r0, r0, 0 - c->sw(a0, 108, a2); // sw a0, 108(a2) - // nop // sll r0, r0, 0 - c->sw(r0, 124, a2); // sw r0, 124(a2) - c->daddiu(a3, a3, 22); // daddiu a3, a3, 22 - c->addiu(t0, r0, 0); // addiu t0, r0, 0 - c->mov64(t1, v1); // or t1, v1, r0 - c->lw(t6, 68, at); // lw t6, 68(at) - c->mov64(a2, a1); // or a2, a1, r0 - c->addiu(t8, r0, 128); // addiu t8, r0, 128 - // nop // sll r0, r0, 0 - c->lq(t2, 0, t6); // lq t2, 0(t6) - // nop // sll r0, r0, 0 - c->lq(t3, 16, t6); // lq t3, 16(t6) - // nop // sll r0, r0, 0 - c->lq(t4, 32, t6); // lq t4, 32(t6) - // nop // sll r0, r0, 0 - c->lq(t5, 48, t6); // lq t5, 48(t6) - // nop // sll r0, r0, 0 - c->lq(t6, 64, t6); // lq t6, 64(t6) - - block_8: - c->daddu(a2, a2, t8); // daddu a2, a2, t8 - c->lbu(t7, 0, a3); // lbu t7, 0(a3) - c->daddiu(t1, t1, -1); // daddiu t1, t1, -1 - c->sq(t2, 0, a2); // sq t2, 0(a2) - c->daddu(t8, t7, t7); // daddu t8, t7, t7 - c->sw(t0, 12, a2); // sw t0, 12(a2) - c->daddu(t8, t8, t7); // daddu t8, t8, t7 - c->sq(t3, 16, a2); // sq t3, 16(a2) - c->daddiu(t8, t8, 9); // daddiu t8, t8, 9 - c->sw(t7, 28, a2); // sw t7, 28(a2) - c->daddiu(a3, a3, 1); // daddiu a3, a3, 1 - // nop // sll r0, r0, 0 - c->daddu(t0, t0, t8); // daddu t0, t0, t8 - c->sq(t4, 32, a2); // sq t4, 32(a2) - c->addiu(t8, r0, 80); // addiu t8, r0, 80 - c->sq(t5, 48, a2); // sq t5, 48(a2) - bc = ((s64)c->sgpr64(t1)) > 0; // bgtz t1, L62 - c->sq(t6, 64, a2); // sq t6, 64(a2) - if (bc) {goto block_8;} // branch non-likely - - c->ori(a3, t7, 32768); // ori a3, t7, 32768 - c->sw(a0, 52, at); // sw a0, 52(at) - // nop // sll r0, r0, 0 - c->sw(a3, 28, a2); // sw a3, 28(a2) - // nop // sll r0, r0, 0 - c->sw(v1, 92, a1); // sw v1, 92(a1) - // nop // sll r0, r0, 0 - c->sw(a0, 108, a1); // sw a0, 108(a1) - //beq r0, r0, L65 // beq r0, r0, L65 - c->sw(r0, 124, a1); // sw r0, 124(a1) - goto block_13; // branch always - - - block_10: - // nop // sll r0, r0, 0 - c->lq(a3, 16, a2); // lq a3, 16(a2) - // nop // sll r0, r0, 0 - c->lq(t0, 32, a2); // lq t0, 32(a2) - // nop // sll r0, r0, 0 - c->lq(t1, 48, a2); // lq t1, 48(a2) - // nop // sll r0, r0, 0 - c->lq(t2, 64, a2); // lq t2, 64(a2) - // nop // sll r0, r0, 0 - c->sq(a3, 16, a1); // sq a3, 16(a1) - // nop // sll r0, r0, 0 - c->sq(t0, 32, a1); // sq t0, 32(a1) - // nop // sll r0, r0, 0 - c->sq(t1, 48, a1); // sq t1, 48(a1) - // nop // sll r0, r0, 0 - c->sq(t2, 64, a1); // sq t2, 64(a1) - // nop // sll r0, r0, 0 - c->lq(a3, 12032, at); // lq a3, 12032(at) - // nop // sll r0, r0, 0 - c->lq(t0, 11936, at); // lq t0, 11936(at) - // nop // sll r0, r0, 0 - c->sq(a3, 80, a1); // sq a3, 80(a1) - // nop // sll r0, r0, 0 - c->sq(t0, 96, a1); // sq t0, 96(a1) - // nop // sll r0, r0, 0 - c->sq(t0, 112, a1); // sq t0, 112(a1) - c->mov64(a3, v1); // or a3, v1, r0 - c->lw(t5, 68, at); // lw t5, 68(at) - c->mov64(t0, a1); // or t0, a1, r0 - c->addiu(t7, r0, 128); // addiu t7, r0, 128 - // nop // sll r0, r0, 0 - c->lq(t1, 0, t5); // lq t1, 0(t5) - // nop // sll r0, r0, 0 - c->lq(t2, 16, t5); // lq t2, 16(t5) - // nop // sll r0, r0, 0 - c->lq(t3, 32, t5); // lq t3, 32(t5) - // nop // sll r0, r0, 0 - c->lq(t4, 48, t5); // lq t4, 48(t5) - // nop // sll r0, r0, 0 - c->lq(t5, 64, t5); // lq t5, 64(t5) - c->daddu(a2, a2, t7); // daddu a2, a2, t7 - // nop // sll r0, r0, 0 - - block_11: - c->daddu(t0, t0, t7); // daddu t0, t0, t7 - c->lwu(t8, 12, a2); // lwu t8, 12(a2) - c->daddiu(a3, a3, -1); // daddiu a3, a3, -1 - c->lwu(t7, 28, a2); // lwu t7, 28(a2) - c->daddiu(a2, a2, 80); // daddiu a2, a2, 80 - c->sq(t1, 0, t0); // sq t1, 0(t0) - // nop // sll r0, r0, 0 - c->sw(t8, 12, t0); // sw t8, 12(t0) - // nop // sll r0, r0, 0 - c->sq(t2, 16, t0); // sq t2, 16(t0) - c->daddu(t8, t8, t6); // daddu t8, t8, t6 - c->sw(t7, 28, t0); // sw t7, 28(t0) - // nop // sll r0, r0, 0 - c->sq(t3, 32, t0); // sq t3, 32(t0) - c->addiu(t7, r0, 80); // addiu t7, r0, 80 - c->sq(t4, 48, t0); // sq t4, 48(t0) - bc = ((s64)c->sgpr64(a3)) > 0; // bgtz a3, L64 - c->sq(t5, 64, t0); // sq t5, 64(t0) - if (bc) {goto block_11;} // branch non-likely - - // nop // sll r0, r0, 0 - c->sw(a0, 52, at); // sw a0, 52(at) - // nop // sll r0, r0, 0 - c->sw(v1, 92, a1); // sw v1, 92(a1) - // nop // sll r0, r0, 0 - c->sw(a0, 108, a1); // sw a0, 108(a1) - // nop // sll r0, r0, 0 - c->sw(r0, 124, a1); // sw r0, 124(a1) - - block_13: - c->gprs[v0].du64[0] = 0; // or v0, r0, r0 - c->ld(ra, 12432, at); // ld ra, 12432(at) - c->lq(gp, 12544, at); // lq gp, 12544(at) - c->lq(s5, 12528, at); // lq s5, 12528(at) - c->lq(s4, 12512, at); // lq s4, 12512(at) - c->lq(s3, 12496, at); // lq s3, 12496(at) - c->lq(s2, 12480, at); // lq s2, 12480(at) - c->lq(s1, 12464, at); // lq s1, 12464(at) - c->lq(s0, 12448, at); // lq s0, 12448(at) - //jr ra // jr ra - c->daddiu(sp, sp, 128); // daddiu sp, sp, 128 - goto end_of_function; // return - - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - end_of_function: - return c->gprs[v0].du64[0]; -} - -void link() { - cache.fake_scratchpad_data = intern_from_c("*fake-scratchpad-data*").c(); - gLinkedFunctionTable.reg("generic-prepare-dma-double", execute, 256); -} - -} // namespace generic_prepare_dma_double -} // namespace Mips2C - -//--------------------------MIPS2C--------------------- -#include "game/mips2c/mips2c_private.h" - -namespace Mips2C::jak1 { -namespace generic_light_proc { -struct Cache { - void* fake_scratchpad_data; // *fake-scratchpad-data* -} cache; - - -void vcallms0(ExecutionContext* c) { - // this function does lighting calculations for 4 vertices. - // the input colors (u8's) are in vf5, vf6, vf7, vf8 - -// fmt::print("nromal:\n {}\n {}\n {}\n {}\n\n", c->vf_src(vf1).vf.print(), c->vf_src(vf2).vf.print(), c->vf_src(vf3).vf.print(), c->vf_src(vf4).vf.print()); -// //fmt::print("dir:\n {}\n {}\n {}\n\n", c->vf_src(vf10).vf.print(), c->vf_src(vf11).vf.print(), c->vf_src(vf12).vf.print()); -// -// // hack to see normals as colors -// c->vfs[vf21].vf.move(Mask::xyzw, c->vf_src(vf17).vf); -// c->vfs[vf22].vf.move(Mask::xyzw, c->vf_src(vf18).vf); -// c->vfs[vf23].vf.move(Mask::xyzw, c->vf_src(vf19).vf); -// c->vfs[vf24].vf.move(Mask::xyzw, c->vf_src(vf20).vf); -// for (int vec = 0; vec < 4; vec++) { -// for (int cmp = 0; cmp < 3; cmp++) { -//// c->vfs[vf17 + vec].f[cmp] = vec * 16 + cmp * 4; -// u32 val = 128 + 127 * c->vfs[vf1 + vec].f[cmp]; -// fmt::print("{} ", val); -// memcpy(&c->vfs[vf17 + vec].f[cmp], &val, 4); -// } -// } -// fmt::print("\n"); -// return; - - // move.xyzw vf21, vf17 | mulax.xyzw ACC, vf10, vf01 - c->acc.vf.mula(Mask::xyzw, c->vf_src(vf10).vf, c->vf_src(vf01).vf.x()); c->vfs[vf21].vf.move(Mask::xyzw, c->vf_src(vf17).vf); - // move.xyzw vf22, vf18 | madday.xyzw ACC, vf11, vf01 - c->acc.vf.madda(Mask::xyzw, c->vfs[vf11].vf, c->vfs[vf01].vf.y()); c->vfs[vf22].vf.move(Mask::xyzw, c->vf_src(vf18).vf); - // move.xyzw vf23, vf19 | maddz.xyzw vf01, vf12, vf01 - c->acc.vf.madd(Mask::xyzw, c->vfs[vf01].vf, c->vf_src(vf12).vf, c->vf_src(vf01).vf.z()); c->vfs[vf23].vf.move(Mask::xyzw, c->vf_src(vf19).vf); - // move.xyzw vf24, vf20 | mulax.xyzw ACC, vf10, vf02 - c->acc.vf.mula(Mask::xyzw, c->vf_src(vf10).vf, c->vf_src(vf02).vf.x()); c->vfs[vf24].vf.move(Mask::xyzw, c->vf_src(vf20).vf); - // nop | itof0.xyzw vf17, vf05 - c->vfs[vf17].vf.itof0(Mask::xyzw, c->vf_src(vf05).vf); - // nop | itof0.xyzw vf18, vf06 - c->vfs[vf18].vf.itof0(Mask::xyzw, c->vf_src(vf06).vf); - // nop | itof0.xyzw vf19, vf07 - c->vfs[vf19].vf.itof0(Mask::xyzw, c->vf_src(vf07).vf); - // nop | itof0.xyzw vf20, vf08 - c->vfs[vf20].vf.itof0(Mask::xyzw, c->vf_src(vf08).vf); - - //fmt::print("light in:\n {}\n {}\n {}\n {}\n\n", c->vf_src(vf17).vf.print(), c->vf_src(vf18).vf.print(), c->vf_src(vf19).vf.print(), c->vf_src(vf20).vf.print()); - - // nop | madday.xyzw ACC, vf11, vf02 - c->acc.vf.madda(Mask::xyzw, c->vfs[vf11].vf, c->vfs[vf02].vf.y()); - // nop | maddz.xyzw vf02, vf12, vf02 - c->acc.vf.madd(Mask::xyzw, c->vfs[vf02].vf, c->vf_src(vf12).vf, c->vf_src(vf02).vf.z()); - // nop | mulax.xyzw ACC, vf10, vf03 - c->acc.vf.mula(Mask::xyzw, c->vf_src(vf10).vf, c->vf_src(vf03).vf.x()); - // nop | madday.xyzw ACC, vf11, vf03 - c->acc.vf.madda(Mask::xyzw, c->vfs[vf11].vf, c->vfs[vf03].vf.y()); - // nop | maddz.xyzw vf03, vf12, vf03 - c->acc.vf.madd(Mask::xyzw, c->vfs[vf03].vf, c->vf_src(vf12).vf, c->vf_src(vf03).vf.z()); - // nop | mulax.xyzw ACC, vf10, vf04 - c->acc.vf.mula(Mask::xyzw, c->vf_src(vf10).vf, c->vf_src(vf04).vf.x()); - // nop | madday.xyzw ACC, vf11, vf04 - c->acc.vf.madda(Mask::xyzw, c->vfs[vf11].vf, c->vfs[vf04].vf.y()); - // nop | maddz.xyzw vf04, vf12, vf04 - c->acc.vf.madd(Mask::xyzw, c->vfs[vf04].vf, c->vf_src(vf12).vf, c->vf_src(vf04).vf.z()); - // nop | maxx.xyzw vf01, vf01, vf00 - c->vfs[vf01].vf.max(Mask::xyzw, c->vf_src(vf01).vf, c->vf_src(vf00).vf.x()); - // nop | maxx.xyzw vf02, vf02, vf00 - c->vfs[vf02].vf.max(Mask::xyzw, c->vf_src(vf02).vf, c->vf_src(vf00).vf.x()); - // nop | maxx.xyzw vf03, vf03, vf00 - c->vfs[vf03].vf.max(Mask::xyzw, c->vf_src(vf03).vf, c->vf_src(vf00).vf.x()); - // nop | maxx.xyzw vf04, vf04, vf00 - c->vfs[vf04].vf.max(Mask::xyzw, c->vf_src(vf04).vf, c->vf_src(vf00).vf.x()); - // nop | mulaw.xyzw ACC, vf13, vf00 - c->acc.vf.mula(Mask::xyzw, c->vf_src(vf13).vf, c->vf_src(vf00).vf.w()); - // nop | maddax.xyzw ACC, vf14, vf01 - c->acc.vf.madda(Mask::xyzw, c->vfs[vf14].vf, c->vfs[vf01].vf.x()); - // nop | madday.xyzw ACC, vf15, vf01 - c->acc.vf.madda(Mask::xyzw, c->vfs[vf15].vf, c->vfs[vf01].vf.y()); - // nop | maddz.xyzw vf01, vf16, vf01 - c->acc.vf.madd(Mask::xyzw, c->vfs[vf01].vf, c->vf_src(vf16).vf, c->vf_src(vf01).vf.z()); - // nop | mulaw.xyzw ACC, vf13, vf00 - c->acc.vf.mula(Mask::xyzw, c->vf_src(vf13).vf, c->vf_src(vf00).vf.w()); - // nop | maddax.xyzw ACC, vf14, vf02 - c->acc.vf.madda(Mask::xyzw, c->vfs[vf14].vf, c->vfs[vf02].vf.x()); - // nop | madday.xyzw ACC, vf15, vf02 - c->acc.vf.madda(Mask::xyzw, c->vfs[vf15].vf, c->vfs[vf02].vf.y()); - // nop | maddz.xyzw vf02, vf16, vf02 - c->acc.vf.madd(Mask::xyzw, c->vfs[vf02].vf, c->vf_src(vf16).vf, c->vf_src(vf02).vf.z()); - // nop | mulaw.xyzw ACC, vf13, vf00 - c->acc.vf.mula(Mask::xyzw, c->vf_src(vf13).vf, c->vf_src(vf00).vf.w()); - // nop | maddax.xyzw ACC, vf14, vf03 - c->acc.vf.madda(Mask::xyzw, c->vfs[vf14].vf, c->vfs[vf03].vf.x()); - // nop | madday.xyzw ACC, vf15, vf03 - c->acc.vf.madda(Mask::xyzw, c->vfs[vf15].vf, c->vfs[vf03].vf.y()); - // nop | maddz.xyzw vf03, vf16, vf03 - c->acc.vf.madd(Mask::xyzw, c->vfs[vf03].vf, c->vf_src(vf16).vf, c->vf_src(vf03).vf.z()); - // nop | mulaw.xyzw ACC, vf13, vf00 - c->acc.vf.mula(Mask::xyzw, c->vf_src(vf13).vf, c->vf_src(vf00).vf.w()); - // nop | maddax.xyzw ACC, vf14, vf04 - c->acc.vf.madda(Mask::xyzw, c->vfs[vf14].vf, c->vfs[vf04].vf.x()); - // nop | madday.xyzw ACC, vf15, vf04 - c->acc.vf.madda(Mask::xyzw, c->vfs[vf15].vf, c->vfs[vf04].vf.y()); - // nop | maddz.xyzw vf04, vf16, vf04 - c->acc.vf.madd(Mask::xyzw, c->vfs[vf04].vf, c->vf_src(vf16).vf, c->vf_src(vf04).vf.z()); - // nop | mul.xyzw vf17, vf17, vf01 - c->vfs[vf17].vf.mul(Mask::xyzw, c->vf_src(vf17).vf, c->vf_src(vf01).vf); - // nop | mul.xyzw vf18, vf18, vf02 - c->vfs[vf18].vf.mul(Mask::xyzw, c->vf_src(vf18).vf, c->vf_src(vf02).vf); - // nop | mul.xyzw vf19, vf19, vf03 - c->vfs[vf19].vf.mul(Mask::xyzw, c->vf_src(vf19).vf, c->vf_src(vf03).vf); - // nop | mul.xyzw vf20, vf20, vf04 - c->vfs[vf20].vf.mul(Mask::xyzw, c->vf_src(vf20).vf, c->vf_src(vf04).vf); - // nop | minix.xyzw vf17, vf17, vf09 - c->vfs[vf17].vf.mini(Mask::xyzw, c->vf_src(vf17).vf, c->vf_src(vf09).vf.x()); - // nop | minix.xyzw vf18, vf18, vf09 - c->vfs[vf18].vf.mini(Mask::xyzw, c->vf_src(vf18).vf, c->vf_src(vf09).vf.x()); - // nop | minix.xyzw vf19, vf19, vf09 - c->vfs[vf19].vf.mini(Mask::xyzw, c->vf_src(vf19).vf, c->vf_src(vf09).vf.x()); - // nop | minix.xyzw vf20, vf20, vf09 - c->vfs[vf20].vf.mini(Mask::xyzw, c->vf_src(vf20).vf, c->vf_src(vf09).vf.x()); - //fmt::print("light:\n {}\n {}\n {}\n {}\n\n", c->vf_src(vf17).vf.print(), c->vf_src(vf18).vf.print(), c->vf_src(vf19).vf.print(), c->vf_src(vf20).vf.print()); - - - - // nop | ftoi0.xyzw vf17, vf17 - c->vfs[vf17].vf.ftoi0(Mask::xyzw, c->vf_src(vf17).vf); - // nop | ftoi0.xyzw vf18, vf18 - c->vfs[vf18].vf.ftoi0(Mask::xyzw, c->vf_src(vf18).vf); - // nop | ftoi0.xyzw vf19, vf19 :e - c->vfs[vf19].vf.ftoi0(Mask::xyzw, c->vf_src(vf19).vf); - // nop | ftoi0.xyzw vf20, vf20 - c->vfs[vf20].vf.ftoi0(Mask::xyzw, c->vf_src(vf20).vf); -} - -u64 execute(void* ctxt) { - auto* c = (ExecutionContext*)ctxt; - bool bc = false; - c->daddiu(sp, sp, -96); // daddiu sp, sp, -96 - c->sd(ra, 12432, at); // sd ra, 12432(at) - c->sq(s2, 12448, at); // sq s2, 12448(at) - c->sq(s3, 12464, at); // sq s3, 12464(at) - c->sq(s4, 12480, at); // sq s4, 12480(at) - c->sq(s5, 12496, at); // sq s5, 12496(at) - c->sq(gp, 12512, at); // sq gp, 12512(at) - get_fake_spad_addr(at, cache.fake_scratchpad_data, 0, c);// lui at, 28672 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->lw(v1, 60, at); // lw v1, 60(at) - // nop // sll r0, r0, 0 - c->lw(a1, 52, at); // lw a1, 52(at) - // nop // sll r0, r0, 0 - c->lw(a0, 4, v1); // lw a0, 4(v1) - // nop // sll r0, r0, 0 - c->lw(t0, 0, v1); // lw t0, 0(v1) - // nop // sll r0, r0, 0 - c->lw(t3, 20, at); // lw t3, 20(at) - // nop // sll r0, r0, 0 - c->lw(v1, 24, at); // lw v1, 24(at) - c->addiu(a3, r0, 255); // addiu a3, r0, 255 - c->lw(t2, 28, at); // lw t2, 28(at) - c->addiu(a2, r0, 256); // addiu a2, r0, 256 - c->lui(t1, -2); // lui t1, -2 - c->mov64(t4, a1); // or t4, a1, r0 - c->lqc2(vf10, 12688, at); // lqc2 vf10, 12688(at) - c->ori(a1, t1, 65534); // ori a1, t1, 65534 - c->lqc2(vf11, 12704, at); // lqc2 vf11, 12704(at) - c->pextlw(a1, a1, a1); // pextlw a1, a1, a1 - c->lqc2(vf12, 12720, at); // lqc2 vf12, 12720(at) - c->pextlw(t1, a0, a0); // pextlw t1, a0, a0 - c->lqc2(vf15, 12752, at); // lqc2 vf15, 12752(at) - c->pextlw(a0, a1, a1); // pextlw a0, a1, a1 - c->lqc2(vf14, 12736, at); // lqc2 vf14, 12736(at) - c->pextlw(a1, t1, t1); // pextlw a1, t1, t1 - c->lqc2(vf16, 12768, at); // lqc2 vf16, 12768(at) - c->pcpyh(a3, a3); // pcpyh a3, a3 - c->lqc2(vf13, 12784, at); // lqc2 vf13, 12784(at) - c->pcpyh(t1, a2); // pcpyh t1, a2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pcpyld(a2, a3, a3); // pcpyld a2, a3, a3 - c->lqc2(vf9, 12144, at); // lqc2 vf9, 12144(at) - c->pcpyld(a3, t1, t1); // pcpyld a3, t1, t1 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->ldr(t1, 0, t0); // ldr t1, 0(t0) - // nop // sll r0, r0, 0 - c->ldl(t1, 7, t0); // ldl t1, 7(t0) - // nop // sll r0, r0, 0 - c->daddiu(t0, t0, 8); // daddiu t0, t0, 8 - c->pextlh(t1, r0, t1); // pextlh t1, r0, t1 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pand(t5, t1, a2); // pand t5, t1, a2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->psllw(t5, t5, 5); // psllw t5, t5, 5 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->paddw(s5, t5, a1); // paddw s5, t5, a1 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->dsrl32(t9, s5, 0); // dsrl32 t9, s5, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pcpyud(s4, s5, r0); // pcpyud s4, s5, r0 - c->lq(t6, 0, s5); // lq t6, 0(s5) - c->dsrl32(ra, s4, 0); // dsrl32 ra, s4, 0 - c->lq(t7, 0, t9); // lq t7, 0(t9) - c->pand(t8, t1, a3); // pand t8, t1, a3 - c->lq(t5, 0, s4); // lq t5, 0(s4) - c->psraw(gp, t8, 8); // psraw gp, t8, 8 - c->lq(t8, 0, ra); // lq t8, 0(ra) - c->pextuw(s3, t7, t6); // pextuw s3, t7, t6 - c->lq(s5, 16, s5); // lq s5, 16(s5) - c->pextuw(s2, t8, t5); // pextuw s2, t8, t5 - c->lq(t9, 16, t9); // lq t9, 16(t9) - c->pcpyud(s3, s3, s2); // pcpyud s3, s3, s2 - c->lq(s4, 16, s4); // lq s4, 16(s4) - c->pand(s3, s3, a0); // pand s3, s3, a0 - c->lq(ra, 16, ra); // lq ra, 16(ra) - c->por(s3, s3, gp); // por s3, s3, gp - c->mov128_vf_gpr(vf1, s5); // qmtc2.ni vf1, s5 - c->pextub(gp, r0, s5); // pextub gp, r0, s5 - c->sq(s3, 0, t2); // sq s3, 0(t2) - c->pextub(s5, r0, t9); // pextub s5, r0, t9 - c->mov128_vf_gpr(vf2, t9); // qmtc2.ni vf2, t9 - c->pextub(t9, r0, s4); // pextub t9, r0, s4 - c->mov128_vf_gpr(vf3, s4); // qmtc2.ni vf3, s4 - c->pextub(s4, r0, ra); // pextub s4, r0, ra - c->mov128_vf_gpr(vf4, ra); // qmtc2.ni vf4, ra - c->pextuh(ra, r0, gp); // pextuh ra, r0, gp - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pextuh(gp, r0, s5); // pextuh gp, r0, s5 - c->mov128_vf_gpr(vf5, ra); // qmtc2.ni vf5, ra - c->pextuh(t9, r0, t9); // pextuh t9, r0, t9 - c->mov128_vf_gpr(vf6, gp); // qmtc2.ni vf6, gp - c->pextuh(ra, r0, s4); // pextuh ra, r0, s4 - c->mov128_vf_gpr(vf7, t9); // qmtc2.ni vf7, t9 - c->prot3w(t8, t8); // prot3w t8, t8 - c->mov128_vf_gpr(vf8, ra); // qmtc2.ni vf8, ra - c->prot3w(t7, t7); // prot3w t7, t7 - // Unknown instr: vcallms 0 - vcallms0(c); - c->pextuw(t9, t7, t6); // pextuw t9, t7, t6 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pcpyld(t7, t5, t7); // pcpyld t7, t5, t7 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pcpyld(t6, t9, t6); // pcpyld t6, t9, t6 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->daddiu(t2, t2, 16); // daddiu t2, t2, 16 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pextuw(t5, t8, t5); // pextuw t5, t8, t5 - c->sq(t7, 16, t3); // sq t7, 16(t3) - c->pcpyld(t5, t8, t5); // pcpyld t5, t8, t5 - c->sq(t6, 0, t3); // sq t6, 0(t3) - // nop // sll r0, r0, 0 - c->sq(t5, 32, t3); // sq t5, 32(t3) - c->daddiu(t3, t3, 48); // daddiu t3, t3, 48 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->daddiu(t4, t4, -4); // daddiu t4, t4, -4 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - bc = ((s64)c->sgpr64(t4)) <= 0; // blez t4, L74 - c->mfc1(r0, f31); // mfc1 r0, f31 - if (bc) {goto block_2;} // branch non-likely - - - block_1: - // nop // sll r0, r0, 0 - c->ldr(t1, 0, t0); // ldr t1, 0(t0) - // nop // sll r0, r0, 0 - c->ldl(t1, 7, t0); // ldl t1, 7(t0) - // nop // sll r0, r0, 0 - c->daddiu(t0, t0, 8); // daddiu t0, t0, 8 - c->pextlh(t1, r0, t1); // pextlh t1, r0, t1 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pand(t5, t1, a2); // pand t5, t1, a2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->psllw(t5, t5, 5); // psllw t5, t5, 5 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->paddw(s5, t5, a1); // paddw s5, t5, a1 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->dsrl32(t9, s5, 0); // dsrl32 t9, s5, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pcpyud(s4, s5, r0); // pcpyud s4, s5, r0 - c->lq(t6, 0, s5); // lq t6, 0(s5) - c->dsrl32(ra, s4, 0); // dsrl32 ra, s4, 0 - c->lq(t7, 0, t9); // lq t7, 0(t9) - c->pand(t8, t1, a3); // pand t8, t1, a3 - c->lq(t5, 0, s4); // lq t5, 0(s4) - c->psraw(gp, t8, 8); // psraw gp, t8, 8 - c->lq(t8, 0, ra); // lq t8, 0(ra) - c->pextuw(s3, t7, t6); // pextuw s3, t7, t6 - c->lq(s5, 16, s5); // lq s5, 16(s5) - c->pextuw(s2, t8, t5); // pextuw s2, t8, t5 - c->lq(t9, 16, t9); // lq t9, 16(t9) - c->pcpyud(s3, s3, s2); // pcpyud s3, s3, s2 - c->lq(s4, 16, s4); // lq s4, 16(s4) - c->pand(s3, s3, a0); // pand s3, s3, a0 - c->lq(ra, 16, ra); // lq ra, 16(ra) - c->por(s3, s3, gp); // por s3, s3, gp - c->mov128_vf_gpr(vf1, s5); // qmtc2.ni vf1, s5 - c->pextub(gp, r0, s5); // pextub gp, r0, s5 - c->sq(s3, 0, t2); // sq s3, 0(t2) - c->pextub(s5, r0, t9); // pextub s5, r0, t9 - c->mov128_vf_gpr(vf2, t9); // qmtc2.ni vf2, t9 - c->pextub(t9, r0, s4); // pextub t9, r0, s4 - c->mov128_vf_gpr(vf3, s4); // qmtc2.ni vf3, s4 - c->pextub(s4, r0, ra); // pextub s4, r0, ra - c->mov128_vf_gpr(vf4, ra); // qmtc2.ni vf4, ra - c->pextuh(ra, r0, gp); // pextuh ra, r0, gp - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pextuh(gp, r0, s5); // pextuh gp, r0, s5 - c->mov128_vf_gpr(vf5, ra); // qmtc2.ni vf5, ra - c->pextuh(t9, r0, t9); // pextuh t9, r0, t9 - c->mov128_vf_gpr(vf6, gp); // qmtc2.ni vf6, gp - c->pextuh(ra, r0, s4); // pextuh ra, r0, s4 - c->mov128_vf_gpr(vf7, t9); // qmtc2.ni vf7, t9 - c->prot3w(t8, t8); // prot3w t8, t8 - c->mov128_vf_gpr(vf8, ra); // qmtc2.ni vf8, ra - c->prot3w(t7, t7); // prot3w t7, t7 - // Unknown instr: vcallms 0 - vcallms0(c); - c->pextuw(t9, t7, t6); // pextuw t9, t7, t6 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pcpyld(t7, t5, t7); // pcpyld t7, t5, t7 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pcpyld(t6, t9, t6); // pcpyld t6, t9, t6 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pextuw(t5, t8, t5); // pextuw t5, t8, t5 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pcpyld(t5, t8, t5); // pcpyld t5, t8, t5 - c->sq(t6, 0, t3); // sq t6, 0(t3) - // nop // sll r0, r0, 0 - c->sq(t7, 16, t3); // sq t7, 16(t3) - // nop // sll r0, r0, 0 - c->sq(t5, 32, t3); // sq t5, 32(t3) - c->daddiu(t2, t2, 16); // daddiu t2, t2, 16 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->daddiu(t3, t3, 48); // daddiu t3, t3, 48 - c->mov128_gpr_vf(t7, vf21); // qmfc2.ni t7, vf21 - // nop // sll r0, r0, 0 - c->mov128_gpr_vf(t8, vf22); // qmfc2.ni t8, vf22 - // nop // sll r0, r0, 0 - c->mov128_gpr_vf(t5, vf23); // qmfc2.ni t5, vf23 - // nop // sll r0, r0, 0 - c->mov128_gpr_vf(t6, vf24); // qmfc2.ni t6, vf24 - c->ppach(t7, t8, t7); // ppach t7, t8, t7 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->ppach(t5, t6, t5); // ppach t5, t6, t5 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->ppacb(t5, t5, t7); // ppacb t5, t5, t7 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->sq(t5, 0, v1); // sq t5, 0(v1) - c->daddiu(t4, t4, -4); // daddiu t4, t4, -4 - c->daddiu(v1, v1, 16); // daddiu v1, v1, 16 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - bc = ((s64)c->sgpr64(t4)) > 0; // bgtz t4, L73 - // nop // sll r0, r0, 0 - if (bc) {goto block_1;} // branch non-likely - - - block_2: - // nop // sll r0, r0, 0 - // nop // vnop - // nop // sll r0, r0, 0 - c->mov128_gpr_vf(a2, vf17); // qmfc2.ni a2, vf17 - // nop // sll r0, r0, 0 - c->mov128_gpr_vf(a3, vf18); // qmfc2.ni a3, vf18 - // nop // sll r0, r0, 0 - c->mov128_gpr_vf(a0, vf19); // qmfc2.ni a0, vf19 - // nop // sll r0, r0, 0 - c->mov128_gpr_vf(a1, vf20); // qmfc2.ni a1, vf20 - c->ppach(a2, a3, a2); // ppach a2, a3, a2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->ppach(a0, a1, a0); // ppach a0, a1, a0 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->ppacb(a0, a0, a2); // ppacb a0, a0, a2 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->sq(a0, 0, v1); // sq a0, 0(v1) - c->gprs[v0].du64[0] = 0; // or v0, r0, r0 - c->ld(ra, 12432, at); // ld ra, 12432(at) - c->lq(gp, 12512, at); // lq gp, 12512(at) - c->lq(s5, 12496, at); // lq s5, 12496(at) - c->lq(s4, 12480, at); // lq s4, 12480(at) - c->lq(s3, 12464, at); // lq s3, 12464(at) - c->lq(s2, 12448, at); // lq s2, 12448(at) - //jr ra // jr ra - c->daddiu(sp, sp, 96); // daddiu sp, sp, 96 - goto end_of_function; // return - - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - end_of_function: - return c->gprs[v0].du64[0]; -} - -void link() { - cache.fake_scratchpad_data = intern_from_c("*fake-scratchpad-data*").c(); - gLinkedFunctionTable.reg("generic-light-proc", execute, 128); -} - -} // namespace generic_light_proc -} // namespace Mips2C - -//--------------------------MIPS2C--------------------- -#include "game/mips2c/mips2c_private.h" - -namespace Mips2C::jak1 { -namespace generic_envmap_proc { -struct Cache { - void* fake_scratchpad_data; // *fake-scratchpad-data* -} cache; - -void vcallms48(ExecutionContext* c) { - // nop | mulx.xyzw vf13, vf09, vf31 - c->vfs[vf13].vf.mul(Mask::xyzw, c->vf_src(vf09).vf, c->vf_src(vf31).vf.x()); - // nop | subw.z vf21, vf21, vf00 - c->vfs[vf21].vf.sub(Mask::z, c->vf_src(vf21).vf, c->vf_src(vf00).vf.w()); - // nop | addy.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.y()); - // nop | mulx.xyz vf08, vf08, vf30 - c->vfs[vf08].vf.mul(Mask::xyz, c->vf_src(vf08).vf, c->vf_src(vf30).vf.x()); - // nop | addw.xy vf05, vf05, vf31 - c->vfs[vf05].vf.add(Mask::xy, c->vf_src(vf05).vf, c->vf_src(vf31).vf.w()); - // nop | mul.xyz vf30, vf21, vf13 - c->vfs[vf30].vf.mul(Mask::xyz, c->vf_src(vf21).vf, c->vf_src(vf13).vf); - // nop | addz.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.z()); - // nop | add.xyz vf08, vf08, vf16 - c->vfs[vf08].vf.add(Mask::xyz, c->vf_src(vf08).vf, c->vf_src(vf16).vf); - // move.xyzw vf28, vf27 | ftoi12.xy vf17, vf05 - c->vfs[vf17].vf.ftoi12(Mask::xy, c->vf_src(vf05).vf); c->vfs[vf28].vf.move(Mask::xyzw, c->vf_src(vf27).vf); - // move.xyzw vf02, vf22 | addy.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.y()); c->vfs[vf02].vf.move(Mask::xyzw, c->vf_src(vf22).vf); - // rsqrt Q, vf31.z, vf29.x | mul.xyz vf06, vf06, Q - c->vfs[vf06].vf.mul(Mask::xyz, c->vf_src(vf06).vf, c->Q); c->Q = c->vf_src(vf31).vf.z() / std::sqrt(c->vf_src(vf29).vf.x()); - // nop | mul.xyz vf29, vf08, vf08 - c->vfs[vf29].vf.mul(Mask::xyz, c->vf_src(vf08).vf, c->vf_src(vf08).vf); - // nop | mulx.xyz vf01, vf21, vf28 - c->vfs[vf01].vf.mul(Mask::xyz, c->vf_src(vf21).vf, c->vf_src(vf28).vf.x()); - // nop | addz.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.z()); - // nop | mulx.xyzw vf14, vf10, vf31 - c->vfs[vf14].vf.mul(Mask::xyzw, c->vf_src(vf10).vf, c->vf_src(vf31).vf.x()); - // nop | subw.z vf02, vf02, vf00 - c->vfs[vf02].vf.sub(Mask::z, c->vf_src(vf02).vf, c->vf_src(vf00).vf.w()); - // nop | addy.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.y()); - // nop | mulx.xyz vf01, vf01, vf30 - c->vfs[vf01].vf.mul(Mask::xyz, c->vf_src(vf01).vf, c->vf_src(vf30).vf.x()); - // nop | addw.xy vf06, vf06, vf31 - c->vfs[vf06].vf.add(Mask::xy, c->vf_src(vf06).vf, c->vf_src(vf31).vf.w()); - // nop | mul.xyz vf30, vf02, vf14 - c->vfs[vf30].vf.mul(Mask::xyz, c->vf_src(vf02).vf, c->vf_src(vf14).vf); - // nop | addz.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.z()); - // nop | add.xyz vf01, vf01, vf13 - c->vfs[vf01].vf.add(Mask::xyz, c->vf_src(vf01).vf, c->vf_src(vf13).vf); - // nop | ftoi12.xy vf18, vf06 - c->vfs[vf18].vf.ftoi12(Mask::xy, c->vf_src(vf06).vf); - // nop | addy.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.y()); - // rsqrt Q, vf31.z, vf29.x | mul.xyz vf07, vf07, Q - c->vfs[vf07].vf.mul(Mask::xyz, c->vf_src(vf07).vf, c->Q); c->Q = c->vf_src(vf31).vf.z() / std::sqrt(c->vf_src(vf29).vf.x()); - // move.xyzw vf03, vf23 | mul.xyz vf29, vf01, vf01 - c->vfs[vf29].vf.mul(Mask::xyz, c->vf_src(vf01).vf, c->vf_src(vf01).vf); c->vfs[vf03].vf.move(Mask::xyzw, c->vf_src(vf23).vf); - // nop | muly.xyz vf02, vf02, vf28 - c->vfs[vf02].vf.mul(Mask::xyz, c->vf_src(vf02).vf, c->vf_src(vf28).vf.y()); - // nop | addz.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.z()); - // nop | mulx.xyzw vf15, vf11, vf31 - c->vfs[vf15].vf.mul(Mask::xyzw, c->vf_src(vf11).vf, c->vf_src(vf31).vf.x()); - // nop | subw.z vf03, vf03, vf00 - c->vfs[vf03].vf.sub(Mask::z, c->vf_src(vf03).vf, c->vf_src(vf00).vf.w()); - // nop | addy.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.y()); - // nop | mulx.xyz vf02, vf02, vf30 - c->vfs[vf02].vf.mul(Mask::xyz, c->vf_src(vf02).vf, c->vf_src(vf30).vf.x()); - // nop | addw.xy vf07, vf07, vf31 - c->vfs[vf07].vf.add(Mask::xy, c->vf_src(vf07).vf, c->vf_src(vf31).vf.w()); - // nop | mul.xyz vf30, vf03, vf15 - c->vfs[vf30].vf.mul(Mask::xyz, c->vf_src(vf03).vf, c->vf_src(vf15).vf); - // nop | addz.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.z()); - // nop | add.xyz vf02, vf02, vf14 - c->vfs[vf02].vf.add(Mask::xyz, c->vf_src(vf02).vf, c->vf_src(vf14).vf); - // nop | ftoi12.xy vf19, vf07 - c->vfs[vf19].vf.ftoi12(Mask::xy, c->vf_src(vf07).vf); - // nop | addy.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.y()); - // rsqrt Q, vf31.z, vf29.x | mul.xyz vf08, vf08, Q - c->vfs[vf08].vf.mul(Mask::xyz, c->vf_src(vf08).vf, c->Q); c->Q = c->vf_src(vf31).vf.z() / std::sqrt(c->vf_src(vf29).vf.x()); - // move.xyzw vf04, vf24 | mul.xyz vf29, vf02, vf02 - c->vfs[vf29].vf.mul(Mask::xyz, c->vf_src(vf02).vf, c->vf_src(vf02).vf); c->vfs[vf04].vf.move(Mask::xyzw, c->vf_src(vf24).vf); - // nop | mulz.xyz vf03, vf03, vf28 - c->vfs[vf03].vf.mul(Mask::xyz, c->vf_src(vf03).vf, c->vf_src(vf28).vf.z()); - // nop | addz.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.z()); - // nop | mulx.xyzw vf16, vf12, vf31 - c->vfs[vf16].vf.mul(Mask::xyzw, c->vf_src(vf12).vf, c->vf_src(vf31).vf.x()); - // nop | subw.z vf04, vf04, vf00 - c->vfs[vf04].vf.sub(Mask::z, c->vf_src(vf04).vf, c->vf_src(vf00).vf.w()); - // nop | addy.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.y()); - // nop | mulx.xyz vf03, vf03, vf30 - c->vfs[vf03].vf.mul(Mask::xyz, c->vf_src(vf03).vf, c->vf_src(vf30).vf.x()); - // nop | addw.xy vf08, vf08, vf31 - c->vfs[vf08].vf.add(Mask::xy, c->vf_src(vf08).vf, c->vf_src(vf31).vf.w()); - // nop | mul.xyz vf30, vf04, vf16 - c->vfs[vf30].vf.mul(Mask::xyz, c->vf_src(vf04).vf, c->vf_src(vf16).vf); - // nop | addz.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.z()); - // nop | add.xyz vf03, vf03, vf15 - c->vfs[vf03].vf.add(Mask::xyz, c->vf_src(vf03).vf, c->vf_src(vf15).vf); - // nop | ftoi12.xy vf20, vf08 - c->vfs[vf20].vf.ftoi12(Mask::xy, c->vf_src(vf08).vf); - // nop | addy.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.y()); - // rsqrt Q, vf31.z, vf29.x | mul.xyz vf05, vf01, Q - c->vfs[vf05].vf.mul(Mask::xyz, c->vf_src(vf01).vf, c->Q); c->Q = c->vf_src(vf31).vf.z() / std::sqrt(c->vf_src(vf29).vf.x()); - // move.xyzw vf06, vf02 | mul.xyz vf29, vf03, vf03 - c->vfs[vf29].vf.mul(Mask::xyz, c->vf_src(vf03).vf, c->vf_src(vf03).vf); c->vfs[vf06].vf.move(Mask::xyzw, c->vf_src(vf02).vf); - // move.xyzw vf07, vf03 | mulw.xyz vf08, vf04, vf28 :e - c->vfs[vf08].vf.mul(Mask::xyz, c->vf_src(vf04).vf, c->vf_src(vf28).vf.w()); c->vfs[vf07].vf.move(Mask::xyzw, c->vf_src(vf03).vf); - // nop | addz.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.z()); - -} - -u64 execute(void* ctxt) { - auto* c = (ExecutionContext*)ctxt; - bool bc = false; - c->daddiu(sp, sp, -128); // daddiu sp, sp, -128 - c->sd(ra, 12432, at); // sd ra, 12432(at) - c->sq(s0, 12448, at); // sq s0, 12448(at) - c->sq(s1, 12464, at); // sq s1, 12464(at) - c->sq(s2, 12480, at); // sq s2, 12480(at) - c->sq(s3, 12496, at); // sq s3, 12496(at) - c->sq(s4, 12512, at); // sq s4, 12512(at) - c->sq(s5, 12528, at); // sq s5, 12528(at) - c->sq(gp, 12544, at); // sq gp, 12544(at) - get_fake_spad_addr(at, cache.fake_scratchpad_data, 0, c);// lui at, 28672 - // nop // sll r0, r0, 0 - c->lw(v1, 60, at); // lw v1, 60(at) - // nop // sll r0, r0, 0 - c->lw(a0, 52, at); // lw a0, 52(at) - // nop // sll r0, r0, 0 - c->lw(t0, 0, v1); // lw t0, 0(v1) - // nop // sll r0, r0, 0 - c->lw(a2, 4, v1); // lw a2, 4(v1) - // nop // sll r0, r0, 0 - c->lw(t3, 32, at); // lw t3, 32(at) - // nop // sll r0, r0, 0 - c->lw(v1, 36, at); // lw v1, 36(at) - // nop // sll r0, r0, 0 - c->addiu(t1, r0, 255); // addiu t1, r0, 255 - c->addiu(a3, r0, 256); // addiu a3, r0, 256 - c->lui(a1, -2); // lui a1, -2 - c->lui(t2, 16256); // lui t2, 16256 - c->ori(a1, a1, 65534); // ori a1, a1, 65534 - c->mtc1(f0, t2); // mtc1 f0, t2 - c->daddiu(t2, a0, 3); // daddiu t2, a0, 3 - c->sra(t5, t2, 2); // sra t5, t2, 2 - c->lq(t2, 12048, at); // lq t2, 12048(at) - c->sra(t4, t5, 2); // sra t4, t5, 2 - c->andi(t5, t5, 3); // andi t5, t5, 3 - bc = c->sgpr64(t4) == 0; // beq t4, r0, L68 - // nop // sll r0, r0, 0 - if (bc) {goto block_2;} // branch non-likely - - - block_1: - c->daddiu(t3, t3, 64); // daddiu t3, t3, 64 - c->sq(t2, -64, t3); // sq t2, -64(t3) - // nop // sll r0, r0, 0 - c->sq(t2, -48, t3); // sq t2, -48(t3) - c->daddiu(t4, t4, -1); // daddiu t4, t4, -1 - c->sq(t2, -32, t3); // sq t2, -32(t3) - bc = ((s64)c->sgpr64(t4)) > 0; // bgtz t4, L67 - c->sq(t2, -16, t3); // sq t2, -16(t3) - if (bc) {goto block_1;} // branch non-likely - - - block_2: - bc = c->sgpr64(t5) == 0; // beq t5, r0, L69 - c->daddiu(t4, t5, -1); // daddiu t4, t5, -1 - if (bc) {goto block_6;} // branch non-likely - - bc = c->sgpr64(t4) == 0; // beq t4, r0, L69 - c->sq(t2, 0, t3); // sq t2, 0(t3) - if (bc) {goto block_6;} // branch non-likely - - c->daddiu(t3, t3, 16); // daddiu t3, t3, 16 - c->daddiu(t4, t4, -1); // daddiu t4, t4, -1 - bc = c->sgpr64(t4) == 0; // beq t4, r0, L69 - c->sq(t2, 0, t3); // sq t2, 0(t3) - if (bc) {goto block_6;} // branch non-likely - - c->daddiu(t3, t3, 16); // daddiu t3, t3, 16 - c->daddiu(t4, t4, -1); // daddiu t4, t4, -1 - // nop // sll r0, r0, 0 - c->sq(t2, 0, t3); // sq t2, 0(t3) - - block_6: - c->daddiu(a0, a0, -4); // daddiu a0, a0, -4 - c->lqc2(vf31, 12016, at); // lqc2 vf31, 12016(at) - c->pextlw(a1, a1, a1); // pextlw a1, a1, a1 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pextlw(a1, a1, a1); // pextlw a1, a1, a1 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pextlw(a2, a2, a2); // pextlw a2, a2, a2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pextlw(a2, a2, a2); // pextlw a2, a2, a2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pcpyh(t2, t1); // pcpyh t2, t1 - c->ld(t1, 0, t0); // ld t1, 0(t0) - c->pcpyld(t2, t2, t2); // pcpyld t2, t2, t2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pcpyh(a3, a3); // pcpyh a3, a3 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pcpyld(a3, a3, a3); // pcpyld a3, a3, a3 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->daddiu(t0, t0, 8); // daddiu t0, t0, 8 - c->sq(t2, 112, at); // sq t2, 112(at) - c->pextlh(t1, r0, t1); // pextlh t1, r0, t1 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pand(t2, t1, t2); // pand t2, t1, t2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->psllw(t2, t2, 5); // psllw t2, t2, 5 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->paddw(t5, t2, a2); // paddw t5, t2, a2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->dsrl32(t6, t5, 0); // dsrl32 t6, t5, 0 - c->lwc1(f4, 24, t5); // lwc1 f4, 24(t5) - c->pcpyud(t7, t5, r0); // pcpyud t7, t5, r0 - c->lwc1(f3, 24, t6); // lwc1 f3, 24(t6) - c->dsrl32(t8, t7, 0); // dsrl32 t8, t7, 0 - c->lwc1(f2, 24, t7); // lwc1 f2, 24(t7) - c->pand(t1, t1, a3); // pand t1, t1, a3 - c->lwc1(f1, 24, t8); // lwc1 f1, 24(t8) - c->psraw(t2, t1, 8); // psraw t2, t1, 8 - c->lq(t1, 16, t5); // lq t1, 16(t5) - c->mov128_gpr_gpr(s0, t2); // por s0, t2, r0 - c->subs(f4, f4, f0); // sub.s f4, f4, f0 - c->divs(f4, f0, f4); // div.s f4, f0, f4 - c->lq(t2, 16, t6); // lq t2, 16(t6) - // nop // sll r0, r0, 0 - c->lq(t3, 16, t7); // lq t3, 16(t7) - // nop // sll r0, r0, 0 - c->lq(t4, 16, t8); // lq t4, 16(t8) - // nop // sll r0, r0, 0 - c->lq(t5, 0, t5); // lq t5, 0(t5) - // nop // sll r0, r0, 0 - c->lq(t6, 0, t6); // lq t6, 0(t6) - // nop // sll r0, r0, 0 - c->lq(t7, 0, t7); // lq t7, 0(t7) - // nop // sll r0, r0, 0 - c->lq(t8, 0, t8); // lq t8, 0(t8) - c->muls(f4, f4, f0); // mul.s f4, f4, f0 - // nop // sll r0, r0, 0 - c->subs(f3, f3, f0); // sub.s f3, f3, f0 - // nop // sll r0, r0, 0 - c->divs(f3, f0, f3); // div.s f3, f0, f3 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mfc1(t9, f4); // mfc1 t9, f4 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->muls(f3, f3, f0); // mul.s f3, f3, f0 - // nop // sll r0, r0, 0 - c->subs(f2, f2, f0); // sub.s f2, f2, f0 - // nop // sll r0, r0, 0 - c->divs(f2, f0, f2); // div.s f2, f0, f2 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mfc1(ra, f3); // mfc1 ra, f3 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->muls(f2, f2, f0); // mul.s f2, f2, f0 - // nop // sll r0, r0, 0 - c->subs(f1, f1, f0); // sub.s f1, f1, f0 - // nop // sll r0, r0, 0 - c->divs(f1, f0, f1); // div.s f1, f0, f1 - // nop // sll r0, r0, 0 - c->pextlw(t9, ra, t9); // pextlw t9, ra, t9 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mfc1(ra, f2); // mfc1 ra, f2 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mfc1(gp, f1); // mfc1 gp, f1 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->pextlw(ra, gp, ra); // pextlw ra, gp, ra - // nop // sll r0, r0, 0 - c->pcpyld(t9, ra, t9); // pcpyld t9, ra, t9 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf21, t1); // qmtc2.ni vf21, t1 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf22, t2); // qmtc2.ni vf22, t2 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf23, t3); // qmtc2.ni vf23, t3 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf24, t4); // qmtc2.ni vf24, t4 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf9, t5); // qmtc2.ni vf9, t5 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf10, t6); // qmtc2.ni vf10, t6 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf11, t7); // qmtc2.ni vf11, t7 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf12, t8); // qmtc2.ni vf12, t8 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf27, t9); // qmtc2.ni vf27, t9 - c->lq(t2, 112, at); // lq t2, 112(at) - // Unknown instr: vcallms 48 - vcallms48(c); - // nop // sll r0, r0, 0 - c->ld(t1, 0, t0); // ld t1, 0(t0) - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->daddiu(t0, t0, 8); // daddiu t0, t0, 8 - c->pextlh(t1, r0, t1); // pextlh t1, r0, t1 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pand(t2, t1, t2); // pand t2, t1, t2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->psllw(t2, t2, 5); // psllw t2, t2, 5 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->paddw(t5, t2, a2); // paddw t5, t2, a2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->dsrl32(t6, t5, 0); // dsrl32 t6, t5, 0 - c->lwc1(f3, 24, t5); // lwc1 f3, 24(t5) - c->pcpyud(t7, t5, r0); // pcpyud t7, t5, r0 - c->lwc1(f2, 24, t6); // lwc1 f2, 24(t6) - c->dsrl32(t8, t7, 0); // dsrl32 t8, t7, 0 - c->lwc1(f1, 24, t7); // lwc1 f1, 24(t7) - c->pand(t1, t1, a3); // pand t1, t1, a3 - c->lwc1(f4, 24, t8); // lwc1 f4, 24(t8) - c->psraw(t2, t1, 8); // psraw t2, t1, 8 - c->lq(t1, 16, t5); // lq t1, 16(t5) - c->mov128_gpr_gpr(s1, t2); // por s1, t2, r0 - c->subs(f5, f3, f0); // sub.s f5, f3, f0 - c->subs(f3, f2, f0); // sub.s f3, f2, f0 - // nop // sll r0, r0, 0 - c->subs(f2, f1, f0); // sub.s f2, f1, f0 - // nop // sll r0, r0, 0 - c->subs(f1, f4, f0); // sub.s f1, f4, f0 - // nop // sll r0, r0, 0 - c->divs(f4, f0, f5); // div.s f4, f0, f5 - c->lq(t2, 16, t6); // lq t2, 16(t6) - // nop // sll r0, r0, 0 - c->lq(t3, 16, t7); // lq t3, 16(t7) - // nop // sll r0, r0, 0 - c->lq(t4, 16, t8); // lq t4, 16(t8) - // nop // sll r0, r0, 0 - c->lq(t5, 0, t5); // lq t5, 0(t5) - // nop // sll r0, r0, 0 - c->lq(t6, 0, t6); // lq t6, 0(t6) - // nop // sll r0, r0, 0 - c->lq(t7, 0, t7); // lq t7, 0(t7) - // nop // sll r0, r0, 0 - c->lq(t8, 0, t8); // lq t8, 0(t8) - c->muls(f4, f4, f0); // mul.s f4, f4, f0 - // nop // sll r0, r0, 0 - c->divs(f3, f0, f3); // div.s f3, f0, f3 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mfc1(t9, f4); // mfc1 t9, f4 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->muls(f3, f3, f0); // mul.s f3, f3, f0 - // nop // sll r0, r0, 0 - c->divs(f2, f0, f2); // div.s f2, f0, f2 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mfc1(ra, f3); // mfc1 ra, f3 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->muls(f2, f2, f0); // mul.s f2, f2, f0 - // nop // sll r0, r0, 0 - c->divs(f1, f0, f1); // div.s f1, f0, f1 - // nop // sll r0, r0, 0 - c->pextlw(t9, ra, t9); // pextlw t9, ra, t9 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mfc1(ra, f2); // mfc1 ra, f2 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mfc1(gp, f1); // mfc1 gp, f1 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf21, t1); // qmtc2.ni vf21, t1 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf9, t5); // qmtc2.ni vf9, t5 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf10, t6); // qmtc2.ni vf10, t6 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf11, t7); // qmtc2.ni vf11, t7 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf12, t8); // qmtc2.ni vf12, t8 - c->pextlw(t1, gp, ra); // pextlw t1, gp, ra - // nop // sll r0, r0, 0 - c->pcpyld(t1, t1, t9); // pcpyld t1, t1, t9 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf22, t2); // qmtc2.ni vf22, t2 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf23, t3); // qmtc2.ni vf23, t3 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf24, t4); // qmtc2.ni vf24, t4 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf27, t1); // qmtc2.ni vf27, t1 - c->lq(t2, 112, at); // lq t2, 112(at) - // Unknown instr: vcallms 48 - vcallms48(c); - // nop // sll r0, r0, 0 - c->ld(t1, 0, t0); // ld t1, 0(t0) - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->daddiu(t0, t0, 8); // daddiu t0, t0, 8 - c->pextlh(t1, r0, t1); // pextlh t1, r0, t1 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pand(t2, t1, t2); // pand t2, t1, t2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->psllw(t2, t2, 5); // psllw t2, t2, 5 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->paddw(t2, t2, a2); // paddw t2, t2, a2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->dsrl32(t3, t2, 0); // dsrl32 t3, t2, 0 - c->lwc1(f3, 24, t2); // lwc1 f3, 24(t2) - c->pcpyud(t7, t2, r0); // pcpyud t7, t2, r0 - c->lwc1(f2, 24, t3); // lwc1 f2, 24(t3) - c->dsrl32(t8, t7, 0); // dsrl32 t8, t7, 0 - c->lwc1(f1, 24, t7); // lwc1 f1, 24(t7) - c->pand(t1, t1, a3); // pand t1, t1, a3 - c->lwc1(f4, 24, t8); // lwc1 f4, 24(t8) - c->psraw(t4, t1, 8); // psraw t4, t1, 8 - c->lq(t1, 16, t2); // lq t1, 16(t2) - c->mov128_gpr_gpr(s2, t4); // por s2, t4, r0 - c->subs(f5, f3, f0); // sub.s f5, f3, f0 - c->subs(f3, f2, f0); // sub.s f3, f2, f0 - // nop // sll r0, r0, 0 - c->subs(f2, f1, f0); // sub.s f2, f1, f0 - // nop // sll r0, r0, 0 - c->subs(f1, f4, f0); // sub.s f1, f4, f0 - // nop // sll r0, r0, 0 - c->divs(f4, f0, f5); // div.s f4, f0, f5 - c->lq(t4, 16, t3); // lq t4, 16(t3) - // nop // sll r0, r0, 0 - c->lq(t5, 16, t7); // lq t5, 16(t7) - // nop // sll r0, r0, 0 - c->lq(t6, 16, t8); // lq t6, 16(t8) - // nop // sll r0, r0, 0 - c->lq(t2, 0, t2); // lq t2, 0(t2) - // nop // sll r0, r0, 0 - c->lq(t3, 0, t3); // lq t3, 0(t3) - // nop // sll r0, r0, 0 - c->lq(t7, 0, t7); // lq t7, 0(t7) - // nop // sll r0, r0, 0 - c->lq(t8, 0, t8); // lq t8, 0(t8) - c->muls(f4, f4, f0); // mul.s f4, f4, f0 - // nop // sll r0, r0, 0 - c->divs(f3, f0, f3); // div.s f3, f0, f3 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mfc1(t9, f4); // mfc1 t9, f4 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->muls(f3, f3, f0); // mul.s f3, f3, f0 - // nop // sll r0, r0, 0 - c->divs(f2, f0, f2); // div.s f2, f0, f2 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mfc1(ra, f3); // mfc1 ra, f3 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->muls(f2, f2, f0); // mul.s f2, f2, f0 - // nop // sll r0, r0, 0 - c->divs(f1, f0, f1); // div.s f1, f0, f1 - // nop // sll r0, r0, 0 - c->pextlw(t9, ra, t9); // pextlw t9, ra, t9 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->mfc1(ra, f2); // mfc1 ra, f2 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf21, t1); // qmtc2.ni vf21, t1 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf22, t4); // qmtc2.ni vf22, t4 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf23, t5); // qmtc2.ni vf23, t5 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf24, t6); // qmtc2.ni vf24, t6 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mfc1(t1, f1); // mfc1 t1, f1 - c->pextlw(t1, t1, ra); // pextlw t1, t1, ra - // nop // sll r0, r0, 0 - c->pcpyld(t1, t1, t9); // pcpyld t1, t1, t9 - c->mov128_vf_gpr(vf9, t2); // qmtc2.ni vf9, t2 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf10, t3); // qmtc2.ni vf10, t3 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf11, t7); // qmtc2.ni vf11, t7 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf12, t8); // qmtc2.ni vf12, t8 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf27, t1); // qmtc2.ni vf27, t1 - // nop // sll r0, r0, 0 - c->mov128_gpr_vf(t4, vf17); // qmfc2.ni t4, vf17 - // nop // sll r0, r0, 0 - c->mov128_gpr_vf(t5, vf18); // qmfc2.ni t5, vf18 - // nop // sll r0, r0, 0 - c->mov128_gpr_vf(t6, vf19); // qmfc2.ni t6, vf19 - bc = ((s64)c->sgpr64(a0)) <= 0; // blez a0, L71 - c->mov128_gpr_vf(t7, vf20); // qmfc2.ni t7, vf20 - if (bc) {goto block_8;} // branch non-likely - - - block_7: - c->lq(t2, 112, at); // lq t2, 112(at) - // Unknown instr: vcallms 48 - vcallms48(c); - c->daddiu(a0, a0, -4); // daddiu a0, a0, -4 - c->ld(t1, 0, t0); // ld t1, 0(t0) - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->daddiu(v1, v1, 16); // daddiu v1, v1, 16 - c->daddiu(t0, t0, 8); // daddiu t0, t0, 8 - c->pextlh(t1, r0, t1); // pextlh t1, r0, t1 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pand(t2, t1, t2); // pand t2, t1, t2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->psllw(t2, t2, 5); // psllw t2, t2, 5 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->paddw(t9, t2, a2); // paddw t9, t2, a2 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->dsrl32(ra, t9, 0); // dsrl32 ra, t9, 0 - c->lwc1(f1, 24, t9); // lwc1 f1, 24(t9) - c->pcpyud(s4, t9, r0); // pcpyud s4, t9, r0 - c->lwc1(f4, 24, ra); // lwc1 f4, 24(ra) - c->subs(f3, f1, f0); // sub.s f3, f1, f0 - // nop // sll r0, r0, 0 - c->dsrl32(s3, s4, 0); // dsrl32 s3, s4, 0 - c->lwc1(f1, 24, s4); // lwc1 f1, 24(s4) - c->pand(t1, t1, a3); // pand t1, t1, a3 - c->lwc1(f2, 24, s3); // lwc1 f2, 24(s3) - c->psraw(s5, t1, 8); // psraw s5, t1, 8 - c->lq(t1, 16, t9); // lq t1, 16(t9) - c->divs(f3, f0, f3); // div.s f3, f0, f3 - c->lq(t2, 16, ra); // lq t2, 16(ra) - c->mov128_gpr_gpr(gp, s0); // por gp, s0, r0 - c->subs(f4, f4, f0); // sub.s f4, f4, f0 - c->ppach(t4, r0, t4); // ppach t4, r0, t4 - c->lq(t3, 16, s4); // lq t3, 16(s4) - c->ppach(t5, r0, t5); // ppach t5, r0, t5 - c->lq(t8, 16, s3); // lq t8, 16(s3) - c->ppach(t6, r0, t6); // ppach t6, r0, t6 - c->lq(t9, 0, t9); // lq t9, 0(t9) - c->ppach(t7, r0, t7); // ppach t7, r0, t7 - c->lq(ra, 0, ra); // lq ra, 0(ra) - c->pextlw(t4, t5, t4); // pextlw t4, t5, t4 - c->lq(t5, 0, s4); // lq t5, 0(s4) - c->pextlw(t6, t7, t6); // pextlw t6, t7, t6 - c->lq(t7, 0, s3); // lq t7, 0(s3) - c->mov128_gpr_gpr(s0, s1); // por s0, s1, r0 - c->muls(f5, f3, f0); // mul.s f5, f3, f0 - c->mov128_gpr_gpr(s1, s2); // por s1, s2, r0 - c->divs(f3, f0, f4); // div.s f3, f0, f4 - c->pcpyld(t4, t6, t4); // pcpyld t4, t6, t4 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pand(t4, t4, a1); // pand t4, t4, a1 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->mov128_gpr_gpr(s2, s5); // por s2, s5, r0 - c->mfc1(t6, f5); // mfc1 t6, f5 - c->subs(f4, f1, f0); // sub.s f4, f1, f0 - // nop // sll r0, r0, 0 - c->por(t4, t4, gp); // por t4, t4, gp - // nop // sll r0, r0, 0 - c->subs(f1, f2, f0); // sub.s f1, f2, f0 - // nop // sll r0, r0, 0 - c->muls(f2, f3, f0); // mul.s f2, f3, f0 - c->sq(t4, -16, v1); // sq t4, -16(v1) - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->divs(f3, f0, f4); // div.s f3, f0, f4 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mfc1(t4, f2); // mfc1 t4, f2 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->muls(f2, f3, f0); // mul.s f2, f3, f0 - // nop // sll r0, r0, 0 - c->pextlw(t4, t4, t6); // pextlw t4, t4, t6 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->divs(f1, f0, f1); // div.s f1, f0, f1 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mfc1(t6, f2); // mfc1 t6, f2 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf21, t1); // qmtc2.ni vf21, t1 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf22, t2); // qmtc2.ni vf22, t2 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf23, t3); // qmtc2.ni vf23, t3 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf24, t8); // qmtc2.ni vf24, t8 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - c->mfc1(t1, f1); // mfc1 t1, f1 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf9, t9); // qmtc2.ni vf9, t9 - c->pextlw(t1, t1, t6); // pextlw t1, t1, t6 - c->mov128_vf_gpr(vf10, ra); // qmtc2.ni vf10, ra - c->pcpyld(t1, t1, t4); // pcpyld t1, t1, t4 - c->mov128_vf_gpr(vf11, t5); // qmtc2.ni vf11, t5 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf12, t7); // qmtc2.ni vf12, t7 - // nop // sll r0, r0, 0 - c->mov128_vf_gpr(vf27, t1); // qmtc2.ni vf27, t1 - // nop // sll r0, r0, 0 - c->mov128_gpr_vf(t4, vf17); // qmfc2.ni t4, vf17 - // nop // sll r0, r0, 0 - c->mov128_gpr_vf(t5, vf18); // qmfc2.ni t5, vf18 - // nop // sll r0, r0, 0 - c->mov128_gpr_vf(t6, vf19); // qmfc2.ni t6, vf19 - bc = ((s64)c->sgpr64(a0)) > 0; // bgtz a0, L70 - c->mov128_gpr_vf(t7, vf20); // qmfc2.ni t7, vf20 - if (bc) {goto block_7;} // branch non-likely - - - block_8: - c->daddiu(v1, v1, 16); // daddiu v1, v1, 16 - // nop // sll r0, r0, 0 - c->ppach(t4, r0, t4); // ppach t4, r0, t4 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->ppach(t5, r0, t5); // ppach t5, r0, t5 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->ppach(t6, r0, t6); // ppach t6, r0, t6 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->ppach(t7, r0, t7); // ppach t7, r0, t7 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pextlw(t4, t5, t4); // pextlw t4, t5, t4 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pextlw(t6, t7, t6); // pextlw t6, t7, t6 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pcpyld(t4, t6, t4); // pcpyld t4, t6, t4 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->pand(t4, t4, a1); // pand t4, t4, a1 - c->mfc1(r0, f31); // mfc1 r0, f31 - c->por(t4, t4, s0); // por t4, t4, s0 - c->mfc1(r0, f31); // mfc1 r0, f31 - // nop // sll r0, r0, 0 - c->sq(t4, -16, v1); // sq t4, -16(v1) - c->gprs[v0].du64[0] = 0; // or v0, r0, r0 - c->ld(ra, 12432, at); // ld ra, 12432(at) - c->lq(gp, 12544, at); // lq gp, 12544(at) - c->lq(s5, 12528, at); // lq s5, 12528(at) - c->lq(s4, 12512, at); // lq s4, 12512(at) - c->lq(s3, 12496, at); // lq s3, 12496(at) - c->lq(s2, 12480, at); // lq s2, 12480(at) - c->lq(s1, 12464, at); // lq s1, 12464(at) - c->lq(s0, 12448, at); // lq s0, 12448(at) - //jr ra // jr ra - c->daddiu(sp, sp, 128); // daddiu sp, sp, 128 - goto end_of_function; // return - - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - end_of_function: - return c->gprs[v0].du64[0]; -} - -void link() { - cache.fake_scratchpad_data = intern_from_c("*fake-scratchpad-data*").c(); - gLinkedFunctionTable.reg("generic-envmap-proc", execute, 256); -} - -} // namespace generic_envmap_proc -} // namespace Mips2C - - -//--------------------------MIPS2C--------------------- -#include "game/mips2c/mips2c_private.h" - -namespace Mips2C::jak1 { -namespace generic_prepare_dma_single { -struct Cache { - void* fake_scratchpad_data; // *fake-scratchpad-data* -} cache; - -u64 execute(void* ctxt) { - auto* c = (ExecutionContext*)ctxt; - bool bc = false; - c->daddiu(sp, sp, -32); // daddiu sp, sp, -32 - c->sq(gp, 12448, at); // sq gp, 12448(at) - get_fake_spad_addr(at, cache.fake_scratchpad_data, 0, c);// lui at, 28672 - // nop // sll r0, r0, 0 - c->lw(t1, 60, at); // lw t1, 60(at) - // nop // sll r0, r0, 0 - c->lw(t8, 64, at); // lw t8, 64(at) - // nop // sll r0, r0, 0 - c->lw(v1, 40, at); // lw v1, 40(at) - // nop // sll r0, r0, 0 - c->lw(t3, 72, at); // lw t3, 72(at) - // nop // sll r0, r0, 0 - c->lh(a1, 18, t1); // lh a1, 18(t1) - c->mov64(a0, v1); // or a0, v1, r0 - c->lbu(a2, 16, t1); // lbu a2, 16(t1) - bc = c->sgpr64(t8) == 0; // beq t8, r0, L55 - c->lq(t4, 11744, at); // lq t4, 11744(at) - if (bc) {goto block_10;} // branch non-likely - - // nop // sll r0, r0, 0 - c->lq(t9, 11808, at); // lq t9, 11808(at) - // nop // sll r0, r0, 0 - c->lq(gp, 11824, at); // lq gp, 11824(at) - c->mov64(a3, a2); // or a3, a2, r0 - c->lq(t7, 11840, at); // lq t7, 11840(at) - bc = c->sgpr64(t3) != 0; // bne t3, r0, L49 - c->lq(t2, 11856, at); // lq t2, 11856(at) - if (bc) {goto block_3;} // branch non-likely - - // nop // sll r0, r0, 0 - c->lq(t0, 11872, at); // lq t0, 11872(at) - // nop // sll r0, r0, 0 - c->lq(t5, 11888, at); // lq t5, 11888(at) - //beq r0, r0, L50 // beq r0, r0, L50 - c->lq(t6, 11936, at); // lq t6, 11936(at) - goto block_4; // branch always - - - block_3: - // nop // sll r0, r0, 0 - c->lq(t0, 12032, at); // lq t0, 12032(at) - // nop // sll r0, r0, 0 - c->lq(t5, 11936, at); // lq t5, 11936(at) - // nop // sll r0, r0, 0 - c->lq(t6, 11936, at); // lq t6, 11936(at) - - block_4: - // nop // sll r0, r0, 0 - c->sq(t4, 0, a0); // sq t4, 0(a0) - // nop // sll r0, r0, 0 - c->sq(t9, 16, a0); // sq t9, 16(a0) - // nop // sll r0, r0, 0 - c->sq(gp, 32, a0); // sq gp, 32(a0) - c->mov64(t4, t8); // or t4, t8, r0 - c->sq(t7, 48, a0); // sq t7, 48(a0) - c->daddiu(t1, t1, 22); // daddiu t1, t1, 22 - c->sq(t2, 64, a0); // sq t2, 64(a0) - c->addiu(t2, r0, 0); // addiu t2, r0, 0 - c->sq(t0, 80, a0); // sq t0, 80(a0) - c->addiu(t0, r0, 128); // addiu t0, r0, 128 - c->sq(t5, 96, a0); // sq t5, 96(a0) - bc = c->sgpr64(t3) != 0; // bne t3, r0, L52 - c->sq(t6, 112, a0); // sq t6, 112(a0) - if (bc) {goto block_7;} // branch non-likely - - - block_5: - c->daddu(a0, a0, t0); // daddu a0, a0, t0 - c->lq(t0, 0, t4); // lq t0, 0(t4) - c->daddiu(a3, a3, -1); // daddiu a3, a3, -1 - c->lbu(t3, 0, t1); // lbu t3, 0(t1) - // nop // sll r0, r0, 0 - c->lq(t5, 16, t4); // lq t5, 16(t4) - c->daddu(t7, t3, t3); // daddu t7, t3, t3 - c->lq(t6, 32, t4); // lq t6, 32(t4) - c->daddu(t8, t7, t3); // daddu t8, t7, t3 - c->lq(t7, 48, t4); // lq t7, 48(t4) - c->daddiu(t9, t8, 9); // daddiu t9, t8, 9 - c->lq(t8, 64, t4); // lq t8, 64(t4) - c->daddiu(t4, t4, 80); // daddiu t4, t4, 80 - c->sq(t0, 0, a0); // sq t0, 0(a0) - // nop // sll r0, r0, 0 - c->sw(t2, 12, a0); // sw t2, 12(a0) - c->daddiu(t1, t1, 1); // daddiu t1, t1, 1 - c->sq(t5, 16, a0); // sq t5, 16(a0) - c->daddu(t2, t2, t9); // daddu t2, t2, t9 - c->sw(t3, 28, a0); // sw t3, 28(a0) - // nop // sll r0, r0, 0 - c->sq(t6, 32, a0); // sq t6, 32(a0) - c->addiu(t0, r0, 80); // addiu t0, r0, 80 - c->sq(t7, 48, a0); // sq t7, 48(a0) - bc = ((s64)c->sgpr64(a3)) > 0; // bgtz a3, L51 - c->sq(t8, 64, a0); // sq t8, 64(a0) - if (bc) {goto block_5;} // branch non-likely - - //beq r0, r0, L54 // beq r0, r0, L54 - // nop // sll r0, r0, 0 - goto block_9; // branch always - - - block_7: - // nop // sll r0, r0, 0 - c->lw(t3, 68, at); // lw t3, 68(at) - // nop // sll r0, r0, 0 - c->lq(t4, 0, t3); // lq t4, 0(t3) - // nop // sll r0, r0, 0 - c->lq(t5, 16, t3); // lq t5, 16(t3) - // nop // sll r0, r0, 0 - c->lq(t6, 32, t3); // lq t6, 32(t3) - // nop // sll r0, r0, 0 - c->lq(t7, 48, t3); // lq t7, 48(t3) - // nop // sll r0, r0, 0 - c->lq(t8, 64, t3); // lq t8, 64(t3) - - block_8: - c->daddu(a0, a0, t0); // daddu a0, a0, t0 - c->lbu(t3, 0, t1); // lbu t3, 0(t1) - c->daddiu(a3, a3, -1); // daddiu a3, a3, -1 - c->sq(t4, 0, a0); // sq t4, 0(a0) - c->daddu(t0, t3, t3); // daddu t0, t3, t3 - c->sw(t2, 12, a0); // sw t2, 12(a0) - c->daddu(t0, t0, t3); // daddu t0, t0, t3 - c->sq(t5, 16, a0); // sq t5, 16(a0) - c->daddiu(t0, t0, 9); // daddiu t0, t0, 9 - c->sw(t3, 28, a0); // sw t3, 28(a0) - c->daddiu(t1, t1, 1); // daddiu t1, t1, 1 - // nop // sll r0, r0, 0 - c->daddu(t2, t2, t0); // daddu t2, t2, t0 - c->sq(t6, 32, a0); // sq t6, 32(a0) - c->addiu(t0, r0, 80); // addiu t0, r0, 80 - c->sq(t7, 48, a0); // sq t7, 48(a0) - bc = ((s64)c->sgpr64(a3)) > 0; // bgtz a3, L53 - c->sq(t8, 64, a0); // sq t8, 64(a0) - if (bc) {goto block_8;} // branch non-likely - - - block_9: - c->ori(a3, t3, 32768); // ori a3, t3, 32768 - c->sw(a1, 52, at); // sw a1, 52(at) - // nop // sll r0, r0, 0 - c->sw(a3, 28, a0); // sw a3, 28(a0) - // nop // sll r0, r0, 0 - c->sw(a2, 92, a0); // sw a2, 92(a0) - // nop // sll r0, r0, 0 - c->sw(a1, 108, v1); // sw a1, 108(v1) - //beq r0, r0, L56 // beq r0, r0, L56 - c->sw(r0, 124, v1); // sw r0, 124(v1) - goto block_11; // branch always - - - block_10: - c->dsll(a3, a2, 2); // dsll a3, a2, 2 - c->sq(t4, 0, a0); // sq t4, 0(a0) - c->daddu(a3, a3, a2); // daddu a3, a3, a2 - c->sw(a1, 108, v1); // sw a1, 108(v1) - c->dsll(a3, a3, 4); // dsll a3, a3, 4 - // nop // sll r0, r0, 0 - c->daddiu(t0, a3, 128); // daddiu t0, a3, 128 - c->sw(a1, 52, at); // sw a1, 52(at) - - block_11: - c->dsll(t1, a2, 2); // dsll t1, a2, 2 - c->lw(a3, 12, v1); // lw a3, 12(v1) - c->daddu(a2, t1, a2); // daddu a2, t1, a2 - c->lw(t1, 84, at); // lw t1, 84(at) - c->daddiu(a2, a2, 7); // daddiu a2, a2, 7 - // nop // sll r0, r0, 0 - c->or_(t2, a3, t1); // or t2, a3, t1 - // nop // sll r0, r0, 0 - c->sll(t3, a2, 16); // sll t3, a2, 16 - c->xori(a3, t1, 38); // xori a3, t1, 38 - c->or_(t1, t2, t3); // or t1, t2, t3 - c->daddiu(a2, a2, 1); // daddiu a2, a2, 1 - // nop // sll r0, r0, 0 - c->sw(t1, 12, v1); // sw t1, 12(v1) - // nop // sll r0, r0, 0 - c->sw(a3, 84, at); // sw a3, 84(at) - c->daddiu(a1, a1, 3); // daddiu a1, a1, 3 - c->daddu(a0, a0, t0); // daddu a0, a0, t0 - c->dsra(a1, a1, 2); // dsra a1, a1, 2 - c->daddiu(a2, a0, 32); // daddiu a2, a0, 32 - c->dsll(t0, a1, 2); // dsll t0, a1, 2 - // nop // sll r0, r0, 0 - c->daddu(a3, t0, t0); // daddu a3, t0, t0 - c->dsll(a1, t0, 2); // dsll a1, t0, 2 - c->daddu(a3, a3, t0); // daddu a3, a3, t0 - c->daddiu(a1, a1, 15); // daddiu a1, a1, 15 - c->dsll(a3, a3, 2); // dsll a3, a3, 2 - c->dsra(a1, a1, 4); // dsra a1, a1, 4 - c->daddiu(a3, a3, 15); // daddiu a3, a3, 15 - c->dsll(t1, a1, 4); // dsll t1, a1, 4 - c->dsra(a3, a3, 4); // dsra a3, a3, 4 - c->lw(a1, 88, at); // lw a1, 88(at) - c->dsll(a3, a3, 4); // dsll a3, a3, 4 - // nop // sll r0, r0, 0 - c->daddu(a3, a2, a3); // daddu a3, a2, a3 - c->lw(t2, 11968, at); // lw t2, 11968(at) - c->daddu(a2, a3, t1); // daddu a2, a3, t1 - c->sq(r0, 0, a3); // sq r0, 0(a3) - c->daddiu(a2, a2, 16); // daddiu a2, a2, 16 - c->sq(r0, -16, a3); // sq r0, -16(a3) - c->daddu(t1, a2, t1); // daddu t1, a2, t1 - c->sq(r0, -32, a3); // sq r0, -32(a3) - c->daddiu(t1, t1, 16); // daddiu t1, t1, 16 - c->sq(r0, 0, a2); // sq r0, 0(a2) - c->subu(t3, t1, v1); // subu t3, t1, v1 - c->sq(r0, -16, a2); // sq r0, -16(a2) - c->sra(t3, t3, 4); // sra t3, t3, 4 - c->sq(r0, 0, a0); // sq r0, 0(a0) - // nop // sll r0, r0, 0 - c->sh(t3, 0, v1); // sh t3, 0(v1) - c->daddiu(v1, t3, 1); // daddiu v1, t3, 1 - c->sq(r0, 0, t1); // sq r0, 0(t1) - // nop // sll r0, r0, 0 - c->sq(r0, -16, t1); // sq r0, -16(t1) - // nop // sll r0, r0, 0 - c->daddiu(t5, a1, 1); // daddiu t5, a1, 1 - c->daddiu(t4, a1, 2); // daddiu t4, a1, 2 - c->lw(t3, 11988, at); // lw t3, 11988(at) - // nop // sll r0, r0, 0 - c->lw(t7, 11972, at); // lw t7, 11972(at) - c->dsll(t0, t0, 16); // dsll t0, t0, 16 - c->lw(t6, 11976, at); // lw t6, 11976(at) - c->or_(t4, t7, t4); // or t4, t7, t4 - c->lw(t7, 11980, at); // lw t7, 11980(at) - c->or_(t5, t6, t5); // or t5, t6, t5 - c->lw(t6, 11984, at); // lw t6, 11984(at) - // nop // sll r0, r0, 0 - c->lw(t8, 11992, at); // lw t8, 11992(at) - c->mov64(t9, a1); // or t9, a1, r0 - c->sw(t2, 8, a0); // sw t2, 8(a0) - c->or_(t2, t7, t9); // or t2, t7, t9 - c->sw(t8, 0, t1); // sw t8, 0(t1) - c->daddiu(a0, a0, 16); // daddiu a0, a0, 16 - c->sw(r0, 4, t1); // sw r0, 4(t1) - c->or_(t4, t4, t0); // or t4, t4, t0 - c->sw(t6, 8, t1); // sw t6, 8(t1) - c->daddiu(a3, a3, 16); // daddiu a3, a3, 16 - c->sw(t3, 12, t1); // sw t3, 12(t1) - c->or_(t1, t5, t0); // or t1, t5, t0 - c->sw(t4, -4, a0); // sw t4, -4(a0) - c->or_(t0, t2, t0); // or t0, t2, t0 - c->sw(t1, -4, a3); // sw t1, -4(a3) - c->daddiu(a2, a2, 16); // daddiu a2, a2, 16 - c->sw(t0, -4, a2); // sw t0, -4(a2) - c->addiu(t0, r0, 567); // addiu t0, r0, 567 - c->sw(v1, 56, at); // sw v1, 56(at) - bc = c->sgpr64(a1) != c->sgpr64(t0); // bne a1, t0, L57 - c->daddiu(v1, a1, 279); // daddiu v1, a1, 279 - if (bc) {goto block_13;} // branch non-likely - - // nop // sll r0, r0, 0 - c->addiu(v1, r0, 9); // addiu v1, r0, 9 - - block_13: - // nop // sll r0, r0, 0 - c->sw(v1, 88, at); // sw v1, 88(at) - // nop // sll r0, r0, 0 - c->sw(a0, 20, at); // sw a0, 20(at) - // nop // sll r0, r0, 0 - c->sw(a3, 24, at); // sw a3, 24(at) - // nop // sll r0, r0, 0 - c->sw(a2, 28, at); // sw a2, 28(at) - c->gprs[v0].du64[0] = 0; // or v0, r0, r0 - c->lq(gp, 12448, at); // lq gp, 12448(at) - //jr ra // jr ra - c->daddiu(sp, sp, 32); // daddiu sp, sp, 32 - goto end_of_function; // return - - // nop // sll r0, r0, 0 - // nop // sll r0, r0, 0 - end_of_function: - return c->gprs[v0].du64[0]; -} - -void link() { - cache.fake_scratchpad_data = intern_from_c("*fake-scratchpad-data*").c(); - gLinkedFunctionTable.reg("generic-prepare-dma-single", execute, 128); -} - -} // namespace generic_prepare_dma_single -} // namespace Mips2C - diff --git a/game/mips2c/jak1_functions/generic_effect2.cpp b/game/mips2c/jak1_functions/generic_effect2.cpp deleted file mode 100644 index 40a8cd13f9..0000000000 --- a/game/mips2c/jak1_functions/generic_effect2.cpp +++ /dev/null @@ -1,127 +0,0 @@ -//--------------------------MIPS2C--------------------- - -#include "game/kernel/jak1/kscheme.h" -#include "game/mips2c/mips2c_private.h" -using namespace jak1; -namespace Mips2C::jak1 { - -// clang-format off -void vcallms48(ExecutionContext* c) { - // nop | mulx.xyzw vf13, vf09, vf31 - c->vfs[vf13].vf.mul(Mask::xyzw, c->vf_src(vf09).vf, c->vf_src(vf31).vf.x()); - // nop | subw.z vf21, vf21, vf00 - c->vfs[vf21].vf.sub(Mask::z, c->vf_src(vf21).vf, c->vf_src(vf00).vf.w()); - // nop | addy.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.y()); - // nop | mulx.xyz vf08, vf08, vf30 - c->vfs[vf08].vf.mul(Mask::xyz, c->vf_src(vf08).vf, c->vf_src(vf30).vf.x()); - // nop | addw.xy vf05, vf05, vf31 - c->vfs[vf05].vf.add(Mask::xy, c->vf_src(vf05).vf, c->vf_src(vf31).vf.w()); - // nop | mul.xyz vf30, vf21, vf13 - c->vfs[vf30].vf.mul(Mask::xyz, c->vf_src(vf21).vf, c->vf_src(vf13).vf); - // nop | addz.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.z()); - // nop | add.xyz vf08, vf08, vf16 - c->vfs[vf08].vf.add(Mask::xyz, c->vf_src(vf08).vf, c->vf_src(vf16).vf); - // move.xyzw vf28, vf27 | ftoi12.xy vf17, vf05 - c->vfs[vf17].vf.ftoi12(Mask::xy, c->vf_src(vf05).vf); c->vfs[vf28].vf.move(Mask::xyzw, c->vf_src(vf27).vf); - // move.xyzw vf02, vf22 | addy.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.y()); c->vfs[vf02].vf.move(Mask::xyzw, c->vf_src(vf22).vf); - // rsqrt Q, vf31.z, vf29.x | mul.xyz vf06, vf06, Q - c->vfs[vf06].vf.mul(Mask::xyz, c->vf_src(vf06).vf, c->Q); c->Q = c->vf_src(vf31).vf.z() / std::sqrt(c->vf_src(vf29).vf.x()); - // nop | mul.xyz vf29, vf08, vf08 - c->vfs[vf29].vf.mul(Mask::xyz, c->vf_src(vf08).vf, c->vf_src(vf08).vf); - // nop | mulx.xyz vf01, vf21, vf28 - c->vfs[vf01].vf.mul(Mask::xyz, c->vf_src(vf21).vf, c->vf_src(vf28).vf.x()); - // nop | addz.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.z()); - // nop | mulx.xyzw vf14, vf10, vf31 - c->vfs[vf14].vf.mul(Mask::xyzw, c->vf_src(vf10).vf, c->vf_src(vf31).vf.x()); - // nop | subw.z vf02, vf02, vf00 - c->vfs[vf02].vf.sub(Mask::z, c->vf_src(vf02).vf, c->vf_src(vf00).vf.w()); - // nop | addy.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.y()); - // nop | mulx.xyz vf01, vf01, vf30 - c->vfs[vf01].vf.mul(Mask::xyz, c->vf_src(vf01).vf, c->vf_src(vf30).vf.x()); - // nop | addw.xy vf06, vf06, vf31 - c->vfs[vf06].vf.add(Mask::xy, c->vf_src(vf06).vf, c->vf_src(vf31).vf.w()); - // nop | mul.xyz vf30, vf02, vf14 - c->vfs[vf30].vf.mul(Mask::xyz, c->vf_src(vf02).vf, c->vf_src(vf14).vf); - // nop | addz.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.z()); - // nop | add.xyz vf01, vf01, vf13 - c->vfs[vf01].vf.add(Mask::xyz, c->vf_src(vf01).vf, c->vf_src(vf13).vf); - // nop | ftoi12.xy vf18, vf06 - c->vfs[vf18].vf.ftoi12(Mask::xy, c->vf_src(vf06).vf); - // nop | addy.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.y()); - // rsqrt Q, vf31.z, vf29.x | mul.xyz vf07, vf07, Q - c->vfs[vf07].vf.mul(Mask::xyz, c->vf_src(vf07).vf, c->Q); c->Q = c->vf_src(vf31).vf.z() / std::sqrt(c->vf_src(vf29).vf.x()); - // move.xyzw vf03, vf23 | mul.xyz vf29, vf01, vf01 - c->vfs[vf29].vf.mul(Mask::xyz, c->vf_src(vf01).vf, c->vf_src(vf01).vf); c->vfs[vf03].vf.move(Mask::xyzw, c->vf_src(vf23).vf); - // nop | muly.xyz vf02, vf02, vf28 - c->vfs[vf02].vf.mul(Mask::xyz, c->vf_src(vf02).vf, c->vf_src(vf28).vf.y()); - // nop | addz.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.z()); - // nop | mulx.xyzw vf15, vf11, vf31 - c->vfs[vf15].vf.mul(Mask::xyzw, c->vf_src(vf11).vf, c->vf_src(vf31).vf.x()); - // nop | subw.z vf03, vf03, vf00 - c->vfs[vf03].vf.sub(Mask::z, c->vf_src(vf03).vf, c->vf_src(vf00).vf.w()); - // nop | addy.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.y()); - // nop | mulx.xyz vf02, vf02, vf30 - c->vfs[vf02].vf.mul(Mask::xyz, c->vf_src(vf02).vf, c->vf_src(vf30).vf.x()); - // nop | addw.xy vf07, vf07, vf31 - c->vfs[vf07].vf.add(Mask::xy, c->vf_src(vf07).vf, c->vf_src(vf31).vf.w()); - // nop | mul.xyz vf30, vf03, vf15 - c->vfs[vf30].vf.mul(Mask::xyz, c->vf_src(vf03).vf, c->vf_src(vf15).vf); - // nop | addz.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.z()); - // nop | add.xyz vf02, vf02, vf14 - c->vfs[vf02].vf.add(Mask::xyz, c->vf_src(vf02).vf, c->vf_src(vf14).vf); - // nop | ftoi12.xy vf19, vf07 - c->vfs[vf19].vf.ftoi12(Mask::xy, c->vf_src(vf07).vf); - // nop | addy.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.y()); - // rsqrt Q, vf31.z, vf29.x | mul.xyz vf08, vf08, Q - c->vfs[vf08].vf.mul(Mask::xyz, c->vf_src(vf08).vf, c->Q); c->Q = c->vf_src(vf31).vf.z() / std::sqrt(c->vf_src(vf29).vf.x()); - // move.xyzw vf04, vf24 | mul.xyz vf29, vf02, vf02 - c->vfs[vf29].vf.mul(Mask::xyz, c->vf_src(vf02).vf, c->vf_src(vf02).vf); c->vfs[vf04].vf.move(Mask::xyzw, c->vf_src(vf24).vf); - // nop | mulz.xyz vf03, vf03, vf28 - c->vfs[vf03].vf.mul(Mask::xyz, c->vf_src(vf03).vf, c->vf_src(vf28).vf.z()); - // nop | addz.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.z()); - // nop | mulx.xyzw vf16, vf12, vf31 - c->vfs[vf16].vf.mul(Mask::xyzw, c->vf_src(vf12).vf, c->vf_src(vf31).vf.x()); - // nop | subw.z vf04, vf04, vf00 - c->vfs[vf04].vf.sub(Mask::z, c->vf_src(vf04).vf, c->vf_src(vf00).vf.w()); - // nop | addy.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.y()); - // nop | mulx.xyz vf03, vf03, vf30 - c->vfs[vf03].vf.mul(Mask::xyz, c->vf_src(vf03).vf, c->vf_src(vf30).vf.x()); - // nop | addw.xy vf08, vf08, vf31 - c->vfs[vf08].vf.add(Mask::xy, c->vf_src(vf08).vf, c->vf_src(vf31).vf.w()); - // nop | mul.xyz vf30, vf04, vf16 - c->vfs[vf30].vf.mul(Mask::xyz, c->vf_src(vf04).vf, c->vf_src(vf16).vf); - // nop | addz.x vf29, vf29, vf29 - c->vfs[vf29].vf.add(Mask::x, c->vf_src(vf29).vf, c->vf_src(vf29).vf.z()); - // nop | add.xyz vf03, vf03, vf15 - c->vfs[vf03].vf.add(Mask::xyz, c->vf_src(vf03).vf, c->vf_src(vf15).vf); - // nop | ftoi12.xy vf20, vf08 - c->vfs[vf20].vf.ftoi12(Mask::xy, c->vf_src(vf08).vf); - // nop | addy.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.y()); - // rsqrt Q, vf31.z, vf29.x | mul.xyz vf05, vf01, Q - c->vfs[vf05].vf.mul(Mask::xyz, c->vf_src(vf01).vf, c->Q); c->Q = c->vf_src(vf31).vf.z() / std::sqrt(c->vf_src(vf29).vf.x()); - // move.xyzw vf06, vf02 | mul.xyz vf29, vf03, vf03 - c->vfs[vf29].vf.mul(Mask::xyz, c->vf_src(vf03).vf, c->vf_src(vf03).vf); c->vfs[vf06].vf.move(Mask::xyzw, c->vf_src(vf02).vf); - // move.xyzw vf07, vf03 | mulw.xyz vf08, vf04, vf28 :e - c->vfs[vf08].vf.mul(Mask::xyz, c->vf_src(vf04).vf, c->vf_src(vf28).vf.w()); c->vfs[vf07].vf.move(Mask::xyzw, c->vf_src(vf03).vf); - // nop | addz.x vf30, vf30, vf30 - c->vfs[vf30].vf.add(Mask::x, c->vf_src(vf30).vf, c->vf_src(vf30).vf.z()); - -} - -} // namespace Mips2C - - diff --git a/game/mips2c/jak1_functions/generic_merc.cpp b/game/mips2c/jak1_functions/generic_merc.cpp index 06182938ce..fa274b7b83 100644 --- a/game/mips2c/jak1_functions/generic_merc.cpp +++ b/game/mips2c/jak1_functions/generic_merc.cpp @@ -3,21 +3,9 @@ using namespace jak1; namespace Mips2C::jak1 { -namespace generic_prepare_dma_single { -u64 execute(void* ctxt); -} -namespace generic_prepare_dma_double { -u64 execute(void* ctxt); -} namespace mercneric_convert { u64 execute(void* ctxt); } -namespace generic_light_proc { -u64 execute(void* ctxt); -} -namespace generic_envmap_proc { -u64 execute(void* ctxt); -} namespace high_speed_reject { u64 execute(void* ctxt); } @@ -292,8 +280,24 @@ struct Cache { void* merc_globals; // *merc-globals* void* merc_death_spawn; // merc-death-spawn void* vector_matrix; // vector-matrix*! + void* generic_effect_stats; // *generic-effect-stats* + void* note_prepare_dma_single; // generic-note-prepare-dma-single + void* note_prepare_dma_double; // generic-note-prepare-dma-double + void* note_light_proc; // generic-note-light-proc + void* note_envmap_proc; // generic-note-envmap-proc } cache; +static u32 symbol_value(void* symbol) { + return *reinterpret_cast(symbol); +} + +static void call_generic_effect(ExecutionContext* c, u32 address, void* note_symbol) { + if (symbol_value(cache.generic_effect_stats) != c->gprs[s7].du32[0]) { + c->jalr(symbol_value(note_symbol)); + } + c->jalr(address); +} + u64 execute(void* ctxt) { auto* c = (ExecutionContext*)ctxt; bool bc = false; @@ -1033,20 +1037,17 @@ u64 execute(void* ctxt) { c->lwu(t9, 7336, v1); // lwu t9, 7336(v1) call_addr = c->gprs[t9].du32[0]; // function call: c->sll(v0, ra, 0); // sll v0, ra, 0 - // c->jalr(call_addr); // jalr ra, t9 - generic_prepare_dma_double::execute(c); + call_generic_effect(c, call_addr, cache.note_prepare_dma_double); get_fake_spad_addr(v1, cache.fake_scratchpad_data, 0, c);// lui v1, 28672 c->lwu(t9, 7340, v1); // lwu t9, 7340(v1) call_addr = c->gprs[t9].du32[0]; // function call: c->sll(v0, ra, 0); // sll v0, ra, 0 - // c->jalr(call_addr); // jalr ra, t9 - generic_light_proc::execute(c); + call_generic_effect(c, call_addr, cache.note_light_proc); get_fake_spad_addr(v1, cache.fake_scratchpad_data, 0, c);// lui v1, 28672 c->lwu(t9, 7344, v1); // lwu t9, 7344(v1) call_addr = c->gprs[t9].du32[0]; // function call: c->sll(v0, ra, 0); // sll v0, ra, 0 - // c->jalr(call_addr); // jalr ra, t9 - generic_envmap_proc::execute(c); + call_generic_effect(c, call_addr, cache.note_envmap_proc); c->lw(v1, 40, at); // lw v1, 40(at) c->lw(a0, 56, at); // lw a0, 56(at) c->mov64(a3, v1); // or a3, v1, r0 @@ -1142,20 +1143,17 @@ u64 execute(void* ctxt) { c->lwu(t9, 7336, v1); // lwu t9, 7336(v1) call_addr = c->gprs[t9].du32[0]; // function call: c->sll(v0, ra, 0); // sll v0, ra, 0 - // c->jalr(call_addr); // jalr ra, t9 - generic_prepare_dma_double::execute(c); + call_generic_effect(c, call_addr, cache.note_prepare_dma_double); get_fake_spad_addr(v1, cache.fake_scratchpad_data, 0, c);// lui v1, 28672 c->lwu(t9, 7340, v1); // lwu t9, 7340(v1) call_addr = c->gprs[t9].du32[0]; // function call: c->sll(v0, ra, 0); // sll v0, ra, 0 - // c->jalr(call_addr); // jalr ra, t9 - generic_light_proc::execute(c); + call_generic_effect(c, call_addr, cache.note_light_proc); get_fake_spad_addr(v1, cache.fake_scratchpad_data, 0, c);// lui v1, 28672 c->lwu(t9, 7344, v1); // lwu t9, 7344(v1) call_addr = c->gprs[t9].du32[0]; // function call: c->sll(v0, ra, 0); // sll v0, ra, 0 - // c->jalr(call_addr); // jalr ra, t9 - generic_envmap_proc::execute(c); + call_generic_effect(c, call_addr, cache.note_envmap_proc); c->lw(v1, 40, at); // lw v1, 40(at) c->lw(a0, 56, at); // lw a0, 56(at) c->mov64(a3, v1); // or a3, v1, r0 @@ -1231,14 +1229,12 @@ u64 execute(void* ctxt) { c->lwu(t9, 7332, v1); // lwu t9, 7332(v1) call_addr = c->gprs[t9].du32[0]; // function call: c->sll(v0, ra, 0); // sll v0, ra, 0 - // c->jalr(call_addr); // jalr ra, t9 - generic_prepare_dma_single::execute(c); + call_generic_effect(c, call_addr, cache.note_prepare_dma_single); get_fake_spad_addr(v1, cache.fake_scratchpad_data, 0, c);// lui v1, 28672 c->lwu(t9, 7340, v1); // lwu t9, 7340(v1) call_addr = c->gprs[t9].du32[0]; // function call: c->sll(v0, ra, 0); // sll v0, ra, 0 - // c->jalr(call_addr); // jalr ra, t9 - generic_light_proc::execute(c); + call_generic_effect(c, call_addr, cache.note_light_proc); c->lw(v1, 40, at); // lw v1, 40(at) c->lw(a0, 56, at); // lw a0, 56(at) c->mov64(a3, v1); // or a3, v1, r0 @@ -1343,6 +1339,11 @@ void link() { cache.merc_globals = intern_from_c("*merc-globals*").c(); cache.merc_death_spawn = intern_from_c("merc-death-spawn").c(); cache.vector_matrix = intern_from_c("vector-matrix*!").c(); + cache.generic_effect_stats = intern_from_c("*generic-effect-stats*").c(); + cache.note_prepare_dma_single = intern_from_c("generic-note-prepare-dma-single").c(); + cache.note_prepare_dma_double = intern_from_c("generic-note-prepare-dma-double").c(); + cache.note_light_proc = intern_from_c("generic-note-light-proc").c(); + cache.note_envmap_proc = intern_from_c("generic-note-envmap-proc").c(); gLinkedFunctionTable.reg("generic-merc-execute-asm", execute, 1024); } @@ -3826,6 +3827,3 @@ void link() { } // namespace high_speed_reject } // namespace Mips2C - - - diff --git a/game/mips2c/jak1_functions/generic_tie.cpp b/game/mips2c/jak1_functions/generic_tie.cpp index 7bed623479..2e2b814838 100644 --- a/game/mips2c/jak1_functions/generic_tie.cpp +++ b/game/mips2c/jak1_functions/generic_tie.cpp @@ -2077,7 +2077,7 @@ u64 execute(void* ctxt) { call_addr = c->gprs[v1].du32[0]; // function call: // Unknown instr: sllv v0, ra, r0 // c->jalr(call_addr); // jalr ra, v1 - generic_prepare_dma_double::execute(c); + // generic_prepare_dma_double::execute(c); c->lw(v1, 752, at); // lw v1, 752(at) call_addr = c->gprs[v1].du32[0]; // function call: // Unknown instr: sllv v0, ra, r0 diff --git a/game/mips2c/mips2c_table.cpp b/game/mips2c/mips2c_table.cpp index 12f80348f2..35c0893ba4 100644 --- a/game/mips2c/mips2c_table.cpp +++ b/game/mips2c/mips2c_table.cpp @@ -76,11 +76,7 @@ namespace setup_blerc_chains_for_one_fragment { extern void link(); } namespace generic_merc_init_asm { extern void link(); } namespace generic_merc_execute_asm { extern void link(); } namespace mercneric_convert { extern void link(); } -namespace generic_prepare_dma_double { extern void link(); } -namespace generic_light_proc { extern void link(); } -namespace generic_envmap_proc { extern void link(); } namespace high_speed_reject { extern void link(); } -namespace generic_prepare_dma_single { extern void link(); } namespace init_ocean_far_regs { extern void link(); } namespace render_ocean_quad { extern void link(); } namespace draw_large_polygon_ocean { extern void link(); } @@ -428,9 +424,6 @@ PerGameVersion>> gMips2C {"generic-merc", {jak1::generic_merc_init_asm::link, jak1::generic_merc_execute_asm::link, jak1::mercneric_convert::link, jak1::high_speed_reject::link}}, - {"generic-effect", - {jak1::generic_prepare_dma_double::link, jak1::generic_light_proc::link, - jak1::generic_envmap_proc::link, jak1::generic_prepare_dma_single::link}}, {"ocean", {jak1::init_ocean_far_regs::link, jak1::render_ocean_quad::link, jak1::draw_large_polygon_ocean::link}}, diff --git a/goal_src/jak1/engine/gfx/generic/generic-effect.gc b/goal_src/jak1/engine/gfx/generic/generic-effect.gc index 12e8faf564..62d54a89ad 100644 --- a/goal_src/jak1/engine/gfx/generic/generic-effect.gc +++ b/goal_src/jak1/engine/gfx/generic/generic-effect.gc @@ -8,39 +8,50 @@ (require "engine/gfx/generic/generic-h.gc") (define-extern *generic-envmap-texture* texture) -;; The effect processors, and the packet builders that frame them. +;; GENERIC converts expanded fragment geometry into the DMA/VIF packets consumed by the Generic VU1 +;; renderer. MERC and Generic TIE first build a GSF source buffer in main memory. Its header supplies +;; strip and draw-point counts, ptr-iks supplies the indexed draw order, and ptr-verts addresses the +;; expanded 32-byte position/texture/normal/color records. Camera, lighting, packet-template, shader, +;; and output-cursor state lives in the shared generic-work scratchpad page. ;; -;; A GENERIC packet is built in three steps, always in this order. A packet builder lays down the -;; header at cur-outbuf and records in generic-saves where each of the VIF unpack streams will start. -;; One or more effect processors then walk the expanded vertices in the GSF buffer and fill those -;; streams. Finally the caller hands cur-outbuf to the fromSPR channel. Nothing is passed in -;; registers: the builders and the processors talk to each other exclusively through generic-saves, -;; which is why they can be called through function pointers from another assembly function. +;; Each packet is assembled in one of the two scratchpad banks selected by saves.cur-outbuf: ;; -;; Every vertex processor in this file has the same shape, and it is worth reading once here rather -;; than eight times below: +;; 1. A packet builder writes the outer DMA/VIF prefix, VU1 header, and per-strip A+D shaders. +;; 2. It reserves packed position, color, and texture-coordinate payloads and publishes their +;; scratchpad addresses through generic-saves. +;; 3. The selected effect processors gather GSF vertices in draw-point order and fill those spans. +;; 4. The caller copies saves.qwc quadwords from cur-outbuf through fromSPR to saves.basep, the +;; current write cursor in the frame's main-memory DMA buffer, then flips the scratchpad bank. ;; -;; - Registers are saved into fx-buf.work.storage2 rather than onto the stack. The stack is still -;; reserved and still unused. Note that the first save happens before `lui at` reloads the page -;; register, so these entries require `at` to already hold the scratchpad page - it does, because -;; every caller's last act is a store into this same page. -;; - The drawing order is gsf-info.ptr-iks: one gsf-ik per draw point, an index byte and a -;; kick-suppression byte. Four draw points are consumed per pass, and the end pointer is -;; ptr-iks + 8 * ceil(num-dps / 4), so a count that is not a multiple of four is rounded up and -;; the last pass reads a little past the list. -;; - The four indices become four vertex addresses in one register: pextlh widens the ik halfwords -;; to words, an AND against a broadcast 255 isolates the index, a shift by five scales it by the -;; 32-byte gsf-vertex, and one packed add against the broadcast ptr-verts finishes it. srl32 and -;; pcpyud then peel the four addresses off for four ordinary loads. This is the reason the -;; processors are indexed at all: a vertex drawn twice is converted once. -;; - The kick byte is extracted from the same widened halfwords with a broadcast 256 and a shift, -;; and ends up in bit 0 of the texture coordinate, where VU1 reads it and turns it into the GS -;; ADC bit. +;; The resulting main-memory block is already a complete Generic DMA/VIF fragment. When the frame's +;; foreground bucket executes, VIF1 uploads its header, shaders, and vertex streams into the selected +;; VU1 banks; VU1 transforms the vertices and emits the final GIF stream to the GS. ;; -;; What differs between processors is only which streams they fill and where the color comes from. +;; Builders and processors take no arguments because generic-saves is their shared interface. Stream +;; sizes are rounded to four draw points, matching the original four-at-a-time processors and keeping +;; each payload qword-aligned; the final group may therefore consume converter-provided padding. +;; Each gsf-ik contributes an 8-bit vertex index and an 8-bit no-kick flag. The processors place the +;; flag in bit 0 of S, which VU1 removes from the coordinate and converts to the GS ADC bit. Authored +;; texture coordinates must consequently keep bit 0 clear. +;; +;; Single-pass packet: +;; DMA/VIF | VU header | shaders | VIF+position | VIF+color | VIF+texture | MSCAL/FLUSH +;; +;; Base + environment packet: +;; DMA/VIF | base header | base shaders | position | color | texture +;; | VIF switch | env header | env shaders | env color | env texture +;; | DMA REF(position) | MSCAL +;; +;; The environment pass carries only a second color and texture-coordinate stream. Its DMA REF points +;; back to the base positions after fromSPR has copied them to main memory, so the geometry is uploaded +;; once and shaded twice. generic-light-proc fills the base streams; generic-envmap-proc fills the two +;; environment streams. (define *target-lock* (the-as symbol 0)) +;; Immutable packet templates and arithmetic constants. generic-work-init copies this block into +;; scratchpad; packet builders then patch transfer counts, VU destinations, and DMA addresses in the +;; private copy without modifying these defaults. (define *generic-consts* (new 'static 'generic-consts @@ -102,25 +113,24 @@ :light-consts (new 'static 'vector :x 255.0 :y 8388608.0))) + (defun generic-work-init ((sink generic-dma-foreground-sink)) - "Copy the Generic packet templates into scratchpad, restore the sink's VU1 input and GIF cursors, - select the first EE output buffer, and build the per-draw environment-map shader." - ;; Everything the packet builders and processors read out of the constant block starts here, so this - ;; runs once per generic pass and before anything else. The environment-map shader is built rather - ;; than authored because it is the same texture for every object in the level - only its GS state - ;; needs saying, and saying it once is cheaper than shipping a copy per model. - ;; copy to scratchpad copy of the work + "Initialize Generic's per-sink scratchpad state: copy the packet templates, restore sink's VU1 + cursors, select the first output bank, and build the shared environment-map A+D shader." + ;; Seed the private constants before any builder patches its packet templates. (quad-copy! (the-as pointer (-> (scratchpad-object terrain-context) work foreground generic-work fx-buf work consts)) (the-as pointer *generic-consts*) 27) - ;; set buffer addresses + ;; A foreground sink owns its VU1 input/header rotation across calls. Continue that rotation and + ;; start packet construction in the first of the two scratchpad output banks. (set! (-> (scratchpad-object terrain-context) work foreground generic-work saves gifbuf-adr) (-> sink state gifbuf-adr)) (set! (-> (scratchpad-object terrain-context) work foreground generic-work saves inbuf-adr) (-> sink state inbuf-adr)) (set! (-> (scratchpad-object terrain-context) work foreground generic-work saves cur-outbuf) (the-as uint (+ 8192 (scratchpad-object int)))) - ;; initialize the environment map "shader". This isn't a real shader, but just texturing settings. + ;; Every environment strip uses the same texture state, so build one shared A+D shader from the + ;; currently selected environment texture and copy it into each packet that needs the second pass. (let ((envmap-shader (-> (scratchpad-object terrain-context) work @@ -148,50 +158,24 @@ (none)) (#when PC_PORT - ;; VU0 work is performed by the native Generic renderer. + ;; The PC Generic renderer performs the VU0 stages natively. These stubs retain the EE-facing + ;; initialization interface without uploading a microprogram that the host renderer will not run. (defun upload-vu0-program ((func vu-function) (wait-counter pointer)) - "Upload func to VU0 in blocks of at most 127 instruction pairs, waiting for VIF0 DMA to become - idle and charging every busy poll to wait-counter. The DMA construction is highly optimized." + "PC stub for the synchronous EE VU0-program upload interface." (none)) (defun generic-upload-vu0 () - "Build and start an asynchronous VIF0 DMA chain which uploads the Generic VU0 program. This - entry does not wait for completion, and its one-off chain construction is less optimized than - upload-vu0-program." - (none))) - -(#unless PC_PORT - (defun generic-upload-vu0 () - "Build and start an asynchronous VIF0 DMA chain which uploads the Generic VU0 program. This - entry does not wait for completion, and its one-off chain construction is less optimized than - upload-vu0-program." - (let ((dma-buf *vu0-dma-list*)) - (set! (-> dma-buf base) (-> dma-buf data)) - (set! (-> dma-buf end) (&-> dma-buf data-buffer (-> dma-buf allocated-length))) - (dma-buffer-add-vu-function dma-buf generic-vu0-block 0) - (let ((end-tag (the-as (pointer uint64) (-> dma-buf base)))) - (set! (-> end-tag 0) #x70000000) - (set! (-> end-tag 1) 0) - (set! (-> dma-buf base) (&+ (the-as pointer end-tag) 16))) - (sync.l) - (dma-buffer-send-chain (the-as dma-bank-source #x10008000) dma-buf)) + "PC stub for the asynchronous Generic VU0 upload used by the EE renderer." (none))) (defun generic-initialize-without-sink ((camera-transform matrix) (lights vu-lights)) - "Synchronously upload the Generic VU0 program, copy camera-transform into scratchpad, and - optionally copy seven VU lighting vectors. This entry does not initialize or update a foreground - sink. The Generic VU0 block is loaded at program address zero." + "Initialize Generic's camera and optional seven-vector lighting block without changing a sink's + VU1 cursors. On EE, synchronously load the Generic VU0 program at address zero first." (upload-vu0-program generic-vu0-block (the-as pointer #x70000064)) - (let (;(a2-0 (+ #x2e20 (the-as int (the-as terrain-context #x70000000)))) - (work-matrix (-> (scratchpad-object terrain-context) work foreground generic-work fx-buf work consts matrix))) - ;;(set! (-> (the-as (pointer uint128) work-matrix)) right) + (let ((work-matrix (-> (scratchpad-object terrain-context) work foreground generic-work fx-buf work consts matrix))) (matrix-copy! work-matrix camera-transform) - ;;(s.q! (+ work-matrix 16) up) - ;; (s.q! (+ work-matrix 32) forward) - ;;(s.q! (+ work-matrix 48) translation) ) (if lights - ;;(quad-copy! (the-as pointer (+ #x3190 #x70000000)) (the-as pointer lights) 7) (quad-copy! (the pointer (-> (scratchpad-object terrain-context) work foreground generic-work fx-buf work lights)) (the-as pointer lights) 7)) @@ -199,11 +183,10 @@ (none)) (defun generic-initialize ((sink generic-dma-foreground-sink) (camera-transform matrix) (lights vu-lights)) - "Initialize Generic scratchpad state for sink, start the VU0 upload, copy camera-transform, and - optionally copy seven VU lighting vectors." + "Initialize sink's Generic packet state, start the EE VU0 upload when applicable, and copy the + camera transform and optional seven-vector lighting block into scratchpad." (generic-work-init sink) (generic-upload-vu0) - ;;(let ((a2-1 (+ #x2e20 (the-as int #x70000000))) (matrix-copy! (-> (scratchpad-object terrain-context) work foreground generic-work fx-buf work consts matrix) camera-transform) (if lights (quad-copy! (the-as pointer (-> (scratchpad-object terrain-context) work foreground generic-work fx-buf work lights)) @@ -213,10 +196,10 @@ (none)) (defun generic-wrapup ((sink generic-dma-foreground-sink)) - "Save the Generic VU1 input-bank and GIF-buffer cursors from scratchpad back into sink." - ;; The two cursors are per bucket, not per pass. If MERC and then TIE both feed this sink in one - ;; frame, the second pass has to continue the same rotation the first left off at, or it will upload - ;; into a bank the GIF is still reading out of. + "Save Generic's next VU1 input-bank and header-buffer destinations back into sink." + ;; These cursors belong to the foreground sink, not to one conversion pass. If MERC and Generic TIE + ;; both append to the same bucket, the later producer must continue the rotation or it can overwrite + ;; a VU1 bank that an earlier packet is still using. (set! (-> sink state gifbuf-adr) (-> (scratchpad-object terrain-context) work foreground generic-work saves gifbuf-adr)) (set! (-> sink state inbuf-adr) @@ -224,12 +207,11 @@ (none)) (defun generic-none () - "Perform no Generic vertex processing." - ;; Installed where a processor slot has to be filled but there is nothing for it to do. + "Satisfy an effect-processor slot without producing or modifying a vertex stream." (none)) (defun generic-post-debug () - "Print the first 16 quadwords of the shared GSF buffer as four hexadecimal words per row." + "Dump the first 256 bytes of the shared GSF buffer as sixteen hexadecimal quadwords." (when #t (let ((buffer *gsf-buffer*)) (dotimes (i 16) @@ -245,4633 +227,479 @@ 0 (none)) -(#unless PC_PORT - ;; Generic work begins 16 bytes into scratchpad. The assembly keeps the scratchpad page in at, so - ;; offsets 20 through 108 name ptr-vtxs through from-spr-waits in generic-saves. The converters - ;; update those cursors directly because their inner loops share them with the DMA packet builders. - ;; Local vectors used by generic-debug-light-proc. The assembly addresses these labels relative to fp. - (asm-data - (label generic-debug-light-normal-range) - (word #xbf800000 #x3f800000 0 0) - (label generic-debug-light-color-center) - (word 0 0 0 #x43000000) - (label generic-debug-light-color-scale) - (word #x437f0000 #x437f0000 #x437f0000 0)) +;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;; +;; Plain GOAL effect implementations +;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;; - (defun generic-debug-light-proc () - "Replace Generic vertex colors with packed, clamped normal components while preserving the other - expanded vertex attributes." - (declare (asm-func none) (allow-saved-regs)) - ;; A debug processor: it replaces every vertex color with the vertex's own normal, so a surface is - ;; tinted by which way it faces. Clamp xyz into [-1, 1], scale by 127 and bias by 128 so that -1 - ;; becomes 1 and +1 becomes 255, then pack the three floats down to three bytes with ftoi0, ppach - ;; and ppacb. A wrong result looks like flat color, or like color that does not change as the - ;; object turns. - ;; - ;; Everything else about the output is the ordinary layout, which is the point: drop this in place - ;; of generic-light-proc and the geometry is unchanged, only shaded differently. Nothing installs - ;; it in the shipped game. - (asm-block build-range-constants - (label generic-debug-light-proc-entry) - (add.i sp sp -80) - (s.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (s.d fp at (+ (generic-work-offset fx-buf work storage2 data 0) 8)) - (m fp t9) - (s.q s3 at (generic-work-offset fx-buf work storage2 data 1)) - (s.q s4 at (generic-work-offset fx-buf work storage2 data 2)) - (s.q s5 at (generic-work-offset fx-buf work storage2 data 3)) - (s.q gp at (generic-work-offset fx-buf work storage2 data 4)) - (add.i v1 fp generic-debug-light-color-scale) - (add.i a0 fp generic-debug-light-color-center) - (add.i a1 fp generic-debug-light-normal-range) - (l.vf vf5 v1) - (l.vf vf6 a0) - (l.vf vf7 a1) - (m v1 vf7) - (lui at #x7000) - (nop!) - (nop!) - (rlet ((gsf-buf :reg a0 :type gsf-buffer)) - (l.w gsf-buf at (generic-work-offset saves gsf-buf)) - (nop!) - (l.w a3 at (generic-work-offset saves num-dps)) - (nop!) - (l.w v1 (-> gsf-buf info ptr-iks)) - (nop!) - (l.w t1 (-> gsf-buf info ptr-verts))) - (nop!) - (l.w a0 at (generic-work-offset saves ptr-vtxs)) - (nop!) - (l.w a1 at (generic-work-offset saves ptr-clrs)) - (add.i t2 r0 255) - (l.w a2 at (generic-work-offset saves ptr-texs)) - (add.i t3 r0 256) ;; not a DMA value: 255 and 256 broadcast into the index and kick-bit masks - (lui t0 -2) - (add.i t4 a3 -4) - (m a3 a3) - (nop!) - (ori t0 t0 #xfffe) - (mmi-nop!) - (pextlw t0 t0 t0) - (mmi-nop!) - (pextlw t0 t0 t0) - (mmi-nop!) - (pextlw t1 t1 t1) - (mmi-nop!) - (pextlw t1 t1 t1) - (mmi-nop!) - (pcpyh t2 t2) - (mmi-nop!) - (pcpyld t2 t2 t2) - (mmi-nop!) - (pcpyh t3 t3) - (mmi-nop!) - (pcpyld t3 t3 t3)) - ;; Four vertices a pass. The clamp, scale and bias happen in VU0 registers because vf5-vf7 already - ;; hold the three constants; only the pack down to bytes is EE work. - (asm-block vertex-loop - (label generic-debug-light-proc-vertex-loop) - (add.i a3 a3 -4) - (l.dr t4 v1) - (add.i a0 a0 48) - (l.dl t4 v1 7) - (add.i a2 a2 16) - (add.i v1 v1 8) - (pextlh t4 r0 t4) - (mmi-nop!) - (and.q t5 t4 t2) - (mmi-nop!) - (sll.w t5 t5 5) - (mmi-nop!) - (add.w s4 t5 t1) - (mmi-nop!) - (srl32 s5 s4 0) - (l.q t6 s4) - (pcpyud gp s4 r0) - (l.q t7 s5) - (srl32 t9 gp 0) - (l.q t5 gp) - (and.q ra t4 t3) - (l.q t8 t9) - (sra.w ra ra 8) - (l.vf vf1 s4 16) - (pextuw s4 t7 t6) - (l.vf vf2 s5 16) - (pextuw s5 t8 t5) - (l.vf vf3 gp 16) - (pcpyud gp s4 s5) - (l.vf vf4 t9 16) - (and.q t9 gp t0) - (nop!) - (or.q t9 t9 ra) - (nop!) - (nop!) - (max.x.vf.xyz vf1 vf1 vf7) - (nop!) - (max.x.vf.xyz vf2 vf2 vf7) - (nop!) - (max.x.vf.xyz vf3 vf3 vf7) - (nop!) - (max.x.vf.xyz vf4 vf4 vf7) - (nop!) - (min.y.vf.xyz vf1 vf1 vf7) - (nop!) - (min.y.vf.xyz vf2 vf2 vf7) - (nop!) - (min.y.vf.xyz vf3 vf3 vf7) - (nop!) - (min.y.vf.xyz vf4 vf4 vf7) - (nop!) - (mul.vf vf1 vf1 vf5) - (nop!) - (mul.vf vf2 vf2 vf5) - (nop!) - (mul.vf vf3 vf3 vf5) - (nop!) - (mul.vf vf4 vf4 vf5) - (nop!) - (add.vf vf1 vf1 vf6) - (nop!) - (add.vf vf2 vf2 vf6) - (nop!) - (add.vf vf3 vf3 vf6) - (nop!) - (add.vf vf4 vf4 vf6) - (nop!) - (ftoi.vf vf1 vf1) - (nop!) - (ftoi.vf vf2 vf2) - (nop!) - (ftoi.vf vf3 vf3) - (nop!) - (ftoi.vf vf4 vf4) - (nop!) - (m gp vf1) - (nop!) - (m s5 vf2) - (nop!) - (m s4 vf3) - (nop!) - (m ra vf4) - (ppach gp r0 gp) - (mmi-nop!) - (ppach s5 r0 s5) - (mmi-nop!) - (ppach s4 r0 s4) - (mmi-nop!) - (ppach ra r0 ra) - (mmi-nop!) - (ppacb gp r0 gp) - (mmi-nop!) - (ppacb s3 r0 s5) - (mmi-nop!) - (ppacb s5 r0 s4) - (mmi-nop!) - (ppacb ra r0 ra) - (mmi-nop!) - (pextlw gp s3 gp) - (nop!) - (add.i a1 a1 16) - (add.i s4 a3 -4) - (pextlw ra ra s5) - (mmi-nop!) - (pcpyld ra ra gp) - (s.q t9 a2 -16) - (prot3w t8 t8) - (s.q ra a1 -16) - (prot3w t7 t7) - (mmi-nop!) - (pextuw t9 t7 t6) - (mmi-nop!) - (pcpyld t7 t5 t7) - (mmi-nop!) - (pcpyld t6 t9 t6) - (mmi-nop!) - (pextuw t5 t8 t5) - (s.q t6 a0 -48) - (pcpyld t5 t8 t5) - (s.q t7 a0 -32) - (b.gt a3 r0 generic-debug-light-proc-vertex-loop :delay (s.q t5 a0 -16)) - (m v0 r0) - (l.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (l.d fp at (+ (generic-work-offset fx-buf work storage2 data 0) 8)) - (l.q gp at (generic-work-offset fx-buf work storage2 data 4)) - (l.q s5 at (generic-work-offset fx-buf work storage2 data 3)) - (l.q s4 at (generic-work-offset fx-buf work storage2 data 2)) - (l.q s3 at (generic-work-offset fx-buf work storage2 data 1)) - (jr ra :delay (add.i sp sp 80)) - (nop!) - (nop!) - (nop!)) - ) +;; Portable versions of the packet builders and per-draw-point processors used by the PC converters. +;; They preserve the original scratchpad ABI: builders publish raw stream addresses in generic-saves, +;; processors fill those addresses, and the surrounding MERC/TIE code performs the fromSPR copy. - (defun generic-none-dma-wait () - "Wait until the scratchpad-to-memory DMA channel is idle without processing vertices." - (declare (asm-func none)) - ;; The other do-nothing processor, for the case where there is no work but the previous packet is - ;; still being copied out. Waiting here rather than at the top of the next pass keeps the wait out - ;; of the common path. - ;; - ;; The sixteen-instruction no-op pad is the shape of a poll written for a channel expected to be - ;; busy: a loaded word is not usable for two more instructions, and re-reading a hardware register - ;; faster than this buys nothing. Unlike the other waits in this file it charges no counter. - (asm-block address-the-channel - (label generic-none-dma-wait-entry) - (lui v1 #x1000) - (ori v1 v1 #xd400)) - ;; Sixteen instructions a pass, almost all of them empty. The channel is expected to be busy. - (asm-block poll-dma - (label generic-none-dma-wait-poll-dma) - (l.w a0 v1) - (nop!) - (nop!) - (nop!) - (and.i a0 a0 DMA-CHCR-STR) - (nop!) - (b.z a0 generic-none-dma-wait-dma-idle :delay (nop!)) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (b generic-none-dma-wait-poll-dma :delay (nop!))) - ;; Return zero. There was never anything to convert. - ;; Return zero. There was never anything to convert. - (asm-block dma-idle - (label generic-none-dma-wait-dma-idle) - (m v0 r0) - (jr ra :delay (add sp sp r0)) - (nop!) - (nop!)) - ) +;; Optional counters for comparing the portable effect path with the surrounding conversion cost. +(define *generic-effect-stats* #f) - (defun generic-copy-vtx-dclr-dtex () - "Expand the GSF vertex stream while applying its packed delta-color and delta-texture attributes." - (declare (asm-func none) (allow-saved-regs)) - ;; The plainest of the processors: it fills all three streams and applies the draw-point attribute - ;; lanes, and does no shading at all. A vertex's color is its own dclr added to its own clr, and - ;; its coordinate its dtex added to its own tex; nothing is looked up and no VU0 call happens. - ;; - ;; This is what generic-no-light-dproc does with the delta pass folded in, and nothing installs - ;; either of them in Jak 1 - the shipped no-light path goes through generic-no-light-dproc. Kept - ;; because it is the shortest complete statement of what a processor has to produce. - (asm-block load-stream-cursors - (label generic-copy-vtx-dclr-dtex-entry) - (add.i sp sp -96) - (s.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (s.q s2 at (generic-work-offset fx-buf work storage2 data 1)) - (s.q s3 at (generic-work-offset fx-buf work storage2 data 2)) - (s.q s4 at (generic-work-offset fx-buf work storage2 data 3)) - (s.q s5 at (generic-work-offset fx-buf work storage2 data 4)) - (s.q gp at (generic-work-offset fx-buf work storage2 data 5)) - (lui at #x7000) - (nop!) - (nop!) - (rlet ((gsf-buf :reg a0 :type gsf-buffer)) - (l.w gsf-buf at (generic-work-offset saves gsf-buf)) - (nop!) - (l.w a1 at (generic-work-offset saves num-dps)) - (nop!) - (l.w v1 (-> gsf-buf info ptr-iks)) - (add.i a1 a1 3) - (l.w a2 (-> gsf-buf info ptr-verts))) - (sra a0 a1 2) - (nop!) - (sll a0 a0 3) - (add.i a3 r0 255) - (lui a1 -2) - (add.i t1 r0 256) ;; not a DMA value: 255 and 256 broadcast into the index and kick-bit masks - (ori a1 a1 #xfffe) - (add a0 v1 a0) - (pextlw a1 a1 a1) - (mmi-nop!) - (pextlw a1 a1 a1) - (mmi-nop!) - (pextlw a2 a2 a2) - (l.w t2 at (generic-work-offset saves ptr-vtxs)) - (pextlw a2 a2 a2) - (l.w t3 at (generic-work-offset saves ptr-clrs)) - (pcpyh a3 a3) - (l.w t4 at (generic-work-offset saves ptr-texs)) - (pcpyld a3 a3 a3) - (l.q t0 at (generic-work-offset fx-buf work consts texture-offset)) - (pcpyh t1 t1) - (l.d t5 v1) - (pcpyld t1 t1 t1) - (mmi-nop!) - ;; Prime the indexed lookup before entering the four-vertex loop. Each source halfword selects - ;; one 32-byte GSF vertex record; the four selected records are gathered through EE packed loads. - (pextlh t6 r0 t5) - (mmi-nop!) - (and.q t5 t6 a3) - (mmi-nop!) - (sll.w t5 t5 5) - (mmi-nop!) - (add.i t2 t2 -48) - (add.i t3 t3 -16) - (b generic-copy-vtx-dclr-dtex-prime-vertex-loop :delay (add.i t4 t4 -16))) - (asm-block vertex-loop - (label generic-copy-vtx-dclr-dtex-vertex-loop) - ;; Drain the previous group while the next four packed indices are decoded. - (pextlh t6 r0 ra) - (s.q t5 t2) - (and.q t5 t6 a3) - (s.q t9 t2 16) - (sll.w t5 t5 5) - (s.q t7 t2 32)) - ;; The loop is rotated: this is the body, entered directly from the setup above, and the short block - ;; before it stores the group this pass gathered. So the stores at the top of a pass belong to the - ;; previous four vertices, and the block above is reached once more than this one. - (asm-block prime-vertex-loop - (label generic-copy-vtx-dclr-dtex-prime-vertex-loop) - ;; The low five index bits select a byte offset within the GSF work area. The upper bits carry - ;; the signed color/texture deltas which are merged after the source records have been gathered. - (add.w s4 t5 a2) - (mmi-nop!) - (srl32 gp s4 0) - (add.i t2 t2 48) - (pcpyud ra s4 r0) - (l.q t5 s4) - (srl32 t7 ra 0) - (add.i t3 t3 16) - (and.q t6 t6 t1) - (l.w t9 s4 16) - (sra.w t8 t6 8) - (l.q s2 gp 16) - (add.i t4 t4 16) - (l.q s5 gp 16) - (add.i v1 v1 8) - (l.q s3 gp 16) - (nop!) - (l.q t6 ra) - (pextlw s2 s2 t9) - (l.q t9 gp) - (pextlw s3 s3 s5) - (l.q s5 t7) - (pcpyld s3 s3 s2) - (l.w s4 s4 20) - (nop!) - (l.w gp gp 20) - (add.h s3 s3 t0) - (l.w ra ra 20) - (and.q s3 s3 a1) - (l.w t7 t7 20) - (or.q s3 s3 t8) - (s.w s4 t3) - (prot3w t8 s5) - (s.q s3 t4) - (prot3w t9 t9) - (s.w gp t3 4) - (pextuw gp t9 t5) - (s.w ra t3 8) - (pcpyld t9 t6 t9) - (l.d ra v1) - (pcpyld t5 gp t5) - (mmi-nop!) - (pextuw t6 t8 t6) - (s.w t7 t3 12) - (b.ne v1 a0 generic-copy-vtx-dclr-dtex-vertex-loop :delay (pcpyld t7 t8 t6)) - (nop!) - (s.q t5 t2) - (nop!) - (s.q t9 t2 16) - (nop!) - (s.q t7 t2 32) - (m v0 r0) - (l.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (l.q gp at (generic-work-offset fx-buf work storage2 data 5)) - (l.q s5 at (generic-work-offset fx-buf work storage2 data 4)) - (l.q s4 at (generic-work-offset fx-buf work storage2 data 3)) - (l.q s3 at (generic-work-offset fx-buf work storage2 data 2)) - (l.q s2 at (generic-work-offset fx-buf work storage2 data 1)) - (jr ra :delay (add.i sp sp 96)) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!)) - ) +(deftype generic-effect-debug-stats (structure) + ((timer stopwatch :inline) + (light-vertices uint32) + (envmap-vertices uint32) + (prepare-single-calls uint32) + (prepare-double-calls uint32))) - (defun generic-light ((gsf-buf gsf-buffer) (shaders pointer)) - "Build one lit Generic pass using shaders, then start the scratchpad-to-memory DMA." - (declare (asm-func none)) - ;; The whole calling convention in twenty instructions, which is the reason to read it: store the - ;; two arguments into generic-saves, clear is-envmap so the base GIF template is used, call the - ;; single-pass builder, call the lighting processor, hand the buffer to fromSPR. Any caller that - ;; wants a different set of effects substitutes different calls in the middle and changes nothing - ;; else. - ;; - ;; Nothing installs this in Jak 1 - mercneric and generic-TIE both do the same three steps by hand - ;; so they can choose per fragment - but it is the readable version of what they do. - (asm-block build-vertex-packet - (rlet ((from-spr :reg a2 :type dma-bank-spr)) - (label generic-light-entry) - (add.i sp sp -16) - (s.d ra at (generic-work-offset fx-buf work storage data 0)) - (lui at #x7000) - (s.w a1 at (generic-work-offset saves ptr-shaders)) - (s.w a0 at (generic-work-offset saves gsf-buf)) - (s.w r0 at (generic-work-offset saves is-envmap)) - (m! t9 generic-prepare-dma-single) - (jalr ra t9 :delay (sll v0 ra 0)) - (m! t9 generic-light-proc) - (jalr ra t9 :delay (sll v0 ra 0)) - (l.w v1 at (generic-work-offset saves cur-outbuf)) - (l.w a0 at (generic-work-offset saves qwc)) - (m a3 v1) - (lui at #x7000) - (lui from-spr #x1000) - (l.wu a1 at (generic-work-offset saves basep)) - (ori from-spr from-spr #xd000) - (l.w t1 (-> from-spr chcr)) - (nop!) - (add.i t0 at (generic-work-offset saves from-spr-waits)) - (and.i a3 a3 #x3fff) - (and.i t1 t1 DMA-CHCR-STR) - (nop!) - (b.z t1 generic-light-start-dma :delay (nop!)) - (m t1 from-spr) - (nop!))) - (asm-block wait-for-dma - (label generic-light-wait-for-dma) - (l.w t2 t0) - (nop!) - (l.w t3 t1) - (nop!) - (and.i t3 t3 DMA-CHCR-STR) - (add.i t2 t2 1) - (b.nz t3 generic-light-wait-for-dma :delay (s.w t2 t0)) - (m t0 r0)) - ;; Program the channel, advance saves.basep by exactly what it will deliver, and switch to the - ;; other output buffer. Nothing waits for this transfer - the wait above is what makes that safe. - (asm-block start-dma - (rlet ((from-spr :reg a2 :type dma-bank-spr)) - (label generic-light-start-dma) - (sll t0 a0 4) - (s.w a3 (-> from-spr sadr)) - (nop!) - (s.w a1 (-> from-spr madr)) - (add.i a3 r0 DMA-CHCR-STR) - (s.w a0 (-> from-spr qwc)) - (add a0 a1 t0) - (s.w a3 (-> from-spr chcr)) - (nop!) - (s.w a0 at (generic-work-offset saves basep)) - (m a0 r0) - (xor.i v1 v1 GENERIC-OUTBUF-FLIP) - (s.w v1 at (generic-work-offset saves cur-outbuf)) - (m v0 r0) - (l.d ra at (generic-work-offset fx-buf work storage data 0)) - (jr ra :delay (add.i sp sp 16)) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!))) - ) +(deftype vector3s-packed (structure) + ((x float) + (y float) + (z float)) + :pack-me) - (defun generic-envmap-only-proc () - "Build the environment-map-only Generic vertex stream, including reflected texture coordinates, - environment colors, strip metadata, and its final DMA packet." - (declare (asm-func none) (allow-saved-regs)) - ;; The complete single-pass converter for "the reflection is the only thing drawn": it fills all three - ;; base cursors itself - positions, the flat environment tint as the color, and the reflected - ;; coordinate - so the object is drawn once, as pure reflection, with no base texture underneath. - ;; - ;; It is dead, and so is its other half. saves.is-envmap is cleared everywhere in the game and set - ;; nowhere, and its only reader is generic-prepare-dma-single - so the envmap branch of that builder, - ;; which swaps in the unfogged GIF tag and repeats the one shared environment shader per strip, is - ;; exactly the packet this processor wants and is never reached either. Together they are a coherent - ;; path that was never wired up. - ;; The reflection is a sphere map, and it is worth reading the VU0 side once because the EE half is - ;; unreadable without it. Per vertex, entry 48 reflects the eye direction about the vertex normal - ;; (dot the two, scale the normal by it, add), normalizes the result, multiplies by the 0.5 in - ;; consts.envmap.consts.z and adds the 0.5 in .w - so a reflection pointing straight at the camera - ;; lands at the centre of the texture and one pointing away lands at the edge - then converts x and y - ;; to the GS's 12-bit fixed point with ftoi12. The texture is clamped in both directions, which is - ;; what keeps the edge of the sphere from wrapping. - ;; - ;; So the output is one coordinate pair per draw point that slides across the environment texture as - ;; the object turns. If it were wrong the symptom is unmistakable: the reflection either stops moving - ;; with the object, or smears radially because the normalize went wrong, or clamps to a single edge - ;; color because the bias is off. - ;; - ;; Four vertices per pass, and entry 48 has the same one-group lag as the lighting entry - it copies - ;; its previous results forward at the top - so the coordinates the EE reads out after a call belong - ;; to the group submitted by the call before it. Everything in the loop is one group behind and the - ;; tail publishes the last one. - ;; - ;; Its conversion loop is unrolled three times, and the three tails are the three phases of that - ;; unroll rather than three vertex counts - the names are misleading. The pipeline is three groups - ;; deep and parks its live state in saves.envmap.verts and .kicks at *constant* displacements, so - ;; rotating which slot triple a pass uses is register renaming the ISA cannot do against fixed - ;; offsets. Hence three copies of the body, each hard-coding its phase, and three exits, each - ;; draining the slots its phase left the data in. All three write a full group of four; nothing is - ;; lane-masked anywhere. Which one you land in is ((ceil(dps / 4) - 1) mod 3). - ;; - ;; The state has to live in memory rather than registers because this processor also emits positions, - ;; so it must keep whole quadwords alive across the VU0 call - twelve of them plus three kick masks. - ;; The fourth kicks slot is declared and never touched. - (asm-block load-stream-cursors - (label generic-envmap-only-proc-entry) - (add.i sp sp -128) - (s.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (s.q s0 at (generic-work-offset fx-buf work storage2 data 1)) - (s.q s1 at (generic-work-offset fx-buf work storage2 data 2)) - (s.q s2 at (generic-work-offset fx-buf work storage2 data 3)) - (s.q s3 at (generic-work-offset fx-buf work storage2 data 4)) - (s.q s4 at (generic-work-offset fx-buf work storage2 data 5)) - (s.q s5 at (generic-work-offset fx-buf work storage2 data 6)) - (s.q gp at (generic-work-offset fx-buf work storage2 data 7)) - (lui at #x7000) - (nop!) - (rlet ((gsf-buf :reg a0 :type gsf-buffer)) - (l.w gsf-buf at (generic-work-offset saves gsf-buf)) - (nop!) - (l.w v1 at (generic-work-offset saves num-dps)) - (nop!) - (l.w t1 (-> gsf-buf info ptr-iks)) - (nop!) - (l.w a0 (-> gsf-buf info ptr-verts))) - (nop!) - (l.w a3 at (generic-work-offset saves ptr-vtxs)) - (nop!) - (l.w t4 at (generic-work-offset saves ptr-clrs)) - (nop!) - (l.w a2 at (generic-work-offset saves ptr-texs)) - (nop!) - (add.i t2 r0 255) - (add.i a1 r0 256) ;; not a DMA value: 255 and 256 broadcast into the index and kick-bit masks - (lui t3 -2) - (lui t0 #x3f80) - (ori t3 t3 #xfffe) - (m f0 t0) - (add.i t0 v1 3) - (sra t6 t0 2) - (l.q t0 at (generic-work-offset fx-buf work consts envmap colors)) - (sra t5 t6 2) - (and.i t6 t6 3) - (b.z t5 generic-envmap-only-proc-clear-tail :delay (nop!))) - (asm-block clear-four-loop - (label generic-envmap-only-proc-clear-four-loop) - ;; Environment colors default to the per-draw value. Fill four destinations at a time, then - ;; handle the remaining one to three vertices without overrunning the allocated stream. - (add.i t4 t4 64) - (s.q t0 t4 -64) - (nop!) - (s.q t0 t4 -48) - (add.i t5 t5 -1) - (s.q t0 t4 -32) - (b.gt t5 r0 generic-envmap-only-proc-clear-four-loop :delay (s.q t0 t4 -16))) - (asm-block clear-tail - (label generic-envmap-only-proc-clear-tail) - (b.z t6 generic-envmap-only-proc-prime-conversion :delay (add.i t5 t6 -1)) - (b.z t5 generic-envmap-only-proc-prime-conversion :delay (s.q t0 t4)) - (add.i t4 t4 16) - (add.i t5 t5 -1) - (b.z t5 generic-envmap-only-proc-prime-conversion :delay (s.q t0 t4)) - (add.i t4 t4 16) - (add.i t5 t5 -1) - (nop!) - (s.q t0 t4)) - (asm-block prime-conversion - (label generic-envmap-only-proc-prime-conversion) - ;; Load the camera matrix, environment constants, packed index masks, and four source records. - ;; The first block is primed here so the loop can overlap EE packing with the next VU0 call. - (add.i t0 v1 -4) - (l.vf vf31 at (generic-work-offset fx-buf work consts envmap consts)) - (pextlw v1 t3 t3) - (mmi-nop!) - (pextlw v1 v1 v1) - (mmi-nop!) - (pextlw a0 a0 a0) - (mmi-nop!) - (pextlw a0 a0 a0) - (mmi-nop!) - (pcpyh t3 t2) - (l.dr t2 t1) - (pcpyld t3 t3 t3) - (l.dl t2 t1 7) - (pcpyh a1 a1) - (mmi-nop!) - (pcpyld a1 a1 a1) - (mmi-nop!) - (add.i t1 t1 8) - (s.q t3 at (generic-work-offset saves envmap index-mask)) - (pextlh t2 r0 t2) - (mmi-nop!) - (and.q t3 t2 t3) - (mmi-nop!) - (sll.w t3 t3 5) - (mmi-nop!) - (add.w t7 t3 a0) - (mmi-nop!) - (srl32 t8 t7 0) - (l.s f4 t7 24) - (pcpyud t9 t7 r0) - (l.s f3 t8 24) - (srl32 ra t9 0) - (l.s f2 t9 24) - (and.q t3 t2 a1) - (l.s f1 ra 24) - (sra.w gp t3 8) - (l.q t3 t7 16) - (sub.s f4 f4 f0) - (nop!) - (div.s f4 f0 f4) - (l.q t4 t8 16) - (nop!) - (l.q t5 t9 16) - (nop!) - (l.q t6 ra 16) - (nop!) - (l.q t7 t7) - (nop!) - (l.q t8 t8) - (nop!) - (l.q t9 t9) - (nop!) - (l.q ra ra) - (mul.s f4 f4 f0) - (s.q t7 at (generic-work-offset saves envmap verts 0)) - (sub.s f3 f3 f0) - (nop!) - (div.s f3 f0 f3) - (s.q t8 at (generic-work-offset saves envmap verts 1)) - (nop!) - (s.q t9 at (generic-work-offset saves envmap verts 2)) - (nop!) - (s.q ra at (generic-work-offset saves envmap verts 3)) - (nop!) - (m s5 f4) - (nop!) - (s.q gp at (generic-work-offset saves envmap kicks 0)) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f3 f3 f0) - (nop!) - (sub.s f2 f2 f0) - (nop!) - (div.s f2 f0 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m gp f3) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f2 f2 f0) - (nop!) - (sub.s f1 f1 f0) - (nop!) - (div.s f1 f0 f1) - (nop!) - (pextlw gp gp s5) - (nop!) - (nop!) - (nop!) - (nop!) - (m s5 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m s4 f1) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (pextlw s5 s4 s5) - (nop!) - (pcpyld gp s5 gp) - (nop!) - (nop!) - (m.ni vf21 t3) - (nop!) - (m.ni vf22 t4) - (nop!) - (m.ni vf23 t5) - (nop!) - (m.ni vf24 t6) - (nop!) - (m.ni vf9 t7) - (nop!) - (m.ni vf10 t8) - (nop!) - (m.ni vf11 t9) - (nop!) - (m.ni vf12 ra) - (nop!) - (m.ni vf27 gp) - (l.q t3 at (generic-work-offset saves envmap index-mask)) - (callms GENERIC-VU0-ENVMAP) - (nop!) - (l.dr t2 t1) - (nop!) - (l.dl t2 t1 7) - (nop!) - (add.i t1 t1 8) - (pextlh t2 r0 t2) - (mmi-nop!) - (and.q t3 t2 t3) - (mmi-nop!) - (sll.w t3 t3 5) - (mmi-nop!) - (add.w t7 t3 a0) - (mmi-nop!) - (srl32 t8 t7 0) - (l.s f4 t7 24) - (pcpyud ra t7 r0) - (l.s f3 t8 24) - (srl32 gp ra 0) - (l.s f2 ra 24) - (and.q t3 t2 a1) - (l.s f1 gp 24) - (sra.w t9 t3 8) - (l.q t3 t7 16) - (sub.s f4 f4 f0) - (nop!) - (sub.s f3 f3 f0) - (nop!) - (sub.s f2 f2 f0) - (nop!) - (sub.s f1 f1 f0) - (nop!) - (div.s f4 f0 f4) - (l.q t4 t8 16) - (nop!) - (l.q t5 ra 16) - (nop!) - (l.q t6 gp 16) - (nop!) - (l.q t7 t7) - (nop!) - (l.q t8 t8) - (nop!) - (l.q ra ra) - (nop!) - (l.q gp gp) - (mul.s f4 f4 f0) - (s.q t7 at (generic-work-offset saves envmap verts 4)) - (div.s f3 f0 f3) - (s.q t8 at (generic-work-offset saves envmap verts 5)) - (nop!) - (s.q ra at (generic-work-offset saves envmap verts 6)) - (nop!) - (s.q gp at (generic-work-offset saves envmap verts 7)) - (nop!) - (m s5 f4) - (nop!) - (s.q t9 at (generic-work-offset saves envmap kicks 1)) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f3 f3 f0) - (nop!) - (div.s f2 f0 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m t9 f3) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f2 f2 f0) - (nop!) - (div.s f1 f0 f1) - (nop!) - (pextlw t9 t9 s5) - (nop!) - (nop!) - (m s5 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m s4 f1) - (nop!) - (nop!) - (nop!) - (m.ni vf21 t3) - (nop!) - (m.ni vf9 t7) - (nop!) - (m.ni vf10 t8) - (nop!) - (m.ni vf11 ra) - (nop!) - (m.ni vf12 gp) - (pextlw t3 s4 s5) - (nop!) - (pextlw t3 t3 t9) - (nop!) - (nop!) - (m.ni vf22 t4) - (nop!) - (m.ni vf23 t5) - (nop!) - (m.ni vf24 t6) - (nop!) - (m.ni vf27 t3) - (l.q t3 at (generic-work-offset saves envmap index-mask)) - (callms GENERIC-VU0-ENVMAP) - (nop!) - (l.dr t2 t1) - (nop!) - (l.dl t2 t1 7) - (nop!) - (add.i t1 t1 8) - (pextlh t2 r0 t2) - (mmi-nop!) - (and.q t3 t2 t3) - (mmi-nop!) - (sll.w t3 t3 5) - (mmi-nop!) - (add.w t5 t3 a0) - (mmi-nop!) - (srl32 t6 t5 0) - (l.s f4 t5 24) - (pcpyud t7 t5 r0) - (l.s f3 t6 24) - (srl32 t8 t7 0) - (l.s f2 t7 24) - (and.q t3 t2 a1) - (l.s f1 t8 24) - (sra.w t3 t3 8) - (l.q t4 t5 16) - (sub.s f4 f4 f0) - (nop!) - (sub.s f3 f3 f0) - (nop!) - (sub.s f2 f2 f0) - (nop!) - (sub.s f1 f1 f0) - (nop!) - (div.s f4 f0 f4) - (l.q t9 t6 16) - (nop!) - (l.q ra t7 16) - (nop!) - (l.q gp t8 16) - (nop!) - (l.q t5 t5) - (nop!) - (l.q t6 t6) - (nop!) - (l.q t7 t7) - (nop!) - (l.q t8 t8) - (mul.s f4 f4 f0) - (s.q t5 at (generic-work-offset saves envmap verts 8)) - (div.s f3 f0 f3) - (s.q t6 at (generic-work-offset saves envmap verts 9)) - (nop!) - (s.q t7 at (generic-work-offset saves envmap verts 10)) - (nop!) - (s.q t8 at (generic-work-offset saves envmap verts 11)) - (nop!) - (m s5 f4) - (nop!) - (s.q t3 at (generic-work-offset saves envmap kicks 2)) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f3 f3 f0) - (nop!) - (div.s f2 f0 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m s4 f3) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f2 f2 f0) - (nop!) - (div.s f1 f0 f1) - (l.q t3 at (generic-work-offset saves envmap kicks 0)) - (pextlw s5 s4 s5) - (l.q s4 at (generic-work-offset saves envmap verts 1)) - (nop!) - (l.q s2 at (generic-work-offset saves envmap verts 3)) - (nop!) - (m s3 f2) - (nop!) - (m.ni vf21 t4) - (nop!) - (m.ni vf22 t9) - (nop!) - (m.ni vf23 ra) - (nop!) - (m.ni vf24 gp) - (nop!) - (l.q t4 at (generic-work-offset saves envmap verts 2)) - (nop!) - (m s1 f1) - (prot3w ra s2) - (l.q t9 at (generic-work-offset saves envmap verts 0)) - (prot3w gp s4) - (m.ni vf9 t5) - (pextlw t5 s1 s3) - (m.ni vf10 t6) - (pcpyld t5 t5 s5) - (m.ni vf11 t7) - (pextuw t7 gp t9) - (m.ni vf12 t8) - (pcpyld t6 t4 gp) - (m.ni vf27 t5) - (pcpyld s5 t7 t9) - (m.ni t9 vf17) - (pextuw t4 ra t4) - (m.ni t7 vf18) - (pcpyld ra ra t4) - (m.ni gp vf19) - (b.le t0 r0 generic-envmap-only-proc-finish-four-tail :delay (m.ni t8 vf20))) - (asm-block convert-four-loop - (label generic-envmap-only-proc-convert-four-loop) - ;; Entry 48 transforms four normals and eye vectors into reflected ST coordinates. While VU0 - ;; works, the EE writes the preceding group's position, color, texture, and strip metadata. - (l.q s4 at (generic-work-offset saves envmap index-mask)) - (callms GENERIC-VU0-ENVMAP) - (add.i t0 t0 -4) - (l.dr t2 t1) - (add.i t4 a3 48) - (l.dl t2 t1 7) - (add.i a3 a2 16) - (add.i t5 t1 8) - (pextlh t2 r0 t2) - (s.q s5 t4 -48) - (and.q a2 t2 s4) - (s.q t6 t4 -32) - (sll.w a2 a2 5) - (s.q ra t4 -16) - (add.w s2 a2 a0) - (mmi-nop!) - (srl32 s3 s2 0) - (l.s f1 s2 24) - (pcpyud s4 s2 r0) - (l.s f4 s3 24) - (sub.s f3 f1 f0) - (nop!) - (srl32 s5 s4 0) - (l.s f2 s4 24) - (and.q a2 t2 a1) - (l.s f1 s5 24) - (sra.w a2 a2 8) - (l.q t1 s2 16) - (div.s f3 f0 f3) - (l.q t6 s3 16) - (sub.s f4 f4 f0) - (nop!) - (ppach s1 r0 t9) - (l.q t9 s4 16) - (ppach v0 r0 t7) - (l.q ra s5 16) - (ppach s0 r0 gp) - (l.q t7 s2) - (ppach s2 r0 t8) - (l.q t8 s3) - (pextlw s3 v0 s1) - (l.q gp s4) - (pextlw s4 s2 s0) - (l.q s5 s5) - (mul.s f5 f3 f0) - (s.q t7 at (generic-work-offset saves envmap verts 0)) - (div.s f3 f0 f4) - (s.q t8 at (generic-work-offset saves envmap verts 1)) - (pcpyld s4 s4 s3) - (s.q gp at (generic-work-offset saves envmap verts 2)) - (and.q s4 s4 v1) - (s.q s5 at (generic-work-offset saves envmap verts 3)) - (or.q s4 s4 t3) - (m t3 f5) - (sub.s f2 f2 f0) - (s.q a2 at (generic-work-offset saves envmap kicks 0)) - (nop!) - (s.q s4 a3 -16) - (sub.s f1 f1 f0) - (nop!) - (mul.s f3 f3 f0) - (nop!) - (nop!) - (nop!) - (div.s f2 f0 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (m a2 f3) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f2 f2 f0) - (nop!) - (pextlw t3 a2 t3) - (l.q a2 at (generic-work-offset saves envmap kicks 1)) - (nop!) - (l.q s4 at (generic-work-offset saves envmap verts 5)) - (div.s f1 f0 f1) - (l.q s2 at (generic-work-offset saves envmap verts 7)) - (nop!) - (m s3 f2) - (nop!) - (m.ni vf21 t1) - (nop!) - (m.ni vf22 t6) - (nop!) - (m.ni vf23 t9) - (nop!) - (m.ni vf24 ra) - (nop!) - (l.q ra at (generic-work-offset saves envmap verts 6)) - (nop!) - (l.q t9 at (generic-work-offset saves envmap verts 4)) - (prot3w t1 s2) - (m s2 f1) - (prot3w t6 s4) - (m.ni vf9 t7) - (pextlw t7 s2 s3) - (m.ni vf10 t8) - (pcpyld t3 t7 t3) - (m.ni vf11 gp) - (pextuw t7 t6 t9) - (m.ni vf12 s5) - (pcpyld t6 ra t6) - (m.ni vf27 t3) - (pcpyld gp t7 t9) - (m.ni t9 vf17) - (pextuw t3 t1 ra) - (m.ni t8 vf18) - (pcpyld ra t1 t3) - (m.ni t7 vf19) - (b.le t0 r0 generic-envmap-only-proc-finish-three-tail :delay (m.ni t3 vf20)) - (l.q s5 at (generic-work-offset saves envmap index-mask)) - (callms GENERIC-VU0-ENVMAP) - (add.i t0 t0 -4) - (l.dr t2 t5) - (add.i t4 t4 48) - (l.dl t2 t5 7) - (add.i t1 a3 16) - (add.i t5 t5 8) - (pextlh t2 r0 t2) - (s.q gp t4 -48) - (and.q a3 t2 s5) - (s.q t6 t4 -32) - (sll.w a3 a3 5) - (s.q ra t4 -16) - (add.w s2 a3 a0) - (mmi-nop!) - (srl32 s3 s2 0) - (l.s f1 s2 24) - (pcpyud gp s2 r0) - (l.s f4 s3 24) - (sub.s f3 f1 f0) - (nop!) - (srl32 s5 gp 0) - (l.s f2 gp 24) - (and.q a3 t2 a1) - (l.s f1 s5 24) - (sra.w s4 a3 8) - (l.q a3 s2 16) - (div.s f3 f0 f3) - (l.q t6 s3 16) - (sub.s f4 f4 f0) - (nop!) - (ppach s1 r0 t9) - (l.q t9 gp 16) - (ppach v0 r0 t8) - (l.q ra s5 16) - (ppach s0 r0 t7) - (l.q t7 s2) - (ppach s2 r0 t3) - (l.q t8 s3) - (pextlw t3 v0 s1) - (l.q gp gp) - (pextlw s3 s2 s0) - (l.q s5 s5) - (mul.s f5 f3 f0) - (s.q t7 at (generic-work-offset saves envmap verts 4)) - (div.s f3 f0 f4) - (s.q t8 at (generic-work-offset saves envmap verts 5)) - (pcpyld t3 s3 t3) - (s.q gp at (generic-work-offset saves envmap verts 6)) - (and.q t3 t3 v1) - (s.q s5 at (generic-work-offset saves envmap verts 7)) - (or.q t3 t3 a2) - (m a2 f5) - (sub.s f2 f2 f0) - (s.q s4 at (generic-work-offset saves envmap kicks 1)) - (nop!) - (s.q t3 t1 -16) - (sub.s f1 f1 f0) - (nop!) - (mul.s f3 f3 f0) - (nop!) - (nop!) - (nop!) - (div.s f2 f0 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (m t3 f3) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f2 f2 f0) - (nop!) - (pextlw a2 t3 a2) - (l.q t3 at (generic-work-offset saves envmap kicks 2)) - (nop!) - (l.q s4 at (generic-work-offset saves envmap verts 9)) - (div.s f1 f0 f1) - (l.q s2 at (generic-work-offset saves envmap verts 11)) - (nop!) - (m s3 f2) - (nop!) - (m.ni vf21 a3) - (nop!) - (m.ni vf22 t6) - (nop!) - (m.ni vf23 t9) - (nop!) - (m.ni vf24 ra) - (nop!) - (l.q t6 at (generic-work-offset saves envmap verts 10)) - (nop!) - (l.q t9 at (generic-work-offset saves envmap verts 8)) - (prot3w a3 s2) - (m s2 f1) - (prot3w ra s4) - (m.ni vf9 t7) - (pextlw t7 s2 s3) - (m.ni vf10 t8) - (pcpyld a2 t7 a2) - (m.ni vf11 gp) - (pextuw t7 ra t9) - (m.ni vf12 s5) - (pcpyld ra t6 ra) - (m.ni vf27 a2) - (pcpyld s5 t7 t9) - (m.ni t8 vf17) - (pextuw a2 a3 t6) - (m.ni t9 vf18) - (pcpyld gp a3 a2) - (m.ni t6 vf19) - (b.le t0 r0 generic-envmap-only-proc-finish-two-tail :delay (m.ni t7 vf20)) - (l.q s4 at (generic-work-offset saves envmap index-mask)) - (callms GENERIC-VU0-ENVMAP) - (add.i t0 t0 -4) - (l.dr t2 t5) - (add.i a3 t4 48) - (l.dl t2 t5 7) - (add.i a2 t1 16) - (add.i t1 t5 8) - (pextlh t2 r0 t2) - (s.q s5 a3 -48) - (and.q t4 t2 s4) - (s.q ra a3 -32) - (sll.w t4 t4 5) - (s.q gp a3 -16) - (add.w s3 t4 a0) - (mmi-nop!) - (srl32 s4 s3 0) - (l.s f1 s3 24) - (pcpyud ra s3 r0) - (l.s f4 s4 24) - (sub.s f3 f1 f0) - (nop!) - (srl32 gp ra 0) - (l.s f2 ra 24) - (and.q t4 t2 a1) - (l.s f1 gp 24) - (sra.w s5 t4 8) - (l.q t4 s3 16) - (div.s f3 f0 f3) - (l.q t5 s4 16) - (sub.s f4 f4 f0) - (nop!) - (ppach s2 r0 t8) - (l.q t8 ra 16) - (ppach s0 r0 t9) - (l.q t9 gp 16) - (ppach s1 r0 t6) - (l.q t6 s3) - (ppach s3 r0 t7) - (l.q t7 s4) - (pextlw s4 s0 s2) - (l.q ra ra) - (pextlw s3 s3 s1) - (l.q gp gp) - (mul.s f5 f3 f0) - (s.q t6 at (generic-work-offset saves envmap verts 8)) - (div.s f3 f0 f4) - (s.q t7 at (generic-work-offset saves envmap verts 9)) - (pcpyld s4 s3 s4) - (s.q ra at (generic-work-offset saves envmap verts 10)) - (and.q s4 s4 v1) - (s.q gp at (generic-work-offset saves envmap verts 11)) - (or.q s4 s4 t3) - (m t3 f5) - (sub.s f2 f2 f0) - (s.q s5 at (generic-work-offset saves envmap kicks 2)) - (nop!) - (s.q s4 a2 -16) - (sub.s f1 f1 f0) - (nop!) - (mul.s f3 f3 f0) - (nop!) - (nop!) - (nop!) - (div.s f2 f0 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (m s5 f3) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f2 f2 f0) - (nop!) - (pextlw s5 s5 t3) - (l.q t3 at (generic-work-offset saves envmap kicks 0)) - (nop!) - (l.q s4 at (generic-work-offset saves envmap verts 1)) - (div.s f1 f0 f1) - (l.q s2 at (generic-work-offset saves envmap verts 3)) - (nop!) - (m s3 f2) - (nop!) - (m.ni vf21 t4) - (nop!) - (m.ni vf22 t5) - (nop!) - (m.ni vf23 t8) - (nop!) - (m.ni vf24 t9) - (nop!) - (l.q t5 at (generic-work-offset saves envmap verts 2)) - (nop!) - (l.q t8 at (generic-work-offset saves envmap verts 0)) - (prot3w t4 s2) - (m s2 f1) - (prot3w t9 s4) - (m.ni vf9 t6) - (pextlw t6 s2 s3) - (m.ni vf10 t7) - (pcpyld t7 t6 s5) - (m.ni vf11 ra) - (pextuw ra t9 t8) - (m.ni vf12 gp) - (pcpyld t6 t5 t9) - (m.ni vf27 t7) - (pcpyld s5 ra t8) - (m.ni t9 vf17) - (pextuw t5 t4 t5) - (m.ni t7 vf18) - (pcpyld ra t4 t5) - (m.ni gp vf19) - (b.gt t0 r0 generic-envmap-only-proc-convert-four-loop :delay (m.ni t8 vf20))) - (asm-block finish-four-tail - (label generic-envmap-only-proc-finish-four-tail) - ;; Complete the group already returned by VU0. The three exits below select the valid lanes of - ;; the partially filled group and retain the same packed stream boundaries as a full group. - (add.i a1 a3 48) - (s.q s5 a1 -48) - (add.i a0 a2 16) - (s.q t6 a1 -32) - (ppach a2 r0 t9) - (s.q ra a1 -16) - (ppach a1 r0 t7) - (mmi-nop!) - (ppach a3 r0 gp) - (mmi-nop!) - (ppach t0 r0 t8) - (mmi-nop!) - (pextlw a1 a1 a2) - (mmi-nop!) - (pextlw a2 t0 a3) - (mmi-nop!) - (pcpyld a1 a2 a1) - (mmi-nop!) - (and.q v1 a1 v1) - (mmi-nop!) - (or.q v1 v1 t3) - (mmi-nop!) - (nop!) - (mmi-nop!) - (b generic-envmap-only-proc-finish :delay (s.q v1 a0 -16))) - (asm-block finish-three-tail - (label generic-envmap-only-proc-finish-three-tail) - (add.i a1 t4 48) - (s.q gp a1 -48) - (add.i a0 a3 16) - (s.q t6 a1 -32) - (ppach a3 r0 t9) - (s.q ra a1 -16) - (ppach a1 r0 t8) - (mmi-nop!) - (ppach t0 r0 t7) - (mmi-nop!) - (ppach t1 r0 t3) - (mmi-nop!) - (pextlw a1 a1 a3) - (mmi-nop!) - (pextlw a3 t1 t0) - (mmi-nop!) - (pcpyld a1 a3 a1) - (mmi-nop!) - (and.q v1 a1 v1) - (mmi-nop!) - (or.q v1 v1 a2) - (mmi-nop!) - (nop!) - (mmi-nop!) - (b generic-envmap-only-proc-finish :delay (s.q v1 a0 -16))) - (asm-block finish-two-tail - (label generic-envmap-only-proc-finish-two-tail) - (add.i a1 t4 48) - (s.q s5 a1 -48) - (add.i a0 t1 16) - (s.q ra a1 -32) - (ppach a2 r0 t8) - (s.q gp a1 -16) - (ppach a1 r0 t9) - (mmi-nop!) - (ppach a3 r0 t6) - (mmi-nop!) - (ppach t0 r0 t7) - (mmi-nop!) - (pextlw a1 a1 a2) - (mmi-nop!) - (pextlw a2 t0 a3) - (mmi-nop!) - (pcpyld a1 a2 a1) - (mmi-nop!) - (and.q v1 a1 v1) - (mmi-nop!) - (or.q v1 v1 t3) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (s.q v1 a0 -16)) - (asm-block finish - (label generic-envmap-only-proc-finish) - (m v0 r0) - (l.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (l.q gp at (generic-work-offset fx-buf work storage2 data 7)) - (l.q s5 at (generic-work-offset fx-buf work storage2 data 6)) - (l.q s4 at (generic-work-offset fx-buf work storage2 data 5)) - (l.q s3 at (generic-work-offset fx-buf work storage2 data 4)) - (l.q s2 at (generic-work-offset fx-buf work storage2 data 3)) - (l.q s1 at (generic-work-offset fx-buf work storage2 data 2)) - (l.q s0 at (generic-work-offset fx-buf work storage2 data 1)) - (jr ra :delay (add.i sp sp 128)) - (nop!) - (nop!)) - ) +(deftype gsf-ik-packed (structure) + ((index uint8) + (no-kick uint8)) + :pack-me) - ;; The same wrapper with the lighting processor swapped for the unlit one, so a model's authored - ;; vertex colors survive to the GS untouched. - (defun generic-no-light ((gsf-buf gsf-buffer) (shaders pointer)) - "Build one Generic pass without lighting using shaders, then start the scratchpad-to-memory DMA." - (declare (asm-func none)) - ;; Install the GSF and shader pointers, prepare one packet, expand source colors without running VU - ;; lighting, and submit the result through the scratchpad DMA channel. - (asm-block build-vertex-packet - (rlet ((from-spr :reg a2 :type dma-bank-spr)) - (label generic-no-light-entry) - (add.i sp sp -16) - (s.d ra at (generic-work-offset fx-buf work storage data 0)) - (lui at #x7000) - (s.w a1 at (generic-work-offset saves ptr-shaders)) - (s.w a0 at (generic-work-offset saves gsf-buf)) - (s.w r0 at (generic-work-offset saves is-envmap)) - (m! t9 generic-prepare-dma-single) - (jalr ra t9 :delay (sll v0 ra 0)) - (m! t9 generic-no-light-proc) - (jalr ra t9 :delay (sll v0 ra 0)) - (l.w v1 at (generic-work-offset saves cur-outbuf)) - (l.w a0 at (generic-work-offset saves qwc)) - (m a3 v1) - (lui at #x7000) - (lui from-spr #x1000) - (l.wu a1 at (generic-work-offset saves basep)) - (ori from-spr from-spr #xd000) - (l.w t1 (-> from-spr chcr)) - (nop!) - (add.i t0 at (generic-work-offset saves from-spr-waits)) - (and.i a3 a3 #x3fff) - (and.i t1 t1 DMA-CHCR-STR) - (nop!) - (b.z t1 generic-no-light-start-dma :delay (nop!)) - (m t1 from-spr) - (nop!))) - ;; The previous packet may still be on its way out of scratchpad. Every busy poll is charged to - ;; saves.from-spr-waits so a frame bound on scratchpad bandwidth says so. - (asm-block wait-for-dma - (label generic-no-light-wait-for-dma) - (l.w t2 t0) - (nop!) - (l.w t3 t1) - (nop!) - (and.i t3 t3 DMA-CHCR-STR) - (add.i t2 t2 1) - (b.nz t3 generic-no-light-wait-for-dma :delay (s.w t2 t0)) - (m t0 r0)) - ;; Program the channel, advance saves.basep by exactly what it will deliver, and switch to the - ;; other output buffer. Nothing waits for this transfer - the wait above is what makes that safe. - (asm-block start-dma - (rlet ((from-spr :reg a2 :type dma-bank-spr)) - (label generic-no-light-start-dma) - (sll t0 a0 4) - (s.w a3 (-> from-spr sadr)) - (nop!) - (s.w a1 (-> from-spr madr)) - (add.i a3 r0 DMA-CHCR-STR) - (s.w a0 (-> from-spr qwc)) - (add a0 a1 t0) - (s.w a3 (-> from-spr chcr)) - (nop!) - (s.w a0 at (generic-work-offset saves basep)) - (m a0 r0) - (xor.i v1 v1 GENERIC-OUTBUF-FLIP) - (s.w v1 at (generic-work-offset saves cur-outbuf)) - (m v0 r0) - (l.d ra at (generic-work-offset fx-buf work storage data 0)) - (jr ra :delay (add.i sp sp 16)) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!))) - ) +(define *generic-effect-debug-stats-data* (new 'global 'generic-effect-debug-stats)) - ;; And the two-pass version: the double builder reserves a second set of stream cursors and a second - ;; packet header, the envmap processor fills them with reflected coordinates and the environment - ;; tint, and the GS lays that pass over the unlit one. This is the shape generic-TIE uses, spelled - ;; out in one place. - (defun generic-no-light+envmap ((gsf-buf gsf-buffer) (shaders pointer)) - "Build both ordinary and environment-map Generic passes without lighting, apply interpolation and - delta attributes using shaders, then start the scratchpad-to-memory DMA." - (declare (asm-func none)) - ;; Prepare the paired base and environment-map packets. The base pass keeps source lighting, while - ;; interpolation, deltas, and reflected texture coordinates are added before the shared scratch output - ;; is submitted. - (asm-block build-two-pass-packet - (rlet ((from-spr :reg a2 :type dma-bank-spr)) - (label generic-no-light+envmap-entry) - (add.i sp sp -16) - (s.d ra at (generic-work-offset fx-buf work storage data 0)) - (lui at #x7000) - (s.w a1 at (generic-work-offset saves ptr-shaders)) - (s.w a0 at (generic-work-offset saves gsf-buf)) - (m! t9 generic-prepare-dma-double) - (jalr ra t9 :delay (sll v0 ra 0)) - (m! t9 generic-envmap-dproc) - (jalr ra t9 :delay (sll v0 ra 0)) - (m! t9 generic-interp-dproc) - (jalr ra t9 :delay (sll v0 ra 0)) - (m! t9 generic-no-light-dproc) - (jalr ra t9 :delay (sll v0 ra 0)) - (l.w v1 at (generic-work-offset saves cur-outbuf)) - (l.w a0 at (generic-work-offset saves qwc)) - (m a3 v1) - (nop!) - (lui at #x7000) - (lui from-spr #x1000) - (l.wu a1 at (generic-work-offset saves basep)) - (ori from-spr from-spr #xd000) - (l.w t1 (-> from-spr chcr)) - (nop!) - (add.i t0 at (generic-work-offset saves from-spr-waits)) - (and.i a3 a3 #x3fff) - (and.i t1 t1 DMA-CHCR-STR) - (nop!) - (b.z t1 generic-no-light+envmap-start-dma :delay (nop!)) - (m t1 from-spr) - (nop!))) - ;; The previous packet may still be on its way out of scratchpad. Every busy poll is charged to - ;; saves.from-spr-waits so a frame bound on scratchpad bandwidth says so. - (asm-block wait-for-dma - (label generic-no-light+envmap-wait-for-dma) - (l.w t2 t0) - (nop!) - (l.w t3 t1) - (nop!) - (and.i t3 t3 DMA-CHCR-STR) - (add.i t2 t2 1) - (b.nz t3 generic-no-light+envmap-wait-for-dma :delay (s.w t2 t0)) - (m t0 r0)) - ;; Program the channel, advance saves.basep by exactly what it will deliver, and switch to the - ;; other output buffer. Nothing waits for this transfer - the wait above is what makes that safe. - (asm-block start-dma - (rlet ((from-spr :reg a2 :type dma-bank-spr)) - (label generic-no-light+envmap-start-dma) - (sll t0 a0 4) - (s.w a3 (-> from-spr sadr)) - (nop!) - (s.w a1 (-> from-spr madr)) - (add.i a3 r0 DMA-CHCR-STR) - (s.w a0 (-> from-spr qwc)) - (add a0 a1 t0) - (s.w a3 (-> from-spr chcr)) - (nop!) - (s.w a0 at (generic-work-offset saves basep)) - (m a0 r0) - (xor.i v1 v1 GENERIC-OUTBUF-FLIP) - (s.w v1 at (generic-work-offset saves cur-outbuf)) - (m v0 r0) - (l.d ra at (generic-work-offset fx-buf work storage data 0)) - (jr ra :delay (add.i sp sp 16)) - (nop!) - (nop!))) - ) +(defun generic-rounded-draw-points ((count int)) + "Round count up to the four-draw-point groups used by Generic's qword-aligned streams." + (declare (inline)) + (logand (+ count 3) -4)) - (defun generic-no-light-dproc () - "Expand GSF vertices without lighting while applying their delta color and texture attributes." - (declare (asm-func none) (allow-saved-regs)) - ;; The unlit processor the shipped game installs, and the one that fans generic-TIE's whole packet out - ;; to draw points. It emits five streams from each vertex record, and they are five separate streams - ;; rather than anything combined: - ;; - ;; ptr-vtxs the position, three quadwords per four vertices - ;; ptr-clrs clr, the vertex's own color - ;; ptr-texs tex plus consts.texture-offset, with the kick flag in bit 0 of S - ;; ptr-env-clrs dclr, which generic-envmap-dproc filled with the environment tint - ;; ptr-env-texs dtex, the reflected coordinate, likewise, with the kick flag added - ;; - ;; Normals are not copied - nothing downstream wants them, which is also what lets envmap-dproc - ;; overwrite two of their lanes. So this is not "no lighting" in the sense of no shading: it is the - ;; processor that publishes whatever the passes before it deposited in the vertex records, which for - ;; TIE means the time-of-day palette color, the subdivision blend, and the reflection. - ;; - ;; If it were wrong the symptom would be right geometry with wrong colors - flat, or carrying the - ;; previous object's palette, or a reflection that does not line up with the base pass. - (asm-block load-stream-cursors - (label generic-no-light-dproc-entry) - (add.i sp sp -128) - (s.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (s.q s0 at (generic-work-offset fx-buf work storage2 data 1)) - (s.q s1 at (generic-work-offset fx-buf work storage2 data 2)) - (s.q s2 at (generic-work-offset fx-buf work storage2 data 3)) - (s.q s3 at (generic-work-offset fx-buf work storage2 data 4)) - (s.q s4 at (generic-work-offset fx-buf work storage2 data 5)) - (s.q s5 at (generic-work-offset fx-buf work storage2 data 6)) - (s.q gp at (generic-work-offset fx-buf work storage2 data 7)) - (lui at #x7000) - (nop!) - (rlet ((gsf-buf :reg a1 :type gsf-buffer)) - (l.w gsf-buf at (generic-work-offset saves gsf-buf)) - (nop!) - (l.w a0 at (generic-work-offset saves num-dps)) - (nop!) - (l.w v1 (-> gsf-buf info ptr-iks)) - (nop!) - (l.w a2 (-> gsf-buf info ptr-verts))) - (add.i a0 a0 3) - (nop!) - (sra a0 a0 2) - (nop!) - (sll a0 a0 3) - (add.i a3 r0 255) - (lui a1 -2) - (add.i t1 r0 256) ;; not a DMA value: 255 and 256 broadcast into the index and kick-bit masks - (ori a1 a1 #xfffe) - (add a0 v1 a0) - (pextlw a1 a1 a1) - (l.w t2 at (generic-work-offset saves ptr-vtxs)) - (pextlw a1 a1 a1) - (l.w t3 at (generic-work-offset saves ptr-clrs)) - (pextlw a2 a2 a2) - (l.w t5 at (generic-work-offset saves ptr-texs)) - (pextlw a2 a2 a2) - (l.w t4 at (generic-work-offset saves ptr-env-clrs)) - (pcpyh a3 a3) - (l.w t6 at (generic-work-offset saves ptr-env-texs)) - (pcpyld a3 a3 a3) - (l.q t0 at (generic-work-offset fx-buf work consts texture-offset)) - (pcpyh t1 t1) - (l.d t7 v1) - (pcpyld t1 t1 t1) - (mmi-nop!) - (pextlh t8 r0 t7) - (mmi-nop!) - (and.q t7 t8 a3) - (mmi-nop!) - (sll.w t7 t7 5) - (mmi-nop!) - (add.i t2 t2 -48) - (add.i t3 t3 -16) - (add.i t4 t4 -16) - (add.i t5 t5 -16) - (b generic-no-light-dproc-prime-vertex-loop :delay (add.i t6 t6 -16)) - (nop!)) - (asm-block vertex-loop - (label generic-no-light-dproc-vertex-loop) - ;; Store the preceding group while decoding the next four packed vertex references. - (pextlh t8 r0 gp) - (s.q t7 t2) - (and.q t7 t8 a3) - (s.q t9 t2 16) - (sll.w t7 t7 5) - (s.q ra t2 32)) - ;; The loop is rotated: this is the body, entered directly from the setup above, and the short block - ;; before it stores the group this pass gathered. So the stores at the top of a pass belong to the - ;; previous four vertices, and the block above is reached once more than this one. - (asm-block prime-vertex-loop - (label generic-no-light-dproc-prime-vertex-loop) - ;; Gather source position, normal, color, and texture data, then add the signed delta fields - ;; carried in the indexed stream before writing the expanded records. - (add.w s3 t7 a2) - (mmi-nop!) - (srl32 s2 s3 0) - (add.i t2 t2 48) - (pcpyud s5 s3 r0) - (l.q t7 s3) - (srl32 s4 s5 0) - (add.i t3 t3 16) - (and.q t8 t8 t1) - (l.q t9 s2) - (sra.w gp t8 8) - (l.q t8 s5) - (pextuw s1 t9 t7) - (l.q ra s4) - (add.i t5 t5 16) - (add.i v1 v1 8) - (add.i t4 t4 16) - (add.i t6 t6 16) - (pextuw s0 ra t8) - (l.q s3 s3 16) - (pcpyud s1 s1 s0) - (l.q s2 s2 16) - (add.h s0 s1 t0) - (l.q s1 s5 16) - (and.q s5 s0 a1) - (l.q s0 s4 16) - (pextlw s4 s2 s3) - (mmi-nop!) - (pextuw s3 s2 s3) - (mmi-nop!) - (pextlw s2 s0 s1) - (mmi-nop!) - (pextuw s0 s0 s1) - (mmi-nop!) - (pcpyld s1 s2 s4) - (mmi-nop!) - (pcpyud s4 s4 s2) - (mmi-nop!) - (pcpyud s3 s3 s0) - (s.q s4 t4) - (and.q s4 s1 a1) - (s.q s3 t3) - (or.q s4 s4 gp) - (mmi-nop!) - (or.q gp s5 gp) - (s.q s4 t6) - (prot3w ra ra) - (s.q gp t5) - (prot3w t9 t9) - (mmi-nop!) - (pextuw s5 t9 t7) - (mmi-nop!) - (pcpyld t9 t8 t9) - (l.d gp v1) - (pcpyld t7 s5 t7) - (mmi-nop!) - (pextuw t8 ra t8) - (mmi-nop!) - (b.ne v1 a0 generic-no-light-dproc-vertex-loop :delay (pcpyld ra ra t8)) - (nop!) - (s.q t7 t2) - (nop!) - (s.q t9 t2 16) - (nop!) - (s.q ra t2 32) - (m v0 r0) - (l.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (l.q gp at (generic-work-offset fx-buf work storage2 data 7)) - (l.q s5 at (generic-work-offset fx-buf work storage2 data 6)) - (l.q s4 at (generic-work-offset fx-buf work storage2 data 5)) - (l.q s3 at (generic-work-offset fx-buf work storage2 data 4)) - (l.q s2 at (generic-work-offset fx-buf work storage2 data 3)) - (l.q s1 at (generic-work-offset fx-buf work storage2 data 2)) - (l.q s0 at (generic-work-offset fx-buf work storage2 data 1)) - (jr ra :delay (add.i sp sp 128)) - (nop!) - (nop!) - (nop!)) - ) +(defun generic-next-inbuf ((address int)) + "Return the VU1 input bank after address, wrapping across Generic's three-bank ring." + (declare (inline)) + (if (= address GENERIC-VU1-INBUF-LAST) GENERIC-VU1-INBUF-FIRST (+ address GENERIC-VU1-INBUF-STEP))) - (defun generic-no-light-dproc-only () - "Apply delta color and texture data to expanded GSF vertices without producing the ordinary position - and normal streams." - (declare (asm-func none) (allow-saved-regs)) - ;; The same color and coordinate work with the position and normal copies left out, and reading - ;; the *environment* stream cursors rather than the base ones. It is the second pass of a two-pass - ;; draw: the geometry has already been written by the first pass and only needs a second set of - ;; attributes laid over it, so copying the positions again would be pure cost. - ;; - ;; Nothing installs it in Jak 1. Where the same job is needed, generic-envmap-dproc does it and - ;; computes the coordinates as well. - (asm-block load-stream-cursors - (label generic-no-light-dproc-only-entry) - (add.i sp sp -32) - (s.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (s.q gp at (generic-work-offset fx-buf work storage2 data 1)) - (lui at #x7000) - (rlet ((gsf-buf :reg a0 :type gsf-buffer)) - (l.w gsf-buf at (generic-work-offset saves gsf-buf)) - (nop!) - (l.w a1 at (generic-work-offset saves num-dps)) - (nop!) - (l.w v1 (-> gsf-buf info ptr-iks)) - (add.i a1 a1 3) - (l.w a2 (-> gsf-buf info ptr-verts))) - (nop!) - (sra a0 a1 2) - (sll a0 a0 3) - (add.i a3 r0 255) - (lui a1 -2) - (add.i t1 r0 256) ;; not a DMA value: 255 and 256 broadcast into the index and kick-bit masks - (ori a1 a1 #xfffe) - (add a0 v1 a0) - (pextlw a1 a1 a1) - (mmi-nop!) - (pextlw a1 a1 a1) - (mmi-nop!) - (pextlw a2 a2 a2) - (mmi-nop!) - (pextlw a2 a2 a2) - (l.w t2 at (generic-work-offset saves ptr-env-clrs)) - (pcpyh a3 a3) - (l.w t3 at (generic-work-offset saves ptr-env-texs)) - (pcpyld a3 a3 a3) - (l.q t0 at (generic-work-offset fx-buf work consts texture-offset)) - (pcpyh t1 t1) - (l.d t4 v1) - (pcpyld t1 t1 t1) - (mmi-nop!) - (pextlh t5 r0 t4) - (mmi-nop!) - (and.q t4 t5 a3) - (mmi-nop!) - (sll.w t4 t4 5) - (mmi-nop!) - (nop!) - (add.i t2 t2 -16) - (b generic-no-light-dproc-only-prime-vertex-loop :delay (add.i t3 t3 -16))) - (asm-block vertex-loop - (label generic-no-light-dproc-only-vertex-loop) - ;; This shortened path drains only the delta color and texture pair from the preceding group. - (pextlh t5 r0 t5) - (s.d t7 t2) - (and.q t6 t5 a3) - (s.d t4 t2 8) - (sll.w t4 t6 5) - (mmi-nop!)) - ;; The loop is rotated: this is the body, entered directly from the setup above, and the short block - ;; before it stores the group this pass gathered. So the stores at the top of a pass belong to the - ;; previous four vertices, and the block above is reached once more than this one. - (asm-block prime-vertex-loop - (label generic-no-light-dproc-only-prime-vertex-loop) - ;; Resolve four source records and merge their delta fields; position and normal cursors are - ;; deliberately untouched. - (add.w t8 t4 a2) - (mmi-nop!) - (srl32 t7 t8 0) - (mmi-nop!) - (pcpyud t6 t8 r0) - (l.wu t9 t8 16) - (srl32 t4 t6 0) - (add.i t2 t2 16) - (and.q t5 t5 t1) - (l.wu gp t7 16) - (sra.w t5 t5 8) - (l.wu ra t6 16) - (pextlw t9 gp t9) - (l.wu gp t4 16) - (add.i t3 t3 16) - (add.i v1 v1 8) - (pextlw ra gp ra) - (l.wu t8 t8 20) - (pcpyld t9 ra t9) - (l.wu t7 t7 20) - (add.h t9 t9 t0) - (l.wu t6 t6 20) - (and.q t9 t9 a1) - (l.wu t4 t4 20) - (or.q t9 t9 t5) - (l.d t5 v1) - (pextlw t7 t7 t8) - (s.q t9 t3) - (b.ne v1 a0 generic-no-light-dproc-only-vertex-loop :delay (pextlw t4 t4 t6)) - (nop!) - (s.d t7 t2) - (nop!) - (s.d t4 t2 8) - (m v0 r0) - (l.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (l.q gp at (generic-work-offset fx-buf work storage2 data 1)) - (jr ra :delay (add.i sp sp 32)) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!)) - ) +(defun generic-copy-effect-shader! ((dst adgif-shader) (src adgif-shader) (kick-offset int) (vertex-count int)) + "Copy src's five A+D quadwords to dst and patch the strip's VU kick offset and vertex count." + (declare (inline)) + (quad-copy! (the-as pointer dst) (the-as pointer src) 5) + (set! (-> dst quad 0 word 3) (the-as uint kick-offset)) + (set! (-> dst quad 1 word 3) (the-as uint vertex-count)) + (none)) - (defun generic-no-light-proc () - "Expand indexed GSF vertices without lighting, preserving source colors while reconstructing the - position, normal, and texture streams." - (declare (asm-func none) (allow-saved-regs)) - ;; Fills all three streams and passes the authored colors through untouched - no lighting, and no - ;; draw-point attribute lanes either. The difference from generic-no-light-dproc is exactly that: - ;; this one ignores dclr and dtex, so a model drawn with it looks the same whatever the earlier - ;; passes computed. - ;; - ;; Nothing installs it. It is the base case the two dproc variants are built from. - (asm-block load-stream-cursors - (label generic-no-light-proc-entry) - (add.i sp sp -96) - (s.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (s.q s2 at (generic-work-offset fx-buf work storage2 data 1)) - (s.q s3 at (generic-work-offset fx-buf work storage2 data 2)) - (s.q s4 at (generic-work-offset fx-buf work storage2 data 3)) - (s.q s5 at (generic-work-offset fx-buf work storage2 data 4)) - (s.q gp at (generic-work-offset fx-buf work storage2 data 5)) - (lui at #x7000) - (nop!) - (nop!) - (rlet ((gsf-buf :reg a0 :type gsf-buffer)) - (l.w gsf-buf at (generic-work-offset saves gsf-buf)) - (nop!) - (l.w a1 at (generic-work-offset saves num-dps)) - (nop!) - (l.w v1 (-> gsf-buf info ptr-iks)) - (add.i a1 a1 3) - (l.w a2 (-> gsf-buf info ptr-verts))) - (sra a0 a1 2) - (nop!) - (sll a0 a0 3) - (add.i a3 r0 255) - (lui a1 -2) - (add.i t1 r0 256) ;; not a DMA value: 255 and 256 broadcast into the index and kick-bit masks - (ori a1 a1 #xfffe) - (add a0 v1 a0) - (pextlw a1 a1 a1) - (mmi-nop!) - (pextlw a1 a1 a1) - (mmi-nop!) - (pextlw a2 a2 a2) - (l.w t2 at (generic-work-offset saves ptr-vtxs)) - (pextlw a2 a2 a2) - (l.w t3 at (generic-work-offset saves ptr-clrs)) - (pcpyh a3 a3) - (l.w t4 at (generic-work-offset saves ptr-texs)) - (pcpyld a3 a3 a3) - (l.q t0 at (generic-work-offset fx-buf work consts texture-offset)) - (pcpyh t1 t1) - (l.d t5 v1) - (pcpyld t1 t1 t1) - (mmi-nop!) - (pextlh t6 r0 t5) - (mmi-nop!) - (and.q t5 t6 a3) - (mmi-nop!) - (sll.w t5 t5 5) - (mmi-nop!) - (add.i t2 t2 -48) - (add.i t3 t3 -16) - (b generic-no-light-proc-prime-vertex-loop :delay (add.i t4 t4 -16))) - (asm-block vertex-loop - (label generic-no-light-proc-vertex-loop) - ;; Store the previous four expanded vertices as the next group is gathered. - (pextlh t6 r0 ra) - (s.q t5 t2) - (and.q t5 t6 a3) - (s.q t7 t2 16) - (sll.w t5 t5 5) - (s.q t8 t2 32)) - ;; The loop is rotated: this is the body, entered directly from the setup above, and the short block - ;; before it stores the group this pass gathered. So the stores at the top of a pass belong to the - ;; previous four vertices, and the block above is reached once more than this one. - (asm-block prime-vertex-loop - (label generic-no-light-proc-prime-vertex-loop) - ;; Strip the packed index/control bits, gather four source records, preserve their colors, and - ;; rebuild the 48-byte position/normal stream plus the 16-byte texture stream. - (add.w gp t5 a2) - (mmi-nop!) - (srl32 s4 gp 0) - (add.i t2 t2 48) - (pcpyud s5 gp r0) - (l.q t5 gp) - (srl32 t8 s5 0) - (add.i t3 t3 16) - (and.q t6 t6 t1) - (l.q t7 s4) - (sra.w ra t6 8) - (l.q t6 s5) - (pextuw s3 t7 t5) - (l.q t9 t8) - (add.i t4 t4 16) - (add.i v1 v1 8) - (pextuw s2 t9 t6) - (l.w gp gp 28) - (pcpyud s3 s3 s2) - (l.w s4 s4 28) - (add.h s3 s3 t0) - (l.w s5 s5 28) - (and.q s3 s3 a1) - (l.w t8 t8 28) - (or.q ra s3 ra) - (s.w gp t3) - (prot3w t9 t9) - (s.q ra t4) - (prot3w t7 t7) - (s.w s4 t3 4) - (pextuw gp t7 t5) - (s.w s5 t3 8) - (pcpyld t7 t6 t7) - (l.d ra v1) - (pcpyld t5 gp t5) - (mmi-nop!) - (pextuw t6 t9 t6) - (s.w t8 t3 12) - (b.ne v1 a0 generic-no-light-proc-vertex-loop :delay (pcpyld t8 t9 t6)) - (nop!) - (s.q t5 t2) - (nop!) - (s.q t7 t2 16) - (nop!) - (s.q t8 t2 32) - (m v0 r0) - (l.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (l.q gp at (generic-work-offset fx-buf work storage2 data 5)) - (l.q s5 at (generic-work-offset fx-buf work storage2 data 4)) - (l.q s4 at (generic-work-offset fx-buf work storage2 data 3)) - (l.q s3 at (generic-work-offset fx-buf work storage2 data 2)) - (l.q s2 at (generic-work-offset fx-buf work storage2 data 1)) - (jr ra :delay (add.i sp sp 96)) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!)) - ) +(defun generic-write-stream-tag! ((stream uint32) (vif0-tag uint) (vif1-tag uint)) + "Clear the 16-byte prefix immediately before stream and install its two trailing VIF commands." + (declare (inline)) + (let ((tag (the-as dma-packet (- stream 16)))) + (set! (-> tag quad) (the-as uint128 0)) + (set! (-> tag vif0) (the-as vif-tag vif0-tag)) + (set! (-> tag vif1) (the-as vif-tag vif1-tag))) + (none)) - (defun generic-interp-dproc () - "Apply the pending Generic interpolation job to a range of expanded vertex delta attributes." - (declare (asm-func none)) - ;; The level-of-detail blend for the *environment coordinate*, and the only processor driven by a job - ;; record rather than by the whole vertex list. saves.ptr-interp-job names a range - num vertices - ;; starting at first - and a second source; for each vertex in that range the dtex that - ;; generic-envmap-dproc computed becomes its own weighted by morph-z plus the source's weighted by - ;; morph-w. Only byte 16 of each record is read and written; colors are not touched here, because - ;; TIE's color blend was already done by the converter when it looked the palette up. - ;; - ;; Eight halfword lanes is four (S, T) pairs, so one pmulth and one pmaddh cover four vertices at a - ;; time. The two weights sum to 256 rather than to 1.0, so what follows them is a shift, not a divide. - ;; - ;; Three early exits, all normal rather than error paths: no job at all, a job-type this build does - ;; not implement, or a weight already at zero because the object sits at the near edge of its LOD band - ;; and nothing is collapsing. Without this pass a subdividing TIE surface would pop its reflection - ;; every time it crossed an LOD boundary - the geometry eases across, and the reflection has to ease - ;; with it. - (asm-block read-interpolation-job - (label generic-interp-dproc-entry) - (lui at #x7000) - (nop!) - (l.w v1 at (generic-work-offset saves ptr-interp-job)) - (nop!) - (rlet ((gsf-buf :reg a0 :type gsf-buffer)) - (l.w gsf-buf at (generic-work-offset saves gsf-buf)) - (b.z v1 generic-interp-dproc-finish :delay (nop!)) - (nop!) - (l.w t0 v1 8) - (nop!) - (l.w a2 (-> gsf-buf info ptr-verts))) - (nop!) - (l.h a0 v1) - (nop!) - (l.h a1 v1 2) - (b.nz a0 generic-interp-dproc-finish :delay (l.h a0 v1 4)) - (sll t1 a0 5) - (l.h a0 v1 12) - (add.i a3 a1 7) - (l.h a1 v1 14) - (b.z a1 generic-interp-dproc-finish :delay (add v1 t1 a2)) - (pextlh a0 a0 a0) - (mmi-nop!) - (pextlw a0 a0 a0) - (mmi-nop!) - (pcpyld a0 a0 a0) - (mmi-nop!) - (pextlh a1 a1 a1) - (mmi-nop!) - (pextlw a1 a1 a1) - (mmi-nop!) - (pcpyld a1 a1 a1) - (mmi-nop!) - (pextlw a2 a2 a2) - (mmi-nop!) - (pcpyld a2 a2 a2) - (mmi-nop!) - (sra a3 a3 3) - (nop!) - (sll a3 a3 4) - (l.d t1 t0) - (add a3 t0 a3) - (add.i t0 t0 8) - (pextlb t1 r0 t1) - (mmi-nop!) - (sll.h t2 t1 5) - (mmi-nop!) - (pextuh t1 r0 t2) - (mmi-nop!) - (pextlh t2 r0 t2) - (mmi-nop!) - (add.w t2 t2 a2) - (mmi-nop!) - (b generic-interp-dproc-prime-vertex-loop :delay (pcpyud t5 t2 r0))) - (asm-block vertex-loop - (label generic-interp-dproc-vertex-loop) - ;; Four records are in flight. Rotate the packed products, write one word into each 32-byte - ;; attribute record, and advance to the next eight source indices. - (srl32 t5 t6 0) - (srl32 t4 t3 0) - (pextuh t1 r0 t2) - (s.w t6 v1 16) - (pextlh t2 r0 t2) - (s.w t5 v1 48) - (add.w t2 t2 a2) - (s.w t3 v1 80) - (pcpyud t5 t2 r0) - (s.w t4 v1 112) - (add.i t0 t0 8) - (add.i v1 v1 128)) - ;; The loop is rotated: this is the body, entered directly from the setup above, and the short block - ;; before it stores the group this pass gathered. So the stores at the top of a pass belong to the - ;; previous four vertices, and the block above is reached once more than this one. - (asm-block prime-vertex-loop - (label generic-interp-dproc-prime-vertex-loop) - ;; Multiply the two signed source attributes by the complementary morph weights, add the - ;; products through the multimedia accumulator, and gather the four low words for storage. - (add.w t1 t1 a2) - (l.wu t3 t2 16) - (pcpyud t6 t1 r0) - (l.wu t4 t5 16) - (srl32 t8 t2 0) - (l.wu t2 t1 16) - (srl32 t9 t5 0) - (l.wu t5 t6 16) - (srl32 t7 t1 0) - (l.wu t1 t8 16) - (srl32 t8 t6 0) - (l.wu t6 t9 16) - (pextlw t4 t4 t3) - (l.wu t3 t7 16) - (pextlw t2 t5 t2) - (l.wu t5 t8 16) - (pcpyld t2 t2 t4) - (l.wu t4 v1 16) - (pextlw t6 t6 t1) - (l.wu t1 v1 48) - (pextlw t3 t5 t3) - (mmi-nop!) - (pcpyld t3 t3 t6) - (l.wu t5 v1 80) - (pextlw t1 t1 t4) - (l.wu t4 v1 112) - (pmulth r0 t2 a1) - (mmi-nop!) - (pextlw t2 t4 t5) - (mmi-nop!) - (pmaddh r0 t3 a1) - (mmi-nop!) - (pcpyld t1 t2 t1) - (l.d t2 t0) - (pmaddh r0 t1 a0) - (mmi-nop!) - (pextlb t1 r0 t2) - (mmi-nop!) - (sll.h t2 t1 5) - (mmi-nop!) - (pmfhl.lw t3) - (mmi-nop!) - (pmfhl.uw t1) - (mmi-nop!) - (sra.w t3 t3 8) - (mmi-nop!) - (sra.w t1 t1 8) - (mmi-nop!) - (pinteh t6 t1 t3) - (mmi-nop!) - (b.ne t0 a3 generic-interp-dproc-vertex-loop :delay (pcpyud t3 t6 r0)) - (srl32 a0 t6 0) - (s.w t6 v1 16) - (srl32 a1 t3 0) - (s.w a0 v1 48) - (nop!) - (s.w t3 v1 80) - (nop!) - (s.w a1 v1 112)) - ;; Nothing to blend: no job, a job type this build does not implement, or a weight already at zero. - (asm-block finish - (label generic-interp-dproc-finish) - (jr ra :delay (add sp sp r0)) - (nop!) - (nop!) - (nop!) - (nop!)) - ) +;; Build one scratchpad-resident Generic fragment from the GSF header and the shader pointers in +;; generic-saves. is-envmap selects the base or environment GIF header and shader source. A zero +;; ptr-shaders means an earlier pass already prepared the header and shader list in this output bank; +;; this call preserves them while rebuilding the outer tag, vertex-stream prefixes, and launch tail. +;; +;; The output payloads are V3-32 positions (12 bytes per draw point), V4-8 colors (4 bytes), and V2-16 +;; texture coordinates (4 bytes). Their VIF NUM fields and storage spans use the four-point padded +;; count. On return, saves holds the three writable scratchpad cursors, the flat fromSPR size, and the +;; next VU1 input/header destinations; a processor still has to fill the payloads before copy-out. +(defun generic-prepare-dma-single-new () + "Lay out one Generic VIF fragment in cur-outbuf and publish its three vertex-payload cursors." + (let* ((work (-> (scratchpad-object terrain-context) work foreground generic-work)) + (saves (-> work saves)) + (consts (-> work fx-buf work consts)) + (gsf (-> saves gsf-buf)) + (packet (the-as generic-texbuf (-> saves cur-outbuf))) + (out (the-as pointer packet)) + (out-address (the-as uint32 packet)) + (header (-> packet header)) + (num-strips (the-as int (-> gsf header num-strips))) + (num-dps (the-as int (-> gsf header num-dps))) + (padded-dps (generic-rounded-draw-points num-dps)) + (shader-end (&+ out (+ 128 (* num-strips 80)))) + (position-bytes (* padded-dps 12)) + (attribute-bytes (* padded-dps 4)) + (position-stream (the-as uint32 (&+ shader-end 16))) + (color-stream (the-as uint32 (&+ shader-end (+ 48 position-bytes)))) + (texture-stream (the-as uint32 (&+ shader-end (+ 64 position-bytes attribute-bytes)))) + (control (the-as uint32 (&+ shader-end (+ 64 position-bytes (* attribute-bytes 2))))) + (shaders (the-as (inline-array adgif-shader) (&+ out 128))) + (control-tags (the-as (pointer vif-tag) control)) + (inbuf (the-as int (-> saves inbuf-adr))) + (header-qwc (+ 7 (* num-strips 5)))) + ;; Always refresh the outer CNT/VIF prefix. Refresh the VU header and shaders only when the caller + ;; supplied a new base shader list; the zero-pointer path deliberately reuses the bank's contents. + (set! (-> packet tag quad) (-> consts dma-header quad)) + (when (nonzero? (-> saves ptr-shaders)) + (matrix-copy! (-> header matrix) (-> consts matrix)) + (cond + ((nonzero? (-> saves is-envmap)) + (set! (-> header strgif qword quad) (-> consts envmap strgif qword quad)) + (set! (-> header adnop1 quad) (-> consts adcmds 3 quad)) + (set! (-> header adnop2 quad) (-> consts adcmds 3 quad))) + (else + (set! (-> header strgif qword quad) (-> consts base-strgif qword quad)) + (set! (-> header adnop1 quad) (-> consts adcmds 0 quad)) + (set! (-> header adnop2 quad) (-> consts adcmds 3 quad)))) - (defun generic-envmap-dproc () - "Generate environment-map texture deltas for the expanded GSF vertices using the current transform - and Generic VU0 entry 48." - (declare (asm-func none)) - ;; The environment-map processor generic-TIE installs, and a genuinely different operation from - ;; generic-envmap-proc rather than the same one with a flag. It writes no VIF stream at all: it walks - ;; the vertex array *linearly*, gsf-header.num-vtxs of them four at a time, and deposits the reflected - ;; coordinate into each vertex's own dtex and the environment tint into its dclr. It never looks at - ;; the index list and never sees num-dps. - ;; - ;; Three reasons it has to work that way. The coordinate has to exist as a per-vertex record before - ;; generic-interp-dproc runs, because that is what interp-dproc blends - a stream cannot be - ;; interpolated, and without the blend a subdividing TIE surface would pop its reflection at every - ;; LOD boundary. TIE fragments share vertices heavily, so num-vtxs is well below num-dps and doing - ;; the VU0 work per vertex is strictly less work; generic-no-light-dproc fans the result out to draw - ;; points afterwards for nothing. And it is a leaf - no saved registers, no stack - because all of - ;; its state fits in the temporaries. - ;; - ;; It destroys the normal it is reading: dtex and dclr are bytes 16 to 23, which are nrm.x and nrm.y. - ;; That is safe only because the write cursor trails the load cursor by exactly three groups, nrm.z - ;; at byte 24 is never written, and nothing downstream reads normals. It also writes up to three - ;; vertices past the end of the array and reads up to three groups past it, which is part of what the - ;; GSF buffer's spare kilobyte is for. - ;; The environment map is a normal-based sphere map, and the EE half is unreadable without knowing - ;; the VU0 half. consts.matrix is the projection matrix alone, so pos and nrm in a gsf-vertex are - ;; already in camera space with +z forward. Writing p for the position, n for the normal and - ;; k = (0, 0, 1) for the camera axis, entry 48 computes - ;; - ;; m = n - k - ;; r = p + m * (m.p) / m.z - ;; ST = 0.5 * r.xy / |r| + 0.5 - ;; - ;; which is the reflection of the eye vector in the plane perpendicular to m: the one that swaps the - ;; camera axis for the normal. It comes out that cheaply because |n| = 1 makes |m|^2 = -2 m.z - ;; exactly, so the 2/|m|^2 a Householder reflection needs is already sitting in m.z and the only - ;; division in the whole thing is the EE-side 1/(n.z - 1) that the FPU chain below computes, four per - ;; group, crossed in together as vf27. - ;; - ;; Two things follow. Unit-length normals are a hard precondition - nothing normalizes n. And for a - ;; vertex on the optical axis the answer is just n, so what lands at the centre of the texture is a - ;; *normal* facing the camera, not a reflection; a true mirror lookup would swing at twice the rate. - ;; The 0.5 scale and 0.5 bias are consts.envmap.consts z and w, and the shader clamps in both - ;; directions, which is what keeps the rim of the sphere from wrapping. n.z = +1 would divide by - ;; zero, but that is a normal pointing directly away from the camera, and front-facing geometry has - ;; n.z in [-1, 0). - ;; - ;; ftoi12 is a transport format rather than a GS coordinate: VU1 converts it back to float before - ;; emitting ST, so the resolution is 1/4096 - 1/2048 in S once its low bit is taken for the kick flag. - ;; - ;; A wrong result is unmistakable. Sign flip on the reciprocal and the reflection slides the wrong - ;; way, mirrored through the texture centre; non-unit normals and it drifts as the model scales; - ;; missing bias and half of every object collapses to one clamped edge color; missing normalize and - ;; distant geometry smears radially and pins at the edge. - ;; - ;; Four vertices per pass, and entry 48 runs one group behind the caller like the lighting entry does - ;; - but structurally rather than by copying its outputs aside. Its last two pairs start the fourth - ;; vertex and the first thirteen pairs of the *next* call finish it, with vf05-vf08, vf29 and vf30 - ;; carrying the pipeline across the call boundary. So the first call's output is garbage, which is - ;; why every caller calls entry 48 twice before reading anything, and why no other VU0 entry may be - ;; used between two of these calls. - (asm-block load-stream-cursors - (label generic-envmap-dproc-entry) - (nop!) - (lui at #x7000) - (lui v1 #x3f80) - (m f0 v1) - (rlet ((gsf-buf :reg a1 :type gsf-buffer)) - (l.wu gsf-buf at (generic-work-offset saves gsf-buf)) - (nop!) - (l.w v1 at (generic-work-offset fx-buf work consts envmap colors)) - (nop!) - (l.w a2 (-> gsf-buf info ptr-verts))) - (m a0 a2) - (l.hu a1 a1 20) - (nop!) - (l.s f4 a2 24) - (add.i a1 a1 -4) - (l.s f3 a2 56) - (nop!) - (l.s f2 a2 88) - (nop!) - (l.s f1 a2 120) - (nop!) - (l.q t2 a2 16) - (sub.s f4 f4 f0) - (l.q t3 a2 48) - (div.s f4 f0 f4) - (l.q t4 a2 80) - (nop!) - (l.q t5 a2 112) - (nop!) - (l.vf vf31 at (generic-work-offset fx-buf work consts envmap consts)) - (nop!) - (l.q t6 a2) - (nop!) - (l.q a3 a2 32) - (nop!) - (l.q t0 a2 64) - (nop!) - (l.q t1 a2 96) - (mul.s f4 f4 f0) - (m vf21 t2) - (sub.s f3 f3 f0) - (m.ni vf22 t3) - (div.s f3 f0 f3) - (m.ni vf23 t4) - (nop!) - (m.ni vf24 t5) - (nop!) - (m.ni vf9 t6) - (sub.s f2 f2 f0) - (m t2 f4) - (sub.s f1 f1 f0) - (m.ni vf10 a3) - (nop!) - (m.ni vf11 t0) - (nop!) - (m.ni vf12 t1) - (mul.s f3 f3 f0) - (nop!) - (div.s f2 f0 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m a3 f3) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f2 f2 f0) - (nop!) - (div.s f1 f0 f1) - (nop!) - (pextlw a3 a3 t2) - (mmi-nop!) - (nop!) - (nop!) - (nop!) - (m t0 f2) - (nop!) - (nop!) - (nop!) - (add.i a2 a2 128) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m t1 f1) - (pextlw t0 t1 t0) - (mmi-nop!) - (pcpyld a3 t0 a3) - (mmi-nop!) - (nop!) - (m.ni vf27 a3) - (nop!) - (callms GENERIC-VU0-ENVMAP) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (l.s f4 a2 24) - (nop!) - (l.s f3 a2 56) - (nop!) - (l.s f2 a2 88) - (nop!) - (l.s f1 a2 120) - (nop!) - (l.q a3 a2 16) - (sub.s f4 f4 f0) - (nop!) - (sub.s f3 f3 f0) - (nop!) - (sub.s f2 f2 f0) - (nop!) - (sub.s f1 f1 f0) - (nop!) - (div.s f4 f0 f4) - (l.q t0 a2 48) - (nop!) - (l.q t1 a2 80) - (nop!) - (l.q t2 a2 112) - (nop!) - (l.q t3 a2) - (nop!) - (l.q t4 a2 32) - (nop!) - (l.q t5 a2 64) - (nop!) - (l.q t6 a2 96) - (mul.s f4 f4 f0) - (nop!) - (div.s f3 f0 f3) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m t7 f4) - (nop!) - (add.i a2 a2 128) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f3 f3 f0) - (nop!) - (div.s f2 f0 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m t8 f3) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f2 f2 f0) - (nop!) - (div.s f1 f0 f1) - (nop!) - (pextlw t7 t8 t7) - (nop!) - (nop!) - (m t8 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m t9 f1) - (nop!) - (nop!) - (nop!) - (m.ni vf21 a3) - (nop!) - (m.ni vf9 t3) - (nop!) - (m.ni vf10 t4) - (nop!) - (m.ni vf11 t5) - (nop!) - (m.ni vf12 t6) - (pextlw a3 t9 t8) - (nop!) - (pcpyld a3 a3 t7) - (nop!) - (nop!) - (m.ni vf22 t0) - (nop!) - (m.ni vf23 t1) - (nop!) - (m.ni vf24 t2) - (nop!) - (m.ni vf27 a3) - (nop!) - (callms GENERIC-VU0-ENVMAP) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (l.s f1 a2 24) - (sub.s f1 f1 f0) - (l.s f2 a2 56) - (div.s f3 f0 f1) - (l.s f5 a2 88) - (nop!) - (l.s f1 a2 120) - (nop!) - (l.q a3 a2 16) - (nop!) - (nop!) - (sub.s f4 f2 f0) - (nop!) - (sub.s f2 f5 f0) - (nop!) - (sub.s f1 f1 f0) - (nop!) - (mul.s f3 f3 f0) - (l.q t0 a2 48) - (div.s f4 f0 f4) - (l.q t1 a2 80) - (nop!) - (l.q t2 a2 112) - (nop!) - (l.q t3 a2) - (nop!) - (l.q t4 a2 32) - (nop!) - (l.q t5 a2 64) - (nop!) - (l.q t6 a2 96) - (nop!) - (m t7 f3) - (mul.s f3 f4 f0) - (nop!) - (div.s f2 f0 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (add.i a2 a2 128) - (nop!) - (m t8 f3) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f2 f2 f0) - (nop!) - (div.s f1 f0 f1) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m t9 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (pextlw t7 t8 t7) - (nop!) - (nop!) - (nop!) - (nop!) - (m t8 f1) - (nop!) - (nop!) - (nop!) - (nop!) - (pextlw t8 t8 t9) - (nop!) - (pcpyld t7 t8 t7) - (nop!) - (nop!) - (nop!) - (nop!) - (m.ni vf21 a3) - (nop!) - (m.ni vf22 t0) - (nop!) - (m.ni vf23 t1) - (nop!) - (m.ni vf24 t2) - (nop!) - (m.ni vf9 t3) - (nop!) - (m.ni vf10 t4) - (nop!) - (m.ni vf11 t5) - (nop!) - (m.ni vf12 t6) - (nop!) - (m.ni vf27 t7) - (nop!) - (m.ni t1 vf17) - (nop!) - (m.ni t2 vf18) - (nop!) - (m.ni t0 vf19) - (b.le a1 r0 generic-envmap-dproc-finish-tail :delay (m.ni a3 vf20))) - (asm-block vertex-loop - (label generic-envmap-dproc-vertex-loop) - ;; VU0 entry 48 converts the next four normal/eye pairs while the EE packs and writes the - ;; preceding reflected texture coordinates. - (ppach t1 r0 t1) - (callms GENERIC-VU0-ENVMAP) - (ppach t2 r0 t2) - (mmi-nop!) - (ppach t0 r0 t0) - (mmi-nop!) - (ppach a3 r0 a3) - (mmi-nop!) - (nop!) - (s.w t1 a0 16) - (nop!) - (s.w t2 a0 48) - (nop!) - (s.w t0 a0 80) - (nop!) - (s.w a3 a0 112) - (nop!) - (l.s f4 a2 24) - (nop!) - (l.s f3 a2 56) - (nop!) - (l.s f2 a2 88) - (nop!) - (l.s f1 a2 120) - (nop!) - (l.q a3 a2 16) - (sub.s f4 f4 f0) - (s.w v1 a0 20) - (sub.s f3 f3 f0) - (s.w v1 a0 52) - (sub.s f2 f2 f0) - (s.w v1 a0 84) - (sub.s f1 f1 f0) - (s.w v1 a0 116) - (div.s f4 f0 f4) - (l.q t3 a2 48) - (nop!) - (l.q t4 a2 80) - (nop!) - (l.q t5 a2 112) - (nop!) - (l.q t6 a2) - (nop!) - (l.q t2 a2 32) - (nop!) - (l.q t0 a2 64) - (nop!) - (l.q t1 a2 96) - (add.i a1 a1 -4) - (add.i a0 a0 128) - (mul.s f4 f4 f0) - (nop!) - (div.s f3 f0 f3) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m t7 f4) - (nop!) - (add.i a2 a2 128) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f3 f3 f0) - (nop!) - (div.s f2 f0 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m t8 f3) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f2 f2 f0) - (nop!) - (div.s f1 f0 f1) - (nop!) - (pextlw t7 t8 t7) - (nop!) - (nop!) - (m t8 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m t9 f1) - (nop!) - (m.ni vf21 a3) - (nop!) - (m.ni vf22 t3) - (nop!) - (m.ni vf23 t4) - (nop!) - (m.ni vf24 t5) - (nop!) - (m.ni vf9 t6) - (pextlw a3 t9 t8) - (m.ni vf10 t2) - (pcpyld a3 a3 t7) - (m.ni vf11 t0) - (nop!) - (m.ni vf12 t1) - (nop!) - (m.ni vf27 a3) - (nop!) - (m.ni t1 vf17) - (nop!) - (m.ni t2 vf18) - (nop!) - (m.ni t0 vf19) - (b.gt a1 r0 generic-envmap-dproc-vertex-loop :delay (m.ni a3 vf20))) - (asm-block finish-tail - (label generic-envmap-dproc-finish-tail) - ;; Drain the final VU0 result and write only the valid lanes in the last group. - (ppach a1 r0 t1) - (s.w v1 a0 20) - (ppach a2 r0 t2) - (s.w v1 a0 52) - (ppach t0 r0 t0) - (s.w a1 a0 16) - (ppach a1 r0 a3) - (s.w a2 a0 48) - (nop!) - (s.w t0 a0 80) - (nop!) - (s.w a1 a0 112) - (nop!) - (s.w v1 a0 84) - (nop!) - (s.w v1 a0 116) - (m v0 r0) - (jr ra :delay (add sp sp r0)) - (nop!) - (nop!) - (nop!)) - ) + ;; A base pass owns one authored shader per strip. A standalone environment pass repeats the + ;; shared environment shader but patches each copy with that strip's count and running kick. + (let ((strip-table (the-as (pointer uint8) (&-> gsf header strip-table 0))) + (shader-source (the-as (inline-array adgif-shader) + (if (nonzero? (-> saves is-envmap)) + (-> saves ptr-env-shader) + (-> saves ptr-shaders)))) + (kick-offset 0)) + (dotimes (strip num-strips) + (let ((strip-count (the-as int (-> strip-table strip)))) + (generic-copy-effect-shader! + (-> shaders strip) + (-> shader-source (if (nonzero? (-> saves is-envmap)) 0 strip)) + kick-offset + strip-count) + (+! kick-offset (+ 9 (* strip-count 3))))) + (when (nonzero? num-strips) + (logior! (-> shaders (- num-strips 1) quad 1 word 3) #x8000))) + (set! (-> header strips) (the-as uint num-strips)) + (set! (-> header dps) (the-as uint num-dps)) + (set! (-> header kickoff) 0)) - (defun generic-prepare-dma-single () - "Build one Generic DMA packet layout, choose the rotating VU1 input and GIF buffers, and initialize - its output header before vertex conversion." - (declare (asm-func none) (allow-saved-regs)) - ;; Builds the packet the processors are about to fill, and it is worth knowing the shape exactly - ;; because everything downstream is measured against it. - ;; - ;; Eight quadwords of header first, which is a generic-texbuf: the DMA tag, then the four rows of the - ;; camera matrix, the triangle-strip GIF tag, and two A+D quadwords. Those two are GS no-ops whose - ;; unused upper words carry the draw-point count and the offset of the first kick, so the counts ride - ;; inside a packet the GS is willing to accept. - ;; - ;; Then one five-quadword adgif shader per strip, each with two software-owned words patched: +28 - ;; takes that strip's vertex count out of gsf-header.strip-table, and +12 takes the running VU-memory - ;; offset at which the strip will be kicked. That offset accumulates three quadwords per vertex plus - ;; nine per strip - the nine being the prologue VU1 writes into the input bank ahead of each strip's - ;; vertices. The last shader's count gets bit 15 set as the terminator. Total upload is - ;; 7 + 5 * num-strips quadwords, which is what the VIF tag is patched with at the end. - ;; - ;; Finally it flips gifbuf-adr to the other header buffer, derives the three stream cursors the - ;; processors write into, and records the quadword count the caller will hand to fromSPR. - ;; - ;; is-envmap chooses between an authored shader per strip and one shared shader repeated - the - ;; environment map is one texture for the whole level - and between the base GIF tag and the envmap - ;; one. That is the only difference between the two paths. - (asm-block select-buffers - (label generic-prepare-dma-single-entry) - (add.i sp sp -32) - (s.q gp at (generic-work-offset fx-buf work storage2 data 1)) - (lui at #x7000) - (nop!) - (l.w t1 at (generic-work-offset saves gsf-buf)) - (nop!) - (l.w t8 at (generic-work-offset saves ptr-shaders)) - (nop!) - (l.w v1 at (generic-work-offset saves cur-outbuf)) - (nop!) - (l.w t3 at (generic-work-offset saves is-envmap)) - (nop!) - (l.h a1 t1 18) - (m a0 v1) - (l.bu a2 t1 16) - (b.z t8 generic-prepare-dma-single-empty-packet :delay (l.q t4 at (generic-work-offset fx-buf work consts dma-header))) - (nop!) - (l.q t9 at (generic-work-offset fx-buf work consts matrix vector 0)) - (nop!) - (l.q gp at (generic-work-offset fx-buf work consts matrix vector 1)) - (m a3 a2) - (l.q t7 at (generic-work-offset fx-buf work consts matrix vector 2)) - (b.nz t3 generic-prepare-dma-single-select-second-gif-buffer :delay (l.q t2 at (generic-work-offset fx-buf work consts matrix vector 3))) - (nop!) - (l.q t0 at (generic-work-offset fx-buf work consts base-strgif)) - (nop!) - (l.q t5 at (generic-work-offset fx-buf work consts adcmds 0)) - (b generic-prepare-dma-single-write-header :delay (l.q t6 at (generic-work-offset fx-buf work consts adcmds 3)))) - ;; The environment-map pass takes its own GIF tag and uses the same A+D quadword twice. Nothing - ;; about a GIF buffer is selected here despite what the label has always said. - (asm-block select-envmap-templates - (label generic-prepare-dma-single-select-second-gif-buffer) - (nop!) - (l.q t0 at (generic-work-offset fx-buf work consts envmap strgif)) - (nop!) - (l.q t5 at (generic-work-offset fx-buf work consts adcmds 3)) - (nop!) - (l.q t6 at (generic-work-offset fx-buf work consts adcmds 3))) - (asm-block write-header - (label generic-prepare-dma-single-write-header) - ;; Write the fixed VIF/GIF prefix. The selected shader path below appends either one record per - ;; strip or repeated copies of the shared environment-map shader. - (nop!) - (s.q t4 a0) - (nop!) - (s.q t9 a0 16) - (nop!) - (s.q gp a0 32) - (m t4 t8) - (s.q t7 a0 48) - (add.i t1 t1 22) - (s.q t2 a0 64) - (add.i t2 r0 0) - (s.q t0 a0 80) - (add.i t0 r0 128) - (s.q t5 a0 96) - (b.nz t3 generic-prepare-dma-single-copy-env-shader :delay (s.q t6 a0 112))) - (asm-block copy-shader-loop - (label generic-prepare-dma-single-copy-shader-loop) - ;; Each strip byte selects an authored shader. Copy its five qwords and patch the strip-specific - ;; AD and GIF words into the packet. - (add a0 a0 t0) - (l.q t0 t4) - (add.i a3 a3 -1) - (l.bu t3 t1) - (nop!) - (l.q t5 t4 16) - (add t7 t3 t3) - (l.q t6 t4 32) - (add t8 t7 t3) - (l.q t7 t4 48) - (add.i t9 t8 9) - (l.q t8 t4 64) - (add.i t4 t4 80) - (s.q t0 a0) - (nop!) - (s.w t2 a0 12) - (add.i t1 t1 1) - (s.q t5 a0 16) - (add t2 t2 t9) - (s.w t3 a0 28) - (nop!) - (s.q t6 a0 32) - (add.i t0 r0 80) - (s.q t7 a0 48) - (b.gt a3 r0 generic-prepare-dma-single-copy-shader-loop :delay (s.q t8 a0 64)) - (b generic-prepare-dma-single-finish-shaders :delay (nop!))) - (asm-block copy-env-shader - (label generic-prepare-dma-single-copy-env-shader) - (nop!) - (l.w t3 at (generic-work-offset saves ptr-env-shader)) - (nop!) - (l.q t4 t3) - (nop!) - (l.q t5 t3 16) - (nop!) - (l.q t6 t3 32) - (nop!) - (l.q t7 t3 48) - (nop!) - (l.q t8 t3 64)) - (asm-block copy-env-shader-loop - (label generic-prepare-dma-single-copy-env-shader-loop) - ;; Environment-map strips all share one five-qword shader record; only the strip count and - ;; packet cursor change. - (add a0 a0 t0) - (l.bu t3 t1) - (add.i a3 a3 -1) - (s.q t4 a0) - (add t0 t3 t3) - (s.w t2 a0 12) - (add t0 t0 t3) - (s.q t5 a0 16) - (add.i t0 t0 9) - (s.w t3 a0 28) - (add.i t1 t1 1) - (nop!) - (add t2 t2 t0) - (s.q t6 a0 32) - (add.i t0 r0 80) - (s.q t7 a0 48) - (b.gt a3 r0 generic-prepare-dma-single-copy-env-shader-loop :delay (s.q t8 a0 64))) - (asm-block finish-shaders - (label generic-prepare-dma-single-finish-shaders) - (ori a3 t3 #x8000) - (s.w a1 at (generic-work-offset saves num-dps)) - (nop!) - (s.w a3 a0 28) - (nop!) - (s.w a2 a0 92) - (nop!) - (s.w a1 v1 108) - (b generic-prepare-dma-single-finish-packet :delay (s.w r0 v1 124))) - ;; No shaders means nothing to draw. Write the DMA tag alone and still publish the draw-point count - ;; and the stream cursors, so the caller's send tail and the processors after it stay valid rather - ;; than needing a special case. - (asm-block empty-packet - (label generic-prepare-dma-single-empty-packet) - (sll a3 a2 2) - (s.q t4 a0) - (add a3 a3 a2) - (s.w a1 v1 108) - (sll a3 a3 4) - (nop!) - (add.i t0 a3 128) - (s.w a1 at (generic-work-offset saves num-dps))) - (asm-block finish-packet - (label generic-prepare-dma-single-finish-packet) - ;; Finish the VIF unpack tags, derive all output-stream cursors, and rotate the input-bank - ;; selector for the next call. - (sll t1 a2 2) - (l.w a3 v1 12) - (add a2 t1 a2) - (l.w t1 at (generic-work-offset saves gifbuf-adr)) - (add.i a2 a2 7) - (nop!) - (or t2 a3 t1) - (nop!) - (sll t3 a2 16) - (xor.i a3 t1 GENERIC-VU1-HEADER-FLIP) - (or t1 t2 t3) - (add.i a2 a2 1) - (nop!) - (s.w t1 v1 12) - (nop!) - (s.w a3 at (generic-work-offset saves gifbuf-adr)) - (add.i a1 a1 3) - (add a0 a0 t0) - (sra a1 a1 2) - (add.i a2 a0 32) - (sll t0 a1 2) - (nop!) - (add a3 t0 t0) - (sll a1 t0 2) - (add a3 a3 t0) - (add.i a1 a1 15) - (sll a3 a3 2) - (sra a1 a1 4) - (add.i a3 a3 15) - (sll t1 a1 4) - (sra a3 a3 4) - (l.w a1 at (generic-work-offset saves inbuf-adr)) - (sll a3 a3 4) - (nop!) - (add a3 a2 a3) - (l.w t2 at (generic-work-offset fx-buf work consts stcycle-tag)) - (add a2 a3 t1) - (s.q r0 a3) - (add.i a2 a2 16) - (s.q r0 a3 -16) - (add t1 a2 t1) - (s.q r0 a3 -32) - (add.i t1 t1 16) - (s.q r0 a2) - (sub t3 t1 v1) - (s.q r0 a2 -16) - (sra t3 t3 4) - (s.q r0 a0) - (nop!) - (s.h t3 v1) - (add.i v1 t3 1) - (s.q r0 t1) - (nop!) - (s.q r0 t1 -16) - (nop!) - (add.i t5 a1 1) - (add.i t4 a1 2) - (l.w t3 at (generic-work-offset fx-buf work consts flush-tag)) - (nop!) - (l.w t7 at (generic-work-offset fx-buf work consts unpack-vtx-tag)) - (sll t0 t0 16) - (l.w t6 at (generic-work-offset fx-buf work consts unpack-clr-tag)) - (or t4 t7 t4) - (l.w t7 at (generic-work-offset fx-buf work consts unpack-tex-tag)) - (or t5 t6 t5) - (l.w t6 at (generic-work-offset fx-buf work consts mscal-tag)) - (nop!) - (l.w t8 at (generic-work-offset fx-buf work consts reset-cycle-tag)) - (m t9 a1) - (s.w t2 a0 8) - (or t2 t7 t9) - (s.w t8 t1) - (add.i a0 a0 16) - (s.w r0 t1 4) - (or t4 t4 t0) - (s.w t6 t1 8) - (add.i a3 a3 16) - (s.w t3 t1 12) - (or t1 t5 t0) - (s.w t4 a0 -4) - (or t0 t2 t0) - (s.w t1 a3 -4) - (add.i a2 a2 16) - (s.w t0 a2 -4) - (add.i t0 r0 GENERIC-VU1-INBUF-LAST) - (s.w v1 at (generic-work-offset saves qwc)) - (b.ne a1 t0 generic-prepare-dma-single-select-next-input :delay (add.i v1 a1 GENERIC-VU1-INBUF-STEP)) - (nop!) - (add.i v1 r0 9)) - (asm-block select-next-input - (label generic-prepare-dma-single-select-next-input) - (nop!) - (s.w v1 at (generic-work-offset saves inbuf-adr)) - (nop!) - (s.w a0 at (generic-work-offset saves ptr-vtxs)) - (nop!) - (s.w a3 at (generic-work-offset saves ptr-clrs)) - (nop!) - (s.w a2 at (generic-work-offset saves ptr-texs)) - (m v0 r0) - (l.q gp at (generic-work-offset fx-buf work storage2 data 1)) - (jr ra :delay (add.i sp sp 32)) - (nop!) - (nop!)) - ) + ;; Upload this fragment's VU header to the sink's current header bank, then select the other bank + ;; for the next single-pass fragment. + (set! (-> packet tag vif1) + (the-as vif-tag + (logior (the-as uint (-> packet tag vif1)) + (the-as uint (-> saves gifbuf-adr)) + (shl header-qwc 16)))) + (set! (-> saves gifbuf-adr) (logxor (-> saves gifbuf-adr) GENERIC-VU1-HEADER-FLIP)) - (defun generic-prepare-dma-double () - "Build the paired ordinary and environment-map Generic DMA packet layouts, choose rotating VU1 input - and GIF buffers, and initialize both output headers before vertex conversion." - (declare (asm-func none) (allow-saved-regs)) - ;; The same packet twice over, because an environment-mapped object is drawn as two GS passes over the - ;; same geometry: the base color, then the reflection blended on top. So this lays down two - ;; generic-texbuf headers, two shader lists - authored shaders for the base pass, the shared - ;; environment shader repeated for the second - and two sets of stream cursors: the base three in - ;; saves.ptr-vtxs/-clrs/-texs and the second pair in saves.ptr-env-clrs/-env-texs. - ;; - ;; Both selectors therefore advance twice, two header buffers consumed and two input banks, because - ;; each pass is an independent VU1 invocation with its own matrix, shaders and kick. That is the reason - ;; the input banks are triple buffered rather than double: a two-pass draw needs two banks in flight - ;; while the GIF is still reading out a third. - ;; - ;; The second pass needs no copy of the positions, and the mechanism is worth knowing: instead of a - ;; second twelve-bytes-per-vertex block in scratchpad, it emits a DMA ref tag pointing at - ;; basep + the first pass's position offset - where those same bytes will already be sitting in main - ;; memory once this buffer has been copied out. So the second pass uploads four bytes of color and - ;; four of coordinate per draw point and nothing else, which is what makes it cheap enough to be - ;; worth having. It is also why its final MSCAL needs a cnt tag of its own rather than riding in a - ;; bare VIF quadword the way the single-pass builder's does. - ;; - ;; Two MSCALs, not one: consts.vif-header is patched into a kick-and-switch quadword whose first word - ;; fires the base pass the moment its streams are in, and whose last word unpacks the second header - ;; into the other VU1 header buffer. And since two flips of gifbuf-adr are the identity, this builder - ;; never writes that cursor back at all - only the input bank advances, twice. - (asm-block select-buffers - (label generic-prepare-dma-double-entry) - (add.i sp sp -128) - (s.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (s.q s0 at (generic-work-offset fx-buf work storage2 data 1)) - (s.q s1 at (generic-work-offset fx-buf work storage2 data 2)) - (s.q s2 at (generic-work-offset fx-buf work storage2 data 3)) - (s.q s3 at (generic-work-offset fx-buf work storage2 data 4)) - (s.q s4 at (generic-work-offset fx-buf work storage2 data 5)) - (s.q s5 at (generic-work-offset fx-buf work storage2 data 6)) - (s.q gp at (generic-work-offset fx-buf work storage2 data 7)) - (lui at #x7000) - (nop!) - (nop!) - (l.w a3 at (generic-work-offset saves gsf-buf)) - (nop!) - (l.bu v1 a3 16) - (nop!) - (l.h a0 a3 18) - (nop!) - (l.w t7 at (generic-work-offset saves cur-outbuf)) - (sll a2 v1 2) - (add.i a1 a0 3) - (add a2 a2 v1) - (add.i t0 r0 -4) - (and t6 a1 t0) - (sll a1 a2 4) - (add a2 t6 t6) - (add.i a1 a1 112) - (add a2 a2 t6) - (sll t0 t6 2) - (nop!) - (add.i t0 t0 15) - (sll a2 a2 2) - (sra t0 t0 4) - (add.i a2 a2 15) - (sll t0 t0 4) - (sra t9 a2 4) - (sll t2 t9 4) - (m a2 t7) - (sra t8 a1 4) - (add t1 a2 a1) - (nop!) - (add gp t1 t2) - (nop!) - (add t2 gp t0) - (nop!) - (add t3 t2 t0) - (nop!) - (add ra t3 a1) - (nop!) - (add t4 ra t0) - (nop!) - (add t5 t4 t0) - (add.i t0 t1 32) - (add.i t1 gp 64) - (add.i t2 t2 80) - (add.i a1 t3 80) - (add.i t3 ra 112) - (add.i t4 t4 128) - (add.i t5 t5 128) - (s.q r0 t0 -16) - (s.q r0 t1 -32) - (s.q r0 t1 -16) - (s.q r0 t2 -32) - (s.q r0 t2 -16) - (s.q r0 a1 -16) - (s.q r0 t3 -16) - (s.q r0 t4 -32) - (s.q r0 t4 -16) - (s.q r0 t5 -16) - (l.q ra at (generic-work-offset fx-buf work consts dma-header)) - (l.q gp at (generic-work-offset fx-buf work consts vif-header)) - (l.q s5 at (generic-work-offset fx-buf work consts dma-ref-vtxs)) - (l.q s4 at (generic-work-offset fx-buf work consts dma-cnt-call)) - (l.w s3 at (generic-work-offset fx-buf work consts mscal-tag)) - (s.q ra a2) - (s.q gp a1) - (s.q s5 t5) - (s.h t9 t5) - (s.q s4 t5 16) - (s.w s3 t5 24) - (sub t9 t5 t7) - (sra t9 t9 4) - (add.i t9 t9 -1) - (s.h t9 t7) - (add.i t7 t9 3) - (s.w t7 at (generic-work-offset saves qwc)) - (l.w t7 at (generic-work-offset saves basep)) - (sub t9 t0 a2) - (add t7 t7 t9) - (nop!) - (l.w t9 a2 12) - (sll ra t8 16) - (l.w t8 at (generic-work-offset saves gifbuf-adr)) - (nop!) - (l.w gp a1 12) - (or s4 t9 t8) - (l.w t9 at (generic-work-offset saves inbuf-adr)) - (xor.i s5 t8 GENERIC-VU1-HEADER-FLIP) - (nop!) - (or s4 s4 ra) - (l.w t8 at (generic-work-offset fx-buf work consts stcycle-tag)) - (nop!) - (l.w s3 at (generic-work-offset fx-buf work consts flush-tag)) - (or gp gp s5) - (s.w s4 a2 12) - (or s5 gp ra) - (l.w ra at (generic-work-offset fx-buf work consts unpack-vtx-tag)) - (add.i gp t9 1) - (s.w s5 a1 12) - (add.i s4 t9 2) - (l.w s3 at (generic-work-offset fx-buf work consts unpack-clr-tag)) - (sll t6 t6 16) - (l.w s5 at (generic-work-offset fx-buf work consts unpack-tex-tag)) - (or s4 ra s4) - (l.w ra at (generic-work-offset fx-buf work consts mscal-tag)) - (or s3 s3 gp) - (l.w gp at (generic-work-offset fx-buf work consts reset-cycle-tag)) - (m s2 t9) - (s.w t8 t0 -8) - (or s5 s5 s2) - (s.w gp a1 4) - (or s4 s4 t6) - (s.w ra a1) - (or s3 s3 t6) - (s.w s4 t0 -4) - (or s5 s5 t6) - (s.w s3 t1 -4) - (add.i s4 r0 GENERIC-VU1-INBUF-LAST) - (s.w s5 t2 -4) - (b.ne t9 s4 generic-prepare-dma-double-select-next-gif-buffer :delay (add.i t9 t9 GENERIC-VU1-INBUF-STEP)) - (nop!) - (add.i t9 r0 9)) - (asm-block select-next-gif-buffer - (label generic-prepare-dma-double-select-next-gif-buffer) - (add.i s1 t9 1) - (l.w s0 at (generic-work-offset fx-buf work consts unpack-clr-tag)) - (m s3 t9) - (l.w s2 at (generic-work-offset fx-buf work consts unpack-tex-tag)) - (add.i s5 t9 2) - (l.w s4 at (generic-work-offset fx-buf work consts unpack-vtx-tag)) - (or s1 s0 s1) - (s.w t8 t3 -8) - (or t8 s2 s3) - (s.w gp t5 24) - (or gp s4 s5) - (nop!) - (or s5 s1 t6) - (s.w ra t5 28) - (or t8 t8 t6) - (s.w s5 t3 -4) - (or ra gp t6) - (s.w t8 t4 -4) - (add.i t6 r0 GENERIC-VU1-INBUF-LAST) - (s.w t7 t5 4) - (nop!) - (s.w ra t5 12) - (b.ne t9 t6 generic-prepare-dma-double-select-next-input :delay (add.i t5 t9 GENERIC-VU1-INBUF-STEP)) - (nop!) - (add.i t5 r0 9)) - (asm-block select-next-input - (label generic-prepare-dma-double-select-next-input) - (nop!) - (s.w t5 at (generic-work-offset saves inbuf-adr)) - (nop!) - (s.w t0 at (generic-work-offset saves ptr-vtxs)) - (nop!) - (s.w t1 at (generic-work-offset saves ptr-clrs)) - (nop!) - (s.w t2 at (generic-work-offset saves ptr-texs)) - (nop!) - (s.w t3 at (generic-work-offset saves ptr-env-clrs)) - (nop!) - (s.w t4 at (generic-work-offset saves ptr-env-texs)) - (nop!) - (l.w t0 at (generic-work-offset saves ptr-shaders)) - (nop!) - (l.q t1 at (generic-work-offset fx-buf work consts matrix vector 0)) - (b.z t0 generic-prepare-dma-double-empty-packet :delay (l.q t2 at (generic-work-offset fx-buf work consts matrix vector 1))) - (nop!) - (l.q t3 at (generic-work-offset fx-buf work consts matrix vector 2)) - (nop!) - (l.q t4 at (generic-work-offset fx-buf work consts matrix vector 3)) - (nop!) - (s.q t1 a2 16) - (nop!) - (s.q t2 a2 32) - (nop!) - (s.q t3 a2 48) - (nop!) - (s.q t4 a2 64) - (nop!) - (s.q t1 a1 16) - (nop!) - (s.q t2 a1 32) - (nop!) - (s.q t3 a1 48) - (nop!) - (s.q t4 a1 64) - (nop!) - (l.q t1 at (generic-work-offset fx-buf work consts base-strgif)) - (nop!) - (l.q t2 at (generic-work-offset fx-buf work consts adcmds 0)) - (nop!) - (l.q t3 at (generic-work-offset fx-buf work consts adcmds 3)) - (nop!) - (l.q t4 at (generic-work-offset fx-buf work consts envmap strgif)) - (nop!) - (s.q t1 a2 80) - (nop!) - (s.q t2 a2 96) - (nop!) - (s.q t3 a2 112) - (nop!) - (s.q t4 a1 80) - (nop!) - (s.q t3 a1 96) - (nop!) - (s.q t3 a1 112) - (add.i t2 a3 22) - (add.i t3 r0 0) - (m t4 v1) - (add.i t6 r0 128) - (m t1 a2) - (nop!)) - (asm-block copy-shader-loop - (label generic-prepare-dma-double-copy-shader-loop) - ;; Build the ordinary pass first, copying the selected five-qword shader for each strip. - (add t1 t1 t6) - (l.q t6 t0) - (add.i t4 t4 -1) - (l.bu t5 t2) - (nop!) - (l.q t7 t0 16) - (add t9 t5 t5) - (l.q t8 t0 32) - (add ra t9 t5) - (l.q t9 t0 48) - (add.i gp ra 9) - (l.q ra t0 64) - (add.i t0 t0 80) - (s.q t6 t1) - (nop!) - (s.w t3 t1 12) - (add.i t2 t2 1) - (s.q t7 t1 16) - (add t3 t3 gp) - (s.w t5 t1 28) - (nop!) - (s.q t8 t1 32) - (add.i t6 r0 80) - (s.q t9 t1 48) - (b.gt t4 r0 generic-prepare-dma-double-copy-shader-loop :delay (s.q ra t1 64)) - (ori t0 t5 #x8000) - (s.w a0 at (generic-work-offset saves num-dps)) - (nop!) - (s.w t0 t1 28) - (nop!) - (s.w v1 a2 92) - (nop!) - (s.w a0 a2 108) - (nop!) - (s.w r0 a2 124) - (add.i a3 a3 22) - (add.i t0 r0 0) - (m t1 v1) - (l.w t6 at (generic-work-offset saves ptr-env-shader)) - (m a2 a1) - (add.i t8 r0 128) - (nop!) - (l.q t2 t6) - (nop!) - (l.q t3 t6 16) - (nop!) - (l.q t4 t6 32) - (nop!) - (l.q t5 t6 48) - (nop!) - (l.q t6 t6 64)) - (asm-block copy-env-shader-loop - (label generic-prepare-dma-double-copy-env-shader-loop) - ;; Append the matching environment-map pass using the shared shader and its own stream cursors. - (add a2 a2 t8) - (l.bu t7 a3) - (add.i t1 t1 -1) - (s.q t2 a2) - (add t8 t7 t7) - (s.w t0 a2 12) - (add t8 t8 t7) - (s.q t3 a2 16) - (add.i t8 t8 9) - (s.w t7 a2 28) - (add.i a3 a3 1) - (nop!) - (add t0 t0 t8) - (s.q t4 a2 32) - (add.i t8 r0 80) - (s.q t5 a2 48) - (b.gt t1 r0 generic-prepare-dma-double-copy-env-shader-loop :delay (s.q t6 a2 64)) - (ori a3 t7 #x8000) - (s.w a0 at (generic-work-offset saves num-dps)) - (nop!) - (s.w a3 a2 28) - (nop!) - (s.w v1 a1 92) - (nop!) - (s.w a0 a1 108) - (b generic-prepare-dma-double-finish :delay (s.w r0 a1 124))) - (asm-block empty-packet - (label generic-prepare-dma-double-empty-packet) - (nop!) - (l.q a3 a2 16) - (nop!) - (l.q t0 a2 32) - (nop!) - (l.q t1 a2 48) - (nop!) - (l.q t2 a2 64) - (nop!) - (s.q a3 a1 16) - (nop!) - (s.q t0 a1 32) - (nop!) - (s.q t1 a1 48) - (nop!) - (s.q t2 a1 64) - (nop!) - (l.q a3 at (generic-work-offset fx-buf work consts envmap strgif)) - (nop!) - (l.q t0 at (generic-work-offset fx-buf work consts adcmds 3)) - (nop!) - (s.q a3 a1 80) - (nop!) - (s.q t0 a1 96) - (nop!) - (s.q t0 a1 112) - (m a3 v1) - (l.w t5 at (generic-work-offset saves ptr-env-shader)) - (m t0 a1) - (add.i t7 r0 128) - (nop!) - (l.q t1 t5) - (nop!) - (l.q t2 t5 16) - (nop!) - (l.q t3 t5 32) - (nop!) - (l.q t4 t5 48) - (nop!) - (l.q t5 t5 64) - (add a2 a2 t7) - (nop!)) - (asm-block copy-existing-packet-loop - (label generic-prepare-dma-double-copy-existing-packet-loop) - ;; The empty-input case mirrors the already prepared packet headers into the second pass so - ;; both chains retain valid termination and cursor state. - (add t0 t0 t7) - (l.wu t8 a2 12) - (add.i a3 a3 -1) - (l.wu t7 a2 28) - (add.i a2 a2 80) - (s.q t1 t0) - (nop!) - (s.w t8 t0 12) - (nop!) - (s.q t2 t0 16) - (add t8 t8 t6) - (s.w t7 t0 28) - (nop!) - (s.q t3 t0 32) - (add.i t7 r0 80) - (s.q t4 t0 48) - (b.gt a3 r0 generic-prepare-dma-double-copy-existing-packet-loop :delay (s.q t5 t0 64)) - (nop!) - (s.w a0 at (generic-work-offset saves num-dps)) - (nop!) - (s.w v1 a1 92) - (nop!) - (s.w a0 a1 108) - (nop!) - (s.w r0 a1 124)) - (asm-block finish - (label generic-prepare-dma-double-finish) - (m v0 r0) - (l.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (l.q gp at (generic-work-offset fx-buf work storage2 data 7)) - (l.q s5 at (generic-work-offset fx-buf work storage2 data 6)) - (l.q s4 at (generic-work-offset fx-buf work storage2 data 5)) - (l.q s3 at (generic-work-offset fx-buf work storage2 data 4)) - (l.q s2 at (generic-work-offset fx-buf work storage2 data 3)) - (l.q s1 at (generic-work-offset fx-buf work storage2 data 2)) - (l.q s0 at (generic-work-offset fx-buf work storage2 data 1)) - (jr ra :delay (add.i sp sp 128)) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!)) - ) + ;; Prefix each dense payload with its UNPACK and clear the intervening qwords to VIF NOPs. STCYCL + ;; interleaves the separately uploaded position, color, and coordinate fields in VU memory. + (generic-write-stream-tag! + position-stream + (-> consts stcycle-tag) + (logior (-> consts unpack-vtx-tag) (shl padded-dps 16) (+ inbuf 2))) + (set! (-> (the-as (pointer uint128) (- color-stream 32)) 0) (the-as uint128 0)) + (generic-write-stream-tag! + color-stream (the-as uint 0) (logior (-> consts unpack-clr-tag) (shl padded-dps 16) (+ inbuf 1))) + (set! (-> (the-as (pointer uint128) (- texture-stream 32)) 0) (the-as uint128 0)) + (generic-write-stream-tag! + texture-stream (the-as uint 0) (logior (-> consts unpack-tex-tag) (shl padded-dps 16) inbuf)) + (set! (-> (the-as (pointer uint128) control) 0) (the-as uint128 0)) + (set! (-> control-tags 0) (the-as vif-tag (-> consts reset-cycle-tag))) + (set! (-> control-tags 2) (the-as vif-tag (-> consts mscal-tag))) + (set! (-> control-tags 3) (the-as vif-tag (-> consts flush-tag))) - (defun generic-envmap-proc () - "Generate the Generic environment-map pass from expanded vertices: transform normals and eye vectors, - call VU0 entry 48 for reflected ST coordinates, and build the extra color and texture stream." - (declare (asm-func none) (allow-saved-regs)) - ;; The environment-map processor mercneric installs. It writes exactly one thing: the second pass's - ;; coordinate stream at saves.ptr-env-texs, in draw-point order. The base streams belong to - ;; generic-light-proc, which mercneric calls first, so a lit environment-mapped character is two - ;; walks over the index list rather than one. - ;; - ;; The second pass's color stream is not per vertex at all - every draw point in a fragment gets the - ;; same tint out of consts.envmap.colors - so it is filled wholesale before conversion starts, four - ;; quadwords and therefore sixteen draw points per pass, with a three-quadword remainder ladder. - ;; Keeping it out of the inner loop is worth more than the stores cost. - ;; The environment map is a normal-based sphere map, and the EE half is unreadable without knowing - ;; the VU0 half. consts.matrix is the projection matrix alone, so pos and nrm in a gsf-vertex are - ;; already in camera space with +z forward. Writing p for the position, n for the normal and - ;; k = (0, 0, 1) for the camera axis, entry 48 computes - ;; - ;; m = n - k - ;; r = p + m * (m.p) / m.z - ;; ST = 0.5 * r.xy / |r| + 0.5 - ;; - ;; which is the reflection of the eye vector in the plane perpendicular to m: the one that swaps the - ;; camera axis for the normal. It comes out that cheaply because |n| = 1 makes |m|^2 = -2 m.z - ;; exactly, so the 2/|m|^2 a Householder reflection needs is already sitting in m.z and the only - ;; division in the whole thing is the EE-side 1/(n.z - 1) that the FPU chain below computes, four per - ;; group, crossed in together as vf27. - ;; - ;; Two things follow. Unit-length normals are a hard precondition - nothing normalizes n. And for a - ;; vertex on the optical axis the answer is just n, so what lands at the centre of the texture is a - ;; *normal* facing the camera, not a reflection; a true mirror lookup would swing at twice the rate. - ;; The 0.5 scale and 0.5 bias are consts.envmap.consts z and w, and the shader clamps in both - ;; directions, which is what keeps the rim of the sphere from wrapping. n.z = +1 would divide by - ;; zero, but that is a normal pointing directly away from the camera, and front-facing geometry has - ;; n.z in [-1, 0). - ;; - ;; ftoi12 is a transport format rather than a GS coordinate: VU1 converts it back to float before - ;; emitting ST, so the resolution is 1/4096 - 1/2048 in S once its low bit is taken for the kick flag. - ;; - ;; A wrong result is unmistakable. Sign flip on the reciprocal and the reflection slides the wrong - ;; way, mirrored through the texture centre; non-unit normals and it drifts as the model scales; - ;; missing bias and half of every object collapses to one clamped edge color; missing normalize and - ;; distant geometry smears radially and pins at the edge. - ;; - ;; Four vertices per pass, and entry 48 runs one group behind the caller like the lighting entry does - ;; - but structurally rather than by copying its outputs aside. Its last two pairs start the fourth - ;; vertex and the first thirteen pairs of the *next* call finish it, with vf05-vf08, vf29 and vf30 - ;; carrying the pipeline across the call boundary. So the first call's output is garbage, which is - ;; why every caller calls entry 48 twice before reading anything, and why no other VU0 entry may be - ;; used between two of these calls. - ;; - ;; The kick flags need the same two-group delay the coordinates do, which is what s0, s1 and s2 are: - ;; a three-deep shift register moved along once per pass. That delay line plus the four-address - ;; bookkeeping is why this function needs the saved registers at all. - (asm-block load-stream-cursors - (label generic-envmap-proc-entry) - (add.i sp sp -128) - (s.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (s.q s0 at (generic-work-offset fx-buf work storage2 data 1)) - (s.q s1 at (generic-work-offset fx-buf work storage2 data 2)) - (s.q s2 at (generic-work-offset fx-buf work storage2 data 3)) - (s.q s3 at (generic-work-offset fx-buf work storage2 data 4)) - (s.q s4 at (generic-work-offset fx-buf work storage2 data 5)) - (s.q s5 at (generic-work-offset fx-buf work storage2 data 6)) - (s.q gp at (generic-work-offset fx-buf work storage2 data 7)) - (lui at #x7000) - (nop!) - (rlet ((gsf-buf :reg v1 :type gsf-buffer)) - (l.w gsf-buf at (generic-work-offset saves gsf-buf)) - (nop!) - (l.w a0 at (generic-work-offset saves num-dps)) - (nop!) - (l.w t0 (-> gsf-buf info ptr-iks)) - (nop!) - (l.w a2 (-> gsf-buf info ptr-verts))) - (nop!) - (l.w t3 at (generic-work-offset saves ptr-env-clrs)) - (nop!) - (l.w v1 at (generic-work-offset saves ptr-env-texs)) - (nop!) - (add.i t1 r0 255) - (add.i a3 r0 256) ;; not a DMA value: 255 and 256 broadcast into the index and kick-bit masks - (lui a1 -2) - (lui t2 #x3f80) - (ori a1 a1 #xfffe) - (m f0 t2) - (add.i t2 a0 3) - (sra t5 t2 2) - (l.q t2 at (generic-work-offset fx-buf work consts envmap colors)) - (sra t4 t5 2) - (and.i t5 t5 3) - (b.z t4 generic-envmap-proc-clear-tail :delay (nop!))) - (asm-block clear-four-loop - (label generic-envmap-proc-clear-four-loop) - ;; Seed the environment color stream four vertices at a time. - (add.i t3 t3 64) - (s.q t2 t3 -64) - (nop!) - (s.q t2 t3 -48) - (add.i t4 t4 -1) - (s.q t2 t3 -32) - (b.gt t4 r0 generic-envmap-proc-clear-four-loop :delay (s.q t2 t3 -16))) - (asm-block clear-tail - (label generic-envmap-proc-clear-tail) - (b.z t5 generic-envmap-proc-prime-conversion :delay (add.i t4 t5 -1)) - (b.z t4 generic-envmap-proc-prime-conversion :delay (s.q t2 t3)) - (add.i t3 t3 16) - (add.i t4 t4 -1) - (b.z t4 generic-envmap-proc-prime-conversion :delay (s.q t2 t3)) - (add.i t3 t3 16) - (add.i t4 t4 -1) - (nop!) - (s.q t2 t3)) - (asm-block prime-conversion - (label generic-envmap-proc-prime-conversion) - ;; Prime four source vertices, the camera transform, and environment-map constants before - ;; starting the overlapped EE/VU0 loop. - (add.i a0 a0 -4) - (l.vf vf31 at (generic-work-offset fx-buf work consts envmap consts)) - (pextlw a1 a1 a1) - (mmi-nop!) - (pextlw a1 a1 a1) - (mmi-nop!) - (pextlw a2 a2 a2) - (mmi-nop!) - (pextlw a2 a2 a2) - (mmi-nop!) - (pcpyh t2 t1) - (l.d t1 t0) - (pcpyld t2 t2 t2) - (mmi-nop!) - (pcpyh a3 a3) - (mmi-nop!) - (pcpyld a3 a3 a3) - (mmi-nop!) - (add.i t0 t0 8) - (s.q t2 at (generic-work-offset saves envmap index-mask)) - (pextlh t1 r0 t1) - (mmi-nop!) - (and.q t2 t1 t2) - (mmi-nop!) - (sll.w t2 t2 5) - (mmi-nop!) - (add.w t5 t2 a2) - (mmi-nop!) - (srl32 t6 t5 0) - (l.s f4 t5 24) - (pcpyud t7 t5 r0) - (l.s f3 t6 24) - (srl32 t8 t7 0) - (l.s f2 t7 24) - (and.q t1 t1 a3) - (l.s f1 t8 24) - (sra.w t2 t1 8) - (l.q t1 t5 16) - (m.q s0 t2) - (sub.s f4 f4 f0) - (div.s f4 f0 f4) - (l.q t2 t6 16) - (nop!) - (l.q t3 t7 16) - (nop!) - (l.q t4 t8 16) - (nop!) - (l.q t5 t5) - (nop!) - (l.q t6 t6) - (nop!) - (l.q t7 t7) - (nop!) - (l.q t8 t8) - (mul.s f4 f4 f0) - (nop!) - (sub.s f3 f3 f0) - (nop!) - (div.s f3 f0 f3) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m t9 f4) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f3 f3 f0) - (nop!) - (sub.s f2 f2 f0) - (nop!) - (div.s f2 f0 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m ra f3) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f2 f2 f0) - (nop!) - (sub.s f1 f1 f0) - (nop!) - (div.s f1 f0 f1) - (nop!) - (pextlw t9 ra t9) - (nop!) - (nop!) - (nop!) - (nop!) - (m ra f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m gp f1) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (pextlw ra gp ra) - (nop!) - (pcpyld t9 ra t9) - (nop!) - (nop!) - (m.ni vf21 t1) - (nop!) - (m.ni vf22 t2) - (nop!) - (m.ni vf23 t3) - (nop!) - (m.ni vf24 t4) - (nop!) - (m.ni vf9 t5) - (nop!) - (m.ni vf10 t6) - (nop!) - (m.ni vf11 t7) - (nop!) - (m.ni vf12 t8) - (nop!) - (m.ni vf27 t9) - (l.q t2 at (generic-work-offset saves envmap index-mask)) - (callms GENERIC-VU0-ENVMAP) - (nop!) - (l.d t1 t0) - (nop!) - (nop!) - (nop!) - (add.i t0 t0 8) - (pextlh t1 r0 t1) - (mmi-nop!) - (and.q t2 t1 t2) - (mmi-nop!) - (sll.w t2 t2 5) - (mmi-nop!) - (add.w t5 t2 a2) - (mmi-nop!) - (srl32 t6 t5 0) - (l.s f3 t5 24) - (pcpyud t7 t5 r0) - (l.s f2 t6 24) - (srl32 t8 t7 0) - (l.s f1 t7 24) - (and.q t1 t1 a3) - (l.s f4 t8 24) - (sra.w t2 t1 8) - (l.q t1 t5 16) - (m.q s1 t2) - (sub.s f5 f3 f0) - (sub.s f3 f2 f0) - (nop!) - (sub.s f2 f1 f0) - (nop!) - (sub.s f1 f4 f0) - (nop!) - (div.s f4 f0 f5) - (l.q t2 t6 16) - (nop!) - (l.q t3 t7 16) - (nop!) - (l.q t4 t8 16) - (nop!) - (l.q t5 t5) - (nop!) - (l.q t6 t6) - (nop!) - (l.q t7 t7) - (nop!) - (l.q t8 t8) - (mul.s f4 f4 f0) - (nop!) - (div.s f3 f0 f3) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m t9 f4) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f3 f3 f0) - (nop!) - (div.s f2 f0 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m ra f3) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f2 f2 f0) - (nop!) - (div.s f1 f0 f1) - (nop!) - (pextlw t9 ra t9) - (nop!) - (nop!) - (m ra f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m gp f1) - (nop!) - (nop!) - (nop!) - (m.ni vf21 t1) - (nop!) - (m.ni vf9 t5) - (nop!) - (m.ni vf10 t6) - (nop!) - (m.ni vf11 t7) - (nop!) - (m.ni vf12 t8) - (pextlw t1 gp ra) - (nop!) - (pcpyld t1 t1 t9) - (nop!) - (nop!) - (m.ni vf22 t2) - (nop!) - (m.ni vf23 t3) - (nop!) - (m.ni vf24 t4) - (nop!) - (m.ni vf27 t1) - (l.q t2 at (generic-work-offset saves envmap index-mask)) - (callms GENERIC-VU0-ENVMAP) - (nop!) - (l.d t1 t0) - (nop!) - (nop!) - (nop!) - (add.i t0 t0 8) - (pextlh t1 r0 t1) - (mmi-nop!) - (and.q t2 t1 t2) - (mmi-nop!) - (sll.w t2 t2 5) - (mmi-nop!) - (add.w t2 t2 a2) - (mmi-nop!) - (srl32 t3 t2 0) - (l.s f3 t2 24) - (pcpyud t7 t2 r0) - (l.s f2 t3 24) - (srl32 t8 t7 0) - (l.s f1 t7 24) - (and.q t1 t1 a3) - (l.s f4 t8 24) - (sra.w t4 t1 8) - (l.q t1 t2 16) - (m.q s2 t4) - (sub.s f5 f3 f0) - (sub.s f3 f2 f0) - (nop!) - (sub.s f2 f1 f0) - (nop!) - (sub.s f1 f4 f0) - (nop!) - (div.s f4 f0 f5) - (l.q t4 t3 16) - (nop!) - (l.q t5 t7 16) - (nop!) - (l.q t6 t8 16) - (nop!) - (l.q t2 t2) - (nop!) - (l.q t3 t3) - (nop!) - (l.q t7 t7) - (nop!) - (l.q t8 t8) - (mul.s f4 f4 f0) - (nop!) - (div.s f3 f0 f3) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m t9 f4) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f3 f3 f0) - (nop!) - (div.s f2 f0 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m ra f3) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f2 f2 f0) - (nop!) - (div.s f1 f0 f1) - (nop!) - (pextlw t9 ra t9) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (m ra f2) - (nop!) - (m.ni vf21 t1) - (nop!) - (m.ni vf22 t4) - (nop!) - (m.ni vf23 t5) - (nop!) - (m.ni vf24 t6) - (nop!) - (nop!) - (nop!) - (m t1 f1) - (pextlw t1 t1 ra) - (nop!) - (pcpyld t1 t1 t9) - (m.ni vf9 t2) - (nop!) - (m.ni vf10 t3) - (nop!) - (m.ni vf11 t7) - (nop!) - (m.ni vf12 t8) - (nop!) - (m.ni vf27 t1) - (nop!) - (m.ni t4 vf17) - (nop!) - (m.ni t5 vf18) - (nop!) - (m.ni t6 vf19) - (b.le a0 r0 generic-envmap-proc-finish-tail :delay (m.ni t7 vf20))) - (asm-block vertex-loop - (label generic-envmap-proc-vertex-loop) - ;; Entry 48 produces reflection coordinates for the next group while the EE packs the previous - ;; group's reflected ST, environment color, and source texture records. - (l.q t2 at (generic-work-offset saves envmap index-mask)) - (callms GENERIC-VU0-ENVMAP) - (add.i a0 a0 -4) - (l.d t1 t0) - (nop!) - (nop!) - (add.i v1 v1 16) - (add.i t0 t0 8) - (pextlh t1 r0 t1) - (mmi-nop!) - (and.q t2 t1 t2) - (mmi-nop!) - (sll.w t2 t2 5) - (mmi-nop!) - (add.w t9 t2 a2) - (mmi-nop!) - (srl32 ra t9 0) - (l.s f1 t9 24) - (pcpyud s4 t9 r0) - (l.s f4 ra 24) - (sub.s f3 f1 f0) - (nop!) - (srl32 s3 s4 0) - (l.s f1 s4 24) - (and.q t1 t1 a3) - (l.s f2 s3 24) - (sra.w s5 t1 8) - (l.q t1 t9 16) - (div.s f3 f0 f3) - (l.q t2 ra 16) - (m.q gp s0) - (sub.s f4 f4 f0) - (ppach t4 r0 t4) - (l.q t3 s4 16) - (ppach t5 r0 t5) - (l.q t8 s3 16) - (ppach t6 r0 t6) - (l.q t9 t9) - (ppach t7 r0 t7) - (l.q ra ra) - (pextlw t4 t5 t4) - (l.q t5 s4) - (pextlw t6 t7 t6) - (l.q t7 s3) - (m.q s0 s1) - (mul.s f5 f3 f0) - (m.q s1 s2) - (div.s f3 f0 f4) - (pcpyld t4 t6 t4) - (mmi-nop!) - (and.q t4 t4 a1) - (mmi-nop!) - (m.q s2 s5) - (m t6 f5) - (sub.s f4 f1 f0) - (nop!) - (or.q t4 t4 gp) - (nop!) - (sub.s f1 f2 f0) - (nop!) - (mul.s f2 f3 f0) - (s.q t4 v1 -16) - (nop!) - (nop!) - (div.s f3 f0 f4) - (nop!) - (nop!) - (nop!) - (nop!) - (m t4 f2) - (nop!) - (nop!) - (nop!) - (nop!) - (mul.s f2 f3 f0) - (nop!) - (pextlw t4 t4 t6) - (nop!) - (nop!) - (nop!) - (div.s f1 f0 f1) - (nop!) - (nop!) - (m t6 f2) - (nop!) - (m.ni vf21 t1) - (nop!) - (m.ni vf22 t2) - (nop!) - (m.ni vf23 t3) - (nop!) - (m.ni vf24 t8) - (nop!) - (nop!) - (nop!) - (nop!) - (nop!) - (m t1 f1) - (nop!) - (m.ni vf9 t9) - (pextlw t1 t1 t6) - (m.ni vf10 ra) - (pcpyld t1 t1 t4) - (m.ni vf11 t5) - (nop!) - (m.ni vf12 t7) - (nop!) - (m.ni vf27 t1) - (nop!) - (m.ni t4 vf17) - (nop!) - (m.ni t5 vf18) - (nop!) - (m.ni t6 vf19) - (b.gt a0 r0 generic-envmap-proc-vertex-loop :delay (m.ni t7 vf20))) - (asm-block finish-tail - (label generic-envmap-proc-finish-tail) - ;; Drain the final VU0 group and write only the remaining vertices. - (add.i v1 v1 16) - (nop!) - (ppach t4 r0 t4) - (mmi-nop!) - (ppach t5 r0 t5) - (mmi-nop!) - (ppach t6 r0 t6) - (mmi-nop!) - (ppach t7 r0 t7) - (mmi-nop!) - (pextlw t4 t5 t4) - (mmi-nop!) - (pextlw t6 t7 t6) - (mmi-nop!) - (pcpyld t4 t6 t4) - (mmi-nop!) - (and.q t4 t4 a1) - (mmi-nop!) - (or.q t4 t4 s0) - (mmi-nop!) - (nop!) - (s.q t4 v1 -16) - (m v0 r0) - (l.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (l.q gp at (generic-work-offset fx-buf work storage2 data 7)) - (l.q s5 at (generic-work-offset fx-buf work storage2 data 6)) - (l.q s4 at (generic-work-offset fx-buf work storage2 data 5)) - (l.q s3 at (generic-work-offset fx-buf work storage2 data 4)) - (l.q s2 at (generic-work-offset fx-buf work storage2 data 3)) - (l.q s1 at (generic-work-offset fx-buf work storage2 data 2)) - (l.q s0 at (generic-work-offset fx-buf work storage2 data 1)) - (jr ra :delay (add.i sp sp 128)) - (nop!) - (nop!) - (nop!) - (nop!)) - ) + ;; The outer CNT tag excludes its own qword; saves.qwc includes it because fromSPR copies the whole + ;; flat block. Publish the payload addresses only after the layout and counts are final. + (let ((dma-qwc (/ (- control out-address) 16))) + (set! (-> packet tag dma qwc) (the-as uint dma-qwc)) + (set! (-> saves qwc) (the-as uint (+ dma-qwc 1)))) + (set! (-> saves inbuf-adr) (the-as uint (generic-next-inbuf inbuf))) + (set! (-> saves num-dps) (the-as uint num-dps)) + (set! (-> saves ptr-vtxs) position-stream) + (set! (-> saves ptr-clrs) color-stream) + (set! (-> saves ptr-texs) texture-stream) + 0 + (none))) - (defun generic-light-proc () - "Expand indexed GSF vertices in groups of four, run Generic VU0 lighting entry zero, and write the - lit colors plus reconstructed position, normal, and texture streams." - (declare (asm-func none) (allow-saved-regs)) - ;; The lit processor, and the one mercneric installs for every fragment. Four vertices per pass: - ;; their normals cross into vf1-vf4 and their packed colors into vf5-vf8, VU0 entry zero dots each - ;; normal against the three light directions, clamps the negative side away, sums the ambient and - ;; the three directional colors, modulates the vertex's own color by the result and saturates at - ;; 255. The seven light quadwords and the clamp ceiling were copied into scratchpad by - ;; generic-initialize. - ;; - ;; The one thing to know before reading the loop: entry zero copies vf17-vf20 into vf21-vf24 as its - ;; very first instruction, so the colors the EE reads out after a call belong to the group - ;; submitted by the *previous* call. Everything here is therefore one group behind, and the tail - ;; block exists to publish the last group with nothing new to submit. - ;; - ;; A wrong result looks like a character lit from the wrong direction, or clipping to white where - ;; the saturate should have held it at 255. - (asm-block load-cursors-and-lights - (label generic-light-proc-entry) - (add.i sp sp -96) - (s.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (s.q s2 at (generic-work-offset fx-buf work storage2 data 1)) - (s.q s3 at (generic-work-offset fx-buf work storage2 data 2)) - (s.q s4 at (generic-work-offset fx-buf work storage2 data 3)) - (s.q s5 at (generic-work-offset fx-buf work storage2 data 4)) - (s.q gp at (generic-work-offset fx-buf work storage2 data 5)) - (lui at #x7000) - (nop!) - (nop!) - (rlet ((gsf-buf :reg v1 :type gsf-buffer)) - (l.w gsf-buf at (generic-work-offset saves gsf-buf)) - (nop!) - (l.w a1 at (generic-work-offset saves num-dps)) - (nop!) - (l.w a0 (-> gsf-buf info ptr-verts)) - (nop!) - (l.w t0 (-> gsf-buf info ptr-iks))) - (nop!) - (l.w t3 at (generic-work-offset saves ptr-vtxs)) - (nop!) - (l.w v1 at (generic-work-offset saves ptr-clrs)) - (add.i a3 r0 255) - (l.w t2 at (generic-work-offset saves ptr-texs)) - (add.i a2 r0 256) ;; not a DMA value: 255 and 256 broadcast into the index and kick-bit masks - (lui t1 -2) - (m t4 a1) - (l.vf vf10 at (generic-work-offset fx-buf work lights direction 0)) - (ori a1 t1 #xfffe) - (l.vf vf11 at (generic-work-offset fx-buf work lights direction 1)) - (pextlw a1 a1 a1) - (l.vf vf12 at (generic-work-offset fx-buf work lights direction 2)) - (pextlw t1 a0 a0) - (l.vf vf15 at (generic-work-offset fx-buf work lights color 1)) - (pextlw a0 a1 a1) - (l.vf vf14 at (generic-work-offset fx-buf work lights color 0)) - (pextlw a1 t1 t1) - (l.vf vf16 at (generic-work-offset fx-buf work lights color 2)) - (pcpyh a3 a3) - (l.vf vf13 at (generic-work-offset fx-buf work lights ambient)) - (pcpyh t1 a2) - (mmi-nop!) - (pcpyld a2 a3 a3) - (l.vf vf9 at (generic-work-offset fx-buf work consts light-consts)) - (pcpyld a3 t1 t1) - (mmi-nop!) - (nop!) - (l.dr t1 t0) - (nop!) - (l.dl t1 t0 7) - (nop!) - (add.i t0 t0 8) - (pextlh t1 r0 t1) - (mmi-nop!) - (and.q t5 t1 a2) - (mmi-nop!) - (sll.w t5 t5 5) - (mmi-nop!) - (add.w s5 t5 a1) - (mmi-nop!) - (srl32 t9 s5 0) - (mmi-nop!) - (pcpyud s4 s5 r0) - (l.q t6 s5) - (srl32 ra s4 0) - (l.q t7 t9) - (and.q t8 t1 a3) - (l.q t5 s4) - (sra.w gp t8 8) - (l.q t8 ra) - (pextuw s3 t7 t6) - (l.q s5 s5 16) - (pextuw s2 t8 t5) - (l.q t9 t9 16) - (pcpyud s3 s3 s2) - (l.q s4 s4 16) - (and.q s3 s3 a0) - (l.q ra ra 16) - (or.q s3 s3 gp) - (m.ni vf1 s5) - (pextub gp r0 s5) - (s.q s3 t2) - (pextub s5 r0 t9) - (m.ni vf2 t9) - (pextub t9 r0 s4) - (m.ni vf3 s4) - (pextub s4 r0 ra) - (m.ni vf4 ra) - (pextuh ra r0 gp) - (mmi-nop!) - (pextuh gp r0 s5) - (m.ni vf5 ra) - (pextuh t9 r0 t9) - (m.ni vf6 gp) - (pextuh ra r0 s4) - (m.ni vf7 t9) - (prot3w t8 t8) - (m.ni vf8 ra) - (prot3w t7 t7) - (callms GENERIC-VU0-LIGHT) - (pextuw t9 t7 t6) - (mmi-nop!) - (pcpyld t7 t5 t7) - (mmi-nop!) - (pcpyld t6 t9 t6) - (mmi-nop!) - (add.i t2 t2 16) - (mmi-nop!) - (nop!) - (mmi-nop!) - (pextuw t5 t8 t5) - (s.q t7 t3 16) - (pcpyld t5 t8 t5) - (s.q t6 t3) - (nop!) - (s.q t5 t3 32) - (add.i t3 t3 48) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (add.i t4 t4 -4) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (b.le t4 r0 generic-light-proc-finish-tail :delay (mmi-nop!))) - (asm-block vertex-loop - (label generic-light-proc-vertex-loop) - ;; Gather four indexed vertices and call VU0 entry zero. The EE simultaneously drains the - ;; preceding lit colors and reconstructs its position, normal, and texture streams. - (nop!) - (l.dr t1 t0) - (nop!) - (l.dl t1 t0 7) - (nop!) - (add.i t0 t0 8) - (pextlh t1 r0 t1) - (mmi-nop!) - (and.q t5 t1 a2) - (mmi-nop!) - (sll.w t5 t5 5) - (mmi-nop!) - (add.w s5 t5 a1) - (mmi-nop!) - (srl32 t9 s5 0) - (mmi-nop!) - (pcpyud s4 s5 r0) - (l.q t6 s5) - (srl32 ra s4 0) - (l.q t7 t9) - (and.q t8 t1 a3) - (l.q t5 s4) - (sra.w gp t8 8) - (l.q t8 ra) - (pextuw s3 t7 t6) - (l.q s5 s5 16) - (pextuw s2 t8 t5) - (l.q t9 t9 16) - (pcpyud s3 s3 s2) - (l.q s4 s4 16) - (and.q s3 s3 a0) - (l.q ra ra 16) - (or.q s3 s3 gp) - (m.ni vf1 s5) - (pextub gp r0 s5) - (s.q s3 t2) - (pextub s5 r0 t9) - (m.ni vf2 t9) - (pextub t9 r0 s4) - (m.ni vf3 s4) - (pextub s4 r0 ra) - (m.ni vf4 ra) - (pextuh ra r0 gp) - (mmi-nop!) - (pextuh gp r0 s5) - (m.ni vf5 ra) - (pextuh t9 r0 t9) - (m.ni vf6 gp) - (pextuh ra r0 s4) - (m.ni vf7 t9) - (prot3w t8 t8) - (m.ni vf8 ra) - (prot3w t7 t7) - (callms GENERIC-VU0-LIGHT) - (pextuw t9 t7 t6) - (mmi-nop!) - (pcpyld t7 t5 t7) - (mmi-nop!) - (pcpyld t6 t9 t6) - (mmi-nop!) - (pextuw t5 t8 t5) - (mmi-nop!) - (pcpyld t5 t8 t5) - (s.q t6 t3) - (nop!) - (s.q t7 t3 16) - (nop!) - (s.q t5 t3 32) - (add.i t2 t2 16) - (mmi-nop!) - (add.i t3 t3 48) - (m.ni t7 vf21) - (nop!) - (m.ni t8 vf22) - (nop!) - (m.ni t5 vf23) - (nop!) - (m.ni t6 vf24) - (ppach t7 t8 t7) - (mmi-nop!) - (ppach t5 t6 t5) - (mmi-nop!) - (ppacb t5 t5 t7) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (s.q t5 v1) - (add.i t4 t4 -4) - (add.i v1 v1 16) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (nop!) - (mmi-nop!) - (b.gt t4 r0 generic-light-proc-vertex-loop :delay (nop!))) - (asm-block finish-tail - (label generic-light-proc-finish-tail) - ;; Pack the final VU0 colors after the main loop has consumed every complete group. - (nop!) - (vnop) - (nop!) - (m.ni a2 vf17) - (nop!) - (m.ni a3 vf18) - (nop!) - (m.ni a0 vf19) - (nop!) - (m.ni a1 vf20) - (ppach a2 a3 a2) - (mmi-nop!) - (ppach a0 a1 a0) - (mmi-nop!) - (ppacb a0 a0 a2) - (mmi-nop!) - (nop!) - (s.q a0 v1) - (m v0 r0) - (l.d ra at (generic-work-offset fx-buf work storage2 data 0)) - (l.q gp at (generic-work-offset fx-buf work storage2 data 5)) - (l.q s5 at (generic-work-offset fx-buf work storage2 data 4)) - (l.q s4 at (generic-work-offset fx-buf work storage2 data 3)) - (l.q s3 at (generic-work-offset fx-buf work storage2 data 2)) - (l.q s2 at (generic-work-offset fx-buf work storage2 data 1)) - (jr ra :delay (add.i sp sp 96)) - (nop!) - (nop!) - (nop!)) - ) - (defun generic-dma-from-spr ((scratch-address int) (qwc int)) - "Wait for the scratchpad-to-memory DMA channel, then transfer qwc quadwords from scratch-address to - the current Generic DMA output and advance that output cursor." - (declare (asm-func none)) - ;; The standalone form of the send tail that the three wrappers below have inlined: hand one - ;; finished packet to the fromSPR channel and advance the main-memory write cursor by exactly what - ;; it will deliver. The scratchpad address register takes only the low fourteen bits of an offset - ;; into the page, which is what the mask is for. - ;; - ;; The wait is before programming the channel, not after starting it, so the copy overlaps whatever - ;; the caller does next. Each busy poll is charged to saves.from-spr-waits, and that is how a frame - ;; spent waiting on scratchpad bandwidth shows up in the timing display instead of looking like - ;; slow conversion. - (asm-block address-the-channel - (rlet ((from-spr :reg a2 :type dma-bank-spr)) - (label generic-dma-from-spr-entry) - (nop!) - (lui at #x7000) - (lui from-spr #x1000) - (l.wu v1 at (generic-work-offset saves basep)) - (ori from-spr from-spr #xd000) - (l.w t0 (-> from-spr chcr)) - (nop!) - (add.i a3 at (generic-work-offset saves from-spr-waits)) - (and.i a0 a0 #x3fff) - (and.i t0 t0 DMA-CHCR-STR) - (nop!) - (b.z t0 generic-dma-from-spr-start-dma :delay (nop!)) - (m t0 from-spr) - (nop!))) - (asm-block wait-for-dma - (label generic-dma-from-spr-wait-for-dma) - (l.w t1 a3) - (nop!) - (l.w t2 t0) - (nop!) - (and.i t2 t2 DMA-CHCR-STR) - (add.i t1 t1 1) - (b.nz t2 generic-dma-from-spr-wait-for-dma :delay (s.w t1 a3)) - (m a3 r0)) - ;; Program the channel, advance saves.basep by exactly what it will deliver, and switch to the - ;; other output buffer. Nothing waits for this transfer - the wait above is what makes that safe. - (asm-block start-dma - (rlet ((from-spr :reg a2 :type dma-bank-spr)) - (label generic-dma-from-spr-start-dma) - (sll a3 a1 4) - (s.w a0 (-> from-spr sadr)) - (nop!) - (s.w v1 (-> from-spr madr)) - (add.i a0 r0 DMA-CHCR-STR) - (s.w a1 (-> from-spr qwc)) - (add v1 v1 a3) - (s.w a0 (-> from-spr chcr)) - (nop!) - (s.w v1 at (generic-work-offset saves basep)) - (m v0 r0) - (jr ra :delay (add sp sp r0)) - (nop!) - (nop!))) - ) +;; Build the base and environment fragments for a lit environment-mapped model in one scratchpad +;; block. Both fragments share the transform, strip topology, and positions. The base fragment owns +;; the V3-32 payload; the environment fragment carries only V4-8 tint and V2-16 reflected coordinates, +;; then a DMA REF reads the base positions from their future address in the main-memory output buffer: +;; saves.basep plus the position payload's offset from cur-outbuf. +;; +;; The base streams use the current VU1 input bank. An inline VIF switch launches that fragment and +;; uploads the environment header to the alternate header buffer. The environment attributes and +;; referenced positions use the next input bank, and a final CNT tag launches the second fragment. +;; Two input banks are consumed. The two header-buffer flips cancel, so gifbuf-adr is unchanged. +(defun generic-prepare-dma-double-new () + "Lay out paired base/environment VIF fragments, with the second reusing the first's positions." + (let* ((work (-> (scratchpad-object terrain-context) work foreground generic-work)) + (saves (-> work saves)) + (consts (-> work fx-buf work consts)) + (gsf (-> saves gsf-buf)) + (base-packet (the-as generic-texbuf (-> saves cur-outbuf))) + (out (the-as pointer base-packet)) + (out-address (the-as uint32 base-packet)) + (base-header (-> base-packet header)) + (num-strips (the-as int (-> gsf header num-strips))) + (num-dps (the-as int (-> gsf header num-dps))) + (padded-dps (generic-rounded-draw-points num-dps)) + (shader-span (+ 112 (* num-strips 80))) + (position-bytes (* padded-dps 12)) + (attribute-bytes (* padded-dps 4)) + (inbuf0 (the-as int (-> saves inbuf-adr))) + (inbuf1 (generic-next-inbuf inbuf0)) + (inbuf2 (generic-next-inbuf inbuf1)) + (gifbuf0 (the-as int (-> saves gifbuf-adr))) + (gifbuf1 (logxor gifbuf0 GENERIC-VU1-HEADER-FLIP)) + (header-qwc (+ 7 (* num-strips 5)))) + ;; Derive every cursor before writing: each fragment has its own header/shader span and attribute + ;; payloads, while the base position span appears only once. + (let* ((position-stream (the-as uint32 (&+ out (+ shader-span 32)))) + (color-stream (the-as uint32 (&+ out (+ shader-span position-bytes 64)))) + (texture-stream + (the-as uint32 (&+ out (+ shader-span position-bytes attribute-bytes 80)))) + (env-packet + (&+ out (+ shader-span position-bytes (* attribute-bytes 2) 80))) + (env-color-stream + (the-as uint32 (&+ env-packet (+ shader-span 32)))) + (env-texture-stream + (the-as uint32 (&+ env-packet (+ shader-span attribute-bytes 48)))) + (position-ref + (&+ env-packet (+ shader-span (* attribute-bytes 2) 48)))) + ;; Seed the outer CNT tag and the inline VIF switch between fragments. + (set! (-> base-packet tag quad) (-> consts dma-header quad)) + (let ((env-prefix (the-as dma-packet env-packet)) + (env-switch (the-as qword env-packet)) + (position-ref-tags (the-as (inline-array dma-packet) position-ref)) + (vif-header-template (the-as qword (&-> consts vif-header 0)))) + ;; The switch launches the base fragment and uploads the environment header. The REF then + ;; fetches the copied-out base positions, and the following CNT carries the second MSCAL. + (set! (-> env-switch quad) (-> vif-header-template quad)) + (set! (-> env-switch word 0) (-> consts mscal-tag)) + (set! (-> env-switch word 1) (-> consts reset-cycle-tag)) + (set! (-> env-prefix vif1) + (the-as vif-tag (logior (the-as uint (-> env-prefix vif1)) gifbuf1 (shl header-qwc 16)))) + (set! (-> position-ref-tags 0 quad) (-> consts dma-ref-vtxs quad)) + (set! (-> position-ref-tags 1 quad) (-> consts dma-cnt-call quad)) + (set! (-> position-ref-tags 0 dma) + (new 'static + 'dma-tag + :id (dma-tag-id ref) + :qwc (/ position-bytes 16) + :addr (the-as int (+ (-> saves basep) (- position-stream out-address))))) + (set! (-> position-ref-tags 0 vif1) + (the-as vif-tag + (logior (-> consts unpack-vtx-tag) (shl padded-dps 16) (+ inbuf1 2)))) + (set! (-> position-ref-tags 1 vif0) (the-as vif-tag (-> consts reset-cycle-tag))) + (set! (-> position-ref-tags 1 vif1) (the-as vif-tag (-> consts mscal-tag)))) - (defun upload-vu0-program ((func vu-function) (wait-counter pointer)) - "Upload func to VU0 in blocks of at most 127 instruction pairs, waiting for VIF0 DMA to become idle - and charging every busy poll to wait-counter. The DMA construction is highly optimized." - (declare (asm-func none)) - ;; VIF0's MPG count field is eight bits and counts instruction pairs, so a program longer than 127 - ;; pairs cannot be one unpack. The loop peels 127 pairs at a time and emits a REF tag per block, - ;; filling in each block's source address, pair count and destination instruction address as it - ;; goes, then appends an END tag. Both the Generic and the mercneric VU0 programs are longer than - ;; 127 pairs, which is why this exists rather than a single tag. - ;; - ;; The two cache write-backs are the load-bearing part: the chain was just written by ordinary EE - ;; stores and is about to be read by the DMA controller, which does not see the data cache. - ;; After VIF0 becomes idle, clear its qword count, install the chain address, and start chain mode. - ;; Each busy poll increments wait-counter. - (asm-block set-up-upload-chain - (label upload-vu0-program-entry) - (m! v1 *vu0-dma-list*) - (lui a2 #x3000) - (l.wu a3 a0 8) - (lui t0 #x1000) - (l.wu t1 a0 4) - (add.i v1 v1 12) - (add.i t2 a0 16) - (m a0 v1) - (lui t3 #x4a00)) - ;; One DMA tag per upload block. VIF0's unpack count is eight bits, so a program longer than 127 - ;; instruction pairs has to arrive as several tags; the loop peels 127 pairs at a time and fills in - ;; each tag's address, count and MPG destination as it goes. The two cache write-backs at the end - ;; are for the chain itself, which the EE has just written and the DMA controller is about to read. - (asm-block append-upload-block - (rlet ((vif0 :reg a2 :type dma-bank-vif)) - (label upload-vu0-program-append-upload-block) - (add.i t4 a3 -127) - (s.w a2 a0) - (max.w t5 t4 r0) - (s.w t2 a0 4) - (sub t4 a3 t5) - (s.w t0 a0 8) - (m a3 t5) - (s.b t4 a0) - (sll t6 t4 17) - (sll t5 t4 1) - (add t6 t6 t1) - (add t1 t1 t5) - (add t5 t6 t3) - (sll t4 t4 4) - (s.w t5 a0 12) - (add t2 t2 t4) - (b.nz a3 upload-vu0-program-append-upload-block :delay (add.i a0 a0 16)) - (lui a2 #x7000) - (lui a3 #x1000) - (s.d a2 a0) - (ori vif0 a3 #x8000) - (s.d r0 a0 8) - (l.w a3 (-> vif0 chcr)) - (m t0 v1) - (sync.l) - (cache dxwbin t0 0) - (sync.l) - (cache dxwbin t0 1) - (sync.l) - (m t0 r0) - (sync.l) - (cache dxwbin a0 0) - (sync.l) - (cache dxwbin a0 1) - (sync.l) - (m a0 r0) - (and.i a0 a3 DMA-CHCR-STR) - (b.z a0 upload-vu0-program-start-vif0 :delay (add.i a0 r0 325)) - (m a3 vif0) - (nop!))) - ;; Charge every busy poll to the caller's counter. a3 is the channel again by now, copied out of - ;; a2 so the store that starts it can use the delay slots either side. - (asm-block wait-for-vif0 - (rlet ((vif0-channel :reg a3 :type dma-bank-vif) - (stall-count-ptr :reg a1) - (stall-count :reg t0) - (status :reg t1)) - (label upload-vu0-program-wait-for-vif0) - (l.w stall-count stall-count-ptr) - (nop!) - (l.w status (-> vif0-channel chcr)) - (nop!) - (and.i status status DMA-CHCR-STR) - (add.i stall-count stall-count 1) - (b.nz status upload-vu0-program-wait-for-vif0 - :delay (s.w stall-count stall-count-ptr)) - (m stall-count-ptr r0))) - ;; Chain mode with the tag transfer enabled, and a sync either side of every register write: this - ;; is the one place the converter hands the EE's own cache-line writes straight to a DMA channel. - (asm-block start-vif0 - (rlet ((vif0 :reg a2 :type dma-bank-vif)) - (label upload-vu0-program-start-vif0) - (sync.l) - (s.w r0 (-> vif0 qwc)) - (s.w v1 (-> vif0 tadr)) - (sync.l) - (s.w a0 (-> vif0 chcr)) - (sync.l) - (m v0 r0) - (jr ra :delay (add sp sp r0)) - (nop!) - (nop!))) - ) + (set! (-> base-packet tag vif1) + (the-as vif-tag + (logior (the-as uint (-> base-packet tag vif1)) gifbuf0 (shl header-qwc 16)))) -) + ;; Write the base position/color/coordinate UNPACKs and the environment color/coordinate + ;; UNPACKs. Cleared qwords between payloads are interpreted as VIF NOPs. + (generic-write-stream-tag! + position-stream + (-> consts stcycle-tag) + (logior (-> consts unpack-vtx-tag) (shl padded-dps 16) (+ inbuf0 2))) + (set! (-> (the-as (pointer uint128) (- color-stream 32)) 0) (the-as uint128 0)) + (generic-write-stream-tag! + color-stream (the-as uint 0) (logior (-> consts unpack-clr-tag) (shl padded-dps 16) (+ inbuf0 1))) + (set! (-> (the-as (pointer uint128) (- texture-stream 32)) 0) (the-as uint128 0)) + (generic-write-stream-tag! + texture-stream (the-as uint 0) (logior (-> consts unpack-tex-tag) (shl padded-dps 16) inbuf0)) + (set! (-> (the-as (pointer uint128) (&+ env-packet -16)) 0) (the-as uint128 0)) + (generic-write-stream-tag! + env-color-stream + (-> consts stcycle-tag) + (logior (-> consts unpack-clr-tag) (shl padded-dps 16) (+ inbuf1 1))) + (set! (-> (the-as (pointer uint128) (- env-texture-stream 32)) 0) (the-as uint128 0)) + (generic-write-stream-tag! + env-texture-stream (the-as uint 0) (logior (-> consts unpack-tex-tag) (shl padded-dps 16) inbuf1)) + (set! (-> (the-as (pointer uint128) (&+ position-ref -16)) 0) (the-as uint128 0)) -(#when PC_PORT - (def-mips2c generic-prepare-dma-single (function none)) - (def-mips2c generic-prepare-dma-double (function none)) - (def-mips2c generic-light-proc (function none)) - (def-mips2c generic-envmap-proc (function none))) + (let ((env-header (-> (the-as generic-texbuf env-packet) header)) + (base-shaders (the-as (inline-array adgif-shader) (&+ out 128))) + (env-shaders (the-as (inline-array adgif-shader) (&+ env-packet 128)))) + (cond + ;; A fresh base shader list requires both complete headers and both per-strip shader lists. + ((nonzero? (-> saves ptr-shaders)) + (matrix-copy! (-> base-header matrix) (-> consts matrix)) + (matrix-copy! (-> env-header matrix) (-> consts matrix)) + (set! (-> base-header strgif qword quad) (-> consts base-strgif qword quad)) + (set! (-> base-header adnop1 quad) (-> consts adcmds 0 quad)) + (set! (-> base-header adnop2 quad) (-> consts adcmds 3 quad)) + (set! (-> env-header strgif qword quad) (-> consts envmap strgif qword quad)) + (set! (-> env-header adnop1 quad) (-> consts adcmds 3 quad)) + (set! (-> env-header adnop2 quad) (-> consts adcmds 3 quad)) + (let ((strip-table (the-as (pointer uint8) (&-> gsf header strip-table 0))) + (base-source (the-as (inline-array adgif-shader) (-> saves ptr-shaders))) + (env-source (the-as (inline-array adgif-shader) (-> saves ptr-env-shader))) + (kick-offset 0)) + (dotimes (strip num-strips) + (let ((strip-count (the-as int (-> strip-table strip)))) + (generic-copy-effect-shader! + (-> base-shaders strip) (-> base-source strip) kick-offset strip-count) + (generic-copy-effect-shader! + (-> env-shaders strip) (-> env-source 0) kick-offset strip-count) + (+! kick-offset (+ 9 (* strip-count 3))))) + (when (nonzero? num-strips) + (logior! (-> base-shaders (- num-strips 1) quad 1 word 3) #x8000) + (logior! (-> env-shaders (- num-strips 1) quad 1 word 3) #x8000))) + (set! (-> base-header strips) (the-as uint num-strips)) + (set! (-> base-header dps) (the-as uint num-dps)) + (set! (-> base-header kickoff) 0)) + ;; With no new base shader pointer, retain the prepared base fragment and derive the + ;; environment fragment from its transform and per-strip kick/count words. + (else + (matrix-copy! (-> env-header matrix) (-> base-header matrix)) + (set! (-> env-header strgif qword quad) (-> consts envmap strgif qword quad)) + (set! (-> env-header adnop1 quad) (-> consts adcmds 3 quad)) + (set! (-> env-header adnop2 quad) (-> consts adcmds 3 quad)) + (let ((env-source (the-as (inline-array adgif-shader) (-> saves ptr-env-shader)))) + (dotimes (strip num-strips) + (generic-copy-effect-shader! + (-> env-shaders strip) + (-> env-source 0) + (the-as int (-> base-shaders strip quad 0 word 3)) + (the-as int (-> base-shaders strip quad 1 word 3))))))) + (set! (-> env-header strips) (the-as uint num-strips)) + (set! (-> env-header dps) (the-as uint num-dps)) + (set! (-> env-header kickoff) 0)) + + ;; Include the final REF and CNT tags in the flat fromSPR span, advance across two VU1 input + ;; banks, and publish the five payload cursors for the two processors. + (let ((dma-qwc (- (/ (&- position-ref out) 16) 1))) + (set! (-> base-packet tag dma qwc) (the-as uint dma-qwc)) + (set! (-> saves qwc) (the-as uint (+ dma-qwc 3)))) + (set! (-> saves inbuf-adr) (the-as uint inbuf2)) + (set! (-> saves num-dps) (the-as uint num-dps)) + (set! (-> saves ptr-vtxs) position-stream) + (set! (-> saves ptr-clrs) color-stream) + (set! (-> saves ptr-texs) texture-stream) + (set! (-> saves ptr-env-clrs) env-color-stream) + (set! (-> saves ptr-env-texs) env-texture-stream) + 0 + (none)))) + + + +;; Fill a prepared base fragment from the main-memory GSF source. ptr-iks supplies draw order and the +;; no-kick bit; each index selects one expanded record from ptr-verts. Copy camera-space XYZ into the +;; dense V3-32 payload and authored ST into V2-16, reserving bit 0 of S for no-kick. For color, clamp +;; each normal/light dot product at zero, combine three directional colors with ambient, modulate the +;; source RGBA channels, clamp them at 255, and pack one V4-8 value per draw point. +;; +;; The padded iterations intentionally consume the converter's trailing index entries so all three +;; output payloads exactly match the VIF NUM written by the packet builder. +(defun generic-light-proc-new () + "Gather and light indexed GSF draw points into a prepared fragment's base vertex payloads." + (let* ((work (-> (scratchpad-object terrain-context) work foreground generic-work)) + (saves (-> work saves)) + (lights (-> work fx-buf work lights)) + (gsf (-> saves gsf-buf)) + ;; These are dense wire-format arrays, not arrays of GOAL object references. + (indices (the-as (inline-array gsf-ik-packed) (-> gsf info ptr-iks))) + (vertices (the-as (inline-array gsf-vertex) (-> gsf info ptr-verts))) + (positions (the-as (inline-array vector3s-packed) (-> saves ptr-vtxs))) + (colors (the-as (inline-array vector4ub) (-> saves ptr-clrs))) + (textures (the-as (inline-array vector2uh) (-> saves ptr-texs))) + (num-dps (the-as int (-> saves num-dps))) + (padded-dps (generic-rounded-draw-points num-dps))) + (dotimes (draw-point padded-dps) + (let* ((index (-> indices draw-point)) + (vertex (-> vertices (-> index index))) + (source-position (-> vertex pos)) + (normal (-> vertex nrm)) + (source-texture (-> vertex tex)) + (source-color (-> vertex clr)) + (output-position (-> positions draw-point)) + (output-texture (-> textures draw-point)) + (output-color (-> colors draw-point)) + (nx (-> normal x)) + (ny (-> normal y)) + (nz (-> normal z)) + (light0 (fmax 0.0 (+ (* nx (-> lights direction 0 x)) + (* ny (-> lights direction 1 x)) + (* nz (-> lights direction 2 x))))) + (light1 (fmax 0.0 (+ (* nx (-> lights direction 0 y)) + (* ny (-> lights direction 1 y)) + (* nz (-> lights direction 2 y))))) + (light2 (fmax 0.0 (+ (* nx (-> lights direction 0 z)) + (* ny (-> lights direction 1 z)) + (* nz (-> lights direction 2 z))))) + (red (fmin 255.0 (* (the float (-> source-color x)) + (+ (-> lights ambient x) (* light0 (-> lights color 0 x)) + (* light1 (-> lights color 1 x)) (* light2 (-> lights color 2 x)))))) + (green (fmin 255.0 (* (the float (-> source-color y)) + (+ (-> lights ambient y) (* light0 (-> lights color 0 y)) + (* light1 (-> lights color 1 y)) (* light2 (-> lights color 2 y)))))) + (blue (fmin 255.0 (* (the float (-> source-color z)) + (+ (-> lights ambient z) (* light0 (-> lights color 0 z)) + (* light1 (-> lights color 1 z)) (* light2 (-> lights color 2 z)))))) + (alpha (fmin 255.0 (* (the float (-> source-color w)) + (+ (-> lights ambient w) (* light0 (-> lights color 0 w)) + (* light1 (-> lights color 1 w)) (* light2 (-> lights color 2 w))))))) + (set! (-> output-position x) (-> source-position x)) + (set! (-> output-position y) (-> source-position y)) + (set! (-> output-position z) (-> source-position z)) + (set! (-> output-texture x) + (the-as uint (logior (logand (-> source-texture x) #xfffe) + (logand (-> index no-kick) 1)))) + (set! (-> output-texture y) (the-as uint (logand (-> source-texture y) #xfffe))) + (set! (-> output-color clr) + (logior (the-as uint (the int red)) + (shl (the int green) 8) + (shl (the int blue) 16) + (shl (the int alpha) 24))))) + 0 + (none))) + +;; Fill a prepared environment fragment from the same indexed GSF source as the base pass. Position +;; and normal are already in camera space. Let m = normal - (0, 0, 1), then reflect the eye vector: +;; +;; r = position + m * dot(m, position) / m.z +;; +;; Normalize r, map XY from [-1, 1] to [0, 1], and scale to the V2-16 stream's 12-bit coordinate +;; range. Copy the fragment's packed environment tint to the V4-8 stream and place no-kick in bit 0 +;; of S. This pass emits no positions; its packet uses the base fragment's DMA REF instead. As with +;; the base processor, padded iterations make the payload sizes agree with their VIF NUM fields. +(defun generic-envmap-proc-new () + "Generate reflected coordinates and packed tint for a prepared environment fragment." + (let* ((work (-> (scratchpad-object terrain-context) work foreground generic-work)) + (saves (-> work saves)) + (consts (-> work fx-buf work consts)) + (gsf (-> saves gsf-buf)) + (indices (the-as (inline-array gsf-ik-packed) (-> gsf info ptr-iks))) + (vertices (the-as (inline-array gsf-vertex) (-> gsf info ptr-verts))) + (colors (the-as (inline-array vector4ub) (-> saves ptr-env-clrs))) + (textures (the-as (inline-array vector2uh) (-> saves ptr-env-texs))) + (num-dps (the-as int (-> saves num-dps))) + (padded-dps (generic-rounded-draw-points num-dps))) + (dotimes (draw-point padded-dps) + (let* ((index (-> indices draw-point)) + (vertex (-> vertices (-> index index))) + (position (-> vertex pos)) + (normal (-> vertex nrm)) + (output-color (-> colors draw-point)) + (output-texture (-> textures draw-point)) + (mx (-> normal x)) + (my (-> normal y)) + (mz (- (-> normal z) 1.0)) + (reflection-scale (/ (+ (* mx (-> position x)) + (* my (-> position y)) + (* mz (-> position z))) + mz)) + (rx (+ (-> position x) (* mx reflection-scale))) + (ry (+ (-> position y) (* my reflection-scale))) + (rz (+ (-> position z) (* mz reflection-scale))) + (inverse-length (/ 1.0 (sqrtf (+ (square rx) (square ry) (square rz))))) + (s (the int (* 4096.0 (+ 0.5 (* 0.5 rx inverse-length))))) + (t (the int (* 4096.0 (+ 0.5 (* 0.5 ry inverse-length)))))) + (set! (-> output-color clr) (the-as uint (-> consts envmap colors x))) + (set! (-> output-texture x) + (the-as uint (logior (logand s #xfffe) + (logand (-> index no-kick) 1)))) + (set! (-> output-texture y) (the-as uint (logand t #xfffe))))) + 0 + (none))) + +(defun generic-note-prepare-dma-single () + "Count one single-fragment packet build when portable-effect statistics are enabled." + (when *generic-effect-stats* (+! (-> *generic-effect-debug-stats-data* prepare-single-calls) 1)) + (none)) + +(defun generic-note-prepare-dma-double () + "Count one paired base/environment packet build when portable-effect statistics are enabled." + (when *generic-effect-stats* (+! (-> *generic-effect-debug-stats-data* prepare-double-calls) 1)) + (none)) + +(defun generic-note-light-proc () + "Add the current fragment's draw-point count to the portable lighting statistic." + (when *generic-effect-stats* + (+! (-> *generic-effect-debug-stats-data* light-vertices) + (-> (scratchpad-object terrain-context) work foreground generic-work saves num-dps))) + (none)) + +(defun generic-note-envmap-proc () + "Add the current fragment's draw-point count to the portable environment statistic." + (when *generic-effect-stats* + (+! (-> *generic-effect-debug-stats-data* envmap-vertices) + (-> (scratchpad-object terrain-context) work foreground generic-work saves num-dps))) + (none)) diff --git a/goal_src/jak1/engine/gfx/generic/generic-h.gc b/goal_src/jak1/engine/gfx/generic/generic-h.gc index c45ec60663..374bb7073d 100644 --- a/goal_src/jak1/engine/gfx/generic/generic-h.gc +++ b/goal_src/jak1/engine/gfx/generic/generic-h.gc @@ -124,7 +124,8 @@ ;; An indexed vertex reference and whether it suppresses the strip kick. ;; One per draw point; together they are the drawing order. ((index uint8) - (no-kick uint8))) + (no-kick uint8)) + :pack-me) ;; Pointers installed at the front of a GSF buffer before the work area is built. The converter lays ;; the three arrays out back to back and publishes them here, so neither the effect processors nor @@ -253,7 +254,7 @@ (ztest-normal ad-cmd :inline) (ztest-opaque ad-cmd :inline) (adcmd-offsets uint8 16) ;; which of the four A+D commands each effect combination needs - (adcmds ad-cmd 4 :overlay-at alpha-opaque) + (adcmds ad-cmd 4 :inline :overlay-at alpha-opaque) (stcycle-tag uint32) ;; STCYCL CL=3 WL=1, which interleaves the three vertex streams (unpack-vtx-tag uint32) ;; UNPACK V3-32, float positions (unpack-clr-tag uint32) ;; UNPACK V4-8 unsigned, colors already in GS RGBAQ order diff --git a/goal_src/jak1/engine/gfx/generic/generic-merc.gc b/goal_src/jak1/engine/gfx/generic/generic-merc.gc index 4e04a7c4be..f20b51020d 100644 --- a/goal_src/jak1/engine/gfx/generic/generic-merc.gc +++ b/goal_src/jak1/engine/gfx/generic/generic-merc.gc @@ -3279,6 +3279,13 @@ insert each completed chain in its foreground bucket, and update wait and DMA-memory statistics." (local-vars (a0-26 int) (a0-28 int)) (when (nonzero? (-> *merc-global-array* count)) + (when *generic-effect-stats* + (set! (-> *generic-effect-debug-stats-data* light-vertices) 0) + (set! (-> *generic-effect-debug-stats-data* envmap-vertices) 0) + (set! (-> *generic-effect-debug-stats-data* prepare-single-calls) 0) + (set! (-> *generic-effect-debug-stats-data* prepare-double-calls) 0) + (stopwatch-init (-> *generic-effect-debug-stats-data* timer)) + (stopwatch-start (-> *generic-effect-debug-stats-data* timer))) (let ((global-buffer-start (-> *display* frames (-> *display* on-screen) frame global-buf base))) ;; set up performance stats (if *debug-segment* @@ -3302,6 +3309,13 @@ ;; - high-speed-reject ;; and also loads the mercneric-vu0-block block with an offset of 280. (generic-merc-init-asm) + ;; Publish the GOAL implementations. MIPS2C calls these addresses through its ordinary + ;; indirect-call bridge. + (let ((calls (-> (scratchpad-object terrain-context) work foreground generic-work in-buf merc shadow))) + (set! (-> calls generic-prepare-dma-single) generic-prepare-dma-single-new) + (set! (-> calls generic-prepare-dma-double) generic-prepare-dma-double-new) + (set! (-> calls generic-light-proc) generic-light-proc-new) + (set! (-> calls generic-envmap-proc) generic-envmap-proc-new)) ;; set a limit, so we don't write off the end of the dma buffer. (set! (-> (scratchpad-object terrain-context) work foreground generic-work in-buf merc shadow write-limit) (&+ (-> dma-buf end) -65536)) @@ -3358,5 +3372,15 @@ (+! (-> dma-usage data 86 used) (&- (-> *display* frames (-> *display* on-screen) frame global-buf base) (the-as uint global-buffer-start))) - (set! (-> dma-usage data 86 total) (-> dma-usage data 86 used)))))) + (set! (-> dma-usage data 86 total) (-> dma-usage data 86 used))))) + (when *generic-effect-stats* + (stopwatch-stop (-> *generic-effect-debug-stats-data* timer)) + (format *stdcon* + "generic: ~D us, light ~D verts, envmap ~D verts, prepare single ~D, double ~D~%" + (the int (* 1000000.0 + (stopwatch-elapsed-seconds (-> *generic-effect-debug-stats-data* timer)))) + (-> *generic-effect-debug-stats-data* light-vertices) + (-> *generic-effect-debug-stats-data* envmap-vertices) + (-> *generic-effect-debug-stats-data* prepare-single-calls) + (-> *generic-effect-debug-stats-data* prepare-double-calls)))) (none)) diff --git a/goal_src/jak1/engine/gfx/generic/generic-tie.gc b/goal_src/jak1/engine/gfx/generic/generic-tie.gc index df78d535d2..bce7bdff9e 100644 --- a/goal_src/jak1/engine/gfx/generic/generic-tie.gc +++ b/goal_src/jak1/engine/gfx/generic/generic-tie.gc @@ -3347,10 +3347,12 @@ ;; cached as addresses. TIE always wants this exact set: two packet headers, reflected ;; coordinates, the subdivision blend, and an unlit color combine. (let ((calls (-> (scratchpad-object terrain-context) work foreground generic-work in-buf tie shadow calls))) - (set! (-> calls generic-prepare-dma-double) generic-prepare-dma-double) - (set! (-> calls generic-envmap-dproc) generic-envmap-dproc) - (set! (-> calls generic-interp-dproc) generic-interp-dproc) - (set! (-> calls generic-no-light-dproc) generic-no-light-dproc)) + (set! (-> calls generic-prepare-dma-double) generic-prepare-dma-double-new) + ;; The Generic TIE/dproc path is currently disabled and its PC implementations are absent. + ;; (set! (-> calls generic-envmap-dproc) generic-envmap-dproc) + ;; (set! (-> calls generic-interp-dproc) generic-interp-dproc) + ;; (set! (-> calls generic-no-light-dproc) generic-no-light-dproc) + ) (set! (-> (scratchpad-object terrain-context) work foreground generic-work saves time-of-day-color r) (the int (-> *time-of-day-context* current-sun env-color x))) (set! (-> (scratchpad-object terrain-context) work foreground generic-work saves time-of-day-color g)