refactor: refactor VU1 (#191)

* feat: implement fix and changes based on dark cloud report
fix: fix GS AFAIL for RGB/alpha/Z, ZMSK
fix: fix VU1 flags mask and pipeline
fix: small VU1 cache fix
feat: __ct__, __sinit_ are not sillent stubs anymore

* feat: fix song JP pulling

* feat: sound update for lotR

* feat: prevent guest execution to be very slow

* fix: small gs size bug

* feat: refactor VU
fix: fix cliping and other issues on gs
fix: fix wrong vu0 register on recompiler

* fix fix ACC scheduler stall
feat: remove unused test
fix: .fix overflow e underflow on FMAC

* feat: small setting  for windows test
This commit is contained in:
Ranieri
2026-08-05 14:50:24 -03:00
committed by GitHub
parent 61300792a0
commit f49ca4edbc
34 changed files with 5506 additions and 1015 deletions
+126 -22
View File
@@ -7,6 +7,7 @@
#include <fstream>
#include <regex>
#include <sstream>
#include <utility>
using namespace ps2recomp;
@@ -823,32 +824,71 @@ void register_code_generator_tests()
t.IsTrue(ctc1Code.find("ignored") == std::string::npos, "CTC1 FCR31 should not be ignored");
});
tc.Run("VU CReg access uses CFC2/CTC2", [](TestCase &t) {
CodeGenerator gen({}, {});
tc.Run("VU CFC2/CTC2 access VI registers directly", [](TestCase& t)
{
CodeGenerator gen({}, {});
Instruction cfc2{};
cfc2.opcode = OPCODE_COP2;
cfc2.rs = COP2_CFC2;
cfc2.rt = 2;
cfc2.rd = VU0_CR_STATUS;
Instruction cfc2{};
cfc2.opcode = OPCODE_COP2;
cfc2.rs = COP2_CFC2;
cfc2.rt = 2;
cfc2.rd = 11;
std::string cfc2Code = gen.translateInstruction(cfc2);
printGeneratedCode("VU CReg access uses CFC2/CTC2 (CFC2)", cfc2Code);
t.IsTrue(cfc2Code.find("SET_GPR_U32(ctx, 2") != std::string::npos, "CFC2 should write to rt");
t.IsTrue(cfc2Code.find("ctx->vu0_status") != std::string::npos, "CFC2 STATUS should read vu0_status");
t.IsTrue(cfc2Code.find("Unimplemented CFC2 VU CReg") == std::string::npos, "CFC2 should not hit unimplemented CReg path");
std::string cfc2Code = gen.translateInstruction(cfc2);
printGeneratedCode("VU CFC2/CTC2 access VI registers directly (CFC2)", cfc2Code);
Instruction ctc2{};
ctc2.opcode = OPCODE_COP2;
ctc2.rs = COP2_CTC2;
ctc2.rt = 3;
ctc2.rd = VU0_CR_ITOP;
t.IsTrue(cfc2Code.find("SET_GPR_U32(ctx, 2") != std::string::npos, "CFC2 should write to rt");
std::string ctc2Code = gen.translateInstruction(ctc2);
printGeneratedCode("VU CReg access uses CFC2/CTC2 (CTC2)", ctc2Code);
t.IsTrue(ctc2Code.find("ctx->vu0_itop") != std::string::npos, "CTC2 ITOP should write vu0_itop");
t.IsTrue(ctc2Code.find("GPR_U32(ctx, 3) & 0x3FF") != std::string::npos, "CTC2 ITOP should mask to 10 bits");
t.IsTrue(ctc2Code.find("Unimplemented CTC2 VU CReg") == std::string::npos, "CTC2 should not hit unimplemented CReg path");
t.IsTrue(cfc2Code.find("ctx->vi[11]") != std::string::npos, "CFC2 VI11 should read VI11");
t.IsTrue(cfc2Code.find("vu0_cmsar1") == std::string::npos, "CFC2 VI11 must not read CMSAR1");
t.IsTrue(cfc2Code.find("Unimplemented") == std::string::npos, "CFC2 VI11 should be implemented");
Instruction ctc2{};
ctc2.opcode = OPCODE_COP2;
ctc2.rs = COP2_CTC2;
ctc2.rt = 3;
ctc2.rd = 4;
std::string ctc2Code = gen.translateInstruction(ctc2);
printGeneratedCode("VU CFC2/CTC2 access VI registers directly (CTC2)", ctc2Code);
t.IsTrue(ctc2Code.find("ctx->vi[4]") != std::string::npos, "CTC2 VI4 should write VI4");
t.IsTrue(ctc2Code.find("static_cast<uint16_t>(GPR_U32(ctx, 3))") != std::string::npos, "CTC2 VI4 should store the low 16 bits");
t.IsTrue(ctc2Code.find("vu0_i") == std::string::npos, "CTC2 VI4 must not write the I register");
t.IsTrue(ctc2Code.find("Unimplemented") == std::string::npos, "CTC2 VI4 should be implemented");
});
tc.Run("VU special control registers use hardware indices", [](TestCase& t)
{
CodeGenerator gen({}, {});
Instruction cfc2{};
cfc2.opcode = OPCODE_COP2;
cfc2.rs = COP2_CFC2;
cfc2.rt = 2;
cfc2.rd = VU0_CR_STATUS;
std::string cfc2Code = gen.translateInstruction(cfc2);
printGeneratedCode("VU special control registers use hardware indices (STATUS)", cfc2Code);
t.IsTrue(cfc2Code.find("SET_GPR_U32(ctx, 2") != std::string::npos,"CFC2 should write to rt");
t.IsTrue(cfc2Code.find("ctx->vu0_status") != std::string::npos, "CFC2 STATUS should read vu0_status");
t.IsTrue(cfc2Code.find("Unimplemented") == std::string::npos,"CFC2 STATUS should be implemented");
Instruction ctc2{};
ctc2.opcode = OPCODE_COP2;
ctc2.rs = COP2_CTC2;
ctc2.rt = 3;
ctc2.rd = VU0_CR_FBRST;
std::string ctc2Code = gen.translateInstruction(ctc2);
printGeneratedCode("VU special control registers use hardware indices (FBRST)", ctc2Code);
t.IsTrue(ctc2Code.find("ctx->vu0_fbrst") != std::string::npos, "CTC2 register 28 should write FBRST");
t.IsTrue(ctc2Code.find("vu0_itop") == std::string::npos, "CTC2 register 28 must not write ITOP");
t.IsTrue(ctc2Code.find("Unimplemented") == std::string::npos, "CTC2 FBRST should be implemented");
});
tc.Run("scalar logical immediates emit low64 operations", [](TestCase &t) {
@@ -1095,6 +1135,70 @@ void register_code_generator_tests()
t.IsTrue(out.find("ctx->vu0_vf[25]") == std::string::npos, "S1 q/i must not use rs(format) as register index");
});
tc.Run("VU0 destination MADD and MSUB forms preserve ACC", [](TestCase &t) {
Instruction inst{};
inst.rt = 7;
inst.rd = 11;
inst.sa = 3;
inst.function = 0;
inst.vectorInfo.vectorField = 0xE;
CodeGenerator gen({}, {});
const std::vector<std::pair<const char *, std::string>> emitted = {
{"MADD field", gen.translateVU_VMADD_Field(inst)},
{"MADD", gen.translateVU_VMADD(inst)},
{"MADDq", gen.translateVU_VMADDq(inst)},
{"MADDi", gen.translateVU_VMADDi(inst)},
{"MSUB field", gen.translateVU_VMSUB_Field(inst)},
{"MSUB", gen.translateVU_VMSUB(inst)},
{"MSUBq", gen.translateVU_VMSUBq(inst)},
{"MSUBi", gen.translateVU_VMSUBi(inst)},
{"OPMSUB", gen.translateVU_VOPMSUB(inst)},
};
for (const auto &[name, code] : emitted)
{
const std::string message =
std::string(name) + " writes VF and must not overwrite ACC";
t.IsTrue(code.find("ctx->vu0_acc = res") == std::string::npos,
message.c_str());
t.IsTrue(code.find("PS2_VADD(ctx->vu0_acc") != std::string::npos ||
code.find("PS2_VSUB(ctx->vu0_acc") != std::string::npos,
(std::string(name) + " must still read ACC").c_str());
}
const std::string madda = gen.translateVU_VMADDA(inst);
t.IsTrue(madda.find("ctx->vu0_acc =") != std::string::npos,
"MADDA must continue writing ACC");
});
tc.Run("VU0 OPMULA and OPMSUB use cross-product lane permutations", [](TestCase &t) {
Instruction inst{};
inst.rt = 7;
inst.rd = 11;
inst.sa = 3;
inst.vectorInfo.vectorField = 0xE;
CodeGenerator gen({}, {});
const std::string opmula = gen.translateVU_VOPMULA(inst);
const std::string opmsub = gen.translateVU_VOPMSUB(inst);
for (const std::string *code : {&opmula, &opmsub})
{
t.IsTrue(code->find("_MM_SHUFFLE(3,0,2,1)") != std::string::npos,
"OPM source Fs must be permuted to y,z,x");
t.IsTrue(code->find("_MM_SHUFFLE(3,1,0,2)") != std::string::npos,
"OPM source Ft must be permuted to z,x,y");
t.IsTrue(code->find("PS2_VMUL(fs_yzx, ft_zxy)") != std::string::npos,
"OPM product must use the permuted operands");
}
t.IsTrue(opmula.find("ctx->vu0_acc =") != std::string::npos,
"OPMULA must write the permuted product to ACC");
t.IsTrue(opmsub.find("ctx->vu0_acc = res") == std::string::npos,
"OPMSUB must preserve ACC after producing the cross product");
});
tc.Run("VU0 S2 vector ops use rd as source and rt as destination", [](TestCase &t) {
Instruction inst{};
inst.opcode = OPCODE_COP2;
+625 -69
View File
@@ -5,6 +5,7 @@
#include "ps2_syscalls.h"
#include "runtime/ps2_gs_gpu.h"
#include "runtime/ps2_gs_memory.h"
#include "runtime/ps2_gs_rasterizer.h"
#include "runtime/ps2_gs_psmct32.h"
#include "runtime/ps2_gs_psmt4.h"
#include "runtime/ps2_gs_psmt8.h"
@@ -297,6 +298,57 @@ namespace
t.Equals(probe, expectedBase, message);
runtime.guestFree(probe);
}
struct GsPixelTestResult
{
uint32_t framebuffer = 0u;
uint32_t depth = 0u;
};
GsPixelTestResult drawGsPixelForTests(uint8_t framePsm,
uint64_t testReg,
bool zmask,
uint32_t initialFramebuffer,
uint32_t initialDepth,
uint8_t sourceAlpha)
{
constexpr uint32_t kFrameBlock = 0u;
constexpr uint32_t kDepthBlock = 32u;
constexpr uint32_t kSourceDepth = 0x22222222u;
std::vector<uint8_t> vram(PS2_GS_VRAM_SIZE, 0u);
GS gs;
gs.init(vram.data(), static_cast<uint32_t>(vram.size()), nullptr);
gs.WriteVram(framePsm, kFrameBlock, 1u, 0u, 0u, initialFramebuffer);
gs.WriteVram(GS_PSM_Z32, kDepthBlock, 1u, 0u, 0u, initialDepth);
const uint64_t frame =
(1ull << 16) |
(static_cast<uint64_t>(framePsm) << 24);
const uint64_t zbuf =
1ull |
(static_cast<uint64_t>(zmask ? 1u : 0u) << 32);
const uint64_t rgbaq =
(0x12ull << 0) |
(0x34ull << 8) |
(0x56ull << 16) |
(static_cast<uint64_t>(sourceAlpha) << 24) |
(0x3F800000ull << 32);
gs.writeRegister(GS_REG_FRAME_1, frame);
gs.writeRegister(GS_REG_ZBUF_1, zbuf);
gs.writeRegister(GS_REG_SCISSOR_1, 0ull);
gs.writeRegister(GS_REG_TEST_1, testReg);
gs.writeRegister(GS_REG_PRIM, static_cast<uint64_t>(GS_PRIM_POINT));
gs.writeRegister(GS_REG_RGBAQ, rgbaq);
gs.writeRegister(GS_REG_XYZ2, static_cast<uint64_t>(kSourceDepth) << 32);
return {
gs.ReadVram(framePsm, kFrameBlock, 1u, 0u, 0u),
gs.ReadVram(GS_PSM_Z32, kDepthBlock, 1u, 0u, 0u),
};
}
}
void register_ps2_gs_tests()
@@ -623,6 +675,126 @@ void register_ps2_gs_tests()
"context-targeted clear should leave the other context framebuffer untouched");
});
tc.Run("XYZ3 culls a triangle strip primitive without desynchronizing the vertex queue", [](TestCase &t)
{
std::vector<uint8_t> vram(PS2_GS_VRAM_SIZE, 0u);
GS gs;
gs.init(vram.data(), static_cast<uint32_t>(vram.size()), nullptr);
constexpr uint32_t kColor = 0xFF0000FFu;
constexpr uint64_t kFrame =
(1ull << 16) |
(static_cast<uint64_t>(GS_PSM_CT32) << 24);
constexpr uint64_t kZbuf = (1ull << 32);
constexpr uint64_t kScissor =
(6ull << 16) |
(6ull << 48);
auto xyz = [](uint32_t x, uint32_t y) -> uint64_t
{
return static_cast<uint64_t>(x * 16u) |
(static_cast<uint64_t>(y * 16u) << 16);
};
gs.writeRegister(GS_REG_FRAME_1, kFrame);
gs.writeRegister(GS_REG_ZBUF_1, kZbuf);
gs.writeRegister(GS_REG_SCISSOR_1, kScissor);
gs.writeRegister(GS_REG_XYOFFSET_1, 0ull);
gs.writeRegister(GS_REG_TEST_1, 0x30000ull);
gs.writeRegister(GS_REG_PRIM, static_cast<uint64_t>(GS_PRIM_TRISTRIP));
gs.writeRegister(GS_REG_RGBAQ, kColor);
// ABC is rejected by XYZ3. D must then draw BCD, not stale ABC.
gs.writeRegister(GS_REG_XYZ2, xyz(0u, 0u));
gs.writeRegister(GS_REG_XYZ2, xyz(6u, 0u));
gs.writeRegister(GS_REG_XYZ3, xyz(0u, 6u));
gs.writeRegister(GS_REG_XYZ2, xyz(6u, 6u));
t.Equals(readReferencePSMCT32Pixel(vram, 0u, 1u, 1u, 1u), 0u,
"XYZ3 should suppress the completed ABC triangle");
t.Equals(readReferencePSMCT32Pixel(vram, 0u, 1u, 4u, 4u), kColor,
"the next XYZ2 should draw BCD from the advanced strip queue");
});
tc.Run("GS fog blends the shaded color toward FOGCOL before framebuffer blending", [](TestCase &t)
{
auto renderFoggedPoint = [](bool fogEnabled, uint8_t fog, uint32_t fogColor = 0u) -> uint32_t
{
std::vector<uint8_t> vram(PS2_GS_VRAM_SIZE, 0u);
GS gs;
gs.init(vram.data(), static_cast<uint32_t>(vram.size()), nullptr);
constexpr uint64_t kFrame =
(1ull << 16) |
(static_cast<uint64_t>(GS_PSM_CT32) << 24);
constexpr uint64_t kZbuf = (1ull << 32);
constexpr uint64_t kWhite = 0x80FFFFFFull;
gs.writeRegister(GS_REG_FRAME_1, kFrame);
gs.writeRegister(GS_REG_ZBUF_1, kZbuf);
gs.writeRegister(GS_REG_SCISSOR_1, 0ull);
gs.writeRegister(GS_REG_XYOFFSET_1, 0ull);
gs.writeRegister(GS_REG_TEST_1, 0x30000ull);
gs.writeRegister(GS_REG_FOGCOL, fogColor);
gs.writeRegister(
GS_REG_PRIM,
static_cast<uint64_t>(GS_PRIM_POINT) |
(static_cast<uint64_t>(fogEnabled ? 1u : 0u) << 5));
gs.writeRegister(GS_REG_RGBAQ, kWhite);
gs.writeRegister(GS_REG_FOG, static_cast<uint64_t>(fog) << 56);
gs.writeRegister(GS_REG_XYZ2, 0ull);
return readReferencePSMCT32Pixel(vram, 0u, 1u, 0u, 0u);
};
t.Equals(renderFoggedPoint(false, 0x80u), 0x80FFFFFFu,
"FOG and FOGCOL must not affect primitives with FGE disabled");
t.Equals(renderFoggedPoint(true, 0x80u), 0x807F7F7Fu,
"F=0x80 over black FOGCOL should halve the point RGB and preserve alpha");
t.Equals(renderFoggedPoint(true, 0x00u), 0x80000000u,
"F=0 should replace the point RGB with black FOGCOL");
t.Equals(renderFoggedPoint(true, 0x00u, 0x00302010u), 0x802F1F0Fu,
"F=0 should replace point RGB with the programmed FOGCOL");
});
tc.Run("PRMODE supplies primitive attributes while PRMODECONT AC is clear", [](TestCase &t)
{
auto renderPoint = [](bool usePrmodeAttributes) -> uint32_t
{
std::vector<uint8_t> vram(PS2_GS_VRAM_SIZE, 0u);
GS gs;
gs.init(vram.data(), static_cast<uint32_t>(vram.size()), nullptr);
constexpr uint64_t kFrame =
(1ull << 16) |
(static_cast<uint64_t>(GS_PSM_CT32) << 24);
constexpr uint64_t kZbuf = (1ull << 32);
gs.writeRegister(GS_REG_FRAME_1, kFrame);
gs.writeRegister(GS_REG_ZBUF_1, kZbuf);
gs.writeRegister(GS_REG_SCISSOR_1, 0ull);
gs.writeRegister(GS_REG_XYOFFSET_1, 0ull);
gs.writeRegister(GS_REG_TEST_1, 0x30000ull);
gs.writeRegister(GS_REG_FOGCOL, 0ull);
gs.writeRegister(GS_REG_RGBAQ, 0x80FFFFFFull);
gs.writeRegister(GS_REG_FOG, 0ull);
gs.writeRegister(GS_REG_PRMODE, 1ull << 5);
gs.writeRegister(GS_REG_PRMODECONT, usePrmodeAttributes ? 0ull : 1ull);
// FGE is clear in PRIM. AC decides whether that clear bit or
// PRMODE's set bit supplies the effective fog enable.
gs.writeRegister(GS_REG_PRIM, static_cast<uint64_t>(GS_PRIM_POINT));
gs.writeRegister(GS_REG_XYZ2, 0ull);
return readReferencePSMCT32Pixel(vram, 0u, 1u, 0u, 0u);
};
t.Equals(renderPoint(true), 0x80000000u,
"AC=0 should retain FGE from PRMODE across a PRIM write");
t.Equals(renderPoint(false), 0x80FFFFFFu,
"AC=1 should source FGE from PRIM instead of PRMODE");
});
tc.Run("PABE bypasses alpha blend for low-alpha source pixels", [](TestCase &t)
{
std::vector<uint8_t> vram(PS2_GS_VRAM_SIZE, 0u);
@@ -1557,6 +1729,20 @@ void register_ps2_gs_tests()
t.Equals(regs.display2, display2, "A+D should write GS DISPLAY2");
});
tc.Run("reserved PSM 0x3F uses null VRAM handlers", [](TestCase &t)
{
std::vector<uint8_t> vram(PS2_GS_VRAM_SIZE, 0xA5u);
GS gs;
gs.init(vram.data(), static_cast<uint32_t>(vram.size()), nullptr);
t.Equals(gs.ReadVram(0x3Fu, 0u, 1u, 0u, 0u), 0u,
"reserved PSM reads should use the null handler");
gs.WriteVram(0x3Fu, 0u, 1u, 0u, 0u, 0x0005180Bu);
t.Equals(static_cast<uint32_t>(vram[0]), 0xA5u,
"reserved PSM writes should leave VRAM unchanged");
});
tc.Run("PSMT4 address mapping matches GS manual layout", [](TestCase &t)
{
constexpr uint32_t kBaseBlock = 0u;
@@ -2488,6 +2674,161 @@ void register_ps2_gs_tests()
"T8 CSM1 CLUT sampling should read CT32-uploaded palette entries through GS swizzled addressing");
});
tc.Run("GS T8 CSM1 applies CSA and masks CSA bit 4 for CT32 CLUTs", [](TestCase &t)
{
std::vector<uint8_t> vram(PS2_GS_VRAM_SIZE, 0u);
GS gs;
gs.init(vram.data(), static_cast<uint32_t>(vram.size()), nullptr);
constexpr uint32_t kTexTbp = 64u;
constexpr uint32_t kClutCbp = 128u;
constexpr uint64_t kFrameReg =
(0ull << 0) |
(1ull << 16) |
(static_cast<uint64_t>(GS_PSM_CT32) << 24);
constexpr uint64_t kZbuf = (1ull << 32);
constexpr uint64_t kTex0 =
(static_cast<uint64_t>(kTexTbp) << 0) |
(1ull << 14) |
(static_cast<uint64_t>(GS_PSM_T8) << 20) |
(0ull << 26) |
(0ull << 30) |
(1ull << 34) |
(1ull << 35) |
(static_cast<uint64_t>(kClutCbp) << 37) |
(static_cast<uint64_t>(GS_PSM_CT32) << 51) |
(17ull << 56);
constexpr uint64_t kPrim =
static_cast<uint64_t>(GS_PRIM_SPRITE) |
(1ull << 4) |
(1ull << 8);
constexpr uint32_t kExpectedColor = 0xFF204080u;
constexpr uint32_t kWrongNoCsaColor = 0xFF00FF00u;
constexpr uint32_t kWrongBit4Color = 0xFFFF0000u;
const uint32_t texOff = GSPSMT8::addrPSMT8(kTexTbp, 1u, 0u, 0u);
vram[texOff] = 0u;
// CSA=17 is CSA=1 for a CT32 CLUT. Logical entry 16 is at
// physical CSM1 entry 8 after address bits 3 and 4 are swapped.
gs.WriteVram(GS_PSM_CT32, kClutCbp, 1u, 0u, 0u, kWrongNoCsaColor);
gs.WriteVram(GS_PSM_CT32, kClutCbp, 1u, 8u, 0u, kExpectedColor);
gs.WriteVram(GS_PSM_CT32, kClutCbp, 1u, 8u, 16u, kWrongBit4Color);
gs.writeRegister(GS_REG_FRAME_1, kFrameReg);
gs.writeRegister(GS_REG_ZBUF_1, kZbuf);
gs.writeRegister(GS_REG_SCISSOR_1, 0ull);
gs.writeRegister(GS_REG_XYOFFSET_1, 0ull);
gs.writeRegister(GS_REG_TEST_1, 0x30000ull);
gs.writeRegister(GS_REG_ALPHA_1, 0ull);
gs.writeRegister(GS_REG_TEX0_1, kTex0);
gs.writeRegister(GS_REG_PRIM, kPrim);
gs.writeRegister(GS_REG_RGBAQ, 0x80808080ull);
gs.writeRegister(GS_REG_UV, 0ull);
gs.writeRegister(GS_REG_XYZ2, 0ull);
gs.writeRegister(GS_REG_UV, 0ull);
gs.writeRegister(GS_REG_XYZ2, 0ull);
uint32_t pixel = 0u;
std::memcpy(&pixel, vram.data(), sizeof(pixel));
t.Equals(pixel, kExpectedColor,
"T8 CSM1 should offset by CSA while CT32 ignores the fifth CSA bit");
});
tc.Run("GS T4 CSM1 preserves CSA bit 4 for CT16 CLUTs", [](TestCase &t)
{
std::vector<uint8_t> vram(PS2_GS_VRAM_SIZE, 0u);
GS gs;
gs.init(vram.data(), static_cast<uint32_t>(vram.size()), nullptr);
constexpr uint32_t kTexTbp = 64u;
constexpr uint32_t kClutCbp = 128u;
constexpr uint64_t kFrameReg =
(0ull << 0) |
(1ull << 16) |
(static_cast<uint64_t>(GS_PSM_CT32) << 24);
constexpr uint64_t kZbuf = (1ull << 32);
constexpr uint64_t kTex0 =
(static_cast<uint64_t>(kTexTbp) << 0) |
(1ull << 14) |
(static_cast<uint64_t>(GS_PSM_T4) << 20) |
(0ull << 26) |
(0ull << 30) |
(1ull << 34) |
(1ull << 35) |
(static_cast<uint64_t>(kClutCbp) << 37) |
(static_cast<uint64_t>(GS_PSM_CT16) << 51) |
(16ull << 56);
constexpr uint64_t kTexa = (0x80ull << 32);
constexpr uint64_t kPrim =
static_cast<uint64_t>(GS_PRIM_SPRITE) |
(1ull << 4) |
(1ull << 8);
constexpr uint16_t kExpectedRed = 0x801Fu;
constexpr uint16_t kWrongGreen = 0x83E0u;
constexpr uint32_t kExpectedColor = 0x800000F8u;
writePSMT4Texel(vram, kTexTbp, 1u, 0u, 0u, 1u);
// CSA=16 selects the upper half of a CT16 CLUT. CSM1 swaps bits
// 3 and 4 but must preserve address bit 8.
gs.WriteVram(GS_PSM_CT16, kClutCbp, 1u, 1u, 0u, kWrongGreen);
gs.WriteVram(GS_PSM_CT16, kClutCbp, 1u, 1u, 16u, kExpectedRed);
gs.writeRegister(GS_REG_FRAME_1, kFrameReg);
gs.writeRegister(GS_REG_ZBUF_1, kZbuf);
gs.writeRegister(GS_REG_SCISSOR_1, 0ull);
gs.writeRegister(GS_REG_XYOFFSET_1, 0ull);
gs.writeRegister(GS_REG_TEST_1, 0x30000ull);
gs.writeRegister(GS_REG_ALPHA_1, 0ull);
gs.writeRegister(GS_REG_TEX0_1, kTex0);
gs.writeRegister(GS_REG_TEXA, kTexa);
gs.writeRegister(GS_REG_PRIM, kPrim);
gs.writeRegister(GS_REG_RGBAQ, 0x80808080ull);
gs.writeRegister(GS_REG_UV, 0ull);
gs.writeRegister(GS_REG_XYZ2, 0ull);
gs.writeRegister(GS_REG_UV, 0ull);
gs.writeRegister(GS_REG_XYZ2, 0ull);
uint32_t pixel = 0u;
std::memcpy(&pixel, vram.data(), sizeof(pixel));
t.Equals(pixel, kExpectedColor,
"CT16 CSM1 should retain CSA[4] instead of aliasing the upper palette onto the lower one");
});
tc.Run("GS TEX0 dimensions saturate at 1024 pixels", [](TestCase &t)
{
std::vector<uint8_t> vram(PS2_GS_VRAM_SIZE, 0u);
GS gs;
gs.init(vram.data(), static_cast<uint32_t>(vram.size()), nullptr);
GSRasterizer rasterizer;
constexpr uint32_t kTexTbp = 64u;
constexpr uint64_t kTex0 =
(static_cast<uint64_t>(kTexTbp) << 0) |
(16ull << 14) |
(static_cast<uint64_t>(GS_PSM_CT32) << 20) |
(15ull << 26) |
(15ull << 30) |
(1ull << 34) |
(1ull << 35);
constexpr uint64_t kPrim =
static_cast<uint64_t>(GS_PRIM_TRIANGLE) |
(1ull << 4);
constexpr uint32_t kExpectedColor = 0xFF3366CCu;
constexpr uint32_t kUnsaturatedColor = 0xFF00FF00u;
gs.WriteVram(GS_PSM_CT32, kTexTbp, 16u, 1u, 0u, kExpectedColor);
gs.WriteVram(GS_PSM_CT32, kTexTbp, 16u, 32u, 0u, kUnsaturatedColor);
gs.writeRegister(GS_REG_TEX0_1, kTex0);
gs.writeRegister(GS_REG_PRIM, kPrim);
const uint32_t sampled =
rasterizer.sampleTexture(&gs, 1.0f / 1024.0f, 0.0f, 1.0f, 0u, 0u);
t.Equals(sampled, kExpectedColor,
"TW/TH values above 10 should address a 1024-pixel texture instead of growing beyond GS limits");
});
tc.Run("GS TEX2 updates CLUT state independently from TEX0", [](TestCase &t)
{
std::vector<uint8_t> vram(PS2_GS_VRAM_SIZE, 0u);
@@ -2975,97 +3316,312 @@ void register_ps2_gs_tests()
"linear filtering should preserve the shared opaque alpha from the CLUT entries");
});
tc.Run("GS alpha test AFAIL framebuffer-only still writes the pixel", [](TestCase &t)
tc.Run("GS CLAMP modes transform texture coordinates before sampling", [](TestCase &t)
{
std::vector<uint8_t> vram(PS2_GS_VRAM_SIZE, 0u);
GS gs;
gs.init(vram.data(), static_cast<uint32_t>(vram.size()), nullptr);
auto renderConstantUv = [](uint64_t clampReg,
uint16_t fixedU,
uint16_t fixedV) -> uint32_t
{
std::vector<uint8_t> vram(PS2_GS_VRAM_SIZE, 0u);
GS gs;
gs.init(vram.data(), static_cast<uint32_t>(vram.size()), nullptr);
constexpr uint64_t kFrame =
(0ull << 0) |
(1ull << 16) |
(static_cast<uint64_t>(GS_PSM_CT32) << 24);
constexpr uint64_t kZbuf = (1ull << 32);
constexpr uint64_t kScissor =
(0ull << 0) |
(0ull << 16) |
(0ull << 32) |
(0ull << 48);
constexpr uint64_t kTest =
1ull | // ATE
(5ull << 1) | // ATST = GEQUAL
(0x80ull << 4) | // AREF
(1ull << 12) | // AFAIL = FB_ONLY
(1ull << 17); // ZTST = ALWAYS
constexpr uint64_t kPrim =
static_cast<uint64_t>(GS_PRIM_POINT);
constexpr uint64_t kRgbaq =
(0x12ull << 0) |
(0x34ull << 8) |
(0x56ull << 16) |
(0x00ull << 24) |
(0x3F800000ull << 32); // q = 1.0f
constexpr uint32_t kTexTbp = 64u;
constexpr uint32_t kTexel0 = 0x800000FFu;
constexpr uint32_t kTexel1 = 0x8000FF00u;
constexpr uint32_t kTexel2 = 0x80FF0000u;
constexpr uint32_t kTexel3 = 0x80FFFFFFu;
constexpr uint32_t kTexelV3 = 0x80FFFF00u;
constexpr uint64_t kFrame =
(1ull << 16) |
(static_cast<uint64_t>(GS_PSM_CT32) << 24);
constexpr uint64_t kZbuf = (1ull << 32);
constexpr uint64_t kTex0 =
(static_cast<uint64_t>(kTexTbp) << 0) |
(1ull << 14) |
(static_cast<uint64_t>(GS_PSM_CT32) << 20) |
(2ull << 26) |
(2ull << 30) |
(1ull << 34) |
(1ull << 35);
constexpr uint64_t kPrim =
static_cast<uint64_t>(GS_PRIM_TRIANGLE) |
(1ull << 4) |
(1ull << 8);
constexpr uint64_t kRgbaq = 0x3F80000080808080ull;
gs.writeRegister(GS_REG_FRAME_1, kFrame);
gs.writeRegister(GS_REG_ZBUF_1, kZbuf);
gs.writeRegister(GS_REG_SCISSOR_1, kScissor);
gs.writeRegister(GS_REG_TEST_1, kTest);
gs.writeRegister(GS_REG_PRIM, kPrim);
gs.writeRegister(GS_REG_RGBAQ, kRgbaq);
gs.writeRegister(GS_REG_XYZ2, 0ull);
writeReferencePSMCT32Pixel(vram, kTexTbp, 1u, 0u, 0u, kTexel0);
writeReferencePSMCT32Pixel(vram, kTexTbp, 1u, 1u, 0u, kTexel1);
writeReferencePSMCT32Pixel(vram, kTexTbp, 1u, 2u, 0u, kTexel2);
writeReferencePSMCT32Pixel(vram, kTexTbp, 1u, 3u, 0u, kTexel3);
writeReferencePSMCT32Pixel(vram, kTexTbp, 1u, 0u, 3u, kTexelV3);
uint32_t pixel = 0u;
std::memcpy(&pixel, vram.data(), sizeof(pixel));
t.Equals(pixel, 0x00563412u,
"AFAIL=FB_ONLY should still update the framebuffer when the alpha test fails");
gs.writeRegister(GS_REG_FRAME_1, kFrame);
gs.writeRegister(GS_REG_ZBUF_1, kZbuf);
gs.writeRegister(GS_REG_SCISSOR_1, (3ull << 16) | (3ull << 48));
gs.writeRegister(GS_REG_XYOFFSET_1, 0ull);
gs.writeRegister(GS_REG_TEST_1, 0x30000ull);
gs.writeRegister(GS_REG_TEX0_1, kTex0);
gs.writeRegister(GS_REG_CLAMP_1, clampReg);
gs.writeRegister(GS_REG_PRIM, kPrim);
gs.writeRegister(GS_REG_RGBAQ, kRgbaq);
const uint64_t uv =
static_cast<uint64_t>(fixedU) |
(static_cast<uint64_t>(fixedV) << 16);
gs.writeRegister(GS_REG_UV, uv);
gs.writeRegister(GS_REG_XYZ2, 0ull);
gs.writeRegister(GS_REG_UV, uv);
gs.writeRegister(GS_REG_XYZ2, 32ull);
gs.writeRegister(GS_REG_UV, uv);
gs.writeRegister(GS_REG_XYZ2, (32ull << 16));
return readReferencePSMCT32Pixel(vram, 0u, 1u, 0u, 0u);
};
constexpr uint64_t kClamp = 1ull;
constexpr uint64_t kRegionClamp =
2ull |
(1ull << 4) |
(2ull << 14);
constexpr uint64_t kRegionRepeat =
3ull |
(1ull << 4) |
(2ull << 14);
t.Equals(renderConstantUv(0ull, 4u * 16u, 0u), 0x800000FFu,
"REPEAT should wrap texel 4 to texel 0 for a four-wide texture");
t.Equals(renderConstantUv(0ull, 0u, 4u * 16u), 0x800000FFu,
"REPEAT should wrap texel row 4 to row 0 for a four-high texture");
t.Equals(renderConstantUv(kClamp, 4u * 16u, 0u), 0x80FFFFFFu,
"CLAMP should hold texel 4 at the last texel");
t.Equals(renderConstantUv(kRegionClamp, 3u * 16u, 0u), 0x80FF0000u,
"REGION_CLAMP should hold texel 3 at MAXU=2");
t.Equals(renderConstantUv(kRegionRepeat, 4u * 16u, 0u), 0x80FF0000u,
"REGION_REPEAT should calculate (U & UMSK) | UFIX");
});
tc.Run("GS alpha test AFAIL RGB-only preserves destination alpha", [](TestCase &t)
tc.Run("GS STQ triangle interpolation divides homogeneous coordinates after DDA", [](TestCase &t)
{
std::vector<uint8_t> vram(PS2_GS_VRAM_SIZE, 0u);
GS gs;
gs.init(vram.data(), static_cast<uint32_t>(vram.size()), nullptr);
constexpr uint32_t kTexTbp = 64u;
constexpr uint64_t kFrame =
(0ull << 0) |
(1ull << 16) |
(static_cast<uint64_t>(GS_PSM_CT32) << 24);
constexpr uint64_t kZbuf = (1ull << 32);
constexpr uint64_t kScissor =
(0ull << 0) |
(0ull << 16) |
(0ull << 32) |
(0ull << 48);
constexpr uint64_t kTest =
1ull | // ATE
(5ull << 1) | // ATST = GEQUAL
(0x80ull << 4) | // AREF
(3ull << 12) | // AFAIL = RGB_ONLY
(1ull << 17); // ZTST = ALWAYS
constexpr uint64_t kTex0 =
(static_cast<uint64_t>(kTexTbp) << 0) |
(1ull << 14) |
(static_cast<uint64_t>(GS_PSM_CT32) << 20) |
(2ull << 26) |
(1ull << 34) |
(1ull << 35);
constexpr uint64_t kPrim =
static_cast<uint64_t>(GS_PRIM_POINT);
constexpr uint64_t kRgbaq =
(0x12ull << 0) |
(0x34ull << 8) |
(0x56ull << 16) |
(0x00ull << 24) |
(0x3F800000ull << 32); // q = 1.0f
constexpr uint32_t kExisting = 0xAB030201u;
static_cast<uint64_t>(GS_PRIM_TRIANGLE) |
(1ull << 4);
constexpr uint32_t kAffineTexel = 0x800000FFu;
constexpr uint32_t kHomogeneousTexel = 0x8000FF00u;
std::memcpy(vram.data(), &kExisting, sizeof(kExisting));
auto packFloat = [](float value) -> uint32_t
{
uint32_t bits = 0u;
std::memcpy(&bits, &value, sizeof(bits));
return bits;
};
auto packSt = [&](float s, float tVal) -> uint64_t
{
return static_cast<uint64_t>(packFloat(s)) |
(static_cast<uint64_t>(packFloat(tVal)) << 32);
};
auto packRgbaq = [&](float q) -> uint64_t
{
return 0x80808080ull |
(static_cast<uint64_t>(packFloat(q)) << 32);
};
writeReferencePSMCT32Pixel(vram, kTexTbp, 1u, 1u, 0u, kAffineTexel);
writeReferencePSMCT32Pixel(vram, kTexTbp, 1u, 2u, 0u, kHomogeneousTexel);
gs.writeRegister(GS_REG_FRAME_1, kFrame);
gs.writeRegister(GS_REG_ZBUF_1, kZbuf);
gs.writeRegister(GS_REG_SCISSOR_1, kScissor);
gs.writeRegister(GS_REG_TEST_1, kTest);
gs.writeRegister(GS_REG_SCISSOR_1, (4ull << 16) | (4ull << 48));
gs.writeRegister(GS_REG_XYOFFSET_1, 0ull);
gs.writeRegister(GS_REG_TEST_1, 0x30000ull);
gs.writeRegister(GS_REG_TEX0_1, kTex0);
gs.writeRegister(GS_REG_CLAMP_1, 1ull);
gs.writeRegister(GS_REG_PRIM, kPrim);
gs.writeRegister(GS_REG_RGBAQ, kRgbaq);
gs.writeRegister(GS_REG_XYZ2, 0ull);
uint32_t pixel = 0u;
std::memcpy(&pixel, vram.data(), sizeof(pixel));
t.Equals(pixel, 0xAB563412u,
"AFAIL=RGB_ONLY should update RGB while preserving destination alpha");
gs.writeRegister(GS_REG_ST, packSt(0.0f, 0.0f));
gs.writeRegister(GS_REG_RGBAQ, packRgbaq(1.0f));
gs.writeRegister(GS_REG_XYZ2, 0ull);
gs.writeRegister(GS_REG_ST, packSt(2.0f, 0.0f));
gs.writeRegister(GS_REG_RGBAQ, packRgbaq(2.0f));
gs.writeRegister(GS_REG_XYZ2, 64ull);
gs.writeRegister(GS_REG_ST, packSt(0.0f, 0.0f));
gs.writeRegister(GS_REG_RGBAQ, packRgbaq(1.0f));
gs.writeRegister(GS_REG_XYZ2, (64ull << 16));
const uint32_t pixel =
readReferencePSMCT32Pixel(vram, 0u, 1u, 1u, 1u);
t.Equals(pixel, kHomogeneousTexel,
"the DDA should interpolate S=0.75 and Q=1.375, selecting texel 2 after S/Q");
});
tc.Run("GS alpha-test AFAIL independently masks framebuffer and depth", [](TestCase &t)
{
constexpr uint32_t kInitialFramebuffer = 0xAB030201u;
constexpr uint32_t kInitialDepth = 0x11111111u;
constexpr uint64_t kTestBase =
1ull | // ATE
(5ull << 1) | // ATST = GEQUAL
(0x80ull << 4) | // AREF
(1ull << 16) | // ZTE
(1ull << 17); // ZTST = ALWAYS
const GsPixelTestResult keep =
drawGsPixelForTests(GS_PSM_CT32, kTestBase | (0ull << 12), false,
kInitialFramebuffer, kInitialDepth, 0x00u);
t.Equals(keep.framebuffer, kInitialFramebuffer,
"AFAIL=KEEP should preserve the framebuffer");
t.Equals(keep.depth, kInitialDepth,
"AFAIL=KEEP should preserve depth");
const GsPixelTestResult framebufferOnly =
drawGsPixelForTests(GS_PSM_CT32, kTestBase | (1ull << 12), false,
kInitialFramebuffer, kInitialDepth, 0x00u);
t.Equals(framebufferOnly.framebuffer, 0x00563412u,
"AFAIL=FB_ONLY should update RGBA");
t.Equals(framebufferOnly.depth, kInitialDepth,
"AFAIL=FB_ONLY should preserve depth");
const GsPixelTestResult depthOnly =
drawGsPixelForTests(GS_PSM_CT32, kTestBase | (2ull << 12), false,
kInitialFramebuffer, kInitialDepth, 0x00u);
t.Equals(depthOnly.framebuffer, kInitialFramebuffer,
"AFAIL=ZB_ONLY should preserve the framebuffer");
t.Equals(depthOnly.depth, 0x22222222u,
"AFAIL=ZB_ONLY should update depth");
const GsPixelTestResult rgbOnly =
drawGsPixelForTests(GS_PSM_CT32, kTestBase | (3ull << 12), false,
kInitialFramebuffer, kInitialDepth, 0x00u);
t.Equals(rgbOnly.framebuffer, 0xAB563412u,
"AFAIL=RGB_ONLY should preserve destination alpha on CT32");
t.Equals(rgbOnly.depth, kInitialDepth,
"AFAIL=RGB_ONLY should preserve depth");
});
tc.Run("GS RGB_ONLY falls back to FB_ONLY outside CT32", [](TestCase &t)
{
constexpr uint32_t kInitialDepth = 0x11111111u;
constexpr uint64_t kTest =
1ull |
(5ull << 1) |
(0x80ull << 4) |
(3ull << 12) |
(1ull << 16) |
(1ull << 17);
const GsPixelTestResult ct24 =
drawGsPixelForTests(GS_PSM_CT24, kTest, false,
0x00030201u, kInitialDepth, 0x00u);
t.Equals(ct24.framebuffer, 0x00563412u,
"RGB_ONLY should write the full CT24 framebuffer pixel");
t.Equals(ct24.depth, kInitialDepth,
"RGB_ONLY-as-FB_ONLY should preserve CT24 depth");
const GsPixelTestResult ct16 =
drawGsPixelForTests(GS_PSM_CT16, kTest, false,
0x8001u, kInitialDepth, 0x00u);
t.Equals(ct16.framebuffer, 0x28C2u,
"RGB_ONLY should write RGB and alpha for CT16");
t.Equals(ct16.depth, kInitialDepth,
"RGB_ONLY-as-FB_ONLY should preserve CT16 depth");
});
tc.Run("GS ZMSK suppresses depth without suppressing framebuffer writes", [](TestCase &t)
{
constexpr uint64_t kTest =
1ull |
(5ull << 1) |
(0x80ull << 4) |
(1ull << 16) |
(1ull << 17);
const GsPixelTestResult result =
drawGsPixelForTests(GS_PSM_CT32, kTest, true,
0xAB030201u, 0x11111111u, 0x80u);
t.Equals(result.framebuffer, 0x80563412u,
"a passing alpha test should write the framebuffer");
t.Equals(result.depth, 0x11111111u,
"ZMSK should preserve depth");
});
tc.Run("GS DATE and DATM inspect the framebuffer-format alpha bit", [](TestCase &t)
{
constexpr uint32_t kInitialDepth = 0x11111111u;
constexpr uint64_t kTestBase =
(1ull << 14) | // DATE
(1ull << 16) | // ZTE
(1ull << 17); // ZTST = ALWAYS
const GsPixelTestResult ct32ZeroPass =
drawGsPixelForTests(GS_PSM_CT32, kTestBase, false,
0x00030201u, kInitialDepth, 0x80u);
t.Equals(ct32ZeroPass.framebuffer, 0x80563412u,
"DATM=0 should accept a clear CT32 alpha bit");
t.Equals(ct32ZeroPass.depth, 0x22222222u,
"a passing CT32 DATE should allow depth");
const GsPixelTestResult ct32OneFail =
drawGsPixelForTests(GS_PSM_CT32, kTestBase, false,
0x80030201u, kInitialDepth, 0x80u);
t.Equals(ct32OneFail.framebuffer, 0x80030201u,
"DATM=0 should reject a set CT32 alpha bit");
t.Equals(ct32OneFail.depth, kInitialDepth,
"a failing CT32 DATE should reject depth");
const GsPixelTestResult ct32OnePass =
drawGsPixelForTests(GS_PSM_CT32, kTestBase | (1ull << 15), false,
0x80030201u, kInitialDepth, 0x80u);
t.Equals(ct32OnePass.framebuffer, 0x80563412u,
"DATM=1 should accept a set CT32 alpha bit");
const GsPixelTestResult ct16ZeroPass =
drawGsPixelForTests(GS_PSM_CT16, kTestBase, false,
0x0001u, kInitialDepth, 0x80u);
t.Equals(ct16ZeroPass.framebuffer, 0xA8C2u,
"DATM=0 should accept a clear CT16 alpha bit");
const GsPixelTestResult ct16OneFail =
drawGsPixelForTests(GS_PSM_CT16, kTestBase, false,
0x8001u, kInitialDepth, 0x80u);
t.Equals(ct16OneFail.framebuffer, 0x8001u,
"DATM=0 should reject a set CT16 alpha bit");
t.Equals(ct16OneFail.depth, kInitialDepth,
"a failing CT16 DATE should reject depth");
const GsPixelTestResult ct16OnePass =
drawGsPixelForTests(GS_PSM_CT16, kTestBase | (1ull << 15), false,
0x8001u, kInitialDepth, 0x80u);
t.Equals(ct16OnePass.framebuffer, 0xA8C2u,
"DATM=1 should accept a set CT16 alpha bit");
const GsPixelTestResult ct24DatmZero =
drawGsPixelForTests(GS_PSM_CT24, kTestBase, false,
0x00030201u, kInitialDepth, 0x80u);
const GsPixelTestResult ct24DatmOne =
drawGsPixelForTests(GS_PSM_CT24, kTestBase | (1ull << 15), false,
0x00030201u, kInitialDepth, 0x80u);
t.Equals(ct24DatmZero.framebuffer, 0x00563412u,
"CT24 DATE should pass for DATM=0");
t.Equals(ct24DatmOne.framebuffer, 0x00563412u,
"CT24 DATE should pass for DATM=1");
t.Equals(ct24DatmOne.depth, 0x22222222u,
"CT24 DATE should not block depth");
});
tc.Run("GS triangle fan subpixel quad fills rows without interior holes", [](TestCase &t)
+63
View File
@@ -440,6 +440,69 @@ void register_ps2_iop_tests()
"reset should restore per-instance service state");
});
tc.Run("LotR sound update completes queued PlayStream slots", [](TestCase &t)
{
FakeIopHost host;
ps2x::iop::IopSubsystem subsystem(host);
std::string error;
t.IsTrue(subsystem.configure({"SLUS_205.78", 0u, 0u}, &error),
"LotR profile should configure");
constexpr uint32_t kSendAddress = 0x0800u;
constexpr uint32_t kReceiveAddress = 0x1000u;
constexpr uint16_t kStreamSlot = 7u;
const std::array<uint16_t, 10> playStreamPacket = {
1u, // command count
1u, // PlayStream
7u, // argument count
0u,
static_cast<uint16_t>(kStreamSlot << 8u),
0u,
0u,
0u,
0u,
0u,
};
t.IsTrue(host.writeGuest(kSendAddress,
playStreamPacket.data(),
sizeof(playStreamPacket)),
"PlayStream command packet should fit in guest memory");
ps2x::iop::RpcRequest request{};
request.sid = 0x00012345u;
request.send = {kSendAddress, sizeof(playStreamPacket)};
request.receive = {kReceiveAddress, 0x100u};
t.IsTrue(subsystem.handleRpc(request).handled,
"LotR sound service should handle PlayStream");
t.Equals(host.readWord(kReceiveAddress), 1u,
"PlayStream response should expose one active record");
const uint32_t packedStream = host.readWord(kReceiveAddress + 4u);
t.Equals((packedStream >> 4u) & 0x3Fu,
static_cast<uint32_t>(kStreamSlot),
"active record should identify the queued EE stream slot");
t.Equals(host.readWord(kReceiveAddress + 0x24u), 1u,
"response counter should follow the active record");
const std::array<uint16_t, 5> statusPacket = {
1u, // command count
9u, // GetStatus
2u, // argument count
kStreamSlot,
0u,
};
t.IsTrue(host.writeGuest(kSendAddress, statusPacket.data(), sizeof(statusPacket)),
"GetStatus command packet should fit in guest memory");
request.send.size = sizeof(statusPacket);
t.IsTrue(subsystem.handleRpc(request).handled,
"LotR sound service should handle the following status update");
t.Equals(host.readWord(kReceiveAddress), 0u,
"the update after PlayStream should report no active records");
t.Equals(host.readWord(kReceiveAddress + 4u), 2u,
"empty response counter should return to the base offset");
});
tc.Run("TSNDDRV uses profile checksum bindings without writing invalid ports", [](TestCase &t)
{
FakeIopHost host(0x02000000u);
+188
View File
@@ -155,6 +155,102 @@ static bool writeMinimalMipsElfWithJalFallbackTarget(const std::filesystem::path
return writer.save(elfPath.string());
}
static bool writeMinimalMipsElfWithInitializer(const std::filesystem::path &elfPath,
const std::string &functionName,
uint32_t initializerTarget)
{
ELFIO::elfio writer;
writer.create(ELFIO::ELFCLASS32, ELFIO::ELFDATA2LSB);
writer.set_os_abi(ELFIO::ELFOSABI_NONE);
writer.set_type(ELFIO::ET_EXEC);
writer.set_machine(ELFIO::EM_MIPS);
writer.set_entry(0x00100000u);
ELFIO::section *text = writer.sections.add(".text");
text->set_type(ELFIO::SHT_PROGBITS);
text->set_flags(ELFIO::SHF_ALLOC | ELFIO::SHF_EXECINSTR);
text->set_addr_align(4);
text->set_address(0x00100000u);
const std::array<uint32_t, 2> textWords = {
0x03E00008u, // jr $ra
0x00000000u, // nop
};
text->set_data(reinterpret_cast<const char *>(textWords.data()),
static_cast<ELFIO::Elf_Word>(textWords.size() * sizeof(uint32_t)));
ELFIO::section *ctors = writer.sections.add(".ctors");
ctors->set_type(ELFIO::SHT_PROGBITS);
ctors->set_flags(ELFIO::SHF_ALLOC | ELFIO::SHF_WRITE);
ctors->set_addr_align(4);
ctors->set_address(0x00200000u);
ctors->set_data(reinterpret_cast<const char *>(&initializerTarget),
static_cast<ELFIO::Elf_Word>(sizeof(initializerTarget)));
ELFIO::section *strtab = writer.sections.add(".strtab");
strtab->set_type(ELFIO::SHT_STRTAB);
strtab->set_addr_align(1);
ELFIO::section *symtab = writer.sections.add(".symtab");
symtab->set_type(ELFIO::SHT_SYMTAB);
symtab->set_info(1);
symtab->set_link(strtab->get_index());
symtab->set_addr_align(4);
symtab->set_entry_size(writer.get_default_entry_size(ELFIO::SHT_SYMTAB));
ELFIO::symbol_section_accessor symbols(writer, symtab);
ELFIO::string_section_accessor strings(strtab);
symbols.add_symbol(strings, "", 0, 0,
ELFIO::STB_LOCAL, ELFIO::STT_NOTYPE, 0, ELFIO::SHN_UNDEF);
symbols.add_symbol(strings, functionName.c_str(), text->get_address(), text->get_size(),
ELFIO::STB_GLOBAL, ELFIO::STT_FUNC, 0, text->get_index());
ELFIO::segment *textSegment = writer.segments.add();
textSegment->set_type(ELFIO::PT_LOAD);
textSegment->set_flags(ELFIO::PF_R | ELFIO::PF_X);
textSegment->set_align(0x1000);
textSegment->add_section_index(text->get_index(), text->get_addr_align());
ELFIO::segment *dataSegment = writer.segments.add();
dataSegment->set_type(ELFIO::PT_LOAD);
dataSegment->set_flags(ELFIO::PF_R | ELFIO::PF_W);
dataSegment->set_align(0x1000);
dataSegment->add_section_index(ctors->get_index(), ctors->get_addr_align());
return writer.save(elfPath.string());
}
static bool writeRecompilerTestConfig(const std::filesystem::path &configPath,
const std::filesystem::path &elfPath,
const std::filesystem::path &outputPath,
const std::vector<std::string> &skip,
const std::vector<std::string> &stubs = {})
{
std::ofstream config(configPath);
if (!config)
return false;
config << "[general]\n";
config << "input = \"" << elfPath.generic_string() << "\"\n";
config << "output = \"" << outputPath.generic_string() << "\"\n";
config << "skip = [";
for (size_t i = 0; i < skip.size(); ++i)
{
if (i != 0u)
config << ", ";
config << '"' << skip[i] << '"';
}
config << "]\n";
config << "stubs = [";
for (size_t i = 0; i < stubs.size(); ++i)
{
if (i != 0u)
config << ", ";
config << '"' << stubs[i] << '"';
}
config << "]\n";
return static_cast<bool>(config);
}
void register_ps2_recompiler_tests()
{
MiniTest::Case("PS2Recompiler", [](TestCase &tc)
@@ -890,6 +986,98 @@ void register_ps2_recompiler_tests()
"__sbprintf should be left for recompilation");
});
tc.Run("initializer skips fall back to guest recompilation", [](TestCase &t) {
const std::string uniqueSuffix =
std::to_string(std::chrono::steady_clock::now().time_since_epoch().count());
const std::filesystem::path tempRoot =
std::filesystem::temp_directory_path() / ("ps2recomp-initializer-" + uniqueSuffix);
const std::filesystem::path elfPath = tempRoot / "initializer.elf";
const std::filesystem::path configPath = tempRoot / "initializer.toml";
const std::filesystem::path outputPath = tempRoot / "output";
std::filesystem::create_directories(tempRoot);
const bool elfWritten =
writeMinimalMipsElfWithInitializer(elfPath, "__sinit_test.cpp", 0x00100000u);
const bool configWritten =
writeRecompilerTestConfig(configPath, elfPath, outputPath, {"__sinit_test.cpp"});
t.IsTrue(elfWritten && configWritten,
"initializer regression inputs should be generated");
if (elfWritten && configWritten)
{
PS2Recompiler recompiler(configPath.string());
t.IsTrue(recompiler.initialize(),
"initializer regression config should initialize");
t.IsTrue(recompiler.recompile(),
"a decodable skipped initializer should use guest fallback");
const RecompilerReporter::Counters &counters = recompiler.reportCounters();
t.Equals(counters.correctnessCriticalGuestFallbacks, static_cast<size_t>(1u),
"the ignored initializer skip should be reported");
t.Equals(counters.correctnessCriticalFailures, static_cast<size_t>(0u),
"guest fallback should avoid a correctness-critical failure");
t.Equals(counters.functionsSkipped, static_cast<size_t>(0u),
"the initializer should not remain skipped");
t.Equals(counters.functionsRecompiled, static_cast<size_t>(1u),
"the original initializer body should be recompiled");
}
std::error_code removeError;
std::filesystem::remove_all(tempRoot, removeError);
});
tc.Run("missing constructor-table targets fail recompilation", [](TestCase &t) {
const std::string uniqueSuffix =
std::to_string(std::chrono::steady_clock::now().time_since_epoch().count());
const std::filesystem::path tempRoot =
std::filesystem::temp_directory_path() / ("ps2recomp-missing-initializer-" + uniqueSuffix);
const std::filesystem::path elfPath = tempRoot / "initializer.elf";
const std::filesystem::path configPath = tempRoot / "initializer.toml";
const std::filesystem::path outputPath = tempRoot / "output";
std::filesystem::create_directories(tempRoot);
const bool elfWritten =
writeMinimalMipsElfWithInitializer(elfPath, "ordinary_entry", 0x00100040u);
const bool configWritten =
writeRecompilerTestConfig(configPath, elfPath, outputPath, {});
t.IsTrue(elfWritten && configWritten,
"missing-initializer regression inputs should be generated");
if (elfWritten && configWritten)
{
{
PS2Recompiler recompiler(configPath.string());
t.IsTrue(recompiler.initialize(),
"missing-initializer regression config should initialize");
t.IsFalse(recompiler.recompile(),
"an unresolved .ctors target should be correctness-fatal");
t.Equals(recompiler.reportCounters().correctnessCriticalFailures,
static_cast<size_t>(1u),
"the unresolved constructor target should appear in the report");
}
const bool overrideWritten =
writeRecompilerTestConfig(
configPath, elfPath, outputPath, {},
{"memclr@0x00100040"});
t.IsTrue(overrideWritten,
"manual initializer override config should be generated");
if (overrideWritten)
{
PS2Recompiler overridden(configPath.string());
t.IsTrue(overridden.initialize(),
"manual initializer override should initialize");
t.IsTrue(overridden.recompile(),
"a resolved address-bound handler should satisfy the constructor target");
t.Equals(overridden.reportCounters().functionsStubbed,
static_cast<size_t>(1u),
"the resolved manual initializer should be emitted as a stub binding");
}
}
std::error_code removeError;
std::filesystem::remove_all(tempRoot, removeError);
});
tc.Run("respect max length for .cpp filenames", [](TestCase& t) {
t.IsTrue(PS2Recompiler::ClampFilenameLength("ReallyLongFunctionNameReallyLongFunctionNameReallyLongFunctionName_0x12345678",".cpp",50).length() <= 50,"Function name must be max 50 characters");
+148 -2
View File
@@ -82,12 +82,38 @@ namespace
0x28u;
}
uint32_t makeVuIaddiu(uint8_t it, uint8_t is, int16_t immediate)
{
return (0x08u << 25) |
(static_cast<uint32_t>(it & 0xFu) << 16) |
(static_cast<uint32_t>(is & 0xFu) << 11) |
(static_cast<uint32_t>(immediate) & 0x7FFu);
}
uint32_t makeVuLowerSpecial(uint8_t specialOp, uint8_t is,
uint8_t it = 0u, uint8_t dest = 0u)
{
return (0x40u << 25) |
(static_cast<uint32_t>(dest & 0xFu) << 21) |
(static_cast<uint32_t>(it & 0x1Fu) << 16) |
(static_cast<uint32_t>(is & 0x1Fu) << 11) |
(static_cast<uint32_t>(specialOp & 0x7Cu) << 4) |
static_cast<uint32_t>(specialOp & 0x3u) |
0x3Cu;
}
void writeVuInstructionPair(uint8_t *code, uint32_t pc, uint32_t lower, uint32_t upper)
{
std::memcpy(code + pc, &lower, sizeof(lower));
std::memcpy(code + pc + sizeof(lower), &upper, sizeof(upper));
}
uint64_t packVuInstructionPair(uint32_t lower, uint32_t upper)
{
return static_cast<uint64_t>(lower) |
(static_cast<uint64_t>(upper) << 32);
}
bool hasSignedRdWrite(const std::string &generated, uint8_t rd)
{
if (rd == 0u)
@@ -182,7 +208,7 @@ namespace
}
bool shouldPreempt = false;
for (int attempt = 0; attempt < 256 &&
for (int attempt = 0; attempt < 2048 &&
!shouldPreempt;
++attempt)
{
@@ -571,6 +597,30 @@ void register_ps2_runtime_expansion_tests()
t.IsFalse(innerPending, "inner scope must stay untouched");
});
tc.Run("guest preemption policy amortizes uncontended back-edge checks", [](TestCase &t)
{
PS2Runtime runtime;
uint32_t firstPreemptionCall = 0u;
// Use a fresh host thread so this assertion starts with a fresh
// thread-local back-edge counter.
std::thread worker([&]()
{
for (uint32_t call = 1u; call <= 32768u; ++call)
{
if (runtime.shouldPreemptGuestExecution())
{
firstPreemptionCall = call;
break;
}
}
});
worker.join();
t.Equals(firstPreemptionCall, 16384u,
"uncontended parser loops should amortize dispatcher handoffs across many back edges");
});
tc.Run("guest preemption policy requests a dispatcher handoff when another guest thread contends", [](TestCase &t)
{
PS2Runtime runtime;
@@ -1693,7 +1743,7 @@ void register_ps2_runtime_expansion_tests()
writeVuInstructionPair(code, 8u, 0u, makeVuAdd(0xFu, 2u, 1u, 1u));
writeVuInstructionPair(code, 16u, makeVuSq(0xFu, 2u, 0u, 1), kVuEndNop);
R5900Context ctx;
R5900Context ctx{};
runtime.executeVU0Microprogram(runtime.memory().getRDRAM(), &ctx, 0u);
float output[4]{};
@@ -1709,6 +1759,102 @@ void register_ps2_runtime_expansion_tests()
t.Equals(static_cast<uint32_t>(ctx.vi[0]), 0u, "VU0 VI0 should remain zero");
});
tc.Run("VU0 microprogram preserves the architectural RNG state", [](TestCase &t)
{
PS2Runtime runtime;
t.IsTrue(runtime.memory().initialize(), "PS2Memory initialize should succeed");
t.IsTrue(runtime.syncCoreSubsystems(), "runtime core subsystems should bind");
uint8_t *const code = runtime.memory().getVU0Code();
std::memset(code, 0, PS2_VU0_CODE_SIZE);
constexpr uint32_t kVuUpperNop = 0x000002FFu;
constexpr uint32_t kVuUpperEndNop = 0x400002FFu;
writeVuInstructionPair(
code, 0u,
makeVuLowerSpecial(0x40u, 0u, 1u, 0x8u),
kVuUpperEndNop); // RNEXT.x vf1
writeVuInstructionPair(code, 8u, 0u, kVuUpperNop);
constexpr uint32_t seed = 0x3FC00000u;
const uint32_t x = (seed >> 4) & 1u;
const uint32_t y = (seed >> 22) & 1u;
const uint32_t expected =
(((seed << 1) ^ x ^ y) & 0x007FFFFFu) | 0x3F800000u;
R5900Context ctx{};
ctx.vu0_r = _mm_castsi128_ps(
_mm_set1_epi32(static_cast<int32_t>(seed)));
runtime.executeVU0Microprogram(runtime.memory().getRDRAM(), &ctx, 0u);
alignas(16) uint32_t rWords[4]{};
_mm_storeu_si128(reinterpret_cast<__m128i *>(rWords),
_mm_castps_si128(ctx.vu0_r));
t.Equals(rWords[0], expected, "VU0 micro RNG should advance the imported R seed");
t.Equals(rWords[1], expected, "VU0 R should remain replicated for macro-mode access");
alignas(16) uint32_t vf1Words[4]{};
_mm_storeu_si128(reinterpret_cast<__m128i *>(vf1Words),
_mm_castps_si128(ctx.vu0_vf[1]));
t.Equals(vf1Words[0], expected, "RNEXT should expose the same R value through VF1.x");
});
tc.Run("VU0 direct MicroMem writes invalidate the fixed decode cache", [](TestCase &t)
{
PS2Runtime runtime;
t.IsTrue(runtime.memory().initialize(), "PS2Memory initialize should succeed");
t.IsTrue(runtime.syncCoreSubsystems(), "runtime core subsystems should bind");
constexpr uint32_t kVuUpperNop = 0x000002FFu;
constexpr uint32_t kVuUpperEndNop = 0x400002FFu;
runtime.memory().write64(
PS2_VU0_CODE_BASE,
packVuInstructionPair(makeVuIaddiu(1u, 0u, 1), kVuUpperEndNop));
runtime.memory().write64(
PS2_VU0_CODE_BASE + 8u,
packVuInstructionPair(0u, kVuUpperNop));
R5900Context first{};
runtime.executeVU0Microprogram(runtime.memory().getRDRAM(), &first, 0u);
t.Equals(static_cast<uint32_t>(first.vi[1]), 1u,
"first cached VU0 microprogram should execute");
runtime.memory().write64(
PS2_VU0_CODE_BASE,
packVuInstructionPair(makeVuIaddiu(1u, 0u, 2), kVuUpperEndNop));
R5900Context second{};
runtime.executeVU0Microprogram(runtime.memory().getRDRAM(), &second, 0u);
t.Equals(static_cast<uint32_t>(second.vi[1]), 2u,
"VU0 cache should rebuild after a direct MicroMem write");
});
tc.Run("VU0 FBRST TE gates a T-bit microprogram stop", [](TestCase &t)
{
PS2Runtime runtime;
t.IsTrue(runtime.memory().initialize(), "PS2Memory initialize should succeed");
t.IsTrue(runtime.syncCoreSubsystems(), "runtime core subsystems should bind");
uint8_t *const code = runtime.memory().getVU0Code();
std::memset(code, 0, PS2_VU0_CODE_SIZE);
constexpr uint32_t kVuUpperNop = 0x000002FFu;
writeVuInstructionPair(
code, 0u, makeVuIaddiu(1u, 0u, 7),
kVuUpperNop | 0x08000000u);
writeVuInstructionPair(
code, 8u, makeVuIaddiu(2u, 0u, 9),
kVuUpperNop);
R5900Context ctx{};
ctx.vu0_fbrst = 1u << 3; // TE0
runtime.executeVU0Microprogram(runtime.memory().getRDRAM(), &ctx, 0u);
t.Equals(static_cast<uint32_t>(ctx.vi[1]), 7u,
"the T-marked instruction should execute");
t.Equals(static_cast<uint32_t>(ctx.vi[2]), 0u,
"TE0 should stop VU0 before the following instruction");
t.IsTrue((ctx.vu0_vpu_stat & (1u << 2)) != 0u,
"VPU-STAT should report a VU0 T-bit stop");
t.Equals(ctx.vu0_tpc, 8u,
"TPC should point at the first instruction not executed");
});
tc.Run("GS sprite draw applies XYOFFSET and fully-outside scissor should not render", [](TestCase &t)
{
std::vector<uint8_t> vram(PS2_GS_VRAM_SIZE, 0u);
+5 -5
View File
@@ -162,7 +162,7 @@ namespace
constexpr uint32_t K_DTX_DISPATCH_RESULT_ADDR = 0x0002D800u;
constexpr uint32_t K_DTX_DISPATCH_RESULT_MARKER = 0xD15CA7C1u;
void lotrSoundEndCallbackShouldNotRun(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime)
void lotrSoundEndCallback(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime)
{
(void)rdram;
(void)runtime;
@@ -637,7 +637,7 @@ void register_ps2_sif_rpc_tests()
PS2Runtime::setIoPaths(oldPaths);
});
tc.Run("LotR sound RPC completes HLE callback without invoking guest loop", [](TestCase &t)
tc.Run("LotR sound RPC invokes guest callback to consume HLE response", [](TestCase &t)
{
TestEnv env;
configureProfile(env, "SLUS_205.78");
@@ -648,7 +648,7 @@ void register_ps2_sif_rpc_tests()
constexpr uint32_t kRecvAddr = 0x0003C000u;
constexpr uint32_t kEndFunc = 0x001FFD70u;
env.runtime.registerFunction(kEndFunc, lotrSoundEndCallbackShouldNotRun);
env.runtime.registerFunction(kEndFunc, lotrSoundEndCallback);
g_lotrSoundCallbackHits = 0u;
SifInitRpc(env.rdram.data(), &env.ctx, &env.runtime);
@@ -681,8 +681,8 @@ void register_ps2_sif_rpc_tests()
SifCallRpc(env.rdram.data(), &env.ctx, &env.runtime);
t.Equals(getRegS32(env.ctx, 2), KE_OK, "SifCallRpc should succeed for LotR sound RPC");
t.Equals(g_lotrSoundCallbackHits.load(), 0u,
"HLE-completed LotR sound callback should not invoke the guest callback");
t.Equals(g_lotrSoundCallbackHits.load(), 1u,
"LotR SOUND_JP callback should consume the HLE response");
t.Equals(readGuestStruct<uint32_t>(env.rdram.data(), kRecvAddr + 0u), 0u,
"LotR sound response should report no active stream records");
t.IsTrue(readGuestStruct<uint32_t>(env.rdram.data(), kRecvAddr + 4u) != 0u,
File diff suppressed because it is too large Load Diff