mirror of
https://github.com/ran-j/PS2Recomp.git
synced 2026-10-03 19:08:24 -04:00
refactor: refactor VU1 (#191)
* feat: implement fix and changes based on dark cloud report fix: fix GS AFAIL for RGB/alpha/Z, ZMSK fix: fix VU1 flags mask and pipeline fix: small VU1 cache fix feat: __ct__, __sinit_ are not sillent stubs anymore * feat: fix song JP pulling * feat: sound update for lotR * feat: prevent guest execution to be very slow * fix: small gs size bug * feat: refactor VU fix: fix cliping and other issues on gs fix: fix wrong vu0 register on recompiler * fix fix ACC scheduler stall feat: remove unused test fix: .fix overflow e underflow on FMAC * feat: small setting for windows test
This commit is contained in:
@@ -23,6 +23,21 @@ namespace
|
||||
static constexpr uint32_t kHostFrameWidth = 640u;
|
||||
static constexpr uint32_t kHostFrameHeight = 512u;
|
||||
|
||||
GSPrimReg decodePrimRegister(uint64_t value)
|
||||
{
|
||||
GSPrimReg prim{};
|
||||
prim.type = static_cast<GSPrimType>(value & 0x7u);
|
||||
prim.iip = ((value >> 3) & 1u) != 0u;
|
||||
prim.tme = ((value >> 4) & 1u) != 0u;
|
||||
prim.fge = ((value >> 5) & 1u) != 0u;
|
||||
prim.abe = ((value >> 6) & 1u) != 0u;
|
||||
prim.aa1 = ((value >> 7) & 1u) != 0u;
|
||||
prim.fst = ((value >> 8) & 1u) != 0u;
|
||||
prim.ctxt = ((value >> 9) & 1u) != 0u;
|
||||
prim.fix = ((value >> 10) & 1u) != 0u;
|
||||
return prim;
|
||||
}
|
||||
|
||||
uint16_t encodeFramePixelPSMCT16(uint8_t r, uint8_t g, uint8_t b, uint8_t a)
|
||||
{
|
||||
return static_cast<uint16_t>(((r >> 3) & 0x1Fu) |
|
||||
@@ -109,7 +124,8 @@ namespace
|
||||
|
||||
bool validatePackedGifPacket(const uint8_t *data, uint32_t sizeBytes)
|
||||
{
|
||||
return visitPackedGifPacket(data, sizeBytes, [](const PackedGifPacketTag &) { return true; });
|
||||
return visitPackedGifPacket(data, sizeBytes, [](const PackedGifPacketTag &)
|
||||
{ return true; });
|
||||
}
|
||||
|
||||
void decodeDisplaySize(uint64_t display64, uint32_t &outWidth, uint32_t &outHeight)
|
||||
@@ -265,7 +281,7 @@ namespace
|
||||
return count;
|
||||
}
|
||||
|
||||
bool clearFramebufferRect(GS* gs, const GSContext &ctx, uint32_t rgba)
|
||||
bool clearFramebufferRect(GS *gs, const GSContext &ctx, uint32_t rgba)
|
||||
{
|
||||
if (ctx.frame.fbw == 0u)
|
||||
{
|
||||
@@ -365,7 +381,7 @@ GS::GS()
|
||||
|
||||
InitLookupTables();
|
||||
|
||||
for (usz i = 0; i < 0x3F; ++i)
|
||||
for (usz i = 0; i < m_read_vram_funcs.size(); ++i)
|
||||
{
|
||||
switch (i)
|
||||
{
|
||||
@@ -444,6 +460,8 @@ void GS::reset()
|
||||
std::lock_guard<std::recursive_mutex> lock(m_stateMutex);
|
||||
std::memset(m_ctx, 0, sizeof(m_ctx));
|
||||
m_prim = {};
|
||||
m_primRegister = {};
|
||||
m_prmodeRegister = {};
|
||||
m_curR = 0x80;
|
||||
m_curG = 0x80;
|
||||
m_curB = 0x80;
|
||||
@@ -454,6 +472,9 @@ void GS::reset()
|
||||
m_curU = 0;
|
||||
m_curV = 0;
|
||||
m_curFog = 0;
|
||||
m_fogR = 0;
|
||||
m_fogG = 0;
|
||||
m_fogB = 0;
|
||||
m_prmodecont = true;
|
||||
m_pabe = false;
|
||||
m_texa = {0u, false, 0u};
|
||||
@@ -553,7 +574,6 @@ GSDebugSnapshot GS::getDebugSnapshot() const
|
||||
return snapshot;
|
||||
}
|
||||
|
||||
|
||||
std::vector<GSDebugHistoryEntry> GS::getDebugHistory() const
|
||||
{
|
||||
std::lock_guard<std::recursive_mutex> lock(m_stateMutex);
|
||||
@@ -1315,7 +1335,6 @@ void GS::processGIFPacket(const uint8_t *data, uint32_t sizeBytes)
|
||||
}
|
||||
});
|
||||
|
||||
|
||||
uint32_t offset = 0;
|
||||
while (offset + 16 <= sizeBytes)
|
||||
{
|
||||
@@ -1394,7 +1413,7 @@ bool GS::processNativePackedGIFPacket(const uint8_t *data, uint32_t sizeBytes)
|
||||
return false;
|
||||
|
||||
const bool processed = visitPackedGifPacket(data, sizeBytes, [&](const PackedGifPacketTag &tag)
|
||||
{
|
||||
{
|
||||
m_curQ = 1.0f;
|
||||
|
||||
recordGifTagDebugEventUnlocked(sizeBytes, tag.nloop, GIF_FMT_PACKED, tag.nreg);
|
||||
@@ -1415,8 +1434,7 @@ bool GS::processNativePackedGIFPacket(const uint8_t *data, uint32_t sizeBytes)
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
});
|
||||
return true; });
|
||||
|
||||
if (!processed)
|
||||
return false;
|
||||
@@ -1775,15 +1793,17 @@ void GS::writeRegister(uint8_t regAddr, uint64_t value)
|
||||
{
|
||||
case GS_REG_PRIM:
|
||||
{
|
||||
m_prim.type = static_cast<GSPrimType>(value & 0x7);
|
||||
m_prim.iip = ((value >> 3) & 1) != 0;
|
||||
m_prim.tme = ((value >> 4) & 1) != 0;
|
||||
m_prim.fge = ((value >> 5) & 1) != 0;
|
||||
m_prim.abe = ((value >> 6) & 1) != 0;
|
||||
m_prim.aa1 = ((value >> 7) & 1) != 0;
|
||||
m_prim.fst = ((value >> 8) & 1) != 0;
|
||||
m_prim.ctxt = ((value >> 9) & 1) != 0;
|
||||
m_prim.fix = ((value >> 10) & 1) != 0;
|
||||
m_primRegister = decodePrimRegister(value);
|
||||
if (m_prmodecont)
|
||||
{
|
||||
m_prim = m_primRegister;
|
||||
}
|
||||
else
|
||||
{
|
||||
// PRIM always selects the primitive topology. With AC=0, all
|
||||
// rendering attributes remain sourced from PRMODE.
|
||||
m_prim.type = m_primRegister.type;
|
||||
}
|
||||
m_vtxCount = 0;
|
||||
m_vtxIndex = 0;
|
||||
break;
|
||||
@@ -1912,21 +1932,24 @@ void GS::writeRegister(uint8_t regAddr, uint64_t value)
|
||||
break;
|
||||
}
|
||||
case GS_REG_PRMODECONT:
|
||||
{
|
||||
m_prmodecont = (value & 1) != 0;
|
||||
const GSPrimType type = m_primRegister.type;
|
||||
m_prim = m_prmodecont ? m_primRegister : m_prmodeRegister;
|
||||
m_prim.type = type;
|
||||
break;
|
||||
}
|
||||
case GS_REG_PRMODE:
|
||||
{
|
||||
m_prmodeRegister = decodePrimRegister(value);
|
||||
if (!m_prmodecont)
|
||||
{
|
||||
m_prim.iip = ((value >> 3) & 1) != 0;
|
||||
m_prim.tme = ((value >> 4) & 1) != 0;
|
||||
m_prim.fge = ((value >> 5) & 1) != 0;
|
||||
m_prim.abe = ((value >> 6) & 1) != 0;
|
||||
m_prim.aa1 = ((value >> 7) & 1) != 0;
|
||||
m_prim.fst = ((value >> 8) & 1) != 0;
|
||||
m_prim.ctxt = ((value >> 9) & 1) != 0;
|
||||
m_prim.fix = ((value >> 10) & 1) != 0;
|
||||
const GSPrimType type = m_primRegister.type;
|
||||
m_prim = m_prmodeRegister;
|
||||
m_prim.type = type;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GS_REG_TEXCLUT:
|
||||
m_texclut.cbw = static_cast<uint8_t>(value & 0x3Fu);
|
||||
m_texclut.cou = static_cast<uint8_t>((value >> 6) & 0x3Fu);
|
||||
@@ -2041,9 +2064,13 @@ void GS::writeRegister(uint8_t regAddr, uint64_t value)
|
||||
case GS_REG_PABE:
|
||||
m_pabe = (value & 1u) != 0u;
|
||||
break;
|
||||
case GS_REG_FOGCOL:
|
||||
m_fogR = static_cast<uint8_t>(value & 0xFFu);
|
||||
m_fogG = static_cast<uint8_t>((value >> 8) & 0xFFu);
|
||||
m_fogB = static_cast<uint8_t>((value >> 16) & 0xFFu);
|
||||
break;
|
||||
case GS_REG_TEXFLUSH:
|
||||
case GS_REG_SCANMSK:
|
||||
case GS_REG_FOGCOL:
|
||||
case GS_REG_DIMX:
|
||||
case GS_REG_DTHE:
|
||||
case GS_REG_COLCLAMP:
|
||||
@@ -2180,7 +2207,6 @@ void GS::performLocalToLocalTransfer()
|
||||
}
|
||||
break;
|
||||
|
||||
|
||||
// left -> right
|
||||
// bottom -> top (invert y)
|
||||
case 1:
|
||||
@@ -2271,9 +2297,6 @@ void GS::vertexKick(bool drawing)
|
||||
}
|
||||
});
|
||||
|
||||
if (!drawing)
|
||||
return;
|
||||
|
||||
int needed = 0;
|
||||
switch (m_prim.type)
|
||||
{
|
||||
@@ -2305,8 +2328,11 @@ void GS::vertexKick(bool drawing)
|
||||
if (m_vtxCount < needed)
|
||||
return;
|
||||
|
||||
m_rasterizer.drawPrimitive(this);
|
||||
recordDrawDebugEventUnlocked(needed);
|
||||
if (drawing)
|
||||
{
|
||||
m_rasterizer.drawPrimitive(this);
|
||||
recordDrawDebugEventUnlocked(needed);
|
||||
}
|
||||
|
||||
switch (m_prim.type)
|
||||
{
|
||||
|
||||
@@ -14,7 +14,6 @@
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
#include <sstream>
|
||||
|
||||
using namespace GSInternal;
|
||||
|
||||
@@ -27,8 +26,8 @@ namespace
|
||||
|
||||
u16 Rgba8888ToRgba5551(u32 c)
|
||||
{
|
||||
uint32_t r = ((c >> 0) & 0xFF) >> 3;
|
||||
uint32_t g = ((c >> 8) & 0xFF) >> 3;
|
||||
uint32_t r = ((c >> 0) & 0xFF) >> 3;
|
||||
uint32_t g = ((c >> 8) & 0xFF) >> 3;
|
||||
uint32_t b = ((c >> 16) & 0xFF) >> 3;
|
||||
uint32_t a = ((c >> 24) & 0xFF) >> 7;
|
||||
|
||||
@@ -37,8 +36,8 @@ namespace
|
||||
|
||||
u32 Rgba5551ToRgba8888(u16 c)
|
||||
{
|
||||
u32 r = ((c >> 0) & 0x1F) << 3;
|
||||
u32 g = ((c >> 5) & 0x1F) << 3;
|
||||
u32 r = ((c >> 0) & 0x1F) << 3;
|
||||
u32 g = ((c >> 5) & 0x1F) << 3;
|
||||
u32 b = ((c >> 10) & 0x1F) << 3;
|
||||
u32 a = ((c >> 15) & 0x01) << 7;
|
||||
|
||||
@@ -101,6 +100,28 @@ namespace
|
||||
std::atomic<uint32_t> s_debugPixelCount{0};
|
||||
std::atomic<uint32_t> s_debugContext1PrimitiveCount{0};
|
||||
std::atomic<uint32_t> s_debugFbp150PixelCount{0};
|
||||
|
||||
int wrapTextureCoordinate(int coordinate,
|
||||
int textureSize,
|
||||
uint8_t mode,
|
||||
uint16_t regionMin,
|
||||
uint16_t regionMax)
|
||||
{
|
||||
switch (mode & 0x3u)
|
||||
{
|
||||
case 0: // REPEAT
|
||||
return static_cast<int>(static_cast<uint32_t>(coordinate) & static_cast<uint32_t>(textureSize - 1));
|
||||
case 1: // CLAMP
|
||||
return clampInt(coordinate, 0, textureSize - 1);
|
||||
case 2: // REGION_CLAMP
|
||||
return std::min(std::max(coordinate, static_cast<int>(regionMin)), static_cast<int>(regionMax));
|
||||
case 3: // REGION_REPEAT
|
||||
return static_cast<int>((static_cast<uint32_t>(coordinate) & static_cast<uint32_t>(regionMin)) | static_cast<uint32_t>(regionMax));
|
||||
default:
|
||||
return coordinate;
|
||||
}
|
||||
}
|
||||
|
||||
bool passesAlphaTest(uint64_t testReg, uint8_t alpha)
|
||||
{
|
||||
if ((testReg & 0x1u) == 0u)
|
||||
@@ -132,29 +153,67 @@ namespace
|
||||
}
|
||||
}
|
||||
|
||||
struct AlphaTestResult
|
||||
struct PixelWriteMask
|
||||
{
|
||||
bool writeFramebuffer;
|
||||
bool preserveDestinationAlpha;
|
||||
bool writeRgb = true;
|
||||
bool writeAlpha = true;
|
||||
bool writeDepth = true;
|
||||
|
||||
bool writesFramebuffer() const
|
||||
{
|
||||
return writeRgb || writeAlpha;
|
||||
}
|
||||
|
||||
bool writesAnything() const
|
||||
{
|
||||
return writesFramebuffer() || writeDepth;
|
||||
}
|
||||
};
|
||||
|
||||
AlphaTestResult classifyAlphaTest(uint64_t testReg, uint8_t alpha)
|
||||
PixelWriteMask classifyAlphaTest(uint64_t testReg, uint8_t alpha, uint8_t framePsm)
|
||||
{
|
||||
const bool pass = passesAlphaTest(testReg, alpha);
|
||||
if (pass)
|
||||
return {true, false};
|
||||
return {};
|
||||
|
||||
// TEST.AFAIL controls what happens when the alpha comparison fails.
|
||||
switch (static_cast<uint8_t>((testReg >> 12) & 0x3u))
|
||||
{
|
||||
case 1: // FB_ONLY
|
||||
return {true, false};
|
||||
case 3: // RGB_ONLY
|
||||
return {true, true};
|
||||
case 0: // KEEP
|
||||
return {true, true, false};
|
||||
case 2: // ZB_ONLY
|
||||
return {false, false, true};
|
||||
case 3: // RGB_ONLY
|
||||
// RGB_ONLY is only distinct for RGBA32. The GS treats it as
|
||||
// FB_ONLY for RGB24 and RGBA16 framebuffers.
|
||||
if (framePsm == GS_PSM_CT32)
|
||||
return {true, false, false};
|
||||
return {true, true, false};
|
||||
case 0: // KEEP
|
||||
default:
|
||||
return {false, false};
|
||||
return {false, false, false};
|
||||
}
|
||||
}
|
||||
|
||||
bool passesDestinationAlphaTest(uint64_t testReg, uint8_t framePsm, uint32_t rawFramebufferPixel)
|
||||
{
|
||||
const bool date = ((testReg >> 14) & 0x1u) != 0u;
|
||||
if (!date)
|
||||
return true;
|
||||
|
||||
const bool datm = ((testReg >> 15) & 0x1u) != 0u;
|
||||
switch (framePsm)
|
||||
{
|
||||
case GS_PSM_CT32:
|
||||
return (((rawFramebufferPixel >> 31) & 0x1u) != 0u) == datm;
|
||||
case GS_PSM_CT16:
|
||||
case GS_PSM_CT16S:
|
||||
return (((rawFramebufferPixel >> 15) & 0x1u) != 0u) == datm;
|
||||
case GS_PSM_CT24:
|
||||
// RGB24 has no destination alpha, so DATE always passes.
|
||||
return true;
|
||||
default:
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -218,36 +277,52 @@ namespace
|
||||
|
||||
uint32_t swizzleClutIndexCSM1(uint32_t index)
|
||||
{
|
||||
return (index & 0xE7u) | ((index & 0x08u) << 1u) | ((index & 0x10u) >> 1u);
|
||||
// CSM1 swaps address bits 3 and 4. Preserve the remaining bits:
|
||||
// 16-bit CLUTs expose a ninth address bit through CSA[4].
|
||||
return (index & ~0x18u) | ((index & 0x08u) << 1u) | ((index & 0x10u) >> 1u);
|
||||
}
|
||||
|
||||
// TODO: clut cache
|
||||
uint32_t resolveClutIndex(uint8_t index, uint8_t csm, uint8_t csa, uint8_t sourcePsm)
|
||||
uint32_t resolveClutIndex(uint8_t index, uint8_t cpsm, uint8_t csm, uint8_t csa, uint8_t sourcePsm)
|
||||
{
|
||||
uint32_t clutIndex = static_cast<uint32_t>(index);
|
||||
|
||||
// CSM2 addresses the source directly through TEXCLUT. CSA is required
|
||||
// to be zero there, so it must not offset the source coordinates.
|
||||
if (csm != 0u)
|
||||
return (sourcePsm == GS_PSM_T4 ||
|
||||
sourcePsm == GS_PSM_T4HH ||
|
||||
sourcePsm == GS_PSM_T4HL)
|
||||
? (clutIndex & 0x0Fu)
|
||||
: clutIndex;
|
||||
|
||||
const bool is16BitClut = cpsm == GS_PSM_CT16 || cpsm == GS_PSM_CT16S;
|
||||
const uint32_t csaMask = is16BitClut ? 0x1Fu : 0x0Fu;
|
||||
const uint32_t clutIndexMask = is16BitClut ? 0x1FFu : 0x0FFu;
|
||||
const uint32_t clutBase = (static_cast<uint32_t>(csa) & csaMask) << 4u;
|
||||
|
||||
switch (sourcePsm)
|
||||
{
|
||||
case GS_PSM_T4:
|
||||
case GS_PSM_T4HH:
|
||||
case GS_PSM_T4HL:
|
||||
{
|
||||
clutIndex = (static_cast<uint32_t>(csa) << 4u) | (clutIndex & 0x0Fu);
|
||||
|
||||
if (csm == 0u)
|
||||
clutIndex = swizzleClutIndexCSM1(clutIndex);
|
||||
}
|
||||
break;
|
||||
clutIndex = clutBase + (clutIndex & 0x0Fu);
|
||||
break;
|
||||
case GS_PSM_T8:
|
||||
case GS_PSM_T8H:
|
||||
if (csm == 0)
|
||||
clutIndex = swizzleClutIndexCSM1(clutIndex);
|
||||
clutIndex = clutBase + clutIndex;
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
return clutIndex;
|
||||
}
|
||||
|
||||
return clutIndex;
|
||||
return swizzleClutIndexCSM1(clutIndex & clutIndexMask);
|
||||
}
|
||||
|
||||
int textureDimension(uint8_t exponent)
|
||||
{
|
||||
// TEX0.TW/TH saturate at 1024 pixels on the GS.
|
||||
return 1 << std::min<uint32_t>(exponent, 10u);
|
||||
}
|
||||
|
||||
bool tex1UsesLinearFilter(uint64_t tex1)
|
||||
@@ -399,7 +474,7 @@ void GSRasterizer::drawPrimitive(GS *gs)
|
||||
const auto &ctx = gs->activeContext();
|
||||
int px = static_cast<int>(v.x) - (ctx.xyoffset.ofx >> 4);
|
||||
int py = static_cast<int>(v.y) - (ctx.xyoffset.ofy >> 4);
|
||||
writePixel(gs, px, py, static_cast<u32>(v.z), v.r, v.g, v.b, v.a);
|
||||
writePixel(gs, px, py, static_cast<u32>(v.z), v.r, v.g, v.b, v.a, v.fog);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
@@ -407,51 +482,72 @@ void GSRasterizer::drawPrimitive(GS *gs)
|
||||
}
|
||||
}
|
||||
|
||||
void GSRasterizer::writePixel(GS *gs, int x, int y, int z, uint8_t r, uint8_t g, uint8_t b, uint8_t a)
|
||||
void GSRasterizer::writePixel(GS *gs, int x, int y, int z, uint8_t r, uint8_t g, uint8_t b, uint8_t a, uint8_t fog)
|
||||
{
|
||||
const auto &ctx = gs->activeContext();
|
||||
|
||||
if (x < ctx.scissor.x0 || x > ctx.scissor.x1 ||
|
||||
y < ctx.scissor.y0 || y > ctx.scissor.y1)
|
||||
if (x < ctx.scissor.x0 || x > ctx.scissor.x1 || y < ctx.scissor.y0 || y > ctx.scissor.y1)
|
||||
return;
|
||||
|
||||
const AlphaTestResult alphaTest = classifyAlphaTest(ctx.test, a);
|
||||
if (gs->m_prim.fge)
|
||||
{
|
||||
const uint32_t inverseFog = 255u - fog;
|
||||
auto applyFog = [&](uint8_t input, uint8_t fogColor) -> uint8_t
|
||||
{
|
||||
return static_cast<uint8_t>(((static_cast<uint32_t>(fog) * input) >> 8) + ((inverseFog * fogColor) >> 8));
|
||||
};
|
||||
|
||||
if (!alphaTest.writeFramebuffer)
|
||||
return;
|
||||
r = applyFog(r, gs->m_fogR);
|
||||
g = applyFog(g, gs->m_fogG);
|
||||
b = applyFog(b, gs->m_fogB);
|
||||
}
|
||||
|
||||
u8* vram = gs->m_vram;
|
||||
|
||||
const u32 fbp = GSInternal::framePageBaseToBlock(ctx.frame.fbp);
|
||||
const u32 fbw = std::max<u32>(ctx.frame.fbw, 1u);
|
||||
const u32 fbp = GSInternal::framePageBaseToBlock(ctx.frame.fbp);
|
||||
const u32 fbw = std::max<u32>(ctx.frame.fbw, 1u);
|
||||
const u32 fpsm = ctx.frame.psm;
|
||||
const u32 fmsk = ctx.frame.fbmsk;
|
||||
const u32 zbp = GSInternal::framePageBaseToBlock(ctx.zbuf.zbp);
|
||||
const u32 zpsm = ctx.zbuf.psm;
|
||||
|
||||
const PixelWriteMask writeMask = classifyAlphaTest(ctx.test, a, static_cast<uint8_t>(fpsm));
|
||||
if (!writeMask.writesAnything())
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
const uint32_t ztestMethod = static_cast<uint32_t>((ctx.test >> 17) & 3u);
|
||||
const bool alphaBlendEnabled = gs->m_prim.abe;
|
||||
const bool destinationAlpha = alphaTest.preserveDestinationAlpha;
|
||||
const bool preserveDestinationAlpha = writeMask.writeRgb && !writeMask.writeAlpha && fpsm == GS_PSM_CT32;
|
||||
const bool destinationAlphaTestNeedsRead = ((ctx.test >> 14) & 0x1u) != 0u && (fpsm == GS_PSM_CT32 || fpsm == GS_PSM_CT16 || fpsm == GS_PSM_CT16S);
|
||||
|
||||
// small optimization, avoid reading the framebuffer for simple draws
|
||||
// TODO: only one address lookup for rmw
|
||||
const bool frmw = (ctx.frame.fbmsk != 0) || alphaBlendEnabled || destinationAlpha;
|
||||
const bool frmw = destinationAlphaTestNeedsRead || (writeMask.writesFramebuffer() && ((ctx.frame.fbmsk != 0) || alphaBlendEnabled || preserveDestinationAlpha));
|
||||
|
||||
u32 rawFramebufferPixel = 0;
|
||||
u32 fbrgba = 0;
|
||||
if (frmw)
|
||||
{
|
||||
fbrgba = gs->ReadVram(fpsm, fbp, fbw, x, y);
|
||||
rawFramebufferPixel = gs->ReadVram(fpsm, fbp, fbw, x, y);
|
||||
fbrgba = rawFramebufferPixel;
|
||||
|
||||
if (bitsPerPixel(fpsm) == 16)
|
||||
{
|
||||
fbrgba = Rgba5551ToRgba8888(fbrgba);
|
||||
}
|
||||
else if (fpsm == GS_PSM_CT24)
|
||||
{
|
||||
// The GS supplies 0x80 as destination alpha for RGB24 blending.
|
||||
fbrgba |= 0x80000000u;
|
||||
}
|
||||
}
|
||||
|
||||
uint ztest_method = (ctx.test >> 17) & 3;
|
||||
|
||||
if (!passesDestinationAlphaTest(ctx.test, static_cast<uint8_t>(fpsm), rawFramebufferPixel))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
bool zpass = false;
|
||||
switch (ztest_method)
|
||||
uint32_t storedZ = 0u;
|
||||
switch (ztestMethod)
|
||||
{
|
||||
case 0:
|
||||
zpass = false;
|
||||
@@ -460,10 +556,12 @@ void GSRasterizer::writePixel(GS *gs, int x, int y, int z, uint8_t r, uint8_t g,
|
||||
zpass = true;
|
||||
break;
|
||||
case 2:
|
||||
zpass = z >= gs->ReadVram(zpsm, zbp, fbw, x, y);
|
||||
storedZ = gs->ReadVram(zpsm, zbp, fbw, x, y);
|
||||
zpass = static_cast<uint32_t>(z) >= storedZ;
|
||||
break;
|
||||
case 3:
|
||||
zpass = z > gs->ReadVram(zpsm, zbp, fbw, x, y);
|
||||
storedZ = gs->ReadVram(zpsm, zbp, fbw, x, y);
|
||||
zpass = static_cast<uint32_t>(z) > storedZ;
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -472,81 +570,79 @@ void GSRasterizer::writePixel(GS *gs, int x, int y, int z, uint8_t r, uint8_t g,
|
||||
return;
|
||||
}
|
||||
|
||||
const u8 srcR = r;
|
||||
const u8 srcG = g;
|
||||
const u8 srcB = b;
|
||||
|
||||
if (gs->m_prim.abe)
|
||||
if (writeMask.writesFramebuffer())
|
||||
{
|
||||
uint8_t dr = fbrgba & 0xFF;
|
||||
uint8_t dg = (fbrgba >> 8) & 0xFF;
|
||||
uint8_t db = (fbrgba >> 16) & 0xFF;
|
||||
uint8_t da = (fbrgba >> 24) & 0xFF;
|
||||
const u8 srcR = r;
|
||||
const u8 srcG = g;
|
||||
const u8 srcB = b;
|
||||
|
||||
// PABE disables alpha blending when the source alpha MSB is clear.
|
||||
if (!(gs->m_pabe && (a & 0x80u) == 0u))
|
||||
if (gs->m_prim.abe)
|
||||
{
|
||||
uint64_t alphaReg = ctx.alpha;
|
||||
uint8_t asel = alphaReg & 3;
|
||||
uint8_t bsel = (alphaReg >> 2) & 3;
|
||||
uint8_t csel = (alphaReg >> 4) & 3;
|
||||
uint8_t dsel = (alphaReg >> 6) & 3;
|
||||
uint8_t fix = static_cast<uint8_t>((alphaReg >> 32) & 0xFF);
|
||||
uint8_t dr = fbrgba & 0xFF;
|
||||
uint8_t dg = (fbrgba >> 8) & 0xFF;
|
||||
uint8_t db = (fbrgba >> 16) & 0xFF;
|
||||
uint8_t da = (fbrgba >> 24) & 0xFF;
|
||||
|
||||
auto pickRGB = [&](uint8_t sel, int cs, int cd) -> int
|
||||
// PABE disables alpha blending when the source alpha MSB is clear.
|
||||
if (!(gs->m_pabe && (a & 0x80u) == 0u))
|
||||
{
|
||||
if (sel == 0)
|
||||
return cs;
|
||||
if (sel == 1)
|
||||
return cd;
|
||||
return 0;
|
||||
};
|
||||
int cAlpha = (csel == 0) ? a : (csel == 1) ? da
|
||||
: fix;
|
||||
uint64_t alphaReg = ctx.alpha;
|
||||
uint8_t asel = alphaReg & 3;
|
||||
uint8_t bsel = (alphaReg >> 2) & 3;
|
||||
uint8_t csel = (alphaReg >> 4) & 3;
|
||||
uint8_t dsel = (alphaReg >> 6) & 3;
|
||||
uint8_t fix = static_cast<uint8_t>((alphaReg >> 32) & 0xFF);
|
||||
|
||||
r = clampU8(((pickRGB(asel, r, dr) - pickRGB(bsel, r, dr)) * cAlpha >> 7) + pickRGB(dsel, r, dr));
|
||||
g = clampU8(((pickRGB(asel, g, dg) - pickRGB(bsel, g, dg)) * cAlpha >> 7) + pickRGB(dsel, g, dg));
|
||||
b = clampU8(((pickRGB(asel, b, db) - pickRGB(bsel, b, db)) * cAlpha >> 7) + pickRGB(dsel, b, db));
|
||||
auto pickRGB = [&](uint8_t sel, int cs, int cd) -> int
|
||||
{
|
||||
if (sel == 0)
|
||||
return cs;
|
||||
if (sel == 1)
|
||||
return cd;
|
||||
return 0;
|
||||
};
|
||||
int cAlpha = (csel == 0) ? a : (csel == 1) ? da
|
||||
: fix;
|
||||
|
||||
r = clampU8(((pickRGB(asel, r, dr) - pickRGB(bsel, r, dr)) * cAlpha >> 7) + pickRGB(dsel, r, dr));
|
||||
g = clampU8(((pickRGB(asel, g, dg) - pickRGB(bsel, g, dg)) * cAlpha >> 7) + pickRGB(dsel, g, dg));
|
||||
b = clampU8(((pickRGB(asel, b, db) - pickRGB(bsel, b, db)) * cAlpha >> 7) + pickRGB(dsel, b, db));
|
||||
}
|
||||
else
|
||||
{
|
||||
r = srcR;
|
||||
g = srcG;
|
||||
b = srcB;
|
||||
}
|
||||
}
|
||||
else
|
||||
|
||||
if (writeMask.writeAlpha && (ctx.fba & 0x1ull) != 0ull && ctx.frame.psm != GS_PSM_CT24)
|
||||
{
|
||||
r = srcR;
|
||||
g = srcG;
|
||||
b = srcB;
|
||||
a = static_cast<uint8_t>(a | 0x80u);
|
||||
}
|
||||
|
||||
u32 pixel = pack32(r, g, b, a);
|
||||
|
||||
if (ctx.frame.fbmsk != 0)
|
||||
{
|
||||
pixel = (pixel & ~ctx.frame.fbmsk) | (fbrgba & ctx.frame.fbmsk);
|
||||
}
|
||||
|
||||
if (preserveDestinationAlpha)
|
||||
{
|
||||
pixel = (pixel & 0x00FFFFFFu) | (fbrgba & 0xFF000000u);
|
||||
}
|
||||
|
||||
// format conversion
|
||||
if (bitsPerPixel(fpsm) == 16)
|
||||
{
|
||||
pixel = Rgba8888ToRgba5551(pixel);
|
||||
}
|
||||
|
||||
gs->WriteVram(fpsm, fbp, fbw, x, y, pixel);
|
||||
}
|
||||
|
||||
u32 fbmask = ctx.frame.fbmsk;
|
||||
bool zmask = ctx.zbuf.zmask;
|
||||
|
||||
if (!alphaTest.preserveDestinationAlpha &&
|
||||
(ctx.fba & 0x1ull) != 0ull &&
|
||||
ctx.frame.psm != GS_PSM_CT24)
|
||||
{
|
||||
a = static_cast<uint8_t>(a | 0x80u);
|
||||
}
|
||||
|
||||
u32 pixel = pack32(r, g, b, a);
|
||||
|
||||
if (fbmask != 0)
|
||||
{
|
||||
pixel = (pixel & ~fbmask) | (fbrgba & fbmask);
|
||||
}
|
||||
|
||||
if (alphaTest.preserveDestinationAlpha)
|
||||
{
|
||||
pixel = (pixel & 0x00FFFFFFu) | (fbrgba & 0xFF000000u);
|
||||
}
|
||||
|
||||
// format conversion
|
||||
if (bitsPerPixel(fpsm) == 16)
|
||||
{
|
||||
pixel = Rgba8888ToRgba5551(pixel);
|
||||
}
|
||||
|
||||
gs->WriteVram(fpsm, fbp, fbw, x, y, pixel);
|
||||
|
||||
if (!zmask)
|
||||
if (writeMask.writeDepth && !ctx.zbuf.zmask)
|
||||
{
|
||||
gs->WriteVram(zpsm, zbp, fbw, x, y, z);
|
||||
}
|
||||
@@ -560,12 +656,11 @@ uint32_t GSRasterizer::lookupCLUT(GS *gs,
|
||||
uint8_t csa,
|
||||
uint8_t sourcePsm)
|
||||
{
|
||||
const uint32_t clutIndex = resolveClutIndex(index, csm, csa, sourcePsm);
|
||||
const uint32_t clutIndex = resolveClutIndex(index, cpsm, csm, csa, sourcePsm);
|
||||
const uint32_t clutWidth = (gs->m_texclut.cbw != 0u) ? static_cast<uint32_t>(gs->m_texclut.cbw) : 1u;
|
||||
const uint32_t clutX = static_cast<uint32_t>(gs->m_texclut.cou) + (clutIndex & 0x0Fu);
|
||||
const uint32_t clutY = static_cast<uint32_t>(gs->m_texclut.cov) + (clutIndex >> 4);
|
||||
|
||||
|
||||
switch (cpsm)
|
||||
{
|
||||
case GS_PSM_CT32:
|
||||
@@ -588,8 +683,15 @@ uint32_t GSRasterizer::sampleTexture(GS *gs, float s, float t, float q, uint16_t
|
||||
const auto &ctx = gs->activeContext();
|
||||
const auto &tex = ctx.tex0;
|
||||
|
||||
int texW = 1 << tex.tw;
|
||||
int texH = 1 << tex.th;
|
||||
const int texW = textureDimension(tex.tw);
|
||||
const int texH = textureDimension(tex.th);
|
||||
const uint64_t clamp = ctx.clamp;
|
||||
const uint8_t wrapU = static_cast<uint8_t>(clamp & 0x3u);
|
||||
const uint8_t wrapV = static_cast<uint8_t>((clamp >> 2) & 0x3u);
|
||||
const uint16_t minU = static_cast<uint16_t>((clamp >> 4) & 0x3FFu);
|
||||
const uint16_t maxU = static_cast<uint16_t>((clamp >> 14) & 0x3FFu);
|
||||
const uint16_t minV = static_cast<uint16_t>((clamp >> 24) & 0x3FFu);
|
||||
const uint16_t maxV = static_cast<uint16_t>((clamp >> 34) & 0x3FFu);
|
||||
|
||||
float texUf, texVf;
|
||||
if (gs->m_prim.fst)
|
||||
@@ -606,8 +708,8 @@ uint32_t GSRasterizer::sampleTexture(GS *gs, float s, float t, float q, uint16_t
|
||||
|
||||
auto samplePoint = [&](int sampleU, int sampleV) -> uint32_t
|
||||
{
|
||||
sampleU = clampInt(sampleU, 0, texW - 1);
|
||||
sampleV = clampInt(sampleV, 0, texH - 1);
|
||||
sampleU = wrapTextureCoordinate(sampleU, texW, wrapU, minU, maxU);
|
||||
sampleV = wrapTextureCoordinate(sampleV, texH, wrapV, minV, maxV);
|
||||
|
||||
u32 out = gs->ReadVram(tex.psm, tex.tbp0, tex.tbw, sampleU, sampleV);
|
||||
|
||||
@@ -747,12 +849,8 @@ void GSRasterizer::drawSprite(GS *gs)
|
||||
if (gs->m_prim.tme)
|
||||
{
|
||||
const auto &tex = ctx.tex0;
|
||||
int texW = 1 << tex.tw;
|
||||
int texH = 1 << tex.th;
|
||||
if (texW == 0)
|
||||
texW = 1;
|
||||
if (texH == 0)
|
||||
texH = 1;
|
||||
const int texW = textureDimension(tex.tw);
|
||||
const int texH = textureDimension(tex.th);
|
||||
|
||||
float u0f, v0f, u1f, v1f;
|
||||
if (gs->m_prim.fst)
|
||||
@@ -799,10 +897,7 @@ void GSRasterizer::drawSprite(GS *gs)
|
||||
}
|
||||
else
|
||||
{
|
||||
texel = sampleTexture(gs,
|
||||
texUf / static_cast<float>(texW),
|
||||
texVf / static_cast<float>(texH),
|
||||
1.0f, 0u, 0u);
|
||||
texel = sampleTexture(gs, texUf / static_cast<float>(texW), texVf / static_cast<float>(texH), 1.0f, 0u, 0u);
|
||||
}
|
||||
|
||||
uint8_t tr = static_cast<uint8_t>(texel & 0xFF);
|
||||
@@ -811,7 +906,7 @@ void GSRasterizer::drawSprite(GS *gs)
|
||||
uint8_t ta = static_cast<uint8_t>((texel >> 24) & 0xFF);
|
||||
|
||||
const TextureCombineResult color = combineTexture(tex, r, g, b, a, tr, tg, tb, ta);
|
||||
writePixel(gs, x, y, z1, color.r, color.g, color.b, color.a);
|
||||
writePixel(gs, x, y, z1, color.r, color.g, color.b, color.a, v1.fog);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -819,7 +914,7 @@ void GSRasterizer::drawSprite(GS *gs)
|
||||
{
|
||||
for (int y = drawY0; y <= drawY1; ++y)
|
||||
for (int x = drawX0; x <= drawX1; ++x)
|
||||
writePixel(gs, x, y, z1, r, g, b, a);
|
||||
writePixel(gs, x, y, z1, r, g, b, a, v1.fog);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -904,15 +999,12 @@ void GSRasterizer::drawTriangle(GS *gs)
|
||||
}
|
||||
else
|
||||
{
|
||||
const float invQ0 = 1.0f / fabsQ(v0.q);
|
||||
const float invQ1 = 1.0f / fabsQ(v1.q);
|
||||
const float invQ2 = 1.0f / fabsQ(v2.q);
|
||||
const float sOverQ = (v0.s * invQ0) * w0 + (v1.s * invQ1) * w1 + (v2.s * invQ2) * w2;
|
||||
const float tOverQ = (v0.t * invQ0) * w0 + (v1.t * invQ1) * w1 + (v2.t * invQ2) * w2;
|
||||
const float invQ = invQ0 * w0 + invQ1 * w1 + invQ2 * w2;
|
||||
iq = (std::fabs(invQ) > 1.0e-8f) ? (1.0f / invQ) : 1.0f;
|
||||
is = sOverQ * iq;
|
||||
it = tOverQ * iq;
|
||||
// The GS DDA interpolates the homogeneous S, T and Q
|
||||
// values. Texel coordinates are calculated from S/Q and
|
||||
// T/Q only after interpolation.
|
||||
is = v0.s * w0 + v1.s * w1 + v2.s * w2;
|
||||
it = v0.t * w0 + v1.t * w1 + v2.t * w2;
|
||||
iq = v0.q * w0 + v1.q * w1 + v2.q * w2;
|
||||
iu = 0;
|
||||
iv = 0;
|
||||
}
|
||||
@@ -937,7 +1029,8 @@ void GSRasterizer::drawTriangle(GS *gs)
|
||||
a = color.a;
|
||||
}
|
||||
|
||||
writePixel(gs, x, y, static_cast<u32>(z + 0.5), r, g, b, a);
|
||||
const uint8_t fog = clampU8(static_cast<int>(v0.fog * w0 + v1.fog * w1 + v2.fog * w2));
|
||||
writePixel(gs, x, y, static_cast<u32>(z + 0.5), r, g, b, a, fog);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -987,8 +1080,8 @@ void GSRasterizer::drawLine(GS *gs)
|
||||
}
|
||||
|
||||
double z = (v0.z + (v1.z - v0.z) * t);
|
||||
|
||||
writePixel(gs, x0, y0, static_cast<u32>(z), r, g, b, a);
|
||||
const uint8_t fog = clampU8(static_cast<int>(v0.fog + (v1.fog - v0.fog) * t));
|
||||
writePixel(gs, x0, y0, static_cast<u32>(z), r, g, b, a, fog);
|
||||
|
||||
if (x0 == x1 && y0 == y1)
|
||||
break;
|
||||
|
||||
@@ -351,6 +351,7 @@ bool PS2Memory::initialize(size_t ramSize)
|
||||
m_vu1Data = new uint8_t[PS2_VU1_DATA_SIZE];
|
||||
std::memset(m_vu1Code, 0, PS2_VU1_CODE_SIZE);
|
||||
std::memset(m_vu1Data, 0, PS2_VU1_DATA_SIZE);
|
||||
markVU0CodeModified();
|
||||
markVU1CodeModified();
|
||||
|
||||
// Initialize VIF registers
|
||||
@@ -765,7 +766,9 @@ void PS2Memory::write8(uint32_t address, uint8_t value)
|
||||
{
|
||||
(void)vuLimit;
|
||||
vuMem[vuOffset] = value;
|
||||
if (vuMem == m_vu1Code)
|
||||
if (vuMem == m_vu0Code)
|
||||
markVU0CodeModified();
|
||||
else if (vuMem == m_vu1Code)
|
||||
markVU1CodeModified();
|
||||
return;
|
||||
}
|
||||
@@ -806,7 +809,9 @@ void PS2Memory::write16(uint32_t address, uint16_t value)
|
||||
if (uint8_t *vuMem = mapVuMemory(physAddr, sizeof(uint16_t), vuOffset, vuLimit))
|
||||
{
|
||||
storeScalar<uint16_t>(vuMem, vuOffset, vuLimit, value, "write16 vu", address);
|
||||
if (vuMem == m_vu1Code)
|
||||
if (vuMem == m_vu0Code)
|
||||
markVU0CodeModified();
|
||||
else if (vuMem == m_vu1Code)
|
||||
markVU1CodeModified();
|
||||
return;
|
||||
}
|
||||
@@ -868,7 +873,9 @@ void PS2Memory::write32(uint32_t address, uint32_t value)
|
||||
if (uint8_t *vuMem = mapVuMemory(physAddr, sizeof(uint32_t), vuOffset, vuLimit))
|
||||
{
|
||||
storeScalar<uint32_t>(vuMem, vuOffset, vuLimit, value, "write32 vu", address);
|
||||
if (vuMem == m_vu1Code)
|
||||
if (vuMem == m_vu0Code)
|
||||
markVU0CodeModified();
|
||||
else if (vuMem == m_vu1Code)
|
||||
markVU1CodeModified();
|
||||
return;
|
||||
}
|
||||
@@ -921,7 +928,9 @@ void PS2Memory::write64(uint32_t address, uint64_t value)
|
||||
if (uint8_t *vuMem = mapVuMemory(physAddr, sizeof(uint64_t), vuOffset, vuLimit))
|
||||
{
|
||||
storeScalar<uint64_t>(vuMem, vuOffset, vuLimit, value, "write64 vu", address);
|
||||
if (vuMem == m_vu1Code)
|
||||
if (vuMem == m_vu0Code)
|
||||
markVU0CodeModified();
|
||||
else if (vuMem == m_vu1Code)
|
||||
markVU1CodeModified();
|
||||
return;
|
||||
}
|
||||
@@ -962,7 +971,9 @@ void PS2Memory::write128(uint32_t address, __m128i value)
|
||||
{
|
||||
inRange(vuOffset, sizeof(__m128i), vuLimit, "write128 vu", address);
|
||||
_mm_storeu_si128(reinterpret_cast<__m128i *>(vuMem + vuOffset), value);
|
||||
if (vuMem == m_vu1Code)
|
||||
if (vuMem == m_vu0Code)
|
||||
markVU0CodeModified();
|
||||
else if (vuMem == m_vu1Code)
|
||||
markVU1CodeModified();
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -209,6 +209,7 @@ namespace
|
||||
ctx->vu0_mac_flags = 0;
|
||||
ctx->vu0_status = 0;
|
||||
ctx->vu0_q = 1.0f;
|
||||
ctx->vu0_r = _mm_castsi128_ps(_mm_set1_epi32(0x3F800000));
|
||||
ctx->vu0_vpu_stat = 0;
|
||||
ctx->vu0_vpu_stat2 = 0;
|
||||
}
|
||||
@@ -230,11 +231,16 @@ namespace
|
||||
state.q = ctx->vu0_q;
|
||||
state.p = ctx->vu0_p;
|
||||
state.i = ctx->vu0_i;
|
||||
alignas(16) uint32_t rWords[4]{};
|
||||
_mm_storeu_si128(reinterpret_cast<__m128i *>(rWords), _mm_castps_si128(ctx->vu0_r));
|
||||
state.r = 0x3F800000u | (rWords[0] & 0x007FFFFFu);
|
||||
state.pc = ctx->vu0_pc;
|
||||
state.mac = ctx->vu0_mac_flags;
|
||||
state.clip = ctx->vu0_clip_flags;
|
||||
state.status = ctx->vu0_status;
|
||||
state.itop = ctx->vu0_itop;
|
||||
state.dBitEnabled = (ctx->vu0_fbrst & (1u << 2)) != 0u;
|
||||
state.tBitEnabled = (ctx->vu0_fbrst & (1u << 3)) != 0u;
|
||||
|
||||
state.vf[0][0] = 0.0f;
|
||||
state.vf[0][1] = 0.0f;
|
||||
@@ -258,6 +264,7 @@ namespace
|
||||
ctx->vu0_q = state.q;
|
||||
ctx->vu0_p = state.p;
|
||||
ctx->vu0_i = state.i;
|
||||
ctx->vu0_r = _mm_castsi128_ps(_mm_set1_epi32(static_cast<int32_t>(state.r)));
|
||||
ctx->vu0_mac_flags = state.mac;
|
||||
ctx->vu0_clip_flags = state.clip;
|
||||
ctx->vu0_clip_flags2 = state.clip;
|
||||
@@ -265,7 +272,7 @@ namespace
|
||||
ctx->vu0_itop = state.itop;
|
||||
ctx->vu0_pc = state.pc;
|
||||
ctx->vu0_tpc = state.pc;
|
||||
ctx->vu0_vpu_stat = 0;
|
||||
ctx->vu0_vpu_stat = (ctx->vu0_vpu_stat & 0xFF00u) | (state.stoppedByD ? (1u << 1) : 0u) | (state.stoppedByT ? (1u << 2) : 0u);
|
||||
ctx->vu0_vpu_stat2 = 0;
|
||||
|
||||
ctx->vu0_vf[0] = _mm_set_ps(1.0f, 0.0f, 0.0f, 0.0f);
|
||||
@@ -523,6 +530,9 @@ PS2Runtime::PS2Runtime()
|
||||
|
||||
// R0 is always zero in MIPS
|
||||
m_cpuContext.r[0] = _mm_set1_epi32(0);
|
||||
m_cpuContext.vu0_vf[0] = _mm_set_ps(1.0f, 0.0f, 0.0f, 0.0f);
|
||||
m_cpuContext.vu0_q = 1.0f;
|
||||
m_cpuContext.vu0_r = _mm_castsi128_ps(_mm_set1_epi32(0x3F800000));
|
||||
|
||||
// Stack pointer (SP) and global pointer (GP) will be set by the loaded ELF
|
||||
|
||||
@@ -647,13 +657,31 @@ bool PS2Runtime::syncCoreSubsystems()
|
||||
{ m_gs.processGIFPacket(data, size); });
|
||||
m_memory.setGifArbiter(&m_gifArbiter);
|
||||
m_memory.setVu1MscalCallback([this](uint32_t startPC, uint32_t top, uint32_t itop)
|
||||
{ m_vu1.execute(m_memory.getVU1Code(), PS2_VU1_CODE_SIZE,
|
||||
m_memory.getVU1Data(), PS2_VU1_DATA_SIZE,
|
||||
m_gs, &m_memory, startPC, top, itop, 65536); });
|
||||
{
|
||||
m_vu1.state().dBitEnabled =
|
||||
(m_cpuContext.vu0_fbrst & (1u << 10)) != 0u;
|
||||
m_vu1.state().tBitEnabled =
|
||||
(m_cpuContext.vu0_fbrst & (1u << 11)) != 0u;
|
||||
m_vu1.execute(m_memory.getVU1Code(), PS2_VU1_CODE_SIZE,
|
||||
m_memory.getVU1Data(), PS2_VU1_DATA_SIZE,
|
||||
m_gs, &m_memory, startPC, top, itop, 65536);
|
||||
m_cpuContext.vu0_vpu_stat =
|
||||
(m_cpuContext.vu0_vpu_stat & ~0x0600u) |
|
||||
(m_vu1.state().stoppedByD ? 0x0200u : 0u) |
|
||||
(m_vu1.state().stoppedByT ? 0x0400u : 0u); });
|
||||
m_memory.setVu1MscntCallback([this](uint32_t top, uint32_t itop)
|
||||
{ m_vu1.resume(m_memory.getVU1Code(), PS2_VU1_CODE_SIZE,
|
||||
m_memory.getVU1Data(), PS2_VU1_DATA_SIZE,
|
||||
m_gs, &m_memory, top, itop, 65536); });
|
||||
{
|
||||
m_vu1.state().dBitEnabled =
|
||||
(m_cpuContext.vu0_fbrst & (1u << 10)) != 0u;
|
||||
m_vu1.state().tBitEnabled =
|
||||
(m_cpuContext.vu0_fbrst & (1u << 11)) != 0u;
|
||||
m_vu1.resume(m_memory.getVU1Code(), PS2_VU1_CODE_SIZE,
|
||||
m_memory.getVU1Data(), PS2_VU1_DATA_SIZE,
|
||||
m_gs, &m_memory, top, itop, 65536);
|
||||
m_cpuContext.vu0_vpu_stat =
|
||||
(m_cpuContext.vu0_vpu_stat & ~0x0600u) |
|
||||
(m_vu1.state().stoppedByD ? 0x0200u : 0u) |
|
||||
(m_vu1.state().stoppedByT ? 0x0400u : 0u); });
|
||||
resetIop();
|
||||
m_vu0.reset();
|
||||
m_vu1.reset();
|
||||
@@ -1157,6 +1185,10 @@ void PS2Runtime::reportMissingFunction(uint8_t *rdram,
|
||||
const uint32_t gp = static_cast<uint32_t>(_mm_extract_epi32(ctx->r[28], 0));
|
||||
const uint32_t a0 = static_cast<uint32_t>(_mm_extract_epi32(ctx->r[4], 0));
|
||||
const uint32_t a1 = static_cast<uint32_t>(_mm_extract_epi32(ctx->r[5], 0));
|
||||
const uint32_t a2 = static_cast<uint32_t>(_mm_extract_epi32(ctx->r[6], 0));
|
||||
const uint32_t a3 = static_cast<uint32_t>(_mm_extract_epi32(ctx->r[7], 0));
|
||||
const uint32_t s0 = static_cast<uint32_t>(_mm_extract_epi32(ctx->r[16], 0));
|
||||
const uint32_t s1 = static_cast<uint32_t>(_mm_extract_epi32(ctx->r[17], 0));
|
||||
const uint32_t v0 = static_cast<uint32_t>(_mm_extract_epi32(ctx->r[2], 0));
|
||||
const uint32_t v1 = static_cast<uint32_t>(_mm_extract_epi32(ctx->r[3], 0));
|
||||
|
||||
@@ -1194,6 +1226,27 @@ void PS2Runtime::reportMissingFunction(uint8_t *rdram,
|
||||
readGuestU32Offset(a0, 0x08u, a0Word8) &&
|
||||
readGuestU32Offset(a0, 0x0cu, a0WordC);
|
||||
|
||||
uint32_t s0Word0 = 0u;
|
||||
uint32_t s0Word4 = 0u;
|
||||
uint32_t s0Word8 = 0u;
|
||||
uint32_t s0WordC = 0u;
|
||||
const bool s0Readable =
|
||||
readGuestU32Offset(s0, 0x00u, s0Word0) &&
|
||||
readGuestU32Offset(s0, 0x04u, s0Word4) &&
|
||||
readGuestU32Offset(s0, 0x08u, s0Word8) &&
|
||||
readGuestU32Offset(s0, 0x0cu, s0WordC);
|
||||
|
||||
uint32_t recordWord0 = 0u;
|
||||
uint32_t recordWord4 = 0u;
|
||||
uint32_t recordWord8 = 0u;
|
||||
uint32_t recordWordC = 0u;
|
||||
const bool recordReadable =
|
||||
s0Readable && s0Word4 != 0u &&
|
||||
readGuestU32Offset(s0Word4, 0x00u, recordWord0) &&
|
||||
readGuestU32Offset(s0Word4, 0x04u, recordWord4) &&
|
||||
readGuestU32Offset(s0Word4, 0x08u, recordWord8) &&
|
||||
readGuestU32Offset(s0Word4, 0x0cu, recordWordC);
|
||||
|
||||
uint32_t vtableSlot0 = 0u;
|
||||
uint32_t vtableSlot4 = 0u;
|
||||
uint32_t vtableSlot8 = 0u;
|
||||
@@ -1218,6 +1271,10 @@ void PS2Runtime::reportMissingFunction(uint8_t *rdram,
|
||||
<< " gp=0x" << gp
|
||||
<< " a0=0x" << a0
|
||||
<< " a1=0x" << a1
|
||||
<< " a2=0x" << a2
|
||||
<< " a3=0x" << a3
|
||||
<< " s0=0x" << s0
|
||||
<< " s1=0x" << s1
|
||||
<< " v0=0x" << v0
|
||||
<< " v1=0x" << v1
|
||||
<< " a0Readable=" << (a0Readable ? "yes" : "no")
|
||||
@@ -1225,6 +1282,16 @@ void PS2Runtime::reportMissingFunction(uint8_t *rdram,
|
||||
<< " a0[4]=0x" << a0Word4
|
||||
<< " a0[8]=0x" << a0Word8
|
||||
<< " a0[c]=0x" << a0WordC
|
||||
<< " s0Readable=" << (s0Readable ? "yes" : "no")
|
||||
<< " s0[0]=0x" << s0Word0
|
||||
<< " s0[4]=0x" << s0Word4
|
||||
<< " s0[8]=0x" << s0Word8
|
||||
<< " s0[c]=0x" << s0WordC
|
||||
<< " recordReadable=" << (recordReadable ? "yes" : "no")
|
||||
<< " record[0]=0x" << recordWord0
|
||||
<< " record[4]=0x" << recordWord4
|
||||
<< " record[8]=0x" << recordWord8
|
||||
<< " record[c]=0x" << recordWordC
|
||||
<< " vtableReadable=" << (vtableReadable ? "yes" : "no")
|
||||
<< " vtbl[0]=0x" << vtableSlot0
|
||||
<< " vtbl[4]=0x" << vtableSlot4
|
||||
@@ -2152,16 +2219,19 @@ void PS2Runtime::yieldGuestExecutionAfterWake()
|
||||
{
|
||||
GuestExecutionReleaseScope releaseGuestExecution(this);
|
||||
std::unique_lock<std::mutex> lock(m_guestExecutionHandoffMutex);
|
||||
m_guestExecutionHandoffCv.wait_for(lock, std::chrono::milliseconds(2), [&]()
|
||||
m_guestExecutionHandoffCv.wait_for(lock, std::chrono::milliseconds(1), [&]()
|
||||
{ return m_guestExecutionHandoffEpoch.load(std::memory_order_acquire) != handoffEpoch; });
|
||||
}
|
||||
}
|
||||
|
||||
bool PS2Runtime::shouldPreemptGuestExecution()
|
||||
{
|
||||
constexpr uint32_t kContendedYieldInterval = 1024u;
|
||||
constexpr uint32_t kUncontendedYieldInterval = 16384u;
|
||||
|
||||
thread_local uint32_t s_backEdgeYieldCounter = 0u;
|
||||
const uint32_t waiterCount = m_guestExecutionWaiters.load(std::memory_order_acquire);
|
||||
const uint32_t yieldInterval = (waiterCount != 0u) ? 64u : 100u;
|
||||
const uint32_t yieldInterval = (waiterCount != 0u) ? kContendedYieldInterval : kUncontendedYieldInterval;
|
||||
if (++s_backEdgeYieldCounter < yieldInterval)
|
||||
{
|
||||
return false;
|
||||
|
||||
@@ -144,7 +144,10 @@ void PS2Memory::processVIF0Data(const uint8_t *data, uint32_t sizeBytes)
|
||||
if (destAddr + copyBytes > PS2_VU0_CODE_SIZE)
|
||||
copyBytes = PS2_VU0_CODE_SIZE - destAddr;
|
||||
if (pos + copyBytes <= sizeBytes)
|
||||
{
|
||||
std::memcpy(m_vu0Code + destAddr, data + pos, copyBytes);
|
||||
markVU0CodeModified();
|
||||
}
|
||||
}
|
||||
|
||||
pos += mpgBytes;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -26,132 +26,4 @@ static inline int16_t IMM15(uint32_t i)
|
||||
return (int16_t)(int32_t)((int32_t)(raw << 17) >> 17);
|
||||
}
|
||||
|
||||
|
||||
static inline uint8_t vuUpperVfWriteReg(uint32_t upper)
|
||||
{
|
||||
const uint8_t op = upper & 0x3Fu;
|
||||
const uint8_t dest = DEST(upper);
|
||||
const uint8_t ft = FT(upper);
|
||||
const uint8_t fd = FD(upper);
|
||||
|
||||
if (dest == 0u)
|
||||
return 0u;
|
||||
|
||||
if (op <= 0x2Fu)
|
||||
return fd;
|
||||
|
||||
if (op >= 0x3Cu)
|
||||
{
|
||||
const uint8_t specialOp = static_cast<uint8_t>((upper & 0x3u) | ((upper >> 4) & 0x7Cu));
|
||||
switch (specialOp)
|
||||
{
|
||||
// Upper special ops that write a VF register use FT as destination.
|
||||
case 0x10: // ITOF0
|
||||
case 0x11: // ITOF4
|
||||
case 0x12: // ITOF12
|
||||
case 0x13: // ITOF15
|
||||
case 0x14: // FTOI0
|
||||
case 0x15: // FTOI4
|
||||
case 0x16: // FTOI12
|
||||
case 0x17: // FTOI15
|
||||
case 0x1D: // ABS
|
||||
return ft;
|
||||
default:
|
||||
return 0u; // ACC/NOP/CLIP/etc.
|
||||
}
|
||||
}
|
||||
|
||||
return 0u;
|
||||
}
|
||||
|
||||
static inline void vuSetRegBit(uint32_t &mask, uint8_t reg)
|
||||
{
|
||||
if (reg != 0u && reg < 32u)
|
||||
mask |= (1u << reg);
|
||||
}
|
||||
|
||||
static inline void vuLowerVfReadWriteMasks(uint32_t lower, uint32_t &readMask, uint32_t &writeMask)
|
||||
{
|
||||
readMask = 0u;
|
||||
writeMask = 0u;
|
||||
|
||||
if (lower == 0u || lower == 0x8000033Cu)
|
||||
return;
|
||||
|
||||
const uint8_t opHi = static_cast<uint8_t>((lower >> 25) & 0x7Fu);
|
||||
const uint8_t it = LIT(lower);
|
||||
const uint8_t is = LIS(lower);
|
||||
|
||||
if ((lower & 0x80000000u) != 0u)
|
||||
{
|
||||
const uint8_t funct = lower & 0x3Fu;
|
||||
if (funct >= 0x3Cu && funct <= 0x3Fu)
|
||||
{
|
||||
const uint8_t specialOp = static_cast<uint8_t>((lower & 0x3u) | ((lower >> 4) & 0x7Cu));
|
||||
switch (specialOp)
|
||||
{
|
||||
case 0x30: // MOVE
|
||||
case 0x31: // MR32
|
||||
vuSetRegBit(readMask, is);
|
||||
vuSetRegBit(writeMask, it);
|
||||
return;
|
||||
case 0x34: // LQI
|
||||
case 0x36: // LQD
|
||||
vuSetRegBit(writeMask, it);
|
||||
return;
|
||||
case 0x35: // SQI
|
||||
case 0x37: // SQD
|
||||
vuSetRegBit(readMask, is);
|
||||
return;
|
||||
case 0x38: // DIV
|
||||
case 0x3A: // RSQRT
|
||||
vuSetRegBit(readMask, is);
|
||||
vuSetRegBit(readMask, it);
|
||||
return;
|
||||
case 0x39: // SQRT
|
||||
vuSetRegBit(readMask, it);
|
||||
return;
|
||||
case 0x3C: // MTIR
|
||||
case 0x3E: // ILWR source base is integer, but field source is VF for MTIR only.
|
||||
if (specialOp == 0x3C)
|
||||
vuSetRegBit(readMask, is);
|
||||
return;
|
||||
case 0x3D: // MFIR
|
||||
case 0x64: // MFP
|
||||
vuSetRegBit(writeMask, it);
|
||||
return;
|
||||
default:
|
||||
return;
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
switch (opHi)
|
||||
{
|
||||
case 0x00: // LQ
|
||||
vuSetRegBit(writeMask, it);
|
||||
return;
|
||||
case 0x01: // SQ
|
||||
vuSetRegBit(readMask, is);
|
||||
return;
|
||||
default:
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
static inline bool vuLowerShouldRunBeforeUpper(uint32_t upper, uint32_t lower)
|
||||
{
|
||||
const uint8_t upperWrite = vuUpperVfWriteReg(upper);
|
||||
if (upperWrite == 0u)
|
||||
return false;
|
||||
|
||||
uint32_t lowerReads = 0u;
|
||||
uint32_t lowerWrites = 0u;
|
||||
vuLowerVfReadWriteMasks(lower, lowerReads, lowerWrites);
|
||||
|
||||
const uint32_t upperBit = (1u << upperWrite);
|
||||
return ((lowerReads | lowerWrites) & upperBit) != 0u;
|
||||
}
|
||||
|
||||
#endif
|
||||
#endif
|
||||
|
||||
@@ -7,7 +7,64 @@
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <limits>
|
||||
#include <vector>
|
||||
|
||||
namespace
|
||||
{
|
||||
float vuEatan(float value)
|
||||
{
|
||||
constexpr float coefficients[] = {
|
||||
0.999999344348907f,
|
||||
-0.333298563957214f,
|
||||
0.199465364217758f,
|
||||
-0.13085337519646f,
|
||||
0.096420042216778f,
|
||||
-0.055909886956215f,
|
||||
0.021861229091883f,
|
||||
-0.004054057877511f};
|
||||
constexpr float quarterPi = 0.785398185253143f;
|
||||
|
||||
const float squared = value * value;
|
||||
float polynomial = coefficients[7];
|
||||
for (int index = 6; index >= 0; --index)
|
||||
polynomial = coefficients[index] + squared * polynomial;
|
||||
return quarterPi + value * polynomial;
|
||||
}
|
||||
|
||||
float vuEsin(float value)
|
||||
{
|
||||
constexpr float coefficients[] = {
|
||||
1.0f,
|
||||
-0.166666567325592f,
|
||||
0.008333025500178f,
|
||||
-0.000198074136279f,
|
||||
0.000002601886990f};
|
||||
|
||||
const float squared = value * value;
|
||||
float polynomial = coefficients[4];
|
||||
for (int index = 3; index >= 0; --index)
|
||||
polynomial = coefficients[index] + squared * polynomial;
|
||||
return value * polynomial;
|
||||
}
|
||||
|
||||
float vuEexp(float value)
|
||||
{
|
||||
constexpr float coefficients[] = {
|
||||
0.249998688697815f,
|
||||
0.031257584691048f,
|
||||
0.002591371303424f,
|
||||
0.000171562001924f,
|
||||
0.000005430199963f,
|
||||
0.000000690600018f};
|
||||
|
||||
float polynomial = coefficients[5];
|
||||
for (int index = 4; index >= 0; --index)
|
||||
polynomial = coefficients[index] + value * polynomial;
|
||||
polynomial = 1.0f + value * polynomial;
|
||||
polynomial *= polynomial;
|
||||
polynomial *= polynomial;
|
||||
return polynomial != 0.0f ? 1.0f / polynomial : std::numeric_limits<float>::max();
|
||||
}
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Lower instructions
|
||||
@@ -19,14 +76,15 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
return;
|
||||
|
||||
uint8_t opHi = (instr >> 25) & 0x7F;
|
||||
const uint32_t pcMask = microAddressMask();
|
||||
|
||||
// The lower instruction encoding uses bits 31:25 for the primary opcode
|
||||
switch (opHi)
|
||||
{
|
||||
case 0x00: // LQ (Load Quadword from VU data memory)
|
||||
{
|
||||
uint8_t it = FT(instr); // VF destination
|
||||
uint8_t is = VIS(instr); // VI base
|
||||
uint8_t it = FT(instr); // VF destination
|
||||
uint8_t is = VIS(instr); // VI base
|
||||
uint8_t dest = (instr >> 21) & 0xF;
|
||||
int16_t imm = IMM11(instr);
|
||||
uint32_t addr = ((uint32_t)(int32_t)(m_state.vi[is] + imm)) * 16u;
|
||||
@@ -41,32 +99,24 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
}
|
||||
case 0x01: // SQ (Store Quadword to VU data memory)
|
||||
{
|
||||
uint8_t is = FS(instr); // VF source
|
||||
uint8_t it = VIT(instr); // VI base
|
||||
uint8_t is = FS(instr); // VF source
|
||||
uint8_t it = VIT(instr); // VI base
|
||||
uint8_t dest = (instr >> 21) & 0xF;
|
||||
int16_t imm = IMM11(instr);
|
||||
uint32_t addr = ((uint32_t)(int32_t)(m_state.vi[it] + imm)) * 16u;
|
||||
addr &= (dataSize - 1);
|
||||
if (addr + 16 <= dataSize)
|
||||
{
|
||||
float tmp[4];
|
||||
std::memcpy(tmp, vuData + addr, 16);
|
||||
if (dest & 0x8)
|
||||
tmp[0] = m_state.vf[is][0];
|
||||
if (dest & 0x4)
|
||||
tmp[1] = m_state.vf[is][1];
|
||||
if (dest & 0x2)
|
||||
tmp[2] = m_state.vf[is][2];
|
||||
if (dest & 0x1)
|
||||
tmp[3] = m_state.vf[is][3];
|
||||
std::memcpy(vuData + addr, tmp, 16);
|
||||
uint32_t words[4]{};
|
||||
std::memcpy(words, m_state.vf[is], sizeof(words));
|
||||
queueStore(addr, words, dest);
|
||||
}
|
||||
return;
|
||||
}
|
||||
case 0x04: // ILW (Integer Load Word from VU data memory)
|
||||
{
|
||||
uint8_t it = VIT(instr); // VI destination
|
||||
uint8_t is = VIS(instr); // VI base
|
||||
uint8_t it = VIT(instr); // VI destination
|
||||
uint8_t is = VIS(instr); // VI base
|
||||
uint8_t dest = (instr >> 21) & 0xF;
|
||||
int16_t imm = IMM11(instr);
|
||||
uint32_t addr = ((uint32_t)(int32_t)(m_state.vi[is] + imm)) * 16u;
|
||||
@@ -91,23 +141,17 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
}
|
||||
case 0x05: // ISW (Integer Store Word to VU data memory)
|
||||
{
|
||||
uint8_t it = VIT(instr); // VI source
|
||||
uint8_t is = VIS(instr); // VI base
|
||||
uint8_t it = VIT(instr); // VI source
|
||||
uint8_t is = VIS(instr); // VI base
|
||||
uint8_t dest = (instr >> 21) & 0xF;
|
||||
int16_t imm = IMM11(instr);
|
||||
uint32_t addr = ((uint32_t)(int32_t)(m_state.vi[is] + imm)) * 16u;
|
||||
addr &= (dataSize - 1);
|
||||
if (addr + 16 <= dataSize)
|
||||
{
|
||||
uint32_t val = (uint32_t)(uint16_t)(m_state.vi[it] & 0xFFFF);
|
||||
if (dest & 0x8)
|
||||
std::memcpy(vuData + addr + 0, &val, 4);
|
||||
if (dest & 0x4)
|
||||
std::memcpy(vuData + addr + 4, &val, 4);
|
||||
if (dest & 0x2)
|
||||
std::memcpy(vuData + addr + 8, &val, 4);
|
||||
if (dest & 0x1)
|
||||
std::memcpy(vuData + addr + 12, &val, 4);
|
||||
const uint32_t val = static_cast<uint32_t>(static_cast<uint16_t>(m_state.vi[it] & 0xFFFF));
|
||||
const uint32_t words[4] = {val, val, val, val};
|
||||
queueStore(addr, words, dest);
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -138,7 +182,7 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
}
|
||||
case 0x11: // FCSET
|
||||
{
|
||||
m_state.clip = instr & 0xFFFFFF;
|
||||
queueFcset(instr & 0xFFFFFFu);
|
||||
return;
|
||||
}
|
||||
case 0x12: // FCAND
|
||||
@@ -157,39 +201,35 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
}
|
||||
case 0x14: // FSEQ
|
||||
{
|
||||
uint16_t imm12 = instr & 0xFFF;
|
||||
if (1 != 0)
|
||||
m_state.vi[1] = ((m_state.status & 0xFFF) == imm12) ? 1 : 0;
|
||||
const uint8_t it = VIT(instr);
|
||||
const uint16_t imm12 = static_cast<uint16_t>((((instr >> 21) & 0x1u) << 11) | (instr & 0x7FFu));
|
||||
if (it != 0)
|
||||
m_state.vi[it] = ((m_state.status & 0xFFFu) == imm12) ? 1 : 0;
|
||||
return;
|
||||
}
|
||||
case 0x15: // FSSET
|
||||
{
|
||||
m_state.status = (instr >> 6) & 0xFC0;
|
||||
const uint16_t imm12 = static_cast<uint16_t>((((instr >> 21) & 0x1u) << 11) | (instr & 0x7FFu));
|
||||
queueFsset(imm12);
|
||||
return;
|
||||
}
|
||||
case 0x16: // FSAND
|
||||
{
|
||||
uint16_t imm12 = instr & 0xFFF;
|
||||
if (1 != 0)
|
||||
m_state.vi[1] = (int32_t)(m_state.status & imm12);
|
||||
const uint8_t it = VIT(instr);
|
||||
const uint16_t imm12 = static_cast<uint16_t>((((instr >> 21) & 0x1u) << 11) | (instr & 0x7FFu));
|
||||
if (it != 0)
|
||||
m_state.vi[it] = static_cast<int32_t>((m_state.status & 0xFFFu) & imm12);
|
||||
return;
|
||||
}
|
||||
case 0x17: // FSOR
|
||||
{
|
||||
uint16_t imm12 = instr & 0xFFF;
|
||||
if (1 != 0)
|
||||
m_state.vi[1] = ((m_state.status | imm12) == 0xFFF) ? 1 : 0;
|
||||
return;
|
||||
}
|
||||
case 0x18: // FMAND
|
||||
{
|
||||
uint8_t it = VIT(instr);
|
||||
uint8_t is = VIS(instr);
|
||||
const uint8_t it = VIT(instr);
|
||||
const uint16_t imm12 = static_cast<uint16_t>((((instr >> 21) & 0x1u) << 11) | (instr & 0x7FFu));
|
||||
if (it != 0)
|
||||
m_state.vi[it] = (int32_t)(m_state.mac & (uint32_t)(uint16_t)m_state.vi[is]);
|
||||
m_state.vi[it] = static_cast<int32_t>((m_state.status & 0xFFFu) | imm12);
|
||||
return;
|
||||
}
|
||||
case 0x1A: // FMEQ
|
||||
case 0x18: // FMEQ
|
||||
{
|
||||
uint8_t it = VIT(instr);
|
||||
uint8_t is = VIS(instr);
|
||||
@@ -197,7 +237,15 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
m_state.vi[it] = ((m_state.mac & 0xFFFF) == (uint32_t)(uint16_t)m_state.vi[is]) ? 1 : 0;
|
||||
return;
|
||||
}
|
||||
case 0x1C: // FMOR
|
||||
case 0x1A: // FMAND
|
||||
{
|
||||
uint8_t it = VIT(instr);
|
||||
uint8_t is = VIS(instr);
|
||||
if (it != 0)
|
||||
m_state.vi[it] = (int32_t)(m_state.mac & (uint32_t)(uint16_t)m_state.vi[is]);
|
||||
return;
|
||||
}
|
||||
case 0x1B: // FMOR
|
||||
{
|
||||
uint8_t it = VIT(instr);
|
||||
uint8_t is = VIS(instr);
|
||||
@@ -205,10 +253,17 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
m_state.vi[it] = (int32_t)(m_state.mac | (uint32_t)(uint16_t)m_state.vi[is]);
|
||||
return;
|
||||
}
|
||||
case 0x1C: // FCGET
|
||||
{
|
||||
const uint8_t it = VIT(instr);
|
||||
if (it != 0)
|
||||
m_state.vi[it] = static_cast<int32_t>(m_state.clip & 0x0FFFu);
|
||||
return;
|
||||
}
|
||||
case 0x20: // B (unconditional branch)
|
||||
{
|
||||
int16_t imm = IMM11(instr);
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & 0x3FFF;
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & pcMask;
|
||||
m_state.branchPending = true;
|
||||
m_state.branchTarget = target;
|
||||
m_state.branchDelay = 1;
|
||||
@@ -218,7 +273,7 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
{
|
||||
uint8_t it = VIT(instr);
|
||||
int16_t imm = IMM11(instr);
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & 0x3FFF;
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & pcMask;
|
||||
if (it != 0)
|
||||
m_state.vi[it] = (int32_t)((m_state.pc + 16) / 8);
|
||||
m_state.branchPending = true;
|
||||
@@ -229,7 +284,7 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
case 0x24: // JR
|
||||
{
|
||||
uint8_t is = VIS(instr);
|
||||
uint32_t target = ((uint32_t)(uint16_t)m_state.vi[is] * 8u) & 0x3FFF;
|
||||
uint32_t target = ((uint32_t)(uint16_t)readBranchVi(is) * 8u) & pcMask;
|
||||
m_state.branchPending = true;
|
||||
m_state.branchTarget = target;
|
||||
m_state.branchDelay = 1;
|
||||
@@ -239,7 +294,7 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
{
|
||||
uint8_t it = VIT(instr);
|
||||
uint8_t is = VIS(instr);
|
||||
uint32_t target = ((uint32_t)(uint16_t)m_state.vi[is] * 8u) & 0x3FFF;
|
||||
uint32_t target = ((uint32_t)(uint16_t)readBranchVi(is) * 8u) & pcMask;
|
||||
if (it != 0)
|
||||
m_state.vi[it] = (int32_t)((m_state.pc + 16) / 8);
|
||||
m_state.branchPending = true;
|
||||
@@ -252,12 +307,12 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
uint8_t it = VIT(instr);
|
||||
uint8_t is = VIS(instr);
|
||||
int16_t imm = IMM11(instr);
|
||||
if ((int16_t)m_state.vi[is] == (int16_t)m_state.vi[it])
|
||||
if ((int16_t)readBranchVi(is) == (int16_t)readBranchVi(it))
|
||||
{
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & 0x3FFF;
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & pcMask;
|
||||
m_state.branchPending = true;
|
||||
m_state.branchTarget = target;
|
||||
m_state.branchDelay = 1;
|
||||
m_state.branchTarget = target;
|
||||
m_state.branchDelay = 1;
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -266,12 +321,12 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
uint8_t it = VIT(instr);
|
||||
uint8_t is = VIS(instr);
|
||||
int16_t imm = IMM11(instr);
|
||||
if ((int16_t)m_state.vi[is] != (int16_t)m_state.vi[it])
|
||||
if ((int16_t)readBranchVi(is) != (int16_t)readBranchVi(it))
|
||||
{
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & 0x3FFF;
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & pcMask;
|
||||
m_state.branchPending = true;
|
||||
m_state.branchTarget = target;
|
||||
m_state.branchDelay = 1;
|
||||
m_state.branchTarget = target;
|
||||
m_state.branchDelay = 1;
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -279,12 +334,12 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
{
|
||||
uint8_t is = VIS(instr);
|
||||
int16_t imm = IMM11(instr);
|
||||
if ((int16_t)m_state.vi[is] < 0)
|
||||
if ((int16_t)readBranchVi(is) < 0)
|
||||
{
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & 0x3FFF;
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & pcMask;
|
||||
m_state.branchPending = true;
|
||||
m_state.branchTarget = target;
|
||||
m_state.branchDelay = 1;
|
||||
m_state.branchTarget = target;
|
||||
m_state.branchDelay = 1;
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -292,12 +347,12 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
{
|
||||
uint8_t is = VIS(instr);
|
||||
int16_t imm = IMM11(instr);
|
||||
if ((int16_t)m_state.vi[is] > 0)
|
||||
if ((int16_t)readBranchVi(is) > 0)
|
||||
{
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & 0x3FFF;
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & pcMask;
|
||||
m_state.branchPending = true;
|
||||
m_state.branchTarget = target;
|
||||
m_state.branchDelay = 1;
|
||||
m_state.branchTarget = target;
|
||||
m_state.branchDelay = 1;
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -305,12 +360,12 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
{
|
||||
uint8_t is = VIS(instr);
|
||||
int16_t imm = IMM11(instr);
|
||||
if ((int16_t)m_state.vi[is] <= 0)
|
||||
if ((int16_t)readBranchVi(is) <= 0)
|
||||
{
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & 0x3FFF;
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & pcMask;
|
||||
m_state.branchPending = true;
|
||||
m_state.branchTarget = target;
|
||||
m_state.branchDelay = 1;
|
||||
m_state.branchTarget = target;
|
||||
m_state.branchDelay = 1;
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -318,12 +373,12 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
{
|
||||
uint8_t is = VIS(instr);
|
||||
int16_t imm = IMM11(instr);
|
||||
if ((int16_t)m_state.vi[is] >= 0)
|
||||
if ((int16_t)readBranchVi(is) >= 0)
|
||||
{
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & 0x3FFF;
|
||||
uint32_t target = (m_state.pc + 8 + imm * 8) & pcMask;
|
||||
m_state.branchPending = true;
|
||||
m_state.branchTarget = target;
|
||||
m_state.branchDelay = 1;
|
||||
m_state.branchTarget = target;
|
||||
m_state.branchDelay = 1;
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -338,95 +393,6 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
const uint8_t viD = VID(instr);
|
||||
const uint8_t dest = (instr >> 21) & 0xF;
|
||||
|
||||
auto doXgkick = [&]()
|
||||
{
|
||||
if (!vuData || dataSize < 16u)
|
||||
return;
|
||||
|
||||
auto wrapOffset = [&](uint32_t off) -> uint32_t
|
||||
{
|
||||
return off % dataSize;
|
||||
};
|
||||
|
||||
auto read64Wrap = [&](uint32_t off) -> uint64_t
|
||||
{
|
||||
uint8_t bytes[8];
|
||||
for (uint32_t i = 0; i < 8u; ++i)
|
||||
{
|
||||
bytes[i] = vuData[wrapOffset(off + i)];
|
||||
}
|
||||
uint64_t value = 0;
|
||||
std::memcpy(&value, bytes, sizeof(value));
|
||||
return value;
|
||||
};
|
||||
|
||||
uint32_t addr = ((uint32_t)(uint16_t)m_state.vi[viS]) * 16u;
|
||||
addr = wrapOffset(addr);
|
||||
uint32_t pktOff = addr;
|
||||
uint32_t totalBytes = 0u;
|
||||
bool done = false;
|
||||
|
||||
for (int safety = 0; safety < 256 && !done; ++safety)
|
||||
{
|
||||
uint64_t tagLo = read64Wrap(pktOff);
|
||||
uint32_t nloop = (uint32_t)(tagLo & 0x7FFFu);
|
||||
uint8_t flg = (uint8_t)((tagLo >> 58) & 0x3u);
|
||||
uint32_t nreg = (uint32_t)((tagLo >> 60) & 0xFu);
|
||||
if (nreg == 0u)
|
||||
nreg = 16u;
|
||||
bool eop = ((tagLo >> 15) & 0x1ull) != 0ull;
|
||||
|
||||
uint32_t pktSize = 16u;
|
||||
if (flg == 0u)
|
||||
{
|
||||
pktSize += nloop * nreg * 16u;
|
||||
}
|
||||
else if (flg == 1u)
|
||||
{
|
||||
uint32_t regs = nloop * nreg;
|
||||
pktSize += regs * 8u;
|
||||
if ((regs & 1u) != 0u)
|
||||
pktSize += 8u;
|
||||
}
|
||||
else if (flg == 2u)
|
||||
{
|
||||
pktSize += nloop * 16u;
|
||||
}
|
||||
|
||||
if (pktSize == 0u)
|
||||
break;
|
||||
|
||||
totalBytes += pktSize;
|
||||
pktOff = wrapOffset(pktOff + pktSize);
|
||||
if (eop)
|
||||
done = true;
|
||||
}
|
||||
|
||||
if (totalBytes == 0u)
|
||||
return;
|
||||
|
||||
if (addr + totalBytes <= dataSize)
|
||||
{
|
||||
if (memory)
|
||||
memory->submitGifPacket(GifPathId::Path1, vuData + addr, totalBytes);
|
||||
else
|
||||
gs.processGIFPacket(vuData + addr, totalBytes);
|
||||
}
|
||||
else
|
||||
{
|
||||
std::vector<uint8_t> wrappedPacket(totalBytes);
|
||||
for (uint32_t i = 0; i < totalBytes; ++i)
|
||||
{
|
||||
wrappedPacket[i] = vuData[wrapOffset(addr + i)];
|
||||
}
|
||||
|
||||
if (memory)
|
||||
memory->submitGifPacket(GifPathId::Path1, wrappedPacket.data(), totalBytes);
|
||||
else
|
||||
gs.processGIFPacket(wrappedPacket.data(), totalBytes);
|
||||
}
|
||||
};
|
||||
|
||||
switch (funct)
|
||||
{
|
||||
case 0x30: // IADD
|
||||
@@ -494,17 +460,9 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
addr &= (dataSize - 1);
|
||||
if (addr + 16 <= dataSize)
|
||||
{
|
||||
float tmp[4];
|
||||
std::memcpy(tmp, vuData + addr, 16);
|
||||
if (dest & 0x8)
|
||||
tmp[0] = m_state.vf[vfS][0];
|
||||
if (dest & 0x4)
|
||||
tmp[1] = m_state.vf[vfS][1];
|
||||
if (dest & 0x2)
|
||||
tmp[2] = m_state.vf[vfS][2];
|
||||
if (dest & 0x1)
|
||||
tmp[3] = m_state.vf[vfS][3];
|
||||
std::memcpy(vuData + addr, tmp, 16);
|
||||
uint32_t words[4]{};
|
||||
std::memcpy(words, m_state.vf[vfS], sizeof(words));
|
||||
queueStore(addr, words, dest);
|
||||
}
|
||||
if (viT != 0)
|
||||
m_state.vi[viT] = (int16_t)(m_state.vi[viT] + 1);
|
||||
@@ -532,17 +490,9 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
addr &= (dataSize - 1);
|
||||
if (addr + 16 <= dataSize)
|
||||
{
|
||||
float tmp[4];
|
||||
std::memcpy(tmp, vuData + addr, 16);
|
||||
if (dest & 0x8)
|
||||
tmp[0] = m_state.vf[vfS][0];
|
||||
if (dest & 0x4)
|
||||
tmp[1] = m_state.vf[vfS][1];
|
||||
if (dest & 0x2)
|
||||
tmp[2] = m_state.vf[vfS][2];
|
||||
if (dest & 0x1)
|
||||
tmp[3] = m_state.vf[vfS][3];
|
||||
std::memcpy(vuData + addr, tmp, 16);
|
||||
uint32_t words[4]{};
|
||||
std::memcpy(words, m_state.vf[vfS], sizeof(words));
|
||||
queueStore(addr, words, dest);
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -550,46 +500,64 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
{
|
||||
int fsf = (instr >> 21) & 0x3;
|
||||
int ftf = (instr >> 23) & 0x3;
|
||||
float num = m_state.vf[vfS][fsf];
|
||||
float den = m_state.vf[vfT][ftf];
|
||||
if (den != 0.0f)
|
||||
m_state.q = num / den;
|
||||
const float num = normalizeOperand(m_state.vf[vfS][fsf]);
|
||||
const float den = normalizeOperand(m_state.vf[vfT][ftf]);
|
||||
uint32_t statusDi = 0u;
|
||||
float result = 0.0f;
|
||||
if (den == 0.0f)
|
||||
{
|
||||
statusDi = num == 0.0f ? 0x10u : 0x20u;
|
||||
result = std::signbit(num) != std::signbit(den)
|
||||
? -std::numeric_limits<float>::max()
|
||||
: std::numeric_limits<float>::max();
|
||||
}
|
||||
else
|
||||
m_state.q = (num >= 0.0f) ? std::numeric_limits<float>::max() : -std::numeric_limits<float>::max();
|
||||
{
|
||||
result = num / den;
|
||||
}
|
||||
uint32_t ignoredFlags = 0u;
|
||||
result = normalizeResult(result, ignoredFlags);
|
||||
queueQ(result, 7u, statusDi);
|
||||
return;
|
||||
}
|
||||
case 0x39: // SQRT
|
||||
{
|
||||
int ftf = (instr >> 23) & 0x3;
|
||||
float val = m_state.vf[vfT][ftf];
|
||||
m_state.q = std::sqrt(std::fabs(val));
|
||||
const float val = normalizeOperand(m_state.vf[vfT][ftf]);
|
||||
queueQ(std::sqrt(std::fabs(val)), 7u,
|
||||
val < 0.0f ? 0x10u : 0u);
|
||||
return;
|
||||
}
|
||||
case 0x3A: // RSQRT
|
||||
{
|
||||
int fsf = (instr >> 21) & 0x3;
|
||||
int ftf = (instr >> 23) & 0x3;
|
||||
float num = m_state.vf[vfS][fsf];
|
||||
float den = std::sqrt(std::fabs(m_state.vf[vfT][ftf]));
|
||||
const float num = normalizeOperand(m_state.vf[vfS][fsf]);
|
||||
const float radicand = normalizeOperand(m_state.vf[vfT][ftf]);
|
||||
const float den = std::sqrt(std::fabs(radicand));
|
||||
uint32_t statusDi = radicand < 0.0f ? 0x10u : 0u;
|
||||
float result = 0.0f;
|
||||
if (den != 0.0f)
|
||||
m_state.q = num / den;
|
||||
result = num / den;
|
||||
else
|
||||
m_state.q = std::numeric_limits<float>::max();
|
||||
{
|
||||
statusDi = num == 0.0f ? 0x10u : 0x20u;
|
||||
result = std::signbit(num)
|
||||
? -std::numeric_limits<float>::max()
|
||||
: std::numeric_limits<float>::max();
|
||||
}
|
||||
uint32_t ignoredFlags = 0u;
|
||||
result = normalizeResult(result, ignoredFlags);
|
||||
queueQ(result, 13u, statusDi);
|
||||
return;
|
||||
}
|
||||
case 0x3B: // WAITQ
|
||||
return;
|
||||
case 0x3C: // MTIR (Move To Integer Register)
|
||||
{
|
||||
int comp = 0;
|
||||
if (dest & 0x8)
|
||||
comp = 0;
|
||||
else if (dest & 0x4)
|
||||
comp = 1;
|
||||
else if (dest & 0x2)
|
||||
comp = 2;
|
||||
else
|
||||
comp = 3;
|
||||
// MTIR encodes a two-bit fsf component selector in bits
|
||||
// 22:21. It is not a four-bit destination mask.
|
||||
const uint32_t comp = (instr >> 21) & 0x3u;
|
||||
uint32_t fval;
|
||||
std::memcpy(&fval, &m_state.vf[vfS][comp], 4);
|
||||
if (viT != 0)
|
||||
@@ -635,26 +603,49 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
addr &= (dataSize - 1);
|
||||
if (addr + 16 <= dataSize)
|
||||
{
|
||||
uint32_t val = (uint32_t)(uint16_t)(m_state.vi[viT] & 0xFFFF);
|
||||
if (dest & 0x8)
|
||||
std::memcpy(vuData + addr + 0, &val, 4);
|
||||
if (dest & 0x4)
|
||||
std::memcpy(vuData + addr + 4, &val, 4);
|
||||
if (dest & 0x2)
|
||||
std::memcpy(vuData + addr + 8, &val, 4);
|
||||
if (dest & 0x1)
|
||||
std::memcpy(vuData + addr + 12, &val, 4);
|
||||
const uint32_t val =
|
||||
static_cast<uint32_t>(static_cast<uint16_t>(m_state.vi[viT] & 0xFFFF));
|
||||
const uint32_t words[4] = {val, val, val, val};
|
||||
queueStore(addr, words, dest);
|
||||
}
|
||||
return;
|
||||
}
|
||||
case 0x40: // RNEXT
|
||||
{
|
||||
const uint32_t x = (m_state.r >> 4) & 1u;
|
||||
const uint32_t y = (m_state.r >> 22) & 1u;
|
||||
m_state.r = ((m_state.r << 1) ^ x ^ y) & 0x007FFFFFu;
|
||||
m_state.r |= 0x3F800000u;
|
||||
float value = 0.0f;
|
||||
std::memcpy(&value, &m_state.r, sizeof(value));
|
||||
const float result[4] = {value, value, value, value};
|
||||
applyDest(m_state.vf[vfT], result, dest);
|
||||
return;
|
||||
}
|
||||
case 0x41: // RGET
|
||||
{
|
||||
float value = 0.0f;
|
||||
std::memcpy(&value, &m_state.r, sizeof(value));
|
||||
const float result[4] = {value, value, value, value};
|
||||
applyDest(m_state.vf[vfT], result, dest);
|
||||
return;
|
||||
}
|
||||
case 0x42: // RINIT
|
||||
{
|
||||
const uint32_t component = (instr >> 21) & 3u;
|
||||
uint32_t bits = 0u;
|
||||
std::memcpy(&bits, &m_state.vf[vfS][component], sizeof(bits));
|
||||
m_state.r = 0x3F800000u | (bits & 0x007FFFFFu);
|
||||
return;
|
||||
}
|
||||
case 0x43: // RXOR
|
||||
{
|
||||
const uint32_t component = (instr >> 21) & 3u;
|
||||
uint32_t bits = 0u;
|
||||
std::memcpy(&bits, &m_state.vf[vfS][component], sizeof(bits));
|
||||
m_state.r = 0x3F800000u | ((m_state.r ^ bits) & 0x007FFFFFu);
|
||||
return;
|
||||
}
|
||||
case 0x64: // MFP (Move From P register)
|
||||
{
|
||||
float result[4] = {m_state.p, m_state.p, m_state.p, m_state.p};
|
||||
@@ -674,45 +665,125 @@ void VU1Interpreter::execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSiz
|
||||
return;
|
||||
}
|
||||
case 0x6C: // XGKICK - send GIF packet from VU1 data memory
|
||||
doXgkick();
|
||||
startXgkick(static_cast<uint32_t>(static_cast<uint16_t>(m_state.vi[viS])));
|
||||
return;
|
||||
case 0x70: // ESADD
|
||||
{
|
||||
const float x = normalizeOperand(m_state.vf[vfS][0]);
|
||||
const float y = normalizeOperand(m_state.vf[vfS][1]);
|
||||
const float z = normalizeOperand(m_state.vf[vfS][2]);
|
||||
queueP(x * x + y * y + z * z, 11u);
|
||||
return;
|
||||
}
|
||||
case 0x71: // ERSADD
|
||||
{
|
||||
const float x = normalizeOperand(m_state.vf[vfS][0]);
|
||||
const float y = normalizeOperand(m_state.vf[vfS][1]);
|
||||
const float z = normalizeOperand(m_state.vf[vfS][2]);
|
||||
const float sum = x * x + y * y + z * z;
|
||||
queueP(sum != 0.0f ? 1.0f / sum : sum, 18u);
|
||||
return;
|
||||
}
|
||||
case 0x72: // ELENG
|
||||
{
|
||||
float s = m_state.vf[vfS][0] * m_state.vf[vfS][0] + m_state.vf[vfS][1] * m_state.vf[vfS][1] + m_state.vf[vfS][2] * m_state.vf[vfS][2];
|
||||
m_state.p = std::sqrt(s);
|
||||
const float x = normalizeOperand(m_state.vf[vfS][0]);
|
||||
const float y = normalizeOperand(m_state.vf[vfS][1]);
|
||||
const float z = normalizeOperand(m_state.vf[vfS][2]);
|
||||
queueP(std::sqrt(x * x + y * y + z * z), 18u);
|
||||
return;
|
||||
}
|
||||
case 0x73: // ERLENG
|
||||
{
|
||||
float s = m_state.vf[vfS][0] * m_state.vf[vfS][0] + m_state.vf[vfS][1] * m_state.vf[vfS][1] + m_state.vf[vfS][2] * m_state.vf[vfS][2];
|
||||
float len = std::sqrt(s);
|
||||
m_state.p = (len != 0.0f) ? (1.0f / len) : std::numeric_limits<float>::max();
|
||||
const float x = normalizeOperand(m_state.vf[vfS][0]);
|
||||
const float y = normalizeOperand(m_state.vf[vfS][1]);
|
||||
const float z = normalizeOperand(m_state.vf[vfS][2]);
|
||||
const float len = std::sqrt(x * x + y * y + z * z);
|
||||
queueP(len != 0.0f ? 1.0f / len : len, 24u);
|
||||
return;
|
||||
}
|
||||
case 0x74: // EATANxy
|
||||
{
|
||||
const float x = normalizeOperand(m_state.vf[vfS][0]);
|
||||
const float y = normalizeOperand(m_state.vf[vfS][1]);
|
||||
queueP(x != 0.0f ? vuEatan(y / x) : 0.0f, 54u);
|
||||
return;
|
||||
}
|
||||
case 0x75: // EATANxz
|
||||
{
|
||||
const float x = normalizeOperand(m_state.vf[vfS][0]);
|
||||
const float z = normalizeOperand(m_state.vf[vfS][2]);
|
||||
queueP(x != 0.0f ? vuEatan(z / x) : 0.0f, 54u);
|
||||
return;
|
||||
}
|
||||
case 0x76: // ESUM
|
||||
{
|
||||
float sum = 0.0f;
|
||||
for (uint32_t component = 0; component < 4u; ++component)
|
||||
sum += normalizeOperand(m_state.vf[vfS][component]);
|
||||
queueP(sum, 12u);
|
||||
return;
|
||||
}
|
||||
case 0x77: // ERSQRT
|
||||
{
|
||||
const uint32_t component = (instr >> 21) & 3u;
|
||||
const float value = normalizeOperand(m_state.vf[vfS][component]);
|
||||
float result = value;
|
||||
if (result >= 0.0f)
|
||||
{
|
||||
result = std::sqrt(result);
|
||||
if (result != 0.0f)
|
||||
result = 1.0f / result;
|
||||
}
|
||||
queueP(result, 18u);
|
||||
return;
|
||||
}
|
||||
case 0x78: // ESQRT
|
||||
{
|
||||
const uint32_t component = (instr >> 21) & 3u;
|
||||
const float value = normalizeOperand(m_state.vf[vfS][component]);
|
||||
queueP(value >= 0.0f ? std::sqrt(value) : value, 12u);
|
||||
return;
|
||||
}
|
||||
case 0x79: // ESIN
|
||||
{
|
||||
const uint32_t component = (instr >> 21) & 3u;
|
||||
const float value = normalizeOperand(m_state.vf[vfS][component]);
|
||||
queueP(vuEsin(value), 29u);
|
||||
return;
|
||||
}
|
||||
case 0x7A: // ERCPR
|
||||
{
|
||||
int fsf = (instr >> 21) & 0x3;
|
||||
float val = m_state.vf[vfS][fsf];
|
||||
m_state.p = (val != 0.0f) ? (1.0f / val) : std::numeric_limits<float>::max();
|
||||
const uint32_t component = (instr >> 21) & 3u;
|
||||
const float value = normalizeOperand(m_state.vf[vfS][component]);
|
||||
queueP(value != 0.0f ? 1.0f / value : value, 12u);
|
||||
return;
|
||||
}
|
||||
case 0x7B: // WAITP
|
||||
return;
|
||||
case 0x7D: // EATAN / EATANxy / EATANxz placeholder
|
||||
case 0x7C: // EATAN
|
||||
{
|
||||
const uint32_t component = (instr >> 21) & 3u;
|
||||
queueP(vuEatan(normalizeOperand(m_state.vf[vfS][component])), 54u);
|
||||
return;
|
||||
}
|
||||
case 0x7D: // EEXP
|
||||
{
|
||||
const uint32_t component = (instr >> 21) & 3u;
|
||||
queueP(vuEexp(normalizeOperand(m_state.vf[vfS][component])), 44u);
|
||||
return;
|
||||
}
|
||||
default:
|
||||
reportReservedInstruction(false, instr);
|
||||
return;
|
||||
}
|
||||
}
|
||||
default:
|
||||
reportReservedInstruction(false, instr);
|
||||
return;
|
||||
}
|
||||
}
|
||||
default:
|
||||
reportReservedInstruction(false, instr);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3,12 +3,27 @@
|
||||
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <limits>
|
||||
|
||||
namespace
|
||||
{
|
||||
int32_t vuFloatToInt(float value, float scale)
|
||||
{
|
||||
const double scaled = static_cast<double>(value) * static_cast<double>(scale);
|
||||
if (scaled >= static_cast<double>(std::numeric_limits<int32_t>::max()))
|
||||
return std::numeric_limits<int32_t>::max();
|
||||
if (scaled <= static_cast<double>(std::numeric_limits<int32_t>::min()))
|
||||
return std::numeric_limits<int32_t>::min();
|
||||
return static_cast<int32_t>(scaled);
|
||||
}
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Upper instructions (FMAC pipeline)
|
||||
// ============================================================================
|
||||
void VU1Interpreter::execUpper(uint32_t instr)
|
||||
{
|
||||
m_currentUpperInstruction = instr;
|
||||
uint8_t dest = DEST(instr);
|
||||
uint8_t ft = FT(instr);
|
||||
uint8_t fs = FS(instr);
|
||||
@@ -16,8 +31,20 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
uint8_t op = instr & 0x3F;
|
||||
|
||||
float *vd = m_state.vf[fd];
|
||||
const float *vs = m_state.vf[fs];
|
||||
const float *vt = m_state.vf[ft];
|
||||
float normalizedVs[4];
|
||||
float normalizedVt[4];
|
||||
float normalizedAcc[4];
|
||||
for (uint32_t component = 0; component < 4u; ++component)
|
||||
{
|
||||
normalizedVs[component] = normalizeOperand(m_state.vf[fs][component]);
|
||||
normalizedVt[component] = normalizeOperand(m_state.vf[ft][component]);
|
||||
normalizedAcc[component] = normalizeOperand(m_state.acc[component]);
|
||||
}
|
||||
const float *vs = normalizedVs;
|
||||
const float *vt = normalizedVt;
|
||||
const float *acc = normalizedAcc;
|
||||
const float q = normalizeOperand(m_state.q);
|
||||
const float i = normalizeOperand(m_state.i);
|
||||
float result[4];
|
||||
|
||||
// Upper opcode decoding (bits 5:0 of upper word)
|
||||
@@ -31,7 +58,7 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
float bc = broadcast(vt, op & 3);
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] + bc;
|
||||
applyDest(vd, result, dest);
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
}
|
||||
case 0x04:
|
||||
@@ -42,7 +69,7 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
float bc = broadcast(vt, op & 3);
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] - bc;
|
||||
applyDest(vd, result, dest);
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
}
|
||||
case 0x08:
|
||||
@@ -52,8 +79,8 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
{
|
||||
float bc = broadcast(vt, op & 3);
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] + vs[c] * bc;
|
||||
applyDest(vd, result, dest);
|
||||
result[c] = acc[c] + vs[c] * bc;
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
}
|
||||
case 0x0C:
|
||||
@@ -63,8 +90,8 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
{
|
||||
float bc = broadcast(vt, op & 3);
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] - vs[c] * bc;
|
||||
applyDest(vd, result, dest);
|
||||
result[c] = acc[c] - vs[c] * bc;
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
}
|
||||
case 0x10:
|
||||
@@ -97,83 +124,83 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
float bc = broadcast(vt, op & 3);
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] * bc;
|
||||
applyDest(vd, result, dest);
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
}
|
||||
case 0x1C: // MULq
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] * m_state.q;
|
||||
applyDest(vd, result, dest);
|
||||
result[c] = vs[c] * q;
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x1D: // MAXi
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = (vs[c] > m_state.i) ? vs[c] : m_state.i;
|
||||
result[c] = (vs[c] > i) ? vs[c] : i;
|
||||
applyDest(vd, result, dest);
|
||||
return;
|
||||
case 0x1E: // MULi
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] * m_state.i;
|
||||
applyDest(vd, result, dest);
|
||||
result[c] = vs[c] * i;
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x1F: // MINIi
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = (vs[c] < m_state.i) ? vs[c] : m_state.i;
|
||||
result[c] = (vs[c] < i) ? vs[c] : i;
|
||||
applyDest(vd, result, dest);
|
||||
return;
|
||||
case 0x20: // ADDq
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] + m_state.q;
|
||||
applyDest(vd, result, dest);
|
||||
result[c] = vs[c] + q;
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x21: // MADDq
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] + vs[c] * m_state.q;
|
||||
applyDest(vd, result, dest);
|
||||
result[c] = acc[c] + vs[c] * q;
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x22: // ADDi
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] + m_state.i;
|
||||
applyDest(vd, result, dest);
|
||||
result[c] = vs[c] + i;
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x23: // MADDi
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] + vs[c] * m_state.i;
|
||||
applyDest(vd, result, dest);
|
||||
result[c] = acc[c] + vs[c] * i;
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x24: // SUBq
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] - m_state.q;
|
||||
applyDest(vd, result, dest);
|
||||
result[c] = vs[c] - q;
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x25: // MSUBq
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] - vs[c] * m_state.q;
|
||||
applyDest(vd, result, dest);
|
||||
result[c] = acc[c] - vs[c] * q;
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x26: // SUBi
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] - m_state.i;
|
||||
applyDest(vd, result, dest);
|
||||
result[c] = vs[c] - i;
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x27: // MSUBi
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] - vs[c] * m_state.i;
|
||||
applyDest(vd, result, dest);
|
||||
result[c] = acc[c] - vs[c] * i;
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x28: // ADD
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] + vt[c];
|
||||
applyDest(vd, result, dest);
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x29: // MADD
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] + vs[c] * vt[c];
|
||||
applyDest(vd, result, dest);
|
||||
result[c] = acc[c] + vs[c] * vt[c];
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x2A: // MUL
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] * vt[c];
|
||||
applyDest(vd, result, dest);
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x2B: // MAX
|
||||
for (int c = 0; c < 4; c++)
|
||||
@@ -183,19 +210,19 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
case 0x2C: // SUB
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] - vt[c];
|
||||
applyDest(vd, result, dest);
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x2D: // MSUB
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] - vs[c] * vt[c];
|
||||
applyDest(vd, result, dest);
|
||||
result[c] = acc[c] - vs[c] * vt[c];
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x2E: // OPMSUB
|
||||
result[0] = m_state.acc[0] - vs[1] * vt[2];
|
||||
result[1] = m_state.acc[1] - vs[2] * vt[0];
|
||||
result[2] = m_state.acc[2] - vs[0] * vt[1];
|
||||
result[0] = acc[0] - vs[1] * vt[2];
|
||||
result[1] = acc[1] - vs[2] * vt[0];
|
||||
result[2] = acc[2] - vs[0] * vt[1];
|
||||
result[3] = 0.0f;
|
||||
applyDest(vd, result, dest);
|
||||
applyFmacDest(vd, result, dest);
|
||||
return;
|
||||
case 0x2F: // MINI
|
||||
for (int c = 0; c < 4; c++)
|
||||
@@ -225,7 +252,7 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
float bc = broadcast(vt, specialOp & 3);
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] + bc;
|
||||
applyDestAcc(result, dest);
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
}
|
||||
case 0x04:
|
||||
@@ -236,7 +263,7 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
float bc = broadcast(vt, specialOp & 3);
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] - bc;
|
||||
applyDestAcc(result, dest);
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
}
|
||||
case 0x08:
|
||||
@@ -246,8 +273,8 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
{
|
||||
float bc = broadcast(vt, specialOp & 3);
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] + vs[c] * bc;
|
||||
applyDestAcc(result, dest);
|
||||
result[c] = acc[c] + vs[c] * bc;
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
}
|
||||
case 0x0C:
|
||||
@@ -257,15 +284,15 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
{
|
||||
float bc = broadcast(vt, specialOp & 3);
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] - vs[c] * bc;
|
||||
applyDestAcc(result, dest);
|
||||
result[c] = acc[c] - vs[c] * bc;
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
}
|
||||
case 0x10: // ITOF0
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
int32_t iv;
|
||||
std::memcpy(&iv, &vs[c], 4);
|
||||
std::memcpy(&iv, &m_state.vf[fs][c], 4);
|
||||
result[c] = static_cast<float>(iv);
|
||||
}
|
||||
applyDest(vtDest, result, dest);
|
||||
@@ -274,7 +301,7 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
int32_t iv;
|
||||
std::memcpy(&iv, &vs[c], 4);
|
||||
std::memcpy(&iv, &m_state.vf[fs][c], 4);
|
||||
result[c] = static_cast<float>(iv) / 16.0f;
|
||||
}
|
||||
applyDest(vtDest, result, dest);
|
||||
@@ -283,7 +310,7 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
int32_t iv;
|
||||
std::memcpy(&iv, &vs[c], 4);
|
||||
std::memcpy(&iv, &m_state.vf[fs][c], 4);
|
||||
result[c] = static_cast<float>(iv) / 4096.0f;
|
||||
}
|
||||
applyDest(vtDest, result, dest);
|
||||
@@ -292,7 +319,7 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
int32_t iv;
|
||||
std::memcpy(&iv, &vs[c], 4);
|
||||
std::memcpy(&iv, &m_state.vf[fs][c], 4);
|
||||
result[c] = static_cast<float>(iv) / 32768.0f;
|
||||
}
|
||||
applyDest(vtDest, result, dest);
|
||||
@@ -300,7 +327,7 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
case 0x14: // FTOI0
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
int32_t iv = static_cast<int32_t>(vs[c]);
|
||||
int32_t iv = vuFloatToInt(vs[c], 1.0f);
|
||||
std::memcpy(&result[c], &iv, 4);
|
||||
}
|
||||
applyDest(vtDest, result, dest);
|
||||
@@ -308,7 +335,7 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
case 0x15: // FTOI4
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
int32_t iv = static_cast<int32_t>(vs[c] * 16.0f);
|
||||
int32_t iv = vuFloatToInt(vs[c], 16.0f);
|
||||
std::memcpy(&result[c], &iv, 4);
|
||||
}
|
||||
applyDest(vtDest, result, dest);
|
||||
@@ -316,7 +343,7 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
case 0x16: // FTOI12
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
int32_t iv = static_cast<int32_t>(vs[c] * 4096.0f);
|
||||
int32_t iv = vuFloatToInt(vs[c], 4096.0f);
|
||||
std::memcpy(&result[c], &iv, 4);
|
||||
}
|
||||
applyDest(vtDest, result, dest);
|
||||
@@ -324,7 +351,7 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
case 0x17: // FTOI15
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
int32_t iv = static_cast<int32_t>(vs[c] * 32768.0f);
|
||||
int32_t iv = vuFloatToInt(vs[c], 32768.0f);
|
||||
std::memcpy(&result[c], &iv, 4);
|
||||
}
|
||||
applyDest(vtDest, result, dest);
|
||||
@@ -337,13 +364,13 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
float bc = broadcast(vt, specialOp & 3);
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] * bc;
|
||||
applyDestAcc(result, dest);
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
}
|
||||
case 0x1C: // MULAq
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] * m_state.q;
|
||||
applyDestAcc(result, dest);
|
||||
result[c] = vs[c] * q;
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x1D: // ABS
|
||||
for (int c = 0; c < 4; c++)
|
||||
@@ -352,98 +379,118 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
return;
|
||||
case 0x1E: // MULAi
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] * m_state.i;
|
||||
applyDestAcc(result, dest);
|
||||
result[c] = vs[c] * i;
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x1F: // CLIP
|
||||
{
|
||||
float w = std::fabs(vt[3]);
|
||||
uint32_t flags = 0;
|
||||
if (vs[0] > +w) flags |= 0x01;
|
||||
if (vs[0] < -w) flags |= 0x02;
|
||||
if (vs[1] > +w) flags |= 0x04;
|
||||
if (vs[1] < -w) flags |= 0x08;
|
||||
if (vs[2] > +w) flags |= 0x10;
|
||||
if (vs[2] < -w) flags |= 0x20;
|
||||
m_state.clip = (m_state.clip << 6) | flags;
|
||||
uint32_t wBits = 0u;
|
||||
std::memcpy(&wBits, &m_state.vf[ft][3], sizeof(wBits));
|
||||
const int32_t limit = (wBits & 0x7F800000u) != 0u ? static_cast<int32_t>(wBits & 0x7FFFFFFFu) : 0x007FFFFF;
|
||||
|
||||
const auto exceedsClipPlane = [limit](float value, uint32_t signMask)
|
||||
{
|
||||
uint32_t bits = 0u;
|
||||
std::memcpy(&bits, &value, sizeof(bits));
|
||||
bits ^= signMask;
|
||||
int32_t orderedBits = 0;
|
||||
std::memcpy(&orderedBits, &bits, sizeof(orderedBits));
|
||||
return orderedBits > limit;
|
||||
};
|
||||
|
||||
uint32_t flags = 0u;
|
||||
if (exceedsClipPlane(m_state.vf[fs][0], 0x00000000u))
|
||||
flags |= 0x01u;
|
||||
if (exceedsClipPlane(m_state.vf[fs][0], 0x80000000u))
|
||||
flags |= 0x02u;
|
||||
if (exceedsClipPlane(m_state.vf[fs][1], 0x00000000u))
|
||||
flags |= 0x04u;
|
||||
if (exceedsClipPlane(m_state.vf[fs][1], 0x80000000u))
|
||||
flags |= 0x08u;
|
||||
if (exceedsClipPlane(m_state.vf[fs][2], 0x00000000u))
|
||||
flags |= 0x10u;
|
||||
if (exceedsClipPlane(m_state.vf[fs][2], 0x80000000u))
|
||||
flags |= 0x20u;
|
||||
queueClip(flags);
|
||||
return;
|
||||
}
|
||||
case 0x20: // ADDAq
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] + m_state.q;
|
||||
applyDestAcc(result, dest);
|
||||
result[c] = vs[c] + q;
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x21: // MADDAq
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] + vs[c] * m_state.q;
|
||||
applyDestAcc(result, dest);
|
||||
result[c] = acc[c] + vs[c] * q;
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x22: // ADDAi
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] + m_state.i;
|
||||
applyDestAcc(result, dest);
|
||||
result[c] = vs[c] + i;
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x23: // MADDAi
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] + vs[c] * m_state.i;
|
||||
applyDestAcc(result, dest);
|
||||
result[c] = acc[c] + vs[c] * i;
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x24: // SUBAq
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] - m_state.q;
|
||||
applyDestAcc(result, dest);
|
||||
result[c] = vs[c] - q;
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x25: // MSUBAq
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] - vs[c] * m_state.q;
|
||||
applyDestAcc(result, dest);
|
||||
result[c] = acc[c] - vs[c] * q;
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x26: // SUBAi
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] - m_state.i;
|
||||
applyDestAcc(result, dest);
|
||||
result[c] = vs[c] - i;
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x27: // MSUBAi
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] - vs[c] * m_state.i;
|
||||
applyDestAcc(result, dest);
|
||||
result[c] = acc[c] - vs[c] * i;
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x28: // ADDA
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] + vt[c];
|
||||
applyDestAcc(result, dest);
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x29: // MADDA
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] + vs[c] * vt[c];
|
||||
applyDestAcc(result, dest);
|
||||
result[c] = acc[c] + vs[c] * vt[c];
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x2A: // MULA
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] * vt[c];
|
||||
applyDestAcc(result, dest);
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x2C: // SUBA
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = vs[c] - vt[c];
|
||||
applyDestAcc(result, dest);
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x2D: // MSUBA
|
||||
for (int c = 0; c < 4; c++)
|
||||
result[c] = m_state.acc[c] - vs[c] * vt[c];
|
||||
applyDestAcc(result, dest);
|
||||
result[c] = acc[c] - vs[c] * vt[c];
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x2E: // OPMULA
|
||||
result[0] = vs[1] * vt[2];
|
||||
result[1] = vs[2] * vt[0];
|
||||
result[2] = vs[0] * vt[1];
|
||||
result[3] = 0.0f;
|
||||
applyDestAcc(result, dest);
|
||||
applyFmacDestAcc(result, dest);
|
||||
return;
|
||||
case 0x2F:
|
||||
case 0x30: // NOP
|
||||
return;
|
||||
default:
|
||||
reportReservedInstruction(true, instr);
|
||||
return;
|
||||
}
|
||||
}
|
||||
@@ -453,6 +500,7 @@ void VU1Interpreter::execUpper(uint32_t instr)
|
||||
case 0x32:
|
||||
case 0x33:
|
||||
default:
|
||||
reportReservedInstruction(true, instr);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user