mirror of
https://github.com/ran-j/PS2Recomp.git
synced 2026-09-29 09:50:27 -04:00
refactor: refactor VU1 (#191)
* feat: implement fix and changes based on dark cloud report fix: fix GS AFAIL for RGB/alpha/Z, ZMSK fix: fix VU1 flags mask and pipeline fix: small VU1 cache fix feat: __ct__, __sinit_ are not sillent stubs anymore * feat: fix song JP pulling * feat: sound update for lotR * feat: prevent guest execution to be very slow * fix: small gs size bug * feat: refactor VU fix: fix cliping and other issues on gs fix: fix wrong vu0 register on recompiler * fix fix ACC scheduler stall feat: remove unused test fix: .fix overflow e underflow on FMAC * feat: small setting for windows test
This commit is contained in:
@@ -399,12 +399,15 @@ private:
|
||||
|
||||
GSContext m_ctx[2];
|
||||
GSPrimReg m_prim{};
|
||||
GSPrimReg m_primRegister{};
|
||||
GSPrimReg m_prmodeRegister{};
|
||||
|
||||
uint8_t m_curR = 0x80, m_curG = 0x80, m_curB = 0x80, m_curA = 0x80;
|
||||
float m_curQ = 1.0f;
|
||||
float m_curS = 0.0f, m_curT = 0.0f;
|
||||
uint16_t m_curU = 0, m_curV = 0;
|
||||
uint8_t m_curFog = 0;
|
||||
uint8_t m_fogR = 0, m_fogG = 0, m_fogB = 0;
|
||||
|
||||
bool m_prmodecont = true;
|
||||
bool m_pabe = false;
|
||||
@@ -462,8 +465,9 @@ private:
|
||||
using WriteVramFunc = std::function<void(u8*, uint32_t, uint32_t, uint32_t, uint32_t, uint32_t)>;
|
||||
using ReadVramFunc = std::function<u32(u8*, u32, u32, u32, u32)>;
|
||||
|
||||
std::array<ReadVramFunc, 0x3F> m_read_vram_funcs{ };
|
||||
std::array<WriteVramFunc, 0x3F> m_write_vram_funcs{ };
|
||||
static constexpr size_t kPsmHandlerCount = 1u << 6u;
|
||||
std::array<ReadVramFunc, kPsmHandlerCount> m_read_vram_funcs{ };
|
||||
std::array<WriteVramFunc, kPsmHandlerCount> m_write_vram_funcs{ };
|
||||
};
|
||||
|
||||
inline u32 GS::ReadVram(u32 psm, u32 base, u32 bw, u32 x, u32 y) const
|
||||
|
||||
@@ -9,7 +9,7 @@ class GSRasterizer
|
||||
{
|
||||
public:
|
||||
void drawPrimitive(GS *gs);
|
||||
void writePixel(GS *gs, int x, int y, int z, uint8_t r, uint8_t g, uint8_t b, uint8_t a);
|
||||
void writePixel(GS *gs, int x, int y, int z, uint8_t r, uint8_t g, uint8_t b, uint8_t a, uint8_t fog);
|
||||
uint32_t sampleTexture(GS *gs, float s, float t, float q, uint16_t u, uint16_t v);
|
||||
uint32_t lookupCLUT(GS *gs, uint8_t index, uint32_t cbp, uint8_t cpsm, uint8_t csm, uint8_t csa, uint8_t sourcePsm);
|
||||
|
||||
|
||||
@@ -285,6 +285,7 @@ public:
|
||||
uint64_t gifCopyCount() const { return m_gifCopyCount.load(std::memory_order_relaxed); }
|
||||
uint64_t gsWriteCount() const { return m_gsWriteCount.load(std::memory_order_relaxed); }
|
||||
uint64_t vifWriteCount() const { return m_vifWriteCount.load(std::memory_order_relaxed); }
|
||||
uint64_t getVU0CodeGeneration() const { return m_vu0CodeGeneration.load(std::memory_order_relaxed); }
|
||||
uint64_t getVU1CodeGeneration() const { return m_vu1CodeGeneration.load(std::memory_order_relaxed); }
|
||||
|
||||
// Read/write memory
|
||||
@@ -372,6 +373,7 @@ public:
|
||||
std::atomic<uint64_t> m_gifCopyCount{0};
|
||||
std::atomic<uint64_t> m_gsWriteCount{0};
|
||||
std::atomic<uint64_t> m_vifWriteCount{0};
|
||||
std::atomic<uint64_t> m_vu0CodeGeneration{0};
|
||||
std::atomic<uint64_t> m_vu1CodeGeneration{0};
|
||||
// I/O registers
|
||||
std::unordered_map<uint32_t, uint32_t> m_ioRegisters;
|
||||
@@ -431,6 +433,7 @@ public:
|
||||
|
||||
bool isAddressInRegion(uint32_t address, const CodeRegion ®ion);
|
||||
void markModified(uint32_t address, uint32_t size);
|
||||
void markVU0CodeModified() { m_vu0CodeGeneration.fetch_add(1, std::memory_order_relaxed); }
|
||||
void markVU1CodeModified() { m_vu1CodeGeneration.fetch_add(1, std::memory_order_relaxed); }
|
||||
bool isScratchpad(uint32_t address) const;
|
||||
uint8_t *mapVuMemory(uint32_t physAddr, uint32_t size, uint32_t &offset, uint32_t &limit);
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
#ifndef PS2_VU1_H
|
||||
#define PS2_VU1_H
|
||||
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
|
||||
class GS;
|
||||
class PS2Memory;
|
||||
@@ -15,13 +15,20 @@ struct VU1State
|
||||
float q;
|
||||
float p;
|
||||
float i;
|
||||
uint32_t r;
|
||||
uint32_t pc;
|
||||
uint32_t mac;
|
||||
uint32_t clip;
|
||||
uint32_t status;
|
||||
uint64_t cycles;
|
||||
bool ebit;
|
||||
uint32_t top; // VIF1 TOP visible to VU1 XTOP
|
||||
uint32_t itop; // VIF1 ITOP visible to VU1 XITOP
|
||||
bool haltAfterDelaySlot;
|
||||
bool dBitEnabled;
|
||||
bool tBitEnabled;
|
||||
bool stoppedByD;
|
||||
bool stoppedByT;
|
||||
uint32_t top; // VIF TOP visible to XTOP
|
||||
uint32_t itop; // VIF ITOP visible to XITOP
|
||||
|
||||
bool branchPending;
|
||||
uint32_t branchTarget;
|
||||
@@ -31,7 +38,13 @@ struct VU1State
|
||||
class VU1Interpreter
|
||||
{
|
||||
public:
|
||||
VU1Interpreter();
|
||||
enum class Unit : uint8_t
|
||||
{
|
||||
VU0,
|
||||
VU1
|
||||
};
|
||||
|
||||
explicit VU1Interpreter(Unit unit = Unit::VU1);
|
||||
|
||||
void reset();
|
||||
|
||||
@@ -50,38 +63,236 @@ public:
|
||||
const VU1State &state() const { return m_state; }
|
||||
|
||||
private:
|
||||
enum Pipeline : uint8_t
|
||||
{
|
||||
PipelineNone = 0,
|
||||
PipelineFmac,
|
||||
PipelineLsu,
|
||||
PipelineFdiv,
|
||||
PipelineEfu,
|
||||
PipelineIalu,
|
||||
PipelineBranch,
|
||||
PipelineXgkick
|
||||
};
|
||||
|
||||
struct VfAccess
|
||||
{
|
||||
uint8_t reg = 0;
|
||||
uint8_t lanes = 0;
|
||||
};
|
||||
|
||||
struct InstructionUsage
|
||||
{
|
||||
std::array<VfAccess, 2> vfRead{};
|
||||
VfAccess vfWrite{};
|
||||
uint8_t vfReadCount = 0;
|
||||
uint16_t viRead = 0;
|
||||
uint16_t viWrite = 0;
|
||||
uint8_t accRead = 0;
|
||||
uint8_t accWrite = 0;
|
||||
uint8_t latency = 0;
|
||||
uint8_t vfLatency = 0;
|
||||
uint8_t viLatency = 0;
|
||||
Pipeline pipeline = PipelineNone;
|
||||
bool waitQ = false;
|
||||
bool waitP = false;
|
||||
bool readsClip = false;
|
||||
bool writesClip = false;
|
||||
bool delaysNextBranchRead = false;
|
||||
bool reserved = false;
|
||||
};
|
||||
|
||||
struct DecodedInstructionPair
|
||||
{
|
||||
uint32_t lower = 0;
|
||||
uint32_t upper = 0;
|
||||
InstructionUsage lowerUsage{};
|
||||
InstructionUsage upperUsage{};
|
||||
bool iBit = false;
|
||||
bool eBit = false;
|
||||
bool lowerBeforeUpper = false;
|
||||
bool mBit = false;
|
||||
bool dBit = false;
|
||||
bool tBit = false;
|
||||
uint8_t upperVfShadowReg = 0;
|
||||
uint8_t suppressedLowerVf = 0;
|
||||
};
|
||||
|
||||
struct FlagPipelineEntry
|
||||
{
|
||||
uint64_t readyCycle = 0;
|
||||
uint64_t issueCycle = 0;
|
||||
uint32_t mac = 0;
|
||||
uint32_t status = 0;
|
||||
uint32_t extraSticky = 0;
|
||||
uint32_t clip = 0;
|
||||
bool valid = false;
|
||||
bool writesMac = false;
|
||||
bool writesStatus = false;
|
||||
bool writesSticky = false;
|
||||
bool writesClip = false;
|
||||
};
|
||||
|
||||
struct ScalarPipelineEntry
|
||||
{
|
||||
uint64_t readyCycle = 0;
|
||||
float value = 0.0f;
|
||||
uint32_t statusDi = 0;
|
||||
bool valid = false;
|
||||
};
|
||||
|
||||
struct PendingStore
|
||||
{
|
||||
uint64_t readyCycle = 0;
|
||||
uint32_t address = 0;
|
||||
std::array<uint32_t, 4> words{};
|
||||
uint8_t laneMask = 0;
|
||||
bool valid = false;
|
||||
};
|
||||
|
||||
struct PendingVfWrite
|
||||
{
|
||||
uint64_t readyCycle = 0;
|
||||
uint64_t sequence = 0;
|
||||
std::array<float, 4> value{};
|
||||
uint8_t reg = 0;
|
||||
uint8_t laneMask = 0;
|
||||
bool valid = false;
|
||||
};
|
||||
|
||||
struct PendingViWrite
|
||||
{
|
||||
uint64_t readyCycle = 0;
|
||||
uint64_t sequence = 0;
|
||||
int32_t value = 0;
|
||||
uint8_t reg = 0;
|
||||
bool valid = false;
|
||||
};
|
||||
|
||||
struct PendingAccWrite
|
||||
{
|
||||
uint64_t readyCycle = 0;
|
||||
uint64_t sequence = 0;
|
||||
std::array<float, 4> value{};
|
||||
uint8_t laneMask = 0;
|
||||
bool valid = false;
|
||||
};
|
||||
|
||||
struct XgkickPipeline
|
||||
{
|
||||
static constexpr uint32_t kBufferSize = 0x10000u;
|
||||
std::array<uint8_t, kBufferSize> packet{};
|
||||
uint32_t sourceAddress = 0;
|
||||
uint32_t totalBytes = 0;
|
||||
uint32_t copiedBytes = 0;
|
||||
uint32_t currentTagEnd = 0;
|
||||
uint32_t cycleCredit = 0;
|
||||
uint64_t issueCycle = 0;
|
||||
bool active = false;
|
||||
bool currentTagEop = false;
|
||||
};
|
||||
|
||||
static constexpr uint32_t kFmacLatency = 4u;
|
||||
static constexpr uint32_t kAccForwardLatency = 1u;
|
||||
static constexpr uint32_t kMaxFlagEntries = 8u;
|
||||
static constexpr uint32_t kMaxPendingStores = 8u;
|
||||
static constexpr uint32_t kMaxPendingVfWrites = 16u;
|
||||
static constexpr uint32_t kMaxPendingViWrites = 8u;
|
||||
static constexpr uint32_t kMaxPendingAccWrites = 8u;
|
||||
static constexpr uint32_t kMaxDecodedPairs = 0x4000u / 8u;
|
||||
|
||||
Unit m_unit;
|
||||
VU1State m_state;
|
||||
std::vector<DecodedInstructionPair> m_decodedCodeCache;
|
||||
std::array<DecodedInstructionPair, kMaxDecodedPairs> m_decodedCodeCache{};
|
||||
const uint8_t *m_cachedVuCode = nullptr;
|
||||
const PS2Memory *m_cachedMemory = nullptr;
|
||||
uint32_t m_cachedCodeSize = 0;
|
||||
uint64_t m_cachedCodeGeneration = 0;
|
||||
bool m_decodedCodeCacheValid = false;
|
||||
|
||||
std::array<FlagPipelineEntry, kMaxFlagEntries> m_flagPipeline{};
|
||||
ScalarPipelineEntry m_fdiv{};
|
||||
std::array<ScalarPipelineEntry, 2> m_efu{};
|
||||
std::array<PendingStore, kMaxPendingStores> m_storePipeline{};
|
||||
std::array<PendingVfWrite, kMaxPendingVfWrites> m_vfWritePipeline{};
|
||||
std::array<PendingViWrite, kMaxPendingViWrites> m_viWritePipeline{};
|
||||
std::array<PendingAccWrite, kMaxPendingAccWrites> m_accWritePipeline{};
|
||||
XgkickPipeline m_xgkick{};
|
||||
std::array<std::array<uint64_t, 4>, 32> m_vfReady{};
|
||||
std::array<uint64_t, 16> m_viReady{};
|
||||
std::array<uint64_t, 4> m_accReady{};
|
||||
std::array<std::array<uint64_t, 4>, 32> m_vfLatestWrite{};
|
||||
std::array<uint64_t, 16> m_viLatestWrite{};
|
||||
std::array<uint64_t, 4> m_accLatestWrite{};
|
||||
|
||||
uint64_t m_cycle = 0;
|
||||
uint64_t m_nextWriteSequence = 0;
|
||||
uint64_t m_efuResourceReady = 0;
|
||||
uint32_t m_workingClip = 0;
|
||||
uint32_t m_currentUpperInstruction = 0;
|
||||
int32_t m_viBranchBackupValue = 0;
|
||||
uint8_t m_viBranchBackupReg = 0;
|
||||
bool m_viBranchBackupValid = false;
|
||||
uint8_t *m_activeVuData = nullptr;
|
||||
uint32_t m_activeVuDataSize = 0;
|
||||
GS *m_activeGs = nullptr;
|
||||
PS2Memory *m_activeMemory = nullptr;
|
||||
bool m_stopRequested = false;
|
||||
bool m_pendingHaltD = false;
|
||||
bool m_pendingHaltT = false;
|
||||
|
||||
void run(uint8_t *vuCode, uint32_t codeSize,
|
||||
uint8_t *vuData, uint32_t dataSize,
|
||||
GS &gs, PS2Memory *memory, uint32_t maxCycles);
|
||||
|
||||
InstructionUsage decodeUpperUsage(uint32_t upper) const;
|
||||
InstructionUsage decodeLowerUsage(uint32_t lower) const;
|
||||
static void addVfRead(InstructionUsage &usage, uint8_t reg, uint8_t lanes);
|
||||
static void addVfWrite(InstructionUsage &usage, uint8_t reg, uint8_t lanes);
|
||||
static uint8_t vfReadLanes(const InstructionUsage &usage, uint8_t reg);
|
||||
DecodedInstructionPair decodeInstructionPair(const uint8_t *vuCode, uint32_t pc) const;
|
||||
DecodedInstructionPair getDecodedInstructionPairForPc(const uint8_t *vuCode, uint32_t codeSize,
|
||||
PS2Memory *memory, uint32_t pc);
|
||||
void rebuildDecodedCodeCache(const uint8_t *vuCode, uint32_t codeSize,
|
||||
const PS2Memory *memory, uint64_t generation);
|
||||
DecodedInstructionPair getDecodedInstructionPairForPc(const uint8_t *vuCode, uint32_t codeSize, PS2Memory *memory, uint32_t pc);
|
||||
void rebuildDecodedCodeCache(const uint8_t *vuCode, uint32_t codeSize, const PS2Memory *memory, uint64_t generation);
|
||||
|
||||
void execUpper(uint32_t instr);
|
||||
void execLower(uint32_t instr, uint8_t *vuData, uint32_t dataSize, GS &gs, PS2Memory *memory, uint32_t upperInstr);
|
||||
|
||||
void applyDest(float *dst, const float *result, uint8_t dest);
|
||||
void applyDestAcc(const float *result, uint8_t dest);
|
||||
void applyFmacDest(float *dst, float *result, uint8_t dest);
|
||||
void applyFmacDestAcc(float *result, uint8_t dest);
|
||||
void normalizeFmacResult(float *result, uint8_t dest, uint8_t laneFlags[4]);
|
||||
bool calculateFmacExactResult(uint32_t component, long double &result) const;
|
||||
uint8_t normalizeFmacExactResult(float &value, long double exactResult) const;
|
||||
uint32_t calculateFmacProductSticky(uint8_t dest) const;
|
||||
void updateFmacFlags(const uint8_t laneFlags[4], uint8_t dest, uint32_t extraSticky);
|
||||
void queueFsset(uint16_t immediate);
|
||||
void queueClip(uint32_t clip);
|
||||
void queueFcset(uint32_t clip);
|
||||
void queueQ(float value, uint32_t latency, uint32_t statusDi);
|
||||
void queueP(float value, uint32_t latency);
|
||||
void queueStore(uint32_t address, const uint32_t words[4], uint8_t laneMask);
|
||||
void queueVfWrite(uint8_t reg, uint8_t laneMask, const float value[4], uint32_t latency);
|
||||
void queueViWrite(uint8_t reg, int32_t value, uint32_t latency);
|
||||
void queueAccWrite(uint8_t laneMask, const float value[4], uint32_t latency);
|
||||
void startXgkick(uint32_t qwordAddress);
|
||||
|
||||
void resetScheduler();
|
||||
void commitReadyPipelines();
|
||||
void advanceOneCycle();
|
||||
void advanceTo(uint64_t targetCycle);
|
||||
void flushPipelines();
|
||||
void progressXgkick();
|
||||
void finishXgkick();
|
||||
uint64_t calculatePairReadyCycle(const DecodedInstructionPair &decoded) const;
|
||||
void markPairWrites(const DecodedInstructionPair &decoded);
|
||||
bool pipelinesPending() const;
|
||||
|
||||
float normalizeOperand(float value) const;
|
||||
float normalizeResult(float value, uint32_t &laneFlags) const;
|
||||
uint32_t microAddressMask() const;
|
||||
int32_t readBranchVi(uint8_t reg) const;
|
||||
void recordViWriteForBranch(uint8_t reg, int32_t oldValue);
|
||||
void reportReservedInstruction(bool upper, uint32_t instruction);
|
||||
float broadcast(const float *vf, uint8_t bc);
|
||||
};
|
||||
|
||||
|
||||
Reference in New Issue
Block a user