refactor: refactor runtime and recompile for better workflow and correct generating code and analyze output

This commit is contained in:
Ran-j
2026-01-04 22:21:50 -03:00
parent 2ca84c131a
commit 2e08d5e87b
9 changed files with 1535 additions and 546 deletions
+308 -43
View File
@@ -2,9 +2,104 @@
#include <iostream>
#include <cstring>
#include <stdexcept>
#include <unordered_map>
namespace
{
inline bool isGsPrivReg(uint32_t addr)
{
return addr >= PS2_GS_PRIV_REG_BASE && addr < PS2_GS_PRIV_REG_BASE + PS2_GS_PRIV_REG_SIZE;
}
inline uint64_t *gsRegPtr(GSRegisters &gs, uint32_t addr)
{
uint32_t off = addr - PS2_GS_PRIV_REG_BASE;
switch (off)
{
case 0x0000:
return &gs.pmode;
case 0x0010:
return &gs.smode1;
case 0x0020:
return &gs.smode2;
case 0x0030:
return &gs.srfsh;
case 0x0040:
return &gs.synch1;
case 0x0050:
return &gs.synch2;
case 0x0060:
return &gs.syncv;
case 0x0070:
return &gs.dispfb1;
case 0x0080:
return &gs.display1;
case 0x0090:
return &gs.dispfb2;
case 0x00A0:
return &gs.display2;
case 0x00B0:
return &gs.extbuf;
case 0x00C0:
return &gs.extdata;
case 0x00D0:
return &gs.extwrite;
case 0x00E0:
return &gs.bgcolor;
case 0x1000:
return &gs.csr;
case 0x1010:
return &gs.imr;
case 0x1040:
return &gs.busdir;
case 0x1080:
return &gs.siglblid;
default:
return nullptr;
}
}
inline void logGsWrite(uint32_t addr, uint64_t value)
{
static std::unordered_map<uint32_t, int> logCount;
int &count = logCount[addr];
if (count < 10)
{
std::cout << "[GS] write 0x" << std::hex << addr << " = 0x" << value << std::dec << std::endl;
}
++count;
}
constexpr uint32_t kSchedulerBase = 0x00363a10;
constexpr uint32_t kSchedulerSpan = 0x00000420;
static int g_schedWriteLogCount = 0;
inline void logSchedulerWrite(uint32_t physAddr, uint32_t size, uint64_t value)
{
if (physAddr < kSchedulerBase || physAddr >= kSchedulerBase + kSchedulerSpan)
{
return;
}
if (g_schedWriteLogCount >= 64)
{
return;
}
std::cout << "[sched write" << size << "] addr=0x" << std::hex << physAddr
<< " val=0x" << value << std::dec << std::endl;
++g_schedWriteLogCount;
}
}
// Helpers for GS VRAM addressing (PSMCT32 only in this minimal path).
static inline uint32_t gs_vram_offset(uint32_t basePage, uint32_t x, uint32_t y, uint32_t fbw)
{
// basePage is in 2048-byte units; fbw is in blocks of 64 pixels.
uint32_t strideBytes = fbw * 64 * 4;
return basePage * 2048 + y * strideBytes + x * 4;
}
PS2Memory::PS2Memory()
: m_rdram(nullptr), m_scratchpad(nullptr)
: m_rdram(nullptr), m_scratchpad(nullptr), m_gsVRAM(nullptr), m_seenGifCopy(false)
{
}
@@ -21,6 +116,12 @@ PS2Memory::~PS2Memory()
delete[] m_scratchpad;
m_scratchpad = nullptr;
}
if (m_gsVRAM)
{
delete[] m_gsVRAM;
m_gsVRAM = nullptr;
}
}
bool PS2Memory::initialize(size_t ramSize)
@@ -70,6 +171,20 @@ bool PS2Memory::initialize(size_t ramSize)
// Initialize GS registers
memset(&gs_regs, 0, sizeof(gs_regs));
// Allocate GS VRAM (4MB)
m_gsVRAM = new uint8_t[PS2_GS_VRAM_SIZE];
if (!m_gsVRAM)
{
delete[] m_rdram;
delete[] m_scratchpad;
delete[] iop_ram;
m_rdram = nullptr;
m_scratchpad = nullptr;
iop_ram = nullptr;
return false;
}
std::memset(m_gsVRAM, 0, PS2_GS_VRAM_SIZE);
// Initialize VIF registers
memset(&vif0_regs, 0, sizeof(vif0_regs));
memset(&vif1_regs, 0, sizeof(vif1_regs));
@@ -94,22 +209,17 @@ bool PS2Memory::isScratchpad(uint32_t address) const
uint32_t PS2Memory::translateAddress(uint32_t virtualAddress)
{
// Handle special memory regions
if (isScratchpad(virtualAddress))
{
// Scratchpad is directly mapped
return virtualAddress - PS2_SCRATCHPAD_BASE;
}
// For RDRAM, mask the address to get the physical address
if (virtualAddress < PS2_RAM_SIZE ||
(virtualAddress >= 0x80000000 && virtualAddress < 0x80000000 + PS2_RAM_SIZE))
{
// KSEG0 is directly mapped, just mask out the high bits
return virtualAddress & 0x1FFFFFFF;
}
// For addresses that need TLB lookup
if (virtualAddress >= 0xC0000000)
{
for (const auto &entry : m_tlbEntries)
@@ -128,11 +238,9 @@ uint32_t PS2Memory::translateAddress(uint32_t virtualAddress)
}
}
}
// TLB miss
throw std::runtime_error("TLB miss for address: 0x" + std::to_string(virtualAddress));
}
// Default to simple masking for other addresses
return virtualAddress & 0x1FFFFFFF;
}
@@ -151,24 +259,22 @@ uint8_t PS2Memory::read8(uint32_t address)
}
else if (physAddr >= PS2_IO_BASE && physAddr < PS2_IO_BASE + PS2_IO_SIZE)
{
// IO registers - often not handled byte by byte
uint32_t regAddr = physAddr & ~0x3; // Align to word boundary
uint32_t regAddr = physAddr & ~0x3;
if (m_ioRegisters.find(regAddr) != m_ioRegisters.end())
{
uint32_t value = m_ioRegisters[regAddr];
uint32_t shift = (physAddr & 3) * 8;
return (value >> shift) & 0xFF;
}
return 0; // Unimplemented IO register
return 0;
}
// Handle other memory regions ,for now return 0 for unimplemented regions
// TODO: Handle other memory regions
return 0;
}
uint16_t PS2Memory::read16(uint32_t address)
{
// Check alignment
if (address & 1)
{
throw std::runtime_error("Unaligned 16-bit read at address: 0x" + std::to_string(address));
@@ -187,7 +293,6 @@ uint16_t PS2Memory::read16(uint32_t address)
}
else if (physAddr >= PS2_IO_BASE && physAddr < PS2_IO_BASE + PS2_IO_SIZE)
{
// IO registers - align to word boundary and extract relevant bits
uint32_t regAddr = physAddr & ~0x3;
if (m_ioRegisters.find(regAddr) != m_ioRegisters.end())
{
@@ -195,7 +300,7 @@ uint16_t PS2Memory::read16(uint32_t address)
uint32_t shift = (physAddr & 2) * 8;
return (value >> shift) & 0xFFFF;
}
return 0; // Unimplemented IO register
return 0;
}
return 0;
@@ -203,12 +308,19 @@ uint16_t PS2Memory::read16(uint32_t address)
uint32_t PS2Memory::read32(uint32_t address)
{
// Check alignment
if (address & 3)
{
throw std::runtime_error("Unaligned 32-bit read at address: 0x" + std::to_string(address));
}
if (isGsPrivReg(address))
{
uint64_t *reg = gsRegPtr(gs_regs, address);
uint32_t off = address & 7;
uint64_t val = reg ? *reg : 0;
return (uint32_t)(val >> (off * 8));
}
const bool scratch = isScratchpad(address);
uint32_t physAddr = translateAddress(address);
@@ -222,12 +334,11 @@ uint32_t PS2Memory::read32(uint32_t address)
}
else if (physAddr >= PS2_IO_BASE && physAddr < PS2_IO_BASE + PS2_IO_SIZE)
{
// IO registers
if (m_ioRegisters.find(physAddr) != m_ioRegisters.end())
{
return m_ioRegisters[physAddr];
}
return 0; // Unimplemented IO register
return 0;
}
return 0;
@@ -235,12 +346,17 @@ uint32_t PS2Memory::read32(uint32_t address)
uint64_t PS2Memory::read64(uint32_t address)
{
// Check alignment
if (address & 7)
{
throw std::runtime_error("Unaligned 64-bit read at address: 0x" + std::to_string(address));
}
if (isGsPrivReg(address))
{
uint64_t *reg = gsRegPtr(gs_regs, address);
return reg ? *reg : 0;
}
const bool scratch = isScratchpad(address);
uint32_t physAddr = translateAddress(address);
@@ -259,7 +375,6 @@ uint64_t PS2Memory::read64(uint32_t address)
__m128i PS2Memory::read128(uint32_t address)
{
// Check alignment
if (address & 15)
{
throw std::runtime_error("Unaligned 128-bit read at address: 0x" + std::to_string(address));
@@ -294,6 +409,7 @@ void PS2Memory::write8(uint32_t address, uint8_t value)
else if (physAddr < PS2_RAM_SIZE)
{
m_rdram[physAddr] = value;
logSchedulerWrite(physAddr, 8, value);
}
else if (physAddr >= PS2_IO_BASE && physAddr < PS2_IO_BASE + PS2_IO_SIZE)
{
@@ -304,13 +420,12 @@ void PS2Memory::write8(uint32_t address, uint8_t value)
uint32_t newValue = (m_ioRegisters[regAddr] & mask) | ((uint32_t)value << shift);
m_ioRegisters[regAddr] = newValue;
// Handle potential side effects of IO register writes
// TODO: Handle potential side effects of IO register writes
}
}
void PS2Memory::write16(uint32_t address, uint16_t value)
{
// Check alignment
if (address & 1)
{
throw std::runtime_error("Unaligned 16-bit write at address: 0x" + std::to_string(address));
@@ -326,28 +441,41 @@ void PS2Memory::write16(uint32_t address, uint16_t value)
else if (physAddr < PS2_RAM_SIZE)
{
*reinterpret_cast<uint16_t *>(&m_rdram[physAddr]) = value;
logSchedulerWrite(physAddr, 16, value);
}
else if (physAddr >= PS2_IO_BASE && physAddr < PS2_IO_BASE + PS2_IO_SIZE)
{
// IO registers - handle halfword writes
uint32_t regAddr = physAddr & ~0x3;
uint32_t shift = (physAddr & 2) * 8;
uint32_t mask = ~(0xFFFF << shift);
uint32_t newValue = (m_ioRegisters[regAddr] & mask) | ((uint32_t)value << shift);
m_ioRegisters[regAddr] = newValue;
// Handle potential side effects of IO register writes
// TODO: Handle potential side effects of IO register writes
}
}
void PS2Memory::write32(uint32_t address, uint32_t value)
{
// Check alignment
if (address & 3)
{
throw std::runtime_error("Unaligned 32-bit write at address: 0x" + std::to_string(address));
}
if (isGsPrivReg(address))
{
uint64_t *reg = gsRegPtr(gs_regs, address);
if (reg)
{
uint32_t off = address & 7;
uint64_t mask = 0xFFFFFFFFULL << (off * 8);
uint64_t newVal = (*reg & ~mask) | ((uint64_t)value << (off * 8));
*reg = newVal;
logGsWrite(address, newVal);
}
return;
}
const bool scratch = isScratchpad(address);
uint32_t physAddr = translateAddress(address);
@@ -361,9 +489,16 @@ void PS2Memory::write32(uint32_t address, uint32_t value)
markModified(address, 4);
*reinterpret_cast<uint32_t *>(&m_rdram[physAddr]) = value;
logSchedulerWrite(physAddr, 32, value);
}
else if (physAddr >= PS2_IO_BASE && physAddr < PS2_IO_BASE + PS2_IO_SIZE)
{
static int ioLogCount = 0;
if (ioLogCount < 64)
{
std::cout << "[IO write32] addr=0x" << std::hex << physAddr << " val=0x" << value << std::dec << std::endl;
++ioLogCount;
}
// Handle IO register writes with potential side effects
writeIORegister(physAddr, value);
}
@@ -371,12 +506,22 @@ void PS2Memory::write32(uint32_t address, uint32_t value)
void PS2Memory::write64(uint32_t address, uint64_t value)
{
// Check alignment
if (address & 7)
{
throw std::runtime_error("Unaligned 64-bit write at address: 0x" + std::to_string(address));
}
if (isGsPrivReg(address))
{
uint64_t *reg = gsRegPtr(gs_regs, address);
if (reg)
{
*reg = value;
logGsWrite(address, value);
}
return;
}
const bool scratch = isScratchpad(address);
uint32_t physAddr = translateAddress(address);
@@ -387,10 +532,10 @@ void PS2Memory::write64(uint32_t address, uint64_t value)
else if (physAddr < PS2_RAM_SIZE)
{
*reinterpret_cast<uint64_t *>(&m_rdram[physAddr]) = value;
logSchedulerWrite(physAddr, 64, value);
}
else
{
// Split into two 32-bit writes for other memory regions
write32(address, (uint32_t)value);
write32(address + 4, (uint32_t)(value >> 32));
}
@@ -398,7 +543,6 @@ void PS2Memory::write64(uint32_t address, uint64_t value)
void PS2Memory::write128(uint32_t address, __m128i value)
{
// Check alignment
if (address & 15)
{
throw std::runtime_error("Unaligned 128-bit write at address: 0x" + std::to_string(address));
@@ -415,10 +559,12 @@ void PS2Memory::write128(uint32_t address, __m128i value)
{
_mm_storeu_si128(reinterpret_cast<__m128i *>(&m_rdram[physAddr]), value);
}
else if (physAddr < PS2_GS_VRAM_SIZE)
{
_mm_storeu_si128(reinterpret_cast<__m128i *>(&m_gsVRAM[physAddr]), value);
}
else
{
// Split into smaller writes for other memory regions
// Extract the data using SSE intrinsics
uint64_t lo = _mm_extract_epi64(value, 0);
uint64_t hi = _mm_extract_epi64(value, 1);
@@ -429,9 +575,30 @@ void PS2Memory::write128(uint32_t address, __m128i value)
bool PS2Memory::writeIORegister(uint32_t address, uint32_t value)
{
if (address >= 0x10008000 && address < 0x1000F000)
{
static int dmaLogCount = 0;
if (dmaLogCount < 100)
{
uint32_t channelBase = address & 0xFFFFFF00;
uint32_t offset = address & 0xFF;
std::cout << "[DMA reg] ch=0x" << std::hex << channelBase
<< " off=0x" << offset << " = 0x" << value << std::dec << std::endl;
dmaLogCount++;
if (offset == 0x00 && (value & 0x100))
{
uint32_t madr = m_ioRegisters[channelBase + 0x10];
uint32_t qwc = m_ioRegisters[channelBase + 0x20];
uint32_t tadr = m_ioRegisters[channelBase + 0x30];
std::cout << "[DMA start] ch=0x" << std::hex << channelBase
<< " madr=0x" << madr << " qwc=0x" << qwc
<< " tadr=0x" << tadr << std::dec << std::endl;
m_dmaStartCount.fetch_add(1, std::memory_order_relaxed);
}
}
}
m_ioRegisters[address] = value;
// Now check if this is a special hardware register
if (address >= 0x10000000 && address < 0x10010000)
{
// Timer/counter registers
@@ -441,12 +608,53 @@ bool PS2Memory::writeIORegister(uint32_t address, uint32_t value)
return true;
}
// VIF0/VIF1 registers
if (address >= 0x10003800 && address < 0x10003A00)
{
static int vif0Log = 0;
if (vif0Log < 50)
{
std::cout << "[VIF0] write 0x" << std::hex << address << " = 0x" << value << std::dec << std::endl;
++vif0Log;
}
m_vifWriteCount.fetch_add(1, std::memory_order_relaxed);
}
if (address >= 0x10003C00 && address < 0x10003E00)
{
static int vif1Log = 0;
if (vif1Log < 50)
{
std::cout << "[VIF1] write 0x" << std::hex << address << " = 0x" << value << std::dec << std::endl;
++vif1Log;
}
m_vifWriteCount.fetch_add(1, std::memory_order_relaxed);
}
// DMA registers
if (address >= 0x10008000 && address < 0x1000F000)
{
std::cout << "DMA register write: " << std::hex << address << " = " << value << std::dec << std::endl;
// Check if we need to start a DMA transfer
// Dump current DMA regs for all channels
static bool dumpedDma = false;
if (!dumpedDma)
{
for (int ch = 0; ch < 10; ++ch)
{
uint32_t base = 0x10008000 + ch * 0x100;
uint32_t chcr_v = m_ioRegisters[base + 0x00];
uint32_t madr_v = m_ioRegisters[base + 0x10];
uint32_t qwc_v = m_ioRegisters[base + 0x20];
uint32_t tadr_v = m_ioRegisters[base + 0x30];
std::cout << "[DMA dump] ch" << ch
<< " chcr=0x" << std::hex << chcr_v
<< " madr=0x" << madr_v
<< " qwc=0x" << qwc_v
<< " tadr=0x" << tadr_v << std::dec << std::endl;
}
dumpedDma = true;
}
if ((address & 0xFF) == 0x00)
{ // CHCR registers
if (value & 0x100)
@@ -459,17 +667,81 @@ bool PS2Memory::writeIORegister(uint32_t address, uint32_t value)
<< ", MADR: " << std::hex << madr
<< ", QWC: " << qwc << std::dec << std::endl;
// Would actually start DMA here
// Minimal GIF (channel 2) and VIF1 (channel 1) image transfer: copy from EE memory to GS VRAM.
// Only handles simple linear IMAGE transfers; treats destination as current DISPFBUF1 FBP.
if ((channelBase == 0x1000A000 || channelBase == 0x10009000) && m_gsVRAM)
{
auto doCopy = [&](uint32_t srcAddr, uint32_t qwCount)
{
uint32_t bytes = qwCount * 16;
uint32_t src = translateAddress(srcAddr);
uint32_t basePage = static_cast<uint32_t>(gs_regs.dispfb1 & 0x1FF);
uint32_t dest = basePage * 2048;
std::cout << "[GIF] ch=" << ((channelBase == 0x1000A000) ? 2 : 1)
<< " IMAGE copy bytes=" << bytes
<< " src=0x" << std::hex << srcAddr
<< " (phys 0x" << src << ")"
<< " dest=0x" << dest << std::dec << std::endl;
if (dest + bytes > PS2_GS_VRAM_SIZE)
{
bytes = std::min<uint32_t>(bytes, PS2_GS_VRAM_SIZE - dest);
}
if (src + bytes > PS2_RAM_SIZE)
{
bytes = std::min<uint32_t>(bytes, PS2_RAM_SIZE - src);
}
std::memcpy(m_gsVRAM + dest, m_rdram + src, bytes);
m_seenGifCopy = true;
m_gifCopyCount.fetch_add(1, std::memory_order_relaxed);
};
// Dump GIF tag/header
uint32_t phys = translateAddress(madr);
if (phys + 16 <= PS2_RAM_SIZE)
{
const uint8_t *p = m_rdram + phys;
uint64_t tag0 = *reinterpret_cast<const uint64_t *>(p + 0);
uint64_t tag1 = *reinterpret_cast<const uint64_t *>(p + 8);
std::cout << "[GIF] tag0=0x" << std::hex << tag0 << " tag1=0x" << tag1 << std::dec << std::endl;
}
if (qwc > 0)
{
doCopy(madr, qwc);
}
else
{
// Simple DMA chain walker for one tag from TADR (REF/NEXT).
uint32_t tadr = m_ioRegisters[channelBase + 0x30];
uint32_t physTag = translateAddress(tadr);
if (physTag + 16 <= PS2_RAM_SIZE)
{
const uint8_t *tp = m_rdram + physTag;
uint64_t tag = *reinterpret_cast<const uint64_t *>(tp);
uint16_t tagQwc = static_cast<uint16_t>(tag & 0xFFFF);
uint32_t id = static_cast<uint32_t>((tag >> 28) & 0x7);
uint32_t addr = static_cast<uint32_t>((tag >> 32) & 0x7FFFFFF);
std::cout << "[DMA chain] ch=" << ((channelBase == 0x1000A000) ? 2 : 1)
<< " tag id=0x" << std::hex << id
<< " qwc=" << tagQwc
<< " addr=0x" << addr
<< " raw=0x" << tag << std::dec << std::endl;
if (id == 0 || id == 1 || id == 2)
{
doCopy(addr, tagQwc);
}
}
}
m_ioRegisters[address] &= ~0x100;
}
}
}
return true;
}
// Interrupt control registers
if (address >= 0x10000200 && address < 0x10000300)
{
std::cout << "Interrupt register write: " << std::hex << address << " = " << value << std::dec << std::endl;
// Handle interrupt register side effects
return true;
}
}
@@ -477,7 +749,7 @@ bool PS2Memory::writeIORegister(uint32_t address, uint32_t value)
{
// GS registers
std::cout << "GS register write: " << std::hex << address << " = " << value << std::dec << std::endl;
// Handle GS register side effects
m_gsWriteCount.fetch_add(1, std::memory_order_relaxed);
return true;
}
@@ -492,7 +764,6 @@ uint32_t PS2Memory::readIORegister(uint32_t address)
return it->second;
}
// Special cases for reads from hardware registers that have side effects
if (address >= 0x10000000 && address < 0x10010000)
{
// Timer registers
@@ -535,7 +806,6 @@ void PS2Memory::registerCodeRegion(uint32_t start, uint32_t end)
region.start = start;
region.end = end;
// Initialize the modified bitmap (one bit per 4-byte word)
size_t sizeInWords = (end - start) / 4;
region.modified.resize(sizeInWords, false);
@@ -560,7 +830,6 @@ void PS2Memory::markModified(uint32_t address, uint32_t size)
uint32_t overlapStart = std::max(address, region.start);
uint32_t overlapEnd = std::min(address + size, region.end);
// Mark each 4-byte word in the overlap as modified
for (uint32_t addr = overlapStart; addr < overlapEnd; addr += 4)
{
size_t bitIndex = (addr - region.start) / 4;
@@ -582,11 +851,9 @@ bool PS2Memory::isCodeModified(uint32_t address, uint32_t size)
continue;
}
// Calculate overlap
uint32_t overlapStart = std::max(address, region.start);
uint32_t overlapEnd = std::min(address + size, region.end);
// Check each 4-byte word in the overlap
for (uint32_t addr = overlapStart; addr < overlapEnd; addr += 4)
{
size_t bitIndex = (addr - region.start) / 4;
@@ -609,11 +876,9 @@ void PS2Memory::clearModifiedFlag(uint32_t address, uint32_t size)
continue;
}
// Calculate overlap
uint32_t overlapStart = std::max(address, region.start);
uint32_t overlapEnd = std::min(address + size, region.end);
// Clear flags for each 4-byte word in the overlap
for (uint32_t addr = overlapStart; addr < overlapEnd; addr += 4)
{
size_t bitIndex = (addr - region.start) / 4;
+514 -13
View File
@@ -1,13 +1,18 @@
#include "ps2_runtime.h"
#include "ps2_syscalls.h"
#include "ps2_runtime_macros.h"
#include <iostream>
#include <fstream>
#include <algorithm>
#include <cstring>
#include <atomic>
#include <thread>
#include <unordered_map>
#include "raylib.h"
// From ps2_syscalls.cpp to help keep the scheduler semaphore sane.
extern std::atomic<int> g_schedulerSemaId;
#define ELF_MAGIC 0x464C457F // "\x7FELF" in little endian
#define ET_EXEC 2 // Executable file
@@ -54,11 +59,190 @@ struct ProgramHeader
static constexpr int FB_WIDTH = 640;
static constexpr int FB_HEIGHT = 448;
static constexpr uint32_t DEFAULT_FB_ADDR = 0x00100000; // location in RDRAM the guest will draw to
static constexpr uint32_t DEFAULT_FB_SIZE = FB_WIDTH * FB_HEIGHT * 4;
static void UploadFrame(Texture2D &tex, PS2Runtime *rt)
{
uint8_t *src = rt->memory().getRDRAM() + (DEFAULT_FB_ADDR & 0x1FFFFFFF);
UpdateTexture(tex, src);
// Try to use GS dispfb/display registers to locate the visible buffer.
const GSRegisters &gs = rt->memory().gs();
// DISPFBUF1 fields: FBP (bits 0-8) * 2048 bytes, FBW (bits 10-15) blocks of 64 pixels, PSM (bits 16-20)
uint32_t dispfb = static_cast<uint32_t>(gs.dispfb1 & 0xFFFFFFFFULL);
uint32_t fbp = dispfb & 0x1FF;
uint32_t fbw = (dispfb >> 10) & 0x3F;
uint32_t psm = (dispfb >> 16) & 0x1F;
// DISPLAY1 fields: DX,DY not used here; DW,DH are width/height minus 1 (11 bits each)
uint64_t display64 = gs.display1;
uint32_t dw = static_cast<uint32_t>((display64 >> 23) & 0x7FF);
uint32_t dh = static_cast<uint32_t>((display64 >> 34) & 0x7FF);
// Default to 640x448 if regs look strange.
uint32_t width = (dw + 1);
uint32_t height = (dh + 1);
if (dw == 0)
width = FB_WIDTH;
if (dh == 0)
height = FB_HEIGHT;
if (width > FB_WIDTH)
width = FB_WIDTH;
if (height > FB_HEIGHT)
height = FB_HEIGHT;
static uint64_t prev_dispfb = ~0ull;
static uint64_t prev_display = ~0ull;
static bool vramLogged = false;
if (gs.dispfb1 != prev_dispfb || gs.display1 != prev_display)
{
std::cout << "[GS] dispfb1=0x" << std::hex << gs.dispfb1
<< " display1=0x" << gs.display1 << std::dec << std::endl;
prev_dispfb = gs.dispfb1;
prev_display = gs.display1;
// Allow VRAM peek to re-log when the buffer changes.
vramLogged = false;
}
// Only handle PSMCT32 (0) in this minimal blitter.
if (psm != 0)
{
uint8_t *src = rt->memory().getRDRAM() + (DEFAULT_FB_ADDR & 0x1FFFFFFF);
UpdateTexture(tex, src);
return;
}
constexpr uint32_t DEFAULT_FB_ADDR = 0x00100000;
uint32_t baseBytes = fbp * 2048;
if (fbp == 0)
{
baseBytes = DEFAULT_FB_ADDR;
}
uint32_t strideBytes = (fbw ? fbw : (FB_WIDTH / 64)) * 64 * 4;
uint8_t *rdram = rt->memory().getRDRAM();
uint8_t *gsvram = rt->memory().getGSVRAM();
std::vector<uint8_t> scratch(FB_WIDTH * FB_HEIGHT * 4, 0);
for (uint32_t y = 0; y < height; ++y)
{
uint32_t srcOff = baseBytes + y * strideBytes;
uint32_t dstOff = y * FB_WIDTH * 4;
uint32_t copyW = width * 4;
uint32_t srcIdx = srcOff;
if (!vramLogged)
{
uint32_t sum = 0;
for (int i = 0; i < 32 && (srcIdx + i) < PS2_GS_VRAM_SIZE; ++i)
{
sum += gsvram[srcIdx + i];
}
std::cout << "[VRAM peek] sum first32=0x" << std::hex << sum << std::dec << std::endl;
vramLogged = true;
}
if (srcIdx + copyW <= PS2_GS_VRAM_SIZE && gsvram)
{
std::memcpy(&scratch[dstOff], gsvram + srcIdx, copyW);
}
else
{
uint32_t rdramIdx = srcOff & PS2_RAM_MASK;
if (rdramIdx + copyW > PS2_RAM_SIZE)
copyW = PS2_RAM_SIZE - rdramIdx;
std::memcpy(&scratch[dstOff], rdram + rdramIdx, copyW);
}
}
// Peek first few bytes to see if anything is drawn.
uint32_t peekOff = 0;
uint32_t sum = 0;
for (int i = 0; i < 32; ++i)
{
sum += scratch[peekOff + i];
}
static int peekCount = 0;
if (peekCount < 4)
{
std::cout << "[FB peek] sum first32=0x" << std::hex << sum << std::dec
<< " w=" << width << " h=" << height << std::endl;
++peekCount;
}
UpdateTexture(tex, scratch.data());
}
static void DumpFramebufferSample(PS2Memory &mem)
{
uint32_t base = DEFAULT_FB_ADDR & 0x1FFFFFFF;
uint8_t *ptr = mem.getRDRAM() + base;
uint32_t sum = 0;
for (int i = 0; i < 32; ++i)
{
sum += ptr[i];
}
std::cout << "[FB] addr=0x" << std::hex << DEFAULT_FB_ADDR << " first32 sum=0x" << sum << std::dec
<< " bytes:";
for (int i = 0; i < 16; ++i)
{
std::cout << " " << (int)ptr[i];
}
std::cout << std::dec << std::endl;
// Also dump the thread param block used in InitThread (0x363610 area)
uint32_t tparam = 0x363600 & PS2_RAM_MASK;
uint32_t *tp = reinterpret_cast<uint32_t *>(mem.getRDRAM() + tparam);
std::cout << "[InitThread params] @0x363600: "
<< std::hex << tp[0] << " " << tp[1] << " " << tp[2] << " " << tp[3]
<< " " << tp[4] << " " << tp[5] << " " << tp[6] << std::dec << std::endl;
}
// Trace hook for a suspected thread entry to see what it does.
static PS2Runtime::RecompiledFunction g_entry_10c920 = nullptr;
static void entry_10c920_traced(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime)
{
std::cout << "[trace] enter 0x10c920 sp=0x" << std::hex << ctx->r[29].m128i_u32[0]
<< " gp=0x" << ctx->r[28].m128i_u32[0]
<< " ra=0x" << ctx->r[31].m128i_u32[0] << std::dec << std::endl;
if (g_entry_10c920)
{
g_entry_10c920(rdram, ctx, runtime);
}
std::cout << "[trace] exit 0x10c920 pc=0x" << std::hex << ctx->pc
<< " ra=0x" << ctx->r[31].m128i_u32[0] << std::dec << std::endl;
}
// Hook for sceGsPutDispEnv to capture display buffer setup.
static PS2Runtime::RecompiledFunction g_putDispEnv = nullptr;
static void sceGsPutDispEnv_hook(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime)
{
uint32_t envPtr = GPR_U32(ctx, 4);
const uint8_t *base = runtime->memory().getRDRAM();
constexpr uint32_t MASK = PS2_RAM_SIZE - 1;
const uint64_t *env = reinterpret_cast<const uint64_t *>(base + (envPtr & MASK));
if (env)
{
auto &gs = runtime->memory().gs();
gs.dispfb1 = env[0];
gs.display1 = env[1];
std::cout << "[hook] sceGsPutDispEnv @0x" << std::hex << envPtr
<< " dispfb1=0x" << env[0] << " display1=0x" << env[1] << std::dec << std::endl;
}
if (g_putDispEnv)
{
g_putDispEnv(rdram, ctx, runtime);
}
}
// Hook for sceSifCallRpc to keep IOP RPC loops from stalling the main thread.
static PS2Runtime::RecompiledFunction g_sceSifCallRpc = nullptr;
static void sceSifCallRpc_stub(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime)
{
static int logCount = 0;
if (logCount++ < 5)
{
std::cout << "[stub] sceSifCallRpc fno=0x" << std::hex << GPR_U32(ctx, 5)
<< " mode=0x" << GPR_U32(ctx, 6)
<< " send=0x" << GPR_U32(ctx, 7)
<< " recv=0x" << GPR_U32(ctx, 8) << std::dec << std::endl;
}
SET_GPR_S32(ctx, 2, 0);
}
PS2Runtime::PS2Runtime()
@@ -176,6 +360,24 @@ bool PS2Runtime::loadELF(const std::string &elfPath)
m_loadedModules.push_back(module);
// acccordind to GPT some titles expect cmd_sem_init to see -1 sentinels before creating semaphores.
const uint32_t semaInitAddrs[] = {0x00302c90u, 0x00302c94u, 0x00302c98u, 0x00302c9cu};
bool seeded = false;
for (uint32_t addr : semaInitAddrs)
{
uint32_t physAddr = m_memory.translateAddress(addr);
uint32_t *p = reinterpret_cast<uint32_t *>(m_memory.getRDRAM() + physAddr);
if (*p == 0)
{
*p = 0xFFFFFFFFu;
seeded = true;
}
}
if (seeded)
{
std::cout << "[init] Seeded cmd_sem_init sema IDs to -1" << std::endl;
}
// Debug: peek at some early globals to verify init state
uint32_t dbg_addr = 0x00300000 + 11240;
uint8_t *dbg_base = m_memory.getRDRAM();
@@ -191,6 +393,11 @@ void PS2Runtime::registerFunction(uint32_t address, RecompiledFunction func)
m_functionTable[address] = func;
}
bool PS2Runtime::hasFunction(uint32_t address) const
{
return m_functionTable.find(address) != m_functionTable.end();
}
PS2Runtime::RecompiledFunction PS2Runtime::lookupFunction(uint32_t address)
{
auto it = m_functionTable.find(address);
@@ -218,21 +425,81 @@ void PS2Runtime::SignalException(R5900Context *ctx, PS2Exception exception)
}
}
static PS2Runtime::RecompiledFunction g_entry_10cb00 = nullptr;
static void entry_10cb00_hook(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime)
{
static int logCount = 0;
if (logCount < 16)
{
std::cout << "[cmdq] enqueue cmd=0 tid=" << GPR_U32(ctx, 16)
<< " a0=0x" << std::hex << GPR_U32(ctx, 4) << std::dec << std::endl;
++logCount;
}
if (g_entry_10cb00)
{
g_entry_10cb00(rdram, ctx, runtime);
}
}
static PS2Runtime::RecompiledFunction g_entry_10cb98 = nullptr;
static void entry_10cb98_hook(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime)
{
static int logCount = 0;
if (logCount < 16)
{
std::cout << "[cmdq] enqueue cmd=1 tid=" << GPR_U32(ctx, 16)
<< " a0=0x" << std::hex << GPR_U32(ctx, 4) << std::dec << std::endl;
++logCount;
}
if (g_entry_10cb98)
{
g_entry_10cb98(rdram, ctx, runtime);
}
}
static PS2Runtime::RecompiledFunction g_entry_10cc34 = nullptr;
static void entry_10cc34_hook(uint8_t *rdram, R5900Context *ctx, PS2Runtime *runtime)
{
static int logCount = 0;
if (logCount < 16)
{
std::cout << "[cmdq] enqueue cmd=2 tid=" << GPR_U32(ctx, 16)
<< " a0=0x" << std::hex << GPR_U32(ctx, 4) << std::dec << std::endl;
++logCount;
}
if (g_entry_10cc34)
{
g_entry_10cc34(rdram, ctx, runtime);
}
}
void PS2Runtime::executeVU0Microprogram(uint8_t *rdram, R5900Context *ctx, uint32_t address)
{
std::cout << "VU0 microprogram call to address 0x" << std::hex << address
<< " - not implemented" << std::dec << std::endl;
static std::unordered_map<uint32_t, int> seen;
int &count = seen[address];
if (count < 3)
{
std::cout << "[VU0] microprogram @0x" << std::hex << address
<< " pc=0x" << ctx->pc
<< " ra=0x" << ctx->r[31].m128i_u32[0]
<< std::dec << std::endl;
}
++count;
// mayve implement like this or a vu0_interpreter
// Placeholder for VU0 microprogram execution
// auto microprog = findCompiledMicroprogram(address);
// if (microprog) microprog(rdram, ctx);
// Clear/seed status so dependent code sees "success".
ctx->vu0_clip_flags = 0;
ctx->vu0_clip_flags2 = 0;
ctx->vu0_mac_flags = 0;
ctx->vu0_status = 0;
ctx->vu0_q = 1.0f;
// TODO: Implement a real interpreter. For now, no register mutations beyond defaults.
}
void PS2Runtime::vu0StartMicroProgram(uint8_t *rdram, R5900Context *ctx, uint32_t address)
{
std::cout << "VU0 microprogram call to address 0x" << std::hex << address
<< " - not implemented" << std::dec << std::endl;
// VCALLMS/VCALLMSR paths both end up here; reuse the same minimal stub.
executeVU0Microprogram(rdram, ctx, address);
}
void PS2Runtime::handleSyscall(uint8_t *rdram, R5900Context *ctx)
@@ -305,8 +572,118 @@ void PS2Runtime::run()
Texture2D frameTex = LoadTextureFromImage(blank);
UnloadImage(blank);
if (hasFunction(0x10c920))
{
g_entry_10c920 = lookupFunction(0x10c920);
registerFunction(0x10c920, entry_10c920_traced);
std::cout << "[trace] hooked entry 0x10c920 for logging" << std::endl;
}
else
{
std::cout << "[trace] entry 0x10c920 not registered" << std::endl;
}
if (hasFunction(0x1004b8))
{
g_putDispEnv = lookupFunction(0x1004b8);
registerFunction(0x1004b8, sceGsPutDispEnv_hook);
std::cout << "[hook] wrapped sceGsPutDispEnv at 0x1004b8" << std::endl;
}
if (hasFunction(0x10ed80))
{
g_sceSifCallRpc = lookupFunction(0x10ed80);
registerFunction(0x10ed80, sceSifCallRpc_stub);
std::cout << "[hook] wrapped sceSifCallRpc at 0x10ed80" << std::endl;
}
if (hasFunction(0x10cb00))
{
g_entry_10cb00 = lookupFunction(0x10cb00);
registerFunction(0x10cb00, entry_10cb00_hook);
std::cout << "[hook] wrapped cmd queue (cmd=0) at 0x10cb00" << std::endl;
}
if (hasFunction(0x10cb98))
{
g_entry_10cb98 = lookupFunction(0x10cb98);
registerFunction(0x10cb98, entry_10cb98_hook);
std::cout << "[hook] wrapped cmd queue (cmd=1) at 0x10cb98" << std::endl;
}
if (hasFunction(0x10cc34))
{
g_entry_10cc34 = lookupFunction(0x10cc34);
registerFunction(0x10cc34, entry_10cc34_hook);
std::cout << "[hook] wrapped cmd queue (cmd=2) at 0x10cc34" << std::endl;
}
g_activeThreads.store(1, std::memory_order_relaxed);
// for now if the scheduler sema hasn't been created yet, force InitThread to set it up.
if (hasFunction(0x10c9f8))
{
uint32_t *sched = reinterpret_cast<uint32_t *>(m_memory.getRDRAM() + (0x363a10 & PS2_RAM_MASK));
if (!sched || sched[0] == 0)
{
RecompiledFunction initThread = lookupFunction(0x10c9f8);
R5900Context initCtx{};
std::memset(&initCtx, 0, sizeof(initCtx));
initCtx.r[0] = _mm_set1_epi32(0);
initCtx.r[29] = _mm_set1_epi32(0x02000000);
initCtx.r[28] = _mm_set1_epi32(0x36a7f0);
initCtx.pc = 0x10c9f8;
std::cout << "[autorun] running InitThread pc=0x10c9f8" << std::endl;
initThread(m_memory.getRDRAM(), &initCtx, this);
}
}
// Fallback: if the game's main entry (ps2_main at 0x12b0a0) is registered, start it on a separate host thread.
// The normal bootstrap thread seems to stall before spawning it, so we kick it off manually.
if (hasFunction(0x12b0a0))
{
RecompiledFunction ps2Main = lookupFunction(0x12b0a0);
g_activeThreads.fetch_add(1, std::memory_order_relaxed);
std::thread([=]() mutable
{
R5900Context localCtx{};
std::memset(&localCtx, 0, sizeof(localCtx));
// Set baseline registers similar to the primary thread.
localCtx.r[0] = _mm_set1_epi32(0);
localCtx.r[29] = _mm_set1_epi32(0x02000000); // SP top of RAM
localCtx.r[28] = _mm_set1_epi32(0x36a7f0); // GP from ELF bootstrap
localCtx.pc = 0x12b0a0;
std::cout << "[autorun] starting ps2_main fallback pc=0x12b0a0 sp=0x02000000 gp=0x36a7f0" << std::endl;
try
{
ps2Main(m_memory.getRDRAM(), &localCtx, this);
std::cout << "[autorun] ps2_main returned pc=0x" << std::hex << localCtx.pc
<< " ra=0x" << localCtx.r[31].m128i_u32[0] << std::dec << std::endl;
}
catch (const std::exception &e)
{
std::cerr << "[autorun] ps2_main exception: " << e.what() << std::endl;
}
g_activeThreads.fetch_sub(1, std::memory_order_relaxed); })
.detach();
}
else
{
std::cout << "[autorun] ps2_main not registered; skipping fallback launch" << std::endl;
}
// Dump a small sample
{
uint32_t base = DEFAULT_FB_ADDR & 0x1FFFFFFF;
uint8_t *ptr = m_memory.getRDRAM() + base;
uint32_t sum = 0;
for (int i = 0; i < 32; ++i)
sum += ptr[i];
std::cout << "[FB] addr=0x" << std::hex << DEFAULT_FB_ADDR
<< " first32 sum=0x" << sum << " bytes:";
for (int i = 0; i < 16; ++i)
{
std::cout << " " << (int)ptr[i];
}
std::cout << std::dec << std::endl;
}
std::thread gameThread([&, entryPoint]()
{
try
@@ -321,12 +698,137 @@ void PS2Runtime::run()
}
g_activeThreads.fetch_sub(1, std::memory_order_relaxed); });
static uint32_t lastSchedId = 0;
uint64_t tick = 0;
while (g_activeThreads.load(std::memory_order_relaxed) > 0)
{
{
uint32_t *sched = reinterpret_cast<uint32_t *>(m_memory.getRDRAM() + (0x363a10 & PS2_RAM_MASK));
int known = g_schedulerSemaId.load(std::memory_order_relaxed);
if (sched)
{
if ((sched[0] == 0 || sched[0] > 1000) && known > 0)
sched[0] = static_cast<uint32_t>(known);
// head/tail indices
if (sched[2] > 511)
sched[2] = 0;
if (sched[3] > 511)
sched[3] = 0;
}
}
{
constexpr uint32_t kSchedSpan = 0x420;
static std::vector<uint8_t> schedSnapshot;
static int schedDeltaLogs = 0;
uint8_t *rdram = m_memory.getRDRAM();
uint32_t base = 0x363a10 & PS2_RAM_MASK;
if (schedSnapshot.empty())
{
schedSnapshot.resize(kSchedSpan);
std::memcpy(schedSnapshot.data(), rdram + base, kSchedSpan);
}
else
{
int diffCount = 0;
int detailCount = 0;
for (uint32_t i = 0; i < kSchedSpan; ++i)
{
uint8_t cur = rdram[(base + i) & PS2_RAM_MASK];
uint8_t prev = schedSnapshot[i];
if (cur != prev)
{
schedSnapshot[i] = cur;
++diffCount;
if (schedDeltaLogs < 32 && detailCount < 8)
{
std::cout << "[sched delta] off=0x" << std::hex << i
<< " " << (int)prev << "->" << (int)cur << std::dec << std::endl;
++detailCount;
}
}
}
if (diffCount > 0 && schedDeltaLogs < 32)
{
std::cout << "[sched delta] changed=" << diffCount << std::endl;
++schedDeltaLogs;
}
}
}
if ((tick++ % 120) == 0)
{
std::cout << "[run] activeThreads=" << g_activeThreads.load(std::memory_order_relaxed) << std::endl;
std::cout << "[run] activeThreads=" << g_activeThreads.load(std::memory_order_relaxed);
std::cout << " pc=0x" << std::hex << m_cpuContext.pc
<< " ra=0x" << m_cpuContext.r[31].m128i_u32[0]
<< " sp=0x" << m_cpuContext.r[29].m128i_u32[0]
<< " gp=0x" << m_cpuContext.r[28].m128i_u32[0] << std::dec << std::endl;
}
if ((tick % 600) == 0)
{
static int schedLog = 0;
if (schedLog < 5)
{
uint8_t *rdram = m_memory.getRDRAM();
uint32_t base = 0x363a10 & PS2_RAM_MASK;
uint32_t *p = reinterpret_cast<uint32_t *>(rdram + base);
uint32_t argPtr = 0x363a18 & PS2_RAM_MASK;
uint32_t idx = *reinterpret_cast<uint32_t *>(rdram + argPtr) & 0x1FF;
uint32_t cmdBase = (argPtr + 8) & PS2_RAM_MASK;
uint32_t tidBase = (argPtr + 9) & PS2_RAM_MASK;
uint8_t cmd0 = rdram[cmdBase & PS2_RAM_MASK];
uint8_t tid0 = rdram[tidBase & PS2_RAM_MASK];
uint8_t cmdIdx = rdram[(cmdBase + (idx << 1)) & PS2_RAM_MASK];
uint8_t tidIdx = rdram[(tidBase + (idx << 1)) & PS2_RAM_MASK];
std::cout << "[sched] sema=" << p[0] << " q1=" << p[1] << " head=" << p[2] << " tail=" << p[3]
<< " cmd0=" << (int)cmd0 << "/" << (int)tid0
<< " cmd[idx=" << idx << "]=" << (int)cmdIdx << "/" << (int)tidIdx
<< " lastId=" << lastSchedId << std::endl;
++schedLog;
}
static uint64_t lastDma = 0, lastGif = 0, lastGs = 0, lastVif = 0;
uint64_t curDma = m_memory.dmaStartCount();
uint64_t curGif = m_memory.gifCopyCount();
uint64_t curGs = m_memory.gsWriteCount();
uint64_t curVif = m_memory.vifWriteCount();
if (curDma != lastDma || curGif != lastGif || curGs != lastGs || curVif != lastVif)
{
std::cout << "[hw] dma_starts=" << curDma
<< " gif_copies=" << curGif
<< " gs_writes=" << curGs
<< " vif_writes=" << curVif << std::endl;
lastDma = curDma;
lastGif = curGif;
lastGs = curGs;
lastVif = curVif;
}
}
// Kick the scheduler semaphore (stored at 0x363a10) to simulate VBlank-style ticks.
{
uint32_t schedId = *reinterpret_cast<uint32_t *>(m_memory.getRDRAM() + (0x363a10 & PS2_RAM_MASK));
if (schedId && schedId < 0x1000)
{
lastSchedId = schedId;
}
if (schedId == 0 && lastSchedId != 0)
{
schedId = lastSchedId; // fall back to the last seen non-zero id
}
else if (schedId >= 0x1000 && lastSchedId != 0)
{
// Ignore obviously bogus ids that are likely other data scribbling over the struct.
schedId = lastSchedId;
}
if (schedId)
{
R5900Context semaCtx{};
R5900Context *semaCtxPtr = &semaCtx;
SET_GPR_U32(semaCtxPtr, 4, schedId);
ps2_syscalls::SignalSema(m_memory.getRDRAM(), semaCtxPtr, this);
}
}
UploadFrame(frameTex, this);
@@ -344,7 +846,6 @@ void PS2Runtime::run()
if (g_activeThreads.load(std::memory_order_relaxed) == 0)
{
// Game thread finished on its own
if (gameThread.joinable())
{
gameThread.join();
@@ -352,7 +853,7 @@ void PS2Runtime::run()
}
else
{
// Window was closed while the game thread is still running
if (gameThread.joinable())
{
gameThread.detach();
File diff suppressed because it is too large Load Diff