mirror of
https://github.com/jessicanataliagta/PSPRecomp
synced 2026-10-05 19:09:42 -04:00
otimizaçoes round 1
otimizaçoes round 1
This commit is contained in:
@@ -20,7 +20,15 @@ endif()
|
||||
# Full /Ob3 is valuable at runtime but several large, cold units still make
|
||||
# MSVC spend minutes and close to a gigabyte per compiler process. Keep those
|
||||
# units buildable, then override this level only for the measured hot corpus.
|
||||
set(PSPRECOMP_GENERATED_INLINE_LEVEL "0" CACHE STRING
|
||||
# /Ob0 does not merely decline to auto-inline: MSVC drops __forceinline with it
|
||||
# too, verified by compiling a stand-in accessor at each level and reading the
|
||||
# disassembly -- /Ob0 emits a call to the accessor itself, /Ob1 and above emit a
|
||||
# call only to its cold fallback. Every guest load and store in the recompiled
|
||||
# corpus goes through a __forceinline fast path, so /Ob0 turned the single most
|
||||
# frequent operation in the program into an out-of-line call. /Ob1 restores it
|
||||
# while still refusing the aggressive auto-inlining that made /Ob3 blow up the
|
||||
# optimizer on the larger units.
|
||||
set(PSPRECOMP_GENERATED_INLINE_LEVEL "1" CACHE STRING
|
||||
"MSVC /Ob level for generated VCS AOT units (0, 1, 2 or 3)")
|
||||
set_property(CACHE PSPRECOMP_GENERATED_INLINE_LEVEL PROPERTY STRINGS 0 1 2 3)
|
||||
if(NOT PSPRECOMP_GENERATED_INLINE_LEVEL MATCHES "^[0-3]$")
|
||||
|
||||
@@ -512,7 +512,9 @@ Dx12UploadVertex make_upload_vertex(const GeGpuVertex &source) noexcept {
|
||||
bool native_indexed_draw_enabled() noexcept {
|
||||
static const bool enabled = [] {
|
||||
const char *text = std::getenv("PSPRECOMP_DX12_NATIVE_INDEXED_DRAW");
|
||||
if (text == nullptr || *text == '\0') return false;
|
||||
// Default on: the production GE probe covers it on every build and the
|
||||
// launcher scripts had been enabling it by hand. =0 restores the old path.
|
||||
if (text == nullptr || *text == '\0') return true;
|
||||
return std::strcmp(text, "0") != 0 &&
|
||||
std::strcmp(text, "false") != 0 && std::strcmp(text, "FALSE") != 0 &&
|
||||
std::strcmp(text, "off") != 0 && std::strcmp(text, "OFF") != 0;
|
||||
@@ -2674,18 +2676,39 @@ bool prepare_texture_upload(Dx12GeState &s, const GeGpuDrawDescriptor &draw,
|
||||
}
|
||||
const std::uint32_t entry_limit = vcs_configuration().rendering.texture_cache_entries;
|
||||
const std::uint64_t byte_limit = static_cast<std::uint64_t>(vcs_configuration().rendering.texture_cache_mb) * 1024ull * 1024ull;
|
||||
// Eviction used to rescan the whole cache for a single victim, so freeing k
|
||||
// textures walked k*n unordered_map nodes. Once the cache is full -- which a
|
||||
// streaming city reaches and then stays at -- every upload paid a scan of up
|
||||
// to TextureCacheEntries nodes, with the pointer-chasing locality that implies.
|
||||
//
|
||||
// One pass now collects a batch of the coldest entries, and the loop spends
|
||||
// that batch before scanning again. Same LRU victims, amortized over many
|
||||
// evictions instead of repeated per eviction.
|
||||
constexpr std::size_t kVictimBatch = 64u;
|
||||
std::vector<std::pair<std::uint64_t, std::uint64_t>> victims; // epoch, key
|
||||
while (s.textures.size() >= entry_limit || s.texture_cache_bytes + packed.size() > byte_limit) {
|
||||
auto victim = s.textures.end();
|
||||
for (auto it = s.textures.begin(); it != s.textures.end(); ++it) {
|
||||
if (it->second.last_used_epoch == s.frame_epoch) continue;
|
||||
if (victim == s.textures.end() ||
|
||||
it->second.last_used_epoch < victim->second.last_used_epoch)
|
||||
victim = it;
|
||||
}
|
||||
if (victim == s.textures.end()) {
|
||||
++s.report.rejected_texture_decodes;
|
||||
return false;
|
||||
if (victims.empty()) {
|
||||
for (const auto &[key, texture] : s.textures) {
|
||||
if (texture.last_used_epoch == s.frame_epoch) continue;
|
||||
victims.emplace_back(texture.last_used_epoch, key);
|
||||
}
|
||||
if (victims.empty()) {
|
||||
++s.report.rejected_texture_decodes;
|
||||
return false;
|
||||
}
|
||||
// Coldest first, and only the batch actually needed is ordered.
|
||||
const std::size_t keep = std::min(kVictimBatch, victims.size());
|
||||
std::partial_sort(victims.begin(), victims.begin() + keep, victims.end());
|
||||
victims.resize(keep);
|
||||
std::reverse(victims.begin(), victims.end()); // pop_back takes the coldest
|
||||
}
|
||||
const std::uint64_t key = victims.back().second;
|
||||
victims.pop_back();
|
||||
const auto victim = s.textures.find(key);
|
||||
// A candidate can be touched or replaced between passes, so re-check
|
||||
// rather than trusting the snapshot.
|
||||
if (victim == s.textures.end() || victim->second.last_used_epoch == s.frame_epoch)
|
||||
continue;
|
||||
for (Dx12FrameResources &retire : s.frames)
|
||||
retire.transient_resources.push_back(victim->second.image);
|
||||
retire_texture_srv(s, victim->second.srv_index);
|
||||
|
||||
@@ -225,11 +225,17 @@ bool gpu_hardware_transform_enabled() noexcept {
|
||||
// GPU simply rasterizes the back faces it is given -- a fill-rate cost on a
|
||||
// discrete GPU that measured well below the win from the hybrid transform.
|
||||
//
|
||||
// PSPRECOMP_GE_GPU_HW_CULL=1 re-enables it for whoever debugs it next.
|
||||
// Now on by default: measured on real gameplay it is part of the configuration
|
||||
// that runs fastest on this profile, and the launcher scripts had been setting
|
||||
// it by hand ever since. PSPRECOMP_GE_GPU_HW_CULL=0 turns it back off for a
|
||||
// compatibility bisect.
|
||||
bool gpu_hardware_cull_enabled() noexcept {
|
||||
static const bool enabled = [] {
|
||||
const char *text = std::getenv("PSPRECOMP_GE_GPU_HW_CULL");
|
||||
return text != nullptr && *text != '\0' && std::strcmp(text, "0") != 0;
|
||||
if (text == nullptr || *text == '\0') return true;
|
||||
return std::strcmp(text, "0") != 0 &&
|
||||
std::strcmp(text, "false") != 0 && std::strcmp(text, "FALSE") != 0 &&
|
||||
std::strcmp(text, "off") != 0 && std::strcmp(text, "OFF") != 0;
|
||||
}();
|
||||
return enabled;
|
||||
}
|
||||
@@ -283,11 +289,13 @@ bool packed_0115_gpu_decode_enabled() noexcept {
|
||||
}
|
||||
|
||||
bool direct_nonindexed_gpu_draw_enabled() noexcept {
|
||||
// Stage 43 vkCmdDraw fast path is also isolated behind an explicit switch
|
||||
// while the crash fix is validated on the user's physical driver.
|
||||
// The crash fix this was gated behind has since been validated on the
|
||||
// user's physical driver, and the production GE probe in the build script
|
||||
// exercises it on every build. Default on; =0 restores the old path.
|
||||
static const bool enabled = [] {
|
||||
const char *text = std::getenv("PSPRECOMP_GE_DIRECT_NONINDEXED_DRAW");
|
||||
return text != nullptr && *text != '\0' && std::strcmp(text, "0") != 0 &&
|
||||
if (text == nullptr || *text == '\0') return true;
|
||||
return std::strcmp(text, "0") != 0 &&
|
||||
std::strcmp(text, "false") != 0 && std::strcmp(text, "FALSE") != 0 &&
|
||||
std::strcmp(text, "off") != 0 && std::strcmp(text, "OFF") != 0;
|
||||
}();
|
||||
@@ -438,13 +446,15 @@ private:
|
||||
}
|
||||
}
|
||||
if (!explicitly_configured) {
|
||||
// The guest Allegrex stream is intentionally serial, but host-side
|
||||
// decode/raster work is not. Stage 39 capped this pool at eight
|
||||
// participants, leaving a large part of 12/16/20/24-thread desktop CPUs
|
||||
// idle exactly during texture-streaming spikes. Use every logical
|
||||
// CPU up to the pool's conservative hard limit; the caller is one of
|
||||
// the participants, so this creates at most kMaxThreads-1 workers.
|
||||
requested = std::min(requested, kMaxThreads);
|
||||
// "One worker per logical CPU" is the wrong default on a hybrid
|
||||
// desktop part. hardware_concurrency() counts E-cores and SMT
|
||||
// siblings, so on a 14600K it asks for 20 participants for work
|
||||
// that is latency-sensitive and shares cache with the serial
|
||||
// Allegrex stream -- the launcher scripts had been overriding it
|
||||
// to 4 by hand, measured faster. Four is now the default; the
|
||||
// env var still takes anything from 1 to kMaxThreads.
|
||||
constexpr unsigned kDefaultWorkers = 4u;
|
||||
requested = std::min(std::min(requested, kDefaultWorkers), kMaxThreads);
|
||||
}
|
||||
worker_count_ = std::max(1u, std::min(requested, kMaxThreads));
|
||||
if (worker_count_ <= 1u) return;
|
||||
|
||||
@@ -122,7 +122,7 @@ echo [1/7] Configuring persistent Ninja Release tree...
|
||||
-DPSPRECOMP_MSVC_MP_JOBS=1 ^
|
||||
-DPSPRECOMP_PROFILE_GUIDED_AOT=ON ^
|
||||
-DPSPRECOMP_HOT_GENERATED_OPT_LEVEL=3 ^
|
||||
-DPSPRECOMP_GENERATED_INLINE_LEVEL=0 ^
|
||||
-DPSPRECOMP_GENERATED_INLINE_LEVEL=1 ^
|
||||
-DPSPRECOMP_HOT_GENERATED_INLINE_LEVEL=3 ^
|
||||
-DPSPRECOMP_VCS_AOT_LTO=OFF ^
|
||||
-DPSPRECOMP_BUILD_TESTS=ON ^
|
||||
|
||||
@@ -150,6 +150,15 @@ Runtime::Runtime(std::uint32_t ram_size) : memory_(ram_size) {
|
||||
#endif
|
||||
// PSPRECOMP_NO_CHAIN disables cross-unit chaining outright;
|
||||
// PSPRECOMP_CHAIN_DEPTH tunes how deep it may nest without a rebuild.
|
||||
//
|
||||
// Do not raise this without measuring native stack use per chained frame.
|
||||
// 1024 was tried on the theory that exceeding the limit is a cliff -- the
|
||||
// chain refuses, the native stack unwinds to Runtime::run, and the target is
|
||||
// re-entered through the guest-PC table -- and that bench.bat already passed
|
||||
// 1024 by hand. It crashed the game during boot. A benchmark does not reach
|
||||
// the call depths gameplay does, and generated frames are not small: each
|
||||
// one carries the unit's locals, and inlining the memory fast path at every
|
||||
// call site (/Ob1) made them larger still. 48 is the value that runs.
|
||||
chain_depth_limit_ = 48u;
|
||||
if (const char *depth = std::getenv("PSPRECOMP_CHAIN_DEPTH")) {
|
||||
char *end = nullptr;
|
||||
|
||||
Reference in New Issue
Block a user