otimizaçoes round 1

otimizaçoes round 1
This commit is contained in:
Jessica_Natalia
2026-08-16 00:48:31 -03:00
parent 3274d0c8f6
commit 818261a985
5 changed files with 75 additions and 25 deletions
+9 -1
View File
@@ -20,7 +20,15 @@ endif()
# Full /Ob3 is valuable at runtime but several large, cold units still make
# MSVC spend minutes and close to a gigabyte per compiler process. Keep those
# units buildable, then override this level only for the measured hot corpus.
set(PSPRECOMP_GENERATED_INLINE_LEVEL "0" CACHE STRING
# /Ob0 does not merely decline to auto-inline: MSVC drops __forceinline with it
# too, verified by compiling a stand-in accessor at each level and reading the
# disassembly -- /Ob0 emits a call to the accessor itself, /Ob1 and above emit a
# call only to its cold fallback. Every guest load and store in the recompiled
# corpus goes through a __forceinline fast path, so /Ob0 turned the single most
# frequent operation in the program into an out-of-line call. /Ob1 restores it
# while still refusing the aggressive auto-inlining that made /Ob3 blow up the
# optimizer on the larger units.
set(PSPRECOMP_GENERATED_INLINE_LEVEL "1" CACHE STRING
"MSVC /Ob level for generated VCS AOT units (0, 1, 2 or 3)")
set_property(CACHE PSPRECOMP_GENERATED_INLINE_LEVEL PROPERTY STRINGS 0 1 2 3)
if(NOT PSPRECOMP_GENERATED_INLINE_LEVEL MATCHES "^[0-3]$")
+34 -11
View File
@@ -512,7 +512,9 @@ Dx12UploadVertex make_upload_vertex(const GeGpuVertex &source) noexcept {
bool native_indexed_draw_enabled() noexcept {
static const bool enabled = [] {
const char *text = std::getenv("PSPRECOMP_DX12_NATIVE_INDEXED_DRAW");
if (text == nullptr || *text == '\0') return false;
// Default on: the production GE probe covers it on every build and the
// launcher scripts had been enabling it by hand. =0 restores the old path.
if (text == nullptr || *text == '\0') return true;
return std::strcmp(text, "0") != 0 &&
std::strcmp(text, "false") != 0 && std::strcmp(text, "FALSE") != 0 &&
std::strcmp(text, "off") != 0 && std::strcmp(text, "OFF") != 0;
@@ -2674,18 +2676,39 @@ bool prepare_texture_upload(Dx12GeState &s, const GeGpuDrawDescriptor &draw,
}
const std::uint32_t entry_limit = vcs_configuration().rendering.texture_cache_entries;
const std::uint64_t byte_limit = static_cast<std::uint64_t>(vcs_configuration().rendering.texture_cache_mb) * 1024ull * 1024ull;
// Eviction used to rescan the whole cache for a single victim, so freeing k
// textures walked k*n unordered_map nodes. Once the cache is full -- which a
// streaming city reaches and then stays at -- every upload paid a scan of up
// to TextureCacheEntries nodes, with the pointer-chasing locality that implies.
//
// One pass now collects a batch of the coldest entries, and the loop spends
// that batch before scanning again. Same LRU victims, amortized over many
// evictions instead of repeated per eviction.
constexpr std::size_t kVictimBatch = 64u;
std::vector<std::pair<std::uint64_t, std::uint64_t>> victims; // epoch, key
while (s.textures.size() >= entry_limit || s.texture_cache_bytes + packed.size() > byte_limit) {
auto victim = s.textures.end();
for (auto it = s.textures.begin(); it != s.textures.end(); ++it) {
if (it->second.last_used_epoch == s.frame_epoch) continue;
if (victim == s.textures.end() ||
it->second.last_used_epoch < victim->second.last_used_epoch)
victim = it;
}
if (victim == s.textures.end()) {
++s.report.rejected_texture_decodes;
return false;
if (victims.empty()) {
for (const auto &[key, texture] : s.textures) {
if (texture.last_used_epoch == s.frame_epoch) continue;
victims.emplace_back(texture.last_used_epoch, key);
}
if (victims.empty()) {
++s.report.rejected_texture_decodes;
return false;
}
// Coldest first, and only the batch actually needed is ordered.
const std::size_t keep = std::min(kVictimBatch, victims.size());
std::partial_sort(victims.begin(), victims.begin() + keep, victims.end());
victims.resize(keep);
std::reverse(victims.begin(), victims.end()); // pop_back takes the coldest
}
const std::uint64_t key = victims.back().second;
victims.pop_back();
const auto victim = s.textures.find(key);
// A candidate can be touched or replaced between passes, so re-check
// rather than trusting the snapshot.
if (victim == s.textures.end() || victim->second.last_used_epoch == s.frame_epoch)
continue;
for (Dx12FrameResources &retire : s.frames)
retire.transient_resources.push_back(victim->second.image);
retire_texture_srv(s, victim->second.srv_index);
+22 -12
View File
@@ -225,11 +225,17 @@ bool gpu_hardware_transform_enabled() noexcept {
// GPU simply rasterizes the back faces it is given -- a fill-rate cost on a
// discrete GPU that measured well below the win from the hybrid transform.
//
// PSPRECOMP_GE_GPU_HW_CULL=1 re-enables it for whoever debugs it next.
// Now on by default: measured on real gameplay it is part of the configuration
// that runs fastest on this profile, and the launcher scripts had been setting
// it by hand ever since. PSPRECOMP_GE_GPU_HW_CULL=0 turns it back off for a
// compatibility bisect.
bool gpu_hardware_cull_enabled() noexcept {
static const bool enabled = [] {
const char *text = std::getenv("PSPRECOMP_GE_GPU_HW_CULL");
return text != nullptr && *text != '\0' && std::strcmp(text, "0") != 0;
if (text == nullptr || *text == '\0') return true;
return std::strcmp(text, "0") != 0 &&
std::strcmp(text, "false") != 0 && std::strcmp(text, "FALSE") != 0 &&
std::strcmp(text, "off") != 0 && std::strcmp(text, "OFF") != 0;
}();
return enabled;
}
@@ -283,11 +289,13 @@ bool packed_0115_gpu_decode_enabled() noexcept {
}
bool direct_nonindexed_gpu_draw_enabled() noexcept {
// Stage 43 vkCmdDraw fast path is also isolated behind an explicit switch
// while the crash fix is validated on the user's physical driver.
// The crash fix this was gated behind has since been validated on the
// user's physical driver, and the production GE probe in the build script
// exercises it on every build. Default on; =0 restores the old path.
static const bool enabled = [] {
const char *text = std::getenv("PSPRECOMP_GE_DIRECT_NONINDEXED_DRAW");
return text != nullptr && *text != '\0' && std::strcmp(text, "0") != 0 &&
if (text == nullptr || *text == '\0') return true;
return std::strcmp(text, "0") != 0 &&
std::strcmp(text, "false") != 0 && std::strcmp(text, "FALSE") != 0 &&
std::strcmp(text, "off") != 0 && std::strcmp(text, "OFF") != 0;
}();
@@ -438,13 +446,15 @@ private:
}
}
if (!explicitly_configured) {
// The guest Allegrex stream is intentionally serial, but host-side
// decode/raster work is not. Stage 39 capped this pool at eight
// participants, leaving a large part of 12/16/20/24-thread desktop CPUs
// idle exactly during texture-streaming spikes. Use every logical
// CPU up to the pool's conservative hard limit; the caller is one of
// the participants, so this creates at most kMaxThreads-1 workers.
requested = std::min(requested, kMaxThreads);
// "One worker per logical CPU" is the wrong default on a hybrid
// desktop part. hardware_concurrency() counts E-cores and SMT
// siblings, so on a 14600K it asks for 20 participants for work
// that is latency-sensitive and shares cache with the serial
// Allegrex stream -- the launcher scripts had been overriding it
// to 4 by hand, measured faster. Four is now the default; the
// env var still takes anything from 1 to kMaxThreads.
constexpr unsigned kDefaultWorkers = 4u;
requested = std::min(std::min(requested, kDefaultWorkers), kMaxThreads);
}
worker_count_ = std::max(1u, std::min(requested, kMaxThreads));
if (worker_count_ <= 1u) return;
+1 -1
View File
@@ -122,7 +122,7 @@ echo [1/7] Configuring persistent Ninja Release tree...
-DPSPRECOMP_MSVC_MP_JOBS=1 ^
-DPSPRECOMP_PROFILE_GUIDED_AOT=ON ^
-DPSPRECOMP_HOT_GENERATED_OPT_LEVEL=3 ^
-DPSPRECOMP_GENERATED_INLINE_LEVEL=0 ^
-DPSPRECOMP_GENERATED_INLINE_LEVEL=1 ^
-DPSPRECOMP_HOT_GENERATED_INLINE_LEVEL=3 ^
-DPSPRECOMP_VCS_AOT_LTO=OFF ^
-DPSPRECOMP_BUILD_TESTS=ON ^
+9
View File
@@ -150,6 +150,15 @@ Runtime::Runtime(std::uint32_t ram_size) : memory_(ram_size) {
#endif
// PSPRECOMP_NO_CHAIN disables cross-unit chaining outright;
// PSPRECOMP_CHAIN_DEPTH tunes how deep it may nest without a rebuild.
//
// Do not raise this without measuring native stack use per chained frame.
// 1024 was tried on the theory that exceeding the limit is a cliff -- the
// chain refuses, the native stack unwinds to Runtime::run, and the target is
// re-entered through the guest-PC table -- and that bench.bat already passed
// 1024 by hand. It crashed the game during boot. A benchmark does not reach
// the call depths gameplay does, and generated frames are not small: each
// one carries the unit's locals, and inlining the memory fast path at every
// call site (/Ob1) made them larger still. 48 is the value that runs.
chain_depth_limit_ = 48u;
if (const char *depth = std::getenv("PSPRECOMP_CHAIN_DEPTH")) {
char *end = nullptr;