diff --git a/profiles/vcs/CMakeLists.txt b/profiles/vcs/CMakeLists.txt index be6bf01..75f2225 100644 --- a/profiles/vcs/CMakeLists.txt +++ b/profiles/vcs/CMakeLists.txt @@ -20,7 +20,15 @@ endif() # Full /Ob3 is valuable at runtime but several large, cold units still make # MSVC spend minutes and close to a gigabyte per compiler process. Keep those # units buildable, then override this level only for the measured hot corpus. -set(PSPRECOMP_GENERATED_INLINE_LEVEL "0" CACHE STRING +# /Ob0 does not merely decline to auto-inline: MSVC drops __forceinline with it +# too, verified by compiling a stand-in accessor at each level and reading the +# disassembly -- /Ob0 emits a call to the accessor itself, /Ob1 and above emit a +# call only to its cold fallback. Every guest load and store in the recompiled +# corpus goes through a __forceinline fast path, so /Ob0 turned the single most +# frequent operation in the program into an out-of-line call. /Ob1 restores it +# while still refusing the aggressive auto-inlining that made /Ob3 blow up the +# optimizer on the larger units. +set(PSPRECOMP_GENERATED_INLINE_LEVEL "1" CACHE STRING "MSVC /Ob level for generated VCS AOT units (0, 1, 2 or 3)") set_property(CACHE PSPRECOMP_GENERATED_INLINE_LEVEL PROPERTY STRINGS 0 1 2 3) if(NOT PSPRECOMP_GENERATED_INLINE_LEVEL MATCHES "^[0-3]$") diff --git a/profiles/vcs/host/ge_gpu_backend_dx12.cpp b/profiles/vcs/host/ge_gpu_backend_dx12.cpp index 648a539..d033615 100644 --- a/profiles/vcs/host/ge_gpu_backend_dx12.cpp +++ b/profiles/vcs/host/ge_gpu_backend_dx12.cpp @@ -512,7 +512,9 @@ Dx12UploadVertex make_upload_vertex(const GeGpuVertex &source) noexcept { bool native_indexed_draw_enabled() noexcept { static const bool enabled = [] { const char *text = std::getenv("PSPRECOMP_DX12_NATIVE_INDEXED_DRAW"); - if (text == nullptr || *text == '\0') return false; + // Default on: the production GE probe covers it on every build and the + // launcher scripts had been enabling it by hand. =0 restores the old path. + if (text == nullptr || *text == '\0') return true; return std::strcmp(text, "0") != 0 && std::strcmp(text, "false") != 0 && std::strcmp(text, "FALSE") != 0 && std::strcmp(text, "off") != 0 && std::strcmp(text, "OFF") != 0; @@ -2674,18 +2676,39 @@ bool prepare_texture_upload(Dx12GeState &s, const GeGpuDrawDescriptor &draw, } const std::uint32_t entry_limit = vcs_configuration().rendering.texture_cache_entries; const std::uint64_t byte_limit = static_cast(vcs_configuration().rendering.texture_cache_mb) * 1024ull * 1024ull; + // Eviction used to rescan the whole cache for a single victim, so freeing k + // textures walked k*n unordered_map nodes. Once the cache is full -- which a + // streaming city reaches and then stays at -- every upload paid a scan of up + // to TextureCacheEntries nodes, with the pointer-chasing locality that implies. + // + // One pass now collects a batch of the coldest entries, and the loop spends + // that batch before scanning again. Same LRU victims, amortized over many + // evictions instead of repeated per eviction. + constexpr std::size_t kVictimBatch = 64u; + std::vector> victims; // epoch, key while (s.textures.size() >= entry_limit || s.texture_cache_bytes + packed.size() > byte_limit) { - auto victim = s.textures.end(); - for (auto it = s.textures.begin(); it != s.textures.end(); ++it) { - if (it->second.last_used_epoch == s.frame_epoch) continue; - if (victim == s.textures.end() || - it->second.last_used_epoch < victim->second.last_used_epoch) - victim = it; - } - if (victim == s.textures.end()) { - ++s.report.rejected_texture_decodes; - return false; + if (victims.empty()) { + for (const auto &[key, texture] : s.textures) { + if (texture.last_used_epoch == s.frame_epoch) continue; + victims.emplace_back(texture.last_used_epoch, key); + } + if (victims.empty()) { + ++s.report.rejected_texture_decodes; + return false; + } + // Coldest first, and only the batch actually needed is ordered. + const std::size_t keep = std::min(kVictimBatch, victims.size()); + std::partial_sort(victims.begin(), victims.begin() + keep, victims.end()); + victims.resize(keep); + std::reverse(victims.begin(), victims.end()); // pop_back takes the coldest } + const std::uint64_t key = victims.back().second; + victims.pop_back(); + const auto victim = s.textures.find(key); + // A candidate can be touched or replaced between passes, so re-check + // rather than trusting the snapshot. + if (victim == s.textures.end() || victim->second.last_used_epoch == s.frame_epoch) + continue; for (Dx12FrameResources &retire : s.frames) retire.transient_resources.push_back(victim->second.image); retire_texture_srv(s, victim->second.srv_index); diff --git a/profiles/vcs/host/ge_renderer.cpp b/profiles/vcs/host/ge_renderer.cpp index da636f6..3391a91 100644 --- a/profiles/vcs/host/ge_renderer.cpp +++ b/profiles/vcs/host/ge_renderer.cpp @@ -225,11 +225,17 @@ bool gpu_hardware_transform_enabled() noexcept { // GPU simply rasterizes the back faces it is given -- a fill-rate cost on a // discrete GPU that measured well below the win from the hybrid transform. // -// PSPRECOMP_GE_GPU_HW_CULL=1 re-enables it for whoever debugs it next. +// Now on by default: measured on real gameplay it is part of the configuration +// that runs fastest on this profile, and the launcher scripts had been setting +// it by hand ever since. PSPRECOMP_GE_GPU_HW_CULL=0 turns it back off for a +// compatibility bisect. bool gpu_hardware_cull_enabled() noexcept { static const bool enabled = [] { const char *text = std::getenv("PSPRECOMP_GE_GPU_HW_CULL"); - return text != nullptr && *text != '\0' && std::strcmp(text, "0") != 0; + if (text == nullptr || *text == '\0') return true; + return std::strcmp(text, "0") != 0 && + std::strcmp(text, "false") != 0 && std::strcmp(text, "FALSE") != 0 && + std::strcmp(text, "off") != 0 && std::strcmp(text, "OFF") != 0; }(); return enabled; } @@ -283,11 +289,13 @@ bool packed_0115_gpu_decode_enabled() noexcept { } bool direct_nonindexed_gpu_draw_enabled() noexcept { - // Stage 43 vkCmdDraw fast path is also isolated behind an explicit switch - // while the crash fix is validated on the user's physical driver. + // The crash fix this was gated behind has since been validated on the + // user's physical driver, and the production GE probe in the build script + // exercises it on every build. Default on; =0 restores the old path. static const bool enabled = [] { const char *text = std::getenv("PSPRECOMP_GE_DIRECT_NONINDEXED_DRAW"); - return text != nullptr && *text != '\0' && std::strcmp(text, "0") != 0 && + if (text == nullptr || *text == '\0') return true; + return std::strcmp(text, "0") != 0 && std::strcmp(text, "false") != 0 && std::strcmp(text, "FALSE") != 0 && std::strcmp(text, "off") != 0 && std::strcmp(text, "OFF") != 0; }(); @@ -438,13 +446,15 @@ private: } } if (!explicitly_configured) { - // The guest Allegrex stream is intentionally serial, but host-side - // decode/raster work is not. Stage 39 capped this pool at eight - // participants, leaving a large part of 12/16/20/24-thread desktop CPUs - // idle exactly during texture-streaming spikes. Use every logical - // CPU up to the pool's conservative hard limit; the caller is one of - // the participants, so this creates at most kMaxThreads-1 workers. - requested = std::min(requested, kMaxThreads); + // "One worker per logical CPU" is the wrong default on a hybrid + // desktop part. hardware_concurrency() counts E-cores and SMT + // siblings, so on a 14600K it asks for 20 participants for work + // that is latency-sensitive and shares cache with the serial + // Allegrex stream -- the launcher scripts had been overriding it + // to 4 by hand, measured faster. Four is now the default; the + // env var still takes anything from 1 to kMaxThreads. + constexpr unsigned kDefaultWorkers = 4u; + requested = std::min(std::min(requested, kDefaultWorkers), kMaxThreads); } worker_count_ = std::max(1u, std::min(requested, kMaxThreads)); if (worker_count_ <= 1u) return; diff --git a/profiles/vcs/scripts/build_release_ninja.bat b/profiles/vcs/scripts/build_release_ninja.bat index aa1f120..3d850f8 100644 --- a/profiles/vcs/scripts/build_release_ninja.bat +++ b/profiles/vcs/scripts/build_release_ninja.bat @@ -122,7 +122,7 @@ echo [1/7] Configuring persistent Ninja Release tree... -DPSPRECOMP_MSVC_MP_JOBS=1 ^ -DPSPRECOMP_PROFILE_GUIDED_AOT=ON ^ -DPSPRECOMP_HOT_GENERATED_OPT_LEVEL=3 ^ - -DPSPRECOMP_GENERATED_INLINE_LEVEL=0 ^ + -DPSPRECOMP_GENERATED_INLINE_LEVEL=1 ^ -DPSPRECOMP_HOT_GENERATED_INLINE_LEVEL=3 ^ -DPSPRECOMP_VCS_AOT_LTO=OFF ^ -DPSPRECOMP_BUILD_TESTS=ON ^ diff --git a/src/runtime.cpp b/src/runtime.cpp index c4c9d99..693fedb 100644 --- a/src/runtime.cpp +++ b/src/runtime.cpp @@ -150,6 +150,15 @@ Runtime::Runtime(std::uint32_t ram_size) : memory_(ram_size) { #endif // PSPRECOMP_NO_CHAIN disables cross-unit chaining outright; // PSPRECOMP_CHAIN_DEPTH tunes how deep it may nest without a rebuild. + // + // Do not raise this without measuring native stack use per chained frame. + // 1024 was tried on the theory that exceeding the limit is a cliff -- the + // chain refuses, the native stack unwinds to Runtime::run, and the target is + // re-entered through the guest-PC table -- and that bench.bat already passed + // 1024 by hand. It crashed the game during boot. A benchmark does not reach + // the call depths gameplay does, and generated frames are not small: each + // one carries the unit's locals, and inlining the memory fast path at every + // call site (/Ob1) made them larger still. 48 is the value that runs. chain_depth_limit_ = 48u; if (const char *depth = std::getenv("PSPRECOMP_CHAIN_DEPTH")) { char *end = nullptr;