From 4f1c727356f57cd02d460647b23c692c96c6bcab Mon Sep 17 00:00:00 2001 From: Jessica_Natalia Date: Mon, 17 Aug 2026 15:32:21 -0300 Subject: [PATCH] =?UTF-8?q?otimiza=C3=A7=C3=B5es=20round=208?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit otimizações round 8 --- ..._V5_ASYNC_PRESENTFIX_HANDOFF_2026-08-17.md | 69 +++++++ ..._V5_ASYNC_PRESENTFIX_STATS_2026-08-17.json | 33 ++++ ...S_PERF_V5_ASYNC_VFPU_HANDOFF_2026-08-16.md | 115 ++++++++++++ ...S_PERF_V5_ASYNC_VFPU_STATS_2026-08-16.json | 66 +++++++ profiles/vcs/host/ge_gpu_backend_dx12.cpp | 8 +- profiles/vcs/host/vcs_config.cpp | 8 + profiles/vcs/host/vcs_profile.cpp | 6 +- profiles/vcs/host/vcs_runtime_log.cpp | 8 +- profiles/vcs/host/vcs_vfpu_fast.hpp | 173 ++++++++++++++++++ profiles/vcs/scripts/bench.bat | 4 +- profiles/vcs/scripts/build_release_ninja.bat | 69 ++++++- profiles/vcs/scripts/play.bat | 2 +- profiles/vcs/tests/vcs_config_tests.cpp | 7 + 13 files changed, 558 insertions(+), 10 deletions(-) create mode 100644 docs/VCS_PERF_V5_ASYNC_PRESENTFIX_HANDOFF_2026-08-17.md create mode 100644 docs/VCS_PERF_V5_ASYNC_PRESENTFIX_STATS_2026-08-17.json create mode 100644 docs/VCS_PERF_V5_ASYNC_VFPU_HANDOFF_2026-08-16.md create mode 100644 docs/VCS_PERF_V5_ASYNC_VFPU_STATS_2026-08-16.json create mode 100644 profiles/vcs/host/vcs_vfpu_fast.hpp diff --git a/docs/VCS_PERF_V5_ASYNC_PRESENTFIX_HANDOFF_2026-08-17.md b/docs/VCS_PERF_V5_ASYNC_PRESENTFIX_HANDOFF_2026-08-17.md new file mode 100644 index 0000000..5bcd9a8 --- /dev/null +++ b/docs/VCS_PERF_V5_ASYNC_PRESENTFIX_HANDOFF_2026-08-17.md @@ -0,0 +1,69 @@ +# VCS PERF V5 ASYNC PRESENTFIX — 2026-08-17 + +## Base +Built directly on **V5 ASYNC/VFPU STALLFIX**. All V5 performance work remains enabled: asynchronous GE, parallel vertex decode, 95 VFPU default-prefix fast paths, native block-load leaf optimization, Tier-2 V4 dataflow/SIMD/GPR-shadow work, and AMD/UMA compatibility. + +## Symptom +After the intro videos the game reached the gameplay framebuffer/swapchain creation and then remained black without a crash. The runtime log stopped before any gameplay GE/draw telemetry appeared. + +## Root cause +V5 made `sceDisplayWaitVblank*` call `ge_async_wait_idle()`. That is too strong and is not PSP display semantics. VBlank is not a global GE DrawSync. A GE list may legitimately remain queued/stalled across a vblank and require the Allegrex thread to advance its stall address. Blocking the Allegrex thread in VBlank until `outstanding == 0` can therefore create a circular wait: + +1. Allegrex enters VBlank and waits for all async GE tasks to become idle. +2. GE reaches/depends on a stall transition that requires the guest to continue and update it. +3. Guest cannot update it because it is blocked in VBlank. +4. No crash occurs; the screen remains black indefinitely. + +The previous STALLFIX correctly fixed one worker/UpdateStall commit race but could not fix this architectural barrier. + +## PRESENTFIX +- Removed the global `ge_async_wait_idle()` call from the VBlank path. +- Added an explicit **presentation safe-point gate**. +- VBlank raises `presentation_requested`. +- The GE worker finishes only the segment it is currently executing, then pauses **between tasks/segments**. +- Presentation waits only for `worker_busy == false`, not `outstanding == 0`. +- While presentation owns the gate, the worker cannot start another segment, so `ge_gpu_backend_finish_color_frame()` and related backend presentation operations see coherent state. +- After presentation, the gate is released and queued/stalled work resumes. +- Existing stall-race recovery is preserved. + +This retains real CPU/GE overlap during the frame while avoiding both the deadlock and a renderer concurrency race during final frame presentation. + +## Telemetry +Runtime header now reports: + +`stage=perf-v5-async-presentfix-2026-08-17` + +and: + +`ge_async_stall_race_fix=1 ge_async_present_gate=1` + +Async shutdown census also reports `present_safe_points=`. + +## Build behavior +New one-shot stamp: + +`.vcs_perf_v5_async_presentfix_20260817` + +It invalidates only: +- `vcs_profile*.obj` +- `vcs_runtime_log*.obj` + +No Tier-2 cluster or DX12 backend object is intentionally invalidated. + +## Validation +- `psprecomp_tests`: PASS +- `vcs_config_tests`: PASS +- `vfpu_tier2_tests`: PASS +- `PSPRECOMP_GE_ASYNC=1 vcs_profile_tests`: PASS +- VCSNative target `vcs_profile.cpp` object: PASS +- VCSNative target `vcs_runtime_log.cpp` object: PASS + +A new regression test verifies that presentation can acquire a safe point even while `outstanding == 1`; this specifically prevents the VBlank/GE circular wait from returning. + +## Expected next test +Apply the overlay over the current V5 STALLFIX tree, rebuild, and boot normally with GE async left enabled. The critical result is whether gameplay proceeds past the first post-intro framebuffer. If it does, the resulting runtime log can finally be used to evaluate V5 performance rather than boot stability. + +## Progress estimate +- Overall project: ~91% +- Tier-2: ~95% +- >150 FPS performance-margin goal: ~62% (unchanged; this patch is a correctness fix for the performance architecture, not a measured FPS gain) diff --git a/docs/VCS_PERF_V5_ASYNC_PRESENTFIX_STATS_2026-08-17.json b/docs/VCS_PERF_V5_ASYNC_PRESENTFIX_STATS_2026-08-17.json new file mode 100644 index 0000000..4fedbf3 --- /dev/null +++ b/docs/VCS_PERF_V5_ASYNC_PRESENTFIX_STATS_2026-08-17.json @@ -0,0 +1,33 @@ +{ + "stage": "perf-v5-async-presentfix-2026-08-17", + "base": "perf-v5-async-vfpu-stallfix-2026-08-17", + "root_cause": "sceDisplayWaitVblank* incorrectly called ge_async_wait_idle(), creating a CPU<->GE deadlock when a display list required guest stall advancement across vblank", + "changes": { + "removed_global_ge_idle_barrier_from_vblank": true, + "presentation_safe_point_gate": true, + "worker_pauses_between_list_segments_for_present": true, + "stall_race_fix_preserved": true, + "ge_async_default": true, + "parallel_vertex_decode_default": true, + "vfpu_fast_paths_preserved": 95, + "tier2_hot_blocks_preserved": 1060, + "amd_uma_compat_preserved": true + }, + "validation": { + "psprecomp_tests": "PASS", + "vcs_config_tests": "PASS", + "vfpu_tier2_tests": "PASS", + "vcs_profile_tests_ge_async_1": "PASS", + "vcsnative_vcs_profile_object": "PASS", + "vcsnative_vcs_runtime_log_object": "PASS" + }, + "windows_incremental_rebuild_objects": [ + "vcs_profile.cpp.obj", + "vcs_runtime_log.cpp.obj" + ], + "progress_estimate": { + "overall_percent": 91, + "tier2_percent": 95, + "performance_150fps_goal_percent": 62 + } +} diff --git a/docs/VCS_PERF_V5_ASYNC_VFPU_HANDOFF_2026-08-16.md b/docs/VCS_PERF_V5_ASYNC_VFPU_HANDOFF_2026-08-16.md new file mode 100644 index 0000000..b27f17e --- /dev/null +++ b/docs/VCS_PERF_V5_ASYNC_VFPU_HANDOFF_2026-08-16.md @@ -0,0 +1,115 @@ +# VCSNative PERF V5 ASYNC/VFPU — Handoff — 2026-08-16 + +## Base + +This revision is based directly on **Tier2 V4 AMD/UMA COMPAT (BUILDFIX3 lineage)**. It preserves the AMD/UMA safety policy (aligned packed 0x0115 storage, conservative indexed/merge/indirect policy on AMD UMA, and the UMA MSAA guard). + +## Objective + +V4 telemetry showed that ExecuteIndirect removed thousands of Draw API calls without materially reducing `ge_us`, while heavy gameplay still spent roughly 10–15 ms in guest/AOT CPU and several milliseconds in GE work. V5 therefore stops optimizing Draw submission count and targets overlap/parallelism plus measured-hot VFPU overhead. + +## V5 changes + +### 1. GE async is the DX12 production default + +`vcs_config.cpp` installs `PSPRECOMP_GE_ASYNC=1` for DirectX 12 when the caller has not provided an explicit environment value. `play.bat` does the same without overriding a caller-provided value. + +The existing GE worker is not new experimental code: it already provides ordered task completion, list/idle waits at guest-visible synchronization points, fatal propagation, and a display/vblank visibility boundary. V5 promotes this path from opt-in to the normal DX12 path so Allegrex work can overlap GE list execution instead of paying guest + GE serially whenever the game does not immediately synchronize. + +Recovery/A-B switch: + +```bat +set PSPRECOMP_GE_ASYNC=0 +``` + +### 2. Parallel vertex decode is the DX12 production default + +`PSPRECOMP_GE_PARALLEL_VERTEX_DECODE=1` is installed by the config/play defaults unless explicitly overridden. The existing persistent worker pool remains bounded and only parallelizes sufficiently large CPU-decoded vertex work. Packed 0x0115 GPU-decode traffic remains on its native GPU path. + +Recovery/A-B switch: + +```bat +set PSPRECOMP_GE_PARALLEL_VERTEX_DECODE=0 +``` + +### 3. ExecuteIndirect is no longer enabled by default + +V4 runtime telemetry proved that it can save thousands of Draw calls while leaving GE wall time essentially unchanged. V5 therefore avoids building/uploading indirect argument records by default. The implementation remains available for A/B testing: + +```bat +set PSPRECOMP_DX12_EXECUTE_INDIRECT=1 +``` + +AMD/UMA safe mode still vetoes that aggressive path unless its separate diagnostic override is used. + +### 4. Measured-hot VFPU default-prefix lowering + +A new header, `profiles/vcs/host/vcs_vfpu_fast.hpp`, provides exact fast branches for the common architectural prefix state (S=T=0xE4, D=0) with the original CT helper as fallback for non-default prefixes. + +Final static transformed call sites: **95** + +- destination-prefix writes: 60 +- VSCL: 2 +- VDOT: 19 +- VCMP: 14 +- VCMOV: 0 + +The transformed clusters are Entity, Matrix, Physics, World and Edge43. Geometry and Boundary are intentionally byte-identical to V4. Geometry was kept unchanged because adding extra VFPU template specialization to that already-large TU reintroduced the compiler-time cliff; the runtime-critical Geometry V3/V4 dataflow + SIMD optimizations remain intact. + +No fast-math/FMA reassociation was introduced. Differential tests exercise default-prefix fast branches and randomized non-default-prefix fallback against the original helpers. + +### 5. Verified native VFPU leaf 0x088B1780 now uses block loads + +The already-registered native leaf at `0x088B1780` previously performed sixteen independent 32-bit loads for its first four vector inputs. V5 keeps the verified native lowering but uses four `aot_load32_block` operations instead, collapsing address canonicalization/bounds checking per 16-byte vector while preserving the original scalar fallback on non-RAM paths. The existing `PSPRECOMP_VALIDATE_FAST_088B1780` AOT-reference mechanism remains available for real-game differential validation. This optimization is local to `vcs_native_fast_paths.cpp`, so it does not force the large Tier-2 Geometry TU to rebuild. + +## Existing V4 optimization retained + +- 7 profile-guided Tier-2 clusters / 1,060 hot blocks +- 49 fused direct calls / 13 fused tails / 9 cold exits +- V3 memory/dataflow batching +- 4 ordered mat4 + 19 ordered matvec SIMD sites +- safe GPR shadow on Entity/Boundary/World/Edge43 (3,428 static accesses) +- unwind fix and Tier-2 reentry guard +- DX12 native GE, packed/GPU vertex path, textures, batching and framebuffer feedback path +- AMD/UMA compatibility layer and UMA MSAA crash guard + +## Build behavior + +New stamp: `.vcs_perf_v5_async_vfpu_20260816`. + +Only the five actually modified Tier-2 cluster objects plus `vcs_config`, `vcs_native_fast_paths`, `ge_gpu_backend_dx12`, and `vcs_runtime_log` are explicitly invalidated. Geometry and Boundary are not invalidated and their generated source is byte-identical to V4, preventing the previous long Geometry recompilation cliff. + +## Validation completed in the container + +- `psprecomp_tests`: PASS +- `vcs_profile_tests` with `PSPRECOMP_GE_ASYNC=1`: PASS +- `vfpu_tier2_tests`: PASS +- `vcs_config_tests`: PASS, including DirectX12 async/parallel-decode defaults +- VCSNative target objects for the five modified Tier-2 clusters: PASS +- VCSNative target objects for `vcs_config.cpp`, `vcs_native_fast_paths.cpp`, `ge_gpu_backend_dx12.cpp`, and `vcs_runtime_log.cpp`: PASS on the Linux target/stub path +- generator second run: idempotent (`host_changed=0`, `hook_units_changed=0`) +- Geometry and Boundary generated sources: byte-identical to V4 AMD/UMA base + +The container does not provide the Windows D3D12 SDK/runtime, so the user's VS2022 build remains the authoritative Windows build. V5 does not introduce a new D3D12 command implementation; it changes the default selection of the already-existing ExecuteIndirect path and preserves the V4 AMD code. + +## Expected V5 runtime log header + +```text +stage=perf-v5-async-vfpu-2026-08-16 +... +version=4 ... perf_layer=5 ge_async_default=1 parallel_vertex_decode_default=1 +vfpu_default_prefix_fast=95 ... dx12_execute_indirect_default=0 +amd_uma_compat=1 uma_msaa_guard=1 +``` + +The important PERF values for the next comparison are `fps_avg`, `guest_cpu_us_avg`, `ge_us_avg`, and especially `ge_wait_us_avg`. With async active, `ge_us_avg` is worker CPU time and `ge_wait_us_avg` is the serialized portion paid by the guest/frame. The goal is to reduce wall-frame cost through overlap, not merely make the worker's CPU time disappear from telemetry. + +## Install + +Overlay is relative to **V4 AMD/UMA COMPAT**. Extract it over the current repository, replacing files, then run: + +```bat +profiles\vcs\BUILD_VCS_NINJA.bat +``` + +Do not delete `out` or `.obj` manually. diff --git a/docs/VCS_PERF_V5_ASYNC_VFPU_STATS_2026-08-16.json b/docs/VCS_PERF_V5_ASYNC_VFPU_STATS_2026-08-16.json new file mode 100644 index 0000000..42c872f --- /dev/null +++ b/docs/VCS_PERF_V5_ASYNC_VFPU_STATS_2026-08-16.json @@ -0,0 +1,66 @@ +{ + "name": "VCSNative PERF V5 ASYNC/VFPU", + "date": "2026-08-16", + "base": "PSPRecomp-VCS-TIER2-V4-AMD-UMA-COMPAT-2026-08-16", + "stage": "perf-v5-async-vfpu-2026-08-16", + "perf_layer": 5, + "tier2_version": 4, + "tier2": { + "clusters": 7, + "hot_blocks": 1060, + "generated_cluster_lines": 18027, + "fused_calls": 49, + "fused_tail": 13, + "cold_exits": 9, + "hooks": 25, + "gpr_shadow_clusters": 4, + "gpr_shadow_registers": 24, + "gpr_shadow_occurrences": 3428, + "simd_mat4": 4, + "simd_matvec": 19 + }, + "vfpu_default_prefix_fast": { + "total": 95, + "write_dest": 60, + "vscl": 2, + "vdot": 19, + "vcmp": 14, + "vcmov": 0, + "modified_clusters": [ + "entity", + "matrix", + "physics", + "world", + "edge43" + ], + "geometry_unchanged_from_v4": true, + "boundary_unchanged_from_v4": true + }, + "dx12_defaults": { + "ge_async": true, + "parallel_vertex_decode": true, + "execute_indirect": false + }, + "compatibility_preserved": { + "amd_uma_safe_mode": true, + "uma_msaa_guard": true, + "packed0115_amd_stride": 12 + }, + "validation": { + "psprecomp_tests": "PASS", + "vcs_profile_tests_async_forced": "PASS", + "vfpu_tier2_tests": "PASS", + "vcs_config_tests": "PASS", + "modified_tier2_target_objects": "5/5 PASS", + "changed_host_target_objects_linux": "4/4 PASS", + "generator_idempotent": true, + "geometry_boundary_byte_identical_v4": true, + "full_windows_vcsnative_build": "requires user VS2022" + }, + "native_fast_paths": { + "vfpu_088b1780_block_loads": true, + "scalar_loads_collapsed_from": 16, + "block_loads_to": 4, + "runtime_validation_switch": "PSPRECOMP_VALIDATE_FAST_088B1780" + } +} diff --git a/profiles/vcs/host/ge_gpu_backend_dx12.cpp b/profiles/vcs/host/ge_gpu_backend_dx12.cpp index c1fd1a1..fb4bb2d 100644 --- a/profiles/vcs/host/ge_gpu_backend_dx12.cpp +++ b/profiles/vcs/host/ge_gpu_backend_dx12.cpp @@ -1168,10 +1168,14 @@ bool adjacent_batch_merge_compatible(const Dx12Batch &a, const Dx12Batch &b) noe bool dx12_execute_indirect_enabled() noexcept { static const bool enabled = [] { + // V4 telemetry showed thousands of saved Draw* calls with essentially + // unchanged GE time. Building/uploading indirect records is therefore + // not part of the production fast path until a workload proves a win. + // Keep it as an explicit A/B switch. const char *text = std::getenv("PSPRECOMP_DX12_EXECUTE_INDIRECT"); - return text == nullptr || (*text != '\0' && std::strcmp(text, "0") != 0 && + return text != nullptr && *text != '\0' && std::strcmp(text, "0") != 0 && std::strcmp(text, "false") != 0 && std::strcmp(text, "FALSE") != 0 && - std::strcmp(text, "off") != 0 && std::strcmp(text, "OFF") != 0); + std::strcmp(text, "off") != 0 && std::strcmp(text, "OFF") != 0; }(); return enabled; } diff --git a/profiles/vcs/host/vcs_config.cpp b/profiles/vcs/host/vcs_config.cpp index bb3723d..4208cf4 100644 --- a/profiles/vcs/host/vcs_config.cpp +++ b/profiles/vcs/host/vcs_config.cpp @@ -840,6 +840,14 @@ void initialize_vcs_configuration(const std::filesystem::path &executable_direct config.rendering.backend == RenderingBackend::DirectX12 ? "directx12" : "software"); } + // V5 STABLE RECOVERY2: return every unproven scheduler/CPU-renderer + // experiment to the last gameplay-stable V4 AMD/UMA baseline. Both GE + // async and parallel vertex decode are quarantined in production, and old + // inherited environment flags are deliberately ignored. + if (config.rendering.backend == RenderingBackend::DirectX12) { + set_environment_value("PSPRECOMP_GE_ASYNC", "0"); + set_environment_value("PSPRECOMP_GE_PARALLEL_VERTEX_DECODE", "0"); + } const InternalResolutionDimensions internal = resolve_internal_resolution(config.rendering); if (std::getenv("PSPRECOMP_INTERNAL_WIDTH") == nullptr) diff --git a/profiles/vcs/host/vcs_profile.cpp b/profiles/vcs/host/vcs_profile.cpp index 362ea1f..cf10cea 100644 --- a/profiles/vcs/host/vcs_profile.cpp +++ b/profiles/vcs/host/vcs_profile.cpp @@ -1142,8 +1142,12 @@ GeAsyncWorkerState ge_async{}; thread_local bool ge_async_worker_thread = false; bool ge_async_enabled() noexcept { + // V5 SYNC RECOVERY: the experimental async scheduler is quarantined after + // repeated boot->gameplay deadlocks. Ignore the legacy PSPRECOMP_GE_ASYNC + // variable so a stale shell/BAT cannot silently re-enable the broken path. + // Re-entry is development-only and requires an explicit new opt-in. static const bool enabled = [] { - const char *value = std::getenv("PSPRECOMP_GE_ASYNC"); + const char *value = std::getenv("PSPRECOMP_GE_ASYNC_EXPERIMENTAL"); return value != nullptr && *value != '\0' && std::strcmp(value, "0") != 0; }(); return enabled; diff --git a/profiles/vcs/host/vcs_runtime_log.cpp b/profiles/vcs/host/vcs_runtime_log.cpp index cb5d0df..0a1f07d 100644 --- a/profiles/vcs/host/vcs_runtime_log.cpp +++ b/profiles/vcs/host/vcs_runtime_log.cpp @@ -65,7 +65,7 @@ void runtime_log_initialize(const VcsConfiguration &configuration) { return; } s.file << "VCSNative runtime log\n"; - s.file << "stage=tier2-v4-amd-uma-compat-2026-08-16\n"; + s.file << "stage=perf-v5-stable-recovery2-2026-08-17\n"; s.file << "config=" << configuration.source_path.string() << '\n'; s.file << "started=" << timestamp_now() << '\n'; s.file << "perf_telemetry=" << (configuration.diagnostics.perf_telemetry ? 1 : 0) @@ -79,7 +79,11 @@ void runtime_log_initialize(const VcsConfiguration &configuration) { << " unwind_fix=1 reentry_guard=1 dataflow=1 vfpu_block32=133 mem_runs=35 mem_words=287" << " append32=51 advance32=89 simd_mat4=4 simd_matvec=19" << " gpr_shadow_clusters=4 gpr_shadow_regs=24 gpr_shadow_occurrences=3428 geometry_shadow=0" - << " dx12_execute_indirect_default=1 indirect_buffer_mb=4 amd_uma_compat=1 uma_msaa_guard=1 packed0115_amd_stride=12\n\n"; + << " perf_layer=4 ge_async_default=0 parallel_vertex_decode_default=0" + << " v5_vfpu_fast_quarantined=1 native_vfpu_088b1780_v4=1" + << " dx12_execute_indirect_default=0 indirect_buffer_mb=4" + << " ge_async_quarantined=1 parallel_vertex_decode_quarantined=1 legacy_perf_env_ignored=1" + << " amd_uma_compat=1 uma_msaa_guard=1 packed0115_amd_stride=12\n\n"; if (s.flush_every_line) s.file.flush(); } diff --git a/profiles/vcs/host/vcs_vfpu_fast.hpp b/profiles/vcs/host/vcs_vfpu_fast.hpp new file mode 100644 index 0000000..dc6e08a --- /dev/null +++ b/profiles/vcs/host/vcs_vfpu_fast.hpp @@ -0,0 +1,173 @@ +#pragma once + +#include "psprecomp/allegrex_context.hpp" + +#include +#include +#include + +namespace vcs { + + +// V5 measured-hot VFPU lowering. The PSP resets S/T/D prefixes after each +// consuming instruction; VCS overwhelmingly executes with those architectural +// defaults already installed (S=T=0xE4, D=0). The generic helpers must still +// decode every lane because arbitrary games can program prefixes. Tier-2 can +// cheaply branch around that work in the profile-guided hot regions while +// retaining the exact generic helper as the uncommon fallback. + +template +PSPRECOMP_CONTEXT_FORCEINLINE void tier2_vfpu_write_dest_fast( + psprecomp::AllegrexContext &ctx, const float *source) noexcept { + static_assert(Length >= 1u && Length <= 4u); + if (ctx.vfpu_ctrl[2] == 0u) { + ctx.template write_vfpu_vector_ct(source); + // D was default, but S/T may have been consumed by the instruction. + ctx.eat_vfpu_prefixes(); + return; + } + ctx.template write_vfpu_vector_with_destination_prefix_ct(source); +} + +template +PSPRECOMP_CONTEXT_FORCEINLINE void tier2_vfpu_vscl_fast( + psprecomp::AllegrexContext &ctx) noexcept { + static_assert(Length >= 1u && Length <= 4u); + if (ctx.vfpu_ctrl[0] == 0xE4u && ctx.vfpu_ctrl[1] == 0xE4u && + ctx.vfpu_ctrl[2] == 0u) { + float source[4]{}; + ctx.template read_vfpu_vector_ct(source); + const float scalar = std::bit_cast( + ctx.template vfpu_scalar_bits_ct<(TargetScalarRegister & 0x7Fu)>()); + float result[4]{}; + result[0] = source[0] * scalar; + if constexpr (Length >= 2u) result[1] = source[1] * scalar; + if constexpr (Length >= 3u) result[2] = source[2] * scalar; + if constexpr (Length >= 4u) result[3] = source[3] * scalar; + ctx.template write_vfpu_vector_ct(result); + return; + } + ctx.template execute_vfpu_vscl_ct(); +} + +template +PSPRECOMP_CONTEXT_FORCEINLINE void tier2_vfpu_vdot_fast( + psprecomp::AllegrexContext &ctx) noexcept { + static_assert(Length >= 1u && Length <= 4u); + if (ctx.vfpu_ctrl[0] == 0xE4u && ctx.vfpu_ctrl[1] == 0xE4u && + ctx.vfpu_ctrl[2] == 0u) { + float source[4]{}; + float target[4]{}; + ctx.template read_vfpu_vector_ct(source); + ctx.template read_vfpu_vector_ct(target); + // Preserve the original expression/order; lanes outside Length remain + // exact +0 under default prefixes. + const float result[1]{ + source[0] * target[0] + source[1] * target[1] + + source[2] * target[2] + source[3] * target[3] + }; + ctx.template write_vfpu_vector_ct(result); + return; + } + ctx.template execute_vfpu_vdot_ct(); +} + +template +PSPRECOMP_CONTEXT_FORCEINLINE void tier2_vfpu_vcmp_fast( + psprecomp::AllegrexContext &ctx) noexcept { + static_assert(Length >= 1u && Length <= 4u); + static_assert(Condition < 16u); + if (!(ctx.vfpu_ctrl[0] == 0xE4u && ctx.vfpu_ctrl[1] == 0xE4u && + ctx.vfpu_ctrl[2] == 0u)) { + ctx.template execute_vfpu_vcmp_ct(); + return; + } + + float source[4]{}; + float target[4]{}; + ctx.template read_vfpu_vector_ct(source); + ctx.template read_vfpu_vector_ct(target); + auto compare_lane = [](float sv, float tv) -> bool { + if constexpr (Condition == 0u) return false; + else if constexpr (Condition == 1u) return sv == tv; + else if constexpr (Condition == 2u) return sv < tv; + else if constexpr (Condition == 3u) return sv <= tv; + else if constexpr (Condition == 4u) return true; + else if constexpr (Condition == 5u) return sv != tv; + else if constexpr (Condition == 6u) return sv >= tv; + else if constexpr (Condition == 7u) return sv > tv; + else if constexpr (Condition == 8u) return sv == 0.0f; + else if constexpr (Condition == 9u) return std::isnan(sv); + else if constexpr (Condition == 10u) return std::isinf(sv); + else if constexpr (Condition == 11u) return std::isnan(sv) || std::isinf(sv); + else if constexpr (Condition == 12u) return sv != 0.0f; + else if constexpr (Condition == 13u) return !std::isnan(sv); + else if constexpr (Condition == 14u) return !std::isinf(sv); + else return !(std::isnan(sv) || std::isinf(sv)); + }; + + const bool r0 = compare_lane(source[0], target[0]); + const bool r1 = Length >= 2u ? compare_lane(source[1], target[1]) : false; + const bool r2 = Length >= 3u ? compare_lane(source[2], target[2]) : false; + const bool r3 = Length >= 4u ? compare_lane(source[3], target[3]) : false; + std::uint32_t lane_bits = static_cast(r0); + if constexpr (Length >= 2u) lane_bits |= static_cast(r1) << 1u; + if constexpr (Length >= 3u) lane_bits |= static_cast(r2) << 2u; + if constexpr (Length >= 4u) lane_bits |= static_cast(r3) << 3u; + bool any = r0; + bool all = r0; + if constexpr (Length >= 2u) { any = any || r1; all = all && r1; } + if constexpr (Length >= 3u) { any = any || r2; all = all && r2; } + if constexpr (Length >= 4u) { any = any || r3; all = all && r3; } + constexpr std::uint32_t affected = ((1u << Length) - 1u) | (1u << 4u) | (1u << 5u); + const std::uint32_t update = lane_bits | (static_cast(any) << 4u) | + (static_cast(all) << 5u); + ctx.vfpu_ctrl[3] = (ctx.vfpu_ctrl[3] & ~affected) | (update & affected); + // S/T/D were already defaults, so consuming them requires no stores. +} + +template +PSPRECOMP_CONTEXT_FORCEINLINE void tier2_vfpu_vcmov_fast( + psprecomp::AllegrexContext &ctx) noexcept { + static_assert(Length >= 1u && Length <= 4u); + static_assert(ConditionIndex < 8u); + if (!(ctx.vfpu_ctrl[0] == 0xE4u && ctx.vfpu_ctrl[1] == 0xE4u && + ctx.vfpu_ctrl[2] == 0u)) { + ctx.template execute_vfpu_vcmov_ct(); + return; + } + + float source[4]{}; + float destination[4]{}; + ctx.template read_vfpu_vector_ct(source); + ctx.template read_vfpu_vector_ct(destination); + const std::uint32_t condition_code = ctx.vfpu_ctrl[3]; + if constexpr (ConditionIndex < 6u) { + const bool cc = ((condition_code >> ConditionIndex) & 1u) != 0u; + if (cc == !MoveIfFalse) { + destination[0] = source[0]; + if constexpr (Length >= 2u) destination[1] = source[1]; + if constexpr (Length >= 3u) destination[2] = source[2]; + if constexpr (Length >= 4u) destination[3] = source[3]; + } + } else if constexpr (ConditionIndex == 6u) { + constexpr bool want = !MoveIfFalse; + if ((((condition_code >> 0u) & 1u) != 0u) == want) destination[0] = source[0]; + if constexpr (Length >= 2u) + if ((((condition_code >> 1u) & 1u) != 0u) == want) destination[1] = source[1]; + if constexpr (Length >= 3u) + if ((((condition_code >> 2u) & 1u) != 0u) == want) destination[2] = source[2]; + if constexpr (Length >= 4u) + if ((((condition_code >> 3u) & 1u) != 0u) == want) destination[3] = source[3]; + } + ctx.template write_vfpu_vector_ct(destination); +} + +} // namespace vcs diff --git a/profiles/vcs/scripts/bench.bat b/profiles/vcs/scripts/bench.bat index c4f757e..6b71d8f 100644 --- a/profiles/vcs/scripts/bench.bat +++ b/profiles/vcs/scripts/bench.bat @@ -28,11 +28,11 @@ set "PSPRECOMP_DX12_DEBUG=0" set "PSPRECOMP_DX12_GE_READBACK=0" set "PSPRECOMP_DX12_GE_STRICT=0" set "PSPRECOMP_DX12_TEXTURE_UPLOAD_RING=1" -set "PSPRECOMP_GE_ASYNC=1" +set "PSPRECOMP_GE_ASYNC=0" set "PSPRECOMP_FRAME_LIMIT=0" set "PSPRECOMP_FRAME_TIME_DIAG=1" set "PSPRECOMP_GE_PHASE_DIAG=1" -set "PSPRECOMP_GE_PARALLEL_VERTEX_DECODE=1" +set "PSPRECOMP_GE_PARALLEL_VERTEX_DECODE=0" set "PSPRECOMP_GE_PARALLEL_VERTEX_THRESHOLD=768" set "PSPRECOMP_GE_PARALLEL_VERTEX_MAX_WORKERS=6" set "PSPRECOMP_GE_DIRECT_NONINDEXED_DRAW=1" diff --git a/profiles/vcs/scripts/build_release_ninja.bat b/profiles/vcs/scripts/build_release_ninja.bat index bdcbbf5..3fa9b2a 100644 --- a/profiles/vcs/scripts/build_release_ninja.bat +++ b/profiles/vcs/scripts/build_release_ninja.bat @@ -94,6 +94,11 @@ set "NINJA_STATUS=[%%f/%%t %%p ^| %%e elapsed ^| %%r running] " set "BOOTFIX_STAMP=%BUILD%\.vcs_tier2_bootfix_20260816_v1" set "SUPERBLOCK_STAMP=%BUILD%\.vcs_tier2_v4_150fps_buildfix3_20260816" set "AMD_COMPAT_STAMP=%BUILD%\.vcs_dx12_amd_uma_compat_20260816" +set "PERF_V5_STAMP=%BUILD%\.vcs_perf_v5_async_vfpu_20260816" +set "PERF_V5_STALLFIX_STAMP=%BUILD%\.vcs_perf_v5_async_stallfix_20260817" +set "PERF_V5_PRESENTFIX_STAMP=%BUILD%\.vcs_perf_v5_async_presentfix_20260817" +set "PERF_V5_SYNC_RECOVERY_STAMP=%BUILD%\.vcs_perf_v5_sync_recovery_20260817" +set "PERF_V5_STABLE_RECOVERY2_STAMP=%BUILD%\.vcs_perf_v5_stable_recovery2_20260817" echo ================================================================ echo VCS - NINJA PERFORMANCE INCREMENTAL BUILD @@ -105,7 +110,7 @@ echo CMake: %CMAKE_EXE% echo Ninja: %NINJA_EXE% echo Ninja workers: %JOBS% echo cl.exe /MP: OFF ^(Ninja owns compile parallelism^) -echo Generated AOT: O3, cold /Ob0, measured hot /Ob3; Tier2 V4 + AMD/UMA DX12 compatibility; Geometry /Ob2 + shadow-off guard +echo Generated AOT: O3, cold /Ob0, measured hot /Ob3; V4-stable Tier2 + sync GE; V5 risky paths quarantined; AMD/UMA safe echo Host/core LTCG: ON echo AVX2/fast paths: ON echo ================================================================ @@ -114,7 +119,7 @@ echo [0b/7] Reapplying BOOTFIX-safe Tier-2 transforms (OPT1 semantic transforms call "%PROFILE%\APPLY_TIER2_EXTREME.bat" if errorlevel 1 goto :FAIL -echo [0b2/7] Building profile-guided Tier-2 V4 150FPS second layer... +echo [0b2/7] Building gameplay-stable V4 Tier2 layer with V5 recovery guards... set "PYTHON3_CMD=" py -3 -c "import sys; raise SystemExit(0 if sys.version_info.major == 3 else 1)" >nul 2>&1 if not errorlevel 1 set "PYTHON3_CMD=py -3" @@ -144,6 +149,61 @@ if exist "%BUILD%" if not exist "%AMD_COMPAT_STAMP%" ( del /s /q "%BUILD%\*vcs_runtime_log*.obj" >nul 2>&1 ) +if exist "%BUILD%" if not exist "%PERF_V5_STAMP%" ( + echo. + echo [0c-v5/7] V5 ASYNC/VFPU - invalidating hot clusters + changed host objects once... + rem Hooks/generated units are unchanged. Rebuild only the five modified Tier2 cluster TUs and host policy/backend/log. + del /s /q "%BUILD%\*vcs_tier2_cluster_entity*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_tier2_cluster_matrix*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_tier2_cluster_physics*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_tier2_cluster_world*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_tier2_cluster_edge43*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_config*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_native_fast_paths*.obj" >nul 2>&1 + del /s /q "%BUILD%\*ge_gpu_backend_dx12*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_runtime_log*.obj" >nul 2>&1 +) + +if exist "%BUILD%" if not exist "%PERF_V5_STALLFIX_STAMP%" ( + echo. + echo [0c-v5fix/7] V5 GE ASYNC STALL-RACE FIX - invalidating profile + runtime log once... + rem Hotfix only changes the async GE scheduler/telemetry. Keep all Tier2 and DX12 objects. + del /s /q "%BUILD%\*vcs_profile*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_runtime_log*.obj" >nul 2>&1 +) + +if exist "%BUILD%" if not exist "%PERF_V5_PRESENTFIX_STAMP%" ( + echo. + echo [0c-v5present/7] V5 GE ASYNC PRESENTFIX - invalidating profile + runtime log once... + rem Presentation safe-point fix only changes async GE/display scheduling + metadata. + del /s /q "%BUILD%\*vcs_profile*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_runtime_log*.obj" >nul 2>&1 +) + +if exist "%BUILD%" if not exist "%PERF_V5_SYNC_RECOVERY_STAMP%" ( + echo. + echo [0c-v5sync/7] V5 SYNC RECOVERY - restoring proven GE scheduler + safe defaults once... + rem Only scheduler/config/log changed. Preserve expensive Tier2/DX12 objects. + del /s /q "%BUILD%\*vcs_profile*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_config*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_runtime_log*.obj" >nul 2>&1 +) + +if exist "%BUILD%" if not exist "%PERF_V5_STABLE_RECOVERY2_STAMP%" ( + echo. + echo [0c-v5stable2/7] V5 STABLE RECOVERY2 - quarantining parallel decode + V5 VFPU/native experiments once... + rem Restore only the five V5-modified Tier2 clusters and native/config/log objects. + rem Geometry and Boundary remain untouched to avoid the prior MSVC compile-time cliff. + del /s /q "%BUILD%\*vcs_tier2_cluster_entity*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_tier2_cluster_matrix*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_tier2_cluster_physics*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_tier2_cluster_world*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_tier2_cluster_edge43*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_config*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_native_fast_paths*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_runtime_log*.obj" >nul 2>&1 +) + if exist "%BUILD%" if not exist "%BOOTFIX_STAMP%" ( echo. echo [0c/7] BOOTFIX revision changed - invalidating stale .obj/.pch once... @@ -180,6 +240,11 @@ if errorlevel 1 goto :FAIL >"%BOOTFIX_STAMP%" echo VCS Tier2 BOOTFIX 2026-08-16 v1 >"%SUPERBLOCK_STAMP%" echo VCS Tier2 V4 BUILDFIX3 2026-08-16 >"%AMD_COMPAT_STAMP%" echo VCS DX12 AMD UMA COMPAT 2026-08-16 +>"%PERF_V5_STAMP%" echo VCS PERF V5 ASYNC VFPU 2026-08-16 +>"%PERF_V5_STALLFIX_STAMP%" echo VCS PERF V5 GE ASYNC STALL-RACE FIX 2026-08-17 +>"%PERF_V5_PRESENTFIX_STAMP%" echo VCS PERF V5 GE ASYNC PRESENTFIX 2026-08-17 +>"%PERF_V5_SYNC_RECOVERY_STAMP%" echo VCS PERF V5 SYNC RECOVERY 2026-08-17 +>"%PERF_V5_STABLE_RECOVERY2_STAMP%" echo VCS PERF V5 STABLE RECOVERY2 2026-08-17 echo. echo [2b/7] Building tests and DX12 probes... diff --git a/profiles/vcs/scripts/play.bat b/profiles/vcs/scripts/play.bat index 6a585d7..6f3bb54 100644 --- a/profiles/vcs/scripts/play.bat +++ b/profiles/vcs/scripts/play.bat @@ -38,7 +38,7 @@ set "PSPRECOMP_CONFIG=%PROFILE%\config\VCSNative.ini" set "PSPRECOMP_GE_BACKEND=directx12" set "PSPRECOMP_GE_GPU_TELEMETRY=0" set "PSPRECOMP_GE_GPU_REPORT=0" -set "PSPRECOMP_GE_ASYNC=" +set "PSPRECOMP_GE_ASYNC=0" set "PSPRECOMP_GE_GPU_SKIP_SOFTWARE_RASTER=" set "PSPRECOMP_CHAIN_DEPTH=" set "PSPRECOMP_TIME_TICK_DISPATCHES=" diff --git a/profiles/vcs/tests/vcs_config_tests.cpp b/profiles/vcs/tests/vcs_config_tests.cpp index 30f0eba..9faa896 100644 --- a/profiles/vcs/tests/vcs_config_tests.cpp +++ b/profiles/vcs/tests/vcs_config_tests.cpp @@ -1,6 +1,7 @@ #include "vcs_config.hpp" #include +#include #include #include #include @@ -179,6 +180,12 @@ int main() { << "DayProgression=-0.12\n"; } vcs::initialize_vcs_configuration(root); + const char *async_default = std::getenv("PSPRECOMP_GE_ASYNC"); + const char *parallel_decode_default = std::getenv("PSPRECOMP_GE_PARALLEL_VERTEX_DECODE"); + require(async_default != nullptr && std::string(async_default) == "0", + "DirectX12 sync-recovery must force legacy GE async off"); + require(parallel_decode_default != nullptr && std::string(parallel_decode_default) == "0", + "DirectX12 stable recovery must force parallel vertex decode off"); const auto &clouds = vcs::vcs_configuration().volumetric_clouds; require(clouds.enabled, "ProperShaders.ini VolumetricClouds.Enabled was not parsed"); require(clouds.downscale_div == 4u && clouds.layers == 3u &&