diff --git a/profiles/vcs/host/vcs_runtime_log.cpp b/profiles/vcs/host/vcs_runtime_log.cpp index 0a1f07d..3f39a12 100644 --- a/profiles/vcs/host/vcs_runtime_log.cpp +++ b/profiles/vcs/host/vcs_runtime_log.cpp @@ -65,7 +65,7 @@ void runtime_log_initialize(const VcsConfiguration &configuration) { return; } s.file << "VCSNative runtime log\n"; - s.file << "stage=perf-v5-stable-recovery2-2026-08-17\n"; + s.file << "stage=perf-v6-entity-leaf-inline-crashfix1-2026-08-17\n"; s.file << "config=" << configuration.source_path.string() << '\n'; s.file << "started=" << timestamp_now() << '\n'; s.file << "perf_telemetry=" << (configuration.diagnostics.perf_telemetry ? 1 : 0) @@ -75,11 +75,13 @@ void runtime_log_initialize(const VcsConfiguration &configuration) { << " sample_stride=256 interval_vblanks=300\n"; s.file << "tier2_superblocks=" << (tier2_superblocks_enabled() ? 1 : 0) << " version=4 clusters=7 mask=0x" << std::hex << tier2_cluster_mask() << std::dec - << " hot_blocks=1060 static_fused_calls=49 static_fused_tail=13 hooks=25" + << " hot_blocks=1060 static_fused_calls=36 static_fused_tail=13 hooks=25" + << " entity_leaf_inline_sites=13" << " unwind_fix=1 reentry_guard=1 dataflow=1 vfpu_block32=133 mem_runs=35 mem_words=287" << " append32=51 advance32=89 simd_mat4=4 simd_matvec=19" - << " gpr_shadow_clusters=4 gpr_shadow_regs=24 gpr_shadow_occurrences=3428 geometry_shadow=0" - << " perf_layer=4 ge_async_default=0 parallel_vertex_decode_default=0" + << " gpr_shadow_clusters=4 gpr_shadow_regs=24 gpr_shadow_occurrences=3461 geometry_shadow=0" + << " perf_layer=6 entity_leaf_inline=1 entity_leaf_scheduler_accounting=1 entity_leaf_resume_pc_fix=1" + << " ge_async_default=0 parallel_vertex_decode_default=0" << " v5_vfpu_fast_quarantined=1 native_vfpu_088b1780_v4=1" << " dx12_execute_indirect_default=0 indirect_buffer_mb=4" << " ge_async_quarantined=1 parallel_vertex_decode_quarantined=1 legacy_perf_env_ignored=1" diff --git a/profiles/vcs/host/vcs_tier2_cluster_entity.cpp b/profiles/vcs/host/vcs_tier2_cluster_entity.cpp index f72055c..2747cc3 100644 --- a/profiles/vcs/host/vcs_tier2_cluster_entity.cpp +++ b/profiles/vcs/host/vcs_tier2_cluster_entity.cpp @@ -1436,21 +1436,20 @@ SB_L_08A71100: SB_L_08A71108: ctx.gpr[31] = (0x08A71110u); tier2_gpr_4 = (tier2_gpr_19 | 0u); - if (tier2_return_depth < kTier2ReturnCapacity) { - if (!rt.tier2_enter_fused_transfer<152u, 0x08A65EB4u>(ctx)) { - ctx.pc = 0x08A65EB4u; - TIER2_SB_RETURN(); - } - tier2_return_pc[tier2_return_depth] = 0x08A71110u; - tier2_return_unit[tier2_return_depth] = 155u; - tier2_return_pending_base[tier2_return_depth] = tier2_pending_transfers; - ++tier2_return_depth; - ++tier2_stats.fused_calls; - goto SB_L_08A65EB4; + tier2_gpr_4 = aot_mem.aot_load32(tier2_gpr_4 + static_cast(72)); + tier2_gpr_4 = (tier2_gpr_4 & 14u); + ctx.gpr[2] = (tier2_gpr_4 ^ 8u); + ctx.gpr[2] = (ctx.gpr[2] < static_cast(1) ? 1u : 0u); + // Publish the exact JAL return PC before scheduler accounting. + // A starvation boundary may switch PSP ownership here; the + // resumed context must never observe the stale superblock PC. + ctx.pc = 0x08A71110u; + TIER2_GPR_SYNC_OUT(); + if (!rt.account_inlined_generated_leaf(ctx)) { + tier2_gpr_shadow_valid = false; + TIER2_SB_RETURN(); } - ++tier2_stats.fallbacks; - if (([&]() { TIER2_GPR_SYNC_OUT(); const bool tier2_same_ = (rt.invoke_chained_direct<&recomp_unit_0152_entry, 152u, 433u, 0x08A65EB4u>(ctx, &aot_mem)); if (tier2_same_) TIER2_GPR_SYNC_IN(); else tier2_gpr_shadow_valid = false; return tier2_same_; }()) && ctx.pc == 0x08A71110u) goto SB_L_08A71110; - TIER2_SB_RETURN(); + goto SB_L_08A71110; SB_L_08A71110: { const bool branch_taken = ctx.gpr[2] == 0u; @@ -1464,21 +1463,17 @@ SB_L_08A71110: SB_L_08A71118: ctx.gpr[31] = (0x08A71120u); tier2_gpr_4 = (tier2_gpr_19 | 0u); - if (tier2_return_depth < kTier2ReturnCapacity) { - if (!rt.tier2_enter_fused_transfer<153u, 0x08A68CBCu>(ctx)) { - ctx.pc = 0x08A68CBCu; - TIER2_SB_RETURN(); - } - tier2_return_pc[tier2_return_depth] = 0x08A71120u; - tier2_return_unit[tier2_return_depth] = 155u; - tier2_return_pending_base[tier2_return_depth] = tier2_pending_transfers; - ++tier2_return_depth; - ++tier2_stats.fused_calls; - goto SB_L_08A68CBC; + ctx.gpr[2] = aot_mem.aot_load32(tier2_gpr_4 + static_cast(444)); + // Publish the exact JAL return PC before scheduler accounting. + // A starvation boundary may switch PSP ownership here; the + // resumed context must never observe the stale superblock PC. + ctx.pc = 0x08A71120u; + TIER2_GPR_SYNC_OUT(); + if (!rt.account_inlined_generated_leaf(ctx)) { + tier2_gpr_shadow_valid = false; + TIER2_SB_RETURN(); } - ++tier2_stats.fallbacks; - if (([&]() { TIER2_GPR_SYNC_OUT(); const bool tier2_same_ = (rt.invoke_chained_direct<&recomp_unit_0153_entry, 153u, 100u, 0x08A68CBCu>(ctx, &aot_mem)); if (tier2_same_) TIER2_GPR_SYNC_IN(); else tier2_gpr_shadow_valid = false; return tier2_same_; }()) && ctx.pc == 0x08A71120u) goto SB_L_08A71120; - TIER2_SB_RETURN(); + goto SB_L_08A71120; SB_L_08A71120: { const bool branch_taken = ctx.gpr[2] != tier2_gpr_17; @@ -1493,21 +1488,17 @@ SB_L_08A71128: tier2_gpr_4 = (tier2_gpr_19 | 0u); ctx.gpr[31] = (0x08A71134u); tier2_gpr_5 = (0u | 0u); - if (tier2_return_depth < kTier2ReturnCapacity) { - if (!rt.tier2_enter_fused_transfer<153u, 0x08A68CC4u>(ctx)) { - ctx.pc = 0x08A68CC4u; - TIER2_SB_RETURN(); - } - tier2_return_pc[tier2_return_depth] = 0x08A71134u; - tier2_return_unit[tier2_return_depth] = 155u; - tier2_return_pending_base[tier2_return_depth] = tier2_pending_transfers; - ++tier2_return_depth; - ++tier2_stats.fused_calls; - goto SB_L_08A68CC4; + aot_mem.aot_store32(tier2_gpr_4 + static_cast(444), tier2_gpr_5); + // Publish the exact JAL return PC before scheduler accounting. + // A starvation boundary may switch PSP ownership here; the + // resumed context must never observe the stale superblock PC. + ctx.pc = 0x08A71134u; + TIER2_GPR_SYNC_OUT(); + if (!rt.account_inlined_generated_leaf(ctx)) { + tier2_gpr_shadow_valid = false; + TIER2_SB_RETURN(); } - ++tier2_stats.fallbacks; - if (([&]() { TIER2_GPR_SYNC_OUT(); const bool tier2_same_ = (rt.invoke_chained_direct<&recomp_unit_0153_entry, 153u, 101u, 0x08A68CC4u>(ctx, &aot_mem)); if (tier2_same_) TIER2_GPR_SYNC_IN(); else tier2_gpr_shadow_valid = false; return tier2_same_; }()) && ctx.pc == 0x08A71134u) goto SB_L_08A71134; - TIER2_SB_RETURN(); + goto SB_L_08A71134; SB_L_08A71134: { const bool branch_taken = 0u == 0u; @@ -1521,21 +1512,20 @@ SB_L_08A71134: SB_L_08A7113C: ctx.gpr[31] = (0x08A71144u); tier2_gpr_4 = (tier2_gpr_17 | 0u); - if (tier2_return_depth < kTier2ReturnCapacity) { - if (!rt.tier2_enter_fused_transfer<152u, 0x08A65EB4u>(ctx)) { - ctx.pc = 0x08A65EB4u; - TIER2_SB_RETURN(); - } - tier2_return_pc[tier2_return_depth] = 0x08A71144u; - tier2_return_unit[tier2_return_depth] = 155u; - tier2_return_pending_base[tier2_return_depth] = tier2_pending_transfers; - ++tier2_return_depth; - ++tier2_stats.fused_calls; - goto SB_L_08A65EB4; + tier2_gpr_4 = aot_mem.aot_load32(tier2_gpr_4 + static_cast(72)); + tier2_gpr_4 = (tier2_gpr_4 & 14u); + ctx.gpr[2] = (tier2_gpr_4 ^ 8u); + ctx.gpr[2] = (ctx.gpr[2] < static_cast(1) ? 1u : 0u); + // Publish the exact JAL return PC before scheduler accounting. + // A starvation boundary may switch PSP ownership here; the + // resumed context must never observe the stale superblock PC. + ctx.pc = 0x08A71144u; + TIER2_GPR_SYNC_OUT(); + if (!rt.account_inlined_generated_leaf(ctx)) { + tier2_gpr_shadow_valid = false; + TIER2_SB_RETURN(); } - ++tier2_stats.fallbacks; - if (([&]() { TIER2_GPR_SYNC_OUT(); const bool tier2_same_ = (rt.invoke_chained_direct<&recomp_unit_0152_entry, 152u, 433u, 0x08A65EB4u>(ctx, &aot_mem)); if (tier2_same_) TIER2_GPR_SYNC_IN(); else tier2_gpr_shadow_valid = false; return tier2_same_; }()) && ctx.pc == 0x08A71144u) goto SB_L_08A71144; - TIER2_SB_RETURN(); + goto SB_L_08A71144; SB_L_08A71144: { const bool branch_taken = ctx.gpr[2] == 0u; @@ -1549,21 +1539,17 @@ SB_L_08A71144: SB_L_08A7114C: ctx.gpr[31] = (0x08A71154u); tier2_gpr_4 = (tier2_gpr_17 | 0u); - if (tier2_return_depth < kTier2ReturnCapacity) { - if (!rt.tier2_enter_fused_transfer<153u, 0x08A68CBCu>(ctx)) { - ctx.pc = 0x08A68CBCu; - TIER2_SB_RETURN(); - } - tier2_return_pc[tier2_return_depth] = 0x08A71154u; - tier2_return_unit[tier2_return_depth] = 155u; - tier2_return_pending_base[tier2_return_depth] = tier2_pending_transfers; - ++tier2_return_depth; - ++tier2_stats.fused_calls; - goto SB_L_08A68CBC; + ctx.gpr[2] = aot_mem.aot_load32(tier2_gpr_4 + static_cast(444)); + // Publish the exact JAL return PC before scheduler accounting. + // A starvation boundary may switch PSP ownership here; the + // resumed context must never observe the stale superblock PC. + ctx.pc = 0x08A71154u; + TIER2_GPR_SYNC_OUT(); + if (!rt.account_inlined_generated_leaf(ctx)) { + tier2_gpr_shadow_valid = false; + TIER2_SB_RETURN(); } - ++tier2_stats.fallbacks; - if (([&]() { TIER2_GPR_SYNC_OUT(); const bool tier2_same_ = (rt.invoke_chained_direct<&recomp_unit_0153_entry, 153u, 100u, 0x08A68CBCu>(ctx, &aot_mem)); if (tier2_same_) TIER2_GPR_SYNC_IN(); else tier2_gpr_shadow_valid = false; return tier2_same_; }()) && ctx.pc == 0x08A71154u) goto SB_L_08A71154; - TIER2_SB_RETURN(); + goto SB_L_08A71154; SB_L_08A71154: { const bool branch_taken = ctx.gpr[2] != tier2_gpr_19; @@ -1577,21 +1563,17 @@ SB_L_08A71154: SB_L_08A7115C: ctx.gpr[31] = (0x08A71164u); tier2_gpr_4 = (tier2_gpr_17 | 0u); - if (tier2_return_depth < kTier2ReturnCapacity) { - if (!rt.tier2_enter_fused_transfer<153u, 0x08A68CCCu>(ctx)) { - ctx.pc = 0x08A68CCCu; - TIER2_SB_RETURN(); - } - tier2_return_pc[tier2_return_depth] = 0x08A71164u; - tier2_return_unit[tier2_return_depth] = 155u; - tier2_return_pending_base[tier2_return_depth] = tier2_pending_transfers; - ++tier2_return_depth; - ++tier2_stats.fused_calls; - goto SB_L_08A68CCC; + ctx.gpr[2] = aot_mem.aot_load32(tier2_gpr_4 + static_cast(448)); + // Publish the exact JAL return PC before scheduler accounting. + // A starvation boundary may switch PSP ownership here; the + // resumed context must never observe the stale superblock PC. + ctx.pc = 0x08A71164u; + TIER2_GPR_SYNC_OUT(); + if (!rt.account_inlined_generated_leaf(ctx)) { + tier2_gpr_shadow_valid = false; + TIER2_SB_RETURN(); } - ++tier2_stats.fallbacks; - if (([&]() { TIER2_GPR_SYNC_OUT(); const bool tier2_same_ = (rt.invoke_chained_direct<&recomp_unit_0153_entry, 153u, 102u, 0x08A68CCCu>(ctx, &aot_mem)); if (tier2_same_) TIER2_GPR_SYNC_IN(); else tier2_gpr_shadow_valid = false; return tier2_same_; }()) && ctx.pc == 0x08A71164u) goto SB_L_08A71164; - TIER2_SB_RETURN(); + goto SB_L_08A71164; SB_L_08A71164: { const bool branch_taken = ctx.gpr[2] == tier2_gpr_19; @@ -1606,21 +1588,17 @@ SB_L_08A7116C: tier2_gpr_4 = (tier2_gpr_17 | 0u); ctx.gpr[31] = (0x08A71178u); tier2_gpr_5 = (0u | 0u); - if (tier2_return_depth < kTier2ReturnCapacity) { - if (!rt.tier2_enter_fused_transfer<153u, 0x08A68CC4u>(ctx)) { - ctx.pc = 0x08A68CC4u; - TIER2_SB_RETURN(); - } - tier2_return_pc[tier2_return_depth] = 0x08A71178u; - tier2_return_unit[tier2_return_depth] = 155u; - tier2_return_pending_base[tier2_return_depth] = tier2_pending_transfers; - ++tier2_return_depth; - ++tier2_stats.fused_calls; - goto SB_L_08A68CC4; + aot_mem.aot_store32(tier2_gpr_4 + static_cast(444), tier2_gpr_5); + // Publish the exact JAL return PC before scheduler accounting. + // A starvation boundary may switch PSP ownership here; the + // resumed context must never observe the stale superblock PC. + ctx.pc = 0x08A71178u; + TIER2_GPR_SYNC_OUT(); + if (!rt.account_inlined_generated_leaf(ctx)) { + tier2_gpr_shadow_valid = false; + TIER2_SB_RETURN(); } - ++tier2_stats.fallbacks; - if (([&]() { TIER2_GPR_SYNC_OUT(); const bool tier2_same_ = (rt.invoke_chained_direct<&recomp_unit_0153_entry, 153u, 101u, 0x08A68CC4u>(ctx, &aot_mem)); if (tier2_same_) TIER2_GPR_SYNC_IN(); else tier2_gpr_shadow_valid = false; return tier2_same_; }()) && ctx.pc == 0x08A71178u) goto SB_L_08A71178; - TIER2_SB_RETURN(); + goto SB_L_08A71178; SB_L_08A71178: { const bool branch_taken = 0u == 0u; @@ -1634,21 +1612,20 @@ SB_L_08A71178: SB_L_08A71180: ctx.gpr[31] = (0x08A71188u); tier2_gpr_4 = (tier2_gpr_19 | 0u); - if (tier2_return_depth < kTier2ReturnCapacity) { - if (!rt.tier2_enter_fused_transfer<152u, 0x08A65EA0u>(ctx)) { - ctx.pc = 0x08A65EA0u; - TIER2_SB_RETURN(); - } - tier2_return_pc[tier2_return_depth] = 0x08A71188u; - tier2_return_unit[tier2_return_depth] = 155u; - tier2_return_pending_base[tier2_return_depth] = tier2_pending_transfers; - ++tier2_return_depth; - ++tier2_stats.fused_calls; - goto SB_L_08A65EA0; + tier2_gpr_4 = aot_mem.aot_load32(tier2_gpr_4 + static_cast(72)); + tier2_gpr_4 = (tier2_gpr_4 & 14u); + ctx.gpr[2] = (tier2_gpr_4 ^ 6u); + ctx.gpr[2] = (ctx.gpr[2] < static_cast(1) ? 1u : 0u); + // Publish the exact JAL return PC before scheduler accounting. + // A starvation boundary may switch PSP ownership here; the + // resumed context must never observe the stale superblock PC. + ctx.pc = 0x08A71188u; + TIER2_GPR_SYNC_OUT(); + if (!rt.account_inlined_generated_leaf(ctx)) { + tier2_gpr_shadow_valid = false; + TIER2_SB_RETURN(); } - ++tier2_stats.fallbacks; - if (([&]() { TIER2_GPR_SYNC_OUT(); const bool tier2_same_ = (rt.invoke_chained_direct<&recomp_unit_0152_entry, 152u, 432u, 0x08A65EA0u>(ctx, &aot_mem)); if (tier2_same_) TIER2_GPR_SYNC_IN(); else tier2_gpr_shadow_valid = false; return tier2_same_; }()) && ctx.pc == 0x08A71188u) goto SB_L_08A71188; - TIER2_SB_RETURN(); + goto SB_L_08A71188; SB_L_08A71188: { const bool branch_taken = ctx.gpr[2] == 0u; @@ -1662,21 +1639,17 @@ SB_L_08A71188: SB_L_08A71190: ctx.gpr[31] = (0x08A71198u); tier2_gpr_4 = (tier2_gpr_19 | 0u); - if (tier2_return_depth < kTier2ReturnCapacity) { - if (!rt.tier2_enter_fused_transfer<153u, 0x08A68CD4u>(ctx)) { - ctx.pc = 0x08A68CD4u; - TIER2_SB_RETURN(); - } - tier2_return_pc[tier2_return_depth] = 0x08A71198u; - tier2_return_unit[tier2_return_depth] = 155u; - tier2_return_pending_base[tier2_return_depth] = tier2_pending_transfers; - ++tier2_return_depth; - ++tier2_stats.fused_calls; - goto SB_L_08A68CD4; + ctx.gpr[2] = aot_mem.aot_load32(tier2_gpr_4 + static_cast(2116)); + // Publish the exact JAL return PC before scheduler accounting. + // A starvation boundary may switch PSP ownership here; the + // resumed context must never observe the stale superblock PC. + ctx.pc = 0x08A71198u; + TIER2_GPR_SYNC_OUT(); + if (!rt.account_inlined_generated_leaf(ctx)) { + tier2_gpr_shadow_valid = false; + TIER2_SB_RETURN(); } - ++tier2_stats.fallbacks; - if (([&]() { TIER2_GPR_SYNC_OUT(); const bool tier2_same_ = (rt.invoke_chained_direct<&recomp_unit_0153_entry, 153u, 103u, 0x08A68CD4u>(ctx, &aot_mem)); if (tier2_same_) TIER2_GPR_SYNC_IN(); else tier2_gpr_shadow_valid = false; return tier2_same_; }()) && ctx.pc == 0x08A71198u) goto SB_L_08A71198; - TIER2_SB_RETURN(); + goto SB_L_08A71198; SB_L_08A71198: { const bool branch_taken = ctx.gpr[2] != tier2_gpr_17; @@ -1691,21 +1664,17 @@ SB_L_08A711A0: tier2_gpr_4 = (tier2_gpr_19 | 0u); ctx.gpr[31] = (0x08A711ACu); tier2_gpr_5 = (0u | 0u); - if (tier2_return_depth < kTier2ReturnCapacity) { - if (!rt.tier2_enter_fused_transfer<153u, 0x08A68CDCu>(ctx)) { - ctx.pc = 0x08A68CDCu; - TIER2_SB_RETURN(); - } - tier2_return_pc[tier2_return_depth] = 0x08A711ACu; - tier2_return_unit[tier2_return_depth] = 155u; - tier2_return_pending_base[tier2_return_depth] = tier2_pending_transfers; - ++tier2_return_depth; - ++tier2_stats.fused_calls; - goto SB_L_08A68CDC; + aot_mem.aot_store32(tier2_gpr_4 + static_cast(2116), tier2_gpr_5); + // Publish the exact JAL return PC before scheduler accounting. + // A starvation boundary may switch PSP ownership here; the + // resumed context must never observe the stale superblock PC. + ctx.pc = 0x08A711ACu; + TIER2_GPR_SYNC_OUT(); + if (!rt.account_inlined_generated_leaf(ctx)) { + tier2_gpr_shadow_valid = false; + TIER2_SB_RETURN(); } - ++tier2_stats.fallbacks; - if (([&]() { TIER2_GPR_SYNC_OUT(); const bool tier2_same_ = (rt.invoke_chained_direct<&recomp_unit_0153_entry, 153u, 104u, 0x08A68CDCu>(ctx, &aot_mem)); if (tier2_same_) TIER2_GPR_SYNC_IN(); else tier2_gpr_shadow_valid = false; return tier2_same_; }()) && ctx.pc == 0x08A711ACu) goto SB_L_08A711AC; - TIER2_SB_RETURN(); + goto SB_L_08A711AC; SB_L_08A711AC: { const bool branch_taken = 0u == 0u; @@ -1719,21 +1688,20 @@ SB_L_08A711AC: SB_L_08A711B4: ctx.gpr[31] = (0x08A711BCu); tier2_gpr_4 = (tier2_gpr_17 | 0u); - if (tier2_return_depth < kTier2ReturnCapacity) { - if (!rt.tier2_enter_fused_transfer<152u, 0x08A65EA0u>(ctx)) { - ctx.pc = 0x08A65EA0u; - TIER2_SB_RETURN(); - } - tier2_return_pc[tier2_return_depth] = 0x08A711BCu; - tier2_return_unit[tier2_return_depth] = 155u; - tier2_return_pending_base[tier2_return_depth] = tier2_pending_transfers; - ++tier2_return_depth; - ++tier2_stats.fused_calls; - goto SB_L_08A65EA0; + tier2_gpr_4 = aot_mem.aot_load32(tier2_gpr_4 + static_cast(72)); + tier2_gpr_4 = (tier2_gpr_4 & 14u); + ctx.gpr[2] = (tier2_gpr_4 ^ 6u); + ctx.gpr[2] = (ctx.gpr[2] < static_cast(1) ? 1u : 0u); + // Publish the exact JAL return PC before scheduler accounting. + // A starvation boundary may switch PSP ownership here; the + // resumed context must never observe the stale superblock PC. + ctx.pc = 0x08A711BCu; + TIER2_GPR_SYNC_OUT(); + if (!rt.account_inlined_generated_leaf(ctx)) { + tier2_gpr_shadow_valid = false; + TIER2_SB_RETURN(); } - ++tier2_stats.fallbacks; - if (([&]() { TIER2_GPR_SYNC_OUT(); const bool tier2_same_ = (rt.invoke_chained_direct<&recomp_unit_0152_entry, 152u, 432u, 0x08A65EA0u>(ctx, &aot_mem)); if (tier2_same_) TIER2_GPR_SYNC_IN(); else tier2_gpr_shadow_valid = false; return tier2_same_; }()) && ctx.pc == 0x08A711BCu) goto SB_L_08A711BC; - TIER2_SB_RETURN(); + goto SB_L_08A711BC; SB_L_08A711BC: { const bool branch_taken = ctx.gpr[2] == 0u; @@ -1747,21 +1715,17 @@ SB_L_08A711BC: SB_L_08A711C4: ctx.gpr[31] = (0x08A711CCu); tier2_gpr_4 = (tier2_gpr_17 | 0u); - if (tier2_return_depth < kTier2ReturnCapacity) { - if (!rt.tier2_enter_fused_transfer<153u, 0x08A68CD4u>(ctx)) { - ctx.pc = 0x08A68CD4u; - TIER2_SB_RETURN(); - } - tier2_return_pc[tier2_return_depth] = 0x08A711CCu; - tier2_return_unit[tier2_return_depth] = 155u; - tier2_return_pending_base[tier2_return_depth] = tier2_pending_transfers; - ++tier2_return_depth; - ++tier2_stats.fused_calls; - goto SB_L_08A68CD4; + ctx.gpr[2] = aot_mem.aot_load32(tier2_gpr_4 + static_cast(2116)); + // Publish the exact JAL return PC before scheduler accounting. + // A starvation boundary may switch PSP ownership here; the + // resumed context must never observe the stale superblock PC. + ctx.pc = 0x08A711CCu; + TIER2_GPR_SYNC_OUT(); + if (!rt.account_inlined_generated_leaf(ctx)) { + tier2_gpr_shadow_valid = false; + TIER2_SB_RETURN(); } - ++tier2_stats.fallbacks; - if (([&]() { TIER2_GPR_SYNC_OUT(); const bool tier2_same_ = (rt.invoke_chained_direct<&recomp_unit_0153_entry, 153u, 103u, 0x08A68CD4u>(ctx, &aot_mem)); if (tier2_same_) TIER2_GPR_SYNC_IN(); else tier2_gpr_shadow_valid = false; return tier2_same_; }()) && ctx.pc == 0x08A711CCu) goto SB_L_08A711CC; - TIER2_SB_RETURN(); + goto SB_L_08A711CC; SB_L_08A711CC: { const bool branch_taken = ctx.gpr[2] != tier2_gpr_19; @@ -1776,21 +1740,17 @@ SB_L_08A711D4: tier2_gpr_4 = (tier2_gpr_17 | 0u); ctx.gpr[31] = (0x08A711E0u); tier2_gpr_5 = (0u | 0u); - if (tier2_return_depth < kTier2ReturnCapacity) { - if (!rt.tier2_enter_fused_transfer<153u, 0x08A68CDCu>(ctx)) { - ctx.pc = 0x08A68CDCu; - TIER2_SB_RETURN(); - } - tier2_return_pc[tier2_return_depth] = 0x08A711E0u; - tier2_return_unit[tier2_return_depth] = 155u; - tier2_return_pending_base[tier2_return_depth] = tier2_pending_transfers; - ++tier2_return_depth; - ++tier2_stats.fused_calls; - goto SB_L_08A68CDC; + aot_mem.aot_store32(tier2_gpr_4 + static_cast(2116), tier2_gpr_5); + // Publish the exact JAL return PC before scheduler accounting. + // A starvation boundary may switch PSP ownership here; the + // resumed context must never observe the stale superblock PC. + ctx.pc = 0x08A711E0u; + TIER2_GPR_SYNC_OUT(); + if (!rt.account_inlined_generated_leaf(ctx)) { + tier2_gpr_shadow_valid = false; + TIER2_SB_RETURN(); } - ++tier2_stats.fallbacks; - if (([&]() { TIER2_GPR_SYNC_OUT(); const bool tier2_same_ = (rt.invoke_chained_direct<&recomp_unit_0153_entry, 153u, 104u, 0x08A68CDCu>(ctx, &aot_mem)); if (tier2_same_) TIER2_GPR_SYNC_IN(); else tier2_gpr_shadow_valid = false; return tier2_same_; }()) && ctx.pc == 0x08A711E0u) goto SB_L_08A711E0; - TIER2_SB_RETURN(); + goto SB_L_08A711E0; SB_L_08A711E0: tier2_gpr_4 = (aot_mem.aot_load32(tier2_gpr_29 + static_cast(1576))); diff --git a/profiles/vcs/scripts/build_release_ninja.bat b/profiles/vcs/scripts/build_release_ninja.bat index 3fa9b2a..b2de956 100644 --- a/profiles/vcs/scripts/build_release_ninja.bat +++ b/profiles/vcs/scripts/build_release_ninja.bat @@ -99,6 +99,8 @@ set "PERF_V5_STALLFIX_STAMP=%BUILD%\.vcs_perf_v5_async_stallfix_20260817" set "PERF_V5_PRESENTFIX_STAMP=%BUILD%\.vcs_perf_v5_async_presentfix_20260817" set "PERF_V5_SYNC_RECOVERY_STAMP=%BUILD%\.vcs_perf_v5_sync_recovery_20260817" set "PERF_V5_STABLE_RECOVERY2_STAMP=%BUILD%\.vcs_perf_v5_stable_recovery2_20260817" +set "PERF_V6_ENTITY_LEAF_STAMP=%BUILD%\.vcs_perf_v6_entity_leaf_inline_20260817" +set "PERF_V6_ENTITY_LEAF_FIX1_STAMP=%BUILD%\.vcs_perf_v6_entity_leaf_inline_crashfix1_20260817" echo ================================================================ echo VCS - NINJA PERFORMANCE INCREMENTAL BUILD @@ -110,7 +112,7 @@ echo CMake: %CMAKE_EXE% echo Ninja: %NINJA_EXE% echo Ninja workers: %JOBS% echo cl.exe /MP: OFF ^(Ninja owns compile parallelism^) -echo Generated AOT: O3, cold /Ob0, measured hot /Ob3; V4-stable Tier2 + sync GE; V5 risky paths quarantined; AMD/UMA safe +echo Generated AOT: O3, cold /Ob0, measured hot /Ob3; V6 Entity leaf-inline CRASHFIX1 over V4-stable Tier2; risky async/decode quarantined echo Host/core LTCG: ON echo AVX2/fast paths: ON echo ================================================================ @@ -119,7 +121,7 @@ echo [0b/7] Reapplying BOOTFIX-safe Tier-2 transforms (OPT1 semantic transforms call "%PROFILE%\APPLY_TIER2_EXTREME.bat" if errorlevel 1 goto :FAIL -echo [0b2/7] Building gameplay-stable V4 Tier2 layer with V5 recovery guards... +echo [0b2/7] Building V6 Entity leaf-inline CRASHFIX1 over gameplay-stable V4 Tier2... set "PYTHON3_CMD=" py -3 -c "import sys; raise SystemExit(0 if sys.version_info.major == 3 else 1)" >nul 2>&1 if not errorlevel 1 set "PYTHON3_CMD=py -3" @@ -204,6 +206,23 @@ if exist "%BUILD%" if not exist "%PERF_V5_STABLE_RECOVERY2_STAMP%" ( del /s /q "%BUILD%\*vcs_runtime_log*.obj" >nul 2>&1 ) +if exist "%BUILD%" if not exist "%PERF_V6_ENTITY_LEAF_STAMP%" ( + echo. + echo [0c-v6entity/7] V6 ENTITY LEAF INLINE - invalidating Entity + runtime log once... + rem V6 changes only the Entity Tier2 TU, its generator, and runtime metadata. + rem Keep Geometry and every generated AOT object to preserve the stable incremental build. + del /s /q "%BUILD%\*vcs_tier2_cluster_entity*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_runtime_log*.obj" >nul 2>&1 +) + +if exist "%BUILD%" if not exist "%PERF_V6_ENTITY_LEAF_FIX1_STAMP%" ( + echo. + echo [0c-v6entityfix1/7] V6 ENTITY LEAF INLINE CRASHFIX1 - publishing continuation PC before scheduler boundary... + rem Only Entity codegen and runtime metadata changed. No Geometry/AOT rebuild. + del /s /q "%BUILD%\*vcs_tier2_cluster_entity*.obj" >nul 2>&1 + del /s /q "%BUILD%\*vcs_runtime_log*.obj" >nul 2>&1 +) + if exist "%BUILD%" if not exist "%BOOTFIX_STAMP%" ( echo. echo [0c/7] BOOTFIX revision changed - invalidating stale .obj/.pch once... @@ -245,6 +264,8 @@ if errorlevel 1 goto :FAIL >"%PERF_V5_PRESENTFIX_STAMP%" echo VCS PERF V5 GE ASYNC PRESENTFIX 2026-08-17 >"%PERF_V5_SYNC_RECOVERY_STAMP%" echo VCS PERF V5 SYNC RECOVERY 2026-08-17 >"%PERF_V5_STABLE_RECOVERY2_STAMP%" echo VCS PERF V5 STABLE RECOVERY2 2026-08-17 +>"%PERF_V6_ENTITY_LEAF_STAMP%" echo VCS PERF V6 ENTITY LEAF INLINE 2026-08-17 +>"%PERF_V6_ENTITY_LEAF_FIX1_STAMP%" echo VCS PERF V6 ENTITY LEAF INLINE CRASHFIX1 2026-08-17 echo. echo [2b/7] Building tests and DX12 probes... diff --git a/profiles/vcs/tests/check_v6_entity_leaf_inline.py b/profiles/vcs/tests/check_v6_entity_leaf_inline.py new file mode 100644 index 0000000..87c6555 --- /dev/null +++ b/profiles/vcs/tests/check_v6_entity_leaf_inline.py @@ -0,0 +1,47 @@ +#!/usr/bin/env python3 +from pathlib import Path +import re +import sys + +profile = Path(__file__).resolve().parents[1] +entity = (profile / 'host' / 'vcs_tier2_cluster_entity.cpp').read_text(encoding='utf-8') +gen = (profile / 'tools' / 'build_tier2_superblocks.py').read_text(encoding='utf-8') + +errors = [] +if entity.count('account_inlined_generated_leaf(ctx)') != 13: + errors.append('expected exactly 13 inlined Entity leaf call sites') + +resume_pairs = re.findall( + r'ctx\.pc = (0x[0-9A-F]{8})u;\n\s*TIER2_GPR_SYNC_OUT\(\);\n\s*if \(!rt\.account_inlined_generated_leaf\(ctx\)\)', + entity) +if len(resume_pairs) != 13: + errors.append(f'expected 13 continuation-PC publications before inline scheduler accounting, got {len(resume_pairs)}') +if len(resume_pairs) == 13: + account_positions = [m.start() for m in re.finditer(r'account_inlined_generated_leaf\(ctx\)', entity)] + for index, pos in enumerate(account_positions): + window = entity[max(0, pos - 900):pos] + ra = re.findall(r'ctx\.gpr\[31\] = \((0x[0-9A-F]{8})u\);', window) + pc = re.findall(r'ctx\.pc = (0x[0-9A-F]{8})u;', window) + if not ra or not pc or ra[-1] != pc[-1]: + errors.append(f'inline site {index} does not publish the exact JAL return PC before accounting') +if 'tier2_enter_fused_transfer<152u' in entity: + errors.append('unit 0152 still uses fused-call plumbing inside Entity cluster') +if 'tier2_enter_fused_transfer<153u' in entity: + errors.append('unit 0153 still uses fused-call plumbing inside Entity cluster') +for pc in ('0x08A65EA0', '0x08A65EB4', '0x08A68CBC', '0x08A68CC4', + '0x08A68CCC', '0x08A68CD4', '0x08A68CDC'): + if pc not in gen: + errors.append(f'missing inline leaf mapping {pc}') +# Keep the known-stable V4 Entity shadow ownership set. This is deliberate: +# V6 makes gpr[2] syntactically hotter but must not evict stack pointer gpr[29]. +for reg in (4, 6, 19, 17, 5, 29): + if f'tier2_gpr_{reg} = ctx.gpr[{reg}]' not in entity: + errors.append(f'missing stable shadow register gpr[{reg}]') +if 'tier2_gpr_2 = ctx.gpr[2]' in entity: + errors.append('gpr[2] unexpectedly displaced a stable V4 shadow register') + +if errors: + for e in errors: + print('FAIL:', e) + sys.exit(1) +print('check_v6_entity_leaf_inline: PASS') diff --git a/profiles/vcs/tools/build_tier2_superblocks.py b/profiles/vcs/tools/build_tier2_superblocks.py index 0ac47f7..50e911e 100644 --- a/profiles/vcs/tools/build_tier2_superblocks.py +++ b/profiles/vcs/tools/build_tier2_superblocks.py @@ -1,5 +1,5 @@ #!/usr/bin/env python3 -"""Generate VCS Tier-2 SUPERBLOCK V4 150FPS multi-cluster second-layer AOT. +"""Generate VCS Tier-2 V6 entity-leaf-inline multi-cluster second-layer AOT. V4 is profile-guided and intentionally keeps the original generated corpus as its semantic fallback. It extracts only measured hot control-flow closures, @@ -46,6 +46,42 @@ LOCAL_DISPATCH_SEQUENCE_RE = re.compile( r'(?P=indent)return;') +# Tiny leaf accessors used repeatedly by the 0155 entity-link maintenance loop. +# V6 emits their exact architectural bodies directly into 0155, eliminating +# C++ return-stack/chain-depth plumbing while keeping scheduler accounting at +# the original guest call cadence. If the runtime chain is already at its +# depth limit we retain the ordinary fused-call path below. +ENTITY_INLINE_LEAF_BODIES = { + 0x08A65EA0: (152, [ + 'ctx.gpr[4] = aot_mem.aot_load32(ctx.gpr[4] + static_cast(72));', + 'ctx.gpr[4] = (ctx.gpr[4] & 14u);', + 'ctx.gpr[2] = (ctx.gpr[4] ^ 6u);', + 'ctx.gpr[2] = (ctx.gpr[2] < static_cast(1) ? 1u : 0u);', + ]), + 0x08A65EB4: (152, [ + 'ctx.gpr[4] = aot_mem.aot_load32(ctx.gpr[4] + static_cast(72));', + 'ctx.gpr[4] = (ctx.gpr[4] & 14u);', + 'ctx.gpr[2] = (ctx.gpr[4] ^ 8u);', + 'ctx.gpr[2] = (ctx.gpr[2] < static_cast(1) ? 1u : 0u);', + ]), + 0x08A68CBC: (153, [ + 'ctx.gpr[2] = aot_mem.aot_load32(ctx.gpr[4] + static_cast(444));', + ]), + 0x08A68CC4: (153, [ + 'aot_mem.aot_store32(ctx.gpr[4] + static_cast(444), ctx.gpr[5]);', + ]), + 0x08A68CCC: (153, [ + 'ctx.gpr[2] = aot_mem.aot_load32(ctx.gpr[4] + static_cast(448));', + ]), + 0x08A68CD4: (153, [ + 'ctx.gpr[2] = aot_mem.aot_load32(ctx.gpr[4] + static_cast(2116));', + ]), + 0x08A68CDC: (153, [ + 'aot_mem.aot_store32(ctx.gpr[4] + static_cast(2116), ctx.gpr[5]);', + ]), +} + + @dataclasses.dataclass(frozen=True) class UnitSource: unit: int @@ -337,7 +373,13 @@ def optimize_tier2_gpr_shadow(text: str, cluster_key: str) -> tuple[str, Dict[st body = text[body_start:end] eligible = cluster_key not in {'physics', 'matrix', 'geometry'} counts = collections.Counter(int(x) for x in re.findall(r'ctx\.gpr\[(\d+)\]', body)) - selected = [reg for reg, _ in counts.most_common() if reg != 0][:6] if eligible else [] + if cluster_key == 'entity': + # Preserve the exact V4-stable entity shadow ownership set. V6 leaf + # inlining makes gpr[2] syntactically hotter, but replacing stack pointer + # gpr[29] with it would pessimize the much larger 0154 stack-local body. + selected = [4, 6, 19, 17, 5, 29] + else: + selected = [reg for reg, _ in counts.most_common() if reg != 0][:6] if eligible else [] if selected: decls = ''.join(f' std::uint32_t tier2_gpr_{r} = ctx.gpr[{r}];\n' for r in selected) @@ -559,6 +601,29 @@ def transform_block(block: str, source_unit: int, selected: Mapping[int, Set[int indent = m.group('indent') if target_pc not in selected_all or pc_owner[target_pc] != target_unit: return m.group(0) + if source_unit == 155 and target_pc in ENTITY_INLINE_LEAF_BODIES: + leaf_unit, leaf_body = ENTITY_INLINE_LEAF_BODIES[target_pc] + if leaf_unit != target_unit: + raise RuntimeError(f'entity inline leaf unit mismatch for 0x{target_pc:08X}') + stats['inlined_leaf_sites'] += 1 + body = '\n'.join(f'{indent}{line}' for line in leaf_body) + # The caller has already materialized $ra and arguments exactly as + # the original JAL would. Execute the tiny leaf body in place, then + # preserve the removed call's scheduler cadence. Entity GPR shadow + # publication costs the same sync-out the old fused-return path paid, + # but all chain-depth, return-stack and local-dispatch plumbing is gone. + return ( + f'{body}\n' + f'{indent}// Publish the exact JAL return PC before scheduler accounting.\n' + f'{indent}// A starvation boundary may switch PSP ownership here; the\n' + f'{indent}// resumed context must never observe the stale superblock PC.\n' + f'{indent}ctx.pc = 0x{cont:08X}u;\n' + f'{indent}TIER2_GPR_SYNC_OUT();\n' + f'{indent}if (!rt.account_inlined_generated_leaf(ctx)) {{\n' + f'{indent} tier2_gpr_shadow_valid = false;\n' + f'{indent} TIER2_SB_RETURN();\n' + f'{indent}}}\n' + f'{indent}goto SB_L_{cont:08X};') stats['fused_calls'] += 1 fused_continuations[(source_unit, cont)] = sources[source_unit].entries.get(cont, -1) original = m.group(0).replace('return;', 'TIER2_SB_RETURN();') @@ -914,7 +979,7 @@ def main() -> int: # common file is handwritten by V2 and must remain; generated cluster files # carry all hot code. - print('Tier2 SUPERBLOCK V4 150FPS:', + print('Tier2 V6 ENTITY LEAF INLINE:', f'clusters={len(CLUSTERS)} hook_units_changed={hook_units_changed}', f'blocks={total["blocks"]} lines={total["lines"]}', f'fused_calls={total["fused_calls"]} fused_tail={total["fused_tail"]}',